97 Commits

Author SHA1 Message Date
f050b9dd54 Added roadmap documentation for the upcoming refactoring of the publish stage 2026-05-22 23:19:44 -05:00
9c9cb54339 Implemented multiple campaign support via a campaign directory registry with explicit campaign IDs 2026-05-22 23:01:27 -05:00
7657ec3ad6 CLI cleanup to consolidate session-related subcommands 2026-05-22 22:09:17 -05:00
cee52aa092 Updated transcript artifact names and canonical paths to use a consistent, role-based nomenclature 2026-05-22 19:05:23 -05:00
e920f3a8d5 Cleaned up and removed legacy configuration surfaces 2026-05-22 18:32:14 -05:00
591c529a09 Updated the analyze stage to accept --artifacts as a CLI flag 2026-05-22 18:01:05 -05:00
7324c5a686 Session configuration templates are now proceeded by narratio session init; all other commands require concrete configuration 2026-05-22 17:38:23 -05:00
d0936fb022 Implemented default config/campaign discovery for narratio session init 2026-05-22 11:36:57 -05:00
2aa074c5cf Implemented narratio publish as a shortcut to run the archive stage only 2026-05-22 11:28:38 -05:00
782d0cf3b9 Upgraded the restore command to download previous session artifcats when configured as inputs for the current session analyze stage 2026-05-21 23:31:01 -05:00
083c01cfa0 Implemented narratio analyze as a shortcut to run the analyze stage only
Some checks failed
ci/woodpecker/tag/release Pipeline failed
2026-05-21 23:02:19 -05:00
2937696024 Add clean command 2026-05-21 22:48:18 -05:00
b817a5b772 Implemented shared S3 audio caching for prepare and restore --include-audio 2026-05-21 22:22:08 -05:00
3022f20beb Simplified the output of narratio artifacts list --remote and narratio status --session-id 2026-05-21 21:20:41 -05:00
ca1ded1821 Bugfix for commands that list artifacts in the S3 backend 2026-05-21 21:00:51 -05:00
3752f3ed28 Added remote artifact listing to narratio status 2026-05-21 20:49:28 -05:00
870c2d69d5 Consolidated addition, removal, and listing of locks under a single narratio locks command 2026-05-21 19:36:14 -05:00
135407ba7c Implemented a centralized secret-backed object-store helper 2026-05-21 19:13:10 -05:00
228c348e42 Implemented operations helper commands for validation, locking, and status 2026-05-21 11:50:20 -05:00
a813bd5a50 Fixed a redundant path bug for previous session artifacts 2026-05-21 10:58:49 -05:00
d8f58dce31 Normalize the default configuration discovery paths for all three config files, and update documentation and tests accordingly 2026-05-21 09:55:56 -05:00
7111edeca4 Add archive promotion locks 2026-05-20 21:40:09 -05:00
3aae4bbb12 Add remote session loading 2026-05-20 20:55:13 -05:00
b29d8eeb50 Add campaign configuration support 2026-05-20 20:41:28 -05:00
dffb432537 Removed completed roadmap for previous_session artifacts 2026-05-20 20:15:47 -05:00
2dd38c7913 Refine campaign and remote session roadmap 2026-05-20 20:15:05 -05:00
bc2ade38d9 Finalize previous-session artifact documentation and restore-analyze continuity coverage 2026-05-20 15:17:04 +00:00
5be831eb13 Restore archived previous-session cache files with session state 2026-05-20 15:05:48 +00:00
cae4d99a89 Archive durable previous-session cache files with session state 2026-05-20 15:03:30 +00:00
e09dc0512d Add analyze integration coverage for previous-session inputs 2026-05-20 15:01:27 +00:00
ae82bc1ce0 Resolve canonical previous-session artifact sources from prepared previous cache 2026-05-20 14:59:28 +00:00
01eb7aa1aa Add prepare rerun guidance for unresolved previous-session analyze inputs 2026-05-20 14:55:44 +00:00
2ca700195c Integrate previous-session artifact hydration into prepare stage 2026-05-20 14:53:22 +00:00
2b08c34539 Add prepare helper to hydrate previous-session artifacts from archive 2026-05-20 14:49:22 +00:00
79f1fc1e09 Add helper to collect previous-session artifact input requirements 2026-05-20 14:37:02 +00:00
9c753270bd Add canonical previous-session artifact source parsing and validation 2026-05-20 14:34:23 +00:00
b907cb01aa Add previous-session workspace path helpers and layout support 2026-05-20 14:31:03 +00:00
7824afd4a5 Add previous session ID templating and CLI support 2026-05-20 14:26:32 +00:00
2a4e1e912c Update documentation to include a roadmap for previous session artifact support 2026-05-20 09:12:17 -05:00
dd03c09d75 Fixed a bug in the S3 credential loading for the restore command
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-19 22:51:26 -05:00
5bc8e8683f Documentation update for the restore subcommand 2026-05-19 22:32:55 -05:00
648001a8fe Add workflow integration tests for the restore command 2026-05-19 22:23:25 -05:00
6684774f52 Add restore report and operator summary 2026-05-19 22:15:40 -05:00
f3b63bd5e5 Implement restore execution for the restore subcommand 2026-05-19 22:06:48 -05:00
23d6470b0f Implement restore planning for the restore subcommand 2026-05-19 21:58:38 -05:00
128449040f Implement remote current-state discovery for the restore subcommand 2026-05-19 21:49:19 -05:00
02ab106ade Implement initial CLI command for narratio restore, and extract shared helper functions from the run stages 2026-05-19 21:39:13 -05:00
c128970f58 Updated documentation to remove the completed runtime artifacts roadmap and add a new restore subcommand roadmap 2026-05-19 21:21:06 -05:00
d001baa660 Use artifact source IDs for archive promotion 2026-05-19 20:05:24 -05:00
c5c35cd3b4 Moved example configuration from docs/examples/ to top-level examples/
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-19 19:46:40 -05:00
574b1cde6c Update documentation for the new analyze stage and artifact registry 2026-05-19 19:42:28 -05:00
ebb21b9201 Removed the legacy built-in session_recap from the analyze stage 2026-05-19 19:26:07 -05:00
958f446387 Add archive stage integration test for the new analyze stage features 2026-05-19 19:12:10 -05:00
86caf4b222 Update analyze-stage metadata and manifest output 2026-05-19 19:07:44 -05:00
e38ed8ba97 Refactor the analyze stage to actually produce the configured artifacts 2026-05-19 18:58:28 -05:00
3e79cf4724 Update artifact resolution so configured artifact IDs are resolved through the runtime catalog 2026-05-19 18:49:27 -05:00
859ae1ae10 Add abstractions for the internal artifact catalog 2026-05-19 18:43:55 -05:00
c63ecbab32 Add new configuration fields and CLI flags for the upcoming analyze stage enhancements 2026-05-19 18:36:30 -05:00
8480b74283 Updated the roadmap for configurable artifact generation 2026-05-19 11:21:28 -05:00
087869f7fa Added a roadmap for new work to support configurable artifacts defined at runtime 2026-05-19 09:26:46 -05:00
2b2a314d65 Move documentation for external integrations into the docs/integrations subfolder 2026-05-19 09:12:40 -05:00
08b0f4edc5 Removed legacy transcript artifact aliases 2026-05-19 09:00:41 -05:00
571a289296 Added new internal documentation 2026-05-19 08:47:45 -05:00
9f80635b42 Updated configuration docs to reflect the minimal pipeline config 2026-05-19 08:15:12 -05:00
11a3e174b6 Set default value for workspace.root and updated config documentation 2026-05-19 07:07:19 -05:00
9c5e5d6dc1 Simplified the reference tables in docs/config.md 2026-05-19 06:51:55 -05:00
c4e87f58c7 Complete documentation rebuild 2026-05-18 22:03:59 -05:00
37daab7857 Bugfix involving nested directory creation
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-18 11:52:40 +00:00
2356688cb9 Removed legacy interfaces and old documentation references to the previous on-disk layout 2026-05-18 03:02:22 +00:00
1054b64d9f Implement minimal downstream invalidation after forced upstream reruns 2026-05-18 01:50:51 +00:00
01fb02426c Update the analyze stage to utilize the new artifact package 2026-05-18 01:29:18 +00:00
7dc79e052f Aligned the archive stage with the new work directory layout 2026-05-18 01:13:51 +00:00
cb525c0f72 Implemented run-local stage execution + immediate promotion for core output-producing stages 2026-05-18 00:55:35 +00:00
622677d038 Added run manifest scaffolding and helpers 2026-05-17 21:15:51 +00:00
550288e008 Add campaign-aware workspace path foundation 2026-05-17 20:57:27 +00:00
e58e545686 Audit workspace architecture implementation plan
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-17 13:22:09 -05:00
6ff54c5a0f Documentation update and reorganization 2026-05-17 13:14:28 -05:00
924b5d15c6 Applied a more general bugfix to path-resolution issues in the archive stage 2026-05-17 11:08:11 -05:00
b065663180 Bugfix involving path resolution in the archive stage 2026-05-17 11:03:02 -05:00
3ba564b00f Bugfix involving directory creation during the merge stage 2026-05-17 08:18:57 -05:00
a3986cf0d6 Centralized defaults into internal/config/defaults.go 2026-05-17 07:53:51 -05:00
539601bd16 Updated the merge stage to normalize the per-speaker transcripts before merging them
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-16 23:30:13 -05:00
6ca1c8d6b0 The backend S3 client now resolves credentials from user-configurable environment variables 2026-05-16 23:22:21 -05:00
4b7b50981b Add .gocache to .gitignore and minor documentation cleanup 2026-05-16 23:21:45 -05:00
33f7ae8f2e Simplify downstream tool configuration
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-16 23:30:40 +00:00
d5a9ad38f8 Add post-archive cleanup policies 2026-05-16 23:09:39 +00:00
6fbefb9867 Add session discovery and template support 2026-05-16 22:57:42 +00:00
1665359486 Added a locally generated UX progress report 2026-05-16 20:11:20 +00:00
03f2543927 Created a UX status report 2026-05-16 20:09:53 +00:00
fe9c348092 Document and review S3 archive workflow 2026-05-16 15:24:44 +00:00
f7f8f1a949 Promote current session artifacts to storage 2026-05-16 15:01:01 +00:00
d40c91acde Upload successful run records to storage 2026-05-16 14:43:29 +00:00
ed4dcf1ef7 Updated go.mod 2026-05-16 09:34:43 -05:00
24cce49a70 Download S3 audio during prepare 2026-05-16 14:33:42 +00:00
1e6db89dd4 Add remote storage backend 2026-05-16 14:22:04 +00:00
0454296c81 Add archive storage path configuration 2026-05-16 14:11:59 +00:00
58c6ab2d54 Updated audita configuration to reflect the new audita public CLI
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-16 08:46:48 -05:00
183 changed files with 28820 additions and 3970 deletions

5
.gitignore vendored
View File

@@ -2,6 +2,8 @@
.codex .codex
AGENTS.md AGENTS.md
.DS_Store
# ---> Go # ---> Go
# If you prefer the allow list template instead of the deny list, see community template: # If you prefer the allow list template instead of the deny list, see community template:
# https://github.com/github/gitignore/blob/main/community/Golang/Go.AllowList.gitignore # https://github.com/github/gitignore/blob/main/community/Golang/Go.AllowList.gitignore
@@ -22,6 +24,9 @@ AGENTS.md
# Dependency directories (remove the comment below to include it) # Dependency directories (remove the comment below to include it)
# vendor/ # vendor/
# Go cache
.gocache
# Go workspace file # Go workspace file
go.work go.work
go.work.sum go.work.sum

304
README.md
View File

@@ -1,300 +1,22 @@
# narratio # narratio
`narratio` is a Go orchestration application for processing D&D session audio into transcripts and generated artifacts. Narratio is a Go orchestration application that turns D&D session audio into polished transcripts and generated session artifacts.
## Current Implementation It coordinates transcription, merge/polish/normalize/trim processing, artifact generation, archive publishing, and resumable run state in one operator workflow.
Implemented now:
- strict config loading/validation (`pipeline.yml` and `session.yml`)
- local workspace/session layout, locking, and manifest persistence
- resumable stage control (`run`, `plan`, `resume`, `run-stage`, `status`)
- real `prepare`, `transcribe`, `merge`, `polish`, `normalize`, `trim`, and `analyze` stages
- real WhisperX, Seriatim, and Audita adapters
- real Scriptorium subprocess adapter
- optional Scriptorium render diagnostics (`render_debug`)
Not implemented yet:
- `archive` stage behavior
- `notify` stage behavior
- additional analyze artifacts beyond `session_recap`
- generic DAG orchestration
## Config Files
Narratio expects two YAML files:
- `pipeline.yml`: pipeline/workspace settings
- `session.yml`: per-session settings
Pipeline config lookup for CLI commands:
- if `--config <path>` is provided, Narratio uses that path
- if `--config` is omitted, Narratio searches in this order:
- `/usr/local/etc/narratio/pipeline.yml`
- `/etc/narratio/pipeline.yml`
Optional secrets-from-files config:
- `pipeline.secrets.env_dir` may point to a directory of secret files
- each top-level file with an env-var-style name is loaded as an environment variable:
- file name = env var name
- file contents = env var value (trailing newline/CRLF trimmed)
- process environment wins: existing env vars are not overwritten
- if configured, Narratio fails fast when `env_dir` is missing/unreadable
- relative `env_dir` values resolve from Narratios current working directory
YAML decoding is strict (`KnownFields(true)`), so unknown fields fail fast.
## Canonical Stage Order
1. `prepare`
2. `transcribe`
3. `merge`
4. `polish`
5. `normalize`
6. `trim`
7. `analyze`
8. `archive`
9. `notify`
## Transcript Tiers
- `transcripts/merged.json`: canonical deterministic merged transcript from Seriatim merge
- `transcripts/processed.json`: full raw Audita-polished transcript output
- `transcripts/normalized.json`: Seriatim-normalized transcript from the normalize stage
- `transcripts/trimmed.json`: gameplay-only normalized polished transcript from trim stage
## Normalize Configuration
`pipeline.normalize` is optional. When omitted, Narratio defaults to:
- `output_path: transcripts/normalized.json`
- `output_schema: seriatim-intermediate`
- `report: true`
Allowed `normalize.output_schema` values:
- `seriatim-minimal`
- `seriatim-intermediate`
- `seriatim-full`
`normalize.output_path` is treated as session-workdir-relative when not absolute.
Normalize stage behavior summary:
- normalize runs after `polish` and before `trim`
- normalize resolves `transcripts/processed.json`
- normalize runs Seriatim `normalize` to produce `transcripts/normalized.json`
- normalize diagnostics are written to:
- `artifacts/seriatim.normalize.report.json` (when enabled)
- `logs/seriatim.normalize.stdout.log`
- `logs/seriatim.normalize.stderr.log`
- `config/seriatim.normalize.generated.yml`
## Trim Configuration
`pipeline.trim` is optional. If omitted, no trim config is loaded. If `trim.enabled` is omitted, it defaults to `false`.
When `trim.enabled: true`:
- `trim.output_path` is required
- `trim.bounds.prompt_id` is required
- `trim.bounds.transcript_input_name` is required
- `trim.bounds.output_path` is required
- `trim.bounds.timeout` must be a valid Go duration when provided
- `trim.bounds.render_debug: true` requires `trim.bounds.render_output_path`
- `trim.bounds.profile_id` may be empty to use the prompt default profile
Trim paths are treated as session-workdir-relative when not absolute.
Example trim config:
```yaml
trim:
enabled: true
output_path: "transcripts/trimmed.json"
bounds:
prompt_id: "dnd_session.bounds"
profile_id: ""
transcript_input_name: "transcript"
output_path: "artifacts/session_bounds.json"
timeout: "10m"
render_debug: false
render_output_path: "artifacts/session_bounds.render.json"
seriatim:
report: false
```
Trim behavior summary:
- trim discovers and validates `transcripts/normalized.json`
- trim uses Scriptorium bounds (`dnd_session.bounds` by example config) to produce `artifacts/session_bounds.json`
- bounds IDs are validated against the same normalized transcript ID space that Seriatim trim will consume
- trim converts bounds to Seriatim keep selector (for example `10-868`) and runs Seriatim trim
- if trim is disabled, Narratio copies normalized transcript to trimmed transcript and records `trim_action=copy_disabled`
Trim outputs and diagnostics:
- `artifacts/session_bounds.json`
- `transcripts/trimmed.json`
- `logs/scriptorium.bounds.stdout.log`
- `logs/scriptorium.bounds.stderr.log`
- `config/scriptorium.bounds.generated.yml`
- `logs/seriatim.trim.stdout.log`
- `logs/seriatim.trim.stderr.log`
- `config/seriatim.trim.generated.yml`
- optional bounds render-debug outputs:
- `artifacts/session_bounds.render.json`
- `logs/scriptorium.bounds.render.stdout.log`
- `logs/scriptorium.bounds.render.stderr.log`
- `config/scriptorium.bounds.render.generated.yml`
Render-debug files are diagnostics and are not treated as canonical stage output artifact refs.
## Scriptorium Configuration
`pipeline.scriptorium` is optional. When present, Narratio validates and uses it for analyze-stage artifact generation.
Key points:
- `scriptorium.binary` is required when section is present
- `scriptorium.config_path` is optional
- `scriptorium.timeout` defaults to `10m` when omitted
- `scriptorium.render_debug` enables render diagnostics globally
- artifacts are configured under `scriptorium.artifacts` (map shape supports multiple artifacts)
- enabled artifacts require `prompt_id` and `output_path`
- artifact `render_debug` may override global render setting
- `vars` currently support boolean and string values
Example `session_recap` artifact definition:
```yaml
scriptorium:
binary: "scriptorium"
config_path: "/etc/scriptorium/config.yml"
timeout: "10m"
render_debug: false
artifacts:
session_recap:
enabled: true
prompt_id: "dnd.session_recap"
profile_id: "local-quality" # optional
output_path: "artifacts/session_recap.md"
timeout: "10m"
# render_debug: true # optional per-artifact override
inputs:
transcript:
source: "trimmed_transcript"
required: true
previous_recap:
source: "previous_session_artifact"
artifact: "session_recap"
path: "" # optional; set when available
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: "session_recap"
```
Prompt IDs and profile IDs are configuration values. They are not hardcoded in analyze-stage logic.
Do not put secrets in `pipeline.yml`. If API-key behavior is configured, use env var names only.
If `pipeline.secrets.env_dir` is configured, keep only references and secret files there; secret values are still not written to manifests, generated configs, or Narratio-managed logs.
## Scriptorium Runtime Behavior
Narratio integrates with Scriptorium through the public CLI subprocess contract:
- generation: `scriptorium run`
- diagnostics/testing: `scriptorium render --format json` when `render_debug` is enabled
For the initial implementation, only `session_recap` generation is supported.
Analyze-stage session recap behavior:
- available transcript input sources for configured artifacts: `processed_transcript`, `normalized_transcript`, `trimmed_transcript`
- session recap should use gameplay-only transcript input (`source: trimmed_transcript`)
- Narratio resolves `trimmed_transcript` from trim manifest output (`transcript_trimmed`) or fallback `transcripts/trimmed.json`
- Narratio resolves `normalized_transcript` from normalize manifest output (`transcript_normalized`) or fallback `transcripts/normalized.json`
- missing trimmed transcript fails clearly and advises running trim stage first
- `normalized_transcript` is the preferred full-transcript source for future table/meta-analysis artifacts
- `processed_transcript` remains supported for advanced/debug use cases
- optionally includes `previous_recap` when configured and resolvable
- omits optional previous recap when unavailable
- fails if required inputs are missing
- validates output file exists and is non-empty
Expected session output paths:
- `artifacts/session_recap.md`
- `logs/scriptorium.session_recap.stdout.log`
- `logs/scriptorium.session_recap.stderr.log`
- `config/scriptorium.session_recap.generated.yml`
- `artifacts/session_recap.render.json` when render diagnostics are enabled
## Examples
Starter files:
- `examples/pipeline.minimal.yml`
- `examples/session.minimal.yml`
- `examples/speakers.yml`
## Commands
Run tests:
```bash ```bash
go test ./... narratio run 2026-04-04
``` ```
Plan a run: This command requires discoverable `pipeline.yml` and `session.yml` files (or explicit `--config` and `--session` flags).
```bash ## Documentation
go run ./cmd/narratio plan --session examples/session.minimal.yml
```
Use `--config <path>` to override default pipeline lookup when needed. - [Configuration](docs/config.md)
- [CLI Reference](docs/cli.md)
Run full pipeline: - [Operations and Recovery](docs/operations.md)
- [Troubleshooting](docs/troubleshooting.md)
```bash - [Development Guide](docs/development.md)
go run ./cmd/narratio run --config examples/pipeline.minimal.yml --session examples/session.minimal.yml - [Architecture Principles](docs/architecture.md)
``` - [Internal Component Contracts](docs/internal/README.md)
- [Config Examples](examples/)
Run analyze only:
```bash
go run ./cmd/narratio run-stage --config examples/pipeline.minimal.yml --session examples/session.minimal.yml analyze
```
## Operational Note
Checksum-based stale detection is not implemented yet.
If prepared inputs or prompt/runtime config change, rerun the appropriate upstream stages before relying on downstream artifacts.
Examples:
- glossary/autocorrect/speaker-context changes: rerun at least `merge`, `polish`, `normalize`, `trim`, and `analyze`
- trim bounds prompt/profile/config changes: rerun at least `normalize`, `trim`, and `analyze`
- session recap prompt/profile/input-source changes: rerun `analyze`
## Roadmap
Near-term roadmap:
- extend analyze to additional configured artifacts
- support workflows where later artifacts consume earlier generated artifacts
- keep orchestration explicit without a generic DAG engine
- implement archive and notify backends

View File

@@ -1,333 +0,0 @@
# Narratio Architecture
## 1. Purpose
`narratio` is a Go orchestrator for D&D session processing. It runs a stage-based local pipeline from audio input through transcript processing and artifact generation, with manifest-based skip/force/resume behavior.
Narratio integrates with Scriptorium through the **public CLI** (`scriptorium run` and `scriptorium render`) via synchronous subprocess execution.
## 2. Current Status
Implemented:
- strict `pipeline.yml` + `session.yml` loading with strict YAML field checking (`KnownFields(true)`)
- local workspace/session layout, lock file handling, artifact path helpers, checksums, and atomic writes
- manifest store and stage status transitions for resumable runs
- real `prepare`, `transcribe`, `merge`, and `polish` stages
- real WhisperX HTTP adapter
- real Seriatim subprocess adapter
- real Audita subprocess adapter
- real Scriptorium subprocess adapter
- real `normalize` stage producing `transcripts/normalized.json`
- real `trim` stage producing `transcripts/trimmed.json`
- real `analyze` stage for initial `session_recap` generation
- optional Scriptorium render diagnostics (`render_debug`) before production run
Still placeholder/future:
- `archive` stage behavior
- `notify` stage behavior
- additional Scriptorium artifact types beyond `session_recap`
- artifact-to-artifact workflows beyond the initial single-artifact implementation
- generic stale detection based on input/config checksums
## 3. Pipeline and Stage Boundaries
Canonical stage order:
1. `prepare`
2. `transcribe`
3. `merge`
4. `polish`
5. `normalize`
6. `trim`
7. `analyze`
8. `archive`
9. `notify`
Boundary rules:
- orchestration logic lives in `internal/app`
- stage business logic lives in `internal/stage`
- external-tool CLI construction lives in adapter packages
- Scriptorium CLI details stay in `internal/adapters/scriptorium`
## 4. Scriptorium Integration Model
Integration mode:
- public CLI subprocesses only (no Scriptorium internal Go packages, no HTTP API)
- production generation uses `scriptorium run`
- diagnostics/testing render uses `scriptorium render --format json`
Run invocation shape used by adapter:
```bash
scriptorium run --prompt <prompt_id> --input name=path --out <output_path>
```
Optional flags passed when configured:
- `--config <path>`
- `--profile <profile_id>`
- repeated `--var name=value`
- repeated `--input name=path`
- `--timeout <duration>`
- `--api-key-env <ENV_NAME>` when configured
Render invocation shape used by adapter:
```bash
scriptorium render --prompt <prompt_id> --input name=path --format json --out <render_output_path>
```
Adapter behavior:
- always passes `--out`
- captures stdout/stderr separately
- writes generated invocation metadata YAML (redacted, no secrets)
- treats exit code `0` as success
- treats exit code `1` as failure
- treats exit code `2` as failure with `validation_failed=true` and preserves output metadata when available
- validates successful output files exist and are non-empty
- does not treat non-empty stderr as failure by itself
## 5. Configuration Contract
CLI pipeline config path resolution:
- when `--config <path>` is provided, that path is used
- when `--config` is omitted, Narratio searches defaults in order:
- `/usr/local/etc/narratio/pipeline.yml`
- `/etc/narratio/pipeline.yml`
Optional pipeline secrets directory:
- `pipeline.secrets.env_dir` enables loading environment variables from local files before command execution
- file name = env var name; file contents = env var value (trailing newline/CRLF trimmed)
- only env-var-style file names are considered; other entries are ignored
- existing process environment values are preserved (not overwritten)
- if configured, unreadable/missing `env_dir` fails command execution early
- relative `env_dir` values are resolved from current working directory
`pipeline.scriptorium` is optional. Existing pipelines without Scriptorium continue to work.
`pipeline.trim` is optional. Existing pipelines without trim config continue to work.
`pipeline.normalize` is optional. Existing pipelines without normalize config continue to work.
When `pipeline.normalize` is omitted, defaults are applied:
- `output_path: transcripts/normalized.json`
- `output_schema: seriatim-intermediate`
- `report: true`
When `pipeline.normalize` is present:
- `output_path` must be non-empty
- `output_schema` must be one of `seriatim-minimal`, `seriatim-intermediate`, or `seriatim-full`
- relative `output_path` values are session-workdir-relative paths
- Seriatim binary settings still come from `pipeline.seriatim`
When `pipeline.trim` is present:
- `enabled` is optional and defaults to `false` when omitted
- relative `output_path`, `bounds.output_path`, and `bounds.render_output_path` values are session-workdir-relative paths
- do not store secrets in trim config values
When `pipeline.trim.enabled: true`:
- `output_path` is required and non-empty
- `bounds.prompt_id` is required and non-empty
- `bounds.transcript_input_name` is required and non-empty
- `bounds.output_path` is required and non-empty
- `bounds.timeout` must parse as a Go duration when provided
- `bounds.render_debug: true` requires non-empty `bounds.render_output_path`
- `bounds.profile_id` may be empty to use the prompt default profile
- prompt IDs are config values, not hardcoded stage logic
When `pipeline.scriptorium` is present:
- `binary` is required and non-empty
- `config_path` is optional; when provided it must be non-empty
- `timeout` is optional; when provided it must parse as a Go duration
- default `timeout` is `10m`
- unknown YAML fields fail strict decode
Artifacts are configured as a map under `pipeline.scriptorium.artifacts` so multiple artifacts are possible in the config shape.
For each artifact definition:
- `enabled: true` requires non-empty `prompt_id`
- `enabled: true` requires non-empty `output_path`
- `timeout` must parse as Go duration when present
- optional per-artifact `render_debug` may override global `scriptorium.render_debug`
- `inputs` are named and each input requires non-empty `source`
- inputs may be optional (`required: false`)
- `vars` values currently support `string` and `bool`
Prompt IDs and profile IDs are configuration values, not hardcoded stage logic.
Trim config shape:
```yaml
trim:
enabled: true
output_path: "transcripts/trimmed.json"
bounds:
prompt_id: "dnd_session.bounds"
profile_id: ""
transcript_input_name: "transcript"
output_path: "artifacts/session_bounds.json"
timeout: "10m"
render_debug: false
render_output_path: "artifacts/session_bounds.render.json"
seriatim:
report: false
```
## 6. Transcript Tiers
Narratio currently produces and uses four transcript tiers:
- `transcripts/merged.json`: canonical deterministic merged transcript from Seriatim merge
- `transcripts/processed.json`: full raw Audita-polished transcript output (includes pre/post-game content)
- `transcripts/normalized.json`: normalized transcript generated by Seriatim normalize
- `transcripts/trimmed.json`: gameplay-only normalized polished transcript from trim stage
Trim reads `transcripts/normalized.json`, validates bounds IDs against that same transcript ID space, and writes `transcripts/trimmed.json`.
## 7. Normalize Stage (Current Implementation)
Normalize stage behavior:
- stage order position: after `polish` and before `trim`
- discovers processed transcript from manifest polish outputs (`transcript_processed`) when present, else `work/<session_id>/transcripts/processed.json`
- validates processed transcript JSON shape (`segments` array required)
- runs Seriatim `normalize` to produce normalized transcript
- validates normalized transcript JSON shape (`segments` array required)
- validates normalize report JSON when enabled
Expected normalize outputs and diagnostics:
- `transcripts/normalized.json`
- `artifacts/seriatim.normalize.report.json` (when normalize report is enabled)
- `logs/seriatim.normalize.stdout.log`
- `logs/seriatim.normalize.stderr.log`
- `config/seriatim.normalize.generated.yml`
## 8. Trim Stage (Current Implementation)
Trim stage behavior:
- stage order position: after `normalize` and before `analyze`
- discovers normalized transcript from manifest normalize outputs (`transcript_normalized`) when present, else `work/<session_id>/transcripts/normalized.json`
- validates normalized transcript JSON shape (`segments` array required)
- when `trim.enabled: false` (or trim config omitted), deterministically copies normalized transcript to `transcripts/trimmed.json` and records `trim_action=copy_disabled`
- when `trim.enabled: true`:
- runs Scriptorium bounds prompt using configured `trim.bounds.prompt_id`
- writes bounds output to configured path (typically `artifacts/session_bounds.json`)
- parses and validates bounds output against the same normalized transcript being trimmed
- converts bounds range to Seriatim keep selector (for example `10-868`)
- runs Seriatim `trim` to produce `transcripts/trimmed.json`
- supports no-trim bounds actions (`none`/`copy`) by copying normalized transcript unchanged
- validates trimmed transcript JSON shape (`segments` array required)
Expected trim outputs and diagnostics:
- `artifacts/session_bounds.json`
- `transcripts/trimmed.json`
- `logs/scriptorium.bounds.stdout.log`
- `logs/scriptorium.bounds.stderr.log`
- `config/scriptorium.bounds.generated.yml`
- `logs/seriatim.trim.stdout.log`
- `logs/seriatim.trim.stderr.log`
- `config/seriatim.trim.generated.yml`
- optional bounds render-debug outputs when enabled:
- `artifacts/session_bounds.render.json`
- `logs/scriptorium.bounds.render.stdout.log`
- `logs/scriptorium.bounds.render.stderr.log`
- `config/scriptorium.bounds.render.generated.yml`
Render-debug files are diagnostics. They are recorded in stage metadata/log/config refs and are not treated as canonical stage output artifact refs.
## 9. Analyze Stage (Current Implementation)
The current real analyze implementation supports only `scriptorium.artifacts.session_recap`.
Behavior:
- if `pipeline.scriptorium` is missing, analyze returns a skipped result with metadata
- if no Scriptorium artifacts are enabled, analyze returns a skipped result with metadata
- if enabled artifacts exist but `session_recap` is not enabled, analyze fails clearly
- available transcript input sources for configured artifacts: `processed_transcript`, `normalized_transcript`, `trimmed_transcript`
- `session_recap` should use `trimmed_transcript` input (`transcripts/trimmed.json`) for in-universe recap generation
- `trimmed_transcript` input is resolved from manifest (`trim` output kind `transcript_trimmed`) when available, otherwise fallback path `work/<session_id>/transcripts/trimmed.json`
- `normalized_transcript` input is resolved from manifest (`normalize` output kind `transcript_normalized`) when available, otherwise fallback path `work/<session_id>/transcripts/normalized.json`
- `processed_transcript` input is resolved from manifest (`polish` output kind `transcript_processed`) when available, otherwise fallback path `work/<session_id>/transcripts/processed.json`
- `normalized_transcript` is the preferred full-transcript source for future table/meta-analysis artifacts
- `processed_transcript` remains available for advanced/debug use cases
- transcript inputs are validated as JSON with top-level `segments` array
- configured inputs are resolved by source
- optional `previous_recap` is omitted when unavailable
- required `previous_recap` fails before invocation when unavailable
- vars are built from config + session metadata
- `render_debug` controls pre-run `scriptorium render` diagnostics
- render failure stops stage before production run
- render output is validated as JSON
- production call uses Scriptorium adapter `RunArtifact`
- successful run output must exist and be non-empty
- missing `trimmed_transcript` input for configured `trimmed_transcript` source fails clearly with guidance to run trim stage first
- manifest records output refs, logs, generated config paths, and non-secret provenance metadata
## 10. Session Recap Paths
Current expected paths for `session_recap`:
- artifact output: `artifacts/session_recap.md`
- run stdout log: `logs/scriptorium.session_recap.stdout.log`
- run stderr log: `logs/scriptorium.session_recap.stderr.log`
- run generated invocation/config: `config/scriptorium.session_recap.generated.yml`
- render output (when enabled): `artifacts/session_recap.render.json`
- render stdout log: `logs/scriptorium.session_recap.render.stdout.log`
- render stderr log: `logs/scriptorium.session_recap.render.stderr.log`
- render generated invocation/config: `config/scriptorium.session_recap.render.generated.yml`
## 11. Security and Privacy
- do not store secrets in pipeline YAML, generated invocation YAML, logs, or manifest metadata
- if API-key integration is configured, pass env var names only (never raw key values)
- with `pipeline.secrets.env_dir`, secret file values are loaded into process env only and are not persisted in manifest metadata or generated configs
- avoid logging transcript content or rendered prompt content by default
- treat generated artifacts and logs as potentially sensitive session material
## 12. Operational Caveat (Pre-Stale-Detection)
Checksum-based stale detection is not implemented yet.
If prepared inputs or prompt/runtime configuration change (for example glossary files, prompt IDs, profile IDs, or relevant pipeline settings), rerun the appropriate prior stages to refresh downstream artifacts.
Examples:
- glossary or autocorrect changes usually require rerunning at least `merge`, `polish`, `normalize`, `trim`, and `analyze`
- trim prompt/profile changes require rerunning at least `normalize`, `trim`, and `analyze`
- session recap prompt/profile/input-source changes require rerunning `analyze`
## 13. Roadmap
Planned next steps:
- extend analyze beyond `session_recap` to additional configured artifacts
- support artifact inputs that consume prior generated artifacts
- keep this composable without adding a generic DAG engine in the near term
- implement real `archive` backend behavior
- implement real `notify` backend behavior
- add checksum-based stale detection and stale transitions
Architectural invariants remain:
- strict config decoding/validation
- manifest-driven run control
- clear stage/adapter separation
- configuration-driven prompt/profile/input/vars/output mapping
- Scriptorium integration through public CLI subprocess contract

202
docs/architecture.md Normal file
View File

@@ -0,0 +1,202 @@
# Narratio Architecture
## Purpose
`narratio` is a Go orchestration application for processing D&D session audio into polished transcripts and generated session artifacts.
This document defines the development principles for the project. It is inward-facing: its audience is developers and LLM coding agents. It should guide future changes, not serve as a complete implementation reference.
Implemented component details belong under `docs/internal/`.
## Project Shape
Narratio is a modular, stage-driven orchestrator.
It coordinates specialized downstream systems rather than reimplementing their domains:
- WhisperX handles transcription.
- Seriatim handles deterministic transcript merge/normalization/trim behavior.
- Audita handles transcript correction and polishing.
- Scriptorium handles prompt execution and generated artifacts.
Narratio owns orchestration, configuration loading, session/run state, local and remote path modeling, manifest persistence, stage sequencing, resume behavior, and archive semantics.
Narratio should remain explicit and comprehensible. It is not intended to become a generic workflow engine.
## Core Principles
### Modular and composable
Code should be organized around clear responsibilities. Stages, adapters, config loading, manifest persistence, path construction, and storage behavior should remain separable and independently testable.
### Hexagonal boundaries
External systems should be isolated behind narrow adapters. Stage logic should depend on Narratio-level interfaces and data structures, not on external SDK types, subprocess argument construction, or transport-specific details.
### Standard library preference
Prefer the Go standard library. Add dependencies only when they provide substantial value, are necessary for an external integration, or are a widely used de facto standard.
Accepted examples include a YAML library for configuration and the AWS SDK for S3-compatible storage.
### Explicit orchestration
The pipeline should remain stage-driven and explicit. New behavior should be added through clear stage, adapter, config, or manifest contracts rather than implicit side effects or generic workflow abstraction.
## Stage Design
Each stage should have a clear scope of responsibility.
A stage should define:
- its purpose;
- required input state;
- produced output state;
- config fields it consumes;
- external adapters it uses;
- manifest refs it reads or writes;
- skip, force, and resume behavior;
- failure behavior;
- tests that protect its contract.
Stages should avoid reaching across boundaries. If shared behavior is needed, prefer a helper or service with a narrow interface over duplicating ad hoc logic between stages.
## Transactionality and Resume
A stage should behave transactionally.
A stage is complete only when its outputs have been written, validated, and recorded in the manifest. If a stage fails, Narratio should preserve enough local state for inspection, recovery, and resume.
A failed or incomplete run must not be treated as successful. Later stages should depend on manifest-recorded success, not merely on incidental files existing on disk.
## Manifest Model
The manifest is the durable local ledger for a run.
It should record:
- run identity;
- stage status;
- input and output refs;
- logs and generated config refs;
- checksums or provenance where useful;
- non-secret adapter and archive metadata.
Resume behavior should be manifest-driven. Filesystem state may be inspected and validated, but it should not replace manifest stage state as the source of run progress.
## Adapter Boundaries
Adapters own external integration details.
Expected boundaries:
- WhisperX HTTP details stay in the WhisperX adapter.
- Seriatim CLI construction stays in the Seriatim adapter.
- Audita CLI construction stays in the Audita adapter.
- Scriptorium CLI construction stays in the Scriptorium adapter.
- Object-storage details stay behind the storage adapter interface.
- AWS SDK types stay inside the S3 storage implementation.
Stage code should express intent in Narratio terms and call adapters through narrow contracts.
## Configuration Philosophy
Configuration should be strict, explicit, and operator-friendly.
Principles:
- YAML decoding should reject unknown fields.
- Defaults should be centralized and testable.
- Empty configured values should not silently override meaningful defaults.
- Session templating should remain narrow and deterministic.
- Template support should serve operator convenience, not become a general configuration language.
Narratio should not become a secondary configuration system for downstream tools. Seriatim, Audita, and Scriptorium should own their runtime defaults wherever practical. Narratio should pass required stage-contract paths and explicit operator overrides.
## Path and Storage Discipline
Local and remote paths are part of Narratios application contract.
Code should use centralized path helpers for workspace, spool, session, run, artifact, log, config, and archive paths. Stages should avoid reconstructing canonical paths through scattered string concatenation.
Storage backends should receive explicit bucket-relative keys. Storage implementations should not infer campaign, session, run, or root-prefix semantics.
## Archive Invariants
Archive behavior must preserve a clear commit boundary.
A remote run is current only after the archive stage has successfully uploaded the run record, required promoted outputs, `current/manifest.json`, and finally `current/run_id.txt`.
`current/run_id.txt` is the final remote commit marker and must be written last.
Failed, incomplete, skipped, or uncommitted archive attempts must not be presented as current remote state. Local cleanup is permitted only after successful archive commit and only when explicitly configured.
## Security and Privacy
Narratio handles private campaign material.
Rules:
- Do not store raw secrets in pipeline or session YAML.
- Use environment variable names or secret-file references for secret handling.
- Do not write raw secret values to manifests, logs, generated configs, or archive metadata.
- Treat transcripts, generated artifacts, prompts, reports, and logs as potentially sensitive.
- Avoid logging transcript or prompt content unless there is a deliberate diagnostic reason.
## Diagnostics
Diagnostics should be durable and discoverable, but distinct from canonical outputs.
Logs, reports, generated invocation/config files, and render-debug files support debugging. Transcript tiers and configured artifacts are pipeline products.
Manifest refs should preserve that distinction.
## Determinism
Where practical, Narratio should prefer deterministic behavior:
- stable local path layout;
- stable remote key layout;
- sorted upload order;
- predictable generated config files;
- repeatable command construction;
- tests that do not depend on live external services.
Run IDs and timestamps may be intentionally variable, but surrounding behavior should remain testable.
## Testing Expectations
Core behavior should be testable without live external services.
Tests should cover:
- config loading, defaults, and validation;
- CLI parsing and command construction;
- path helpers;
- manifest transitions;
- stage success, failure, skip, and resume behavior;
- adapter command construction;
- fake storage behavior;
- archive commit ordering;
- example config validity where practical.
Live S3, WhisperX, LLM, or subprocess integration tests should be explicit integration tests, not required for ordinary unit test runs.
## Documentation Expectations
Documentation must follow `docs/documentation/policy.md`.
Current behavior belongs in user-facing docs and `docs/internal/`. Future, planned, aspirational, experimental, or unimplemented work belongs only under `docs/roadmap/`.
`docs/architecture.md` should remain concise and principle-focused. It should not duplicate the full config reference, CLI reference, operations guide, or internal stage documentation.
## Non-Goals
Narratio is not:
- a generic DAG or workflow engine;
- a replacement configuration layer for Seriatim, Audita, or Scriptorium;
- a storage backend abstraction beyond the needs of this pipeline;
- a place to embed raw secrets;
- a place for stage logic to depend directly on AWS SDK types or downstream tool internals;
- a prompt-authoring system.

341
docs/cli.md Normal file
View File

@@ -0,0 +1,341 @@
# CLI
## Shortest Useful Command
```bash
narratio run 2026-04-04
```
This command uses default system discovery for `pipeline.yml`, the pipeline default campaign ID, and local `session.yml`. If local session discovery misses and S3 storage is configured, the positional session ID loads remote `session.yml` from the canonical session prefix.
Default pipeline and session discovery checks system config locations only. Pass `--config`, `--campaign-file`, and `--session` to use files from the current working directory. Pass `--campaign <id>` to select a campaign from `pipeline.campaigns.root`.
Ordinary local and remote `session.yml` files must be concrete YAML. Templates belong to `narratio session init`, which renders a configured campaign template before writing the concrete file.
## Command Overview
Top-level commands:
- `run <session_id>`: execute pipeline stages and persist manifest state.
- `run-stage <stage> <session_id>`: execute exactly one stage.
- `resume <session_id>`: continue from first non-succeeded stage unless forced.
- `analyze <session_id>`: force-rerun the analyze stage.
- `publish <session_id>`: force-rerun the archive stage.
- `clean <session_id>|--all`: remove local workspace/spool state.
- `session <subcommand>`: session-scoped helper commands.
Session subcommands:
- `session init <session_id>`: create local or remote `session.yml`.
- `session validate <session_id>`: run read-only preflight checks.
- `session status <session_id>`: inspect local/remote session state.
- `session plan <session_id>`: validate config, prepare workspace layout, and print stage run/skip decisions.
- `session restore <session_id>`: restore durable local state from committed remote archive state.
- `session artifacts <session_id>`: list effective artifact source IDs.
- `session locks <session_id>`: list archive promotion locks.
- `session locks add <session_id> <source>`: add or update a remote lock.
- `session locks remove <session_id> <source>`: remove a remote lock.
Unknown commands print usage and exit non-zero.
For config semantics, see [docs/config.md](./config.md). For operator lifecycle and recovery, see [docs/operations.md](./operations.md).
## Common Flags
Most session-aware commands accept:
- `--config <path>`: optional explicit `pipeline.yml` path.
- `--campaign <id>`: optional campaign ID selector.
- `--campaign-file <path>`: optional explicit `campaign.yml` path.
- `--session <path>`: optional explicit concrete `session.yml` path.
- `--previous-session-id <value>`: expected previous session identifier.
The positional `<session_id>` is required even when `--session` is provided. It is used as the expected session identity and as the remote session lookup value when local session discovery misses.
## Command Reference
### `run`
```bash
narratio run <session_id> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>] [--force] [--artifacts <name[,name...]>]
```
Purpose:
- Execute configured stages in canonical order.
Success output:
- `narratio run: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
Common failure cases:
- missing system default config/session paths when flags are omitted.
- missing selected campaign under `pipeline.campaigns.root`.
- missing local session plus missing/unavailable remote `session.yml`.
- templated `session.yml`; run `narratio session init` to generate concrete YAML.
- concrete session identity mismatch.
- unknown configured artifact key in `--artifacts`.
### `resume`
```bash
narratio resume <session_id> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>] [--force] [--artifacts <name[,name...]>]
```
Purpose:
- Continue from session-manifest stage status.
Success output:
- `narratio resume: session <session_id> has no remaining stages`
- or `narratio resume: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
### `run-stage`
```bash
narratio run-stage <stage> <session_id> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>] [--force] [--artifacts <name[,name...]>]
```
Valid stage names:
- `prepare`
- `transcribe`
- `merge`
- `polish`
- `normalize`
- `trim`
- `analyze`
- `archive`
- `notify`
Success output:
- `narratio run-stage: stage=<name> executed=<n> skipped=<n> force=<true|false>; manifest=<path>`
`--artifacts` is accepted only for `analyze` and `archive`.
### `analyze`
```bash
narratio analyze <session_id> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>] [--artifacts <name[,name...]>]
```
Purpose:
- Force-rerun the analyze stage.
- Shorter equivalent for `narratio run-stage analyze <session_id> --force`.
`analyze` is force-by-design and does not accept `--force`.
### `publish`
```bash
narratio publish <session_id> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>] [--artifacts <name[,name...]>]
```
Purpose:
- Force-rerun the archive stage.
- Shorter equivalent for `narratio run-stage archive <session_id> --force`.
`publish` is force-by-design and does not accept `--force` or a stage positional argument.
### `clean`
```bash
narratio clean <session_id> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>] [--dry-run] [--clear-cache]
narratio clean --all [--config <pipeline.yml>] [--dry-run] [--clear-cache]
```
Session cleanup deletes:
- `{workspace.root}/work/{campaign}/{session_id}`
- `{spool.root}/{campaign}/{session_id}`
All-session cleanup deletes:
- `{workspace.root}/work`
- the contents of `{spool.root}`, while preserving the spool root directory itself.
Cache behavior:
- cache is preserved by default.
- `--clear-cache` in session mode removes cached S3 audio files for the resolved session.
- `--all --clear-cache` removes the configured Narratio S3 audio cache namespace for the configured bucket/root prefix.
- `--clear-cache` does not delete arbitrary files under `pipeline.cache.root`.
### `session plan`
```bash
narratio session plan <session_id> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>] [--force]
```
Purpose:
- Validate config, load secrets if configured, prepare workdir, and print stage run/skip decisions.
Success output includes:
- `narratio session plan: workdir prepared at <path>`
- one line per stage (`<stage>: run|skip`)
- `totals: run=<n> skip=<n>`
### `session status`
```bash
narratio session status <session_id> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>]
```
Output includes:
- session ID, campaign, workspace, and session config source.
- local manifest state when present.
- remote current archive state when storage is configured.
- catalog-based promoted output availability for expected transcript and artifact sources.
- effective archive locks and conservative next actions.
### `session validate`
```bash
narratio session validate <session_id> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>]
```
Checks include:
- effective config and session source.
- stable input files.
- local or remote audio availability.
- previous-session requirements.
- archive promotions and effective locks.
Warnings do not fail the command. Any `ERROR` finding exits non-zero.
### `session init`
```bash
narratio session init <session_id> --output ./session.yml
narratio session init <session_id> --remote
narratio session init <session_id> --config <pipeline.yml> --campaign icewind --remote
narratio session init <session_id> --config <pipeline.yml> --campaign-file ./campaign.yml --remote
```
Additional flags:
- `--previous-session-id <value>`
- `--date <value>`
- `--title <value>`
- `--audio-s3-prefix <prefix>`: defaults to `audio/` when neither audio flag is provided.
- `--audio-dir <path>`: local audio directory; mutually exclusive with `--audio-s3-prefix`.
- `--force`: overwrite existing local or remote target.
Behavior:
- exactly one of `--output` or `--remote` is required.
- `--config`, `--campaign`, and `--campaign-file` are optional overrides; omitted campaign selection uses `pipeline.campaigns.default_campaign_id`.
- `--campaign <id>` selects a campaign under `pipeline.campaigns.root`.
- `--campaign-file <path>` loads an explicit campaign file.
- if `campaign.yml` sets `session_template_file`, the template path is resolved relative to `campaign.yml` and rendered from init flags.
- if no session template is configured, a minimal concrete session file is generated directly.
- template variables must be supplied by matching flags, and supplied template-related flags must be used by the template.
- remote writes target `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/session.yml`.
- existing local or remote targets fail unless `--force` is passed.
- remote writes use existence checks, not compare-and-swap.
### `session restore`
```bash
narratio session restore <session_id> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>] [--dry-run] [--force] [--include-audio]
```
Purpose:
- Restore durable session state from the committed remote archive current state.
- Default restore installs `manifest.json`, `transcripts/**`, and `artifacts/**` from the current session archive.
- When configured previous-session inputs require it, restore reconstructs `previous/**` from the previous session's committed current archive.
- `audio/**` is restored only with `--include-audio`.
Dry-run output may include planned previous-cache downloads. Existing differing files under `previous/**` follow the normal restore conflict policy and require `--force` to overwrite.
When `--include-audio` is set, S3 audio files are restored through the shared audio cache. Cache hits avoid re-downloading large audio objects.
### `session artifacts`
```bash
narratio session artifacts <session_id> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>] [--remote]
```
Purpose:
- List built-in, configured, previous-session, promoted, and locked artifact sources.
`--remote` checks promoted top-level object availability through the storage adapter. Remote markers appear only in the `Promoted` section, which reports each configured archive promotion destination and includes `dest=<path>` when that destination differs from the source's canonical path.
### `session locks`
```bash
narratio session locks <session_id> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>]
narratio session locks add <session_id> <source> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>] [--reason <text>] [--force]
narratio session locks remove <session_id> <source> [--config <pipeline.yml>] [--campaign <id>] [--campaign-file <campaign.yml>] [--session <session.yml>] [--previous-session-id <id>]
```
Behavior:
- list mode prints effective locks from static `pipeline.archive.locks` and remote `{session_prefix}/locks.yml`.
- `locks add` writes only the remote lock store and fails if the source is already locked by pipeline config.
- `locks remove` removes only remote locks and cannot remove static pipeline locks.
- `locks add --force` is required to update an existing remote lock reason.
## Common Workflows
Default-discovery run:
```bash
narratio run 2026-04-04
```
Run only selected analyze artifacts:
```bash
narratio run 2026-04-04 --artifacts session_recap,player_handout
```
Resume with selected analyze artifacts:
```bash
narratio resume 2026-04-04 --artifacts player_handout
```
Force-rerun analyze with selected artifacts:
```bash
narratio analyze 2026-04-04 --artifacts player_handout
```
Force-rerun archive publishing:
```bash
narratio publish 2026-04-04
```
Preview restore actions without writes:
```bash
narratio session restore 2026-04-04 --dry-run
```
Restore and then force analyze:
```bash
narratio session restore 2026-04-04
narratio analyze 2026-04-04
```
Rehydrate canonical previous-session inputs after artifact-input changes:
```bash
narratio run-stage prepare 2026-04-04 --force
```
Reset local state before testing restore:
```bash
narratio clean 2026-04-04 --dry-run
narratio clean 2026-04-04
narratio session restore 2026-04-04 --include-audio
```
Clean all local sessions while keeping cached S3 audio:
```bash
narratio clean --all
```
## `--artifacts` and `--force`
- `--artifacts` filters which configured artifacts are executable when analyze runs and which configured artifact promotions archive publishes.
- `--artifacts` does not imply `--force`.
- if analyze is already `succeeded` and `--force` is not set, runner-level skip still applies.
- `--artifacts` does not suppress built-in transcript or bounds promotions.

525
docs/config.md Normal file
View File

@@ -0,0 +1,525 @@
# Configuration
## 1. Overview
Narratio loads three YAML files:
- `pipeline.yml`: pipeline-level runtime configuration.
- `campaign.yml`: stable campaign identity and campaign-level input defaults.
- `session.yml`: per-session metadata and input selection, loaded locally or from the configured S3 backend.
These commands load and validate all three files before running:
- `narratio run`
- `narratio resume`
- `narratio run-stage`
- `narratio analyze`
- `narratio publish`
- `narratio session plan`
- `narratio session status`
- `narratio session validate`
- `narratio session restore`
- `narratio session artifacts`
- `narratio session locks`
- `narratio clean <session_id>`
Behavior:
- strict YAML decode is enabled (`KnownFields(true)`): unknown fields fail.
- ordinary local and remote `session.yml` files must be concrete YAML; template placeholders are rejected.
- defaults are applied for optional pipeline fields.
- campaign identity is selected by ID from the pipeline campaign registry unless `--campaign-file` is used.
- campaign-level stable input paths fill missing session input paths.
- session-level stable input paths override campaign-level input paths.
- campaign config may point `session init` to a session template.
- validation enforces required fields, value formats, and cross-field constraints.
## 2. Config file discovery
These commands use the same config discovery behavior:
- `narratio run`
- `narratio resume`
- `narratio run-stage`
- `narratio analyze`
- `narratio publish`
- `narratio session plan`
- `narratio session status`
- `narratio session validate`
- `narratio session restore`
- `narratio session artifacts`
- `narratio session locks`
- `narratio clean <session_id>`
Pipeline config lookup:
- if `--config <path>` is provided, that path is used.
- if omitted, Narratio searches in order:
1. `/usr/local/etc/narratio/pipeline.yml`
2. `/etc/narratio/pipeline.yml`
- first existing file wins.
Campaign config lookup:
- pipeline config is loaded first.
- if `--campaign-file <path>` is provided, that path is used.
- otherwise, if `--campaign <id>` is provided, Narratio loads:
- `{pipeline.campaigns.root}/{id}/campaign.yml`
- otherwise, Narratio uses `pipeline.campaigns.default_campaign_id` and loads:
- `{pipeline.campaigns.root}/{default_campaign_id}/campaign.yml`
- `--campaign` and `--campaign-file` are mutually exclusive.
- campaign IDs must be single path segments, not paths.
Session config lookup:
- if `--session <path>` is provided, that path is used.
- if `--session` is omitted, Narratio searches locally in order:
1. `/usr/local/etc/narratio/session.yml`
2. `/etc/narratio/session.yml`
- first existing local file wins.
- if no local session file is found, a positional `<session_id>` is present, storage is configured, and campaign identity is resolved, Narratio loads remote `session.yml` from:
- `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/session.yml`
- local discovery always runs before remote fallback.
- local files in the current working directory are used only when passed explicitly, for example `--config ./pipeline.yml --campaign-file ./campaign.yml --session ./session.yml`.
## 3. Session templating
Template behavior for local and remote `session.yml` loaded by downstream commands:
- downstream commands do not render templates.
- local and remote `session.yml` must be concrete.
- any `{{ ... }}` placeholder in loaded `session.yml` fails with guidance to run `narratio session init`.
- if concrete `session_id` mismatches the positional `<session_id>`, load fails.
- if concrete `previous_session_id` mismatches `--previous-session-id`, load fails.
Template behavior for `narratio session init`:
- `campaign.yml` may set `session_template_file`.
- relative template paths resolve relative to `campaign.yml`.
- supported init template variables:
- `{{ session_id }}`
- `{{ previous_session_id }}`
- `{{ date }}`
- `{{ title }}`
- `{{ audio_s3_prefix }}`
- `{{ audio_dir }}`
- each template variable must be supplied by the matching `session init` flag.
- template-related flags such as `--date`, `--title`, `--audio-s3-prefix`, `--audio-dir`, and `--previous-session-id` fail if the configured template does not use them.
- rendered output is strict-decoded and validated before it is written locally or remotely.
- if `session_template_file` is omitted, `session init` generates the minimal concrete session YAML directly.
## 4. Minimal config set
### `pipeline.yml`
```yaml
campaigns:
root: /usr/local/share/narratio/campaigns
default_campaign_id: sample-campaign
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"
```
Why this is sufficient:
- `whisperx.transcribe_url` is required.
- `campaigns.default_campaign_id` selects the default campaign when `--campaign` is omitted.
- `workspace.root` defaults to `/var/lib/narratio`.
- optional sections (`seriatim`, `audita`, `archive`, `scriptorium`, `trim`, `normalize`, etc.) receive defaults or stay inactive.
### `campaign.yml`
```yaml
campaign_id: sample-campaign
session_template_file: ./session.template.yml
inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
```
Why this is sufficient:
- `campaign_id` supplies the stable campaign identity.
- stable input files are required and resolve relative to `campaign.yml` when copied during `prepare`.
### `session.yml`
```yaml
session_id: 2026-05-03
inputs:
audio_dir: ./audio
```
Why this is sufficient:
- `session_id` is required.
- `campaign` can be omitted because it is supplied by `campaign.yml`.
- stable input paths can be omitted because `campaign.yml` supplies defaults.
- local `audio_dir` resolves relative to `session.yml`.
Minimal local-file usage:
```bash
narratio run 2026-05-03 --config /path/to/pipeline.yml --campaign sample-campaign --session ./session.yml
narratio run 2026-05-03 --config /path/to/pipeline.yml --campaign-file ./campaign.yml --session ./session.yml
```
Previous-session-enabled variant:
```yaml
session_id: 2026-05-03
previous_session_id: 2026-04-26
inputs:
audio_dir: ./audio
```
```bash
narratio run 2026-05-03 --config /path/to/pipeline.yml --campaign sample-campaign --session ./session.yml --previous-session-id 2026-04-26
```
## 5. Production-oriented config set
### `pipeline.yml`
```yaml
workspace:
root: /var/lib/narratio/workspace
cleanup_after_archive: true
storage:
backend: s3
s3:
bucket: my-dnd-archive
root_prefix: dnd
region: us-east-1
access_key_id_env: OBJECT_STORAGE_KEY_ID
secret_access_key_env: OBJECT_STORAGE_KEY
campaigns:
root: /srv/narratio/campaigns
default_campaign_id: forsaken
spool:
root: /var/spool/narratio
delete_audio_after_archive: true
cache:
root: /var/cache/narratio
s3_audio: true
archive:
enabled: true
upload_run: true
promote_artifacts:
- source: narratio.transcript.final_trimmed
dest: transcripts/final.trimmed.json
required: true
- source: narratio.artifact.session_recap
dest: artifacts/session_recap.md
required: true
locks:
- source: narratio.artifact.session_recap
reason: Final recap was manually edited.
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"
scriptorium:
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
output_path: artifacts/session_recap.md
inputs:
transcript:
source: narratio.transcript.final_trimmed
required: true
previous_recap:
source: narratio.previous_session.artifact.session_recap
required: false
```
### `campaign.yml`
```yaml
campaign_id: forsaken
inputs:
speakers_file: /srv/narratio/campaigns/forsaken/speakers.yml
autocorrect_file: /srv/narratio/campaigns/forsaken/autocorrect.yml
glossary_file: /srv/narratio/campaigns/forsaken/glossary.yml
```
### Local `session.yml`
```yaml
session_id: 2026-05-03
previous_session_id: 2026-04-26
date: 2026-05-03
title: The Black Cabin
inputs:
audio_s3:
prefix: audio/
```
### S3-first session config
For S3-first operation, upload the same `session.yml` content to:
```text
{root_prefix}/campaigns/{campaign}/sessions/{session_id}/session.yml
```
Then run with explicit or discovered pipeline/campaign config and no `--session`:
```bash
narratio run 2026-05-03 --config /usr/local/etc/narratio/pipeline.yml --campaign forsaken --previous-session-id 2026-04-26
```
Operational notes:
- archive promotion is explicit and source-based via `archive.promote_artifacts`.
- `source` is required; `dest` is optional and derived when omitted.
- `archive.locks` skips top-level promotion overwrites for static locked sources while preserving run-local uploads.
- operator-created mutable locks are stored at `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/locks.yml` and are merged with static locks.
- Narratio does not auto-promote all generated analyze artifacts.
- `restore` reads the same config/campaign/session inputs and restore scope is bounded by committed archive current state.
- `clean` removes workspace/spool state by default and preserves `pipeline.cache.root` unless `--clear-cache` is passed.
## 6. Full pipeline reference
| Path | Type | Required | Default |
| --- | --- | --- | --- |
| `pipeline.workspace.root` | string | No | `/var/lib/narratio` |
| `pipeline.workspace.cleanup_after_archive` | bool | No | `false` |
| `pipeline.campaigns.root` | string | No | `/usr/local/share/narratio/campaigns` |
| `pipeline.campaigns.default_campaign_id` | string | No | empty |
| `pipeline.secrets.env_dir` | string | Conditional | none |
| `pipeline.storage.backend` | string | No | empty |
| `pipeline.storage.s3.bucket` | string | Conditional | empty |
| `pipeline.storage.s3.root_prefix` | string | No | `dnd` |
| `pipeline.storage.s3.region` | string | No | empty |
| `pipeline.storage.s3.endpoint` | string | No | empty |
| `pipeline.storage.s3.force_path_style` | bool | No | `false` |
| `pipeline.storage.s3.access_key_id_env` | string | No | `OBJECT_STORAGE_KEY_ID` |
| `pipeline.storage.s3.secret_access_key_env` | string | No | `OBJECT_STORAGE_KEY` |
| `pipeline.spool.root` | string | No | `/var/spool/narratio` |
| `pipeline.spool.delete_audio_after_archive` | bool | No | `false` |
| `pipeline.cache.root` | string | No | `/var/cache/narratio` |
| `pipeline.cache.s3_audio` | bool | No | `true` |
| `pipeline.archive.enabled` | bool | No | `true` |
| `pipeline.archive.upload_run` | bool | No | `true` |
| `pipeline.archive.promote_artifacts[]` | list | No | final-trimmed transcript rule |
| `pipeline.archive.promote_artifacts[].source` | string | Yes (per rule) | none |
| `pipeline.archive.promote_artifacts[].dest` | string | No | derived from source |
| `pipeline.archive.promote_artifacts[].required` | bool | No | `true` |
| `pipeline.archive.locks[]` | list | No | empty |
| `pipeline.archive.locks[].source` | string | Yes (per lock) | none |
| `pipeline.archive.locks[].reason` | string | No | empty |
| `pipeline.whisperx.transcribe_url` | string | Yes | none |
| `pipeline.whisperx.language` | string | No | `en` |
| `pipeline.whisperx.timeout` | duration string | No | `30m` |
| `pipeline.whisperx.retries` | int | No | `3` |
| `pipeline.whisperx.retry_delay` | duration string | No | `2s` |
| `pipeline.whisperx.concurrency` | int | No | `2` |
| `pipeline.seriatim.binary` | string | No | `seriatim` |
| `pipeline.seriatim.timeout` | duration string | No | `10m` |
| `pipeline.seriatim.output_schema` | string | No | `seriatim-intermediate` |
| `pipeline.seriatim.coalesce_gap` | float | No | `3.0` |
| `pipeline.seriatim.report` | bool | No | `true` |
| `pipeline.seriatim.env.overlap_word_run_gap` | float | No | unset |
| `pipeline.seriatim.env.overlap_word_run_reorder_window` | float | No | unset |
| `pipeline.seriatim.env.backchannel_max_duration` | float | No | unset |
| `pipeline.seriatim.env.filler_max_duration` | float | No | unset |
| `pipeline.audita.binary` | string | No | `audita` |
| `pipeline.audita.timeout` | duration string | No | `3h` |
| `pipeline.audita.llm_api_key_env` | string | No | empty |
| `pipeline.audita.modules[]` | list[string] | No | empty |
| `pipeline.audita.base_url` | string | No | empty |
| `pipeline.audita.model` | string | No | empty |
| `pipeline.audita.total_llm_concurrency` | int | No | unset |
| `pipeline.audita.proposal_llm_concurrency` | int | No | unset |
| `pipeline.audita.validation_model` | string | No | empty |
| `pipeline.audita.validation_llm_concurrency` | int | No | unset |
| `pipeline.audita.transcript_description` | string | No | empty |
| `pipeline.audita.config_path` | string | No | empty |
| `pipeline.audita.output_schema` | string | No | empty |
| `pipeline.audita.work_dir_retention` | string | No | empty |
| `pipeline.audita.report` | bool | No | `true` |
| `pipeline.normalize.output_path` | string | No | `transcripts/final.json` |
| `pipeline.normalize.output_schema` | string | No | `seriatim-intermediate` |
| `pipeline.normalize.report` | bool | No | `true` |
| `pipeline.trim.enabled` | bool | No | `false` |
| `pipeline.trim.output_path` | string | Conditional | none |
| `pipeline.trim.bounds.prompt_id` | string | Conditional | none |
| `pipeline.trim.bounds.profile_id` | string | No | empty |
| `pipeline.trim.bounds.transcript_input_name` | string | Conditional | none |
| `pipeline.trim.bounds.output_path` | string | Conditional | none |
| `pipeline.trim.bounds.timeout` | duration string | No | `10m` |
| `pipeline.trim.bounds.render_debug` | bool | No | `false` |
| `pipeline.trim.bounds.render_output_path` | string | Conditional | none |
| `pipeline.trim.seriatim.report` | bool | No | `false` |
| `pipeline.scriptorium.binary` | string | No | `scriptorium` |
| `pipeline.scriptorium.config_path` | string | No | empty |
| `pipeline.scriptorium.timeout` | duration string | No | `10m` |
| `pipeline.scriptorium.render_debug` | bool | No | `false` |
| `pipeline.scriptorium.artifacts` | map | No | empty |
| `pipeline.scriptorium.artifacts.<name>.enabled` | bool | No | `false` |
| `pipeline.scriptorium.artifacts.<name>.depends_on[]` | list[string] | No | empty |
| `pipeline.scriptorium.artifacts.<name>.render_debug` | bool | No | unset |
| `pipeline.scriptorium.artifacts.<name>.prompt_id` | string | Conditional | none |
| `pipeline.scriptorium.artifacts.<name>.profile_id` | string | No | empty |
| `pipeline.scriptorium.artifacts.<name>.output_path` | string | Conditional | none |
| `pipeline.scriptorium.artifacts.<name>.timeout` | duration string | No | empty |
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source` | string | Conditional | none |
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.artifact` | string | No | empty |
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.path` | string | No | empty |
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.required` | bool | No | `false` |
| `pipeline.scriptorium.artifacts.<name>.vars.<key>` | map value | No | empty |
| `pipeline.notification.backend` | string | No | empty |
| `pipeline.notification.recipient` | string | No | empty |
| `pipeline.notification.timeout` | duration string | No | empty |
Scriptorium artifact-key and dependency rules:
- artifact keys must match `^[a-z][a-z0-9_]*$`.
- enabled artifacts require `prompt_id` and `output_path`.
- `output_path` must be relative, traversal-safe, and under `artifacts/`.
- configured artifact input sources use `narratio.artifact.<name>`.
- if input source references `narratio.artifact.<name>`, artifact `<name>` must exist and must be listed in `depends_on`.
- every `depends_on` entry must be a configured artifact key.
- self-dependency is rejected.
- enabled dependency cycles are rejected.
- any artifact referenced by `depends_on` or `narratio.artifact.<name>` source must define `output_path` (even if not enabled).
Allowed `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source` values:
- `narratio.previous_session.artifact.<configured_artifact_key>`
- `narratio.transcript.base`
- `narratio.transcript.polished`
- `narratio.transcript.final`
- `narratio.transcript.final_trimmed`
- `narratio.bounds.session`
- `narratio.artifact.<configured_artifact_key>`
`pipeline.archive.promote_artifacts[].source` values:
- `narratio.transcript.base`
- `narratio.transcript.polished`
- `narratio.transcript.final`
- `narratio.transcript.final_trimmed`
- `narratio.bounds.session`
- `narratio.artifact.<configured_artifact_key>`
`pipeline.archive.locks[].source` accepts the same source values as `pipeline.archive.promote_artifacts[].source`.
Archive promotion destination rules:
- `dest` must be a clean relative path (not absolute, no traversal).
- duplicate `dest` values are rejected.
- if `dest` is omitted:
- built-in sources derive their canonical destination path;
- configured sources derive from `pipeline.scriptorium.artifacts.<name>.output_path`;
- derivation failure is a config validation error.
Archive lock rules:
- locks are source-based and do not accept `dest`.
- duplicate lock sources are rejected.
- static `pipeline.archive.locks` win over remote mutable locks for the same source.
- locked promotions are recorded as intentional skips in archive metadata.
- locked required promotions do not fail archive by default.
- ordinary `--force` reruns do not override locks.
Remote mutable lock store:
- path: `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/locks.yml`.
- strict YAML shape: top-level `locks`, each with `source` and optional `reason`.
- `narratio session locks add` and `narratio session locks remove` mutate only the remote lock store.
- writes use existence checks plus `--force` for updates; they are not compare-and-swap atomic.
Restore-related implications:
- restore remote identity requires archive S3 identity to resolve (`pipeline.storage.s3.bucket` and session prefix derivation inputs).
- restore scope considers committed current state and durable paths (`manifest.json`, `transcripts/**`, `artifacts/**`, `previous/**`, optional `audio/**`).
- S3 audio downloads use `pipeline.spool.root` for active downloads and `pipeline.cache.root` for reusable cached audio when `pipeline.cache.s3_audio` is true.
- `pipeline.cache.root` is durable local cache state. It is not workspace state and is preserved by default by `narratio clean`.
## 7. Full campaign reference
| Path | Type | Required | Default |
| --- | --- | --- | --- |
| `campaign.campaign_id` | string | Yes | none |
| `campaign.session_template_file` | string | No | none |
| `campaign.inputs.speakers_file` | string | Yes | none |
| `campaign.inputs.autocorrect_file` | string | Yes | none |
| `campaign.inputs.glossary_file` | string | Yes | none |
Campaign input paths and `campaign.session_template_file` may be absolute or relative. Relative paths resolve from the directory containing `campaign.yml`.
## 8. Full session reference
| Path | Type | Required | Default |
| --- | --- | --- | --- |
| `session.session_id` | string | Yes | none |
| `session.previous_session_id` | string | No | empty |
| `session.campaign` | string | No | `campaign.campaign_id` |
| `session.date` | string | No | empty |
| `session.title` | string | No | empty |
| `session.inputs.audio_dir` | string | Conditional | empty |
| `session.inputs.audio_files[]` | list[string] | Conditional | empty |
| `session.inputs.audio_s3.prefix` | string | Conditional | none |
| `session.inputs.speakers_file` | string | No | `campaign.inputs.speakers_file` |
| `session.inputs.autocorrect_file` | string | No | `campaign.inputs.autocorrect_file` |
| `session.inputs.glossary_file` | string | No | `campaign.inputs.glossary_file` |
Session input paths may be absolute or relative. Relative audio paths and session-level stable input overrides resolve from the directory containing `session.yml`. If both `campaign.yml` and `session.yml` specify campaign identity, the values must match.
Audio-source rule:
- configure exactly one mode:
- `audio_dir`, or
- `audio_files` (at least one), or
- `audio_s3.prefix`
- `audio_s3` cannot be combined with local audio fields.
Previous-session rule:
- if `session.previous_session_id` is set, it must not equal `session.session_id`.
- canonical previous-session sources (`narratio.previous_session.artifact.<name>`) are hydrated during `prepare` from archive current state when required by enabled configured artifacts.
## 9. Secrets
Narratio supports filesystem-based secret injection via `pipeline.secrets.env_dir`.
Behavior:
- `env_dir` may be absolute or relative.
- relative `env_dir` resolves from current working directory.
- files with valid env-var names (`[A-Za-z_][A-Za-z0-9_]*`) are loaded.
- values are loaded from file contents with trailing newline trimming.
- existing process env vars are preserved.
- invalid names and subdirectories are skipped.
- missing/unreadable `env_dir` fails command execution.
Guidance:
- do not put secret values directly in YAML.
- configure env var names in config and provide values via env/secrets files.
## 10. Examples
Maintained examples:
- `examples/pipeline.minimal.yml`
- `examples/pipeline.production.yml`
- `examples/pipeline.full.annotated.yml`
- `examples/campaigns/sample-campaign/campaign.yml`
- `examples/campaigns/sample-campaign/speakers.yml`
- `examples/campaigns/sample-campaign/autocorrect.yml`
- `examples/campaigns/sample-campaign/glossary.yml`
- `examples/campaigns/sample-campaign/session.template.yml`
- `examples/session.local-audio.yml`
- `examples/session.s3-audio.yml`
These examples are validated by `internal/config` tests.

94
docs/development.md Normal file
View File

@@ -0,0 +1,94 @@
# Development Guide
## Purpose
Canonical contributor workflow and engineering conventions for implemented Narratio behavior.
## Repository layout
- `cmd/narratio/`: CLI entrypoint.
- `internal/app/`: command handlers, plan/run/resume orchestration, cleanup gates, secrets loading.
- `internal/config/`: strict YAML loading, defaults, and validation.
- `internal/stage/`: stage implementations and stage registry/order.
- `internal/adapters/`: external boundary adapters (WhisperX, Seriatim, Audita, Scriptorium, storage, notify).
- `internal/manifest/`: session/run manifest types and persistence.
- `internal/artifacts/`: canonical local/remote path helpers and local artifact store.
- `docs/`: canonical documentation set.
- `examples/`: maintained config examples used by tests.
## Build and test commands
- Run focused CLI behavior checks:
```bash
go test ./internal/app -run TestExecute -v
```
- Run config example load/validate checks:
```bash
go test ./internal/config -run TestExamplesLoadAndValidate -v
```
- Run full test suite:
```bash
go test ./...
```
## Coding conventions
- Keep orchestration explicit and stage-driven; do not introduce generic workflow/DAG abstractions.
- Keep external-system details inside adapter packages; stages should consume Narratio-level contracts only.
- Use centralized path helpers from `internal/artifacts` rather than ad hoc path concatenation.
- Preserve manifest-driven state transitions (`running`, `succeeded`, `failed`, `skipped`, `stale`) as the source of run progress.
- Keep user/operator docs implementation-accurate; planned work belongs only under `docs/roadmap/`.
For design principles and invariants, see [docs/architecture.md](./architecture.md). For stage/adapter contracts, see [docs/internal/README.md](./internal/README.md).
## Dependency policy
- Prefer Go standard library where practical.
- Add third-party dependencies only when they provide clear value for required behavior.
- Keep dependency additions narrow to the boundary package that needs them.
## Change playbooks
### Add config fields
1. Add fields to config structs in `internal/config`.
2. Set defaults in `internal/config/defaults.go` when appropriate.
3. Add validation rules in `internal/config/validate.go`.
4. Add or update load/validate tests in `internal/config/*_test.go`.
5. Update canonical config docs and examples:
- [docs/config.md](./config.md)
- relevant files under `examples/`
### Add CLI flags or commands
1. Update command parsing and behavior in `internal/app`.
2. Add or update command tests (`TestExecute` and command-specific tests).
3. Update [docs/cli.md](./cli.md) and, if operator workflow changes, [docs/operations.md](./operations.md).
Remote-storage commands must obtain object storage through the app-level command object-store helper. Do not call `storage.NewObjectStoreFromConfig` directly from command handlers; the helper loads configured filesystem secrets before constructing the storage adapter.
### Add or modify stages/adapters
1. Implement stage behavior in `internal/stage` with clear input/output boundaries.
2. Keep external transport/subprocess details in `internal/adapters`.
3. Preserve manifest and promotion semantics expected by runner and archive logic.
4. Add/update stage and adapter tests.
5. Update internal component contracts in `docs/internal/`.
### Update examples
1. Keep canonical examples only in `examples/`.
2. Ensure examples load and validate through runtime config paths.
3. Update `internal/config/load_validate_test.go` as needed.
4. Update links in `docs/config.md` if example filenames change.
### Update docs and roadmap
1. Keep implemented behavior in canonical docs (`README`, `docs/*.md`, `docs/internal/`).
2. Keep planned/unimplemented behavior only in `docs/roadmap/`.
3. After completing roadmap items, remove or mark them complete in `docs/roadmap/documentation.md`.
4. Run a link/path sweep before finalizing changes.

View File

@@ -0,0 +1,356 @@
# Go Project Documentation Policy
## Purpose
Project documentation must help four audiences:
1. users who need to run the application;
2. administrators/operators who need to configure and operate it;
3. developers who need to understand and change it safely;
4. LLM coding agents that need clear scope, boundaries, and invariants.
Docs should be accurate, concise, task-oriented, and organized by audience. Prefer links to canonical docs over repetition.
## Core Rules
### 1. Keep docs concise
Each document should cover a defined scope and only the essentials for that scope.
Avoid:
- long background explanations;
- repeated reference material;
- implementation detail in user-facing docs;
- aspirational language outside roadmap docs;
- verbose examples where one minimal example is clearer.
### 2. Document only implemented behavior outside roadmap files
Unimplemented, planned, aspirational, experimental, or future work may be described only under:
- `docs/roadmap/`
No other documentation file, including `README.md`, should describe code, features, modules, stages, commands, config fields, or behaviors that do not currently exist.
If a feature is partial, non-roadmap docs may describe only the implemented portion and its current boundary.
### 3. Use canonical homes
Each type of information should have one canonical location.
Canonical homes:
- project purpose and quickstart: `README.md`
- development principles: `docs/architecture.md`
- configuration reference: `docs/config.md`
- CLI reference: `docs/cli.md`
- operations and recovery: `docs/operations.md`
- troubleshooting: `docs/troubleshooting.md`
- implemented internals: `docs/internal/`
- future work: `docs/roadmap/`
- contributor workflow: `docs/development.md`
- copyable examples: `examples/`
Other files should summarize briefly and link to the canonical source.
### 4. Keep examples real
Examples should be valid, maintained, and free of secrets.
Where practical:
- example configs should load successfully;
- example commands should match real CLI syntax;
- important examples should be covered by tests.
## Documentation Profiles
All projects require:
- `README.md`
- `docs/architecture.md`
Additional docs depend on the project.
### Small library
Recommended:
- `docs/development.md`, if contributor conventions are non-obvious
### Simple CLI
Required:
- `docs/cli.md`
Recommended:
- `docs/development.md`
### Config-driven CLI
Required:
- `docs/cli.md`
- `docs/config.md`
Recommended:
- `examples/`
- `docs/development.md`
### Stateful or operator-facing application
Required:
- `docs/cli.md`, if CLI-based
- `docs/config.md`, if config-driven
- `docs/operations.md`
Recommended:
- `docs/troubleshooting.md`
- `examples/`
- `docs/development.md`
### Modular, staged, service-oriented, or orchestration application
Required:
- `docs/cli.md`, if CLI-based
- `docs/config.md`, if config-driven
- `docs/operations.md`
- `docs/internal/`
- `docs/development.md`
Recommended:
- `docs/troubleshooting.md`
- validated examples under `examples/`
## Required Documents
### README.md
**Audience:** users, administrators, operators
The README is the outward-facing project orientation page.
It should include, in order:
1. concise description;
2. elevator pitch;
3. shortest useful command or usage example;
4. links to targeted docs.
The README should be short. It is not a manual.
The “shortest useful command” means the simplest command that performs the projects core use case. (It does not mean `app --help`.)
### docs/architecture.md
**Audience:** developers, LLM coding agents
`docs/architecture.md` is required for every project.
It is an inward-facing development policy document. It should describe how the project is intended to be built and changed.
It should include:
- project shape;
- core design principles;
- package and boundary philosophy;
- state/persistence philosophy, if applicable;
- external integration philosophy, if applicable;
- error-handling and logging principles;
- testing expectations;
- documentation expectations;
- architectural invariants;
- explicit non-goals, if useful.
For small projects, this file may be brief. It may simply state that the project is intentionally narrow, monolithic, and dependency-light.
### docs/config.md
**Audience:** administrators, operators, advanced users
Required for applications with configuration files.
It should include, in order:
1. config file locations and discovery precedence;
2. minimal working config;
3. production-oriented config;
4. full configuration reference;
5. secrets handling, if applicable;
6. links to maintained examples.
The full configuration reference should be canonical.
### docs/cli.md
**Audience:** users, administrators, operators
Required for CLI applications.
It should include, in order:
1. shortest useful command;
2. command overview;
3. complete flag reference;
4. common workflows;
5. diagnostic or recovery commands, if applicable.
Explain when commands are useful, not just their syntax.
### docs/operations.md
**Audience:** administrators, operators
Required for applications that maintain state, support resume behavior, run multiple stages, write durable artifacts, use remote storage, or require recovery procedures.
It should cover:
- normal workflow;
- filesystem layout;
- remote storage layout, if applicable;
- logs and manifests;
- resume/retry behavior;
- cleanup behavior;
- archive/backup behavior;
- safe recovery procedures;
- operational caveats.
### docs/troubleshooting.md
**Audience:** administrators, operators
Recommended once recurring failure modes exist.
Each entry should include:
- symptom;
- likely cause;
- diagnostic command or inspection step;
- safe fix;
- relevant links.
### docs/development.md
**Audience:** developers, LLM coding agents
Required for projects maintained by humans and LLM coding agents.
It should include:
- repository layout;
- build/test commands;
- coding conventions;
- dependency policy;
- how to add config fields;
- how to add CLI flags;
- how to add stages/modules/adapters, if applicable;
- how to update examples;
- documentation update expectations.
### docs/internal/
**Audience:** developers, LLM coding agents
Required for modular, staged, service-oriented, or orchestration projects.
This directory describes implemented internal components. It is not the roadmap.
Use one file per major component where useful.
Each component doc should include:
1. purpose;
2. inputs and outputs;
3. boundaries;
4. config fields used;
5. external adapters used;
6. state or manifest behavior, if applicable;
7. skip/resume behavior, if applicable;
8. failure behavior;
9. tests to inspect before changing;
10. architectural invariants.
### docs/roadmap/
**Audience:** maintainers, developers, LLM coding agents
This is the only place for planned, future, aspirational, experimental, or unimplemented work.
Roadmap docs should clearly distinguish:
- proposed work;
- accepted plans;
- deferred ideas;
- rejected ideas;
- implementation prompts or task breakdowns, if useful.
Roadmap docs should not be confused with current behavior.
### docs/integrations/
**Audience:** developers, LLM coding agents
Required for projects that depend on external CLIs, APIs, services, protocols, or file formats where the integration contract is important to maintain.
This directory contains concise, versioned reference notes for external integration contracts. It should document only the parts of the external system that this project actually uses.
Use one file per integration where useful.
## Examples Directory
Projects with non-trivial configuration or workflows should include `examples/`.
Useful examples include:
- minimal working config;
- production-oriented config;
- full annotated config;
- local development config;
- remote/object-storage config;
- minimal session/input file.
Examples should be valid, maintained, tested when practical, and linked from relevant docs.
## Security and Privacy
Docs and examples must not include:
- real API keys;
- tokens;
- passwords;
- private keys;
- private environment dumps;
- sensitive user data;
- raw private transcripts;
- private infrastructure details unless intentionally public.
Document secret-handling mechanisms, not actual secret values.
## Maintenance Rules
When docs change, verify the affected behavior.
Where practical:
- load example config files in tests;
- test CLI examples or command parser behavior;
- validate documented flags against real flags;
- remove stale references;
- update links after renames;
- keep roadmap content out of non-roadmap docs.
If documentation and code disagree, fix the documentation and/or open a roadmap item; do not leave aspirational behavior in current-behavior docs.
Documentation is complete only when it matches the current code.
## Documentation Change Checklist
Before merging documentation changes, verify:
- README is concise and orientation-focused.
- `docs/architecture.md` describes development principles.
- Future work appears only under `docs/roadmap/`.
- User-facing docs avoid unnecessary internals.
- Developer-facing docs preserve boundaries and invariants.
- Config examples match the schema.
- CLI examples match real commands and flags.
- Defaults appear in the canonical config reference.
- No secrets or private data are included.
- Links are accurate.

View File

@@ -0,0 +1,15 @@
# Integration Documentation Index
## Audience
Developers and LLM coding agents changing Narratio's external integration contracts.
## Scope
Implemented-only reference notes for the external systems Narratio currently integrates with.
## Integration Docs
- `audita.md`: Audita adapter invocation and validation contract.
- `seriatim.md`: Seriatim normalize/merge/trim adapter contract.
- `scriptorium.md`: Scriptorium run/render adapter contract.
## Canonical Owner
`docs/integrations/` is the canonical home for external integration reference notes per `docs/documentation/policy.md`.

View File

@@ -1,96 +1,66 @@
# Audita Subprocess Operations # Integration: audita
This document describes how parent processes should invoke `audita process` safely in production orchestration. ## Purpose
Define Narratio's adapter contract for transcript polishing via Audita CLI subprocess execution.
## Recommended command form ## Inputs and Outputs
Inputs (`audita.PolishRequest`):
- base transcript path
- glossary path
- output polished transcript path
- optional report path (required when report enabled)
- work dir
- generated config path
- stdout/stderr log paths
- optional module/model/base URL and concurrency knobs
Use explicit file outputs for orchestrated runs: Outputs (`audita.PolishResult`):
- polished transcript path
- optional report path
- generated config path
- stdout/stderr log paths
- exit code, duration, invoked binary
- adapter metadata
```sh ## Boundaries
audita process <transcript.json> \ Owns:
--transcript-description "Brief context that may help resolve ambiguous terms." \ - Deterministic CLI argument construction for `audita process`
--glossary <glossary.yaml> \ - Environment bridging for API credentials
--output <output-transcript.json> \ - Invocation config emission
--report-json <report.json> - Output validation for polished transcript and report
```
Additional flags that may be situationally appropriate: Does not own:
- `--config <path>` to select an explicit versioned config file. - Upstream/downstream stage orchestration
- `--output-schema <bare-segments|audita-v1>` to select transcript output shape. - Credential sourcing policy beyond required env-var presence check
- `--work-dir <dir>` to control diagnostics location.
- `--work-dir-retention <always|auto|never>` to control retained run directories.
- `--total-llm-concurrency`, `--proposal-llm-concurrency`, and `--validation-llm-concurrency` when orchestration needs to set explicit LLM throughput controls.
- `--modules ...` only when intentionally overriding the default sequence.
For config-driven orchestration, validate config files in CI/preflight: ## Config Fields Used
Via `pipeline.audita.*` mapped in app/stage wiring:
- `binary`, `timeout`, `llm_api_key_env`, `modules`, `base_url`, `model`
- `transcript_description`, `config_path`, `output_schema`, `work_dir_retention`
- `total_llm_concurrency`, `proposal_llm_concurrency`, `validation_model`, `validation_llm_concurrency`, `report`
```sh ## External Adapters Used
audita config validate --config <path> - Shared subprocess helper (`internal/adapters/subprocess`) to run CLI and capture logs.
```
## Stdout behavior ## State and Manifest Behavior
- No direct manifest writes.
- Stage-level metadata records adapter provenance and credential-present signal.
- Generated invocation YAML is written when `GeneratedConfigPath` is provided.
- With `--output`: stdout is expected to be empty on success. ## Skip and Resume Behavior
- Without `--output`: stdout contains transcript JSON only on success. - Adapter has no skip/resume logic. Stage/runner controls this.
- Report JSON is never written to stdout.
## Stderr behavior ## Failure Behavior
- Constructor validation fails on invalid binary/timeout/schema/concurrency/URL values.
- Run fails on missing required paths, missing required credential env var, subprocess errors, invalid polished JSON shape, or invalid report JSON.
- Failures preserve stdout/stderr paths in returned result metadata.
- Success path should be quiet or minimal human-readable logs. ## Tests to Inspect Before Changing
- Failure path writes concise human-readable errors. - `internal/adapters/audita/subprocess_test.go`
- When a diagnostics run directory exists, failure stderr includes its path. - `internal/adapters/audita/fake_test.go`
- Prompt/response diagnostic payloads are not streamed to stderr. - `internal/stage/polish_test.go`
## Output file behavior ## Architectural Invariants
- Polished output must be valid JSON with top-level `segments` array.
- `--output` writes transcript JSON in the selected output schema to the provided path. - When report is enabled, report output must be valid JSON.
- Output write failures return nonzero and surface actionable errors. - If `llm_api_key_env` is configured, credential must be present in environment.
- The command does not silently ignore output write errors.
## Report JSON behavior
- `--report-json` writes a machine-readable process report to the requested path.
- Run-directory `report.json` is written independently under diagnostics.
- Best-effort failure reports are emitted when possible without masking the primary failure.
- Report write failures return nonzero with clear stderr messaging.
- Report diagnostics metadata references run-directory artifacts including utilization diagnostics and correction ledger paths when available.
## Diagnostics directory behavior
- Each run creates (when possible) a per-run diagnostics directory.
- Typical artifacts include transcript, normalization, chunking, invocation, effective config, LLM diagnostics, `utilization-diagnostics.json`, `correction-ledger.json`, `report.json`, and `error.log` on failure.
- Failed runs retain diagnostics.
- Under `auto` retention, successful runs with skipped/rejected corrections are retained; clean successful runs may be removed.
## Exit codes
- `0`: success.
- Nonzero: failure (input/schema/config/module/LLM/runtime/output/report/diagnostics errors).
Treat any nonzero as a failed subprocess invocation.
## Timeout and cancellation
- Runtime operations propagate context cancellation and request timeouts through LLM/scheduler paths.
- On cancellation or timeout, the process exits nonzero and should not hang.
- If diagnostics were initialized before failure, failure artifacts remain available for debugging.
## Secret redaction expectations
API keys and configured secret values are redacted from:
- reports (`--report-json` and run-dir `report.json`);
- diagnostics artifacts (including effective config and LLM interaction artifacts);
- surfaced adapter/runtime errors;
- test fixtures and regression outputs.
Parent-process logs should still avoid printing raw environment variables.
## Parent-process pipe guidance
To avoid deadlocks in orchestrators:
- always read both stdout and stderr concurrently when invoking as a subprocess;
- prefer file outputs (`--output`, `--report-json`) for machine workflows;
- treat stderr as human-readable diagnostics, not structured data;
- parse structured results from output/report files.
For Go callers, prefer `exec.CommandContext` with explicit timeout/cancellation and buffered/streamed readers for both pipes.

View File

@@ -1,339 +1,64 @@
# Narratio -> Scriptorium CLI Integration # Integration: scriptorium
## 1. Purpose ## Purpose
Define Narratio's adapter contract for Scriptorium artifact generation and render-debug subprocess invocations.
This document defines how Narratio should invoke Scriptorium through the **public CLI**.
## Inputs and Outputs
This is a **subprocess integration contract**, not an internal Go API contract. Inputs:
- `RunArtifactRequest`: binary, config path, prompt/profile IDs, input map, vars map, timeout, output path, logs/config paths, optional API env and working dir
## 2. Assumptions - `RenderArtifactRequest`: same core fields for render mode
- `scriptorium` is installed and available on `PATH`. Outputs (`ArtifactResult`):
- Scriptorium is configured with `config.yml`. - output path
- `config.yml` provides `prompt_dir`, `profile_dir`, and `schema_dir` as needed. - stdout/stderr log paths
- Prompt and profile libraries are already deployed for the environment. - generated config path
- Narratio provides prepared artifact files (for example polished transcript, glossary, previous recap, campaign notes). - exit code and duration
- Initial integration is synchronous subprocess execution. - command mode (`run` or `render`)
- Narratio remains the orchestrator. - prompt/profile provenance
- validation failure signal
In normal operation, Narratio does not need to pass `--prompt-dir` and `--profile-dir` if they are supplied by Scriptorium config. - adapter metadata
Narratio may pass `--config <PATH>` when it must use a non-default Scriptorium config file. ## Boundaries
Owns:
## 3. Core Commands Narratio May Call - Deterministic CLI arg construction for `scriptorium run` and `scriptorium render`
- Common request validation
Primary commands for subprocess integration: - Invocation config emission
- Output existence/non-empty checks
- `scriptorium run` - Validation-failure mapping for run exit code 2
- `scriptorium render`
Does not own:
For production generation, use `scriptorium run`. - Artifact selection policy (`analyze` stage)
- Bounds semantic validation (`trim` stage)
`scriptorium render` is for debugging, dry-runs, test assertions, and validating command construction without LLM execution.
## Config Fields Used
Note: `scriptorium serve` and HTTP API exist, but they are not the initial integration path. Via `pipeline.scriptorium.*` and stage-level artifact config:
- `binary`, `config_path`, `timeout`, `render_debug`
## 4. Command Selection Guidance - artifact-level `prompt_id`, `profile_id`, `timeout`, `inputs`, `vars`, `output_path`
- Use `run` to generate an output artifact. ## External Adapters Used
- Use `render` to inspect the prepared prompt and effective settings without calling the LLM. - Shared subprocess helper (`internal/adapters/subprocess`).
- Use `render --format json` when Narratio/tests need structured prepare output.
## State and Manifest Behavior
## 5. Recommended `run` Invocation Shape - No direct manifest writes.
- Stage metadata records adapter outputs and command mode.
Production shape: - Generated invocation YAML is written when requested.
```bash ## Skip and Resume Behavior
scriptorium run \ - Adapter has no skip/resume logic. Stage/runner controls execution.
--prompt <prompt_id> \
--input transcript=<processed-transcript-path> \ ## Failure Behavior
--out <output-artifact-path> - Request validation fails for missing binary/prompt/output, invalid timeout, invalid input/var names, or missing required API env var.
``` - Subprocess errors bubble with command context.
- `run` exit code 2 is treated as `ValidationFailed=true` and surfaced as error by calling stage.
Common optional additions: - Successful subprocess still fails if output file is missing/empty.
- `--config <path>`: use a specific Scriptorium config file. ## Tests to Inspect Before Changing
- `--profile <profile_id>`: override prompt default profile. - `internal/adapters/scriptorium/subprocess_test.go`
- `--var name=value` (repeatable): small metadata values. - `internal/adapters/scriptorium/fake_test.go`
- `--input name=path` (repeatable): additional named artifacts. - `internal/stage/analyze_test.go`
- `--timeout <duration>`: per-run timeout override. - `internal/stage/trim_test.go`
- Runtime model override flags (`--llm-base-url`, `--model`, etc.) only for exceptional/operator-directed cases.
## Architectural Invariants
## 6. Recommended `render` Invocation Shape - Both modes require explicit timeout > 0.
- Input/var maps are sorted into deterministic CLI argument order.
Human-readable debug shape: - Run-mode validation failures are represented explicitly, not silently skipped.
```bash
scriptorium render \
--prompt <prompt_id> \
--input transcript=<processed-transcript-path> \
--format text
```
Structured debug/test shape:
```bash
scriptorium render \
--prompt <prompt_id> \
--input transcript=<processed-transcript-path> \
--format json \
--out <render-debug-path>
```
`render` does **not** call the LLM, does **not** validate model output, and does **not** perform repair.
## 7. Inputs
- Pass inputs as repeated `--input name=path` flags.
- `name` must match the Prompt Definition input name.
- Prefer absolute paths, or paths relative to a working directory controlled by Narratio.
- Pass Audita output as the primary transcript input.
- Additional inputs may include glossary, previous recap, campaign notes, event logs, final state maps, or other prompt-specific artifacts.
- Scriptorium reads input files directly; Narratio does not need to inline file content for CLI use.
## 8. Variables
Use repeated `--var name=value` for small metadata values.
Typical examples:
- `session_date`
- `session_id`
- `campaign_name`
- `previous_session_id`
- `output_kind`
Large content belongs in input files, not `--var` values.
## 9. Prompt IDs and Output Artifact Types
Narratio should treat prompt IDs as configuration, not hardcoded business logic.
Narratio config may map stage/output names to prompt IDs, for example:
- session recap prompt
- structured event extraction prompt
- glossary suggestion prompt
- player-facing summary prompt
Prompt IDs used by Narratio should come from the deployed Scriptorium prompt library.
## 10. Profiles
- Prompts may declare `default_profile`.
- Narratio may omit `--profile` to use prompt default profile.
- Narratio may pass `--profile` to force profile selection.
- This enables environment/profile selection like `local-fast`, `local-quality`, `frontier`, `batch`, or test profiles.
- Profile names should generally be Narratio configuration values.
## 11. Runtime Overrides
Supported runtime override flags:
- `--llm-base-url`
- `--model`
- `--api-key-env`
- `--temperature`
- `--max-tokens`
- `--top-p`
- `--timeout`
Guidance:
- Keep normal model/runtime settings in Execution Profiles.
- Use runtime overrides only for explicit per-run exceptions, tests, or operator overrides.
- Never pass raw API keys on the command line.
- `--api-key-env` names an environment variable; Narratio must ensure that variable is set in subprocess environment.
## 12. Config Behavior
- Default config path: `/etc/scriptorium/config.yml`.
- `--config <PATH>` overrides default path.
- Missing default config is allowed by Scriptorium.
- If `--config` is provided explicitly, the file must exist and be valid.
- CLI flags override `config.yml`.
- `config.yml` overrides built-in application defaults.
Narratio can either:
- rely on system default config path, or
- carry an explicit config path and pass `--config`.
## 13. Environment Handling
Subprocess environment recommendations:
- Pass through required API-key environment variables referenced by `api_key_env`.
- Do not pass raw API keys as CLI arguments.
- Avoid logging full environment dumps.
- Capture stdout and stderr separately.
- Use a controlled working directory.
- Prefer absolute artifact paths.
## 14. Output Handling
For `scriptorium run`:
- Use `--out` when Narratio needs durable artifact files.
- Without `--out`, artifact content is written to stdout.
- Preferred orchestration pattern: always use `--out`, then treat the file as stage output artifact.
- Capture stderr for diagnostics.
For `scriptorium render`:
- Use `--out` to store render diagnostics.
- Use `--format json` when tests need to inspect selected profile, effective runtime settings, input hashes, prompt hash, and rendered messages.
## 15. Exit Status and Errors
Current CLI behavior (verified from implementation/tests):
- `0`: success.
- `1`: runtime/parse/config/load/render/generation/IO error.
- `2`: run completed but output validation failed (`ValidationFailed`).
Additional details:
- On `run`, output artifact write happens before exit code selection. If validation fails, artifact may still be written and exit code is `2`.
- `stderr` carries both errors and normal run summary output; non-empty stderr alone does not imply failure.
- `render` returns `0` on success and `1` on failures.
Narratio should treat non-zero exit codes as failed stage execution, but may record generated artifact paths if a run exited `2` and output file exists.
## 16. Recommended Narratio Integration Pattern
1. Build CLI args from Narratio stage configuration.
2. Use subprocess context cancellation/timeout.
3. Pass absolute input paths.
4. Pass `--out` to a session-scoped artifact path.
5. Add `--var` metadata values.
6. Optionally add `--config`.
7. Optionally add `--profile`.
8. Ensure required API-key env vars are present.
9. Run subprocess synchronously.
10. Capture stdout/stderr separately.
11. On success, store output artifact path and invocation metadata in stage artifacts.
12. On failure, store exit code and stderr diagnostics in stage status.
## 17. Suggested Narratio Configuration Shape
Illustrative `pipeline.yml` shape:
```yaml
scriptorium:
binary: scriptorium
config_path: /etc/scriptorium/config.yml
timeout: 10m
render_debug: false
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
profile_id: local-quality # optional
output_path: artifacts/session_recap.md
timeout: 10m
render_debug: false # optional artifact override
inputs:
transcript:
source: trimmed_transcript
required: true
previous_recap:
source: previous_session_artifact
artifact: session_recap
path: "" # optional
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: session_recap
```
The key idea: map Narratio artifact names to prompt ID, optional profile, expected inputs, vars, and output destination.
## 18. Testing Strategy for Narratio Integration
- Use `scriptorium render --format json` to verify command construction without LLM calls.
- Use dedicated test prompt/profile libraries for integration tests.
- Use small fixture transcripts.
- Verify missing-input failure behavior.
- Verify prompt `default_profile` behavior.
- Verify explicit `--profile` override behavior.
- Verify `--config` behavior (default and explicit).
- Verify output file creation when `--out` is used.
- Verify stderr capture on failures.
- Avoid real API keys in tests.
## 19. Security and Privacy Notes
- Never pass raw API keys on command line.
- Do not log full rendered prompts by default; transcripts may contain sensitive content.
- Avoid logging prompt content unless explicit debug mode is enabled.
- Treat generated artifacts as potentially sensitive.
- Use session-scoped, access-controlled output paths.
- `api_key_env` names should come from environment management, not embedded secrets.
## 20. Initial D&D Artifact Generation Examples
These are examples only. Use prompt IDs from the deployed prompt library.
Session recap:
```bash
scriptorium run \
--prompt dnd.session_recap \
--input transcript=/work/session-42/transcript.polished.md \
--input glossary=/work/session-42/glossary.yml \
--out /work/session-42/artifacts/session_recap.md
```
Structured events:
```bash
scriptorium run \
--prompt dnd.structured_events \
--input transcript=/work/session-42/transcript.polished.md \
--out /work/session-42/artifacts/structured_events.json
```
Glossary suggestions:
```bash
scriptorium run \
--prompt dnd.glossary_suggestions \
--input transcript=/work/session-42/transcript.polished.md \
--input previous_recap=/work/session-41/artifacts/session_recap.md \
--out /work/session-42/artifacts/glossary_suggestions.md
```
Player-facing summary:
```bash
scriptorium run \
--prompt dnd.player_summary \
--input transcript=/work/session-42/transcript.polished.md \
--input structured_events=/work/session-42/artifacts/structured_events.json \
--out /work/session-42/artifacts/player_summary.md
```
## 21. Non-Goals
Initial Narratio integration should not:
- call Scriptorium internal Go packages
- use HTTP API as the primary path
- expect Scriptorium to read S3 refs directly
- make Scriptorium responsible for Narratio stage state
- make Scriptorium responsible for notification
- require Scriptorium to understand D&D workflow semantics beyond prompt definitions
## 22. Future Extension Notes
Possible later extensions:
- HTTP API integration
- S3 artifact references if Scriptorium adds S3 reader support
- richer render diagnostics and policy controls
- token budgeting/prompt-size checks
- batch execution if Scriptorium later adds batch support

View File

@@ -1,403 +1,60 @@
# seriatim # Integration: seriatim
`seriatim` merges per-speaker WhisperX-style JSON transcripts into a single JSON transcript that preserves speaker identity and chronological order. ## Purpose
Define Narratio's adapter contract for merge, normalize, and trim subprocess invocations of Seriatim.
The current implementation supports the `merge` command. It reads one or more input JSON files, optionally maps each input file to a canonical speaker using `speakers.yml`, sorts all segments by timestamp, detects and resolves overlaps when word-level timing is available, assigns consecutive numeric `id` values, and writes a merged JSON artifact.
## Inputs and Outputs
## Usage Inputs:
- `MergeRequest`: raw/per-speaker normalized transcript inputs, base output path, optional report, speaker/autocorrect paths, logs/config
Run from source: - `NormalizeRequest`: input transcript, output path, schema, optional report, timeout/log/config
- `TrimRequest`: input transcript, output path, keep selector, timeout/log/config
```sh
go run ./cmd/seriatim merge \ Outputs:
--input-file samples/raw/2026-04-19-Eric_Rakestraw.json \ - `MergeResult`, `NormalizeResult`, `TrimResult` with output paths, logs/config paths, exit code, duration, binary provenance, and metadata.
--input-file samples/raw/2026-04-19-Mike_Brown.json \
--output-file merged.json ## Boundaries
``` Owns:
- Validated deterministic CLI invocation construction
Optional report output: - Optional env tuning propagation for merge
- Invocation config file emission
```sh - JSON output validation
go run ./cmd/seriatim merge \
--input-file eric.json \ Does not own:
--input-file mike.json \ - Transcript input selection/promotion logic (stage-owned)
--output-file merged.json \ - Bounds computation (scriptorium/trim-stage-owned)
--report-file report.json
``` ## Config Fields Used
Via `pipeline.seriatim.*` mapped in app/stage wiring:
## CLI - `binary`, `timeout`, `output_schema`, `coalesce_gap`, `report`
- `env.overlap_word_run_gap`
```text - `env.overlap_word_run_reorder_window`
seriatim merge [flags] - `env.backchannel_max_duration`
``` - `env.filler_max_duration`
Global flags: ## External Adapters Used
- Shared subprocess helper (`internal/adapters/subprocess`).
| Flag | Description |
| --- | --- | ## State and Manifest Behavior
| `--help` | Show command help. | - No direct manifest writes.
| `--version` | Show application version. Local builds default to `dev`; release builds inject the release version. | - Stage metadata consumes adapter result fields and preserves generated config/log references.
`merge` flags: ## Skip and Resume Behavior
- Adapter has no skip/resume logic. Runner controls stage execution.
| Flag | Required | Default | Description |
| --- | --- | --- | --- | ## Failure Behavior
| `--input-file` | Yes | none | Input transcript JSON file. Repeat once per speaker/input file. | - Constructor fails for invalid binary/timeout/output-schema/coalesce-gap.
| `--output-file` | Yes | none | Merged transcript JSON output path. | - Merge fails on missing output path/inputs/report path (if enabled), subprocess errors, invalid merged output JSON, invalid report JSON.
| `--report-file` | No | none | Optional report JSON output path. | - Normalize fails on missing input/output, invalid schema, subprocess errors, invalid final output JSON shape, invalid report JSON.
| `--speakers` | No | none | Speaker map YAML file. When omitted, input file basenames are used as speaker labels. | - Trim fails on missing input/output/keep selector, subprocess errors, invalid final-trimmed output JSON shape.
| `--autocorrect` | No | none | Autocorrect rules YAML file. When omitted, the default `autocorrect` module leaves text unchanged. |
| `--input-reader` | No | `json-files` | Input reader module. | ## Tests to Inspect Before Changing
| `--output-modules` | No | `json` | Comma-separated output modules. | - `internal/adapters/seriatim/subprocess_test.go`
| `--output-schema` | No | `seriatim-intermediate` | JSON output contract. Allowed values are `seriatim-minimal`, `seriatim-intermediate`, and `seriatim-full`. If omitted, the runtime default is used; consumers that depend on a specific shape should set this explicitly. | - `internal/adapters/seriatim/fake_test.go`
| `--preprocessing-modules` | No | `validate-raw,normalize-speakers,trim-text` | Comma-separated preprocessing modules, evaluated in order. | - `internal/stage/merge_test.go`
| `--postprocessing-modules` | No | `detect-overlaps,resolve-overlaps,backchannel,filler,resolve-danglers,coalesce,detect-overlaps,autocorrect,assign-ids,validate-output` | Comma-separated postprocessing modules, evaluated in order. | - `internal/stage/normalize_test.go`
| `--coalesce-gap` | No | `3.0` | Maximum same-speaker gap in seconds for `coalesce`; also used as the `resolve-overlaps` context window. Must be a non-negative float. | - `internal/stage/trim_test.go`
Environment variables: ## Architectural Invariants
- Supported output schemas are limited to `seriatim-minimal`, `seriatim-intermediate`, `seriatim-full`.
| Environment Variable | Default | Description | - Final and final-trimmed outputs must include `segments` arrays.
| --- | --- | --- | - Merge/normalize/trim all route through deterministic subprocess invocation.
| `SERIATIM_OUTPUT_SCHEMA` | `seriatim-intermediate` | Output schema used when `--output-schema` is not explicitly provided. Allowed values are `seriatim-minimal`, `seriatim-intermediate`, and `seriatim-full`. The CLI flag takes precedence. |
| `SERIATIM_OVERLAP_WORD_RUN_GAP` | `1.0` | Maximum gap in seconds between adjacent timed words when `resolve-overlaps` builds word-run replacement segments. Must be a positive float. |
| `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW` | `1.0` | Near-start window in seconds for ordering replacement word runs shortest-first. Must be a positive float. |
| `SERIATIM_BACKCHANNEL_MAX_DURATION` | `2.0` | Maximum duration in seconds for `backchannel` classification. Must be a positive float. |
| `SERIATIM_FILLER_MAX_DURATION` | `1.25` | Maximum duration in seconds for `filler` classification. Must be a positive float. |
## Input JSON Format
Each input file must be valid JSON with a top-level `segments` array. The current parser accepts the WhisperX segment subset needed for merging:
```json
{
"segments": [
{
"start": 1.25,
"end": 3.5,
"text": "Hello there.",
"words": [
{"word": "Hello", "start": 1.25, "end": 1.55, "score": 0.98},
{"word": "there.", "start": 1.7, "end": 2.0}
]
}
]
}
```
Required segment fields:
- `start`: number, must be `>= 0`.
- `end`: number, must be `>= start`.
- `text`: string.
Optional word fields:
- `words`: array of word timing objects.
- `words[].word`: string.
- `words[].start`: optional number, must be `>= 0` when present.
- `words[].end`: optional number, must be `>= start` when present with `start`.
- `words[].score`: optional number.
- `words[].speaker`: optional raw speaker label string.
Word-level timing is preserved internally for overlap resolution. If a word is missing `start` or `end`, seriatim keeps the word text, emits a warning in the optional report, and does not use that word as a timing anchor. Word timing is not emitted in the final JSON artifact.
## Speaker Map Format
`speakers.yml` maps input files to canonical speaker names using ordered substring rules:
This file is optional. If `--speakers` is omitted, `seriatim` uses each input file basename as the segment speaker label.
```yaml
match:
- speaker: "Eric Rakestraw"
match:
- "Eric_Rakestraw"
- "Eric"
- speaker: "Mike Brown"
match:
- "Mike_Brown"
- "mb"
```
For each `--input-file`, `seriatim` takes the file basename and evaluates the rules in order. The first rule with a matching substring wins, and no later rules are evaluated.
For example, this input:
```text
samples/raw/2026-04-19-Eric_Rakestraw.json
```
matches this rule because the basename contains `Eric_Rakestraw`:
```yaml
- speaker: "Eric Rakestraw"
match:
- "Eric_Rakestraw"
```
Important details:
- Matching is against the input file basename, not the full path.
- Matching is case-insensitive.
- Rules are evaluated from first to last.
- Each rule must have a non-empty `speaker`.
- Each rule must have at least one non-empty `match` string.
- Duplicate speaker names are invalid.
- Every input file must match at least one rule or the command fails.
Deprecated old format:
```yaml
inputs:
eric.json:
speaker: "Eric Rakestraw"
```
The old `inputs:` direct mapping format is no longer supported.
## Output JSON Format
`--output-modules json` controls the writer. `--output-schema` controls the JSON contract that writer serializes.
The named schemas are stable public contracts. If a consumer depends on a specific shape, it should request that schema explicitly at runtime. The runtime default selection may change in a future release.
The `seriatim-intermediate` schema is the current default selection when neither `--output-schema` nor `SERIATIM_OUTPUT_SCHEMA` is set. It stays close to the minimal schema, but adds optional `categories` on each segment:
```json
{
"metadata": {
"application": "seriatim",
"version": "dev",
"output_schema": "seriatim-intermediate"
},
"segments": [
{
"id": 1,
"start": 1.25,
"end": 3.5,
"speaker": "Eric Rakestraw",
"text": "Hello there.",
"categories": ["backchannel"]
}
]
}
```
The `seriatim-full` schema uses the full seriatim envelope:
```json
{
"metadata": {
"application": "seriatim",
"version": "dev",
"input_reader": "json-files",
"input_files": ["eric.json", "mike.json"],
"preprocessing_modules": ["validate-raw", "normalize-speakers", "trim-text"],
"postprocessing_modules": ["detect-overlaps", "resolve-overlaps", "backchannel", "filler", "resolve-danglers", "coalesce", "detect-overlaps", "autocorrect", "assign-ids", "validate-output"],
"output_modules": ["json"]
},
"segments": [
{
"id": 1,
"source": "eric.json",
"source_segment_index": 0,
"speaker": "Eric Rakestraw",
"start": 1.25,
"end": 3.5,
"text": "Hello there.",
"overlap_group_id": 1
},
{
"id": 2,
"source": "eric.json",
"source_ref": "word-run:1:1:1",
"derived_from": ["eric.json#0"],
"speaker": "Eric Rakestraw",
"start": 2.0,
"end": 2.5,
"text": "Resolved word run",
"categories": ["backchannel"]
}
],
"overlap_groups": [
{
"id": 1,
"start": 1.25,
"end": 4.0,
"segments": ["eric.json#0", "mike.json#0"],
"speakers": ["Eric Rakestraw", "Mike Brown"],
"class": "unknown",
"resolution": "unresolved"
}
]
}
```
The `seriatim-minimal` schema emits minimal metadata and compact ordered segments:
```json
{
"metadata": {
"application": "seriatim",
"version": "dev",
"output_schema": "seriatim-minimal"
},
"segments": [
{
"id": 1,
"start": 1.25,
"end": 3.5,
"speaker": "Eric Rakestraw",
"text": "Hello there."
}
]
}
```
Minimal output intentionally omits categories, overlap groups, source/provenance fields, and pipeline configuration metadata.
Intermediate output intentionally omits overlap groups and source/provenance fields, but keeps optional `categories` and minimal metadata.
Segments are sorted deterministically by:
```text
(start, end, source, source_segment_index/source_ref, speaker)
```
Final segment IDs are assigned after sorting and start at `1`.
The public Go output contract is available from:
```go
import "gitea.maximumdirect.net/eric/seriatim/schema"
```
The same package embeds machine-readable JSON Schemas in `schema/full-output.schema.json`, `schema/intermediate-output.schema.json`, and `schema/minimal-output.schema.json`. The default `validate-output` postprocessor validates the selected output shape and verifies final segment IDs are present, sequential, and start at `1`.
## Overlap Detection
The default postprocessing pipeline detects overlapping segment groups.
Overlap behavior:
- A strict timing overlap is required: `next.start < current_group_end`.
- Segments that only touch at a boundary are not grouped.
- Groups require at least two distinct speakers.
- Transitive overlaps are grouped together.
- Segments in detected groups receive `overlap_group_id`.
- `overlap_groups[].segments` contains stable references in `source#source_segment_index` format.
- `class` is currently `unknown`.
- `resolution` is `unresolved` until `resolve-overlaps` replaces the group.
## Overlap Resolution
The default postprocessing pipeline runs `detect-overlaps`, then `resolve-overlaps`, then `backchannel`, then `filler`, then `resolve-danglers`, then `coalesce`, then a second `detect-overlaps` pass.
For each detected overlap group, `resolve-overlaps` uses preserved WhisperX word timing to build smaller word-run replacement segments:
- The resolution window expands the detected overlap group by `--coalesce-gap` seconds on both sides.
- Nearby same-speaker context segments are included when they intersect the expanded window and their start or end is within `--coalesce-gap` of the original overlap boundary.
- Once a segment is selected for replacement, all timed words from that segment participate in word-run construction; the window controls segment selection, not per-word clipping.
- Context segments that are part of another detected overlap group are not pulled into the current group.
- Untimed words are included in replacement text in original word order when nearby timed words create a replacement run.
- Untimed words do not affect replacement segment start/end times or word-run gap splitting.
- Words for the same speaker are merged into one run when the gap between adjacent words is no greater than `SERIATIM_OVERLAP_WORD_RUN_GAP`.
- The default word-run gap is `1.0` seconds.
- Set `SERIATIM_OVERLAP_WORD_RUN_GAP` to a positive number of seconds to override the default.
- Near-start replacement word runs are reordered so shorter segments come first when adjacent starts are within `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW`.
- The default word-run reorder window is `1.0` seconds.
- Set `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW` to a positive number of seconds to override the default.
- Replacement segment text is built by joining word text with single spaces.
- Replacement segments include `source_ref` and `derived_from`.
- Replacement segments omit `source_segment_index` because they are derived from one or more original segments.
- Resolved overlap groups are removed before the second detection pass.
- Replacement segments are left without `overlap_group_id` until the second detection pass annotates any remaining overlap.
- If a speaker has no usable word timing in a group, that speaker's original segment is kept.
- If no speakers in a group have usable word timing, the original group and annotations remain unchanged.
## Backchannels
The default pipeline runs `backchannel` before `coalesce`. It tags short acknowledgement segments with:
```json
"categories": ["backchannel"]
```
Backchannel matching is case-insensitive, ignores punctuation for matching and word-count purposes, trims surrounding whitespace, and requires a matching acknowledgement phrase, no more than three whitespace-delimited words, and duration no greater than `SERIATIM_BACKCHANNEL_MAX_DURATION` seconds. The default maximum duration is `2.0` seconds.
## Fillers
The default pipeline runs `filler` after `backchannel` and before `coalesce`. It tags short filler utterances with:
```json
"categories": ["filler"]
```
Filler matching is case-insensitive, ignores punctuation for matching and word-count purposes, trims surrounding whitespace, and requires only filler tokens such as `um`, `uh`, `er`, `erm`, `ah`, `eh`, `hmm`, `mm`, or repeated combinations of those tokens. Matching segments must contain no more than three whitespace-delimited words and have duration no greater than `SERIATIM_FILLER_MAX_DURATION` seconds. The default maximum duration is `1.25` seconds.
## Dangler Resolution
The default pipeline runs `resolve-danglers` before `coalesce` and before the second overlap detection pass. It repairs short derived fragments when they share provenance with a nearby segment:
- Dangling-end fragments have no more than two words and end in punctuation.
- Dangling-start fragments have no more than two words.
- Matching uses same-speaker segments with any shared `derived_from` value.
- Merged segments use `source_ref` values such as `resolve-danglers:1`, keep the target segment's transcript position, and union `derived_from`.
## Coalescing
The default pipeline runs `coalesce` after `resolve-danglers` and before the second overlap detection pass. It merges adjacent same-speaker segments in the transcript's current order when `next.start - current.end <= --coalesce-gap`.
Coalesced segments use `source_ref` values such as `coalesce:1`, include `derived_from`, and omit `source_segment_index`.
Different-speaker backchannel and filler segments do not block coalescing of surrounding same-speaker segments. Same-speaker backchannel and filler segments are merged normally when they are within `--coalesce-gap`. When same-speaker segments are coalesced, any `backchannel` or `filler` category from the merged inputs is dropped from the coalesced segment.
## Autocorrect
Autocorrect is included in the default postprocessing pipeline. If `--autocorrect` is omitted, the module leaves transcript text unchanged and records a skip event in the optional report.
Enable corrections by passing `--autocorrect`:
```sh
go run ./cmd/seriatim merge \
--input-file input.json \
--autocorrect autocorrect.yml \
--output-file merged.json
```
`autocorrect.yml` format:
```yaml
autocorrect:
- target: "Hrank"
match:
- "hrank"
- "Frank"
- target: "Mike Brown"
match:
- "Mike Pat"
```
Matching behavior:
- Matching is case-sensitive.
- Matches apply only to whole tokens, not substrings inside larger words.
- Punctuation and whitespace can surround a match.
- Multi-word and hyphenated matches are supported.
- Duplicate match strings are invalid, including duplicates across separate rules.
## Current Limitations
- Only JSON input is supported.
- Overlap resolution depends on WhisperX word timing; groups without usable word timing remain unresolved.
- Alternate output formats are not implemented yet.
## Release Builds
Local builds record version metadata as `dev`. Release builds should inject the release version with `ldflags`:
```sh
go build -ldflags "-X gitea.maximumdirect.net/eric/seriatim/internal/buildinfo.Version=v1.0.0" ./cmd/seriatim
```

29
docs/internal/README.md Normal file
View File

@@ -0,0 +1,29 @@
# Internal Documentation Index
## Audience
Developers and LLM coding agents changing Narratio internals.
## Scope
Implementation-accurate contracts for workspace/state, manifests, stages, artifact resolution, adapter boundaries, and restore command behavior.
## Component Docs
- `adapters.md`: external adapter map, runtime wiring, and boundary ownership.
- `storage.md`: remote storage backend contracts and object-store invariants.
- `manifest.md`: session/run manifest schemas, lifecycle transitions, and persistence semantics.
- `artifacts.md`: built-in artifact registry, runtime artifact catalog, and source-resolution behavior.
- `workspace.md`: local state model, manifests, run-local layout, promotion, and cleanup invariants.
- `command-restore.md`: restore command discovery/planning/execution/reporting contract.
- `stage-prepare.md`: input materialization and provenance capture.
- `stage-transcribe.md`: WhisperX transcript generation.
- `stage-merge.md`: Seriatim normalization + merge.
- `stage-polish.md`: Audita transcript polishing.
- `stage-normalize.md`: post-polish normalization.
- `stage-trim.md`: bounds-driven transcript trimming.
- `stage-analyze.md`: dependency-ordered Scriptorium artifact generation for selected configured artifacts.
- `stage-archive.md`: archive upload and current-pointer publish contract.
## External Integration Notes
- `../integrations/README.md`: canonical location for external integration contracts (`audita.md`, `seriatim.md`, `scriptorium.md`).
## Canonical Owner
`docs/internal/` is the canonical home for implemented internals per `docs/documentation/policy.md`.

78
docs/internal/adapters.md Normal file
View File

@@ -0,0 +1,78 @@
# Internal: Adapters
## Purpose
Describe the external adapter boundaries used by Narratio stages and app orchestration, including default runtime wiring.
## Inputs and outputs
Inputs:
- Stage requests passed through adapter interfaces (for example transcription, merge/normalize/trim, polish, artifact generation, object-store operations, notifications).
- Resolved config values used to construct default adapters.
Outputs:
- Adapter-specific result structs (paths, metadata, status/attempt info, duration/exit details).
- Adapter errors returned to stage/app orchestration.
## Boundaries
Owns:
- Transport/process/SDK details at system boundaries (`HTTP`, subprocess CLI invocation, AWS SDK calls).
- Request/response contracts in `internal/adapters/*` packages.
Does not own:
- Stage sequencing, skip/force/resume decisions.
- Manifest transition logic.
- Canonical workspace path policy.
## Config fields used
Default wiring and adapter calls consume:
- `pipeline.whisperx.*`
- `pipeline.seriatim.*`
- `pipeline.audita.*`
- `pipeline.scriptorium.*`
- `pipeline.storage.*` and `pipeline.archive.*` (object-store construction/gating)
- `pipeline.notification.*` (sender boundary exists; placeholder behavior today)
## External adapters used
Runtime env boundary fields (`internal/stage.Env`):
- `whisperx.Client`
- `seriatim.Runner`
- `audita.Runner`
- `scriptorium.Runner`
- `storage.ObjectStore`
- `notify.Sender`
Current execution usage:
- Actively used by implemented stages: `WhisperX`, `Seriatim`, `Audita`, `Scriptorium`, `ObjectStore`, `Notifier`.
- Present but not used by implemented stage set: legacy `storage.Backend`.
Default construction in app runner:
- Auto-constructed when not injected: WhisperX HTTP client, Seriatim subprocess runner, Audita subprocess runner, Scriptorium subprocess runner, object store (only when needed), and `notify.NoopSender`.
- Object-store construction goes through app command orchestration so configured filesystem secrets are loaded before the storage adapter is initialized.
- Callers can inject test/fake implementations through `app.RunOptions.Env`.
## State and manifest behavior
- Adapters do not directly mutate session/run manifests.
- Stages and runner own manifest writes and stage status transitions.
- Adapter outputs are persisted indirectly through stage result mapping (outputs/logs/generated configs/metadata).
## Skip and resume behavior
- No adapter-level skip/resume semantics.
- Skip/resume/force behavior is decided by app runner using manifest stage state.
## Failure behavior
- Adapter constructors validate config-derived values and fail early on invalid required inputs.
- Adapter run-time failures are returned to stage code with boundary context and are recorded as stage failures by runner logic.
- Subprocess adapters preserve stdout/stderr and generated-config paths to aid diagnosis.
## Tests to inspect before changing
- `internal/adapters/whisperx/http_test.go`
- `internal/adapters/seriatim/subprocess_test.go`
- `internal/adapters/audita/subprocess_test.go`
- `internal/adapters/scriptorium/subprocess_test.go`
- `internal/adapters/storage/*_test.go`
- `internal/adapters/notify/fake_test.go`
- `internal/app/runner_test.go`
## Architectural invariants
- Stage code depends on adapter interfaces, not transport-specific implementation types.
- External SDK-specific types remain inside adapter implementations.
- Default app wiring must remain deterministic and overrideable via injected env dependencies.

106
docs/internal/artifacts.md Normal file
View File

@@ -0,0 +1,106 @@
# Internal: Artifacts
## Purpose
Define Narratio artifact identity, catalog, and source-resolution behavior for:
- built-in session artifacts;
- configured analyze artifacts;
- canonical previous-session artifact sources.
## Inputs and outputs
Inputs:
- configured input sources (`pipeline.scriptorium.artifacts.*.inputs.*.source`);
- session paths and manifest inputs/outputs;
- runtime catalog state.
Outputs:
- resolved artifact path + provenance (`ResolvedSessionArtifact`);
- runtime catalog entries for built-ins and configured artifacts;
- requirement sets for canonical previous-session inputs.
- canonical S3 session, run, current, session config, session locks, audio, and promoted artifact keys.
## Boundaries
Owns:
- built-in source registry and validation;
- configured artifact catalog identity (`narratio.artifact.<name>`);
- canonical previous-session source parsing and resolution;
- previous-session requirement collection (`CollectPreviousArtifactRequirements`).
Does not own:
- prepare-stage remote hydration;
- stage success/skip transitions;
- archive upload orchestration.
## Built-in IDs
| Artifact ID | Canonical file | Producer stage | Output kind |
| --- | --- | --- | --- |
| `narratio.transcript.base` | `transcripts/base.json` | `merge` | `transcript_base` |
| `narratio.transcript.polished` | `transcripts/polished.json` | `polish` | `transcript_polished` |
| `narratio.transcript.final` | `transcripts/final.json` | `normalize` | `transcript_final` |
| `narratio.transcript.final_trimmed` | `transcripts/final.trimmed.json` | `trim` | `transcript_final_trimmed` |
| `narratio.bounds.session` | `artifacts/session_bounds.json` | `trim` | `session_bounds` |
## Source families
- built-in: `narratio.transcript.*`, `narratio.bounds.session`
- configured artifact: `narratio.artifact.<artifact_key>`
- canonical previous-session artifact: `narratio.previous_session.artifact.<artifact_key>`
## S3 key helpers
- session prefix: `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/`
- session config: `{session_prefix}/session.yml`
- session lock store: `{session_prefix}/locks.yml`
- run prefix: `{session_prefix}/runs/{run_id}/`
- audio prefix: `{session_prefix}/{session.inputs.audio_s3.prefix}`
- current manifest: `{session_prefix}/current/manifest.json`
- current run pointer: `{session_prefix}/current/run_id.txt`
## Runtime catalog model
Catalog entries track:
- `planned`: source is registered for this run;
- `executable`: configured artifact is selected for analyze execution;
- `available`: usable local file exists (generated this run or reused from disk).
Configured artifact provenance values include:
- `generated.current_analyze_run`
- `filesystem.disabled_artifact_output`
Previous-session canonical provenance values include:
- `manifest.inputs.previous_cache`
- `current_session.previous_cache`
## Resolution behavior
- Built-ins resolve via manifest producer outputs first, then canonical fallback paths.
- Configured `narratio.artifact.<name>` sources resolve through catalog availability.
- Canonical previous-session sources resolve to current-session `previous/` cache candidates derived from configured artifact canonical output paths.
- Archive-relative configured artifact paths under `artifacts/` are cached without a redundant nested `artifacts/` segment.
- Previous-session canonical resolution prefers manifest-recorded input paths when present, then filesystem fallback under `previous/artifacts/**`.
## Previous-session requirement scanning
`CollectPreviousArtifactRequirements`:
- scans enabled configured artifacts only;
- includes canonical previous-session sources only;
- deduplicates by artifact key;
- merges required/optional references (`required` wins);
- records deterministic sorted source locations for diagnostics.
## Validation behavior
- transcript built-ins: JSON with top-level `segments` array;
- bounds built-in: valid JSON;
- configured and previous-session artifact files: non-empty text content.
## Failure behavior
- unsupported source or malformed canonical previous source: validation/resolution error;
- known source unavailable: `ErrSessionArtifactNotFound`;
- configured/previous canonical source without catalog: error;
- resolved invalid file content: validation error.
## Tests to inspect before changing
- `internal/artifacts/artifact_resolver_test.go`
- `internal/artifacts/catalog_test.go`
- `internal/artifacts/previous_requirements_test.go`
- `internal/stage/prepare_previous_test.go`
- `internal/stage/analyze_test.go`
## Architectural invariants
- Built-in source IDs are static.
- Configured and previous-session source IDs are artifact-key based and validation-gated.
- Resolution behavior remains deterministic and manifest-aware.

View File

@@ -0,0 +1,106 @@
# Internal: Command Restore
## Purpose
Define the implemented `narratio session restore` contract: committed remote-state discovery, deterministic plan classification, safe file install semantics, and restore reporting.
## Inputs and outputs
Inputs:
- CLI syntax: `narratio session restore <session_id>`.
- CLI flags: `--config`, `--campaign`, `--campaign-file`, `--session`, `--previous-session-id`, `--dry-run`, `--force`, `--include-audio`.
- Resolved/validated `pipeline.yml` and `session.yml`.
- Configured remote object store.
- Remote committed current-state markers (`current/run_id.txt`, `current/manifest.json`).
Outputs:
- Dry-run summary to stdout (plan + counts).
- Non-dry-run completion summary to stdout.
- Local durable session files restored under canonical session root.
- Non-dry-run restore report at `reports/restore-latest.json`.
## Boundaries
Owns:
- Restore command flag parsing and command wiring.
- Remote current-state discovery and identity validation.
- Restore plan construction and conflict classification.
- Restore execution for planned downloads.
- Restore report model and persistence.
Does not own:
- Stage execution orchestration (`run`, `resume`, `run-stage`).
- Archive publish behavior (owned by archive stage).
- Storage transport implementation details (owned by storage adapters).
## Config fields used
- Config/session discovery and templating fields consumed by all commands.
- `pipeline.workspace.root` (local restore target root).
- `pipeline.storage.*` (remote backend + archive identity derivation).
- `pipeline.storage.s3.*` identity components used by archive prefix helpers.
- `pipeline.spool.root` for active audio downloads.
- `pipeline.cache.root` and `pipeline.cache.s3_audio` for reusable S3 audio cache.
- `session.session_id`
- `session.campaign`
## External adapters used
- `storage.ObjectStore` for `Exists`, `List`, `Download`.
- `artifacts.Store` (`LocalStore`) for layout and session lock management.
- `manifest.LocalStore` for manifest decode/validation and identity checks.
## State and manifest behavior
- Restore is not a pipeline run and does not create a run manifest.
- Restore uses committed remote current state only:
- `current/run_id.txt` must exist and be non-empty.
- `current/manifest.json` must decode and match requested session/campaign.
- Non-dry-run writes restore files to canonical session paths.
- With `--include-audio`, restore uses the shared S3 audio cache for `audio/**` objects. Cache hits avoid object downloads; cache misses download through spool, install the work file, and populate cache.
- Manifest install behavior:
- validated before replacement.
- installed last among download actions.
- existing local manifest is preserved if restored manifest validation/install fails.
- Non-dry-run report persists summary/action status metadata in `reports/restore-latest.json`.
Restore path scope:
- includes:
- `manifest.json`
- `transcripts/**`
- `artifacts/**`
- `previous/**`
- `audio/**` only when `--include-audio` is set
- excludes:
- `runs/**`
- `logs/**`
- `reports/**`
- `config/**`
- `inputs/**`
- remote `current/**` pointer files as local restore targets
## Skip and resume behavior
- Restore does not participate in stage skip/resume decisions.
- Restore provides durable local state so subsequent stage commands can resume or rerun based on restored manifest state.
- Audio cache is outside the workspace and is reused across restore and prepare invocations.
- Dry-run is read-only and returns plan output only.
## Failure behavior
- Fails when storage backend is unavailable or archive identity cannot be resolved.
- Fails when remote current pointer/manifest is missing or invalid.
- Fails when remote manifest identity mismatches requested campaign/session.
- Fails on local conflicts unless `--force` is set.
- Fails fast on session lock acquisition conflict for non-dry-run execution.
- On execution failure, previously installed files remain; no rollback is performed.
## Tests to inspect before changing
- `internal/app/restore_test.go`
- `internal/app/restore_discovery_test.go`
- `internal/app/restore_plan_test.go`
- `internal/app/restore_execution_test.go`
- `internal/app/restore_workflow_test.go`
- `internal/artifacts/archive_identity_test.go`
## Architectural invariants
- Restore relies on centralized archive identity/key helpers (`internal/artifacts`) rather than ad hoc key building.
- `current/run_id.txt` is the remote commit marker; restore must not infer committed state from incidental files.
- Local path mapping is traversal-safe and constrained to session root.
- Restore scope is deterministic and path-classified:
- include `manifest.json`, `transcripts/**`, `artifacts/**`, `previous/**`
- include `audio/**` only with `--include-audio`
- exclude `runs/**`, `logs/**`, `reports/**`, `config/**`, `inputs/**`
- Command remains standalone; no implicit `run --restore` behavior.

81
docs/internal/manifest.md Normal file
View File

@@ -0,0 +1,81 @@
# Internal: Manifest
## Purpose
Describe Narratio's durable execution state model for session-level and run-level manifests, including lifecycle transitions and persistence behavior.
## Inputs and outputs
Inputs:
- Session identity and run identity from app orchestration.
- Stage transition events and stage result payloads.
Outputs:
- Session manifest at `{workspace.root}/work/{campaign}/{session_id}/manifest.json`.
- Run manifest at `{workspace.root}/work/{campaign}/{session_id}/runs/{run_id}/manifest.json`.
## Boundaries
Owns:
- Manifest schemas (`Manifest`, `RunManifest`, stage records, error records, input/artifact records).
- Stage status/action transition methods.
- Persistent store contract (`manifest.Store`) and local JSON store implementation.
Does not own:
- Stage implementation details.
- Path construction policy outside manifest file persistence calls.
- CLI command behavior.
## Config fields used
Manifest package itself does not read config directly.
Manifest identity fields are populated by app/stage orchestration from:
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.storage.s3.*` (when archive/S3 identity is set)
## External adapters used
- No external service adapters.
- Uses local filesystem for persistence via `manifest.LocalStore`.
## State and manifest behavior
Session manifest model:
- Tracks durable per-session stage state and provenance (`pending`, `running`, `succeeded`, `failed`, `skipped`, `stale`, `interrupted`).
- Stores resolved inputs, durable artifacts, stage logs/config refs, and stage metadata.
Run manifest model:
- Tracks one invocation (`run_id`) with requested stages and force mode.
- Tracks per-stage action (`run` or `skip`) and per-stage status.
- Tracks overall run status (`running`, `succeeded`, `failed`).
Persistence behavior:
- Load validates required identity/timestamp fields and normalizes maps/records.
- Save updates `updated_at` and writes JSON atomically (temp file + rename).
- Session and run manifests are saved incrementally before/after stage transitions.
Relationship during execution:
- Runner updates both manifests for every stage transition.
- Session manifest is the durable pipeline-progress ledger.
- Run manifest is invocation history and audit record.
- Analyze stage outputs are persisted as `kind=scriptorium_artifact` with `source_id=narratio.artifact.<name>` for configured artifact identity.
## Skip and resume behavior
- Resume and skip decisions are based on session-manifest stage statuses.
- `--force` reruns selected stages and marks downstream succeeded stages as `stale` in session manifest.
- Run manifest records whether each stage was executed or skipped in that invocation.
## Failure behavior
- Stage failure marks both manifests failed for that stage and records error messages/timestamps.
- Save failures are returned immediately and fail the command.
- Invalid/malformed manifest files fail load with explicit validation/decode errors.
## Tests to inspect before changing
- `internal/manifest/manifest_test.go`
- `internal/manifest/run_manifest_test.go`
- `internal/manifest/store_test.go`
- `internal/app/runner_test.go`
- `internal/app/run_control_test.go`
- `internal/app/resume_run_stage_test.go`
## Architectural invariants
- Session manifest is authoritative for stage progression across invocations.
- Run manifest is invocation-scoped and never replaces session manifest as progress authority.
- Manifest writes are atomic and deterministic (JSON + newline, temp rename pattern).

View File

@@ -0,0 +1,80 @@
# Stage: analyze
## Purpose
Execute selected configured Scriptorium artifacts in deterministic dependency order and promote successful outputs to canonical session artifact paths.
## Inputs and outputs
Inputs:
- configured artifact definitions from `pipeline.scriptorium.artifacts`;
- selected artifact filter (`--artifacts`) when provided;
- resolved artifact sources from resolver/catalog.
Source types used by analyze:
- built-ins: `narratio.transcript.*`, `narratio.bounds.session`;
- configured artifacts: `narratio.artifact.<artifact_key>`;
- canonical previous-session artifacts: `narratio.previous_session.artifact.<artifact_key>`.
Outputs:
- promoted configured artifact files at each configured `output_path`;
- stage metadata (`generated_artifacts`, `reused_artifacts`, selected/order info).
## Boundaries
Owns:
- runtime artifact catalog construction;
- selected-artifact planning and dependency ordering;
- per-input resolution and required/optional handling;
- Scriptorium render/run invocation;
- run-local output generation and canonical promotion.
Does not own:
- prepare-time previous-session hydration;
- object-store access for previous-session sources;
- archive promotion policy.
## Config fields used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.scriptorium.binary`
- `pipeline.scriptorium.config_path`
- `pipeline.scriptorium.timeout`
- `pipeline.scriptorium.render_debug`
- `pipeline.scriptorium.artifacts.<name>.*`
## External adapters used
- Scriptorium adapter:
- optional `RenderArtifact` when render-debug is enabled;
- `RunArtifact` for artifact generation.
## State and manifest behavior
- If Scriptorium config is absent, or no artifacts are executable after filtering, analyze returns success metadata with `skipped=true`.
- Builds runtime catalog with built-ins and configured `narratio.artifact.<name>` entries.
- Non-executable configured artifacts may still be marked available from existing canonical output files.
- Resolves canonical previous-session sources from local prepared `previous/` cache:
- prefers manifest-backed previous input paths when present;
- may fall back to current-session `previous/` filesystem paths.
- Analyze does not call object storage for canonical previous-session source resolution.
- Required canonical previous-session input missing:
- fails with guidance to run `narratio run-stage --force prepare`.
- Optional missing sources are omitted from adapter input paths.
## Skip and resume behavior
- Runner-level skip applies when analyze is already `succeeded` and `--force` is not set.
- Analyze is stage-scoped for resume; no per-artifact manifest resume state.
- `--artifacts` filters executable artifacts but does not imply force rerun.
## Failure behavior
- Fails on dependency-order violations, missing required inputs, resolver validation failures, adapter errors, and missing/empty generated outputs.
- Required unavailable configured artifact source (`narratio.artifact.<name>`) fails before invocation.
- Required canonical previous-session source fails with prepare-rerun guidance.
## Tests to inspect before changing
- `internal/stage/analyze_test.go`
- `internal/artifacts/catalog_test.go`
- `internal/artifacts/artifact_resolver_test.go`
- `internal/app/restore_workflow_test.go`
## Architectural invariants
- Canonical previous-session behavior is local-cache only during analyze.
- Generated outputs are validated and promoted before stage success is recorded.
- Resolver/catalog decisions stay deterministic and validation-gated.

View File

@@ -0,0 +1,86 @@
# Stage: archive
## Purpose
Publish durable run/session state to object storage, then atomically advance remote current state.
## Inputs and Outputs
Inputs:
- session manifest and prerequisite stage records
- run root contents under `runs/{run_id}/`
- promotion rules with artifact `source` IDs and archive `dest` paths (`archive.promote_artifacts`)
- effective source-based promotion locks from static config and remote session lock store
- session-level `previous/**` cache files when present
Outputs:
- uploaded run files under `{session_prefix}/runs/{run_id}/...`
- uploaded promoted artifacts under `{session_prefix}/...`
- uploaded session previous-cache files under `{session_prefix}/previous/...` when present
- `{session_prefix}/current/manifest.json`
- `{session_prefix}/current/run_id.txt` written last
## Boundaries
Owns:
- Archive enable/disable gate behavior
- Prerequisite stage success enforcement
- Run file collection and upload (excluding `audio/`)
- Promotion rule resolution and upload
- Promotion lock enforcement
- Session previous-cache file collection/upload
- Commit pointer publish order
Does not own:
- Stage execution before archive
- Post-archive local cleanup policy execution (handled by app cleanup logic)
## Config Fields Used
- `pipeline.archive.enabled`
- `pipeline.archive.upload_run`
- `pipeline.archive.promote_artifacts`
- `pipeline.archive.locks`
- `{session_prefix}/locks.yml` loaded by app orchestration before archive execution
- `pipeline.storage.s3.bucket`
- `pipeline.storage.s3.root_prefix`
- `pipeline.workspace.root`
- `session.campaign`
- `session.session_id`
## External Adapters Used
- Object storage backend (`env.ObjectStore`) for upload/list primitives.
## State and Manifest Behavior
- Requires `prepare`, `transcribe`, `merge`, `polish`, `normalize`, `trim`, and `analyze` status `succeeded`.
- Resolves bucket/prefix from manifest identity first, then config fallback.
- Uploads session `previous/**` files as durable session state when the local `previous/` directory exists.
- Skips top-level promotion uploads for effective locked sources; run-local uploads still publish.
- When selected configured artifact keys are supplied, skips promotion rules for unselected `narratio.artifact.<key>` sources; built-in transcript and bounds promotions still publish.
- Effective locks are the union of `pipeline.archive.locks` and remote `{session_prefix}/locks.yml`; static pipeline locks win on duplicate sources.
- Writes metadata including:
- upload counts/paths
- `previous_files_uploaded` and `previous_uploaded_paths`
- `skipped_unselected_promotions`
- `locked_promotion_count` and `locked_promotions`
- `current_manifest_key`
- `current_run_id_key`
- `current_pointer_written`
- On skipped archive path, returns metadata with `skipped=true` and pointer not written.
## Skip and Resume Behavior
- Stage may self-skip (metadata skip) when archive disabled or run upload disabled.
- Runner-level skip also applies for previously succeeded stage unless forced.
## Failure Behavior
- Fails on missing prerequisite success, missing object store when required, missing run root, missing unlocked required promotion source, upload failures, or pointer write failures.
- Locked required promotions are intentional skips and do not fail archive.
- Pointer semantics are fail-safe: `current/run_id.txt` is not written if prior required uploads fail.
## Tests to Inspect Before Changing
- `internal/stage/archive_test.go`
- `internal/app/post_archive_cleanup_test.go`
## Architectural Invariants
- Run upload excludes `audio/` subtree.
- Session `previous/**` is archiveable durable input/provenance state, not run-local output.
- Ordinary `--force` does not override archive locks.
- Malformed or unreadable remote lock store fails archive-capable execution before promotion.
- `current/manifest.json` uploads before `current/run_id.txt`.
- `current/run_id.txt` is the remote publish commit marker.

View File

@@ -0,0 +1,63 @@
# Stage: merge
## Purpose
Normalize per-speaker raw transcripts and merge them into the base transcript via Seriatim.
## Inputs and Outputs
Inputs:
- `transcripts/raw/*.json`
- `inputs/speakers.yml`
- `inputs/autocorrect.yml`
Outputs:
- `transcripts/base.json`
- optional `artifacts/seriatim.report.json` (when report enabled)
## Boundaries
Owns:
- Raw transcript discovery/validation
- Per-input normalize calls to Seriatim
- Final merge call to Seriatim
- Run-local log/config/report path wiring
- Promotion of base/report outputs to canonical paths
Does not own:
- Transcript polishing or downstream artifact generation
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.seriatim.binary`
- `pipeline.seriatim.timeout`
- `pipeline.seriatim.output_schema`
- `pipeline.seriatim.coalesce_gap`
- `pipeline.seriatim.report`
- `pipeline.seriatim.env.*`
## External Adapters Used
- Seriatim adapter:
- `Normalize` for each raw input
- `Run` for final merge
## State and Manifest Behavior
- Reads transcript inputs from transcribe stage outputs in manifest when present; falls back to canonical raw directory.
- Writes run-local outputs/logs/config under `runs/{run_id}/merge/...` when enabled.
- Promotes canonical base transcript and optional report.
- Records normalized-input provenance and adapter metadata in stage metadata.
## Skip and Resume Behavior
- Runner-level skip applies when already succeeded and not forced.
- Forced rerun of this or upstream stages can stale downstream succeeded stages via runner invalidation.
## Failure Behavior
- Fails on missing/invalid raw transcripts, missing speakers/autocorrect files, normalize failure, merge failure, invalid base output JSON, or invalid report JSON when enabled.
## Tests to Inspect Before Changing
- `internal/stage/merge_test.go`
- `internal/adapters/seriatim/subprocess_test.go`
## Architectural Invariants
- Merge consumes normalized forms of each raw transcript.
- Base transcript must validate before promotion.
- Report output is optional and gated by config.

View File

@@ -0,0 +1,56 @@
# Stage: normalize
## Purpose
Normalize the polished transcript into the full final transcript and optionally emit a normalize report.
## Inputs and Outputs
Inputs:
- `transcripts/polished.json`
Outputs:
- `transcripts/final.json` (or configured normalize output path)
- optional `artifacts/seriatim.normalize.report.json`
## Boundaries
Owns:
- Polished transcript discovery/validation
- Normalize request construction and invocation
- Optional normalize report wiring
- Promotion of final transcript and optional report
Does not own:
- Bounds detection or segment trimming
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.normalize.output_path`
- `pipeline.normalize.output_schema`
- `pipeline.normalize.report`
- `pipeline.seriatim.binary`
- `pipeline.seriatim.timeout`
## External Adapters Used
- Seriatim adapter (`Normalize`).
## State and Manifest Behavior
- Reads polished transcript from polish outputs in manifest when present; falls back to canonical path.
- Uses run-local output/report/log/config paths when run layout is enabled.
- Promotes canonical final transcript and optional normalize report.
- Records adapter/result metadata including source path selection.
## Skip and Resume Behavior
- Runner-level skip applies when already succeeded and not forced.
- Forced reruns can stale downstream succeeded stages.
## Failure Behavior
- Fails on missing/invalid polished transcript, adapter error, invalid final output, or invalid report output when report enabled.
## Tests to Inspect Before Changing
- `internal/stage/normalize_test.go`
- `internal/adapters/seriatim/subprocess_test.go`
## Architectural Invariants
- Final output must validate as transcript-compatible JSON (`segments` array required).
- Default normalize config is applied when `pipeline.normalize` is unset.

View File

@@ -0,0 +1,69 @@
# Stage: polish
## Purpose
Polish the base transcript with Audita and produce a polished transcript for downstream normalization/analyze.
## Inputs and Outputs
Inputs:
- `transcripts/base.json`
- `inputs/glossary.yml`
Outputs:
- `transcripts/polished.json`
- optional `artifacts/audita.report.json` (when report enabled)
## Boundaries
Owns:
- Base transcript discovery/validation
- Audita invocation request construction
- Run-local logs/config/work-dir/report wiring
- Promotion of polished transcript and optional report
Does not own:
- Upstream merge normalization
- Downstream normalize/trim/analyze logic
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.audita.binary`
- `pipeline.audita.timeout`
- `pipeline.audita.llm_api_key_env`
- `pipeline.audita.modules`
- `pipeline.audita.base_url`
- `pipeline.audita.model`
- `pipeline.audita.transcript_description`
- `pipeline.audita.config_path`
- `pipeline.audita.output_schema`
- `pipeline.audita.work_dir_retention`
- `pipeline.audita.total_llm_concurrency`
- `pipeline.audita.proposal_llm_concurrency`
- `pipeline.audita.validation_model`
- `pipeline.audita.validation_llm_concurrency`
- `pipeline.audita.report`
## External Adapters Used
- Audita adapter (`env.Audita.Run`).
## State and Manifest Behavior
- Reads base transcript from merge manifest outputs when available; falls back to canonical base path.
- Uses run-local output/report/log/config/scratch paths when run layout is enabled.
- Promotes canonical `transcripts/polished.json` and optional report.
- Records adapter invocation metadata, credential presence signal, and output provenance in stage metadata.
## Skip and Resume Behavior
- Runner-level skip applies when already succeeded and not forced.
- Forced rerun can stale downstream succeeded stages via runner invalidation.
## Failure Behavior
- Fails on missing/invalid base transcript, missing glossary, adapter error, invalid polished output shape (`segments` array required), or invalid report JSON when enabled.
## Tests to Inspect Before Changing
- `internal/stage/polish_test.go`
- `internal/adapters/audita/subprocess_test.go`
## Architectural Invariants
- Polished transcript must contain a top-level `segments` array.
- Report behavior is strictly config-gated.
- Stage output canonicalization always ends at `transcripts/polished.json`.

View File

@@ -0,0 +1,124 @@
# Stage: prepare
## Purpose
Materialize canonical current-session input state and provenance before downstream stages run.
Prepare owns:
- local input file materialization (`inputs/**`);
- audio input materialization (`audio/**`);
- previous-session cache hydration (`previous/**`) for canonical previous-session artifact sources.
## Inputs and outputs
Inputs:
- resolved config/campaign/session (`pipeline.yml`, `campaign.yml`, `session.yml`);
- remote session provenance when `session.yml` was loaded from S3;
- campaign or session input files (`speakers`, `autocorrect`, `glossary`);
- audio source:
- local: `session.inputs.audio_dir` or `session.inputs.audio_files`;
- S3: `session.inputs.audio_s3.prefix`;
- configured enabled Scriptorium artifact inputs (for previous-session requirement scanning);
- remote previous-session current archive state when previous hydration is required.
Outputs:
- `inputs/campaign.yml`;
- `inputs/session.yml`;
- `inputs/pipeline.resolved.yml`;
- `inputs/speakers.yml`;
- `inputs/autocorrect.yml`;
- `inputs/glossary.yml`;
- `audio/*.flac` in canonical session `audio/`;
- optional `previous/manifest.json`;
- optional `previous/artifacts/**`;
- deterministic `manifest.Inputs` records with checksums and provenance metadata.
## Boundaries
Owns:
- input path resolution and materialization;
- S3 audio list/download/copy flow;
- previous-session artifact requirement collection from enabled configured artifacts;
- previous cache lifecycle when requirements exist (clear and rehydrate managed `previous/` state).
Does not own:
- transcript or artifact generation;
- analyze-stage source resolution;
- archive commit behavior.
## Config fields used
- `session.session_id`
- `session.previous_session_id`
- `session.campaign`
- `session.inputs.speakers_file`
- `session.inputs.autocorrect_file`
- `session.inputs.glossary_file`
- `session.inputs.audio_dir`
- `session.inputs.audio_files`
- `session.inputs.audio_s3.prefix`
- `pipeline.workspace.root`
- `pipeline.spool.root`
- `pipeline.cache.root`
- `pipeline.cache.s3_audio`
- `pipeline.storage.s3.bucket`
- `pipeline.storage.s3.root_prefix`
- `pipeline.scriptorium.artifacts.<name>.enabled`
- `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source`
- `pipeline.scriptorium.artifacts.<name>.inputs.<key>.required`
- `campaign.campaign_id`
- `campaign.inputs.speakers_file`
- `campaign.inputs.autocorrect_file`
- `campaign.inputs.glossary_file`
## External adapters used
- `storage.ObjectStore` for:
- S3 audio listing/downloads;
- previous-session current pointer/manifest/artifact object checks and downloads.
## State and manifest behavior
- Ensures workspace layout exists.
- Materializes canonical input files and audio files.
- For S3 audio, uses run-scoped spool for active downloads and durable cache for reusable audio files; cache hits copy directly to work audio without downloading the object again.
- Records `inputs/session.yml` provenance as local `session_config` or remote `session_config.s3`.
- Resolves campaign-provided stable input paths relative to `campaign.yml`.
- Resolves session-provided stable input overrides relative to `session.yml`.
- Scans enabled configured artifact inputs for canonical sources:
- `narratio.previous_session.artifact.<artifact_key>`
- If one or more canonical previous-session requirements exist:
- clears managed `previous/` state;
- hydrates required/optional previous artifacts from the configured previous sessions committed archive current state;
- writes `previous/manifest.json` and hydrated `previous/artifacts/**`;
- stores archive-relative artifact paths such as `artifacts/session_recap.md` as `previous/artifacts/session_recap.md`, not `previous/artifacts/artifacts/session_recap.md`;
- records hydrated previous inputs in `manifest.Inputs` with source `previous_session_archive.current`.
- If no canonical previous-session requirements exist, prepare does not manage `previous/`.
- `manifest.Inputs` is sorted deterministically by `(kind, path)`.
- S3 audio `manifest.Inputs` retain S3 provenance and include `cache_path`; `spool_path` is present only when the current prepare invocation downloaded the file.
## Required and optional previous-session behavior
- `previous_session_id` unset:
- if any referenced previous artifact is required: fail;
- if all referenced previous artifacts are optional: continue and omit them.
- Previous session archive current pointer or manifest missing:
- if any referenced previous artifact is required: fail;
- if all referenced previous artifacts are optional: continue and omit missing ones.
- Missing required previous artifact object: fail.
- Missing optional previous artifact object: omit.
- Downloaded previous artifacts must validate as non-empty files.
## Skip and resume behavior
- Runner-level skip remains authoritative:
- if `prepare` already succeeded and run is not forced, `prepare` does not run and no hydration/download occurs.
- If `prepare` runs (including with `--force`), it owns managed `previous/` state for canonical previous-session inputs.
## Failure behavior
- Fails on missing required input files, invalid audio-source combinations, empty/duplicate audio inputs, missing object store for S3 modes, and remote access/download/validation errors.
- For required canonical previous-session inputs, analyze-time missing-input guidance is to rerun:
- `narratio run-stage --force prepare`
## Tests to inspect before changing
- `internal/stage/prepare_test.go`
- `internal/stage/prepare_previous_test.go`
- `internal/artifacts/previous_requirements_test.go`
- `internal/app/runner_test.go`
## Architectural invariants
- `audio_dir`/`audio_files` and `audio_s3` are mutually exclusive.
- Storage keys are computed by callers using archive/path helpers; storage adapter receives explicit keys.
- `prepare` is the only stage that hydrates canonical previous-session cache state.

View File

@@ -0,0 +1,58 @@
# Stage: transcribe
## Purpose
Generate per-speaker raw transcripts from prepared audio using WhisperX.
## Inputs and Outputs
Inputs:
- `audio/*.flac` prepared by `prepare`
Outputs:
- `transcripts/raw/<speaker>.json` for each input audio file
## Boundaries
Owns:
- Discovering prepared audio inputs
- Deriving speaker ids from audio basenames
- Parallel WhisperX invocation with bounded concurrency
- Validating produced JSON and promoting run-local outputs
Does not own:
- Transcript merge/polish/normalize/trim/analyze
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.whisperx.transcribe_url`
- `pipeline.whisperx.language`
- `pipeline.whisperx.timeout`
- `pipeline.whisperx.retries`
- `pipeline.whisperx.retry_delay`
- `pipeline.whisperx.concurrency`
## External Adapters Used
- WhisperX adapter (`env.WhisperX.Transcribe`).
## State and Manifest Behavior
- Uses run-local output paths under `runs/{run_id}/transcribe/outputs/...` when run layout is enabled.
- Validates each generated transcript JSON before promotion.
- Promotes canonical outputs to `transcripts/raw/*.json`.
- Records per-file metadata (attempts/status/duration/output path) in stage metadata.
## Skip and Resume Behavior
- Runner-level skip applies for previously succeeded stage unless forced.
- On forced upstream reruns, downstream succeeded stages can be marked `stale` by runner logic.
## Failure Behavior
- Fails if no prepared audio exists, duplicate speaker basenames are detected, adapter output path mismatches expected path, any output JSON is invalid, or one worker fails.
- Cancels in-flight workers after first terminal error.
## Tests to Inspect Before Changing
- `internal/stage/transcribe_test.go`
- `internal/app/whisperx_wiring_test.go`
## Architectural Invariants
- Speaker identity is derived from `.flac` basename and must be unique.
- Every successful speaker output must be valid JSON before promotion.
- Canonical raw transcript set is the only supported merge input surface.

View File

@@ -0,0 +1,75 @@
# Stage: trim
## Purpose
Optionally trim the final transcript to session bounds; always produce a durable final-trimmed transcript.
## Inputs and Outputs
Inputs:
- `transcripts/final.json`
Outputs:
- `transcripts/final.trimmed.json` (or configured trim output path)
- when trim enabled: `artifacts/session_bounds.json`
## Boundaries
Owns:
- Trim-enabled switch behavior
- Bounds generation via Scriptorium artifact run
- Bounds validation against final transcript
- Keep-selector derivation and Seriatim trim invocation
- Copy-through behavior when disabled or bounds indicate unchanged transcript
Does not own:
- Upstream normalization
- Downstream artifact analysis
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.trim.enabled`
- `pipeline.trim.output_path`
- `pipeline.trim.bounds.prompt_id`
- `pipeline.trim.bounds.profile_id`
- `pipeline.trim.bounds.timeout`
- `pipeline.trim.bounds.output_path`
- `pipeline.trim.bounds.transcript_input_name`
- `pipeline.trim.bounds.render_debug`
- `pipeline.trim.bounds.render_output_path`
- `pipeline.seriatim.binary`
- `pipeline.seriatim.timeout`
- `pipeline.scriptorium.binary`
- `pipeline.scriptorium.config_path`
- `pipeline.scriptorium.timeout`
## External Adapters Used
- Scriptorium adapter:
- optional `RenderArtifact` for bounds debug render
- `RunArtifact` for bounds output
- Seriatim adapter:
- `Trim` when bounds indicate trimming is required
## State and Manifest Behavior
- Reads final transcript from normalize manifest outputs when available; falls back to canonical path.
- Uses run-local outputs/logs/reports/config/scratch paths when run layout is enabled.
- Promotes canonical final-trimmed transcript; promotes session bounds when trim enabled.
- Records bounds diagnostics, trim action, keep selector, and adapter metadata.
## Skip and Resume Behavior
- Runner-level skip applies when already succeeded and not forced.
- Forced reruns can stale downstream succeeded stages.
- When `trim.enabled=false`, stage still succeeds by copying final to final-trimmed output.
## Failure Behavior
- Fails on missing/invalid final transcript.
- With trim enabled, fails on missing adapters/config, bounds generation/validation errors, invalid bounds JSON, invalid range/segment ids, trim adapter failures, or invalid final-trimmed output.
## Tests to Inspect Before Changing
- `internal/stage/trim_test.go`
- `internal/adapters/scriptorium/subprocess_test.go`
- `internal/adapters/seriatim/subprocess_test.go`
## Architectural Invariants
- Trim never falls back to polished transcript; final transcript is required input.
- `session_bounds` output exists only for enabled trim path.
- Render-debug artifacts are diagnostics and not declared stage outputs.

75
docs/internal/storage.md Normal file
View File

@@ -0,0 +1,75 @@
# Internal: Storage
## Purpose
Document Narratio's remote storage backend contracts and implementations under `internal/adapters/storage`.
## Inputs and outputs
Inputs:
- Resolved storage config (`pipeline.storage.*`).
- Already-loaded environment variables for configured S3 credentials.
- Bucket-relative object keys and local file paths from app/stage orchestration.
Outputs:
- Listed/downloaded/uploaded object metadata (`ObjectInfo`).
- Existence checks and storage-layer errors.
## Boundaries
Owns:
- Remote object-store interface and implementation details.
- S3 client wiring and API calls.
- Object key normalization and upload/download/list primitives.
Does not own:
- Session/run prefix semantics.
- Archive commit order semantics.
- Manifest updates.
- Filesystem secret loading from `pipeline.secrets.env_dir`.
## Config fields used
- `pipeline.storage.backend`
- `pipeline.storage.s3.bucket`
- `pipeline.storage.s3.region`
- `pipeline.storage.s3.endpoint`
- `pipeline.storage.s3.force_path_style`
- `pipeline.storage.s3.access_key_id_env`
- `pipeline.storage.s3.secret_access_key_env`
## External adapters used
Storage package contracts:
- `ObjectStore` (active remote object-store boundary): `List`, `Download`, `Upload`, `Exists`.
- `Backend` (archive request boundary): currently implemented with `NoopBackend` only.
Implementations:
- `S3Backend`: AWS SDK-backed `ObjectStore` implementation.
- `FakeBackend`: deterministic test `ObjectStore` and archive backend.
- `NoopBackend`: deterministic no-op archive backend for compatibility wiring.
## State and manifest behavior
- Storage implementations are stateless with respect to manifest/session lifecycle.
- Caller supplies fully-qualified bucket-relative keys.
- Storage layer does not infer campaign/session/run/root-prefix semantics.
- Caller controls publish ordering; storage layer executes individual operations in the order invoked.
## Skip and resume behavior
- No storage-level skip/resume behavior.
- Skip/resume decisions are made by stage/app logic before storage calls occur.
## Failure behavior
- `NewObjectStoreFromConfig` fails when no remote backend is configured or required S3 config is missing.
- `S3Backend` constructor fails when required bucket is missing or AWS client setup fails.
- App command orchestration loads configured filesystem secrets before calling the object-store factory.
- CRUD operations return contextual errors (including not-found behavior via `Exists`).
- Key normalization is applied before operations (`\\` to `/`, leading slash trimmed).
- Remote session loading uses `List` to find the exact `session.yml` key and `Download` to materialize it to a local temp file.
## Tests to inspect before changing
- `internal/adapters/storage/factory_test.go`
- `internal/adapters/storage/s3_backend_test.go`
- `internal/adapters/storage/fake_test.go`
- `internal/adapters/storage/keys_test.go`
- `internal/adapters/storage/archive.go` + consumers in stage tests (`prepare`, `archive`)
## Architectural invariants
- Callers pass full bucket-relative keys.
- Storage backends must not prepend or infer narratio prefixes.
- Remote transport details remain isolated to storage adapter implementations.

View File

@@ -0,0 +1,78 @@
# Workspace internals
## Purpose
Define the local durable and run-local workspace model used by stages, manifests, resume, and archive.
## Inputs and Outputs
Inputs:
- `pipeline.workspace.root`
- `session.campaign`
- `session.session_id`
- generated `run_id`
Outputs:
- Session manifest at `{workspace.root}/work/{campaign}/{session_id}/manifest.json`
- Run manifest at `{workspace.root}/work/{campaign}/{session_id}/runs/{run_id}/manifest.json`
- Canonical durable session directories and run-local stage trees
## Boundaries
Owns:
- Session-level path layout (`inputs/`, `audio/`, `transcripts/`, `artifacts/`, `reports/`, `logs/`, `config/`, `current/`, `runs/`, `previous/`)
- `previous/manifest.json` and `previous/artifacts/**` are reserved for previous-session cache state materialized by `prepare` or `restore`
- Run-local stage sandbox layout under `runs/{run_id}/{stage}/`
- Session lock acquisition/release (`.lock`)
Does not own:
- Stage business logic
- Remote archive semantics (documented in `stage-archive.md`)
- CLI argument parsing
## Config Fields Used
- `pipeline.workspace.root`
- `pipeline.workspace.cleanup_after_archive`
- `pipeline.spool.root`
- `pipeline.spool.delete_audio_after_archive`
- `pipeline.cache.root`
- `pipeline.cache.s3_audio`
- `session.campaign`
- `session.session_id`
## External Adapters Used
None directly in this subsystem. Stages may use object storage adapters and then write local outputs into this layout.
## State and Manifest Behavior
- Session state is persisted in the session manifest (`manifest.Manifest`).
- Invocation history is persisted per run in run manifests under `runs/{run_id}/manifest.json`.
- During each run, stage outputs are often written run-local first (`runs/{run_id}/{stage}/outputs/...`) and promoted to canonical session paths after stage success.
- `manifest.Artifacts` entries record `ProducerRunID` for durable outputs.
- For S3 audio sessions, `prepare` records work/cache paths, S3 provenance, and spool path when the invocation downloaded the object.
- `previous/**` is reconstructed from configured previous-session requirements; restore uses the previous session's committed current archive rather than treating current-session archived `previous/**` as authoritative.
- Durable cache state under `pipeline.cache.root` is not workspace state and is preserved by default by `narratio clean`.
- `narratio clean <id>` removes the session work root and session spool root.
- `narratio clean --all` removes all local session work under `workspace.root/work` and spool children under `spool.root`.
- `narratio clean --clear-cache` is the explicit opt-in for deleting matching S3 audio cache entries.
## Skip and Resume Behavior
- Skip/resume decisions are made in `internal/app` (`run_control.go`, `resume.go`) using stage status in the session manifest.
- `--force` reruns selected stages and marks downstream previously-succeeded stages as `stale`.
- Workspace layout is idempotent (`EnsureLayoutFor`) and reused across runs.
## Failure Behavior
- Failures preserve manifests and run-local files for inspection.
- Lock conflicts fail fast via `ErrLockConflict`.
- Cleanup can fail post-archive; failure is recorded in archive stage metadata and returned by the run.
## Tests to Inspect Before Changing
- `internal/artifacts/local_test.go`
- `internal/stage/run_local_test.go`
- `internal/app/run_control_test.go`
- `internal/app/resume_run_stage_test.go`
- `internal/app/post_archive_cleanup_test.go`
## Architectural Invariants
- Session root is campaign-aware: `{workspace.root}/work/{campaign}/{session_id}`.
- Run roots are always nested: `runs/{run_id}` under the session root.
- Run-local output promotion must end in canonical session paths.
- `previous/**` is session-durable state and must not be treated as run-local output scratch state.
- Automatic post-archive cleanup only targets run-scoped directories and must never delete configured root directories.
- Manual `clean` may delete session-scoped directories or the `workspace.root/work` directory, but it must preserve configured root directories and reject unsafe targets.

285
docs/operations.md Normal file
View File

@@ -0,0 +1,285 @@
# Operations
This guide describes the implemented operator lifecycle for Narratio.
For field-level configuration, see [docs/config.md](./config.md). For full command/flag reference, see [docs/cli.md](./cli.md).
## Normal workflow (S3-first path)
1. Create or upload `session.yml`, or pass a local `session.yml` explicitly.
2. Upload session `.flac` files to object storage under the configured session audio prefix.
3. Run Narratio:
```bash
narratio run 2026-04-04
```
4. Read success output:
- `narratio run: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
- use `narratio session status <session_id>` for inspection.
Notes:
- default pipeline/session discovery checks system config locations; campaign selection uses `pipeline.campaigns.default_campaign_id` unless `--campaign <id>` or `--campaign-file <path>` is passed.
- when local `session.yml` discovery misses, positional `<session_id>` loads remote `session.yml` from `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/session.yml`.
- S3 audio mode requires `session.inputs.audio_s3.prefix` and valid object-store access.
Initialize a remote session skeleton:
```bash
narratio session init 2026-04-04 --remote
```
Remote init uses normal default config discovery and writes `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/session.yml`. If `campaign.yml` sets `session_template_file`, init renders that template from the supplied flags and writes concrete YAML. Pass `--config`, `--campaign <id>`, or `--campaign-file <path>` when testing non-system config files. It fails if the object already exists unless `--force` is passed.
Validate before running:
```bash
narratio session validate 2026-04-04
```
## Restore workflow
Use restore when local durable session state is missing or stale and archive current state is authoritative.
Dry-run (no local writes):
```bash
narratio session restore 2026-04-04 --dry-run
```
Execution:
```bash
narratio session restore 2026-04-04
```
Post-restore analyze rerun pattern:
```bash
narratio analyze 2026-04-04
```
Restore source-of-truth:
- remote commit marker: `current/run_id.txt`
- remote current manifest: `current/manifest.json`
- configured previous-session requirements are reconstructed from the previous session's remote `current/` state, not from archived `previous/**` objects in the current session.
Restore default scope:
- includes `manifest.json`, `transcripts/**`, and `artifacts/**` from the current session archive.
- includes `previous/**` only when configured previous-session artifact inputs require it; restore hydrates those files the same way `prepare` would.
- includes `audio/**` only with `--include-audio`
- excludes `runs/**`, `logs/**`, `reports/**`, `config/**`, `inputs/**`, and `current/**` (except remote `current/manifest.json` as source)
Reset local state before restore testing:
```bash
narratio clean 2026-04-04 --dry-run
narratio clean 2026-04-04
narratio session restore 2026-04-04 --include-audio
```
`clean` removes the local session work directory and session spool directory. It preserves the durable S3 audio cache by default, so repeated restore or forced prepare tests do not re-download large audio files.
## Local filesystem layout and state artifacts
Session root:
- `{workspace.root}/work/{campaign}/{session_id}/`
Primary state:
- `manifest.json`: session-level stage state.
- `runs/{run_id}/manifest.json`: invocation-level state.
- `.lock`: session lock while a modifying command is active.
- `inputs/campaign.yml`, `inputs/session.yml`, and `inputs/pipeline.resolved.yml`: materialized config inputs for the run.
Canonical session directories:
- `inputs/`
- `audio/`
- `transcripts/`
- `artifacts/`
- `previous/`
- `reports/`
- `logs/`
- `config/`
- `current/`
- `runs/`
Run-local stage directories:
- `runs/{run_id}/{stage}/` with stage-local `outputs/`, `logs/`, `reports/`, `config/`, `scratch/`.
Behavior:
- directory creation is idempotent.
- stage outputs are generally generated run-local first, then promoted to canonical paths on success.
- restore installs downloaded files to canonical session paths and does not recreate historical run sandboxes.
## Analyze artifact execution lifecycle
Analyze executes configured artifacts from `pipeline.scriptorium.artifacts`.
Execution model:
- executable set = enabled artifacts, filtered by `--artifacts` when provided.
- artifact-to-artifact dependencies are declared via `depends_on`.
- selected artifacts run in deterministic dependency order.
- after each successful artifact run, output is promoted to configured canonical `output_path`.
Configured artifact source reuse:
- a non-executable configured artifact can satisfy inputs if its configured output file already exists and is valid.
- reused configured artifact provenance is `filesystem.disabled_artifact_output`.
`--artifacts` behavior:
- accepted on `run`, `resume`, `run-stage analyze`, `run-stage archive`, `analyze`, and `publish`.
- filters analyze execution and configured artifact promotions.
- built-in transcript and bounds promotions are not filtered.
- does not imply force on `run`, `resume`, or `run-stage`; `narratio analyze` is force-by-design.
- `publish` is force-by-design and accepts `--artifacts` for configured artifact promotions.
Canonical previous-session input behavior:
- canonical sources use `narratio.previous_session.artifact.<artifact_key>`.
- these inputs are hydrated by `prepare` and by `restore`; `analyze` expects the local previous cache to already exist.
- if analyze fails due to missing canonical previous cache, rerun:
- `narratio run-stage prepare <id> --force`
- or `narratio session restore <id>` when remote archive current state is authoritative.
## Remote archive layout and publish contract
Preferred manual publish command:
```bash
narratio publish <id>
```
`publish` is equivalent to `narratio run-stage archive <id> --force`; use `run-stage` when you need the general single-stage command form.
When archive is enabled and run upload is enabled, archive publishes under:
- session prefix: `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/`
- run prefix: `{session_prefix}/runs/{run_id}/`
Archive uploads:
- run record files from run root (excluding `audio/`).
- promoted files from explicit `archive.promote_artifacts` rules.
- mutable session locks from helper commands live at `{session_prefix}/locks.yml`.
Publish order:
1. upload `current/manifest.json`
2. upload `current/run_id.txt` last
`current/run_id.txt` is the remote commit marker.
Archive promotion is explicit and source-based:
- Narratio does not auto-promote all generated analyze artifacts.
- each rule resolves `source` through the artifact resolver/catalog model, then uploads to `dest`.
- missing required promotion sources fail archive stage.
- missing optional promotion sources are skipped.
- invalid resolved artifacts fail archive stage.
- `archive.locks` skips top-level promotion overwrites for locked sources while run-local uploads still publish.
- remote locks from `{session_prefix}/locks.yml` are merged with static `archive.locks`; static locks win on duplicate sources.
- locked required promotions are treated as intentional successful skips and are recorded in archive metadata.
Lock helper behavior:
- `narratio session locks <id>` lists effective static and remote locks.
- `narratio session locks add <id> <source> --reason <text>` writes a remote lock.
- `narratio session locks add <id> <source> --force --reason <text>` updates an existing remote lock reason.
- `narratio session locks remove <id> <source>` removes only a remote lock.
- `locks remove` cannot remove static pipeline locks.
- remote lock writes check whether the lock store exists, but are not compare-and-swap atomic.
## Resume, retry, restore, and safe rerun behavior
Default skip:
- `run` and `run-stage` skip already-succeeded stages unless `--force` is set.
Resume:
- `resume` starts at first non-succeeded stage.
- `resume --force` runs full stage order.
Restore conflict policy:
- restore classifies local differences as conflicts.
- without `--force`, restore fails when conflicts exist.
- with `--force`, conflicting local files are overwritten by remote archive files.
Forced reruns:
- force-rerunning an upstream succeeded stage marks downstream succeeded stages as `stale`.
- ordinary `--force` does not override archive locks.
Safe rerun pattern:
1. rerun the changed stage with `--force`.
2. run `resume` to rebuild downstream stages.
## Cleanup behavior
Automatic post-archive cleanup is considered only when archive stage executed and succeeded.
Automatic cleanup toggles:
- `pipeline.spool.delete_audio_after_archive=true` deletes run-scoped spool audio.
- `pipeline.workspace.cleanup_after_archive=true` deletes run-scoped local run directory.
Manual cleanup:
- `narratio clean <id>` deletes `{workspace.root}/work/{campaign}/{session_id}` and `{spool.root}/{campaign}/{session_id}`.
- `narratio clean --all` deletes all local session work under `{workspace.root}/work` and all spool children under `{spool.root}`.
- `--dry-run` prints targets without deleting.
- `--clear-cache` also removes matching S3 audio cache files. Without it, cache is preserved.
The S3 audio cache under `pipeline.cache.root` is durable input cache state, not workspace or spool state. Automatic cleanup and default manual cleanup do not delete it.
Cleanup eligibility gates:
- archive enabled
- archive run upload enabled
- run record upload completed
- current pointer write completed (`current/run_id.txt` written)
No cleanup for failed/incomplete/unarchived/archive-skipped runs.
## Failure and recovery playbooks
After run failure, Narratio keeps:
- session manifest
- run manifest
- run-local artifacts/logs/config/reports
Failed or incomplete runs remain local-only.
After restore failure:
- already-installed restore files remain in place.
- restore does not roll back prior successful installs.
- existing local manifest is preserved if restored manifest validation/install fails.
Recommended recovery:
1. inspect state:
```bash
narratio session status 2026-04-04
```
This reports local manifest state, committed remote current state, expected remote transcript/artifact availability, and archive locks.
2. for restore-specific checks, run:
```bash
narratio session restore 2026-04-04 --dry-run
```
3. fix root cause (config/input/credentials/storage/service availability).
4. continue with `resume`, or targeted `run-stage <stage> <id> --force` followed by `resume`.
## Restore report
Non-dry-run restore writes a durable report at:
- `reports/restore-latest.json`
Report content includes:
- identity (`campaign`, `session_id`, `run_id`)
- mode flags (`dry_run`, `force`, `include_audio`)
- plan counts and execution counts
- per-action status
Dry-run does not write restore report files.
## Operational caveats
- `session status <session_id>` uses normal config/session loading, including remote session fallback.
- `session status <session_id>` includes the same promoted remote output availability view as `session artifacts <session_id> --remote` when storage is configured.
- local and S3 audio input modes are mutually exclusive.
- archive publish requires upstream stages through `analyze` to be `succeeded`.
- required configured artifact promotions for unselected `--artifacts` keys are skipped intentionally; selected required promotions still fail if their files are missing.
- restore requires configured remote object storage and committed remote current state.

231
docs/roadmap/campaign.md Normal file
View File

@@ -0,0 +1,231 @@
# Roadmap: Campaign Registry
Status: Implemented
## Problem
Narratio currently treats campaign configuration as one selected
`campaign.yml` file:
- command flags use `--campaign <path>`;
- default discovery searches fixed system file locations;
- `campaign.yml` uses `campaign:` as the identity field.
That model works for a single campaign, but it is awkward for installations
that manage multiple campaigns. Operators need to pass file paths or maintain a
single global campaign config, while the newer session-oriented CLI already
uses concise positional session IDs and remote session lookup.
The campaign selection model should become ID-based and pipeline-owned.
Pipeline config should describe where campaigns live, commands should select a
campaign by ID, and each campaign directory should contain its stable campaign
materials.
## Target Model
`pipeline.yml` owns the campaign registry:
campaigns:
root: /usr/local/share/narratio/campaigns
default_campaign_id: dilfs
Campaign files live at the conventional path:
{campaigns.root}/{campaign_id}/campaign.yml
The first implementation should use only the conventional path. Recursive
discovery of every `campaign.yml` under `campaigns.root` is deferred to a
future stage.
Each campaign file uses `campaign_id` as the canonical identity field:
campaign_id: dilfs
session_template_file: ./session.template.yml
inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
Campaign-relative files continue to resolve relative to the selected
`campaign.yml`, including stable input files and `session_template_file`.
The public CLI changes from path-based campaign selection to ID-based campaign
selection:
- `--campaign <id>` selects a campaign ID.
- `--campaign-file <path>` explicitly loads one campaign file for
development, tests, and unusual local workflows.
- `--campaign` and `--campaign-file` are mutually exclusive.
If neither `--campaign` nor `--campaign-file` is passed, Narratio uses
`pipeline.campaigns.default_campaign_id`. If no campaign can be selected,
commands fail clearly before session loading or stage execution.
Resolved campaign ID remains the campaign segment used for:
- workspace paths;
- spool paths;
- S3 session prefixes;
- remote `session.yml` lookup;
- archive locks and promoted output keys;
- session/campaign mismatch validation;
- status, plan, restore, and helper output.
## Compatibility Policy
This is a breaking public/config contract change.
After the cutover:
- `--campaign` no longer accepts a filesystem path;
- default fixed campaign file discovery is removed;
- `campaign:` is no longer accepted in `campaign.yml`;
- `campaign_id:` is required.
Keep `--campaign-file` as the only explicit file override. Do not retain hidden
aliases for the old `--campaign <path>` behavior.
## Implementation Stages
### Stage 1: Add Campaign Registry Selection
Status: Implemented
Add the registry model and switch command loading to resolve campaigns through
pipeline config.
Implementation requirements:
- Add `pipeline.campaigns.root`.
- Add `pipeline.campaigns.default_campaign_id`.
- Add `campaign_id` to campaign config and make it the canonical identity.
- Resolve pipeline config first, then campaign selection.
- Use this selection order:
1. explicit `--campaign-file <path>`;
2. explicit `--campaign <id>`;
3. `pipeline.campaigns.default_campaign_id`;
4. fail clearly.
- For ID selection, load `{campaigns.root}/{campaign_id}/campaign.yml`.
- Validate that the loaded `campaign_id` matches the selected ID.
- Reject `--campaign` with `--campaign-file`.
- Preserve strict YAML decoding.
- Preserve campaign-relative stable input and session template resolution.
- Keep storage details behind the existing storage adapter and object-store
helper.
- Keep remote session lookup and archive key construction based on the
resolved campaign ID.
Acceptance criteria:
- Commands can run with only a pipeline config and the pipeline default
campaign ID.
- Commands can select another campaign with `--campaign <id>`.
- Commands can load a specific file with `--campaign-file <path>`.
- Existing session loading, remote session fallback, prepare materialization,
restore, archive, locks, clean, analyze, and publish behavior continue to use
the same resolved campaign identity.
- No generic config registry framework is introduced.
### Stage 2: Remove Old Single-File Campaign Behavior
Status: Implemented
Remove the old public campaign file model after registry selection is in
place.
Implementation requirements:
- Remove fixed default campaign config discovery from command loading.
- Remove `DefaultCampaignConfigSearchPaths` and related path-only resolution if
no current tests or helpers still need them.
- Remove support for `campaign:` from `campaign.yml`.
- Update validation errors to refer to `campaign_id`.
- Update examples to use campaign directories and `campaign_id`.
- Update current-behavior docs to document:
- `pipeline.campaigns.root`;
- `pipeline.campaigns.default_campaign_id`;
- `campaign_id`;
- `--campaign <id>`;
- `--campaign-file <path>`.
- Update troubleshooting examples that currently pass `--campaign <path>`.
Acceptance criteria:
- `campaign.yml` files with `campaign:` fail strict decoding.
- `--campaign /path/to/campaign.yml` is treated as a campaign ID and fails
unless that ID exists under `campaigns.root`.
- `--campaign-file /path/to/campaign.yml` is the supported file override.
- User-facing docs no longer describe fixed campaign config discovery.
## Test Guidance
Focused tests:
- `go test ./internal/config -v`
- `go test ./internal/app -v`
- `go test ./internal/stage -run Prepare -v`
Full validation:
- `go test ./...`
Config tests to add or update:
- strict decode accepts `pipeline.campaigns.root`;
- strict decode accepts `pipeline.campaigns.default_campaign_id`;
- strict decode accepts `campaign_id`;
- selected campaign ID mismatch fails;
- missing campaign root fails when ID selection is needed;
- missing default campaign ID fails when no explicit campaign selector is
passed;
- old `campaign:` fails after Stage 2.
App tests to add or update:
- `--campaign <id>` resolves `{campaigns.root}/{id}/campaign.yml`;
- omitted `--campaign` uses `pipeline.campaigns.default_campaign_id`;
- `--campaign-file` loads an explicit campaign file;
- `--campaign` plus `--campaign-file` fails;
- remote session fallback uses the resolved campaign ID;
- `session init`, `run`, `run-stage`, `resume`, `analyze`, `publish`, `clean`,
and `session` subcommands all use the same campaign selection path;
- path-based `--campaign` examples and tests are removed after Stage 2.
## Documentation Guidance
Update current-behavior docs only after implementation lands:
- `docs/config.md`
- `docs/cli.md`
- `docs/operations.md`
- `docs/troubleshooting.md`
- relevant files under `docs/internal/`
- `examples/`
Planned campaign registry behavior belongs only in this roadmap until the code,
tests, examples, and current-behavior docs are updated.
## Architecture Guardrails
- Keep Narratio explicit and stage-driven.
- Do not introduce a generic configuration registry or workflow framework.
- Keep YAML decoding strict.
- Keep defaults centralized and testable.
- Keep campaign-relative path resolution centralized.
- Use centralized S3 and workspace path helpers.
- Keep storage details behind `storage.ObjectStore`.
- Keep secret-backed object-store construction in `internal/app`.
- Preserve manifest-driven resume and restore behavior.
- Do not store raw secrets in campaign configs, manifests, logs, generated
configs, or archive metadata.
## Assumptions
- The canonical pipeline schema is grouped under `campaigns`.
- The canonical campaign identity field is `campaign_id`.
- `--campaign` means campaign ID.
- `--campaign-file` is retained as an explicit override.
- Recursive discovery is planned but not part of the first implementation.
- Existing production configs can be migrated from `campaign:` to
`campaign_id:` and from `--campaign <path>` to `--campaign <id>` or
`--campaign-file <path>`.

159
docs/roadmap/cleanup.md Normal file
View File

@@ -0,0 +1,159 @@
# Roadmap: Legacy Config Cleanup
Status: Implemented
## Problem
Narratio's current pipeline config schema still accepts fields that predate the current storage, artifact, and previous-session models:
- `pipeline.storage.bucket`
- `pipeline.storage.prefix`
- `pipeline.analyzer.*`
- `previous_session_artifact`
These names make the config reference harder to trust because they suggest supported behavior that operators should no longer use. The modern interface is:
- `pipeline.storage.s3.*` for remote storage.
- Scriptorium configured artifacts under `pipeline.scriptorium.artifacts`.
- Canonical artifact source IDs such as `narratio.artifact.<configured_artifact_key>`.
- Canonical previous-session artifact sources such as `narratio.previous_session.artifact.<configured_artifact_key>`.
Strict YAML decoding should reject removed legacy fields once this cleanup lands.
## Current State
`pipeline.storage.bucket` and `pipeline.storage.prefix` were inert compatibility fields and have been removed:
- They are no longer present on `config.StorageConfig`.
- Strict decoding rejects them.
- Runtime S3 behavior uses `pipeline.storage.s3.bucket` and `pipeline.storage.s3.root_prefix`.
- No current code reads the top-level storage bucket or prefix fields.
`pipeline.analyzer.*` was legacy code surface and has been removed:
- `config.PipelineConfig` no longer includes analyzer config.
- Strict decoding rejects `pipeline.analyzer`.
- `stage.Env` no longer exposes an analyzer runner, and `internal/adapters/analyzer` has been deleted.
- Modern analyze execution is Scriptorium-backed; the analyzer adapter is not used by current stage execution.
`previous_session_artifact` was a live legacy behavior and has been removed:
- Config validation rejects it as an unsupported Scriptorium input source.
- The analyze stage no longer has path-based previous-artifact resolution through `inputs.<name>.path`.
- Tests cover canonical previous-session sources and the rejection of the legacy source.
- The canonical replacement is `narratio.previous_session.artifact.<configured_artifact_key>`, resolved through the previous-session cache/catalog model.
## Target Model
The pipeline config schema should expose only current behavior:
- Remote storage is configured only through `pipeline.storage.s3.*`.
- Generated artifacts are configured only through `pipeline.scriptorium.artifacts`.
- Scriptorium artifact inputs use canonical source IDs.
- Previous-session artifact inputs use `narratio.previous_session.artifact.<configured_artifact_key>`.
- Unknown legacy fields fail strict YAML decoding.
No compatibility aliases should remain unless a future migration requirement explicitly reintroduces them.
## Cleanup Order
### Stage 1: Remove Inert Storage Compatibility Fields
Status: Implemented
Remove `pipeline.storage.bucket` and `pipeline.storage.prefix`.
Implementation requirements:
- Delete `StorageConfig.Bucket` and `StorageConfig.Prefix`.
- Keep `StorageConfig.Backend` and `StorageConfig.S3`.
- Confirm all runtime storage paths continue to use `storage.s3.bucket` and `storage.s3.root_prefix`.
- Update examples and docs to remove top-level storage `bucket` and `prefix`.
- Add or update strict-decode tests proving `pipeline.storage.bucket` and `pipeline.storage.prefix` are rejected.
Acceptance criteria:
- Existing S3 workflows still pass with `pipeline.storage.s3.bucket`.
- Pipeline configs containing top-level `storage.bucket` or `storage.prefix` fail to load.
- No docs or examples present those fields as available.
### Stage 2: Remove Legacy Analyzer Schema and Adapter Surface
Status: Implemented
Remove the unused analyzer configuration and adapter contract.
Implementation requirements:
- Delete `PipelineConfig.Analyzer`.
- Delete `AnalyzerConfig` and `ArtifactSettings`.
- Remove analyzer timeout validation.
- Remove `stage.Env.Analyzer`.
- Delete `internal/adapters/analyzer` if no remaining code imports it.
- Remove `pipeline.analyzer.*` from tests, examples, and docs.
- Add or update strict-decode tests proving `pipeline.analyzer` is rejected.
Acceptance criteria:
- Analyze behavior remains fully Scriptorium-backed.
- No runtime code imports `internal/adapters/analyzer`.
- Pipeline configs containing `pipeline.analyzer` fail to load.
- Contributor and internal adapter docs no longer list the analyzer adapter.
### Stage 3: Remove Path-Based Previous Session Artifact Source
Status: Implemented
Remove `previous_session_artifact` and require canonical previous-session artifact sources.
Implementation requirements:
- Remove `previous_session_artifact` from supported Scriptorium input sources.
- Remove analyze-stage special-case handling that resolves `inputs.<name>.path` for previous artifacts.
- Keep canonical handling for `narratio.previous_session.artifact.<configured_artifact_key>`.
- Rewrite tests that use `previous_session_artifact` to use canonical sources and prepared previous-cache fixtures.
- Add validation tests proving `previous_session_artifact` is rejected.
- Update docs to remove the legacy path-based source and document only canonical previous-session sources.
Acceptance criteria:
- `pipeline.scriptorium.artifacts.*.inputs.*.source: previous_session_artifact` fails validation.
- Canonical previous-session sources continue to work for required and optional inputs.
- Prepare/restore previous-cache behavior remains unchanged.
- No docs or examples mention `previous_session_artifact` as supported.
## Test Guidance
Run focused tests after each stage:
- `go test ./internal/config -v`
- `go test ./internal/stage -run Analyze -v`
- `go test ./internal/app -v`
- `go test ./...`
For Stage 1, focus on config load/strict-decode and S3 workflow regression tests.
For Stage 2, focus on compile-time removal, config strict-decode tests, and full app/stage tests to catch stale adapter references.
For Stage 3, focus on Scriptorium config validation, analyze-stage input resolution, previous-cache behavior, and restore/analyze workflows.
## Documentation Updates
Update current-behavior docs only after the corresponding code removal lands:
- `docs/config.md`
- `docs/cli.md`, only if command behavior text references removed fields.
- `docs/operations.md`, only if operator workflow text references removed fields.
- `docs/internal/stage-analyze.md`
- `docs/internal/adapters.md`
- `examples/pipeline.full.annotated.yml`
- `examples/pipeline.production.yml`
Do not preserve removed fields in examples as compatibility notes. The goal is to make strict config behavior and documentation line up.
## Assumptions
- This is a hard cleanup; no backward-compatible aliases are retained.
- Current production configs can be migrated to `storage.s3.*`, Scriptorium artifacts, and canonical previous-session sources before this lands.
- Removing the unused analyzer adapter does not block any active stage behavior.
- The cleanup should be implemented in the listed order so inert schema removal is separated from behavior removal.

255
docs/roadmap/cli.md Normal file
View File

@@ -0,0 +1,255 @@
# Roadmap: Session-Oriented CLI Cleanup
Status: Implemented
## Problem
Narratio's public CLI has accumulated too many top-level commands. Several
commands are session-scoped operator helpers, but they currently appear as
independent top-level verbs:
- `plan`
- `status`
- `restore`
- `artifacts list`
- `locks`
- `session validate`
- `session init`
This makes the command surface harder to learn because the CLI does not clearly
separate primary workflow actions from session inspection, initialization,
restore, and helper operations.
## Target Model
Keep primary workflow commands at top level:
- `run`
- `run-stage`
- `resume`
- `analyze`
- `publish`
- `clean`
- `session`
Keep `clean` top-level because it can operate on one session or all local
sessions and is a workspace maintenance command, not only a session helper.
Move session-scoped helper commands under `narratio session` and use positional
session identifiers:
- `narratio session init <session_id> [--remote|--output <path>] [--flags]`
- `narratio session validate <session_id> [--flags]`
- `narratio session status <session_id> [--flags]`
- `narratio session plan <session_id> [--flags]`
- `narratio session restore <session_id> [--flags]`
- `narratio session artifacts <session_id> [--remote] [--flags]`
- `narratio session locks <session_id> [--flags]`
- `narratio session locks add <session_id> <source> [--reason <text>] [--force] [--flags]`
- `narratio session locks remove <session_id> <source> [--flags]`
Update top-level workflow commands to use positional session identifiers:
- `narratio run <session_id> [--flags]`
- `narratio resume <session_id> [--flags]`
- `narratio analyze <session_id> [--flags]`
- `narratio publish <session_id> [--flags]`
- `narratio run-stage <stage> <session_id> [--flags]`
The positional session ID replaces `--session-id` as the primary public
interface. Existing `--config`, `--campaign`, `--session`, and
`--previous-session-id` flags remain available where they are meaningful.
## Command Mapping
| Current command | Target command |
| --- | --- |
| `narratio run --session-id <id>` | `narratio run <id>` |
| `narratio resume --session-id <id>` | `narratio resume <id>` |
| `narratio analyze --session-id <id>` | `narratio analyze <id>` |
| `narratio publish --session-id <id>` | `narratio publish <id>` |
| `narratio run-stage [flags] <stage> --session-id <id>` | `narratio run-stage <stage> <id> [flags]` |
| `narratio plan --session-id <id>` | `narratio session plan <id>` |
| `narratio status --session-id <id>` | `narratio session status <id>` |
| `narratio restore --session-id <id>` | `narratio session restore <id>` |
| `narratio artifacts list --session-id <id>` | `narratio session artifacts <id>` |
| `narratio locks --session-id <id>` | `narratio session locks <id>` |
| `narratio locks add --session-id <id> <source>` | `narratio session locks add <id> <source>` |
| `narratio locks remove --session-id <id> <source>` | `narratio session locks remove <id> <source>` |
| `narratio session validate --session-id <id>` | `narratio session validate <id>` |
| `narratio session init --session-id <id>` | `narratio session init <id>` |
| `narratio clean --session-id <id>` | `narratio clean <id>` |
| `narratio clean --all` | unchanged |
`clean` remains top-level, but its session-scoped form should also move from
`--session-id` to positional `<session_id>` for consistency.
## Compatibility Policy
This is a hard public CLI cleanup after the migration step lands.
During Step 1, old forms may remain as compatibility aliases to keep the
implementation reviewable. During Step 2, remove the old forms from command
dispatch, tests, docs, and examples:
- remove top-level `plan`;
- remove top-level `status`;
- remove top-level `restore`;
- remove top-level `artifacts`;
- remove top-level `locks`;
- remove `--session-id` from the public command syntax for session-aware
commands.
Do not keep long-term deprecated aliases unless a later roadmap explicitly
chooses a compatibility window.
`status --manifest` does not fit the session-oriented command shape. Remove it
from the public CLI in this cleanup. If direct manifest inspection is needed
later, add a separate diagnostic command in a future roadmap rather than keeping
it as a special case in `session status`.
## Implementation Step 1: Add New Session-Oriented Interface
Status: Implemented
Add the target command forms while preserving current behavior internally.
Implementation requirements:
- Add positional session ID parsing helpers in `internal/app`.
- Keep the existing `loadCommandConfig` behavior and populate
`config.SessionLoadOptions.SessionID` from the positional ID.
- Add or update command wrappers:
- `Run(ctx, args, out)` parses `run <session_id>`.
- `Resume(ctx, args, out)` parses `resume <session_id>`.
- `Analyze(ctx, args, out)` parses `analyze <session_id>`.
- `Publish(ctx, args, out)` parses `publish <session_id>`.
- `RunStage(ctx, args, out)` parses `run-stage <stage> <session_id>`.
- `Clean(ctx, args, out)` parses `clean <session_id>` and keeps
`clean --all`.
- Extend `Session(ctx, args, out)` dispatch to support:
- `init <session_id>`
- `validate <session_id>`
- `status <session_id>`
- `plan <session_id>`
- `restore <session_id>`
- `artifacts <session_id>`
- `locks <session_id>`
- `locks add <session_id> <source>`
- `locks remove <session_id> <source>`
- Keep storage access through the existing app-level object-store helper.
- Keep AWS SDK details behind storage adapters.
- Keep the runner, stages, manifest behavior, archive behavior, restore
planning, lock semantics, and artifact catalog behavior unchanged.
Acceptance criteria:
- New forms execute the same code paths and produce equivalent results.
- Positional session ID mismatch with concrete local or remote `session.yml`
fails through existing session identity checks.
- Remote session fallback still uses the positional session ID as the lookup
value.
- Current command tests cover the new forms before old forms are removed.
## Implementation Step 2: Remove Old Public Forms
Status: Implemented
Remove compatibility aliases and make the session-oriented interface the only
documented and supported public CLI.
Implementation requirements:
- Remove top-level dispatch for:
- `plan`
- `status`
- `restore`
- `artifacts`
- `locks`
- Remove `--session-id` flags from public session-aware commands.
- Keep `--previous-session-id` as an expected previous-session identity flag.
- Keep explicit `--session <path>` for loading a local concrete session file,
but still require the positional session ID for commands that operate on a
session.
- Remove `status --manifest`.
- Update usage text and invalid-command errors.
- Update `docs/cli.md` and `docs/operations.md` to use only the new forms.
- Update any roadmap docs that mention old helper command names.
- Update tests to expect old top-level helper commands and `--session-id` forms
to fail.
Acceptance criteria:
- Top-level command list is exactly:
- `run`
- `run-stage`
- `resume`
- `analyze`
- `publish`
- `clean`
- `session`
- All session-oriented commands use `narratio session <subcommand>
<session_id> [--flags]`, except nested lock mutation forms, which use
`narratio session locks add|remove <session_id> <source> [--flags]`.
- `clean <session_id>` and `clean --all` remain top-level.
- Current-behavior docs and tests no longer advertise `--session-id`.
## Test Guidance
Focused tests:
- `go test ./internal/app -run TestExecute -v`
- `go test ./internal/app -run 'Session|Status|Restore|Clean|Locks|Artifacts|Plan|RunStage|Analyze|Publish' -v`
- `go test ./internal/config -v`
Full validation:
- `go test ./...`
Test cases to add or update:
- `run <session_id>` loads local and remote sessions through the existing
config path.
- `resume <session_id>`, `analyze <session_id>`, and `publish <session_id>`
preserve current behavior.
- `run-stage <stage> <session_id>` preserves current run-stage output and
force/artifact-selection behavior.
- `session plan <session_id>` replaces top-level `plan`.
- `session status <session_id>` replaces top-level session status.
- `session validate <session_id>` replaces `session validate --session-id`.
- `session init <session_id>` writes the same local or remote concrete
`session.yml`.
- `session restore <session_id>` preserves restore planning/execution.
- `session artifacts <session_id> --remote` preserves promoted-output
availability reporting.
- `session locks <session_id>`, `session locks add <session_id> <source>`, and
`session locks remove <session_id> <source>` preserve static/remote lock
semantics.
- `clean <session_id>` preserves session cleanup behavior, while `clean --all`
remains unchanged.
- Old top-level helper commands fail after Step 2.
- `--session-id` fails after Step 2.
- `status --manifest` fails after Step 2.
## Documentation Guidance
Update only after implementation lands:
- `docs/cli.md`
- `docs/operations.md`
- any internal docs that list command names or examples
Keep planned behavior only in this roadmap until the command refactor is
implemented.
## Architecture Guardrails
- Keep Narratio explicit and stage-driven.
- Do not introduce a generic workflow or command framework abstraction.
- Reuse existing app command helpers where practical.
- Keep config loading strict and centralized.
- Keep storage details behind `storage.ObjectStore`.
- Keep secret-backed object-store construction in `internal/app`.
- Preserve manifest-driven resume and restore behavior.
- Treat command renaming as a public CLI contract change, not a runtime stage
behavior change.

File diff suppressed because it is too large Load Diff

287
docs/roadmap/publish.md Normal file
View File

@@ -0,0 +1,287 @@
# Roadmap: Publish Contract
Status: Planned
## Problem
Narratio currently uses several terms for one operator-facing concept:
- `archive` is the stage that uploads run state and commits remote current
state.
- `publish` is the convenience command that force-runs the archive stage.
- `promote`, `promoted`, and `promote_artifacts` describe configured top-level
remote output writes.
This mixed vocabulary makes the public contract harder to explain. Operators
should not need to distinguish "archive the run", "publish the run", and
"promote artifacts" when these are all part of the same publish action.
The public model should use:
- `publish` for the stage, command, config section, and action;
- `published` for an expected remote output that exists at its top-level
current destination;
- `publish rules` for the configured source-to-destination output rules;
- `locked` for sources whose top-level published destination must not be
overwritten;
- `run history` for immutable per-run records under `runs/<run_id>/`.
## Target Model
The public stage is `publish`.
The convenience command:
narratio publish <session_id>
is equivalent to:
narratio run-stage publish <session_id> --force
Pipeline configuration uses `publish`:
publish:
enabled: true
upload_run: true
outputs:
- source: narratio.transcript.final_trimmed
- source: narratio.artifact.session_recap
locks:
- source: narratio.artifact.session_recap
reason: Final recap was manually edited.
Publish output rules are source-based. Each rule writes one artifact source to
a top-level remote destination. If `dest` is omitted, Narratio derives the
destination from the artifact registry or configured artifact output path.
The mutable remote lock store remains:
{session_prefix}/locks.yml
Remote availability output uses `published`:
Published:
- narratio.transcript.final_trimmed remote=published
- narratio.artifact.session_recap locked remote=published
The remote key layout is otherwise unchanged:
- immutable run history stays under `{session_prefix}/runs/{run_id}/`;
- current state stays under `{session_prefix}/current/manifest.json`;
- the final commit marker stays `{session_prefix}/current/run_id.txt`;
- `current/run_id.txt` is still written last.
## Compatibility Policy
This is a hard cutover.
After implementation:
- `pipeline.archive` is rejected by strict YAML decoding.
- `pipeline.archive.promote_artifacts` is rejected.
- `pipeline.workspace.cleanup_after_archive` is rejected.
- `pipeline.spool.delete_audio_after_archive` is rejected.
- `narratio run-stage archive <session_id>` is an unknown stage.
- manifests that record an `archive` stage are not migrated.
- old archive/promotion metadata keys are not read as compatibility fallbacks.
Existing remote objects are not moved or renamed. Remote layout remains stable;
the rename changes configuration, stage names, status output, metadata, helper
names, tests, examples, and documentation.
## Implementation Stages
### Stage 1: Public Schema and Stage Cutover
Status: Planned
Switch the public config and stage contract to publish terminology.
Implementation requirements:
- Replace `pipeline.archive` with `pipeline.publish`.
- Replace `archive.promote_artifacts` with `publish.outputs`.
- Keep output rule fields:
- `source`
- `dest`
- `required`
- Replace `pipeline.archive.locks` with `pipeline.publish.locks`.
- Rename post-publish cleanup fields:
- `pipeline.workspace.cleanup_after_publish`
- `pipeline.spool.delete_audio_after_publish`
- Rename the registered stage from `archive` to `publish`.
- Update stage order so `publish` runs after `analyze` and before `notify`.
- Update top-level `narratio publish` to target stage `publish`.
- Keep `run-stage --artifacts <names> publish` support.
- Reject `run-stage --artifacts <names>` for stages other than `analyze` and
`publish`.
- Preserve the remote commit ordering and storage adapter boundaries.
Acceptance criteria:
- `narratio run-stage publish <session_id>` executes the publish stage.
- `narratio publish <session_id>` force-runs the publish stage.
- `narratio run-stage archive <session_id>` fails clearly as an unknown stage.
- Old archive config fields fail strict decoding.
- New publish config fields load, default, and validate.
### Stage 2: Runtime Terminology and Metadata Cutover
Status: Planned
Rename implementation concepts and runtime output to publish terminology.
Implementation requirements:
- Rename archive/promotion config and runtime types conceptually to
publish/output terms.
- Rename the remote key helper intent from promoted artifact to published
output while keeping generated keys unchanged.
- Change helper output:
- `Promoted:` becomes `Published:`
- `remote=promoted` becomes `remote=published`
- lock output uses `published` / `not-published`
- Rename publish-stage metadata, including:
- `promoted_paths` to `published_paths`
- `promoted_files_uploaded` to `published_files_uploaded`
- `skipped_optional_promotions` to `skipped_optional_outputs`
- `skipped_unselected_promotions` to `skipped_unselected_outputs`
- `locked_promotion_count` to `locked_output_count`
- `locked_promotions` to `locked_outputs`
- Update previous-cache and restore logic to use the `publish` stage and
`published_paths` metadata only.
- Keep run-local stage output materialization separate from remote publish
terminology. If local helper names are confusing, rename them to
materialization-oriented names rather than publish names.
Acceptance criteria:
- Status and artifact helper output use `Published:` and `remote=published`.
- Publish metadata contains only publish/output terminology.
- Previous-cache and restore behavior works with publish metadata and does not
depend on old archive metadata.
- Storage adapters still receive explicit keys and no AWS SDK details leak into
app or stage logic.
### Stage 3: Documentation, Examples, and Final Cleanup
Status: Planned
Update implemented-behavior docs and remove stale public terminology after the
runtime cutover lands.
Implementation requirements:
- Update current-behavior docs:
- `docs/config.md`
- `docs/cli.md`
- `docs/operations.md`
- `docs/troubleshooting.md`
- `docs/architecture.md`
- relevant files under `docs/internal/`
- Rename `docs/internal/stage-archive.md` to
`docs/internal/stage-publish.md`.
- Update internal documentation links and references.
- Update examples to use:
- `publish.outputs`
- `publish.locks`
- `cleanup_after_publish`
- `delete_audio_after_publish`
- Update tests and final searches so old terminology remains only in this
roadmap as historical context.
Acceptance criteria:
- Maintained examples load and validate.
- Current-behavior docs describe only implemented publish terminology.
- Internal docs describe run history, published outputs, locks, and current
commit ordering clearly.
- Old user-facing archive/promote wording is removed except where discussing
historical behavior in this roadmap.
## Test Guidance
Focused tests:
- `go test ./internal/config -v`
- `go test ./internal/app -v`
- `go test ./internal/stage -v`
- `go test ./internal/artifacts -v`
Full validation:
- `go test ./...`
Config tests to add or update:
- `publish.outputs` defaults and validates.
- `publish.outputs[].dest` derives from the artifact registry when omitted.
- `publish.locks` validates with the same source rules as publish outputs.
- old `archive` fails strict decode.
- old `promote_artifacts` fails strict decode.
- old cleanup fields fail strict decode.
App and stage tests to add or update:
- stage order uses `publish` before `notify`.
- `run-stage publish` succeeds.
- `run-stage archive` fails clearly.
- `narratio publish` force-runs the `publish` stage.
- `--artifacts` is accepted for `run-stage publish`.
- `--artifacts` error text names `analyze` and `publish`.
- status and artifact list output show `Published:` and `remote=published`.
- lock output says `published` or `not-published`.
- previous-cache and restore use `publish` stage metadata.
Final searches:
- Config/stage names:
- `pipeline.archive`
- `archive:`
- `promote_artifacts`
- `cleanup_after_archive`
- `delete_audio_after_archive`
- User-facing output:
- `Promoted:`
- `remote=promoted`
- `not-promoted`
- Runtime symbols and metadata:
- `ArchiveConfig`
- `ArchivePromotionRule`
- `S3PromotedArtifactKey`
- `promoted_paths`
- `promoted_files_uploaded`
- `locked_promotions`
Expected remaining matches should be limited to this roadmap and narrowly
justified historical references until the roadmap is fully retired.
## Architecture Guardrails
- Keep Narratio explicit and stage-driven.
- Do not introduce a generic workflow or DAG abstraction.
- Keep strict YAML decoding.
- Keep remote path construction centralized.
- Keep storage details behind `storage.ObjectStore`.
- Keep AWS SDK types inside storage adapters.
- Preserve manifest-driven resume and restore behavior.
- Preserve current-state commit ordering with `current/run_id.txt` written
last.
- Keep raw secrets out of configs, manifests, logs, generated configs, and
publish metadata.
- Keep planned behavior only in this roadmap until implementation lands.
## Assumptions
- This is a breaking public/config/stage contract change.
- No compatibility aliases are retained.
- No migration logic is needed for in-progress local manifests.
- No migration logic is needed for old remote manifests.
- Existing remote objects are not moved or renamed.
- `publish` means uploading run history, writing configured published outputs,
and committing current state.
- `run history` is the preferred term for immutable per-run records under
`runs/<run_id>/`.
- `archive` remains acceptable only as a generic English concept in historical
roadmap context, not as a public Narratio command, config field, stage name,
or metadata term after implementation.

210
docs/roadmap/transcripts.md Normal file
View File

@@ -0,0 +1,210 @@
# Roadmap: Transcript Artifact Naming
Status: Implemented
## Problem
Narratio's built-in transcript artifact names and canonical paths currently mix
operator-facing artifact meaning with historical stage and tool terminology:
- `narratio.transcript.merged` maps to `transcripts/merged.json`.
- `narratio.transcript.polished` maps to `transcripts/processed.json`.
- `narratio.transcript.full` maps to `transcripts/normalized.json`.
- `narratio.transcript.trimmed` maps to `transcripts/trimmed.json`.
This makes the public artifact surface harder to reason about. Operators see
`full`, `normalized`, `processed`, `polished`, `merged`, and `trimmed` used in
different places for the same transcript lineage.
The transcript source IDs, canonical paths, and manifest output kinds should
use one vocabulary based on each transcript's role in the session artifact
model.
## Target Model
Built-in transcript artifacts should use these public source IDs, canonical
paths, and manifest output kinds:
| Source ID | Canonical path | Output kind | Meaning |
| --- | --- | --- | --- |
| `narratio.transcript.base` | `transcripts/base.json` | `transcript_base` | First unified transcript produced by merging per-speaker raw transcripts. |
| `narratio.transcript.polished` | `transcripts/polished.json` | `transcript_polished` | Audita-polished transcript. |
| `narratio.transcript.final` | `transcripts/final.json` | `transcript_final` | Full final transcript after normalization. |
| `narratio.transcript.final_trimmed` | `transcripts/final.trimmed.json` | `transcript_final_trimmed` | Trimmed version of the final transcript. |
Stage names remain process-oriented and unchanged:
- `merge`
- `polish`
- `normalize`
- `trim`
Downstream adapter contracts also remain process-oriented. The rename changes
Narratio's artifact model, canonical paths, config examples, archive promotion
sources, lock sources, status output, and documentation. It should not rename
the stages themselves or move external integration details into stage logic.
## Compatibility Policy
This is a hard cutover.
After implementation, these old source IDs should be rejected:
- `narratio.transcript.merged`
- `narratio.transcript.full`
- `narratio.transcript.trimmed`
These old canonical paths should not be compatibility fallbacks:
- `transcripts/merged.json`
- `transcripts/processed.json`
- `transcripts/normalized.json`
- `transcripts/trimmed.json`
Existing remote archives are not migrated automatically. Operators who want
new promoted keys for old sessions should republish those sessions after
updating configuration.
## Implementation Stages
### Stage 1: Centralize Transcript Artifact Naming
Status: Implemented
Consolidate transcript artifact source IDs, canonical paths, and output kinds
in the artifact/path layer before changing runtime behavior.
Implementation requirements:
- Add or consolidate constants/helpers for built-in transcript source IDs.
- Add or consolidate constants/helpers for canonical transcript paths.
- Add or consolidate constants/helpers for transcript manifest output kinds.
- Keep source ID, path, and output-kind mappings in one registry or one
obviously shared artifact model.
- Update artifact registry tests to prove the target mapping.
- Avoid changing stage output behavior in this stage unless the implementation
is simpler and still reviewable.
Acceptance criteria:
- There is one clear source of truth for built-in transcript artifact names,
paths, and output kinds.
- Tests prove the new target mapping in the artifact layer.
- No generic workflow abstraction is introduced.
### Stage 2: Rename Runtime Outputs and Defaults
Status: Implemented
Switch runtime behavior to the new transcript artifact model.
Implementation requirements:
- Update `merge` to write and record `transcripts/base.json` with
`transcript_base`.
- Update `polish` to write and record `transcripts/polished.json` with
`transcript_polished`.
- Update `normalize` to write and record `transcripts/final.json` with
`transcript_final`.
- Update `trim` to write and record `transcripts/final.trimmed.json` with
`transcript_final_trimmed`.
- Update normalize and trim defaults to:
- `pipeline.normalize.output_path: transcripts/final.json`
- `pipeline.trim.output_path: transcripts/final.trimmed.json`
- Update built-in artifact resolution, archive promotion destination
derivation, archive locks, status output, artifact catalog output,
previous-cache resolution, restore planning, and restore execution to use
the new registry values.
- Ensure old source IDs fail config validation.
Acceptance criteria:
- New runs produce the target canonical transcript files.
- Manifest outputs use the target output kinds.
- Archive promotion and lock validation accept new source IDs and reject old
source IDs.
- Status and artifact listing display new source IDs.
- Restore uses the new canonical paths and does not restore old transcript
paths as canonical outputs.
### Stage 3: Update Tests, Examples, and Current Documentation
Status: Implemented
Update all implemented-behavior references after the runtime cutover lands.
Implementation requirements:
- Update examples to use `narratio.transcript.final_trimmed` and
`transcripts/final.trimmed.json` where trimmed final transcript is intended.
- Update examples that refer to full final transcripts to use
`narratio.transcript.final` and `transcripts/final.json`.
- Update `docs/config.md`, `docs/internal/artifacts.md`, stage docs,
CLI examples, operations examples, archive examples, lock examples, and
status/artifact-list examples.
- Add strict validation tests proving old source IDs are rejected.
- Mark roadmap stages implemented only after code, tests, examples, and
current-behavior docs agree.
Acceptance criteria:
- Maintained examples load and validate.
- Current-behavior docs describe only implemented new names.
- Old names remain only in this roadmap as historical/planning context until
this roadmap is retired or archived.
## Test Guidance
Run focused tests while implementing:
- `go test ./internal/artifacts -v`
- `go test ./internal/config -v`
- `go test ./internal/stage -v`
- `go test ./internal/app -v`
Run full validation before finishing:
- `go test ./...`
Run final searches:
- Old source IDs:
- `narratio.transcript.merged`
- `narratio.transcript.full`
- `narratio.transcript.trimmed`
- Old paths:
- `transcripts/merged.json`
- `transcripts/processed.json`
- `transcripts/normalized.json`
- `transcripts/trimmed.json`
- Old output kinds:
- `transcript_merged`
- `transcript_processed`
- `transcript_normalized`
- `transcript_trimmed`
Expected remaining matches should be limited to this roadmap's
historical/planning references until the roadmap is fully completed.
## Architecture Guardrails
- Keep Narratio explicit and stage-driven; do not introduce a generic workflow
or DAG abstraction.
- Keep path and artifact naming in centralized helpers rather than scattered
string concatenation.
- Preserve manifest-driven resume behavior.
- Keep storage details behind storage adapters.
- Do not move Seriatim, Audita, or Scriptorium command details out of their
adapter boundaries.
- Keep current-behavior documentation in sync only after implementation lands;
planned behavior belongs in this roadmap until then.
## Assumptions
- The cutover is intentionally not backward-compatible.
- Existing remote archive objects are not renamed or migrated automatically.
- Stage names and downstream adapter request field names remain unchanged.
- The term `base` is preferred over `merged` for the first unified transcript.
- The term `final` is preferred over `full` or `normalized` for the full final
transcript.
- The trimmed final path is `transcripts/final.trimmed.json`.

379
docs/troubleshooting.md Normal file
View File

@@ -0,0 +1,379 @@
# Troubleshooting
## Purpose
Canonical operator troubleshooting guide for recurring implemented Narratio failures.
## Config file discovery failure
Symptom:
- `run`, `resume`, `run-stage`, `session plan`, or `session restore` fails with config/session not found.
Likely Cause:
- `pipeline.yml` or `session.yml` is missing from system discovery paths.
- the selected campaign ID does not exist under `pipeline.campaigns.root`.
- a local working-directory config file was not passed explicitly.
Diagnostics:
```bash
ls -l /usr/local/etc/narratio/pipeline.yml /etc/narratio/pipeline.yml
ls -l /usr/local/etc/narratio/session.yml /etc/narratio/session.yml
```
Safe Fix:
- pass explicit `--config`, `--campaign <id>`, `--campaign-file <path>`, and `--session` as appropriate.
- or place files in documented discovery paths and set `pipeline.campaigns.default_campaign_id`.
Links:
- [docs/config.md](./config.md)
- [docs/cli.md](./cli.md)
## Templated session file rejected
Symptom:
- load fails with a message that `session.yml must be concrete`.
Likely Cause:
- a template authoring file such as `session.template.yml` was passed to `--session` or uploaded as remote `session.yml`.
- `session.yml` still contains `{{ ... }}` placeholders.
Diagnostics:
```bash
narratio session plan 2026-04-04 --config /path/to/pipeline.yml --campaign-file /path/to/campaign.yml --session ./session.yml
```
Safe Fix:
- generate concrete YAML with `narratio session init`.
- pass the generated concrete `session.yml` to downstream commands or upload it through `session init --remote`.
Links:
- [docs/config.md](./config.md)
## Strict YAML decode or validation failure
Symptom:
- config load fails with unknown field or validation error.
Likely Cause:
- typo/stale field name.
- missing required fields or invalid constraints.
Diagnostics:
```bash
narratio session plan 2026-04-04 --config /path/to/pipeline.yml --campaign-file /path/to/campaign.yml --session /path/to/session.yml
```
Safe Fix:
- align fields/values to canonical config reference and examples.
Links:
- [docs/config.md](./config.md)
- [examples/](../examples/)
## `--artifacts` selection failure
Symptom:
- `run`/`resume`/`run-stage` fails with invalid or unknown artifact selection.
Likely Cause:
- `--artifacts` contains blank names or unknown artifact keys.
- `pipeline.scriptorium.artifacts` missing while using `--artifacts`.
Diagnostics:
```bash
narratio run 2026-04-04 --config /path/to/pipeline.yml --campaign-file /path/to/campaign.yml --session /path/to/session.yml --artifacts player_handout
```
Safe Fix:
- use configured artifact keys only.
- ensure `pipeline.scriptorium.artifacts` is defined.
Links:
- [docs/cli.md](./cli.md)
- [docs/config.md](./config.md)
## `run-stage --artifacts` on unsupported stage
Symptom:
- `run-stage` fails because `--artifacts` is only supported for `analyze` and `archive`.
Likely Cause:
- `--artifacts` was used with a stage other than `analyze` or `archive`.
Diagnostics:
```bash
narratio run-stage polish 2026-04-04 --config /path/to/pipeline.yml --campaign-file /path/to/campaign.yml --session /path/to/session.yml --artifacts session_recap
```
Safe Fix:
- use `--artifacts` only with `run-stage analyze ...` or `run-stage archive ...`.
Links:
- [docs/cli.md](./cli.md)
## Configured artifact dependency/input validation failure
Symptom:
- config validation fails for `depends_on`, `narratio.artifact.<name>` source, or artifact output path.
Likely Cause:
- `narratio.artifact.<name>` source missing matching `depends_on` key.
- dependency references unknown artifact key.
- dependency self-reference or enabled dependency cycle.
- artifact output path missing/invalid/outside `artifacts/` root.
Diagnostics:
```bash
narratio session plan 2026-04-04 --config /path/to/pipeline.yml --campaign-file /path/to/campaign.yml --session /path/to/session.yml
```
Safe Fix:
- ensure artifact-to-artifact inputs have explicit `depends_on` entries using artifact keys.
- ensure referenced artifacts exist and define valid `output_path` values.
- keep output paths relative and under `artifacts/`.
Links:
- [docs/config.md](./config.md)
- [docs/internal/stage-analyze.md](./internal/stage-analyze.md)
## Required configured artifact input unavailable at analyze time
Symptom:
- analyze fails because configured input source is unavailable.
Likely Cause:
- required upstream configured artifact was not selected/executed this run.
- non-executable dependency output file is missing or invalid on disk.
Diagnostics:
```bash
narratio session status 2026-04-04
narratio run-stage analyze 2026-04-04 --config /path/to/pipeline.yml --campaign-file /path/to/campaign.yml --session /path/to/session.yml --artifacts player_handout
```
Safe Fix:
- run analyze with needed artifacts selected.
- or ensure dependency output file exists at configured path and is valid.
Links:
- [docs/operations.md](./operations.md)
- [docs/config.md](./config.md)
## Manifest/status path failure
Symptom:
- `session status` fails because config/session state is missing, unreadable, or invalid.
Likely Cause:
- wrong session ID.
- wrong config/campaign/session file selected.
- manifest removed after cleanup.
Diagnostics:
```bash
narratio session status 2026-04-04
```
Safe Fix:
- use the same session ID and config files that will be used for `run`, `resume`, or `run-stage`.
Links:
- [docs/cli.md](./cli.md)
- [docs/operations.md](./operations.md)
## Session lock conflict (`.lock`)
Symptom:
- `run`, `resume`, `run-stage`, or `session restore` fails with lock conflict for session workdir.
Likely Cause:
- another Narratio process is running same session.
- stale lock from interrupted prior run.
Diagnostics:
```bash
ls -l {workspace.root}/work/{campaign}/{session_id}/.lock
cat {workspace.root}/work/{campaign}/{session_id}/.lock
ps aux | grep narratio
```
Safe Fix:
- wait for active process to finish.
- if no process is active, remove only stale session `.lock` file.
Links:
- [docs/operations.md](./operations.md)
- [docs/internal/workspace.md](./internal/workspace.md)
## Restore remote current pointer or manifest missing
Symptom:
- `session restore` fails with remote current pointer or current manifest errors.
Likely Cause:
- `current/run_id.txt` was never published.
- `current/manifest.json` is missing for the session prefix.
- archive commit did not complete.
Diagnostics:
```bash
narratio session restore 2026-04-04 --config /path/to/pipeline.yml --campaign-file /path/to/campaign.yml --session /path/to/session.yml --dry-run
```
Safe Fix:
- verify archive stage succeeded for the target session.
- rerun/archive from a healthy source workspace so current pointers are published.
Links:
- [docs/operations.md](./operations.md)
- [docs/internal/stage-archive.md](./internal/stage-archive.md)
## Restore manifest identity mismatch
Symptom:
- `session restore` fails because remote manifest session or campaign does not match requested values.
Likely Cause:
- wrong positional session ID or wrong session config selected.
- archive prefix points to a different campaign/session.
Diagnostics:
```bash
narratio session restore 2026-04-04 --config /path/to/pipeline.yml --campaign-file /path/to/campaign.yml --session /path/to/session.yml --dry-run
```
Safe Fix:
- use the correct session config and positional session ID.
- verify campaign/session identity in local config before restore.
Links:
- [docs/config.md](./config.md)
- [docs/operations.md](./operations.md)
## Restore conflict without `--force`
Symptom:
- `session restore` fails with `restore conflict` and conflict counts.
Likely Cause:
- local durable file differs from remote file for one or more planned restore paths.
Diagnostics:
```bash
narratio session restore 2026-04-04 --config /path/to/pipeline.yml --campaign-file /path/to/campaign.yml --session /path/to/session.yml --dry-run
```
Safe Fix:
- review planned conflicts.
- rerun with `--force` only when remote state should overwrite local state.
Links:
- [docs/cli.md](./cli.md)
- [docs/operations.md](./operations.md)
## Restore report expectations
Symptom:
- operator expects restore report file but does not find one.
Likely Cause:
- restore was executed in `--dry-run` mode.
- restore failed before report persistence path (for example lock acquisition failure).
Diagnostics:
```bash
ls -l {workspace.root}/work/{campaign}/{session_id}/reports/restore-latest.json
```
Safe Fix:
- run non-dry-run restore for durable report output.
- resolve lock or early preflight failures and retry.
Links:
- [docs/operations.md](./operations.md)
## Secrets env-dir or credential-env failure
Symptom:
- startup fails loading secrets directory, or stage fails due to missing credential env vars.
Likely Cause:
- invalid `pipeline.secrets.env_dir` path/permissions.
- required credential env var unset/empty.
Diagnostics:
```bash
ls -la /path/to/secrets_dir
env | grep -E 'AUDITA|OBJECT_STORAGE|AWS|SCRIPTORIUM'
```
Safe Fix:
- fix secrets directory and credential env vars.
- keep secret values out of YAML.
Links:
- [docs/config.md](./config.md)
## S3-audio prepare failure
Symptom:
- `prepare` fails in S3 mode (listing/downloading/no audio/backend error).
Likely Cause:
- wrong `session.inputs.audio_s3.prefix`.
- no `.flac` files at resolved prefix.
- invalid/missing object-store credentials or backend config.
- mixed local+S3 audio input config.
Diagnostics:
```bash
narratio run-stage prepare 2026-04-04 --config /path/to/pipeline.yml --campaign-file /path/to/campaign.yml --session /path/to/session.yml
```
Safe Fix:
- configure exactly one audio source mode.
- verify `.flac` files and storage access.
Links:
- [docs/config.md](./config.md)
- [docs/operations.md](./operations.md)
## Archive promotion/current-pointer failure
Symptom:
- archive fails on required promotion source missing or pointer write failure.
Likely Cause:
- required promoted file absent (including analyze outputs not generated for this run).
- storage upload failed before `current/run_id.txt` commit marker write.
Diagnostics:
```bash
narratio session status 2026-04-04
narratio run-stage archive 2026-04-04 --config /path/to/pipeline.yml --campaign-file /path/to/campaign.yml --session /path/to/session.yml
```
Safe Fix:
- rerun or resume upstream stages to generate required files.
- adjust promotion `source`/`dest` rules to match artifacts that must exist.
- retry after storage issue is resolved.
Links:
- [docs/operations.md](./operations.md)
- [docs/config.md](./config.md)
- [docs/internal/stage-archive.md](./internal/stage-archive.md)

View File

@@ -0,0 +1 @@
[]

View File

@@ -1,10 +1,6 @@
session_id: 2026-05-03 campaign_id: sample-campaign
campaign: sample-campaign session_template_file: ./session.template.yml
date: 2026-05-03
title: Sample Session
inputs: inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml glossary_file: ./glossary.yml

View File

@@ -0,0 +1 @@
[]

View File

@@ -0,0 +1,3 @@
session_id: "{{ session_id }}"
inputs:
audio_dir: ./audio

View File

@@ -0,0 +1,5 @@
match:
- speaker: "Eric Rakestraw"
match:
- "Eric_Rakestraw"
- "Eric"

View File

@@ -0,0 +1,175 @@
# Full annotated pipeline example for implemented Narratio config fields.
# Values are safe placeholders and must be adapted per environment.
workspace:
# Optional: defaults to /var/lib/narratio.
root: /var/lib/narratio/workspace
# Optional: remove run-scoped workdir after successful archive commit.
cleanup_after_archive: false
# Optional: local secret file loader (directory of ENV_VAR_NAME files).
# secrets:
# env_dir: ./secrets
storage:
# Optional storage backend selector; use "s3" for archive + S3 audio workflows.
backend: s3
s3:
# Required when using S3 audio or S3 archive uploads.
bucket: my-dnd-archive
# Optional; defaults to "dnd".
root_prefix: dnd
# Optional region/endpoint settings.
region: us-east-1
endpoint: ""
force_path_style: false
# Optional; defaults shown explicitly.
access_key_id_env: OBJECT_STORAGE_KEY_ID
secret_access_key_env: OBJECT_STORAGE_KEY
campaigns:
# Optional; defaults to /usr/local/share/narratio/campaigns.
root: /usr/local/share/narratio/campaigns
# Optional command default when --campaign is omitted.
default_campaign_id: sample-campaign
spool:
# Optional; defaults to /var/spool/narratio.
root: /var/spool/narratio
# Optional cleanup of run-scoped spool audio after successful archive commit.
delete_audio_after_archive: false
archive:
# Optional booleans; defaults are true.
enabled: true
upload_run: true
# Optional promotion rules; sources use Narratio artifact source IDs.
promote_artifacts:
- source: narratio.transcript.final_trimmed
dest: transcripts/final.trimmed.json
required: true
- source: narratio.artifact.session_recap
dest: artifacts/session_recap.md
required: true
- source: narratio.artifact.player_handout
dest: artifacts/player_handout.md
required: false
whisperx:
# Required.
transcribe_url: "https://transcription.example.com/transcribe"
# Optional overrides; defaults shown explicitly.
language: en
timeout: 30m
retries: 3
retry_delay: 2s
concurrency: 2
seriatim:
# Optional overrides; defaults shown explicitly.
binary: seriatim
timeout: 10m
output_schema: seriatim-intermediate
coalesce_gap: 3.0
report: true
env:
# Optional advanced tuning; set only when needed.
overlap_word_run_gap: 1.0
overlap_word_run_reorder_window: 1.0
backchannel_max_duration: 2.0
filler_max_duration: 1.25
audita:
# Optional overrides; defaults shown explicitly where applicable.
binary: audita
timeout: 3h
llm_api_key_env: AUDITA_LLM_API_KEY
modules: [glossary, homophones, spoken_word, grammar]
base_url: ""
model: ""
total_llm_concurrency: 2
proposal_llm_concurrency: 1
validation_model: ""
validation_llm_concurrency: 1
transcript_description: ""
config_path: /usr/local/etc/audita/config.yml
output_schema: audita-v1
work_dir_retention: auto
report: true
normalize:
# Optional; defaults shown explicitly.
output_path: transcripts/final.json
output_schema: seriatim-intermediate
report: true
trim:
# Keep disabled unless bounds prompt integration is configured.
enabled: false
output_path: transcripts/final.trimmed.json
bounds:
prompt_id: dnd.session_bounds
profile_id: local-fast
transcript_input_name: transcript
output_path: reports/session_bounds.json
timeout: 10m
render_debug: false
render_output_path: reports/session_bounds.render.json
seriatim:
report: false
scriptorium:
binary: scriptorium
config_path: /usr/local/etc/scriptorium/config.yml
timeout: 10m
render_debug: false
artifacts:
# Configured artifact keys map to source IDs narratio.artifact.<key>.
session_recap:
enabled: true
prompt_id: dnd.session_recap
profile_id: local-fast
output_path: artifacts/session_recap.md
timeout: 10m
inputs:
transcript:
source: narratio.transcript.final_trimmed
required: true
previous_recap:
source: narratio.previous_session.artifact.session_recap
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: session_recap
# Example dependent artifact:
# - depends_on entries use artifact keys.
# - narratio.artifact.<key> sources require matching depends_on membership.
player_handout:
enabled: true
depends_on:
- session_recap
prompt_id: dnd.player_handout
profile_id: local-fast
output_path: artifacts/player_handout.md
timeout: 10m
inputs:
recap:
source: narratio.artifact.session_recap
required: true
transcript:
source: narratio.transcript.final_trimmed
required: true
vars:
session_id: true
campaign_name: true
output_kind: player_handout
notification:
# Optional notification settings.
backend: ""
recipient: ""
timeout: 30s

View File

@@ -1,114 +1,6 @@
workspace: campaigns:
root: ./tmp/narratio-workspace root: /usr/local/share/narratio/campaigns
default_campaign_id: sample-campaign
storage:
backend: local
secrets:
# Optional: load environment variables from files in this directory.
# File name = env var name; file contents = env var value.
env_dir: /var/local/narratio/secrets
whisperx: whisperx:
transcribe_url: "https://transcription.example.com/transcribe" transcribe_url: "https://transcription.example.com/transcribe"
language: "en"
timeout: "30m"
retries: 3
retry_delay: "2s"
concurrency: 2
seriatim:
binary: "seriatim"
timeout: "10m"
output_schema: "seriatim-intermediate"
coalesce_gap: 3.0
report: true
env:
overlap_word_run_gap: 1.0
overlap_word_run_reorder_window: 1.0
backchannel_max_duration: 2.0
filler_max_duration: 1.25
audita:
binary: "audita"
timeout: "3h"
llm_api_key_env: "AUDITA_LLM_API_KEY"
modules:
- glossary
- homophones
- glossary
- spoken_word
- grammar
- homophones
- glossary
base_url: "https://openrouter.ai/api/v1"
model: "openrouter/google/gemma-4-31b-it"
llm_concurrency: 1
validation_model: ""
validation_llm_concurrency: 1
report: true
normalize:
# Session-workdir-relative when not absolute.
output_path: "transcripts/normalized.json"
output_schema: "seriatim-intermediate"
report: true
trim:
enabled: true
# Session-workdir-relative when not absolute.
output_path: "transcripts/trimmed.json"
bounds:
prompt_id: "dnd_session.bounds"
# Empty means use prompt default profile.
profile_id: ""
transcript_input_name: "transcript"
output_path: "artifacts/session_bounds.json"
timeout: "10m"
render_debug: false
render_output_path: "artifacts/session_bounds.render.json"
seriatim:
report: false
scriptorium:
binary: "scriptorium"
config_path: "/etc/scriptorium/config.yml"
timeout: "10m"
render_debug: false
artifacts:
session_recap:
enabled: true
prompt_id: "dnd.session_recap"
profile_id: "local-quality"
output_path: "artifacts/session_recap.md"
timeout: "10m"
# Optional per-artifact override of global scriptorium.render_debug.
# render_debug: true
inputs:
transcript:
# Available transcript sources:
# - trimmed_transcript (recommended for session_recap)
# - normalized_transcript (recommended for future full-session analysis)
# - processed_transcript (raw Audita-polished output)
source: "trimmed_transcript"
required: true
previous_recap:
source: "previous_session_artifact"
artifact: "session_recap"
# Optional: set when previous recap is available.
path: ""
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: "session_recap"
analyzer:
timeout: 20m
artifacts:
output_dir: artifacts
notification:
timeout: 10s

View File

@@ -0,0 +1,116 @@
workspace:
root: /var/lib/narratio/workspace
cleanup_after_archive: true
storage:
backend: s3
s3:
bucket: my-dnd-archive
root_prefix: dnd
region: us-east-1
access_key_id_env: OBJECT_STORAGE_KEY_ID
secret_access_key_env: OBJECT_STORAGE_KEY
campaigns:
root: /usr/local/share/narratio/campaigns
default_campaign_id: sample-campaign
spool:
root: /var/spool/narratio
delete_audio_after_archive: true
archive:
enabled: true
upload_run: true
promote_artifacts:
- source: narratio.transcript.final_trimmed
dest: transcripts/final.trimmed.json
required: true
- source: narratio.artifact.session_recap
dest: artifacts/session_recap.md
required: true
- source: narratio.artifact.player_handout
dest: artifacts/player_handout.md
required: false
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"
language: en
timeout: 45m
retries: 3
retry_delay: 3s
concurrency: 2
seriatim:
binary: seriatim
timeout: 10m
output_schema: seriatim-intermediate
coalesce_gap: 3.0
report: true
audita:
binary: audita
timeout: 3h
llm_api_key_env: AUDITA_LLM_API_KEY
modules: [glossary, homophones, spoken_word, grammar]
output_schema: audita-v1
work_dir_retention: auto
total_llm_concurrency: 2
proposal_llm_concurrency: 1
validation_llm_concurrency: 1
report: true
normalize:
output_path: transcripts/final.json
output_schema: seriatim-intermediate
report: true
trim:
enabled: false
scriptorium:
binary: scriptorium
config_path: /usr/local/etc/scriptorium/config.yml
timeout: 10m
render_debug: false
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
profile_id: local-fast
output_path: artifacts/session_recap.md
timeout: 10m
inputs:
transcript:
source: narratio.transcript.final_trimmed
required: true
previous_recap:
source: narratio.previous_session.artifact.session_recap
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: session_recap
player_handout:
enabled: true
depends_on:
- session_recap
prompt_id: dnd.player_handout
profile_id: local-fast
output_path: artifacts/player_handout.md
timeout: 10m
inputs:
recap:
source: narratio.artifact.session_recap
required: true
transcript:
source: narratio.transcript.final_trimmed
required: true
vars:
session_id: true
output_kind: player_handout
notification:
timeout: 30s

View File

@@ -0,0 +1,5 @@
session_id: 2026-05-03
date: 2026-05-03
title: Sample Session
inputs:
audio_dir: ./audio

View File

@@ -0,0 +1,6 @@
session_id: 2026-05-03
date: 2026-05-03
title: Sample Session
inputs:
audio_s3:
prefix: audio/

View File

@@ -0,0 +1,3 @@
session_id: "{{ session_id }}"
inputs:
audio_dir: ./audio

25
go.mod
View File

@@ -2,4 +2,27 @@ module gitea.maximumdirect.net/eric/narratio
go 1.25.0 go 1.25.0
require gopkg.in/yaml.v3 v3.0.1 require (
github.com/aws/aws-sdk-go-v2/config v1.32.17
github.com/aws/aws-sdk-go-v2/credentials v1.19.16
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0
github.com/aws/smithy-go v1.25.1
gopkg.in/yaml.v3 v3.0.1
)
require (
github.com/aws/aws-sdk-go-v2 v1.41.7 // indirect
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10 // indirect
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.23 // indirect
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23 // indirect
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23 // indirect
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.24 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.9 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.15 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.23 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.23 // indirect
github.com/aws/aws-sdk-go-v2/service/signin v1.0.11 // indirect
github.com/aws/aws-sdk-go-v2/service/sso v1.30.17 // indirect
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.35.21 // indirect
github.com/aws/aws-sdk-go-v2/service/sts v1.42.1 // indirect
)

36
go.sum
View File

@@ -1,3 +1,39 @@
github.com/aws/aws-sdk-go-v2 v1.41.7 h1:DWpAJt66FmnnaRIOT/8ASTucrvuDPZASqhhLey6tLY8=
github.com/aws/aws-sdk-go-v2 v1.41.7/go.mod h1:4LAfZOPHNVNQEckOACQx60Y8pSRjIkNZQz1w92xpMJc=
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10 h1:gx1AwW1Iyk9Z9dD9F4akX5gnN3QZwUB20GGKH/I+Rho=
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10/go.mod h1:qqY157uZoqm5OXq/amuaBJyC9hgBCBQnsaWnPe905GY=
github.com/aws/aws-sdk-go-v2/config v1.32.17 h1:FpL4/758/diKwqbytU0prpuiu60fgXKUWCpDJtApclU=
github.com/aws/aws-sdk-go-v2/config v1.32.17/go.mod h1:OXqUMzgXytfoF9JaKkhrOYsyh72t9G+MJH8mMRaexOE=
github.com/aws/aws-sdk-go-v2/credentials v1.19.16 h1:r3RJBuU7X9ibt8RHbMjWE6y60QbKBiII6wSrXnapxSU=
github.com/aws/aws-sdk-go-v2/credentials v1.19.16/go.mod h1:6cx7zqDENJDbBIIWX6P8s0h6hqHC8Avbjh9Dseo27ug=
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.23 h1:UuSfcORqNSz/ey3VPRS8TcVH2Ikf0/sC+Hdj400QI6U=
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.23/go.mod h1:+G/OSGiOFnSOkYloKj/9M35s74LgVAdJBSD5lsFfqKg=
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23 h1:GpT/TrnBYuE5gan2cZbTtvP+JlHsutdmlV2YfEyNde0=
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23/go.mod h1:xYWD6BS9ywC5bS3sz9Xh04whO/hzK2plt2Zkyrp4JuA=
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23 h1:bpd8vxhlQi2r1hiueOw02f/duEPTMK59Q4QMAoTTtTo=
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23/go.mod h1:15DfR2nw+CRHIk0tqNyifu3G1YdAOy68RftkhMDDwYk=
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.24 h1:OQqn11BtaYv1WLUowvcA30MpzIu8Ti4pcLPIIyoKZrA=
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.24/go.mod h1:X5ZJyfwVrWA96GzPmUCWFQaEARPR7gCrpq2E92PJwAE=
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.9 h1:FLudkZLt5ci0ozzgkVo8BJGwvqNaZbTWb3UcucAateA=
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.9/go.mod h1:w7wZ/s9qK7c8g4al+UyoF1Sp/Z45UwMGcqIzLWVQHWk=
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.15 h1:ieLCO1JxUWuxTZ1cRd0GAaeX7O6cIxnwk7tc1LsQhC4=
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.15/go.mod h1:e3IzZvQ3kAWNykvE0Tr0RDZCMFInMvhku3qNpcIQXhM=
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.23 h1:pbrxO/kuIwgEsOPLkaHu0O+m4fNgLU8B3vxQ+72jTPw=
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.23/go.mod h1:/CMNUqoj46HpS3MNRDEDIwcgEnrtZlKRaHNaHxIFpNA=
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.23 h1:03xatSQO4+AM1lTAbnRg5OK528EUg744nW7F73U8DKw=
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.23/go.mod h1:M8l3mwgx5ToK7wot2sBBce/ojzgnPzZXUV445gTSyE8=
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0 h1:etqBTKY581iwLL/H/S2sVgk3C9lAsTJFeXWFDsDcWOU=
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0/go.mod h1:L2dcoOgS2VSgbPLvpak2NyUPsO1TBN7M45Z4H7DlRc4=
github.com/aws/aws-sdk-go-v2/service/signin v1.0.11 h1:TdJ+HdzOBhU8+iVAOGUTU63VXopcumCOF1paFulHWZc=
github.com/aws/aws-sdk-go-v2/service/signin v1.0.11/go.mod h1:R82ZRExE/nheo0N+T8zHPcLRTcH8MGsnR3BiVGX0TwI=
github.com/aws/aws-sdk-go-v2/service/sso v1.30.17 h1:7byT8HUWrgoRp6sXjxtZwgOKfhss5fW6SkLBtqzgRoE=
github.com/aws/aws-sdk-go-v2/service/sso v1.30.17/go.mod h1:xNWknVi4Ezm1vg1QsB/5EWpAJURq22uqd38U8qKvOJc=
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.35.21 h1:+1Kl1zx6bWi4X7cKi3VYh29h8BvsCoHQEQ6ST9X8w7w=
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.35.21/go.mod h1:4vIRDq+CJB2xFAXZ+YgGUTiEft7oAQlhIs71xcSeuVg=
github.com/aws/aws-sdk-go-v2/service/sts v1.42.1 h1:F/M5Y9I3nwr2IEpshZgh1GeHpOItExNM9L1euNuh/fk=
github.com/aws/aws-sdk-go-v2/service/sts v1.42.1/go.mod h1:mTNxImtovCOEEuD65mKW7DCsL+2gjEH+RPEAexAzAio=
github.com/aws/smithy-go v1.25.1 h1:J8ERsGSU7d+aCmdQur5Txg6bVoYelvQJgtZehD12GkI=
github.com/aws/smithy-go v1.25.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM= gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=

View File

@@ -1,40 +0,0 @@
package analyzer
import "context"
// NoopRunner is a deterministic no-op analyzer adapter.
type NoopRunner struct{}
// Run returns the requested output path with placeholder metadata.
func (n *NoopRunner) Run(ctx context.Context, req AnalyzeRequest) (AnalyzeResult, error) {
if err := ctx.Err(); err != nil {
return AnalyzeResult{}, err
}
return AnalyzeResult{ArtifactPath: req.OutputPath, Metadata: map[string]any{"placeholder": true}}, nil
}
// FakeRunner captures analyze requests and returns deterministic responses.
type FakeRunner struct {
Requests []AnalyzeRequest
Err error
Result AnalyzeResult
}
// Run records request and returns configured response.
func (f *FakeRunner) Run(ctx context.Context, req AnalyzeRequest) (AnalyzeResult, error) {
if err := ctx.Err(); err != nil {
return AnalyzeResult{}, err
}
f.Requests = append(f.Requests, req)
if f.Err != nil {
return AnalyzeResult{}, f.Err
}
res := f.Result
if res.ArtifactPath == "" {
res.ArtifactPath = req.OutputPath
}
if res.Metadata == nil {
res.Metadata = map[string]any{"fake": true}
}
return res, nil
}

View File

@@ -1,31 +0,0 @@
package analyzer
import (
"context"
"errors"
"testing"
)
func TestFakeRunnerCapturesRequestAndReturnsPath(t *testing.T) {
fake := &FakeRunner{}
req := AnalyzeRequest{ArtifactType: "session-log", OutputPath: "artifacts/session-log.md"}
res, err := fake.Run(context.Background(), req)
if err != nil {
t.Fatalf("Run() error = %v", err)
}
if len(fake.Requests) != 1 || fake.Requests[0].ArtifactType != "session-log" {
t.Fatalf("requests = %#v, want captured request", fake.Requests)
}
if res.ArtifactPath != req.OutputPath {
t.Fatalf("artifact path = %q, want %q", res.ArtifactPath, req.OutputPath)
}
}
func TestFakeRunnerError(t *testing.T) {
fake := &FakeRunner{Err: errors.New("boom")}
_, err := fake.Run(context.Background(), AnalyzeRequest{})
if err == nil {
t.Fatal("expected error, got nil")
}
}

View File

@@ -1,28 +0,0 @@
// Package analyzer declares the adapter contract for artifact analysis generation.
package analyzer
import "context"
// TODO: implement analyzer integration once the analyzer contract is finalized.
// Runner is the adapter boundary for analyzer invocations.
type Runner interface {
Run(ctx context.Context, req AnalyzeRequest) (AnalyzeResult, error)
}
// AnalyzeRequest describes one analyzer artifact generation request.
type AnalyzeRequest struct {
ArtifactType string
ProcessedTranscriptPath string
ContextReferences []string
OutputPath string
GeneratedConfigPath string
StdoutLogPath string
StderrLogPath string
}
// AnalyzeResult describes analyzer output.
type AnalyzeResult struct {
ArtifactPath string
Metadata map[string]any
}

View File

@@ -14,7 +14,7 @@ func TestFakeRunnerCapturesRequestAndReturnsPath(t *testing.T) {
dir := t.TempDir() dir := t.TempDir()
req := PolishRequest{ req := PolishRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "audita.yml"), GeneratedConfigPath: filepath.Join(dir, "config", "audita.yml"),
OutputProcessedPath: filepath.Join(dir, "transcripts", "processed.json"), OutputProcessedPath: filepath.Join(dir, "transcripts", "polished.json"),
StdoutLogPath: filepath.Join(dir, "logs", "audita.stdout.log"), StdoutLogPath: filepath.Join(dir, "logs", "audita.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "audita.stderr.log"), StderrLogPath: filepath.Join(dir, "logs", "audita.stderr.log"),
} }

View File

@@ -24,6 +24,12 @@ type PolishRequest struct {
Modules []string Modules []string
BaseURL string BaseURL string
Model string Model string
TranscriptDescription string
ConfigPath string
OutputSchema string
WorkDirRetention string
TotalLLMConcurrency *int
ProposalLLMConcurrency *int
ValidationModel string ValidationModel string
ValidationLLMConcurrency *int ValidationLLMConcurrency *int
StdoutLogPath string StdoutLogPath string

View File

@@ -21,7 +21,12 @@ type SubprocessRunnerConfig struct {
Modules []string Modules []string
BaseURL string BaseURL string
Model string Model string
LLMConcurrency *int TranscriptDescription string
ConfigPath string
OutputSchema string
WorkDirRetention string
TotalLLMConcurrency *int
ProposalLLMConcurrency *int
ValidationModel string ValidationModel string
ValidationLLMConcurrency *int ValidationLLMConcurrency *int
Report bool Report bool
@@ -35,7 +40,12 @@ type SubprocessRunner struct {
modules []string modules []string
baseURL string baseURL string
model string model string
llmConcurrency *int transcriptDescription string
configPath string
outputSchema string
workDirRetention string
totalLLMConcurrency *int
proposalLLMConcurrency *int
validationModel string validationModel string
validationLLMConcurrency *int validationLLMConcurrency *int
report bool report bool
@@ -49,7 +59,12 @@ func NewSubprocessRunnerFromConfigValues(
modules []string, modules []string,
baseURL string, baseURL string,
model string, model string,
llmConcurrency *int, transcriptDescription string,
configPath string,
outputSchema string,
workDirRetention string,
totalLLMConcurrency *int,
proposalLLMConcurrency *int,
validationModel string, validationModel string,
validationLLMConcurrency *int, validationLLMConcurrency *int,
report bool, report bool,
@@ -68,7 +83,12 @@ func NewSubprocessRunnerFromConfigValues(
Modules: modules, Modules: modules,
BaseURL: baseURL, BaseURL: baseURL,
Model: model, Model: model,
LLMConcurrency: llmConcurrency, TranscriptDescription: transcriptDescription,
ConfigPath: configPath,
OutputSchema: outputSchema,
WorkDirRetention: workDirRetention,
TotalLLMConcurrency: totalLLMConcurrency,
ProposalLLMConcurrency: proposalLLMConcurrency,
ValidationModel: validationModel, ValidationModel: validationModel,
ValidationLLMConcurrency: validationLLMConcurrency, ValidationLLMConcurrency: validationLLMConcurrency,
Report: report, Report: report,
@@ -83,17 +103,12 @@ func NewSubprocessRunner(cfg SubprocessRunnerConfig) (*SubprocessRunner, error)
if cfg.Timeout <= 0 { if cfg.Timeout <= 0 {
return nil, fmt.Errorf("audita timeout must be > 0") return nil, fmt.Errorf("audita timeout must be > 0")
} }
if len(cfg.Modules) == 0 {
return nil, fmt.Errorf("audita modules must include at least one module")
}
for i, module := range cfg.Modules { for i, module := range cfg.Modules {
if strings.TrimSpace(module) == "" { if strings.TrimSpace(module) == "" {
return nil, fmt.Errorf("audita module at index %d is empty", i) return nil, fmt.Errorf("audita module at index %d is empty", i)
} }
} }
if strings.TrimSpace(cfg.BaseURL) == "" { if strings.TrimSpace(cfg.BaseURL) != "" {
return nil, fmt.Errorf("audita base url is required")
}
u, err := url.Parse(cfg.BaseURL) u, err := url.Parse(cfg.BaseURL)
if err != nil || u.Scheme == "" || u.Host == "" { if err != nil || u.Scheme == "" || u.Host == "" {
if err != nil { if err != nil {
@@ -101,15 +116,26 @@ func NewSubprocessRunner(cfg SubprocessRunnerConfig) (*SubprocessRunner, error)
} }
return nil, fmt.Errorf("audita base url %q is invalid", cfg.BaseURL) return nil, fmt.Errorf("audita base url %q is invalid", cfg.BaseURL)
} }
if strings.TrimSpace(cfg.Model) == "" {
return nil, fmt.Errorf("audita model is required")
} }
if cfg.LLMConcurrency != nil && *cfg.LLMConcurrency <= 0 { if cfg.TotalLLMConcurrency != nil && *cfg.TotalLLMConcurrency <= 0 {
return nil, fmt.Errorf("audita llm concurrency must be > 0 when provided") return nil, fmt.Errorf("audita total llm concurrency must be > 0 when provided")
}
if cfg.ProposalLLMConcurrency != nil && *cfg.ProposalLLMConcurrency <= 0 {
return nil, fmt.Errorf("audita proposal llm concurrency must be > 0 when provided")
} }
if cfg.ValidationLLMConcurrency != nil && *cfg.ValidationLLMConcurrency <= 0 { if cfg.ValidationLLMConcurrency != nil && *cfg.ValidationLLMConcurrency <= 0 {
return nil, fmt.Errorf("audita validation llm concurrency must be > 0 when provided") return nil, fmt.Errorf("audita validation llm concurrency must be > 0 when provided")
} }
switch strings.TrimSpace(cfg.OutputSchema) {
case "", "bare-segments", "audita-v1":
default:
return nil, fmt.Errorf("audita output schema must be one of: bare-segments, audita-v1")
}
switch strings.TrimSpace(cfg.WorkDirRetention) {
case "", "always", "auto", "never":
default:
return nil, fmt.Errorf("audita work dir retention must be one of: always, auto, never")
}
modules := make([]string, len(cfg.Modules)) modules := make([]string, len(cfg.Modules))
for i, m := range cfg.Modules { for i, m := range cfg.Modules {
@@ -123,7 +149,12 @@ func NewSubprocessRunner(cfg SubprocessRunnerConfig) (*SubprocessRunner, error)
modules: modules, modules: modules,
baseURL: strings.TrimSpace(cfg.BaseURL), baseURL: strings.TrimSpace(cfg.BaseURL),
model: strings.TrimSpace(cfg.Model), model: strings.TrimSpace(cfg.Model),
llmConcurrency: cfg.LLMConcurrency, transcriptDescription: strings.TrimSpace(cfg.TranscriptDescription),
configPath: strings.TrimSpace(cfg.ConfigPath),
outputSchema: strings.TrimSpace(cfg.OutputSchema),
workDirRetention: strings.TrimSpace(cfg.WorkDirRetention),
totalLLMConcurrency: cfg.TotalLLMConcurrency,
proposalLLMConcurrency: cfg.ProposalLLMConcurrency,
validationModel: strings.TrimSpace(cfg.ValidationModel), validationModel: strings.TrimSpace(cfg.ValidationModel),
validationLLMConcurrency: cfg.ValidationLLMConcurrency, validationLLMConcurrency: cfg.ValidationLLMConcurrency,
report: cfg.Report, report: cfg.Report,
@@ -152,7 +183,7 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
} }
reqModules := req.Modules reqModules := req.Modules
if len(reqModules) == 0 { if reqModules == nil {
reqModules = append([]string(nil), r.modules...) reqModules = append([]string(nil), r.modules...)
} }
args := r.buildArgs(req, reqModules) args := r.buildArgs(req, reqModules)
@@ -168,14 +199,9 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
env["AUDITA_LLM_API_KEY"] = credential env["AUDITA_LLM_API_KEY"] = credential
credentialPresent = true credentialPresent = true
} }
primaryConcurrencyViaEnv := false
if r.llmConcurrency != nil {
env["AUDITA_LLM_CONCURRENCY"] = strconv.Itoa(*r.llmConcurrency)
primaryConcurrencyViaEnv = true
}
if req.GeneratedConfigPath != "" { if req.GeneratedConfigPath != "" {
if err := r.writeInvocationConfig(req, args, reqModules, credentialPresent, primaryConcurrencyViaEnv); err != nil { if err := r.writeInvocationConfig(req, args, reqModules, credentialPresent); err != nil {
return PolishResult{}, fmt.Errorf("write audita invocation config %q: %w", req.GeneratedConfigPath, err) return PolishResult{}, fmt.Errorf("write audita invocation config %q: %w", req.GeneratedConfigPath, err)
} }
} }
@@ -196,7 +222,7 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
req.StderrLogPath, req.StderrLogPath,
) )
wrappedMessage = addSubprocessStreamHint(wrappedMessage, err) wrappedMessage = addSubprocessStreamHint(wrappedMessage, err)
return r.failureResult(req, reqModules, runRes, credentialPresent, primaryConcurrencyViaEnv), fmt.Errorf( return r.failureResult(req, reqModules, runRes, credentialPresent), fmt.Errorf(
"%s: %w", "%s: %w",
wrappedMessage, wrappedMessage,
err, err,
@@ -204,11 +230,11 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
} }
if err := validateProcessedOutput(req.OutputProcessedPath); err != nil { if err := validateProcessedOutput(req.OutputProcessedPath); err != nil {
return r.failureResult(req, reqModules, runRes, credentialPresent, primaryConcurrencyViaEnv), fmt.Errorf("validate audita processed output %q: %w", req.OutputProcessedPath, err) return r.failureResult(req, reqModules, runRes, credentialPresent), fmt.Errorf("validate audita processed output %q: %w", req.OutputProcessedPath, err)
} }
if r.report { if r.report {
if err := validateJSONFile(req.ReportPath); err != nil { if err := validateJSONFile(req.ReportPath); err != nil {
return r.failureResult(req, reqModules, runRes, credentialPresent, primaryConcurrencyViaEnv), fmt.Errorf("validate audita report output %q: %w", req.ReportPath, err) return r.failureResult(req, reqModules, runRes, credentialPresent), fmt.Errorf("validate audita report output %q: %w", req.ReportPath, err)
} }
} }
@@ -227,17 +253,21 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
"modules": reqModules, "modules": reqModules,
"base_url": r.baseURL, "base_url": r.baseURL,
"model": r.model, "model": r.model,
"transcript_description": r.transcriptDescription,
"config_path": r.configPath,
"output_schema": r.outputSchema,
"work_dir_retention": r.workDirRetention,
"validation_model": r.validationModel, "validation_model": r.validationModel,
"total_llm_concurrency": r.totalLLMConcurrency,
"proposal_llm_concurrency": r.proposalLLMConcurrency,
"validation_llm_concurrency": r.validationLLMConcurrency, "validation_llm_concurrency": r.validationLLMConcurrency,
"credential_env_var": r.llmAPIKeyEnv, "credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent, "credential_present": credentialPresent,
"primary_llm_concurrency_via_env": primaryConcurrencyViaEnv,
"primary_llm_concurrency_env_name": "AUDITA_LLM_CONCURRENCY",
}, },
}, nil }, nil
} }
func (r *SubprocessRunner) failureResult(req PolishRequest, modules []string, runRes subprocess.RunResult, credentialPresent bool, primaryConcurrencyViaEnv bool) PolishResult { func (r *SubprocessRunner) failureResult(req PolishRequest, modules []string, runRes subprocess.RunResult, credentialPresent bool) PolishResult {
return PolishResult{ return PolishResult{
ProcessedTranscriptPath: req.OutputProcessedPath, ProcessedTranscriptPath: req.OutputProcessedPath,
ReportPath: req.ReportPath, ReportPath: req.ReportPath,
@@ -253,12 +283,16 @@ func (r *SubprocessRunner) failureResult(req PolishRequest, modules []string, ru
"modules": modules, "modules": modules,
"base_url": r.baseURL, "base_url": r.baseURL,
"model": r.model, "model": r.model,
"transcript_description": r.transcriptDescription,
"config_path": r.configPath,
"output_schema": r.outputSchema,
"work_dir_retention": r.workDirRetention,
"validation_model": r.validationModel, "validation_model": r.validationModel,
"total_llm_concurrency": r.totalLLMConcurrency,
"proposal_llm_concurrency": r.proposalLLMConcurrency,
"validation_llm_concurrency": r.validationLLMConcurrency, "validation_llm_concurrency": r.validationLLMConcurrency,
"credential_env_var": r.llmAPIKeyEnv, "credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent, "credential_present": credentialPresent,
"primary_llm_concurrency_via_env": primaryConcurrencyViaEnv,
"primary_llm_concurrency_env_name": "AUDITA_LLM_CONCURRENCY",
}, },
} }
} }
@@ -269,14 +303,38 @@ func (r *SubprocessRunner) buildArgs(req PolishRequest, modules []string) []stri
req.MergedTranscriptPath, req.MergedTranscriptPath,
"--glossary", req.GlossaryPath, "--glossary", req.GlossaryPath,
"--output", req.OutputProcessedPath, "--output", req.OutputProcessedPath,
"--modules", strings.Join(modules, ","),
"--base-url", r.baseURL,
"--model", r.model,
"--work-dir", req.WorkDir, "--work-dir", req.WorkDir,
} }
if r.baseURL != "" {
args = append(args, "--base-url", r.baseURL)
}
if r.model != "" {
args = append(args, "--model", r.model)
}
if len(modules) > 0 {
args = append(args, "--modules", strings.Join(modules, ","))
}
if r.report { if r.report {
args = append(args, "--report-json", req.ReportPath) args = append(args, "--report-json", req.ReportPath)
} }
if r.transcriptDescription != "" {
args = append(args, "--transcript-description", r.transcriptDescription)
}
if r.configPath != "" {
args = append(args, "--config", r.configPath)
}
if r.outputSchema != "" {
args = append(args, "--output-schema", r.outputSchema)
}
if r.workDirRetention != "" {
args = append(args, "--work-dir-retention", r.workDirRetention)
}
if r.totalLLMConcurrency != nil {
args = append(args, "--total-llm-concurrency", strconv.Itoa(*r.totalLLMConcurrency))
}
if r.proposalLLMConcurrency != nil {
args = append(args, "--proposal-llm-concurrency", strconv.Itoa(*r.proposalLLMConcurrency))
}
if r.validationModel != "" { if r.validationModel != "" {
args = append(args, "--validation-model", r.validationModel) args = append(args, "--validation-model", r.validationModel)
} }
@@ -286,7 +344,7 @@ func (r *SubprocessRunner) buildArgs(req PolishRequest, modules []string) []stri
return args return args
} }
func (r *SubprocessRunner) writeInvocationConfig(req PolishRequest, args []string, modules []string, credentialPresent bool, primaryConcurrencyViaEnv bool) error { func (r *SubprocessRunner) writeInvocationConfig(req PolishRequest, args []string, modules []string, credentialPresent bool) error {
payload := map[string]any{ payload := map[string]any{
"schema": "audita.generated.v1", "schema": "audita.generated.v1",
"binary": r.binary, "binary": r.binary,
@@ -295,7 +353,13 @@ func (r *SubprocessRunner) writeInvocationConfig(req PolishRequest, args []strin
"modules": modules, "modules": modules,
"base_url": r.baseURL, "base_url": r.baseURL,
"model": r.model, "model": r.model,
"transcript_description": r.transcriptDescription,
"config_path": r.configPath,
"output_schema": r.outputSchema,
"work_dir_retention": r.workDirRetention,
"validation_model": r.validationModel, "validation_model": r.validationModel,
"total_llm_concurrency": r.totalLLMConcurrency,
"proposal_llm_concurrency": r.proposalLLMConcurrency,
"validation_llm_concurrency": r.validationLLMConcurrency, "validation_llm_concurrency": r.validationLLMConcurrency,
"report_enabled": r.report, "report_enabled": r.report,
"merged_transcript_path": req.MergedTranscriptPath, "merged_transcript_path": req.MergedTranscriptPath,
@@ -305,10 +369,6 @@ func (r *SubprocessRunner) writeInvocationConfig(req PolishRequest, args []strin
"work_dir": req.WorkDir, "work_dir": req.WorkDir,
"credential_env_var": r.llmAPIKeyEnv, "credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent, "credential_present": credentialPresent,
"primary_llm_concurrency_via_env": primaryConcurrencyViaEnv,
}
if r.llmConcurrency != nil {
payload["llm_concurrency"] = *r.llmConcurrency
} }
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644) return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
} }

View File

@@ -25,7 +25,8 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath) t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath)
wrapper := writeAuditaHelperWrapper(t) wrapper := writeAuditaHelperWrapper(t)
llmConcurrency := 1 totalLLMConcurrency := 3
proposalLLMConcurrency := 2
validationLLMConcurrency := 2 validationLLMConcurrency := 2
runner, err := NewSubprocessRunner(SubprocessRunnerConfig{ runner, err := NewSubprocessRunner(SubprocessRunnerConfig{
Binary: wrapper, Binary: wrapper,
@@ -34,7 +35,12 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
Modules: []string{"glossary", "homophones", "glossary"}, Modules: []string{"glossary", "homophones", "glossary"},
BaseURL: "https://openrouter.ai/api/v1", BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it", Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency, TranscriptDescription: "Campaign Session 42",
ConfigPath: "/etc/audita/config.yml",
OutputSchema: "audita-v1",
WorkDirRetention: "auto",
TotalLLMConcurrency: &totalLLMConcurrency,
ProposalLLMConcurrency: &proposalLLMConcurrency,
ValidationModel: "openrouter/google/gemma-4-31b-it", ValidationModel: "openrouter/google/gemma-4-31b-it",
ValidationLLMConcurrency: &validationLLMConcurrency, ValidationLLMConcurrency: &validationLLMConcurrency,
Report: true, Report: true,
@@ -46,9 +52,9 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
dir := t.TempDir() dir := t.TempDir()
req := PolishRequest{ req := PolishRequest{
GeneratedConfigPath: filepath.Join(dir, "audita.generated.yml"), GeneratedConfigPath: filepath.Join(dir, "audita.generated.yml"),
MergedTranscriptPath: filepath.Join(dir, "merged.json"), MergedTranscriptPath: filepath.Join(dir, "base.json"),
GlossaryPath: filepath.Join(dir, "glossary.yml"), GlossaryPath: filepath.Join(dir, "glossary.yml"),
OutputProcessedPath: filepath.Join(dir, "processed.json"), OutputProcessedPath: filepath.Join(dir, "polished.json"),
ReportPath: filepath.Join(dir, "audita.report.json"), ReportPath: filepath.Join(dir, "audita.report.json"),
WorkDir: filepath.Join(dir, "artifacts", "audita-work"), WorkDir: filepath.Join(dir, "artifacts", "audita-work"),
StdoutLogPath: filepath.Join(dir, "audita.stdout.log"), StdoutLogPath: filepath.Join(dir, "audita.stdout.log"),
@@ -93,11 +99,17 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
"process", req.MergedTranscriptPath, "process", req.MergedTranscriptPath,
"--glossary", req.GlossaryPath, "--glossary", req.GlossaryPath,
"--output", req.OutputProcessedPath, "--output", req.OutputProcessedPath,
"--modules", "glossary,homophones,glossary", "--work-dir", req.WorkDir,
"--base-url", "https://openrouter.ai/api/v1", "--base-url", "https://openrouter.ai/api/v1",
"--model", "openrouter/google/gemma-4-31b-it", "--model", "openrouter/google/gemma-4-31b-it",
"--work-dir", req.WorkDir, "--modules", "glossary,homophones,glossary",
"--report-json", req.ReportPath, "--report-json", req.ReportPath,
"--transcript-description", "Campaign Session 42",
"--config", "/etc/audita/config.yml",
"--output-schema", "audita-v1",
"--work-dir-retention", "auto",
"--total-llm-concurrency", "3",
"--proposal-llm-concurrency", "2",
"--validation-model", "openrouter/google/gemma-4-31b-it", "--validation-model", "openrouter/google/gemma-4-31b-it",
"--validation-llm-concurrency", "2", "--validation-llm-concurrency", "2",
} }
@@ -107,8 +119,8 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
if rec.Env["AUDITA_LLM_API_KEY"] != "super-secret" { if rec.Env["AUDITA_LLM_API_KEY"] != "super-secret" {
t.Fatalf("AUDITA_LLM_API_KEY = %q, want propagated secret", rec.Env["AUDITA_LLM_API_KEY"]) t.Fatalf("AUDITA_LLM_API_KEY = %q, want propagated secret", rec.Env["AUDITA_LLM_API_KEY"])
} }
if rec.Env["AUDITA_LLM_CONCURRENCY"] != "1" { if rec.Env["AUDITA_LLM_CONCURRENCY"] != "" {
t.Fatalf("AUDITA_LLM_CONCURRENCY = %q, want 1", rec.Env["AUDITA_LLM_CONCURRENCY"]) t.Fatalf("AUDITA_LLM_CONCURRENCY = %q, want empty/omitted", rec.Env["AUDITA_LLM_CONCURRENCY"])
} }
cfgData, err := os.ReadFile(req.GeneratedConfigPath) cfgData, err := os.ReadFile(req.GeneratedConfigPath)
@@ -124,7 +136,6 @@ func TestSubprocessRunnerMissingConfiguredCredentialFails(t *testing.T) {
if runtime.GOOS == "windows" { if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh") t.Skip("helper wrapper script uses /bin/sh")
} }
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{ runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t), Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"), Timeout: mustParseAuditaDuration(t, "2s"),
@@ -132,7 +143,6 @@ func TestSubprocessRunnerMissingConfiguredCredentialFails(t *testing.T) {
Modules: []string{"glossary"}, Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1", BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it", Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false, Report: false,
}) })
req := auditaReqForTest(t, false) req := auditaReqForTest(t, false)
@@ -155,7 +165,6 @@ func TestSubprocessRunnerUnconfiguredCredentialEnvOmitsCredential(t *testing.T)
recordPath := filepath.Join(t.TempDir(), "record.json") recordPath := filepath.Join(t.TempDir(), "record.json")
t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath) t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath)
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{ runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t), Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"), Timeout: mustParseAuditaDuration(t, "2s"),
@@ -163,7 +172,6 @@ func TestSubprocessRunnerUnconfiguredCredentialEnvOmitsCredential(t *testing.T)
Modules: []string{"glossary"}, Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1", BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it", Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false, Report: false,
}) })
req := auditaReqForTest(t, false) req := auditaReqForTest(t, false)
@@ -211,6 +219,65 @@ func TestSubprocessRunnerInheritsParentEnvironment(t *testing.T) {
} }
} }
func TestSubprocessRunnerOmitsModulesFlagWhenNotConfigured(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_AUDITA_HELPER", "1")
t.Setenv("AUDITA_HELPER_MODE", "success")
recordPath := filepath.Join(t.TempDir(), "record.json")
t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath)
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "",
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: false,
})
req := auditaReqForTest(t, false)
if _, err := runner.Run(context.Background(), req); err != nil {
t.Fatalf("Run() error = %v", err)
}
rec := readAuditaHelperRecord(t, recordPath)
for i := 0; i < len(rec.Args); i++ {
if rec.Args[i] == "--modules" {
t.Fatalf("args contained --modules unexpectedly: %#v", rec.Args)
}
}
}
func TestSubprocessRunnerOmitsBaseURLAndModelFlagsWhenNotConfigured(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_AUDITA_HELPER", "1")
t.Setenv("AUDITA_HELPER_MODE", "success")
recordPath := filepath.Join(t.TempDir(), "record.json")
t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath)
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "",
Report: false,
})
req := auditaReqForTest(t, false)
if _, err := runner.Run(context.Background(), req); err != nil {
t.Fatalf("Run() error = %v", err)
}
rec := readAuditaHelperRecord(t, recordPath)
for i := 0; i < len(rec.Args); i++ {
if rec.Args[i] == "--base-url" {
t.Fatalf("args contained --base-url unexpectedly: %#v", rec.Args)
}
if rec.Args[i] == "--model" {
t.Fatalf("args contained --model unexpectedly: %#v", rec.Args)
}
}
}
func TestSubprocessRunnerSubprocessFailure(t *testing.T) { func TestSubprocessRunnerSubprocessFailure(t *testing.T) {
if runtime.GOOS == "windows" { if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh") t.Skip("helper wrapper script uses /bin/sh")
@@ -220,7 +287,6 @@ func TestSubprocessRunnerSubprocessFailure(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret") t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json")) t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{ runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t), Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"), Timeout: mustParseAuditaDuration(t, "2s"),
@@ -228,7 +294,6 @@ func TestSubprocessRunnerSubprocessFailure(t *testing.T) {
Modules: []string{"glossary"}, Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1", BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it", Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: true, Report: true,
}) })
req := auditaReqForTest(t, true) req := auditaReqForTest(t, true)
@@ -256,7 +321,6 @@ func TestSubprocessRunnerSubprocessFailureAddsStderrDescriptorHint(t *testing.T)
t.Setenv("OPENAI_KEY_SOURCE", "super-secret") t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json")) t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{ runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t), Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"), Timeout: mustParseAuditaDuration(t, "2s"),
@@ -264,7 +328,6 @@ func TestSubprocessRunnerSubprocessFailureAddsStderrDescriptorHint(t *testing.T)
Modules: []string{"glossary"}, Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1", BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it", Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: true, Report: true,
}) })
req := auditaReqForTest(t, true) req := auditaReqForTest(t, true)
@@ -286,7 +349,6 @@ func TestSubprocessRunnerMissingOutputFails(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret") t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json")) t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{ runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t), Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"), Timeout: mustParseAuditaDuration(t, "2s"),
@@ -294,7 +356,6 @@ func TestSubprocessRunnerMissingOutputFails(t *testing.T) {
Modules: []string{"glossary"}, Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1", BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it", Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false, Report: false,
}) })
req := auditaReqForTest(t, false) req := auditaReqForTest(t, false)
@@ -316,7 +377,6 @@ func TestSubprocessRunnerInvalidOutputJSONFails(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret") t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json")) t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{ runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t), Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"), Timeout: mustParseAuditaDuration(t, "2s"),
@@ -324,7 +384,6 @@ func TestSubprocessRunnerInvalidOutputJSONFails(t *testing.T) {
Modules: []string{"glossary"}, Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1", BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it", Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false, Report: false,
}) })
req := auditaReqForTest(t, false) req := auditaReqForTest(t, false)
@@ -346,7 +405,6 @@ func TestSubprocessRunnerSegmentsMissingFails(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret") t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json")) t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{ runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t), Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"), Timeout: mustParseAuditaDuration(t, "2s"),
@@ -354,7 +412,6 @@ func TestSubprocessRunnerSegmentsMissingFails(t *testing.T) {
Modules: []string{"glossary"}, Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1", BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it", Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false, Report: false,
}) })
req := auditaReqForTest(t, false) req := auditaReqForTest(t, false)
@@ -376,7 +433,6 @@ func TestSubprocessRunnerInvalidReportJSONFails(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret") t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json")) t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{ runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t), Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"), Timeout: mustParseAuditaDuration(t, "2s"),
@@ -384,7 +440,6 @@ func TestSubprocessRunnerInvalidReportJSONFails(t *testing.T) {
Modules: []string{"glossary"}, Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1", BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it", Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: true, Report: true,
}) })
req := auditaReqForTest(t, true) req := auditaReqForTest(t, true)
@@ -398,11 +453,11 @@ func TestSubprocessRunnerInvalidReportJSONFails(t *testing.T) {
} }
func TestSubprocessRunnerConstructorValidation(t *testing.T) { func TestSubprocessRunnerConstructorValidation(t *testing.T) {
_, err := NewSubprocessRunnerFromConfigValues("", "3h", "AUDITA_LLM_API_KEY", []string{"glossary"}, "https://openrouter.ai/api/v1", "openrouter/google/gemma-4-31b-it", nil, "", nil, true) _, err := NewSubprocessRunnerFromConfigValues("", "3h", "AUDITA_LLM_API_KEY", []string{"glossary"}, "https://openrouter.ai/api/v1", "openrouter/google/gemma-4-31b-it", "", "", "", "", nil, nil, "", nil, true)
if err == nil { if err == nil {
t.Fatal("expected binary validation error") t.Fatal("expected binary validation error")
} }
_, err = NewSubprocessRunnerFromConfigValues("audita", "bad", "AUDITA_LLM_API_KEY", []string{"glossary"}, "https://openrouter.ai/api/v1", "openrouter/google/gemma-4-31b-it", nil, "", nil, true) _, err = NewSubprocessRunnerFromConfigValues("audita", "bad", "AUDITA_LLM_API_KEY", []string{"glossary"}, "https://openrouter.ai/api/v1", "openrouter/google/gemma-4-31b-it", "", "", "", "", nil, nil, "", nil, true)
if err == nil { if err == nil {
t.Fatal("expected timeout parse error") t.Fatal("expected timeout parse error")
} }
@@ -516,7 +571,7 @@ func mustAuditaRunner(t *testing.T, cfg SubprocessRunnerConfig) *SubprocessRunne
func auditaReqForTest(t *testing.T, withReport bool) PolishRequest { func auditaReqForTest(t *testing.T, withReport bool) PolishRequest {
t.Helper() t.Helper()
dir := t.TempDir() dir := t.TempDir()
merged := filepath.Join(dir, "merged.json") merged := filepath.Join(dir, "base.json")
glossary := filepath.Join(dir, "glossary.yml") glossary := filepath.Join(dir, "glossary.yml")
writeAuditaTestFile(t, merged, `{"segments":[]}`) writeAuditaTestFile(t, merged, `{"segments":[]}`)
writeAuditaTestFile(t, glossary, "terms: []\n") writeAuditaTestFile(t, glossary, "terms: []\n")
@@ -524,7 +579,7 @@ func auditaReqForTest(t *testing.T, withReport bool) PolishRequest {
GeneratedConfigPath: filepath.Join(dir, "audita.generated.yml"), GeneratedConfigPath: filepath.Join(dir, "audita.generated.yml"),
MergedTranscriptPath: merged, MergedTranscriptPath: merged,
GlossaryPath: glossary, GlossaryPath: glossary,
OutputProcessedPath: filepath.Join(dir, "processed.json"), OutputProcessedPath: filepath.Join(dir, "polished.json"),
WorkDir: filepath.Join(dir, "artifacts", "audita-work"), WorkDir: filepath.Join(dir, "artifacts", "audita-work"),
StdoutLogPath: filepath.Join(dir, "audita.stdout.log"), StdoutLogPath: filepath.Join(dir, "audita.stdout.log"),
StderrLogPath: filepath.Join(dir, "audita.stderr.log"), StderrLogPath: filepath.Join(dir, "audita.stderr.log"),

View File

@@ -31,7 +31,7 @@ func TestSubprocessRunnerRunSuccessBuildsDeterministicArgsAndCapturesLogs(t *tes
ConfigPath: "/etc/scriptorium/config.yml", ConfigPath: "/etc/scriptorium/config.yml",
PromptID: "dnd.session_recap", PromptID: "dnd.session_recap",
ProfileID: "local-quality", ProfileID: "local-quality",
InputPaths: map[string]string{"transcript": filepath.Join(dir, "processed.json"), "other": filepath.Join(dir, "other.md")}, InputPaths: map[string]string{"transcript": filepath.Join(dir, "polished.json"), "other": filepath.Join(dir, "other.md")},
Vars: map[string]string{"session_id": "2026-05-03", "campaign_name": "Icewind Dale"}, Vars: map[string]string{"session_id": "2026-05-03", "campaign_name": "Icewind Dale"},
OutputPath: filepath.Join(dir, "artifacts", "session_recap.md"), OutputPath: filepath.Join(dir, "artifacts", "session_recap.md"),
StdoutLogPath: filepath.Join(dir, "logs", "scriptorium.run.stdout.log"), StdoutLogPath: filepath.Join(dir, "logs", "scriptorium.run.stdout.log"),
@@ -180,7 +180,7 @@ func TestSubprocessRunnerRenderSuccess(t *testing.T) {
req := RenderArtifactRequest{ req := RenderArtifactRequest{
Binary: wrapper, Binary: wrapper,
PromptID: "dnd.session_recap", PromptID: "dnd.session_recap",
InputPaths: map[string]string{"transcript": filepath.Join(dir, "processed.json")}, InputPaths: map[string]string{"transcript": filepath.Join(dir, "polished.json")},
OutputPath: filepath.Join(dir, "artifacts", "session_recap.render.json"), OutputPath: filepath.Join(dir, "artifacts", "session_recap.render.json"),
StdoutLogPath: filepath.Join(dir, "logs", "scriptorium.render.stdout.log"), StdoutLogPath: filepath.Join(dir, "logs", "scriptorium.render.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "scriptorium.render.stderr.log"), StderrLogPath: filepath.Join(dir, "logs", "scriptorium.render.stderr.log"),
@@ -285,7 +285,7 @@ type scriptoriumHelperRecord struct {
func runReqForTest(t *testing.T, binary string) RunArtifactRequest { func runReqForTest(t *testing.T, binary string) RunArtifactRequest {
t.Helper() t.Helper()
dir := t.TempDir() dir := t.TempDir()
transcriptPath := filepath.Join(dir, "processed.json") transcriptPath := filepath.Join(dir, "polished.json")
writeScriptoriumFile(t, transcriptPath, `{"segments":[]}`) writeScriptoriumFile(t, transcriptPath, `{"segments":[]}`)
return RunArtifactRequest{ return RunArtifactRequest{
Binary: binary, Binary: binary,

View File

@@ -14,7 +14,7 @@ func TestFakeRunnerCapturesRequestAndReturnsPath(t *testing.T) {
dir := t.TempDir() dir := t.TempDir()
req := MergeRequest{ req := MergeRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.yml"), GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.yml"),
OutputMergedTranscriptPath: filepath.Join(dir, "transcripts", "merged.json"), OutputMergedTranscriptPath: filepath.Join(dir, "transcripts", "base.json"),
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.stdout.log"), StdoutLogPath: filepath.Join(dir, "logs", "seriatim.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "seriatim.stderr.log"), StderrLogPath: filepath.Join(dir, "logs", "seriatim.stderr.log"),
} }
@@ -57,8 +57,8 @@ func TestFakeRunnerTrimCapturesRequestAndReturnsPath(t *testing.T) {
dir := t.TempDir() dir := t.TempDir()
req := TrimRequest{ req := TrimRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.trim.yml"), GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.trim.yml"),
InputTranscriptPath: filepath.Join(dir, "transcripts", "processed.json"), InputTranscriptPath: filepath.Join(dir, "transcripts", "polished.json"),
OutputTrimmedPath: filepath.Join(dir, "transcripts", "trimmed.json"), OutputTrimmedPath: filepath.Join(dir, "transcripts", "final.trimmed.json"),
KeepSelector: "1-10", KeepSelector: "1-10",
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.trim.stdout.log"), StdoutLogPath: filepath.Join(dir, "logs", "seriatim.trim.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "seriatim.trim.stderr.log"), StderrLogPath: filepath.Join(dir, "logs", "seriatim.trim.stderr.log"),
@@ -105,8 +105,8 @@ func TestFakeRunnerNormalizeCapturesRequestAndReturnsPath(t *testing.T) {
dir := t.TempDir() dir := t.TempDir()
req := NormalizeRequest{ req := NormalizeRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.normalize.yml"), GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.normalize.yml"),
InputTranscriptPath: filepath.Join(dir, "transcripts", "processed.json"), InputTranscriptPath: filepath.Join(dir, "transcripts", "polished.json"),
OutputNormalizedPath: filepath.Join(dir, "transcripts", "normalized.json"), OutputNormalizedPath: filepath.Join(dir, "transcripts", "final.json"),
OutputSchema: "seriatim-intermediate", OutputSchema: "seriatim-intermediate",
ReportPath: filepath.Join(dir, "artifacts", "seriatim.normalize.report.json"), ReportPath: filepath.Join(dir, "artifacts", "seriatim.normalize.report.json"),
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.normalize.stdout.log"), StdoutLogPath: filepath.Join(dir, "logs", "seriatim.normalize.stdout.log"),

View File

@@ -50,7 +50,7 @@ func TestSubprocessRunnerSuccessWithReportArgsAndEnv(t *testing.T) {
req := MergeRequest{ req := MergeRequest{
GeneratedConfigPath: filepath.Join(dir, "seriatim.generated.yml"), GeneratedConfigPath: filepath.Join(dir, "seriatim.generated.yml"),
InputTranscriptPaths: []string{filepath.Join(dir, "a.json"), filepath.Join(dir, "b.json")}, InputTranscriptPaths: []string{filepath.Join(dir, "a.json"), filepath.Join(dir, "b.json")},
OutputMergedTranscriptPath: filepath.Join(dir, "merged.json"), OutputMergedTranscriptPath: filepath.Join(dir, "base.json"),
ReportPath: filepath.Join(dir, "seriatim.report.json"), ReportPath: filepath.Join(dir, "seriatim.report.json"),
SpeakersPath: filepath.Join(dir, "speakers.yml"), SpeakersPath: filepath.Join(dir, "speakers.yml"),
AutocorrectPath: filepath.Join(dir, "autocorrect.yml"), AutocorrectPath: filepath.Join(dir, "autocorrect.yml"),
@@ -732,7 +732,7 @@ func mergeReqForTest(t *testing.T, withReport bool) MergeRequest {
req := MergeRequest{ req := MergeRequest{
GeneratedConfigPath: filepath.Join(dir, "seriatim.generated.yml"), GeneratedConfigPath: filepath.Join(dir, "seriatim.generated.yml"),
InputTranscriptPaths: []string{in1, in2}, InputTranscriptPaths: []string{in1, in2},
OutputMergedTranscriptPath: filepath.Join(dir, "merged.json"), OutputMergedTranscriptPath: filepath.Join(dir, "base.json"),
StdoutLogPath: filepath.Join(dir, "seriatim.stdout.log"), StdoutLogPath: filepath.Join(dir, "seriatim.stdout.log"),
StderrLogPath: filepath.Join(dir, "seriatim.stderr.log"), StderrLogPath: filepath.Join(dir, "seriatim.stderr.log"),
} }
@@ -745,11 +745,11 @@ func mergeReqForTest(t *testing.T, withReport bool) MergeRequest {
func trimReqForTest(t *testing.T) TrimRequest { func trimReqForTest(t *testing.T) TrimRequest {
t.Helper() t.Helper()
dir := t.TempDir() dir := t.TempDir()
input := filepath.Join(dir, "processed.json") input := filepath.Join(dir, "polished.json")
writeSeriatimFile(t, input, `{"schema":"seriatim.intermediate.v1","segments":[]}`) writeSeriatimFile(t, input, `{"schema":"seriatim.intermediate.v1","segments":[]}`)
return TrimRequest{ return TrimRequest{
InputTranscriptPath: input, InputTranscriptPath: input,
OutputTrimmedPath: filepath.Join(dir, "trimmed.json"), OutputTrimmedPath: filepath.Join(dir, "final.trimmed.json"),
KeepSelector: "5-12", KeepSelector: "5-12",
GeneratedConfigPath: filepath.Join(dir, "seriatim.trim.generated.yml"), GeneratedConfigPath: filepath.Join(dir, "seriatim.trim.generated.yml"),
StdoutLogPath: filepath.Join(dir, "seriatim.trim.stdout.log"), StdoutLogPath: filepath.Join(dir, "seriatim.trim.stdout.log"),
@@ -760,12 +760,12 @@ func trimReqForTest(t *testing.T) TrimRequest {
func normalizeReqForTest(t *testing.T, withReport bool) NormalizeRequest { func normalizeReqForTest(t *testing.T, withReport bool) NormalizeRequest {
t.Helper() t.Helper()
dir := t.TempDir() dir := t.TempDir()
input := filepath.Join(dir, "processed.json") input := filepath.Join(dir, "polished.json")
writeSeriatimFile(t, input, `{"schema":"audita.processed.v1","segments":[]}`) writeSeriatimFile(t, input, `{"schema":"audita.processed.v1","segments":[]}`)
req := NormalizeRequest{ req := NormalizeRequest{
InputTranscriptPath: input, InputTranscriptPath: input,
OutputNormalizedPath: filepath.Join(dir, "normalized.json"), OutputNormalizedPath: filepath.Join(dir, "final.json"),
OutputSchema: "seriatim-intermediate", OutputSchema: "seriatim-intermediate",
GeneratedConfigPath: filepath.Join(dir, "seriatim.normalize.generated.yml"), GeneratedConfigPath: filepath.Join(dir, "seriatim.normalize.generated.yml"),
StdoutLogPath: filepath.Join(dir, "seriatim.normalize.stdout.log"), StdoutLogPath: filepath.Join(dir, "seriatim.normalize.stdout.log"),

View File

@@ -0,0 +1,29 @@
package storage
import (
"context"
"fmt"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
// NewObjectStoreFromConfig constructs a remote object store from resolved config.
func NewObjectStoreFromConfig(ctx context.Context, cfg *config.Config) (ObjectStore, error) {
if cfg == nil || cfg.Pipeline == nil {
return nil, fmt.Errorf("pipeline config is required")
}
if strings.EqualFold(strings.TrimSpace(cfg.Pipeline.Storage.Backend), "s3") {
if cfg.Pipeline.Storage.S3 == nil {
return nil, fmt.Errorf("pipeline.storage.s3 is required when pipeline.storage.backend is s3")
}
return NewS3BackendFromConfig(ctx, *cfg.Pipeline.Storage.S3)
}
if cfg.Pipeline.Storage.S3 != nil && strings.TrimSpace(cfg.Pipeline.Storage.S3.Bucket) != "" {
return NewS3BackendFromConfig(ctx, *cfg.Pipeline.Storage.S3)
}
return nil, fmt.Errorf("no remote object store backend is configured")
}

View File

@@ -0,0 +1,56 @@
package storage
import (
"context"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestNewObjectStoreFromConfigBuildsS3WhenBackendIsS3(t *testing.T) {
original := newS3Client
t.Cleanup(func() { newS3Client = original })
newS3Client = func(_ context.Context, _ s3ClientOptions) (s3API, error) {
return &fakeS3API{}, nil
}
store, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
Pipeline: &config.PipelineConfig{
Storage: config.StorageConfig{
Backend: "s3",
S3: &config.StorageS3Config{
Bucket: "my-archive",
},
},
},
})
if err != nil {
t.Fatalf("NewObjectStoreFromConfig() error = %v", err)
}
if _, ok := store.(*S3Backend); !ok {
t.Fatalf("store type = %T, want *S3Backend", store)
}
}
func TestNewObjectStoreFromConfigRequiresS3ConfigWhenBackendIsS3(t *testing.T) {
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
Pipeline: &config.PipelineConfig{
Storage: config.StorageConfig{Backend: "s3"},
},
})
if err == nil || !strings.Contains(err.Error(), "pipeline.storage.s3 is required") {
t.Fatalf("NewObjectStoreFromConfig() error = %v, want missing storage.s3 error", err)
}
}
func TestNewObjectStoreFromConfigNoRemoteBackendConfigured(t *testing.T) {
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
Pipeline: &config.PipelineConfig{
Storage: config.StorageConfig{Backend: "local"},
},
})
if err == nil || !strings.Contains(err.Error(), "no remote object store backend is configured") {
t.Fatalf("NewObjectStoreFromConfig() error = %v, want no-backend error", err)
}
}

View File

@@ -1,6 +1,14 @@
package storage package storage
import "context" import (
"context"
"fmt"
"os"
"path/filepath"
"sort"
"strings"
"time"
)
// NoopBackend is a deterministic no-op archive/storage adapter. // NoopBackend is a deterministic no-op archive/storage adapter.
type NoopBackend struct{} type NoopBackend struct{}
@@ -18,6 +26,28 @@ type FakeBackend struct {
Requests []ArchiveRequest Requests []ArchiveRequest
Err error Err error
Result ArchiveResult Result ArchiveResult
Objects map[string]FakeObject
Uploads []FakeUploadCall
Downloads []FakeDownloadCall
ListErr error
DownloadErr error
UploadErr error
ExistsErr error
}
// FakeUploadCall captures one upload invocation in call order.
type FakeUploadCall struct {
LocalPath string
Key string
Options UploadOptions
}
// FakeDownloadCall captures one download invocation in call order.
type FakeDownloadCall struct {
Key string
LocalPath string
} }
// Archive records request and returns configured response. // Archive records request and returns configured response.
@@ -38,3 +68,152 @@ func (f *FakeBackend) Archive(ctx context.Context, req ArchiveRequest) (ArchiveR
} }
return res, nil return res, nil
} }
// FakeObject is a deterministic fake object-store record.
type FakeObject struct {
Key string
Data []byte
Metadata map[string]string
ETag string
LastModified *time.Time
}
// SeedObject inserts or replaces an object in the fake object store.
func (f *FakeBackend) SeedObject(obj FakeObject) {
if f.Objects == nil {
f.Objects = map[string]FakeObject{}
}
key := normalizeObjectKey(obj.Key)
obj.Key = key
obj.Data = append([]byte(nil), obj.Data...)
obj.Metadata = copyMetadata(obj.Metadata)
f.Objects[key] = obj
}
// List returns deterministic prefix-filtered objects.
func (f *FakeBackend) List(ctx context.Context, prefix string) ([]ObjectInfo, error) {
if err := ctx.Err(); err != nil {
return nil, err
}
if f.ListErr != nil {
return nil, f.ListErr
}
normalizedPrefix := normalizeObjectKey(prefix)
keys := make([]string, 0, len(f.Objects))
for key := range f.Objects {
if strings.HasPrefix(key, normalizedPrefix) {
keys = append(keys, key)
}
}
sort.Strings(keys)
out := make([]ObjectInfo, 0, len(keys))
for _, key := range keys {
obj := f.Objects[key]
out = append(out, ObjectInfo{
Key: obj.Key,
Size: int64(len(obj.Data)),
ETag: obj.ETag,
LastModified: obj.LastModified,
})
}
return out, nil
}
// Download writes one object to a local path.
func (f *FakeBackend) Download(ctx context.Context, key, localPath string) error {
if err := ctx.Err(); err != nil {
return err
}
if f.DownloadErr != nil {
return f.DownloadErr
}
if strings.TrimSpace(localPath) == "" {
return fmt.Errorf("download object: local path is required")
}
obj, ok := f.Objects[normalizeObjectKey(key)]
if !ok {
return fmt.Errorf("download object %q: %w", key, os.ErrNotExist)
}
f.Downloads = append(f.Downloads, FakeDownloadCall{
Key: normalizeObjectKey(key),
LocalPath: localPath,
})
if err := os.MkdirAll(filepath.Dir(localPath), 0o755); err != nil {
return fmt.Errorf("download object %q: create parent directory: %w", key, err)
}
if err := os.WriteFile(localPath, obj.Data, 0o644); err != nil {
return fmt.Errorf("download object %q: write local file: %w", key, err)
}
return nil
}
// Upload reads a local file and stores it under key.
func (f *FakeBackend) Upload(ctx context.Context, localPath, key string, opts UploadOptions) (ObjectInfo, error) {
if err := ctx.Err(); err != nil {
return ObjectInfo{}, err
}
if f.UploadErr != nil {
return ObjectInfo{}, f.UploadErr
}
if strings.TrimSpace(localPath) == "" {
return ObjectInfo{}, fmt.Errorf("upload object: local path is required")
}
if strings.TrimSpace(key) == "" {
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
}
data, err := os.ReadFile(localPath)
if err != nil {
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", key, localPath, err)
}
normalizedKey := normalizeObjectKey(key)
f.Uploads = append(f.Uploads, FakeUploadCall{
LocalPath: localPath,
Key: normalizedKey,
Options: UploadOptions{
Metadata: copyMetadata(opts.Metadata),
ContentType: opts.ContentType,
},
})
now := time.Now().UTC()
obj := FakeObject{
Key: normalizedKey,
Data: data,
Metadata: copyMetadata(opts.Metadata),
LastModified: &now,
}
f.SeedObject(obj)
return ObjectInfo{
Key: normalizedKey,
Size: int64(len(data)),
LastModified: &now,
}, nil
}
// Exists checks object presence.
func (f *FakeBackend) Exists(ctx context.Context, key string) (bool, error) {
if err := ctx.Err(); err != nil {
return false, err
}
if f.ExistsErr != nil {
return false, f.ExistsErr
}
_, ok := f.Objects[normalizeObjectKey(key)]
return ok, nil
}
func copyMetadata(in map[string]string) map[string]string {
if len(in) == 0 {
return nil
}
out := make(map[string]string, len(in))
for k, v := range in {
out[k] = v
}
return out
}

View File

@@ -3,6 +3,9 @@ package storage
import ( import (
"context" "context"
"errors" "errors"
"os"
"path/filepath"
"strings"
"testing" "testing"
) )
@@ -29,3 +32,84 @@ func TestFakeBackendError(t *testing.T) {
t.Fatal("expected error, got nil") t.Fatal("expected error, got nil")
} }
} }
func TestFakeBackendListPrefixFiltering(t *testing.T) {
fake := &FakeBackend{}
fake.SeedObject(FakeObject{Key: "dnd/campaigns/forsaken/audio/a.flac", Data: []byte("a")})
fake.SeedObject(FakeObject{Key: "dnd/campaigns/forsaken/audio/b.flac", Data: []byte("b")})
fake.SeedObject(FakeObject{Key: "dnd/campaigns/other/audio/c.flac", Data: []byte("c")})
items, err := fake.List(context.Background(), "dnd/campaigns/forsaken/audio/")
if err != nil {
t.Fatalf("List() error = %v", err)
}
if len(items) != 2 {
t.Fatalf("List() len = %d, want 2", len(items))
}
if items[0].Key != "dnd/campaigns/forsaken/audio/a.flac" || items[1].Key != "dnd/campaigns/forsaken/audio/b.flac" {
t.Fatalf("List() keys = %#v", items)
}
}
func TestFakeBackendDownload(t *testing.T) {
fake := &FakeBackend{}
fake.SeedObject(FakeObject{Key: "audio/a.flac", Data: []byte("audio-a")})
dst := filepath.Join(t.TempDir(), "nested", "a.flac")
if err := fake.Download(context.Background(), `audio\a.flac`, dst); err != nil {
t.Fatalf("Download() error = %v", err)
}
data, err := os.ReadFile(dst)
if err != nil {
t.Fatalf("ReadFile() error = %v", err)
}
if string(data) != "audio-a" {
t.Fatalf("downloaded content = %q, want %q", string(data), "audio-a")
}
}
func TestFakeBackendUploadAndExists(t *testing.T) {
fake := &FakeBackend{}
local := filepath.Join(t.TempDir(), "upload.txt")
if err := os.WriteFile(local, []byte("payload"), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
info, err := fake.Upload(context.Background(), local, `runs\id\artifact.txt`, UploadOptions{
Metadata: map[string]string{"kind": "artifact"},
})
if err != nil {
t.Fatalf("Upload() error = %v", err)
}
if info.Key != "runs/id/artifact.txt" {
t.Fatalf("Upload() key = %q, want normalized key", info.Key)
}
ok, err := fake.Exists(context.Background(), "runs/id/artifact.txt")
if err != nil {
t.Fatalf("Exists() error = %v", err)
}
if !ok {
t.Fatal("Exists() = false, want true")
}
}
func TestFakeBackendObjectErrors(t *testing.T) {
fake := &FakeBackend{DownloadErr: errors.New("download fail"), UploadErr: errors.New("upload fail"), ListErr: errors.New("list fail"), ExistsErr: errors.New("exists fail")}
if _, err := fake.List(context.Background(), "x"); err == nil || !strings.Contains(err.Error(), "list fail") {
t.Fatalf("List() error = %v, want list fail", err)
}
if err := fake.Download(context.Background(), "x", filepath.Join(t.TempDir(), "x")); err == nil || !strings.Contains(err.Error(), "download fail") {
t.Fatalf("Download() error = %v, want download fail", err)
}
local := filepath.Join(t.TempDir(), "x.txt")
_ = os.WriteFile(local, []byte("x"), 0o644)
if _, err := fake.Upload(context.Background(), local, "x", UploadOptions{}); err == nil || !strings.Contains(err.Error(), "upload fail") {
t.Fatalf("Upload() error = %v, want upload fail", err)
}
if _, err := fake.Exists(context.Background(), "x"); err == nil || !strings.Contains(err.Error(), "exists fail") {
t.Fatalf("Exists() error = %v, want exists fail", err)
}
}

View File

@@ -0,0 +1,8 @@
package storage
import "strings"
func normalizeObjectKey(key string) string {
normalized := strings.ReplaceAll(strings.TrimSpace(key), "\\", "/")
return strings.TrimLeft(normalized, "/")
}

View File

@@ -0,0 +1,21 @@
package storage
import "testing"
func TestNormalizeObjectKey(t *testing.T) {
tests := []struct {
in string
want string
}{
{in: `dnd\campaigns\forsaken\a.flac`, want: "dnd/campaigns/forsaken/a.flac"},
{in: " /dnd/campaigns/forsaken/a.flac ", want: "dnd/campaigns/forsaken/a.flac"},
{in: "//dnd/campaigns/forsaken/a.flac", want: "dnd/campaigns/forsaken/a.flac"},
{in: "", want: ""},
}
for _, tt := range tests {
if got := normalizeObjectKey(tt.in); got != tt.want {
t.Fatalf("normalizeObjectKey(%q) = %q, want %q", tt.in, got, tt.want)
}
}
}

View File

@@ -0,0 +1,32 @@
package storage
import (
"context"
"time"
)
// ObjectStore is a remote object storage boundary used by future prepare/archive work.
//
// Key invariant:
// callers pass full bucket-relative object keys. Backend implementations do not
// infer Narratio session semantics and do not prepend root prefixes.
type ObjectStore interface {
List(ctx context.Context, prefix string) ([]ObjectInfo, error)
Download(ctx context.Context, key, localPath string) error
Upload(ctx context.Context, localPath, key string, opts UploadOptions) (ObjectInfo, error)
Exists(ctx context.Context, key string) (bool, error)
}
// ObjectInfo describes one object in remote storage.
type ObjectInfo struct {
Key string
Size int64
ETag string
LastModified *time.Time
}
// UploadOptions configures optional object upload metadata.
type UploadOptions struct {
Metadata map[string]string
ContentType string
}

View File

@@ -0,0 +1,275 @@
package storage
import (
"context"
"errors"
"fmt"
"io"
"os"
"path/filepath"
"strings"
"time"
awsconfig "github.com/aws/aws-sdk-go-v2/config"
"github.com/aws/aws-sdk-go-v2/credentials"
"github.com/aws/aws-sdk-go-v2/service/s3"
"github.com/aws/aws-sdk-go-v2/service/s3/types"
"github.com/aws/smithy-go"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
type s3API interface {
ListObjectsV2(ctx context.Context, params *s3.ListObjectsV2Input, optFns ...func(*s3.Options)) (*s3.ListObjectsV2Output, error)
GetObject(ctx context.Context, params *s3.GetObjectInput, optFns ...func(*s3.Options)) (*s3.GetObjectOutput, error)
PutObject(ctx context.Context, params *s3.PutObjectInput, optFns ...func(*s3.Options)) (*s3.PutObjectOutput, error)
HeadObject(ctx context.Context, params *s3.HeadObjectInput, optFns ...func(*s3.Options)) (*s3.HeadObjectOutput, error)
}
// S3Backend is an ObjectStore implementation backed by S3-compatible APIs.
type S3Backend struct {
bucket string
client s3API
}
type s3ClientOptions struct {
Region string
Endpoint string
ForcePathStyle bool
AccessKeyID string
SecretKey string
}
var newS3Client = func(ctx context.Context, opts s3ClientOptions) (s3API, error) {
loadOpts := make([]func(*awsconfig.LoadOptions) error, 0, 1)
if strings.TrimSpace(opts.Region) != "" {
loadOpts = append(loadOpts, awsconfig.WithRegion(strings.TrimSpace(opts.Region)))
}
if strings.TrimSpace(opts.AccessKeyID) != "" && strings.TrimSpace(opts.SecretKey) != "" {
loadOpts = append(loadOpts, awsconfig.WithCredentialsProvider(
credentials.NewStaticCredentialsProvider(
strings.TrimSpace(opts.AccessKeyID),
strings.TrimSpace(opts.SecretKey),
"",
),
))
}
awsCfg, err := awsconfig.LoadDefaultConfig(ctx, loadOpts...)
if err != nil {
return nil, fmt.Errorf("load aws config: %w", err)
}
return s3.NewFromConfig(awsCfg, func(o *s3.Options) {
if strings.TrimSpace(opts.Endpoint) != "" {
endpoint := strings.TrimSpace(opts.Endpoint)
o.BaseEndpoint = &endpoint
}
o.UsePathStyle = opts.ForcePathStyle
}), nil
}
// NewS3BackendFromConfig builds an S3 backend from resolved config.
func NewS3BackendFromConfig(ctx context.Context, cfg config.StorageS3Config) (*S3Backend, error) {
bucket := strings.TrimSpace(cfg.Bucket)
if bucket == "" {
return nil, fmt.Errorf("storage.s3.bucket is required")
}
client, err := newS3Client(ctx, s3ClientOptions{
Region: cfg.Region,
Endpoint: cfg.Endpoint,
ForcePathStyle: cfg.ForcePathStyle,
AccessKeyID: s3CredentialFromEnv(orDefaultEnvName(cfg.AccessKeyIDEnv, config.DefaultS3AccessKeyIDEnv)),
SecretKey: s3CredentialFromEnv(orDefaultEnvName(cfg.SecretKeyEnv, config.DefaultS3SecretAccessKeyEnv)),
})
if err != nil {
return nil, fmt.Errorf("build s3 client: %w", err)
}
return &S3Backend{
bucket: bucket,
client: client,
}, nil
}
func s3CredentialFromEnv(envVarName string) string {
name := strings.TrimSpace(envVarName)
if name == "" {
return ""
}
value, ok := os.LookupEnv(name)
if !ok {
return ""
}
return strings.TrimSpace(value)
}
func orDefaultEnvName(name, fallback string) string {
trimmed := strings.TrimSpace(name)
if trimmed == "" {
return fallback
}
return trimmed
}
// List returns objects under prefix.
func (b *S3Backend) List(ctx context.Context, prefix string) ([]ObjectInfo, error) {
normalizedPrefix := normalizeObjectKey(prefix)
out := make([]ObjectInfo, 0)
var token *string
for {
resp, err := b.client.ListObjectsV2(ctx, &s3.ListObjectsV2Input{
Bucket: &b.bucket,
Prefix: &normalizedPrefix,
ContinuationToken: token,
})
if err != nil {
return nil, fmt.Errorf("list objects under %q: %w", normalizedPrefix, err)
}
for _, item := range resp.Contents {
var lastModified *time.Time
if item.LastModified != nil {
t := *item.LastModified
lastModified = &t
}
out = append(out, ObjectInfo{
Key: normalizeObjectKey(valueOrEmpty(item.Key)),
Size: valueOrZeroInt64(item.Size),
ETag: strings.Trim(valueOrEmpty(item.ETag), "\""),
LastModified: lastModified,
})
}
if !valueOrFalseBool(resp.IsTruncated) || resp.NextContinuationToken == nil {
break
}
token = resp.NextContinuationToken
}
return out, nil
}
// Download retrieves one object to localPath, creating parent directories as needed.
func (b *S3Backend) Download(ctx context.Context, key, localPath string) error {
normalizedKey := normalizeObjectKey(key)
if strings.TrimSpace(localPath) == "" {
return fmt.Errorf("download object: local path is required")
}
resp, err := b.client.GetObject(ctx, &s3.GetObjectInput{
Bucket: &b.bucket,
Key: &normalizedKey,
})
if err != nil {
return fmt.Errorf("download object %q: %w", normalizedKey, err)
}
defer resp.Body.Close()
if err := os.MkdirAll(filepath.Dir(localPath), 0o755); err != nil {
return fmt.Errorf("download object %q: create parent directory: %w", normalizedKey, err)
}
dst, err := os.Create(localPath)
if err != nil {
return fmt.Errorf("download object %q: create local file: %w", normalizedKey, err)
}
defer dst.Close()
if _, err := io.Copy(dst, resp.Body); err != nil {
return fmt.Errorf("download object %q: copy body: %w", normalizedKey, err)
}
if err := dst.Sync(); err != nil {
return fmt.Errorf("download object %q: sync local file: %w", normalizedKey, err)
}
return nil
}
// Upload sends a local file to key.
func (b *S3Backend) Upload(ctx context.Context, localPath, key string, opts UploadOptions) (ObjectInfo, error) {
normalizedKey := normalizeObjectKey(key)
if strings.TrimSpace(localPath) == "" {
return ObjectInfo{}, fmt.Errorf("upload object: local path is required")
}
if normalizedKey == "" {
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
}
file, err := os.Open(localPath)
if err != nil {
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", normalizedKey, localPath, err)
}
defer file.Close()
stat, err := file.Stat()
if err != nil {
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: stat local file: %w", normalizedKey, localPath, err)
}
input := &s3.PutObjectInput{
Bucket: &b.bucket,
Key: &normalizedKey,
Body: file,
Metadata: copyMetadata(opts.Metadata),
}
if strings.TrimSpace(opts.ContentType) != "" {
ct := strings.TrimSpace(opts.ContentType)
input.ContentType = &ct
}
resp, err := b.client.PutObject(ctx, input)
if err != nil {
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", normalizedKey, localPath, err)
}
return ObjectInfo{
Key: normalizedKey,
Size: stat.Size(),
ETag: strings.Trim(valueOrEmpty(resp.ETag), "\""),
}, nil
}
// Exists checks whether one object key exists.
func (b *S3Backend) Exists(ctx context.Context, key string) (bool, error) {
normalizedKey := normalizeObjectKey(key)
_, err := b.client.HeadObject(ctx, &s3.HeadObjectInput{
Bucket: &b.bucket,
Key: &normalizedKey,
})
if err == nil {
return true, nil
}
var notFound *types.NotFound
if errors.As(err, &notFound) {
return false, nil
}
var apiErr smithy.APIError
if errors.As(err, &apiErr) {
switch apiErr.ErrorCode() {
case "NotFound", "NoSuchKey", "404":
return false, nil
}
}
return false, fmt.Errorf("head object %q: %w", normalizedKey, err)
}
func valueOrEmpty(v *string) string {
if v == nil {
return ""
}
return *v
}
func valueOrZeroInt64(v *int64) int64 {
if v == nil {
return 0
}
return *v
}
func valueOrFalseBool(v *bool) bool {
if v == nil {
return false
}
return *v
}

View File

@@ -0,0 +1,253 @@
package storage
import (
"context"
"io"
"os"
"path/filepath"
"strings"
"testing"
"time"
"github.com/aws/aws-sdk-go-v2/service/s3"
"github.com/aws/aws-sdk-go-v2/service/s3/types"
"github.com/aws/smithy-go"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
type fakeS3API struct {
listOut *s3.ListObjectsV2Output
listErr error
getBody io.ReadCloser
getErr error
putOut *s3.PutObjectOutput
putErr error
headErr error
lastList *s3.ListObjectsV2Input
lastGet *s3.GetObjectInput
lastPut *s3.PutObjectInput
lastHead *s3.HeadObjectInput
}
func (f *fakeS3API) ListObjectsV2(_ context.Context, params *s3.ListObjectsV2Input, _ ...func(*s3.Options)) (*s3.ListObjectsV2Output, error) {
f.lastList = params
if f.listErr != nil {
return nil, f.listErr
}
if f.listOut == nil {
return &s3.ListObjectsV2Output{}, nil
}
return f.listOut, nil
}
func (f *fakeS3API) GetObject(_ context.Context, params *s3.GetObjectInput, _ ...func(*s3.Options)) (*s3.GetObjectOutput, error) {
f.lastGet = params
if f.getErr != nil {
return nil, f.getErr
}
body := f.getBody
if body == nil {
body = io.NopCloser(strings.NewReader(""))
}
return &s3.GetObjectOutput{Body: body}, nil
}
func (f *fakeS3API) PutObject(_ context.Context, params *s3.PutObjectInput, _ ...func(*s3.Options)) (*s3.PutObjectOutput, error) {
f.lastPut = params
if f.putErr != nil {
return nil, f.putErr
}
if f.putOut == nil {
return &s3.PutObjectOutput{}, nil
}
return f.putOut, nil
}
func (f *fakeS3API) HeadObject(_ context.Context, params *s3.HeadObjectInput, _ ...func(*s3.Options)) (*s3.HeadObjectOutput, error) {
f.lastHead = params
if f.headErr != nil {
return nil, f.headErr
}
return &s3.HeadObjectOutput{}, nil
}
func TestS3BackendListAndKeyNormalization(t *testing.T) {
lastModified := time.Date(2026, 5, 16, 12, 0, 0, 0, time.UTC)
client := &fakeS3API{
listOut: &s3.ListObjectsV2Output{
Contents: []types.Object{
{Key: strPtr(`dnd\campaigns\forsaken\a.flac`), Size: int64Ptr(7), ETag: strPtr(`"abc"`), LastModified: &lastModified},
},
},
}
backend := &S3Backend{bucket: "bucket-1", client: client}
items, err := backend.List(context.Background(), `dnd\campaigns\`)
if err != nil {
t.Fatalf("List() error = %v", err)
}
if len(items) != 1 {
t.Fatalf("List() len = %d, want 1", len(items))
}
if items[0].Key != "dnd/campaigns/forsaken/a.flac" {
t.Fatalf("List() key = %q, want normalized slash key", items[0].Key)
}
if items[0].ETag != "abc" {
t.Fatalf("List() ETag = %q, want %q", items[0].ETag, "abc")
}
if client.lastList == nil || *client.lastList.Prefix != "dnd/campaigns/" {
t.Fatalf("List() prefix = %#v, want normalized prefix", client.lastList)
}
}
func TestS3BackendDownloadCreatesParentDirectory(t *testing.T) {
client := &fakeS3API{getBody: io.NopCloser(strings.NewReader("audio"))}
backend := &S3Backend{bucket: "bucket-1", client: client}
dst := filepath.Join(t.TempDir(), "nested", "clip.flac")
if err := backend.Download(context.Background(), `audio\clip.flac`, dst); err != nil {
t.Fatalf("Download() error = %v", err)
}
data, err := os.ReadFile(dst)
if err != nil {
t.Fatalf("ReadFile() error = %v", err)
}
if string(data) != "audio" {
t.Fatalf("downloaded content = %q, want %q", string(data), "audio")
}
if client.lastGet == nil || *client.lastGet.Key != "audio/clip.flac" {
t.Fatalf("GetObject key = %#v, want normalized key", client.lastGet)
}
}
func TestS3BackendUploadAndExists(t *testing.T) {
client := &fakeS3API{putOut: &s3.PutObjectOutput{ETag: strPtr(`"etag123"`)}}
backend := &S3Backend{bucket: "bucket-1", client: client}
local := filepath.Join(t.TempDir(), "artifact.txt")
if err := os.WriteFile(local, []byte("artifact"), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
info, err := backend.Upload(context.Background(), local, `runs\id\artifact.txt`, UploadOptions{
Metadata: map[string]string{"kind": "artifact"},
})
if err != nil {
t.Fatalf("Upload() error = %v", err)
}
if info.Key != "runs/id/artifact.txt" {
t.Fatalf("Upload key = %q, want normalized key", info.Key)
}
if info.ETag != "etag123" {
t.Fatalf("Upload ETag = %q, want %q", info.ETag, "etag123")
}
if client.lastPut == nil || *client.lastPut.Key != "runs/id/artifact.txt" {
t.Fatalf("PutObject key = %#v, want normalized key", client.lastPut)
}
ok, err := backend.Exists(context.Background(), "runs/id/artifact.txt")
if err != nil {
t.Fatalf("Exists() error = %v", err)
}
if !ok {
t.Fatal("Exists() = false, want true")
}
}
func TestS3BackendUploadMissingLocalFile(t *testing.T) {
backend := &S3Backend{bucket: "bucket-1", client: &fakeS3API{}}
_, err := backend.Upload(context.Background(), filepath.Join(t.TempDir(), "missing.txt"), "key.txt", UploadOptions{})
if err == nil || !strings.Contains(err.Error(), "no such file") {
t.Fatalf("Upload() error = %v, want missing local file error", err)
}
}
func TestS3BackendExistsNotFound(t *testing.T) {
backend := &S3Backend{
bucket: "bucket-1",
client: &fakeS3API{
headErr: &smithy.GenericAPIError{Code: "NotFound", Message: "missing"},
},
}
ok, err := backend.Exists(context.Background(), "missing-key")
if err != nil {
t.Fatalf("Exists() error = %v", err)
}
if ok {
t.Fatal("Exists() = true, want false")
}
}
func TestNewS3BackendFromConfigUsesClientOptions(t *testing.T) {
original := newS3Client
t.Cleanup(func() { newS3Client = original })
t.Setenv("OBJECT_STORAGE_KEY_ID", "id-123")
t.Setenv("OBJECT_STORAGE_KEY", "secret-abc")
var got s3ClientOptions
newS3Client = func(_ context.Context, opts s3ClientOptions) (s3API, error) {
got = opts
return &fakeS3API{}, nil
}
backend, err := NewS3BackendFromConfig(context.Background(), config.StorageS3Config{
Bucket: "my-archive",
Region: "us-east-1",
Endpoint: "http://localhost:9000",
ForcePathStyle: true,
})
if err != nil {
t.Fatalf("NewS3BackendFromConfig() error = %v", err)
}
if backend.bucket != "my-archive" {
t.Fatalf("backend.bucket = %q, want %q", backend.bucket, "my-archive")
}
if got.Region != "us-east-1" || got.Endpoint != "http://localhost:9000" || !got.ForcePathStyle {
t.Fatalf("client options = %#v, want region/endpoint/path-style values", got)
}
if got.AccessKeyID != "id-123" || got.SecretKey != "secret-abc" {
t.Fatalf("client options credentials = %#v, want env-resolved static credentials", got)
}
}
func TestNewS3BackendFromConfigRequiresBucket(t *testing.T) {
_, err := NewS3BackendFromConfig(context.Background(), config.StorageS3Config{})
if err == nil || !strings.Contains(err.Error(), "bucket is required") {
t.Fatalf("NewS3BackendFromConfig() error = %v, want bucket validation", err)
}
}
func TestNewS3BackendFromConfigFallsBackWhenCredentialEnvMissing(t *testing.T) {
original := newS3Client
t.Cleanup(func() { newS3Client = original })
var got s3ClientOptions
newS3Client = func(_ context.Context, opts s3ClientOptions) (s3API, error) {
got = opts
return &fakeS3API{}, nil
}
_, err := NewS3BackendFromConfig(context.Background(), config.StorageS3Config{
Bucket: "my-archive",
Region: "us-east-1",
AccessKeyIDEnv: "MISSING_ACCESS_KEY_ID",
SecretKeyEnv: "MISSING_SECRET_KEY",
})
if err != nil {
t.Fatalf("NewS3BackendFromConfig() error = %v", err)
}
if got.AccessKeyID != "" || got.SecretKey != "" {
t.Fatalf("client options credentials = %#v, want empty fallback values", got)
}
}
func strPtr(v string) *string { return &v }
func int64Ptr(v int64) *int64 { return &v }
var _ s3API = (*fakeS3API)(nil)

View File

@@ -0,0 +1,66 @@
package app
import (
"fmt"
"sort"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
type artifactSelectionFlag struct {
values []string
}
func (f *artifactSelectionFlag) String() string {
return strings.Join(f.values, ",")
}
func (f *artifactSelectionFlag) Set(value string) error {
f.values = append(f.values, value)
return nil
}
func (f *artifactSelectionFlag) Normalize() ([]string, error) {
if len(f.values) == 0 {
return nil, nil
}
seen := map[string]struct{}{}
out := make([]string, 0, len(f.values))
for _, raw := range f.values {
for _, part := range strings.Split(raw, ",") {
name := strings.TrimSpace(part)
if name == "" {
return nil, fmt.Errorf("artifact names must be non-empty")
}
if _, ok := seen[name]; ok {
continue
}
seen[name] = struct{}{}
out = append(out, name)
}
}
sort.Strings(out)
return out, nil
}
func validateSelectedArtifacts(cfg *config.Config, selected []string) error {
if len(selected) == 0 {
return nil
}
if cfg == nil || cfg.Pipeline == nil || cfg.Pipeline.Scriptorium == nil {
return fmt.Errorf("--artifacts requires pipeline.scriptorium.artifacts to be configured")
}
configured := cfg.Pipeline.Scriptorium.Artifacts
if len(configured) == 0 {
return fmt.Errorf("--artifacts requires at least one configured artifact in pipeline.scriptorium.artifacts")
}
for _, name := range selected {
if _, ok := configured[name]; !ok {
return fmt.Errorf("--artifacts includes unknown artifact %q", name)
}
}
return nil
}

View File

@@ -0,0 +1,434 @@
package app
import (
"bytes"
"context"
"os"
"path/filepath"
"strings"
"testing"
"time"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
"gitea.maximumdirect.net/eric/narratio/internal/stage"
)
func TestExecuteRunStageArtifactsUnsupportedStageFails(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"run-stage", "polish", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "session_recap"},
&stdout,
&stderr,
)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), `run-stage: --artifacts is only supported for stages "analyze" and "archive"`) {
t.Fatalf("stderr = %q, want stage-gating error", stderr.String())
}
}
func TestExecuteRunStageArchivePropagatesSelectedArtifacts(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var capturedStages []string
var capturedArtifacts []string
origExecuteStagesFn := executeStagesFn
t.Cleanup(func() {
executeStagesFn = origExecuteStagesFn
})
executeStagesFn = func(_ context.Context, _ *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
for _, s := range stages {
capturedStages = append(capturedStages, s.Name())
}
capturedArtifacts = append([]string(nil), opts.SelectedArtifacts...)
return &RunSummary{ManifestPath: filepath.Join(workspaceRoot, "manifest.json"), Executed: []string{"archive"}}, nil
}
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{
"run-stage", "archive", "2026-05-03",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--artifacts", "session_recap",
},
&stdout,
&stderr,
)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if len(capturedStages) != 1 || capturedStages[0] != "archive" {
t.Fatalf("captured stages = %#v, want [archive]", capturedStages)
}
if strings.Join(capturedArtifacts, ",") != "session_recap" {
t.Fatalf("captured artifacts = %#v, want [session_recap]", capturedArtifacts)
}
}
func TestExecuteUnknownArtifactsFailValidation(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"run", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "unknown_artifact"},
&stdout,
&stderr,
)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), `run: --artifacts includes unknown artifact "unknown_artifact"`) {
t.Fatalf("stderr = %q, want unknown-artifact validation error", stderr.String())
}
}
func TestRunStageArtifactsDoesNotImplyForce(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{}
seed := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
seed.MarkStageSucceeded("analyze", time.Date(2026, 5, 3, 10, 1, 0, 0, time.UTC), nil)
if err := store.Save(context.Background(), manifestPath, seed); err != nil {
t.Fatalf("save manifest: %v", err)
}
var out bytes.Buffer
err := RunStage(
context.Background(),
[]string{"analyze", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "session_recap,session_recap"},
&out,
)
if err != nil {
t.Fatalf("RunStage() error = %v", err)
}
if !strings.Contains(out.String(), "stage=analyze executed=0 skipped=1 force=false") {
t.Fatalf("output = %q, want analyze skip without force", out.String())
}
}
func TestResumeArtifactsWithSucceededAnalyzeSkipsUnlessForced(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{}
seed := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
for _, stageName := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "analyze", "archive", "notify"} {
seed.MarkStageSucceeded(stageName, time.Date(2026, 5, 3, 10, 1, 0, 0, time.UTC), nil)
}
if err := store.Save(context.Background(), manifestPath, seed); err != nil {
t.Fatalf("save manifest: %v", err)
}
var out bytes.Buffer
err := Resume(
context.Background(),
[]string{"2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "session_recap"},
&out,
)
if err != nil {
t.Fatalf("Resume() error = %v", err)
}
if !strings.Contains(out.String(), "has no remaining stages") {
t.Fatalf("output = %q, want no remaining stages", out.String())
}
}
func TestExecuteAnalyzeForceRunsAnalyze(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var capturedStages []string
var capturedForce bool
origExecuteStagesFn := executeStagesFn
t.Cleanup(func() {
executeStagesFn = origExecuteStagesFn
})
executeStagesFn = func(_ context.Context, _ *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
for _, s := range stages {
capturedStages = append(capturedStages, s.Name())
}
capturedForce = opts.Force
return &RunSummary{
ManifestPath: filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json"),
Executed: []string{"analyze"},
}, nil
}
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"analyze", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath},
&stdout,
&stderr,
)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if len(capturedStages) != 1 || capturedStages[0] != "analyze" {
t.Fatalf("captured stages = %#v, want [analyze]", capturedStages)
}
if !capturedForce {
t.Fatal("captured force = false, want true")
}
if !strings.Contains(stdout.String(), "narratio analyze: executed=1 skipped=0 force=true; manifest=") {
t.Fatalf("stdout = %q, want analyze summary", stdout.String())
}
}
func TestExecuteAnalyzePropagatesSelectedArtifacts(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var capturedArtifacts []string
origExecuteStagesFn := executeStagesFn
t.Cleanup(func() {
executeStagesFn = origExecuteStagesFn
})
executeStagesFn = func(_ context.Context, _ *config.Config, _ []stage.Stage, opts RunOptions) (*RunSummary, error) {
capturedArtifacts = append([]string(nil), opts.SelectedArtifacts...)
return &RunSummary{ManifestPath: filepath.Join(workspaceRoot, "manifest.json"), Executed: []string{"analyze"}}, nil
}
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{
"analyze",
"2026-05-03",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--artifacts", "player_handout,session_recap",
},
&stdout,
&stderr,
)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if strings.Join(capturedArtifacts, ",") != "player_handout,session_recap" {
t.Fatalf("captured artifacts = %#v, want sorted selected artifacts", capturedArtifacts)
}
}
func TestExecuteAnalyzeUnknownArtifactFailsValidation(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"analyze", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "unknown_artifact"},
&stdout,
&stderr,
)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), `analyze: --artifacts includes unknown artifact "unknown_artifact"`) {
t.Fatalf("stderr = %q, want unknown-artifact validation error", stderr.String())
}
}
func TestExecuteAnalyzeRejectsPositionalArgsAndForceFlag(t *testing.T) {
cases := []struct {
name string
args []string
want string
}{
{name: "extra positional", args: []string{"analyze", "2026-05-03", "extra"}, want: "analyze: unexpected positional arguments"},
{name: "force flag", args: []string{"analyze", "--force"}, want: "analyze: invalid flags: flag provided but not defined: -force"},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(tc.args, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), tc.want) {
t.Fatalf("stderr = %q, want %q", stderr.String(), tc.want)
}
})
}
}
func TestExecuteAnalyzeMissingConfigUsesRunStageLoadingPath(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"analyze", "2026-05-03"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "analyze: no pipeline config path provided and no default pipeline config found; searched:") {
t.Fatalf("stderr = %q, want pipeline discovery error", stderr.String())
}
}
func TestExecutePublishForceRunsArchive(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var capturedStages []string
var capturedForce bool
var capturedArtifacts []string
origExecuteStagesFn := executeStagesFn
t.Cleanup(func() {
executeStagesFn = origExecuteStagesFn
})
executeStagesFn = func(_ context.Context, _ *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
for _, s := range stages {
capturedStages = append(capturedStages, s.Name())
}
capturedForce = opts.Force
capturedArtifacts = append([]string(nil), opts.SelectedArtifacts...)
return &RunSummary{
ManifestPath: filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json"),
Executed: []string{"archive"},
}, nil
}
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"publish", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "session_recap"},
&stdout,
&stderr,
)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if len(capturedStages) != 1 || capturedStages[0] != "archive" {
t.Fatalf("captured stages = %#v, want [archive]", capturedStages)
}
if !capturedForce {
t.Fatal("captured force = false, want true")
}
if strings.Join(capturedArtifacts, ",") != "session_recap" {
t.Fatalf("captured artifacts = %#v, want [session_recap]", capturedArtifacts)
}
if !strings.Contains(stdout.String(), "narratio publish: executed=1 skipped=0 force=true; manifest=") {
t.Fatalf("stdout = %q, want publish summary", stdout.String())
}
}
func TestExecutePublishRejectsUnsupportedArgsAndFlags(t *testing.T) {
cases := []struct {
name string
args []string
want string
}{
{name: "extra positional", args: []string{"publish", "2026-05-03", "extra"}, want: "publish: unexpected positional arguments"},
{name: "force flag", args: []string{"publish", "--force"}, want: "publish: invalid flags: flag provided but not defined: -force"},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(tc.args, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), tc.want) {
t.Fatalf("stderr = %q, want %q", stderr.String(), tc.want)
}
})
}
}
func TestExecutePublishUnknownArtifactFailsValidation(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"publish", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "unknown_artifact"},
&stdout,
&stderr,
)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), `publish: --artifacts includes unknown artifact "unknown_artifact"`) {
t.Fatalf("stderr = %q, want unknown-artifact validation error", stderr.String())
}
}
func TestExecutePublishMissingConfigUsesRunStageLoadingPath(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"publish", "2026-05-03"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "publish: no pipeline config path provided and no default pipeline config found; searched:") {
t.Fatalf("stderr = %q, want pipeline discovery error", stderr.String())
}
}
func TestExecuteUsageIncludesAnalyzeAndPublish(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(nil, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "analyze") {
t.Fatalf("stderr = %q, want usage to include analyze", stderr.String())
}
if !strings.Contains(stderr.String(), "publish") {
t.Fatalf("stderr = %q, want usage to include publish", stderr.String())
}
}
func writeValidConfigFilesWithScriptoriumArtifacts(t *testing.T, workspaceRoot string) (string, string, string) {
t.Helper()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
f, err := os.OpenFile(pipelinePath, os.O_APPEND|os.O_WRONLY, 0)
if err != nil {
t.Fatalf("open pipeline config for append: %v", err)
}
defer f.Close()
extra := `
scriptorium:
binary: scriptorium
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
output_path: artifacts/session_recap.md
player_handout:
enabled: true
prompt_id: dnd.player_handout
output_path: artifacts/player_handout.md
depends_on:
- session_recap
inputs:
recap:
source: narratio.artifact.session_recap
required: true
`
if _, err := f.WriteString(extra); err != nil {
t.Fatalf("append scriptorium config: %v", err)
}
return pipelinePath, campaignPath, sessionPath
}

View File

@@ -0,0 +1,132 @@
package app
import (
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestArtifactSelectionFlagNormalize(t *testing.T) {
tests := []struct {
name string
inputs []string
want []string
wantErr string
}{
{
name: "single value",
inputs: []string{"session_recap"},
want: []string{"session_recap"},
},
{
name: "repeatable and comma separated values are deduped and sorted",
inputs: []string{"session_recap,player_handout", "session_recap"},
want: []string{"player_handout", "session_recap"},
},
{
name: "empty token fails",
inputs: []string{"session_recap,"},
wantErr: "artifact names must be non-empty",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
var flag artifactSelectionFlag
for _, in := range tt.inputs {
if err := flag.Set(in); err != nil {
t.Fatalf("Set(%q) error = %v", in, err)
}
}
got, err := flag.Normalize()
if tt.wantErr != "" {
if err == nil {
t.Fatalf("Normalize() error = nil, want %q", tt.wantErr)
}
if err.Error() != tt.wantErr {
t.Fatalf("Normalize() error = %q, want %q", err.Error(), tt.wantErr)
}
return
}
if err != nil {
t.Fatalf("Normalize() error = %v", err)
}
if len(got) != len(tt.want) {
t.Fatalf("Normalize() len = %d, want %d; got=%v", len(got), len(tt.want), got)
}
for i := range got {
if got[i] != tt.want[i] {
t.Fatalf("Normalize()[%d] = %q, want %q", i, got[i], tt.want[i])
}
}
})
}
}
func TestValidateSelectedArtifacts(t *testing.T) {
tests := []struct {
name string
cfg *config.Config
selected []string
wantErr string
}{
{
name: "empty selection is accepted",
cfg: &config.Config{},
selected: nil,
},
{
name: "scriptorium required when selected artifacts present",
cfg: &config.Config{Pipeline: &config.PipelineConfig{}},
selected: []string{"session_recap"},
wantErr: "--artifacts requires pipeline.scriptorium.artifacts to be configured",
},
{
name: "unknown selected artifact fails",
cfg: &config.Config{
Pipeline: &config.PipelineConfig{
Scriptorium: &config.ScriptoriumConfig{
Artifacts: map[string]config.ScriptoriumArtifactConfig{
"session_recap": {Enabled: true, PromptID: "dnd.session_recap", OutputPath: "artifacts/session_recap.md"},
},
},
},
},
selected: []string{"player_handout"},
wantErr: `--artifacts includes unknown artifact "player_handout"`,
},
{
name: "known selected artifacts are accepted",
cfg: &config.Config{
Pipeline: &config.PipelineConfig{
Scriptorium: &config.ScriptoriumConfig{
Artifacts: map[string]config.ScriptoriumArtifactConfig{
"session_recap": {Enabled: true, PromptID: "dnd.session_recap", OutputPath: "artifacts/session_recap.md"},
"player_handout": {Enabled: true, PromptID: "dnd.player_handout", OutputPath: "artifacts/player_handout.md"},
},
},
},
},
selected: []string{"player_handout", "session_recap"},
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
err := validateSelectedArtifacts(tt.cfg, tt.selected)
if tt.wantErr != "" {
if err == nil {
t.Fatalf("error = nil, want %q", tt.wantErr)
}
if err.Error() != tt.wantErr {
t.Fatalf("error = %q, want %q", err.Error(), tt.wantErr)
}
return
}
if err != nil {
t.Fatalf("error = %v, want nil", err)
}
})
}
}

View File

@@ -0,0 +1,44 @@
package app
import (
"fmt"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func resolveCampaignConfigPath(pipelineCfg *config.PipelineConfig, campaignIDFlag, campaignFileFlag string) (string, error) {
campaignID := strings.TrimSpace(campaignIDFlag)
campaignFile := strings.TrimSpace(campaignFileFlag)
if campaignID != "" && campaignFile != "" {
return "", fmt.Errorf("--campaign and --campaign-file are mutually exclusive")
}
if campaignFile != "" {
return filepath.Clean(campaignFile), nil
}
if campaignID == "" && pipelineCfg != nil {
campaignID = strings.TrimSpace(pipelineCfg.Campaigns.DefaultCampaignID)
}
if campaignID == "" {
return "", fmt.Errorf("no campaign selected; pass --campaign <id> or set pipeline.campaigns.default_campaign_id")
}
if err := validateCampaignIDToken(campaignID); err != nil {
return "", err
}
if pipelineCfg == nil || strings.TrimSpace(pipelineCfg.Campaigns.Root) == "" {
return "", fmt.Errorf("pipeline.campaigns.root is required to select campaign %q", campaignID)
}
return filepath.Clean(filepath.Join(pipelineCfg.Campaigns.Root, campaignID, "campaign.yml")), nil
}
func validateCampaignIDToken(campaignID string) error {
if filepath.IsAbs(campaignID) ||
strings.Contains(campaignID, "/") ||
strings.Contains(campaignID, `\`) ||
campaignID == "." ||
campaignID == ".." {
return fmt.Errorf("campaign id %q must be a single path segment", campaignID)
}
return nil
}

View File

@@ -0,0 +1,84 @@
package app
import (
"path/filepath"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestResolveCampaignConfigPathCampaignFileWins(t *testing.T) {
explicit := filepath.Join(t.TempDir(), "custom-campaign.yml")
got, err := resolveCampaignConfigPath(&config.PipelineConfig{}, "", explicit)
if err != nil {
t.Fatalf("resolveCampaignConfigPath() error = %v", err)
}
if got != explicit {
t.Fatalf("path = %q, want explicit path %q", got, explicit)
}
}
func TestResolveCampaignConfigPathUsesSelectedCampaignID(t *testing.T) {
dir := t.TempDir()
pipelineCfg := &config.PipelineConfig{}
pipelineCfg.Campaigns.Root = dir
got, err := resolveCampaignConfigPath(pipelineCfg, "icewind", "")
if err != nil {
t.Fatalf("resolveCampaignConfigPath() error = %v", err)
}
want := filepath.Join(dir, "icewind", "campaign.yml")
if got != filepath.Clean(want) {
t.Fatalf("path = %q, want %q", got, filepath.Clean(want))
}
}
func TestResolveCampaignConfigPathUsesDefaultCampaignID(t *testing.T) {
dir := t.TempDir()
pipelineCfg := &config.PipelineConfig{}
pipelineCfg.Campaigns.Root = dir
pipelineCfg.Campaigns.DefaultCampaignID = "dilfs"
got, err := resolveCampaignConfigPath(pipelineCfg, "", "")
if err != nil {
t.Fatalf("resolveCampaignConfigPath() error = %v", err)
}
want := filepath.Join(dir, "dilfs", "campaign.yml")
if got != filepath.Clean(want) {
t.Fatalf("path = %q, want %q", got, filepath.Clean(want))
}
}
func TestResolveCampaignConfigPathRejectsCampaignIDAndFile(t *testing.T) {
_, err := resolveCampaignConfigPath(&config.PipelineConfig{}, "dilfs", filepath.Join(t.TempDir(), "campaign.yml"))
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "mutually exclusive") {
t.Fatalf("error = %q, want mutual exclusion", err.Error())
}
}
func TestResolveCampaignConfigPathRequiresCampaignSelection(t *testing.T) {
_, err := resolveCampaignConfigPath(&config.PipelineConfig{}, "", "")
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "no campaign selected") {
t.Fatalf("error = %q, want missing selection guidance", err.Error())
}
}
func TestResolveCampaignConfigPathRejectsPathLikeCampaignID(t *testing.T) {
pipelineCfg := &config.PipelineConfig{}
pipelineCfg.Campaigns.Root = t.TempDir()
_, err := resolveCampaignConfigPath(pipelineCfg, "../icewind", "")
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "single path segment") {
t.Fatalf("error = %q, want path segment guidance", err.Error())
}
}

347
internal/app/clean.go Normal file
View File

@@ -0,0 +1,347 @@
package app
import (
"context"
"flag"
"fmt"
"io"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
// Clean removes local workspace/spool state while preserving durable cache
// state unless cache cleanup is explicitly requested.
func Clean(ctx context.Context, args []string, out io.Writer) error {
positionalSessionID, args := pullLeadingSessionID(args)
fs := flag.NewFlagSet("clean", flag.ContinueOnError)
fs.SetOutput(io.Discard)
var flags commonConfigFlags
var all bool
var dryRun bool
var clearCache bool
addCommonConfigFlags(fs, &flags)
fs.BoolVar(&all, "all", false, "clean all local session work/spool state")
fs.BoolVar(&dryRun, "dry-run", false, "print cleanup targets without deleting")
fs.BoolVar(&clearCache, "clear-cache", false, "also clear durable S3 audio cache entries")
if err := fs.Parse(args); err != nil {
return fmt.Errorf("clean: invalid flags: %w", err)
}
if positionalSessionID == "" {
if err := applyParsedSessionIDArg("clean", fs, &flags.sessionID); err != nil {
return err
}
} else {
if fs.NArg() != 0 {
return fmt.Errorf("clean: unexpected positional arguments")
}
if err := applyPositionalSessionID("clean", positionalSessionID, &flags.sessionID); err != nil {
return err
}
}
if all {
return cleanAllLocal(flags, dryRun, clearCache, out)
}
return cleanSession(ctx, flags, dryRun, clearCache, out)
}
func cleanSession(ctx context.Context, flags commonConfigFlags, dryRun, clearCache bool, out io.Writer) error {
if strings.TrimSpace(flags.sessionID) == "" {
return fmt.Errorf("clean: session_id is required unless --all is set")
}
cfg, err := loadCommandConfig(ctx, flags.pipelinePath, flags.campaignPath, flags.campaignFilePath, flags.sessionPath, flags.sessionOptions())
if err != nil {
return fmt.Errorf("clean: %w", err)
}
if cfg == nil || cfg.Pipeline == nil || cfg.Session == nil {
return fmt.Errorf("clean: resolved pipeline and session config are required")
}
campaign := strings.TrimSpace(cfg.Session.Campaign)
sessionID := strings.TrimSpace(cfg.Session.SessionID)
if campaign == "" || sessionID == "" {
return fmt.Errorf("clean: campaign and session_id are required")
}
if dryRun {
fmt.Fprintf(out, "Clean plan for %s/%s\n", campaign, sessionID)
} else {
fmt.Fprintf(out, "Cleaned %s/%s\n", campaign, sessionID)
}
workDir := artifacts.SessionWorkDirForCampaign(cfg.Pipeline.Workspace.Root, campaign, sessionID)
spoolDir := artifacts.SessionSpoolDir(cfg.Pipeline.Spool.Root, campaign, sessionID)
if err := reportCleanScopedDir(out, cfg.Pipeline.Workspace.Root, workDir, "clean.workspace.session", dryRun); err != nil {
return fmt.Errorf("clean: %w", err)
}
if err := reportCleanScopedDir(out, cfg.Pipeline.Spool.Root, spoolDir, "clean.spool.session", dryRun); err != nil {
return fmt.Errorf("clean: %w", err)
}
if clearCache {
if err := cleanSessionAudioCache(ctx, cfg, dryRun, out); err != nil {
return fmt.Errorf("clean: %w", err)
}
} else {
fmt.Fprintln(out, "Cache: preserved")
}
return nil
}
func cleanAllLocal(flags commonConfigFlags, dryRun, clearCache bool, out io.Writer) error {
if strings.TrimSpace(flags.campaignPath) != "" ||
strings.TrimSpace(flags.campaignFilePath) != "" ||
strings.TrimSpace(flags.sessionPath) != "" ||
strings.TrimSpace(flags.sessionID) != "" ||
strings.TrimSpace(flags.previousSessionID) != "" {
return fmt.Errorf("clean: --all cannot be combined with --campaign, --campaign-file, --session, a session_id, or --previous-session-id")
}
resolvedPipelinePath, err := resolvePipelineConfigPath(flags.pipelinePath)
if err != nil {
return fmt.Errorf("clean: %w", err)
}
pipelineCfg, err := config.LoadPipeline(resolvedPipelinePath)
if err != nil {
return fmt.Errorf("clean: %w", err)
}
if dryRun {
fmt.Fprintln(out, "Clean plan for all local sessions")
} else {
fmt.Fprintln(out, "Cleaned all local sessions")
}
workRoot := filepath.Join(pipelineCfg.Workspace.Root, config.PathWorkDirSegment)
if err := reportCleanScopedDir(out, pipelineCfg.Workspace.Root, workRoot, "clean.workspace.all", dryRun); err != nil {
return fmt.Errorf("clean: %w", err)
}
if err := reportCleanRootChildren(out, pipelineCfg.Spool.Root, "clean.spool.all", dryRun); err != nil {
return fmt.Errorf("clean: %w", err)
}
if clearCache {
if err := cleanAllAudioCache(pipelineCfg, dryRun, out); err != nil {
return fmt.Errorf("clean: %w", err)
}
} else {
fmt.Fprintln(out, "Cache: preserved")
}
return nil
}
func reportCleanScopedDir(out io.Writer, root, target, policy string, dryRun bool) error {
dir, err := validateScopedDir(root, target, policy)
if err != nil {
return err
}
if dryRun {
if dir.Exists {
fmt.Fprintf(out, "Would delete: %s\n", dir.TargetAbs)
} else {
fmt.Fprintf(out, "Would skip missing: %s\n", dir.TargetAbs)
}
return nil
}
if !dir.Exists {
fmt.Fprintf(out, "Missing: %s\n", dir.TargetAbs)
return nil
}
if err := os.RemoveAll(dir.TargetAbs); err != nil {
return fmt.Errorf("cleanup policy %s: remove %q: %w", policy, dir.TargetAbs, err)
}
fmt.Fprintf(out, "Deleted: %s\n", dir.TargetAbs)
return nil
}
func reportCleanRootChildren(out io.Writer, root, policy string, dryRun bool) error {
rootAbs, entries, err := cleanableRootChildren(root, policy)
if err != nil {
return err
}
if len(entries) == 0 {
if dryRun {
fmt.Fprintf(out, "Would skip empty: %s\n", rootAbs)
} else {
fmt.Fprintf(out, "Empty: %s\n", rootAbs)
}
return nil
}
for _, entry := range entries {
if dryRun {
fmt.Fprintf(out, "Would delete: %s\n", entry)
continue
}
if err := os.RemoveAll(entry); err != nil {
return fmt.Errorf("cleanup policy %s: remove %q: %w", policy, entry, err)
}
fmt.Fprintf(out, "Deleted: %s\n", entry)
}
return nil
}
func cleanableRootChildren(root, policy string) (string, []string, error) {
cleanRoot := strings.TrimSpace(root)
if cleanRoot == "" {
return "", nil, fmt.Errorf("cleanup policy %s: root path is required", policy)
}
rootAbs, err := filepath.Abs(cleanRoot)
if err != nil {
return "", nil, fmt.Errorf("cleanup policy %s: resolve root %q: %w", policy, cleanRoot, err)
}
info, err := os.Lstat(rootAbs)
if err != nil {
if os.IsNotExist(err) {
return rootAbs, nil, nil
}
return "", nil, fmt.Errorf("cleanup policy %s: stat root %q: %w", policy, rootAbs, err)
}
if info.Mode()&os.ModeSymlink != 0 {
return "", nil, fmt.Errorf("cleanup policy %s: refusing to clean symlink root %q", policy, rootAbs)
}
if !info.IsDir() {
return "", nil, fmt.Errorf("cleanup policy %s: root %q is not a directory", policy, rootAbs)
}
entries, err := os.ReadDir(rootAbs)
if err != nil {
return "", nil, fmt.Errorf("cleanup policy %s: read root %q: %w", policy, rootAbs, err)
}
out := make([]string, 0, len(entries))
for _, entry := range entries {
path := filepath.Join(rootAbs, entry.Name())
info, err := os.Lstat(path)
if err != nil {
return "", nil, fmt.Errorf("cleanup policy %s: stat child %q: %w", policy, path, err)
}
if info.Mode()&os.ModeSymlink != 0 {
return "", nil, fmt.Errorf("cleanup policy %s: refusing to delete symlink path %q", policy, path)
}
out = append(out, path)
}
return rootAbs, out, nil
}
func cleanSessionAudioCache(ctx context.Context, cfg *config.Config, dryRun bool, out io.Writer) error {
if cfg.Session.Inputs.AudioS3 == nil {
fmt.Fprintln(out, "Cache: skipped (session does not use audio_s3)")
return nil
}
if cfg.Pipeline.Storage.S3 == nil || strings.TrimSpace(cfg.Pipeline.Storage.S3.Bucket) == "" {
return fmt.Errorf("clear cache requires pipeline.storage.s3.bucket")
}
store, err := newCommandObjectStore(ctx, cfg, nil)
if err != nil {
return fmt.Errorf("initialize object store for cache cleanup: %w", err)
}
sessionPrefix := artifacts.S3SessionPrefix(cfg.Pipeline.Storage.S3.RootPrefix, cfg.Session.Campaign, cfg.Session.SessionID)
audioPrefix := artifacts.S3AudioPrefix(sessionPrefix, cfg.Session.Inputs.AudioS3.Prefix)
objects, err := store.List(ctx, audioPrefix)
if err != nil {
return fmt.Errorf("list s3 audio objects under %q: %w", audioPrefix, err)
}
count := 0
for _, obj := range objects {
key := strings.TrimSpace(obj.Key)
if key == "" || strings.HasSuffix(key, "/") || !cleanIsFlac(key) {
continue
}
cachePath, err := artifacts.S3AudioCachePath(cfg.Pipeline.Cache.Root, cfg.Pipeline.Storage.S3.Bucket, key)
if err != nil {
return err
}
deleted, err := reportCleanScopedFile(out, cfg.Pipeline.Cache.Root, cachePath, "clean.cache.session", dryRun)
if err != nil {
return err
}
if deleted {
count++
}
}
if count == 0 {
fmt.Fprintf(out, "Cache: no cached S3 audio files found for %s\n", audioPrefix)
}
return nil
}
func cleanAllAudioCache(cfg *config.PipelineConfig, dryRun bool, out io.Writer) error {
if cfg.Storage.S3 == nil || strings.TrimSpace(cfg.Storage.S3.Bucket) == "" {
return fmt.Errorf("clear cache requires pipeline.storage.s3.bucket")
}
namespaceDir, err := artifacts.S3AudioCacheNamespaceDir(cfg.Cache.Root, cfg.Storage.S3.Bucket, cfg.Storage.S3.RootPrefix)
if err != nil {
return err
}
return reportCleanScopedDir(out, cfg.Cache.Root, namespaceDir, "clean.cache.all", dryRun)
}
func reportCleanScopedFile(out io.Writer, root, target, policy string, dryRun bool) (bool, error) {
file, err := validateScopedFile(root, target, policy)
if err != nil {
return false, err
}
if dryRun {
if file.Exists {
fmt.Fprintf(out, "Would delete cache file: %s\n", file.TargetAbs)
return true, nil
}
fmt.Fprintf(out, "Would skip missing cache file: %s\n", file.TargetAbs)
return false, nil
}
if !file.Exists {
fmt.Fprintf(out, "Missing cache file: %s\n", file.TargetAbs)
return false, nil
}
if err := os.Remove(file.TargetAbs); err != nil {
return false, fmt.Errorf("cleanup policy %s: remove %q: %w", policy, file.TargetAbs, err)
}
fmt.Fprintf(out, "Deleted cache file: %s\n", file.TargetAbs)
return true, nil
}
func validateScopedFile(root, target, policy string) (scopedDir, error) {
cleanRoot := strings.TrimSpace(root)
cleanTarget := strings.TrimSpace(target)
if cleanRoot == "" {
return scopedDir{}, fmt.Errorf("cleanup policy %s: root path is required", policy)
}
if cleanTarget == "" {
return scopedDir{}, fmt.Errorf("cleanup policy %s: target path is required", policy)
}
rootAbs, err := filepath.Abs(cleanRoot)
if err != nil {
return scopedDir{}, fmt.Errorf("cleanup policy %s: resolve root %q: %w", policy, cleanRoot, err)
}
targetAbs, err := filepath.Abs(cleanTarget)
if err != nil {
return scopedDir{}, fmt.Errorf("cleanup policy %s: resolve target %q: %w", policy, cleanTarget, err)
}
rel, err := filepath.Rel(rootAbs, targetAbs)
if err != nil {
return scopedDir{}, fmt.Errorf("cleanup policy %s: relative path from %q to %q: %w", policy, rootAbs, targetAbs, err)
}
if rel == "." {
return scopedDir{}, fmt.Errorf("cleanup policy %s: refusing to delete root directory %q", policy, rootAbs)
}
if rel == ".." || strings.HasPrefix(rel, ".."+string(filepath.Separator)) {
return scopedDir{}, fmt.Errorf("cleanup policy %s: refusing to delete path outside root: root=%q target=%q", policy, rootAbs, targetAbs)
}
info, err := os.Lstat(targetAbs)
if err != nil {
if os.IsNotExist(err) {
return scopedDir{RootAbs: rootAbs, TargetAbs: targetAbs, Exists: false}, nil
}
return scopedDir{}, fmt.Errorf("cleanup policy %s: stat target %q: %w", policy, targetAbs, err)
}
if info.Mode()&os.ModeSymlink != 0 {
return scopedDir{}, fmt.Errorf("cleanup policy %s: refusing to delete symlink path %q", policy, targetAbs)
}
if info.IsDir() {
return scopedDir{}, fmt.Errorf("cleanup policy %s: target %q is a directory", policy, targetAbs)
}
return scopedDir{RootAbs: rootAbs, TargetAbs: targetAbs, Exists: true}, nil
}
func cleanIsFlac(path string) bool {
return strings.EqualFold(filepath.Ext(path), ".flac")
}

255
internal/app/clean_test.go Normal file
View File

@@ -0,0 +1,255 @@
package app
import (
"bytes"
"os"
"path/filepath"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
)
func TestExecuteCleanSessionDeletesWorkAndSpoolButPreservesCache(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
workDir := artifacts.SessionWorkDirForCampaign(workspaceRoot, "sample-campaign", "2026-05-03")
spoolDir := artifacts.SessionSpoolDir(filepath.Join(workspaceRoot, "spool"), "sample-campaign", "2026-05-03")
cachePath, err := artifacts.S3AudioCachePath(filepath.Join(workspaceRoot, "cache"), "test-bucket", "dnd/campaigns/sample-campaign/sessions/2026-05-03/audio/alice.flac")
if err != nil {
t.Fatalf("S3AudioCachePath() error = %v", err)
}
mustWriteTestFile(t, filepath.Join(workDir, "manifest.json"), "{}")
mustWriteTestFile(t, filepath.Join(spoolDir, "run-1", "audio", "alice.flac"), "audio")
mustWriteTestFile(t, cachePath, "cached-audio")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
cleanAssertMissing(t, workDir)
cleanAssertMissing(t, spoolDir)
cleanAssertExists(t, cachePath)
if !strings.Contains(stdout.String(), "Cache: preserved") {
t.Fatalf("stdout = %q, want cache preserved", stdout.String())
}
}
func TestExecuteCleanSessionDryRunDeletesNothing(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
workDir := artifacts.SessionWorkDirForCampaign(workspaceRoot, "sample-campaign", "2026-05-03")
spoolDir := artifacts.SessionSpoolDir(filepath.Join(workspaceRoot, "spool"), "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workDir, "manifest.json"), "{}")
mustWriteTestFile(t, filepath.Join(spoolDir, "run-1", "audio", "alice.flac"), "audio")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--dry-run"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
cleanAssertExists(t, workDir)
cleanAssertExists(t, spoolDir)
if !strings.Contains(stdout.String(), "Would delete:") {
t.Fatalf("stdout = %q, want dry-run delete plan", stdout.String())
}
}
func TestExecuteCleanMissingSessionPathsSucceeds(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "Missing:") {
t.Fatalf("stdout = %q, want missing path output", stdout.String())
}
}
func TestExecuteCleanSessionClearCacheRemovesOnlyS3AudioCache(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
if err := os.WriteFile(sessionPath, []byte(`session_id: 2026-05-03
inputs:
audio_s3:
prefix: audio/
`), 0o644); err != nil {
t.Fatalf("write session: %v", err)
}
audioKey := "dnd/campaigns/sample-campaign/sessions/2026-05-03/audio/alice.flac"
fake := &storage.FakeBackend{}
fake.SeedObject(storage.FakeObject{Key: audioKey, Data: []byte("audio")})
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{sessionPath})
cacheRoot := filepath.Join(workspaceRoot, "cache")
cachePath, err := artifacts.S3AudioCachePath(cacheRoot, "test-bucket", audioKey)
if err != nil {
t.Fatalf("S3AudioCachePath() error = %v", err)
}
otherCachePath, err := artifacts.S3AudioCachePath(cacheRoot, "test-bucket", "dnd/campaigns/other/sessions/2026-05-03/audio/bob.flac")
if err != nil {
t.Fatalf("S3AudioCachePath() error = %v", err)
}
mustWriteTestFile(t, cachePath, "cached-audio")
mustWriteTestFile(t, otherCachePath, "other-audio")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--clear-cache"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
cleanAssertMissing(t, cachePath)
cleanAssertExists(t, otherCachePath)
if storeInitCalls != 1 {
t.Fatalf("object store init calls = %d, want 1", storeInitCalls)
}
}
func TestExecuteCleanLocalAudioClearCacheIsNoop(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--clear-cache"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "Cache: skipped (session does not use audio_s3)") {
t.Fatalf("stdout = %q, want local audio cache no-op", stdout.String())
}
}
func TestExecuteCleanAllDeletesWorkAndSpoolContentsButPreservesCache(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, _, _ := writeValidConfigFiles(t, workspaceRoot)
workRoot := filepath.Join(workspaceRoot, "work")
spoolRoot := filepath.Join(workspaceRoot, "spool")
cachePath := filepath.Join(workspaceRoot, "cache", "keep.txt")
mustWriteTestFile(t, filepath.Join(workRoot, "sample-campaign", "2026-05-03", "manifest.json"), "{}")
mustWriteTestFile(t, filepath.Join(spoolRoot, "sample-campaign", "2026-05-03", "run-1", "audio", "alice.flac"), "audio")
mustWriteTestFile(t, cachePath, "cache")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "--config", pipelinePath, "--all"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
cleanAssertMissing(t, workRoot)
cleanAssertExists(t, spoolRoot)
cleanAssertMissing(t, filepath.Join(spoolRoot, "sample-campaign"))
cleanAssertExists(t, cachePath)
}
func TestExecuteCleanAllClearCacheRemovesS3AudioNamespaceOnly(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, _, _ := writeValidConfigFiles(t, workspaceRoot)
cacheRoot := filepath.Join(workspaceRoot, "cache")
audioCachePath, err := artifacts.S3AudioCachePath(cacheRoot, "test-bucket", "dnd/campaigns/sample-campaign/sessions/2026-05-03/audio/alice.flac")
if err != nil {
t.Fatalf("S3AudioCachePath() error = %v", err)
}
otherCachePath, err := artifacts.S3AudioCachePath(cacheRoot, "test-bucket", "other-root/campaigns/sample-campaign/sessions/2026-05-03/audio/alice.flac")
if err != nil {
t.Fatalf("S3AudioCachePath() error = %v", err)
}
mustWriteTestFile(t, audioCachePath, "cached-audio")
mustWriteTestFile(t, otherCachePath, "other-cache")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "--config", pipelinePath, "--all", "--clear-cache"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
cleanAssertMissing(t, audioCachePath)
cleanAssertExists(t, otherCachePath)
}
func TestExecuteCleanAllRejectsSessionScopedFlags(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--all"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "--all cannot be combined") {
t.Fatalf("stderr = %q, want --all conflict", stderr.String())
}
}
func TestCleanRequiresSessionID(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "session_id is required unless --all is set") {
t.Fatalf("stderr = %q, want missing session-id", stderr.String())
}
}
func TestCleanRejectsUnsafeTargets(t *testing.T) {
root := t.TempDir()
outside := t.TempDir()
if err := reportCleanScopedDir(&bytes.Buffer{}, root, filepath.Join(outside, "target"), "test.outside", false); err == nil {
t.Fatal("outside target error = nil, want error")
}
if err := reportCleanScopedDir(&bytes.Buffer{}, root, root, "test.root", false); err == nil {
t.Fatal("root target error = nil, want error")
}
filePath := filepath.Join(root, "file.txt")
mustWriteTestFile(t, filePath, "file")
if err := reportCleanScopedDir(&bytes.Buffer{}, root, filePath, "test.file", false); err == nil {
t.Fatal("file target error = nil, want error")
}
symlinkPath := filepath.Join(root, "link")
if err := os.Symlink(filepath.Join(root, "missing"), symlinkPath); err != nil {
t.Fatalf("Symlink() error = %v", err)
}
if err := reportCleanScopedDir(&bytes.Buffer{}, root, symlinkPath, "test.symlink", false); err == nil {
t.Fatal("symlink target error = nil, want error")
}
}
func TestClearIsNotCommandAlias(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clear"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), `unknown command: "clear"`) {
t.Fatalf("stderr = %q, want unknown clear command", stderr.String())
}
}
func cleanAssertExists(t *testing.T, path string) {
t.Helper()
if _, err := os.Stat(path); err != nil {
t.Fatalf("expected %q to exist: %v", path, err)
}
}
func cleanAssertMissing(t *testing.T, path string) {
t.Helper()
if _, err := os.Stat(path); !os.IsNotExist(err) {
t.Fatalf("expected %q to be missing, stat err=%v", path, err)
}
}

View File

@@ -7,7 +7,7 @@ import (
"strings" "strings"
) )
var supportedCommands = []string{"run", "plan", "status", "resume", "run-stage"} var supportedCommands = []string{"run", "run-stage", "resume", "analyze", "publish", "clean", "session"}
// Execute dispatches CLI commands and returns a process exit code. // Execute dispatches CLI commands and returns a process exit code.
func Execute(args []string, stdout, stderr io.Writer) int { func Execute(args []string, stdout, stderr io.Writer) int {
@@ -24,14 +24,18 @@ func Execute(args []string, stdout, stderr io.Writer) int {
switch cmd { switch cmd {
case "run": case "run":
err = Run(ctx, cmdArgs, stdout) err = Run(ctx, cmdArgs, stdout)
case "plan":
err = Plan(ctx, cmdArgs, stdout)
case "status":
err = Status(ctx, cmdArgs, stdout)
case "resume": case "resume":
err = Resume(ctx, cmdArgs, stdout) err = Resume(ctx, cmdArgs, stdout)
case "run-stage": case "run-stage":
err = RunStage(ctx, cmdArgs, stdout) err = RunStage(ctx, cmdArgs, stdout)
case "analyze":
err = Analyze(ctx, cmdArgs, stdout)
case "publish":
err = Publish(ctx, cmdArgs, stdout)
case "session":
err = Session(ctx, cmdArgs, stdout)
case "clean":
err = Clean(ctx, cmdArgs, stdout)
default: default:
fmt.Fprintf(stderr, "unknown command: %q\n\n", cmd) fmt.Fprintf(stderr, "unknown command: %q\n\n", cmd)
printUsage(stderr) printUsage(stderr)

View File

@@ -24,19 +24,18 @@ func TestExecuteValidCommands(t *testing.T) {
})) }))
defer srv.Close() defer srv.Close()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL) pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL)
manifestPath := writeManifestPathForExecute(t)
cases := []struct { cases := []struct {
name string name string
args []string args []string
wantOut string wantOut string
}{ }{
{name: "run", args: []string{"run", "--config", pipelinePath, "--session", sessionPath}, wantOut: "narratio run: session 2026-05-03; executed=9 skipped=0; manifest="}, {name: "run", args: []string{"run", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "narratio run: session 2026-05-03; executed=9 skipped=0; manifest="},
{name: "plan", args: []string{"plan", "--config", pipelinePath, "--session", sessionPath}, wantOut: "prepare: skip\ntranscribe: skip\nmerge: skip\npolish: skip\nnormalize: skip\ntrim: skip\nanalyze: skip\narchive: skip\nnotify: skip"}, {name: "session plan", args: []string{"session", "plan", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "prepare: skip\ntranscribe: skip\nmerge: skip\npolish: skip\nnormalize: skip\ntrim: skip\nanalyze: skip\narchive: skip\nnotify: skip"},
{name: "status", args: []string{"status", "--manifest", manifestPath}, wantOut: "session_id: 2026-05-03"}, {name: "session status", args: []string{"session", "status", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "Session: 2026-05-03"},
{name: "resume", args: []string{"resume", "--config", pipelinePath, "--session", sessionPath}, wantOut: "narratio resume: session 2026-05-03 has no remaining stages"}, {name: "resume", args: []string{"resume", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "narratio resume: session 2026-05-03 has no remaining stages"},
{name: "run-stage", args: []string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "polish"}, wantOut: "narratio run-stage: stage=polish executed=0 skipped=1 force=false; manifest="}, {name: "run-stage", args: []string{"run-stage", "polish", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "narratio run-stage: stage=polish executed=0 skipped=1 force=false; manifest="},
} }
for _, tc := range cases { for _, tc := range cases {
@@ -64,13 +63,13 @@ func TestExecuteMissingRequiredFlags(t *testing.T) {
args []string args []string
want string want string
}{ }{
{name: "run missing flags", args: []string{"run"}, want: "run: --session is required"}, {name: "run missing session", args: []string{"run"}, want: "run: session_id is required"},
{name: "plan missing flags", args: []string{"plan"}, want: "plan: --session is required"}, {name: "plan old top-level removed", args: []string{"plan"}, want: `unknown command: "plan"`},
{name: "status missing flags", args: []string{"status"}, want: "status: --manifest is required"}, {name: "status old top-level removed", args: []string{"status"}, want: `unknown command: "status"`},
{name: "resume missing flags", args: []string{"resume"}, want: "resume: --session is required"}, {name: "resume missing session", args: []string{"resume"}, want: "resume: session_id is required"},
{name: "run-stage missing name", args: []string{"run-stage", "--config", "a", "--session", "b"}, want: "run-stage: expected exactly one stage name"}, {name: "run-stage missing name", args: []string{"run-stage", "--config", "a", "--session", "b"}, want: "run-stage: expected stage name and session_id"},
{name: "run-stage missing config flags", args: []string{"run-stage", "polish"}, want: "run-stage: --session is required"}, {name: "run-stage missing session", args: []string{"run-stage", "polish"}, want: "run-stage: expected stage name and session_id"},
{name: "run missing config uses defaults", args: []string{"run", "--session", "session.yml"}, want: "run: no pipeline config path provided and no default pipeline config found; searched:"}, {name: "run missing config uses defaults", args: []string{"run", "2026-05-03", "--session", "session.yml"}, want: "run: no pipeline config path provided and no default pipeline config found; searched:"},
} }
for _, tc := range cases { for _, tc := range cases {
@@ -94,12 +93,12 @@ func TestExecuteMissingRequiredFlags(t *testing.T) {
func TestExecuteRunStageUnknownFails(t *testing.T) { func TestExecuteRunStageUnknownFails(t *testing.T) {
workspaceRoot := t.TempDir() workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, "https://example.com/transcribe") pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot, "https://example.com/transcribe")
var stdout bytes.Buffer var stdout bytes.Buffer
var stderr bytes.Buffer var stderr bytes.Buffer
code := Execute([]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "unknown"}, &stdout, &stderr) code := Execute([]string{"run-stage", "unknown", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 { if code == 0 {
t.Fatal("exit code = 0, want non-zero") t.Fatal("exit code = 0, want non-zero")
} }
@@ -110,14 +109,14 @@ func TestExecuteRunStageUnknownFails(t *testing.T) {
func TestExecuteRunStageNormalizeIsAccepted(t *testing.T) { func TestExecuteRunStageNormalizeIsAccepted(t *testing.T) {
workspaceRoot := t.TempDir() workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, "https://example.com/transcribe") pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot, "https://example.com/transcribe")
workRoot := filepath.Join(workspaceRoot, "work", "2026-05-03") workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "processed.json"), `{"segments":[{"id":1}]}`) mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "polished.json"), `{"segments":[{"id":1}]}`)
var stdout bytes.Buffer var stdout bytes.Buffer
var stderr bytes.Buffer var stderr bytes.Buffer
code := Execute([]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "normalize"}, &stdout, &stderr) code := Execute([]string{"run-stage", "normalize", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String()) t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
} }
@@ -136,19 +135,19 @@ func TestExecuteRunStageTranscribeUsesConfiguredWhisperXServer(t *testing.T) {
})) }))
defer srv.Close() defer srv.Close()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL) pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL)
var stdout bytes.Buffer var stdout bytes.Buffer
var stderr bytes.Buffer var stderr bytes.Buffer
code := Execute([]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "prepare"}, &stdout, &stderr) code := Execute([]string{"run-stage", "prepare", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("prepare exit code = %d, want 0; stderr=%q", code, stderr.String()) t.Fatalf("prepare exit code = %d, want 0; stderr=%q", code, stderr.String())
} }
stdout.Reset() stdout.Reset()
stderr.Reset() stderr.Reset()
code = Execute([]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "--force", "transcribe"}, &stdout, &stderr) code = Execute([]string{"run-stage", "transcribe", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--force"}, &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("transcribe exit code = %d, want 0; stderr=%q", code, stderr.String()) t.Fatalf("transcribe exit code = %d, want 0; stderr=%q", code, stderr.String())
} }
@@ -156,7 +155,7 @@ func TestExecuteRunStageTranscribeUsesConfiguredWhisperXServer(t *testing.T) {
t.Fatal("expected whisperx server to be called at least once") t.Fatal("expected whisperx server to be called at least once")
} }
outPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "transcripts", "raw", "alice.json") outPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "transcripts", "raw", "alice.json")
data, err := os.ReadFile(outPath) data, err := os.ReadFile(outPath)
if err != nil { if err != nil {
t.Fatalf("ReadFile(%q): %v", outPath, err) t.Fatalf("ReadFile(%q): %v", outPath, err)
@@ -188,6 +187,7 @@ func TestExecuteRunStagePolishLoadsCredentialFromSecretsDir(t *testing.T) {
t.Setenv("GO_WANT_APP_AUDITA_HELPER", "1") t.Setenv("GO_WANT_APP_AUDITA_HELPER", "1")
pipelinePath := filepath.Join(configDir, "pipeline.yml") pipelinePath := filepath.Join(configDir, "pipeline.yml")
campaignPath := writeAppTestCampaignConfig(t, configDir)
sessionPath := filepath.Join(configDir, "session.yml") sessionPath := filepath.Join(configDir, "session.yml")
pipelineYAML := `workspace: pipelineYAML := `workspace:
root: ` + workspaceRoot + ` root: ` + workspaceRoot + `
@@ -202,12 +202,11 @@ seriatim:
audita: audita:
binary: ` + auditaBinary + ` binary: ` + auditaBinary + `
llm_api_key_env: OPENROUTER_API_KEY llm_api_key_env: OPENROUTER_API_KEY
analyzer:
timeout: 20m
notification: notification:
timeout: 10s timeout: 10s
` `
sessionYAML := `session_id: ` + sessionID + ` sessionYAML := `session_id: ` + sessionID + `
campaign: sample-campaign
inputs: inputs:
audio_dir: ./audio audio_dir: ./audio
speakers_file: ./speakers.yml speakers_file: ./speakers.yml
@@ -232,13 +231,13 @@ inputs:
_ = os.Chdir(originalWD) _ = os.Chdir(originalWD)
}) })
workRoot := filepath.Join(workspaceRoot, "work", sessionID) workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", sessionID)
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "merged.json"), `{"schema":"seriatim-intermediate","segments":[]}`) mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "base.json"), `{"schema":"seriatim-intermediate","segments":[]}`)
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "[]\n") mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "[]\n")
var stdout bytes.Buffer var stdout bytes.Buffer
var stderr bytes.Buffer var stderr bytes.Buffer
code := Execute([]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "--force", "polish"}, &stdout, &stderr) code := Execute([]string{"run-stage", "polish", sessionID, "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--force"}, &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String()) t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
} }
@@ -251,6 +250,7 @@ func TestExecuteRunFailsWhenConfiguredSecretsDirMissing(t *testing.T) {
workspaceRoot := t.TempDir() workspaceRoot := t.TempDir()
configDir := t.TempDir() configDir := t.TempDir()
pipelinePath := filepath.Join(configDir, "pipeline.yml") pipelinePath := filepath.Join(configDir, "pipeline.yml")
campaignPath := writeAppTestCampaignConfig(t, configDir)
sessionPath := filepath.Join(configDir, "session.yml") sessionPath := filepath.Join(configDir, "session.yml")
pipelineYAML := `workspace: pipelineYAML := `workspace:
@@ -265,12 +265,11 @@ seriatim:
binary: seriatim binary: seriatim
audita: audita:
binary: audita binary: audita
analyzer:
timeout: 20m
notification: notification:
timeout: 10s timeout: 10s
` `
sessionYAML := `session_id: 2026-05-03 sessionYAML := `session_id: 2026-05-03
campaign: sample-campaign
inputs: inputs:
audio_dir: ./audio audio_dir: ./audio
speakers_file: ./speakers.yml speakers_file: ./speakers.yml
@@ -286,7 +285,7 @@ inputs:
var stdout bytes.Buffer var stdout bytes.Buffer
var stderr bytes.Buffer var stderr bytes.Buffer
code := Execute([]string{"run", "--config", pipelinePath, "--session", sessionPath}, &stdout, &stderr) code := Execute([]string{"run", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 { if code == 0 {
t.Fatal("exit code = 0, want non-zero") t.Fatal("exit code = 0, want non-zero")
} }
@@ -303,16 +302,17 @@ func TestExecuteUsesDefaultPipelineConfigPathWhenConfigFlagOmitted(t *testing.T)
})) }))
defer srv.Close() defer srv.Close()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL) pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL)
originalDefaults := append([]string(nil), config.DefaultPipelineConfigSearchPaths...) originalDefaults := append([]string(nil), config.DefaultPipelineConfigSearchPaths...)
config.DefaultPipelineConfigSearchPaths = []string{pipelinePath} config.DefaultPipelineConfigSearchPaths = []string{pipelinePath}
defer func() { defer func() {
config.DefaultPipelineConfigSearchPaths = originalDefaults config.DefaultPipelineConfigSearchPaths = originalDefaults
}() }()
_ = campaignPath
var stdout bytes.Buffer var stdout bytes.Buffer
var stderr bytes.Buffer var stderr bytes.Buffer
code := Execute([]string{"run", "--session", sessionPath}, &stdout, &stderr) code := Execute([]string{"run", "2026-05-03", "--session", sessionPath}, &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String()) t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
} }
@@ -321,6 +321,86 @@ func TestExecuteUsesDefaultPipelineConfigPathWhenConfigFlagOmitted(t *testing.T)
} }
} }
func TestExecuteMissingCampaignConfigReportsRegistryPath(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
if err := os.Remove(campaignPath); err != nil {
t.Fatalf("remove campaign config: %v", err)
}
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"run", "2026-05-03", "--config", pipelinePath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if stdout.Len() != 0 {
t.Fatalf("stdout = %q, want empty", stdout.String())
}
if !strings.Contains(stderr.String(), "load campaign config") {
t.Fatalf("stderr = %q, want campaign discovery failure", stderr.String())
}
if !strings.Contains(stderr.String(), filepath.ToSlash(filepath.Join("campaigns", "sample-campaign", "campaign.yml"))) {
t.Fatalf("stderr = %q, want campaign registry path", stderr.String())
}
}
func TestExecuteUsesPipelineDefaultCampaignID(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, _, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "status", "2026-05-03", "--config", pipelinePath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "Campaign: sample-campaign") {
t.Fatalf("stdout = %q, want default campaign", stdout.String())
}
}
func TestExecuteCampaignIDSelectsRegistryCampaign(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
campaignRoot := filepath.Dir(filepath.Dir(campaignPath))
otherDir := filepath.Join(campaignRoot, "icewind")
mustWriteTestFile(t, filepath.Join(otherDir, "campaign.yml"), `campaign_id: icewind
inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`)
mustWriteTestFile(t, filepath.Join(otherDir, "speakers.yml"), "match:\n - speaker: Alice\n match: [\"alice\"]\n")
mustWriteTestFile(t, filepath.Join(otherDir, "autocorrect.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(otherDir, "glossary.yml"), "[]\n")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "status", "2026-05-03", "--config", pipelinePath, "--campaign", "icewind", "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "Campaign: icewind") {
t.Fatalf("stdout = %q, want selected campaign", stdout.String())
}
}
func TestExecuteRejectsCampaignIDAndCampaignFile(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "status", "2026-05-03", "--config", pipelinePath, "--campaign", "sample-campaign", "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "mutually exclusive") {
t.Fatalf("stderr = %q, want mutually exclusive error", stderr.String())
}
}
func TestExecuteInvalidCommand(t *testing.T) { func TestExecuteInvalidCommand(t *testing.T) {
var stdout bytes.Buffer var stdout bytes.Buffer
var stderr bytes.Buffer var stderr bytes.Buffer
@@ -357,11 +437,14 @@ func TestExecuteMissingCommand(t *testing.T) {
} }
} }
func writeValidConfigFiles(t *testing.T, workspaceRoot string, transcribeURL ...string) (string, string) { func writeValidConfigFiles(t *testing.T, workspaceRoot string, transcribeURL ...string) (string, string, string) {
t.Helper() t.Helper()
dir := t.TempDir() dir := t.TempDir()
pipelinePath := filepath.Join(dir, "pipeline.yml") pipelinePath := filepath.Join(dir, "pipeline.yml")
campaignRoot := filepath.Join(dir, "campaigns")
campaignDir := filepath.Join(campaignRoot, "sample-campaign")
campaignPath := filepath.Join(campaignDir, "campaign.yml")
sessionPath := filepath.Join(dir, "session.yml") sessionPath := filepath.Join(dir, "session.yml")
url := "https://example.com/transcribe" url := "https://example.com/transcribe"
if len(transcribeURL) > 0 && strings.TrimSpace(transcribeURL[0]) != "" { if len(transcribeURL) > 0 && strings.TrimSpace(transcribeURL[0]) != "" {
@@ -375,8 +458,20 @@ func writeValidConfigFiles(t *testing.T, workspaceRoot string, transcribeURL ...
pipelineYAML := `workspace: pipelineYAML := `workspace:
root: ` + workspaceRoot + ` root: ` + workspaceRoot + `
campaigns:
root: ` + campaignRoot + `
default_campaign_id: sample-campaign
cache:
root: ` + filepath.Join(workspaceRoot, "cache") + `
spool:
root: ` + filepath.Join(workspaceRoot, "spool") + `
storage: storage:
backend: s3 backend: s3
s3:
bucket: test-bucket
archive:
enabled: true
upload_run: false
whisperx: whisperx:
transcribe_url: ` + url + ` transcribe_url: ` + url + `
timeout: 2s timeout: 2s
@@ -391,10 +486,6 @@ seriatim:
report: true report: true
audita: audita:
binary: ` + auditaBinary + ` binary: ` + auditaBinary + `
analyzer:
timeout: 20m
artifacts:
output_dir: artifacts
notification: notification:
timeout: 10s timeout: 10s
` `
@@ -402,6 +493,9 @@ notification:
sessionYAML := `session_id: 2026-05-03 sessionYAML := `session_id: 2026-05-03
inputs: inputs:
audio_dir: ./audio audio_dir: ./audio
`
campaignYAML := `campaign_id: sample-campaign
inputs:
speakers_file: ./speakers.yml speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml glossary_file: ./glossary.yml
@@ -410,16 +504,37 @@ inputs:
if err := os.WriteFile(pipelinePath, []byte(pipelineYAML), 0o644); err != nil { if err := os.WriteFile(pipelinePath, []byte(pipelineYAML), 0o644); err != nil {
t.Fatalf("write pipeline config: %v", err) t.Fatalf("write pipeline config: %v", err)
} }
if err := os.MkdirAll(campaignDir, 0o755); err != nil {
t.Fatalf("create campaign dir: %v", err)
}
if err := os.WriteFile(campaignPath, []byte(campaignYAML), 0o644); err != nil {
t.Fatalf("write campaign config: %v", err)
}
if err := os.WriteFile(sessionPath, []byte(sessionYAML), 0o644); err != nil { if err := os.WriteFile(sessionPath, []byte(sessionYAML), 0o644); err != nil {
t.Fatalf("write session config: %v", err) t.Fatalf("write session config: %v", err)
} }
mustWriteTestFile(t, filepath.Join(dir, "speakers.yml"), "match:\n - speaker: Alice\n match: [\"alice\"]\n") mustWriteTestFile(t, filepath.Join(campaignDir, "speakers.yml"), "match:\n - speaker: Alice\n match: [\"alice\"]\n")
mustWriteTestFile(t, filepath.Join(dir, "autocorrect.yml"), "[]\n") mustWriteTestFile(t, filepath.Join(campaignDir, "autocorrect.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(dir, "glossary.yml"), "[]\n") mustWriteTestFile(t, filepath.Join(campaignDir, "glossary.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(dir, "audio", "alice.flac"), "audio-bytes") mustWriteTestFile(t, filepath.Join(dir, "audio", "alice.flac"), "audio-bytes")
return pipelinePath, sessionPath return pipelinePath, campaignPath, sessionPath
}
func writeAppTestCampaignConfig(t *testing.T, dir string) string {
t.Helper()
campaignPath := filepath.Join(dir, "campaign.yml")
campaignYAML := `campaign_id: sample-campaign
inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
if err := os.WriteFile(campaignPath, []byte(campaignYAML), 0o644); err != nil {
t.Fatalf("write campaign.yml: %v", err)
}
return campaignPath
} }
func writeManifestPathForExecute(t *testing.T) string { func writeManifestPathForExecute(t *testing.T) string {

View File

@@ -0,0 +1,157 @@
package app
import (
"context"
"fmt"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
type pipelineCampaignConfig struct {
PipelinePath string
CampaignPath string
Pipeline *config.PipelineConfig
Campaign *config.CampaignConfig
}
func loadCommandConfig(ctx context.Context, pipelineFlag, campaignFlag, campaignFileFlag, sessionFlag string, sessionOpts config.SessionLoadOptions) (*config.Config, error) {
base, err := loadPipelineCampaignConfig(pipelineFlag, campaignFlag, campaignFileFlag)
if err != nil {
return nil, err
}
if explicitSession := strings.TrimSpace(sessionFlag); explicitSession != "" {
return config.LoadWithSessionOptions(base.PipelinePath, base.CampaignPath, explicitSession, sessionOpts)
}
discoveredSession, err := discoverSessionConfigPathWithCandidates(config.DefaultSessionConfigSearchPaths)
if err != nil {
return nil, err
}
if discoveredSession.Path != "" {
return config.LoadWithSessionOptions(base.PipelinePath, base.CampaignPath, discoveredSession.Path, sessionOpts)
}
sessionID := strings.TrimSpace(sessionOpts.SessionID)
if sessionID == "" {
return nil, missingSessionConfigError(discoveredSession.Searched, "remote session loading requires a session_id")
}
sessionPrefix := artifacts.S3SessionPrefix(base.Pipeline.Storage.S3.RootPrefix, config.CampaignID(base.Campaign), sessionID)
remoteKey := artifacts.S3SessionConfigKey(sessionPrefix)
partialCfg := &config.Config{
Pipeline: base.Pipeline,
Campaign: base.Campaign,
PipelinePath: base.PipelinePath,
CampaignPath: base.CampaignPath,
}
store, err := newCommandObjectStore(ctx, partialCfg, nil)
if err != nil {
return nil, missingSessionConfigError(discoveredSession.Searched, fmt.Sprintf("remote session %q unavailable: %v", remoteKey, err))
}
sessionInfo, err := findRemoteSessionConfig(ctx, store, sessionPrefix, remoteKey)
if err != nil {
return nil, missingSessionConfigError(discoveredSession.Searched, err.Error())
}
sessionTempPath, err := downloadRemoteSessionConfig(ctx, store, remoteKey)
if err != nil {
return nil, missingSessionConfigError(discoveredSession.Searched, fmt.Sprintf("remote session %q download failed: %v", remoteKey, err))
}
sessionBytes, err := os.ReadFile(sessionTempPath)
if err != nil {
return nil, fmt.Errorf("read downloaded remote session %q: %w", sessionTempPath, err)
}
sessionCfg, err := config.LoadSessionBytesWithOptions("s3://"+s3BucketName(base.Pipeline)+"/"+remoteKey, sessionBytes, sessionOpts)
if err != nil {
return nil, err
}
return config.Resolve(
base.PipelinePath,
base.Pipeline,
base.CampaignPath,
base.Campaign,
sessionTempPath,
sessionCfg,
config.SessionSource{
Source: "session_config.s3",
LocalPath: sessionTempPath,
S3Bucket: s3BucketName(base.Pipeline),
S3Key: remoteKey,
S3Size: sessionInfo.Size,
S3ETag: sessionInfo.ETag,
SpoolPath: sessionTempPath,
},
)
}
func loadPipelineCampaignConfig(pipelineFlag, campaignFlag, campaignFileFlag string) (*pipelineCampaignConfig, error) {
resolvedPipelinePath, err := resolvePipelineConfigPath(pipelineFlag)
if err != nil {
return nil, err
}
pipelineCfg, err := config.LoadPipeline(resolvedPipelinePath)
if err != nil {
return nil, err
}
resolvedCampaignPath, err := resolveCampaignConfigPath(pipelineCfg, campaignFlag, campaignFileFlag)
if err != nil {
return nil, err
}
campaignCfg, err := config.LoadCampaign(resolvedCampaignPath)
if err != nil {
return nil, err
}
if selectedID := strings.TrimSpace(campaignFlag); selectedID != "" && strings.TrimSpace(campaignFileFlag) == "" {
if got := config.CampaignID(campaignCfg); got != selectedID {
return nil, fmt.Errorf("campaign config %q invalid: campaign_id %q does not match selected campaign %q", resolvedCampaignPath, got, selectedID)
}
}
return &pipelineCampaignConfig{
PipelinePath: resolvedPipelinePath,
CampaignPath: resolvedCampaignPath,
Pipeline: pipelineCfg,
Campaign: campaignCfg,
}, nil
}
func findRemoteSessionConfig(ctx context.Context, store storage.ObjectStore, sessionPrefix, remoteKey string) (storage.ObjectInfo, error) {
objects, err := store.List(ctx, sessionPrefix)
if err != nil {
return storage.ObjectInfo{}, fmt.Errorf("remote session %q list failed: %w", remoteKey, err)
}
for _, obj := range objects {
if obj.Key == remoteKey {
return obj, nil
}
}
return storage.ObjectInfo{}, fmt.Errorf("remote session %q not found", remoteKey)
}
func downloadRemoteSessionConfig(ctx context.Context, store storage.ObjectStore, remoteKey string) (string, error) {
f, err := os.CreateTemp("", "narratio-session-*.yml")
if err != nil {
return "", fmt.Errorf("create temp file: %w", err)
}
path := f.Name()
if err := f.Close(); err != nil {
return "", fmt.Errorf("close temp file %q: %w", path, err)
}
if err := store.Download(ctx, remoteKey, path); err != nil {
return "", err
}
return filepath.Clean(path), nil
}
func s3BucketName(cfg *config.PipelineConfig) string {
if cfg == nil || cfg.Storage.S3 == nil {
return ""
}
return strings.TrimSpace(cfg.Storage.S3.Bucket)
}

View File

@@ -0,0 +1,21 @@
package app
import (
"context"
"fmt"
"log/slog"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func newCommandObjectStore(ctx context.Context, cfg *config.Config, logger *slog.Logger) (storage.ObjectStore, error) {
if _, err := loadSecretsFromConfig(cfg, logger); err != nil {
return nil, fmt.Errorf("load secrets from files: %w", err)
}
store, err := newObjectStoreFromConfigFn(ctx, cfg)
if err != nil {
return nil, fmt.Errorf("initialize object store backend: %w", err)
}
return store, nil
}

View File

@@ -0,0 +1,165 @@
package app
import (
"context"
"errors"
"os"
"path/filepath"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestNewCommandObjectStoreLoadsSecretsBeforeFactory(t *testing.T) {
accessKeyEnv := "NARRATIO_TEST_COMMAND_STORE_KEY_ID"
secretKeyEnv := "NARRATIO_TEST_COMMAND_STORE_SECRET"
restoreEnvAfterTest(t, accessKeyEnv, secretKeyEnv)
secretsDir := t.TempDir()
mustWriteSecretFile(t, filepath.Join(secretsDir, accessKeyEnv), "loaded-key-id\n")
mustWriteSecretFile(t, filepath.Join(secretsDir, secretKeyEnv), "loaded-secret\n")
cfg := commandObjectStoreTestConfig(secretsDir)
fake := &storage.FakeBackend{}
called := false
origStoreFn := newObjectStoreFromConfigFn
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
called = true
if got := os.Getenv(accessKeyEnv); got != "loaded-key-id" {
return nil, errors.New("access key was not loaded before object store init")
}
if got := os.Getenv(secretKeyEnv); got != "loaded-secret" {
return nil, errors.New("secret key was not loaded before object store init")
}
return fake, nil
}
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
})
store, err := newCommandObjectStore(context.Background(), cfg, nil)
if err != nil {
t.Fatalf("newCommandObjectStore() error = %v", err)
}
if store != fake {
t.Fatalf("store = %#v, want fake backend", store)
}
if !called {
t.Fatal("object store factory was not called")
}
}
func TestNewCommandObjectStorePreservesExistingEnv(t *testing.T) {
accessKeyEnv := "NARRATIO_TEST_COMMAND_STORE_EXISTING_KEY_ID"
secretKeyEnv := "NARRATIO_TEST_COMMAND_STORE_EXISTING_SECRET"
t.Setenv(accessKeyEnv, "existing-key-id")
t.Setenv(secretKeyEnv, "existing-secret")
secretsDir := t.TempDir()
mustWriteSecretFile(t, filepath.Join(secretsDir, accessKeyEnv), "file-key-id\n")
mustWriteSecretFile(t, filepath.Join(secretsDir, secretKeyEnv), "file-secret\n")
cfg := commandObjectStoreTestConfig(secretsDir)
origStoreFn := newObjectStoreFromConfigFn
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
if got := os.Getenv(accessKeyEnv); got != "existing-key-id" {
return nil, errors.New("existing access key was overwritten")
}
if got := os.Getenv(secretKeyEnv); got != "existing-secret" {
return nil, errors.New("existing secret key was overwritten")
}
return &storage.FakeBackend{}, nil
}
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
})
if _, err := newCommandObjectStore(context.Background(), cfg, nil); err != nil {
t.Fatalf("newCommandObjectStore() error = %v", err)
}
}
func TestNewCommandObjectStoreSecretErrorStopsFactory(t *testing.T) {
cfg := commandObjectStoreTestConfig(filepath.Join(t.TempDir(), "missing"))
called := false
origStoreFn := newObjectStoreFromConfigFn
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
called = true
return &storage.FakeBackend{}, nil
}
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
})
_, err := newCommandObjectStore(context.Background(), cfg, nil)
if err == nil {
t.Fatal("expected error, got nil")
}
if called {
t.Fatal("object store factory was called after secret load failure")
}
if !strings.Contains(err.Error(), "load secrets from files") {
t.Fatalf("error = %q, want secret loading context", err.Error())
}
}
func TestNewCommandObjectStoreFactoryErrorIsContextual(t *testing.T) {
cfg := commandObjectStoreTestConfig("")
origStoreFn := newObjectStoreFromConfigFn
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
return nil, errors.New("factory boom")
}
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
})
_, err := newCommandObjectStore(context.Background(), cfg, nil)
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "initialize object store backend") || !strings.Contains(err.Error(), "factory boom") {
t.Fatalf("error = %q, want factory context", err.Error())
}
}
func commandObjectStoreTestConfig(secretsDir string) *config.Config {
cfg := &config.Config{
Pipeline: &config.PipelineConfig{
Storage: config.StorageConfig{
Backend: "s3",
S3: &config.StorageS3Config{
Bucket: "test-bucket",
AccessKeyIDEnv: "NARRATIO_TEST_COMMAND_STORE_KEY_ID",
SecretKeyEnv: "NARRATIO_TEST_COMMAND_STORE_SECRET",
},
},
},
}
if strings.TrimSpace(secretsDir) != "" {
cfg.Pipeline.Secrets = &config.SecretsConfig{EnvDir: secretsDir}
}
return cfg
}
func restoreEnvAfterTest(t *testing.T, names ...string) {
t.Helper()
originals := make(map[string]string, len(names))
present := make(map[string]bool, len(names))
for _, name := range names {
value, ok := os.LookupEnv(name)
originals[name] = value
present[name] = ok
_ = os.Unsetenv(name)
}
t.Cleanup(func() {
for _, name := range names {
if present[name] {
_ = os.Setenv(name, originals[name])
} else {
_ = os.Unsetenv(name)
}
}
})
}

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,962 @@
package app
import (
"bytes"
"context"
"fmt"
"os"
"path/filepath"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
func TestExecuteSessionInitRemoteWritesCanonicalSessionConfig(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{filepath.Join(t.TempDir(), "session.yml")})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "init", "2026-06-07",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--title", "The Black Cabin",
"--remote",
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
key := artifacts.S3SessionConfigKey(artifacts.S3SessionPrefix("dnd", "sample-campaign", "2026-06-07"))
obj, ok := fake.Objects[key]
if !ok {
t.Fatalf("remote session key %q not uploaded; objects=%v", key, fake.Objects)
}
if !strings.Contains(string(obj.Data), `session_id: "2026-06-07"`) || !strings.Contains(string(obj.Data), "prefix: audio/") {
t.Fatalf("remote session data = %q", string(obj.Data))
}
if storeInitCalls != 1 {
t.Fatalf("object store init calls = %d, want 1", storeInitCalls)
}
}
func TestExecuteSessionInitRemoteUsesDefaultConfigDiscovery(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
withDefaultPipelineCampaignConfigs(t, pipelinePath, campaignPath)
fake := &storage.FakeBackend{}
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{filepath.Join(t.TempDir(), "session.yml")})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "init", "2026-06-07",
"--remote",
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
key := artifacts.S3SessionConfigKey(artifacts.S3SessionPrefix("dnd", "sample-campaign", "2026-06-07"))
if _, ok := fake.Objects[key]; !ok {
t.Fatalf("remote session key %q not uploaded; objects=%v", key, fake.Objects)
}
if storeInitCalls != 1 {
t.Fatalf("object store init calls = %d, want 1", storeInitCalls)
}
}
func TestExecuteSessionInitLocalUsesDefaultConfigDiscovery(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
withDefaultPipelineCampaignConfigs(t, pipelinePath, campaignPath)
outputPath := filepath.Join(t.TempDir(), "session.yml")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "init", "2026-06-07",
"--output", outputPath,
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
data, err := os.ReadFile(outputPath)
if err != nil {
t.Fatalf("read generated session: %v", err)
}
if !strings.Contains(string(data), `session_id: "2026-06-07"`) || !strings.Contains(string(data), "prefix: audio/") {
t.Fatalf("generated session = %q", string(data))
}
}
func TestExecuteSessionInitExplicitConfigWinsOverDefaults(t *testing.T) {
workspaceRoot := t.TempDir()
defaultPipeline, defaultCampaign, _ := writeValidConfigFiles(t, workspaceRoot)
withDefaultPipelineCampaignConfigs(t, defaultPipeline, defaultCampaign)
explicitDir := t.TempDir()
explicitCampaign := filepath.Join(explicitDir, "campaign.yml")
if err := os.WriteFile(explicitCampaign, []byte(`campaign_id: explicit-campaign
inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`), 0o644); err != nil {
t.Fatalf("write explicit campaign: %v", err)
}
mustWriteTestFile(t, filepath.Join(explicitDir, "speakers.yml"), "match:\n - speaker: Alice\n match: [\"alice\"]\n")
mustWriteTestFile(t, filepath.Join(explicitDir, "autocorrect.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(explicitDir, "glossary.yml"), "[]\n")
fake := &storage.FakeBackend{}
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{filepath.Join(t.TempDir(), "session.yml")})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "init", "2026-06-07",
"--config", defaultPipeline,
"--campaign-file", explicitCampaign,
"--remote",
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
explicitKey := artifacts.S3SessionConfigKey(artifacts.S3SessionPrefix("dnd", "explicit-campaign", "2026-06-07"))
if _, ok := fake.Objects[explicitKey]; !ok {
t.Fatalf("explicit campaign remote key %q not uploaded; objects=%v", explicitKey, fake.Objects)
}
defaultKey := artifacts.S3SessionConfigKey(artifacts.S3SessionPrefix("dnd", "sample-campaign", "2026-06-07"))
if _, ok := fake.Objects[defaultKey]; ok {
t.Fatalf("default campaign key %q uploaded despite explicit campaign override", defaultKey)
}
}
func TestExecuteSessionInitRequiresSessionID(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
withDefaultPipelineCampaignConfigs(t, pipelinePath, campaignPath)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "init", "--remote"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "session init: session_id is required") {
t.Fatalf("stderr = %q, want session-id required error", stderr.String())
}
}
func TestExecuteSessionInitMissingDefaultConfigReportsSearchedPaths(t *testing.T) {
origPipelineDefaults := append([]string(nil), config.DefaultPipelineConfigSearchPaths...)
config.DefaultPipelineConfigSearchPaths = []string{filepath.Join(t.TempDir(), "missing-pipeline.yml")}
t.Cleanup(func() {
config.DefaultPipelineConfigSearchPaths = origPipelineDefaults
})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "init", "2026-06-07", "--remote"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "session init: no pipeline config path provided and no default pipeline config found; searched:") {
t.Fatalf("stderr = %q, want default pipeline searched-path error", stderr.String())
}
}
func TestExecuteSessionInitRemoteLoadsSecretsBeforeObjectStoreInit(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
withDefaultPipelineCampaignConfigs(t, pipelinePath, campaignPath)
accessKeyEnv := "NARRATIO_TEST_SESSION_INIT_OBJECT_KEY_ID"
secretKeyEnv := "NARRATIO_TEST_SESSION_INIT_OBJECT_SECRET"
restoreEnvAfterTest(t, accessKeyEnv, secretKeyEnv)
secretsDir := t.TempDir()
mustWriteTestFile(t, filepath.Join(secretsDir, accessKeyEnv), "test-key-id\n")
mustWriteTestFile(t, filepath.Join(secretsDir, secretKeyEnv), "test-secret\n")
addSecretsToPipelineConfig(t, pipelinePath, secretsDir, accessKeyEnv, secretKeyEnv)
fake := &storage.FakeBackend{}
origStoreFn := newObjectStoreFromConfigFn
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
if os.Getenv(accessKeyEnv) != "test-key-id" || os.Getenv(secretKeyEnv) != "test-secret" {
return nil, fmt.Errorf("secrets were not loaded before object store init")
}
return fake, nil
}
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "init", "2026-06-07", "--remote"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
}
func TestExecuteSessionInitLocalRendersCampaignTemplate(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
writeSessionInitTemplate(t, campaignPath, `session_id: "{{ session_id }}"
previous_session_id: "{{ previous_session_id }}"
date: "{{ date }}"
title: "{{ title }}"
inputs:
audio_s3:
prefix: "{{ audio_s3_prefix }}"
`)
outputPath := filepath.Join(t.TempDir(), "session.yml")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "init", "2026-06-07",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--previous-session-id", "2026-05-31",
"--date", "2026-06-07",
"--title", "The Black Cabin",
"--audio-s3-prefix", "audio/",
"--output", outputPath,
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
data, err := os.ReadFile(outputPath)
if err != nil {
t.Fatalf("read generated session: %v", err)
}
got := string(data)
for _, want := range []string{
`session_id: "2026-06-07"`,
`previous_session_id: "2026-05-31"`,
`date: "2026-06-07"`,
`title: "The Black Cabin"`,
`prefix: "audio/"`,
} {
if !strings.Contains(got, want) {
t.Fatalf("generated session = %q, want %q", got, want)
}
}
if strings.Contains(got, "{{") {
t.Fatalf("generated session still contains template placeholder: %q", got)
}
}
func TestExecuteSessionInitRemoteRendersCampaignTemplate(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
writeSessionInitTemplate(t, campaignPath, `session_id: "{{ session_id }}"
inputs:
audio_s3:
prefix: audio/
`)
fake := &storage.FakeBackend{}
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{filepath.Join(t.TempDir(), "session.yml")})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "init", "2026-06-07",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--remote",
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
key := artifacts.S3SessionConfigKey(artifacts.S3SessionPrefix("dnd", "sample-campaign", "2026-06-07"))
obj, ok := fake.Objects[key]
if !ok {
t.Fatalf("remote session key %q not uploaded; objects=%v", key, fake.Objects)
}
if strings.Contains(string(obj.Data), "{{") || !strings.Contains(string(obj.Data), `session_id: "2026-06-07"`) {
t.Fatalf("remote session data = %q, want rendered concrete session", string(obj.Data))
}
}
func TestExecuteSessionInitTemplatePathIsCampaignRelative(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
templateDir := filepath.Join(filepath.Dir(campaignPath), "templates")
if err := os.MkdirAll(templateDir, 0o755); err != nil {
t.Fatalf("mkdir template dir: %v", err)
}
templatePath := filepath.Join(templateDir, "session.template.yml")
if err := os.WriteFile(templatePath, []byte(`session_id: "{{ session_id }}"
inputs:
audio_dir: ./audio
`), 0o644); err != nil {
t.Fatalf("write session template: %v", err)
}
addSessionTemplateToCampaign(t, campaignPath, "./templates/session.template.yml")
outputPath := filepath.Join(t.TempDir(), "session.yml")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "init", "2026-06-07",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--output", outputPath,
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
data, err := os.ReadFile(outputPath)
if err != nil {
t.Fatalf("read generated session: %v", err)
}
if !strings.Contains(string(data), `session_id: "2026-06-07"`) {
t.Fatalf("generated session = %q, want campaign-relative template output", string(data))
}
}
func TestExecuteSessionInitTemplateMissingVariableFails(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
writeSessionInitTemplate(t, campaignPath, `session_id: "{{ session_id }}"
date: "{{ date }}"
inputs:
audio_s3:
prefix: audio/
`)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "init", "2026-06-07",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--remote",
}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "missing required template variable value(s): date") {
t.Fatalf("stderr = %q, want missing date variable", stderr.String())
}
}
func TestExecuteSessionInitTemplateUnusedFlagFails(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
writeSessionInitTemplate(t, campaignPath, `session_id: "{{ session_id }}"
inputs:
audio_s3:
prefix: audio/
`)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "init", "2026-06-07",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--title", "Unused Title",
"--remote",
}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "unused template variable value(s): title") {
t.Fatalf("stderr = %q, want unused title variable", stderr.String())
}
}
func TestExecuteSessionInitTemplateStrictDecodeFailure(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
writeSessionInitTemplate(t, campaignPath, `session_id: "{{ session_id }}"
unknown: true
inputs:
audio_s3:
prefix: audio/
`)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "init", "2026-06-07",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--remote",
}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "strict decode failed") {
t.Fatalf("stderr = %q, want strict decode error", stderr.String())
}
}
func TestExecuteSessionValidateLoadsSecretsBeforeObjectStoreInit(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
accessKeyEnv := "NARRATIO_TEST_VALIDATE_OBJECT_KEY_ID"
secretKeyEnv := "NARRATIO_TEST_VALIDATE_OBJECT_SECRET"
restoreEnvAfterTest(t, accessKeyEnv, secretKeyEnv)
secretsDir := t.TempDir()
mustWriteTestFile(t, filepath.Join(secretsDir, accessKeyEnv), "test-key-id\n")
mustWriteTestFile(t, filepath.Join(secretsDir, secretKeyEnv), "test-secret\n")
addSecretsToPipelineConfig(t, pipelinePath, secretsDir, accessKeyEnv, secretKeyEnv)
if err := os.WriteFile(sessionPath, []byte(`session_id: 2026-05-03
inputs:
audio_s3:
prefix: audio/
`), 0o644); err != nil {
t.Fatalf("write session: %v", err)
}
fake := &storage.FakeBackend{}
audioKey := artifacts.S3PromotedArtifactKey(artifacts.S3AudioPrefix(artifacts.S3SessionPrefix("dnd", "sample-campaign", "2026-05-03"), "audio/"), "alice.flac")
fake.SeedObject(storage.FakeObject{Key: audioKey, Data: []byte("audio")})
origStoreFn := newObjectStoreFromConfigFn
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
if os.Getenv(accessKeyEnv) != "test-key-id" || os.Getenv(secretKeyEnv) != "test-secret" {
return nil, fmt.Errorf("secrets were not loaded before object store init")
}
return fake, nil
}
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "validate", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stdout=%q stderr=%q", code, stdout.String(), stderr.String())
}
if !strings.Contains(stdout.String(), "OK audio") {
t.Fatalf("stdout = %q, want OK audio", stdout.String())
}
}
func TestExecuteLocksAddListAndRemoveUseRemoteLockStore(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{sessionPath})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "locks", "add", "2026-05-03", "narratio.transcript.final_trimmed",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--reason", "manual edit",
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("locks add exit code = %d, want 0; stderr=%q", code, stderr.String())
}
key := artifacts.S3SessionLocksKey(artifacts.S3SessionPrefix("dnd", "sample-campaign", "2026-05-03"))
obj, ok := fake.Objects[key]
if !ok {
t.Fatalf("remote locks key %q not uploaded", key)
}
if !strings.Contains(string(obj.Data), "source: narratio.transcript.final_trimmed") || !strings.Contains(string(obj.Data), "reason: manual edit") {
t.Fatalf("lock store data = %q", string(obj.Data))
}
stdout.Reset()
stderr.Reset()
code = Execute([]string{
"session", "locks", "2026-05-03",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("locks list exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "- narratio.transcript.final_trimmed origin=remote") {
t.Fatalf("stdout = %q, want remote lock", stdout.String())
}
stdout.Reset()
stderr.Reset()
code = Execute([]string{
"session", "locks", "remove", "2026-05-03", "narratio.transcript.final_trimmed",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("locks remove exit code = %d, want 0; stderr=%q", code, stderr.String())
}
store, err := config.LoadArchiveLockStoreBytes("locks.yml", fake.Objects[key].Data, nil)
if err != nil {
t.Fatalf("LoadArchiveLockStoreBytes() error = %v", err)
}
if len(store.Locks) != 0 {
t.Fatalf("locks after remove = %#v, want empty", store.Locks)
}
if storeInitCalls != 3 {
t.Fatalf("object store init calls = %d, want 3", storeInitCalls)
}
}
func TestExecuteLocksAddDuplicateRequiresForce(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{sessionPath})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "locks", "add", "2026-05-03", "narratio.transcript.final_trimmed",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--reason", "first",
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("initial locks add exit code = %d, want 0; stderr=%q", code, stderr.String())
}
stdout.Reset()
stderr.Reset()
code = Execute([]string{
"session", "locks", "add", "2026-05-03", "narratio.transcript.final_trimmed",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--reason", "second",
}, &stdout, &stderr)
if code == 0 {
t.Fatal("duplicate locks add exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "pass --force to update") {
t.Fatalf("stderr = %q, want force guidance", stderr.String())
}
stdout.Reset()
stderr.Reset()
code = Execute([]string{
"session", "locks", "add", "2026-05-03", "narratio.transcript.final_trimmed",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--reason", "second",
"--force",
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("forced locks add exit code = %d, want 0; stderr=%q", code, stderr.String())
}
key := artifacts.S3SessionLocksKey(artifacts.S3SessionPrefix("dnd", "sample-campaign", "2026-05-03"))
if !strings.Contains(string(fake.Objects[key].Data), "reason: second") {
t.Fatalf("lock store data = %q, want updated reason", string(fake.Objects[key].Data))
}
}
func TestExecuteLocksRequireSessionID(t *testing.T) {
tests := []struct {
name string
args []string
want string
}{
{"list", []string{"session", "locks"}, "locks: session_id is required"},
{"add", []string{"session", "locks", "add", "narratio.transcript.final_trimmed"}, "locks add: expected session_id and source id"},
{"remove", []string{"session", "locks", "remove", "narratio.transcript.final_trimmed"}, "locks remove: expected session_id and source id"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(tt.args, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), tt.want) {
t.Fatalf("stderr = %q, want %q", stderr.String(), tt.want)
}
})
}
}
func TestExecuteLocksCannotModifyStaticLocks(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
addStaticArchiveLockToPipelineConfig(t, pipelinePath, "narratio.transcript.final_trimmed")
fake := &storage.FakeBackend{}
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{sessionPath})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "locks", "add", "2026-05-03", "narratio.transcript.final_trimmed",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
}, &stdout, &stderr)
if code == 0 {
t.Fatal("locks add static lock exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "locked by pipeline config") {
t.Fatalf("stderr = %q, want static lock error", stderr.String())
}
stdout.Reset()
stderr.Reset()
code = Execute([]string{
"session", "locks", "remove", "2026-05-03", "narratio.transcript.final_trimmed",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
}, &stdout, &stderr)
if code == 0 {
t.Fatal("locks remove static lock exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "locked by pipeline config") {
t.Fatalf("stderr = %q, want static lock error", stderr.String())
}
}
func TestExecuteTopLevelLockAndUnlockAreRemoved(t *testing.T) {
tests := []string{"lock", "unlock"}
for _, cmd := range tests {
t.Run(cmd, func(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{cmd, "narratio.transcript.final_trimmed"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), `unknown command: "`+cmd+`"`) {
t.Fatalf("stderr = %q, want unknown command", stderr.String())
}
})
}
}
func withDefaultPipelineCampaignConfigs(t *testing.T, pipelinePath, campaignPath string) {
t.Helper()
origPipelineDefaults := append([]string(nil), config.DefaultPipelineConfigSearchPaths...)
config.DefaultPipelineConfigSearchPaths = []string{pipelinePath}
t.Cleanup(func() {
config.DefaultPipelineConfigSearchPaths = origPipelineDefaults
})
_ = campaignPath
}
func writeSessionInitTemplate(t *testing.T, campaignPath, templateYAML string) {
t.Helper()
templatePath := filepath.Join(filepath.Dir(campaignPath), "session.template.yml")
if err := os.WriteFile(templatePath, []byte(templateYAML), 0o644); err != nil {
t.Fatalf("write session template: %v", err)
}
addSessionTemplateToCampaign(t, campaignPath, "./session.template.yml")
}
func addSessionTemplateToCampaign(t *testing.T, campaignPath, templateFile string) {
t.Helper()
data, err := os.ReadFile(campaignPath)
if err != nil {
t.Fatalf("read campaign config: %v", err)
}
if strings.Contains(string(data), "session_template_file:") {
t.Fatalf("campaign config already has session_template_file: %q", string(data))
}
updated := "session_template_file: " + templateFile + "\n" + string(data)
if err := os.WriteFile(campaignPath, []byte(updated), 0o644); err != nil {
t.Fatalf("write campaign config: %v", err)
}
}
func TestExecuteArtifactsListRemoteReportsPromotedAvailability(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
addArchivePromotionsToPipeline(t, pipelinePath, `
promote_artifacts:
- source: narratio.transcript.final_trimmed
dest: transcripts/final.trimmed.json
required: true
`)
fake := &storage.FakeBackend{}
trimmedKey := artifacts.S3PromotedArtifactKey(
artifacts.S3SessionPrefix("dnd", "sample-campaign", "2026-05-03"),
"transcripts/final.trimmed.json",
)
fake.SeedObject(storage.FakeObject{Key: trimmedKey, Data: []byte(`{"segments":[]}`)})
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{sessionPath})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "artifacts", "2026-05-03",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--remote",
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "narratio.transcript.final_trimmed remote=promoted") {
t.Fatalf("stdout = %q, want promoted remote availability", stdout.String())
}
}
func TestExecuteArtifactsListRemoteUsesPromotionDestinations(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
addArchivePromotionsToPipeline(t, pipelinePath, `
promote_artifacts:
- source: narratio.transcript.final
dest: transcripts/full.json
required: true
- source: narratio.bounds.session
dest: transcripts/bounds.json
required: true
`)
fake := &storage.FakeBackend{}
sessionPrefix := artifacts.S3SessionPrefix("dnd", "sample-campaign", "2026-05-03")
fake.SeedObject(storage.FakeObject{Key: artifacts.S3PromotedArtifactKey(sessionPrefix, "transcripts/full.json"), Data: []byte(`{"segments":[]}`)})
fake.SeedObject(storage.FakeObject{Key: artifacts.S3PromotedArtifactKey(sessionPrefix, "transcripts/bounds.json"), Data: []byte(`{}`)})
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{sessionPath})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "artifacts", "2026-05-03",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--remote",
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
out := stdout.String()
for _, unwanted := range []string{
"narratio.transcript.final remote=missing",
"narratio.bounds.session remote=missing",
} {
if strings.Contains(out, unwanted) {
t.Fatalf("stdout = %q, did not want catalog remote marker %q", out, unwanted)
}
}
for _, want := range []string{
"narratio.transcript.final dest=transcripts/full.json remote=promoted",
"narratio.bounds.session dest=transcripts/bounds.json remote=promoted",
} {
if !strings.Contains(out, want) {
t.Fatalf("stdout = %q, want %q", out, want)
}
}
}
func TestExecuteStatusReportsRemoteArtifactCatalog(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
addArchivePromotionsToPipeline(t, pipelinePath, `
promote_artifacts:
- source: narratio.transcript.final_trimmed
dest: transcripts/final.trimmed.json
required: true
- source: narratio.transcript.final
dest: transcripts/full.json
required: true
`)
fake := &storage.FakeBackend{}
sessionPrefix := artifacts.S3SessionPrefix("dnd", "sample-campaign", "2026-05-03")
manifestKey, runIDKey := artifacts.ResolveArchiveCurrentStateKeys(sessionPrefix)
trimmedKey := artifacts.S3PromotedArtifactKey(sessionPrefix, "transcripts/final.trimmed.json")
fullKey := artifacts.S3PromotedArtifactKey(sessionPrefix, "transcripts/full.json")
lockKey := artifacts.S3SessionLocksKey(sessionPrefix)
fake.SeedObject(storage.FakeObject{Key: runIDKey, Data: []byte("20260519T010203Z-a1b2c3d4\n")})
fake.SeedObject(storage.FakeObject{Key: manifestKey, Data: restoreManifestJSON(t, "2026-05-03", "sample-campaign")})
fake.SeedObject(storage.FakeObject{Key: trimmedKey, Data: []byte(`{"segments":[]}`)})
fake.SeedObject(storage.FakeObject{Key: fullKey, Data: []byte(`{"segments":[]}`)})
fake.SeedObject(storage.FakeObject{Key: lockKey, Data: []byte("locks:\n - source: narratio.transcript.final_trimmed\n reason: remote review\n")})
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{sessionPath})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "status", "2026-05-03",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{
"Remote outputs:",
"Built-in:",
"Configured:",
"Previous-session:",
"Promoted:",
"narratio.transcript.final_trimmed locked",
"narratio.transcript.final_trimmed locked remote=promoted",
"narratio.transcript.final dest=transcripts/full.json remote=promoted",
} {
if !strings.Contains(out, want) {
t.Fatalf("stdout = %q, want %q", out, want)
}
}
if strings.Contains(out, "narratio.transcript.base remote=missing") {
t.Fatalf("stdout = %q, did not want catalog remote marker", out)
}
}
func TestExecuteStatusReportsRemoteArtifactCatalogErrorsWithoutFailing(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
addArchivePromotionsToPipeline(t, pipelinePath, `
promote_artifacts:
- source: narratio.transcript.final_trimmed
dest: transcripts/final.trimmed.json
required: true
`)
fake := &storage.FakeBackend{ExistsErr: fmt.Errorf("exists failed")}
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{sessionPath})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{
"session", "status", "2026-05-03",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "Remote archive: missing or unavailable:") {
t.Fatalf("stdout = %q, want remote archive unavailable state", out)
}
if !strings.Contains(out, "Remote outputs:") || !strings.Contains(out, "narratio.transcript.final_trimmed remote=error") {
t.Fatalf("stdout = %q, want remote output error state", out)
}
if !strings.Contains(out, "Archive locks: error:") {
t.Fatalf("stdout = %q, want archive locks error", out)
}
}
func TestExecuteArchiveLoadsRemoteLocks(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidArchiveConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
lockKey := artifacts.S3SessionLocksKey(artifacts.S3SessionPrefix("dnd", "sample-campaign", "2026-05-03"))
fake.SeedObject(storage.FakeObject{Key: lockKey, Data: []byte("locks:\n - source: narratio.transcript.final_trimmed\n reason: remote review\n")})
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{sessionPath})
workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
for _, stageName := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "analyze"} {
// The archive stage only checks the manifest statuses and source files.
_ = stageName
}
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "final.trimmed.json"), `{"segments":[]}`)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"run-stage", "archive", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--force"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
promotedKey := artifacts.S3PromotedArtifactKey(artifacts.S3SessionPrefix("dnd", "sample-campaign", "2026-05-03"), "transcripts/final.trimmed.json")
if _, ok := fake.Objects[promotedKey]; ok {
t.Fatalf("locked promoted key %q was uploaded", promotedKey)
}
}
func addArchivePromotionsToPipeline(t *testing.T, pipelinePath, archiveYAML string) {
t.Helper()
data, err := os.ReadFile(pipelinePath)
if err != nil {
t.Fatalf("read pipeline: %v", err)
}
updated := strings.Replace(string(data), " upload_run: false\n", " upload_run: false\n"+archiveYAML, 1)
if updated == string(data) {
t.Fatalf("pipeline %q did not contain archive upload_run marker", pipelinePath)
}
if err := os.WriteFile(pipelinePath, []byte(updated), 0o644); err != nil {
t.Fatalf("write pipeline: %v", err)
}
}
func writeValidArchiveConfigFiles(t *testing.T, workspaceRoot string) (string, string, string) {
t.Helper()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
data, err := os.ReadFile(pipelinePath)
if err != nil {
t.Fatalf("read pipeline: %v", err)
}
updated := strings.Replace(string(data), "upload_run: false", "upload_run: true", 1)
if err := os.WriteFile(pipelinePath, []byte(updated), 0o644); err != nil {
t.Fatalf("write pipeline: %v", err)
}
ctx := context.Background()
cfg, err := config.LoadWithSessionOptions(pipelinePath, campaignPath, sessionPath, config.SessionLoadOptions{})
if err != nil {
t.Fatalf("LoadWithSessionOptions() error = %v", err)
}
store := &manifest.LocalStore{}
m := manifest.New("2026-05-03", nowUTC())
m.Campaign = "sample-campaign"
m.RunID = "20260521T160000Z-test"
for _, name := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "analyze"} {
m.MarkStageSucceeded(name, nowUTC(), nil)
}
path := artifacts.SessionManifestPathForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID)
if err := store.Save(ctx, path, m); err != nil {
t.Fatalf("save manifest: %v", err)
}
runManifestPath := artifacts.SessionRunManifestPathForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID, m.RunID)
if err := os.MkdirAll(filepath.Dir(runManifestPath), 0o755); err != nil {
t.Fatalf("mkdir run manifest: %v", err)
}
if err := os.WriteFile(runManifestPath, []byte("{}\n"), 0o644); err != nil {
t.Fatalf("write run manifest: %v", err)
}
return pipelinePath, campaignPath, sessionPath
}
func addStaticArchiveLockToPipelineConfig(t *testing.T, pipelinePath, source string) {
t.Helper()
data, err := os.ReadFile(pipelinePath)
if err != nil {
t.Fatalf("read pipeline: %v", err)
}
updated := strings.Replace(
string(data),
"archive:\n enabled: true\n upload_run: false\n",
"archive:\n enabled: true\n upload_run: false\n locks:\n - source: "+source+"\n reason: static review\n",
1,
)
if updated == string(data) {
t.Fatalf("archive section not found in pipeline config")
}
if err := os.WriteFile(pipelinePath, []byte(updated), 0o644); err != nil {
t.Fatalf("write pipeline: %v", err)
}
}

View File

@@ -7,6 +7,7 @@ import (
"io" "io"
"log/slog" "log/slog"
"os" "os"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts" "gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config" "gitea.maximumdirect.net/eric/narratio/internal/config"
@@ -16,32 +17,46 @@ import (
// Plan validates configuration, prepares the local workdir, and prints stage order. // Plan validates configuration, prepares the local workdir, and prints stage order.
func Plan(ctx context.Context, args []string, out io.Writer) error { func Plan(ctx context.Context, args []string, out io.Writer) error {
positionalSessionID, args := pullLeadingSessionID(args)
fs := flag.NewFlagSet("plan", flag.ContinueOnError) fs := flag.NewFlagSet("plan", flag.ContinueOnError)
fs.SetOutput(io.Discard) fs.SetOutput(io.Discard)
var pipelinePath string var pipelinePath string
var campaignPath string
var campaignFilePath string
var sessionPath string var sessionPath string
var sessionID string
var previousSessionID string
var force bool var force bool
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)") fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
fs.StringVar(&campaignPath, "campaign", "", "campaign ID")
fs.StringVar(&campaignFilePath, "campaign-file", "", "path to campaign.yml")
fs.StringVar(&sessionPath, "session", "", "path to session.yml") fs.StringVar(&sessionPath, "session", "", "path to session.yml")
fs.StringVar(&previousSessionID, "previous-session-id", "", "expected previous session identifier")
fs.BoolVar(&force, "force", false, "force stage execution (reserved for future behavior)") fs.BoolVar(&force, "force", false, "force stage execution (reserved for future behavior)")
if err := fs.Parse(args); err != nil { if err := fs.Parse(args); err != nil {
return fmt.Errorf("plan: invalid flags: %w", err) return fmt.Errorf("plan: invalid flags: %w", err)
} }
if positionalSessionID == "" {
if err := applyParsedSessionIDArg("plan", fs, &sessionID); err != nil {
return err
}
} else {
if fs.NArg() != 0 { if fs.NArg() != 0 {
return fmt.Errorf("plan: unexpected positional arguments") return fmt.Errorf("plan: unexpected positional arguments")
} }
if sessionPath == "" { if err := applyPositionalSessionID("plan", positionalSessionID, &sessionID); err != nil {
return fmt.Errorf("plan: --session is required") return err
} }
resolvedPipelinePath, err := resolvePipelineConfigPath(pipelinePath)
if err != nil {
return fmt.Errorf("plan: %w", err)
} }
if strings.TrimSpace(sessionID) == "" {
cfg, err := config.Load(resolvedPipelinePath, sessionPath) return fmt.Errorf("plan: session_id is required")
}
cfg, err := loadCommandConfig(ctx, pipelinePath, campaignPath, campaignFilePath, sessionPath, config.SessionLoadOptions{
SessionID: sessionID,
PreviousSessionID: previousSessionID,
})
if err != nil { if err != nil {
return fmt.Errorf("plan: %w", err) return fmt.Errorf("plan: %w", err)
} }
@@ -53,7 +68,7 @@ func Plan(ctx context.Context, args []string, out io.Writer) error {
} }
store := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root) store := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root)
paths, err := store.EnsureLayout(cfg.Session.SessionID) paths, err := store.EnsureLayoutFor(cfg.Session.Campaign, cfg.Session.SessionID)
if err != nil { if err != nil {
return fmt.Errorf("plan: prepare workdir: %w", err) return fmt.Errorf("plan: prepare workdir: %w", err)
} }
@@ -68,7 +83,7 @@ func Plan(ctx context.Context, args []string, out io.Writer) error {
runCount := 0 runCount := 0
skipCount := 0 skipCount := 0
if _, err := fmt.Fprintf(out, "narratio plan: workdir prepared at %s\n", paths.Root); err != nil { if _, err := fmt.Fprintf(out, "narratio session plan: workdir prepared at %s\n", paths.Root); err != nil {
return err return err
} }
for _, d := range decisions { for _, d := range decisions {

View File

@@ -15,16 +15,16 @@ import (
func TestPlanCreatesAndReusesWorkdir(t *testing.T) { func TestPlanCreatesAndReusesWorkdir(t *testing.T) {
workspaceRoot := t.TempDir() workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot) pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var out bytes.Buffer var out bytes.Buffer
args := []string{"--config", pipelinePath, "--session", sessionPath} args := []string{"2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}
if err := Plan(context.Background(), args, &out); err != nil { if err := Plan(context.Background(), args, &out); err != nil {
t.Fatalf("first Plan() error = %v", err) t.Fatalf("first Plan() error = %v", err)
} }
got := out.String() got := out.String()
if !strings.Contains(got, "narratio plan: workdir prepared at") { if !strings.Contains(got, "narratio session plan: workdir prepared at") {
t.Fatalf("first output = %q, want workdir prepared", got) t.Fatalf("first output = %q, want workdir prepared", got)
} }
for _, name := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "analyze", "archive", "notify"} { for _, name := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "analyze", "archive", "notify"} {
@@ -36,7 +36,7 @@ func TestPlanCreatesAndReusesWorkdir(t *testing.T) {
t.Fatalf("first output = %q, want totals", got) t.Fatalf("first output = %q, want totals", got)
} }
sessionWorkdir := artifacts.SessionWorkDir(workspaceRoot, "2026-05-03") sessionWorkdir := artifacts.SessionWorkDirForCampaign(workspaceRoot, "sample-campaign", "2026-05-03")
expectedDirs := []string{ expectedDirs := []string{
sessionWorkdir, sessionWorkdir,
filepath.Join(sessionWorkdir, "inputs"), filepath.Join(sessionWorkdir, "inputs"),
@@ -55,15 +55,15 @@ func TestPlanCreatesAndReusesWorkdir(t *testing.T) {
if err := Plan(context.Background(), args, &out); err != nil { if err := Plan(context.Background(), args, &out); err != nil {
t.Fatalf("second Plan() error = %v", err) t.Fatalf("second Plan() error = %v", err)
} }
if !strings.Contains(out.String(), "narratio plan: workdir prepared at") { if !strings.Contains(out.String(), "narratio session plan: workdir prepared at") {
t.Fatalf("second output = %q, want workdir prepared", out.String()) t.Fatalf("second output = %q, want workdir prepared", out.String())
} }
} }
func TestPlanShowsRunAndSkipFromManifest(t *testing.T) { func TestPlanShowsRunAndSkipFromManifest(t *testing.T) {
workspaceRoot := t.TempDir() workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot) pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json") manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{} store := &manifest.LocalStore{}
m := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC)) m := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
@@ -74,7 +74,7 @@ func TestPlanShowsRunAndSkipFromManifest(t *testing.T) {
} }
var out bytes.Buffer var out bytes.Buffer
if err := Plan(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath}, &out); err != nil { if err := Plan(context.Background(), []string{"2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &out); err != nil {
t.Fatalf("Plan() error = %v", err) t.Fatalf("Plan() error = %v", err)
} }
got := out.String() got := out.String()
@@ -93,6 +93,7 @@ func TestPlanFailsWhenConfiguredSecretsDirMissing(t *testing.T) {
workspaceRoot := t.TempDir() workspaceRoot := t.TempDir()
configDir := t.TempDir() configDir := t.TempDir()
pipelinePath := filepath.Join(configDir, "pipeline.yml") pipelinePath := filepath.Join(configDir, "pipeline.yml")
campaignPath := writeAppTestCampaignConfig(t, configDir)
sessionPath := filepath.Join(configDir, "session.yml") sessionPath := filepath.Join(configDir, "session.yml")
pipelineYAML := `workspace: pipelineYAML := `workspace:
@@ -107,12 +108,11 @@ seriatim:
binary: seriatim binary: seriatim
audita: audita:
binary: audita binary: audita
analyzer:
timeout: 20m
notification: notification:
timeout: 10s timeout: 10s
` `
sessionYAML := `session_id: 2026-05-03 sessionYAML := `session_id: 2026-05-03
campaign: sample-campaign
inputs: inputs:
audio_dir: ./audio audio_dir: ./audio
speakers_file: ./speakers.yml speakers_file: ./speakers.yml
@@ -127,7 +127,7 @@ inputs:
} }
var out bytes.Buffer var out bytes.Buffer
err := Plan(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath}, &out) err := Plan(context.Background(), []string{"2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &out)
if err == nil { if err == nil {
t.Fatal("expected error, got nil") t.Fatal("expected error, got nil")
} }

View File

@@ -0,0 +1,225 @@
package app
import (
"context"
"fmt"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
func runPostArchiveCleanup(ctx context.Context, env *Env, manifestPath string, m *manifest.Manifest, executed []string) error {
if env == nil || env.Config == nil || env.Config.Pipeline == nil || m == nil {
return nil
}
spoolRequested := env.Config.Pipeline.Spool.DeleteAudioAfterArchive
workRequested := env.Config.Pipeline.Workspace.CleanupAfterArchive
if !spoolRequested && !workRequested {
return nil
}
sr := archiveStageRecordForCleanup(m, executed)
if sr == nil {
return nil
}
if sr.Metadata == nil {
sr.Metadata = map[string]any{}
}
sr.Metadata["spool_cleanup_requested"] = spoolRequested
sr.Metadata["workdir_cleanup_requested"] = workRequested
eligible, reason := archiveCleanupEligible(env.Config, sr)
if !eligible {
sr.Metadata["cleanup_skipped"] = true
sr.Metadata["cleanup_skipped_reason"] = reason
if err := env.ManifestStore.Save(ctx, manifestPath, m); err != nil {
return fmt.Errorf("save manifest cleanup skip metadata %q: %w", manifestPath, err)
}
return nil
}
spoolDir := strings.TrimSpace(m.LocalSpoolDir)
if spoolDir == "" {
spoolDir = artifacts.SessionSpoolAudioDir(
env.Config.Pipeline.Spool.Root,
strings.TrimSpace(env.Config.Session.Campaign),
strings.TrimSpace(env.Config.Session.SessionID),
strings.TrimSpace(m.RunID),
)
}
workDir := strings.TrimSpace(m.LocalWorkDir)
if workDir == "" {
workDir = artifacts.SessionRunRootForCampaign(
env.Config.Pipeline.Workspace.Root,
strings.TrimSpace(env.Config.Session.Campaign),
strings.TrimSpace(env.Config.Session.SessionID),
strings.TrimSpace(m.RunID),
)
}
if spoolRequested {
if err := removeRunScopedDir(strings.TrimSpace(env.Config.Pipeline.Spool.Root), spoolDir, "pipeline.spool.delete_audio_after_archive"); err != nil {
sr.Metadata["cleanup_failed"] = true
sr.Metadata["cleanup_failed_policy"] = "pipeline.spool.delete_audio_after_archive"
sr.Metadata["cleanup_failed_path"] = spoolDir
_ = env.ManifestStore.Save(ctx, manifestPath, m)
return err
}
sr.Metadata["spool_cleanup_deleted"] = filepath.Clean(spoolDir)
}
if !workRequested {
sr.Metadata["cleanup_completed"] = true
sr.Metadata["cleanup_skipped"] = false
if err := env.ManifestStore.Save(ctx, manifestPath, m); err != nil {
return fmt.Errorf("save manifest cleanup metadata %q: %w", manifestPath, err)
}
return nil
}
if err := removeRunScopedDir(strings.TrimSpace(env.Config.Pipeline.Workspace.Root), workDir, "pipeline.workspace.cleanup_after_archive"); err != nil {
sr.Metadata["cleanup_failed"] = true
sr.Metadata["cleanup_failed_policy"] = "pipeline.workspace.cleanup_after_archive"
sr.Metadata["cleanup_failed_path"] = workDir
_ = env.ManifestStore.Save(ctx, manifestPath, m)
return err
}
sr.Metadata["workdir_cleanup_deleted"] = filepath.Clean(workDir)
sr.Metadata["cleanup_completed"] = true
sr.Metadata["cleanup_skipped"] = false
return nil
}
func archiveStageRecordForCleanup(m *manifest.Manifest, executed []string) *manifest.StageRecord {
if m == nil {
return nil
}
archiveRan := false
for _, name := range executed {
if name == "archive" {
archiveRan = true
break
}
}
if !archiveRan {
return nil
}
sr := m.Stages["archive"]
if sr == nil || sr.Status != manifest.StatusSucceeded {
return nil
}
return sr
}
func archiveCleanupEligible(cfg *config.Config, sr *manifest.StageRecord) (bool, string) {
if cfg == nil || cfg.Pipeline == nil || cfg.Pipeline.Archive == nil {
return false, "archive configuration is missing"
}
enabled := true
if cfg.Pipeline.Archive.Enabled != nil {
enabled = *cfg.Pipeline.Archive.Enabled
}
if !enabled {
return false, "archive.enabled is false"
}
uploadRun := true
if cfg.Pipeline.Archive.UploadRun != nil {
uploadRun = *cfg.Pipeline.Archive.UploadRun
}
if !uploadRun {
return false, "archive.upload_run is false"
}
if sr == nil || sr.Metadata == nil {
return false, "archive metadata is missing"
}
if skipped, _ := sr.Metadata["skipped"].(bool); skipped {
return false, "archive stage was skipped"
}
if uploaded, _ := sr.Metadata["uploaded"].(bool); !uploaded {
return false, "archive did not upload run record"
}
if pointer, _ := sr.Metadata["current_pointer_written"].(bool); !pointer {
return false, "archive did not write current pointer"
}
if strings.TrimSpace(asString(sr.Metadata["current_run_id_key"])) == "" {
return false, "archive current run pointer key is missing"
}
return true, ""
}
type scopedDir struct {
RootAbs string
TargetAbs string
Exists bool
}
func removeRunScopedDir(root, target, policy string) error {
dir, err := validateScopedDir(root, target, policy)
if err != nil {
return err
}
if !dir.Exists {
return nil
}
if err := os.RemoveAll(dir.TargetAbs); err != nil {
return fmt.Errorf("cleanup policy %s: remove %q: %w", policy, dir.TargetAbs, err)
}
return nil
}
func validateScopedDir(root, target, policy string) (scopedDir, error) {
cleanRoot := strings.TrimSpace(root)
cleanTarget := strings.TrimSpace(target)
if cleanRoot == "" {
return scopedDir{}, fmt.Errorf("cleanup policy %s: root path is required", policy)
}
if cleanTarget == "" {
return scopedDir{}, fmt.Errorf("cleanup policy %s: target path is required", policy)
}
rootAbs, err := filepath.Abs(cleanRoot)
if err != nil {
return scopedDir{}, fmt.Errorf("cleanup policy %s: resolve root %q: %w", policy, cleanRoot, err)
}
targetAbs, err := filepath.Abs(cleanTarget)
if err != nil {
return scopedDir{}, fmt.Errorf("cleanup policy %s: resolve target %q: %w", policy, cleanTarget, err)
}
rel, err := filepath.Rel(rootAbs, targetAbs)
if err != nil {
return scopedDir{}, fmt.Errorf("cleanup policy %s: relative path from %q to %q: %w", policy, rootAbs, targetAbs, err)
}
if rel == "." {
return scopedDir{}, fmt.Errorf("cleanup policy %s: refusing to delete root directory %q", policy, rootAbs)
}
if rel == ".." || strings.HasPrefix(rel, ".."+string(filepath.Separator)) {
return scopedDir{}, fmt.Errorf("cleanup policy %s: refusing to delete path outside root: root=%q target=%q", policy, rootAbs, targetAbs)
}
info, err := os.Lstat(targetAbs)
if err != nil {
if os.IsNotExist(err) {
return scopedDir{RootAbs: rootAbs, TargetAbs: targetAbs, Exists: false}, nil
}
return scopedDir{}, fmt.Errorf("cleanup policy %s: stat target %q: %w", policy, targetAbs, err)
}
if info.Mode()&os.ModeSymlink != 0 {
return scopedDir{}, fmt.Errorf("cleanup policy %s: refusing to delete symlink path %q", policy, targetAbs)
}
if !info.IsDir() {
return scopedDir{}, fmt.Errorf("cleanup policy %s: target %q is not a directory", policy, targetAbs)
}
return scopedDir{RootAbs: rootAbs, TargetAbs: targetAbs, Exists: true}, nil
}
func asString(v any) string {
s, _ := v.(string)
return s
}

View File

@@ -0,0 +1,421 @@
package app
import (
"context"
"errors"
"os"
"path/filepath"
"strings"
"testing"
"time"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
"gitea.maximumdirect.net/eric/narratio/internal/stage"
)
type archiveSuccessStage struct {
metadata map[string]any
}
func (archiveSuccessStage) Name() string { return "archive" }
func (archiveSuccessStage) Declares() stage.IODecl { return stage.IODecl{} }
func (s archiveSuccessStage) Run(_ context.Context, _ *stage.Env, _ *manifest.Manifest) (*stage.StageResult, error) {
md := map[string]any{
"stage": "archive",
"uploaded": true,
"current_pointer_written": true,
"current_run_id_key": "dnd/campaigns/sample-campaign/sessions/2026-05-03/current/run_id.txt",
}
for k, v := range s.metadata {
md[k] = v
}
return &stage.StageResult{Metadata: md}, nil
}
type notifyFailStage struct{}
func (notifyFailStage) Name() string { return "notify" }
func (notifyFailStage) Declares() stage.IODecl { return stage.IODecl{} }
func (notifyFailStage) Run(_ context.Context, _ *stage.Env, _ *manifest.Manifest) (*stage.StageResult, error) {
return nil, errors.New("notify failed")
}
func TestPostArchiveCleanupDisabledKeepsLocalDirs(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = false
cfg.Pipeline.Workspace.CleanupAfterArchive = false
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
assertExists(t, seed.localSourceAudio)
}
func TestPostArchiveCleanupSpoolOnly(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = false
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertMissing(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
assertExists(t, seed.localSourceAudio)
}
func TestPostArchiveCleanupWorkdirOnly(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = false
cfg.Pipeline.Workspace.CleanupAfterArchive = true
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertExists(t, cfg.Pipeline.Workspace.Root)
assertExists(t, seed.otherRunDir)
assertExists(t, seed.previousCachePath)
assertMissing(t, seed.runWorkDir)
assertExists(t, seed.spoolAudioDir)
}
func TestPostArchiveCleanupBothPolicies(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertMissing(t, seed.spoolAudioDir)
assertMissing(t, seed.runWorkDir)
assertExists(t, seed.otherRunDir)
assertExists(t, seed.previousCachePath)
}
func TestPostArchiveCleanupNotRunWhenArchiveFails(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
_, err := executeStages(context.Background(), cfg, []stage.Stage{failingStage{name: "archive", err: errors.New("archive failed")}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}})
if err == nil || !strings.Contains(err.Error(), "stage \"archive\" failed") {
t.Fatalf("executeStages() error = %v, want archive failure", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
func TestPostArchiveCleanupNotRunWhenArchiveSkipped(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{metadata: map[string]any{"skipped": true}}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
func TestPostArchiveCleanupNotRunWhenCurrentPointerMissing(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{metadata: map[string]any{"current_pointer_written": false}}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
func TestPostArchiveCleanupNotRunWhenArchiveUploadDisabled(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
cfg.Pipeline.Archive.UploadRun = boolPtr(false)
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
func TestPostArchiveCleanupWaitsUntilAllStagesSucceed(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
_, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}, notifyFailStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}})
if err == nil || !strings.Contains(err.Error(), "stage \"notify\" failed") {
t.Fatalf("executeStages() error = %v, want notify failure", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
func TestPostArchiveCleanupFailsOnUnsafePath(t *testing.T) {
cfg, _ := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = false
manifestPath := manifestPathFor(cfg)
store := &manifest.LocalStore{}
m, err := store.Load(context.Background(), manifestPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
m.LocalSpoolDir = filepath.Join(filepath.Dir(cfg.Pipeline.Spool.Root), "outside-spool")
if err := store.Save(context.Background(), manifestPath, m); err != nil {
t.Fatalf("Save() error = %v", err)
}
_, err = executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}})
if err == nil || !strings.Contains(err.Error(), "refusing to delete path outside root") {
t.Fatalf("executeStages() error = %v, want safe-path failure", err)
}
}
func TestPostArchiveCleanupNotRunWhenPromotionIsMissing(t *testing.T) {
cfg, seed, runID := archiveStageCleanupFixture(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
cfg.Pipeline.Archive.PromoteArtifacts = []config.ArchivePromotionRule{
{Source: "narratio.transcript.base", Dest: "transcripts/base.json", Required: boolPtr(true)},
}
archiveStageImpl, err := stage.Select("archive")
if err != nil {
t.Fatalf("Select(archive) error = %v", err)
}
_, err = executeStages(context.Background(), cfg, []stage.Stage{archiveStageImpl}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}})
if err == nil || !strings.Contains(err.Error(), "required promotion source unavailable") {
t.Fatalf("executeStages() error = %v, want promotion-missing failure", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
assertExists(t, filepath.Join(seed.runWorkDir, "manifest.json"))
assertExists(t, artifacts.SessionRunRootForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID, runID))
}
func TestPostArchiveCleanupNotRunWhenCurrentManifestUploadFails(t *testing.T) {
cfg, seed, _ := archiveStageCleanupFixture(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
failKey := seed.sessionPrefix + "current/manifest.json"
archiveStageImpl, err := stage.Select("archive")
if err != nil {
t.Fatalf("Select(archive) error = %v", err)
}
_, err = executeStages(context.Background(), cfg, []stage.Stage{archiveStageImpl}, RunOptions{
Env: &Env{ObjectStore: &failKeyStore{delegate: &storage.FakeBackend{}, failKey: failKey}},
})
if err == nil || !strings.Contains(err.Error(), "current manifest") {
t.Fatalf("executeStages() error = %v, want current-manifest failure", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
func TestPostArchiveCleanupNotRunWhenCurrentPointerUploadFails(t *testing.T) {
cfg, seed, _ := archiveStageCleanupFixture(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
failKey := seed.sessionPrefix + "current/run_id.txt"
archiveStageImpl, err := stage.Select("archive")
if err != nil {
t.Fatalf("Select(archive) error = %v", err)
}
_, err = executeStages(context.Background(), cfg, []stage.Stage{archiveStageImpl}, RunOptions{
Env: &Env{ObjectStore: &failKeyStore{delegate: &storage.FakeBackend{}, failKey: failKey}},
})
if err == nil || !strings.Contains(err.Error(), "current run pointer") {
t.Fatalf("executeStages() error = %v, want current-run-pointer failure", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
type cleanupSeed struct {
runWorkDir string
otherRunDir string
spoolAudioDir string
localSourceAudio string
previousCachePath string
sessionPrefix string
}
func cleanupFixtureConfig(t *testing.T) (*config.Config, cleanupSeed) {
t.Helper()
cfg := testConfig(t)
cfg.Pipeline.Archive = &config.ArchiveConfig{Enabled: boolPtr(true), UploadRun: boolPtr(true)}
cfg.Pipeline.Spool.Root = filepath.Join(t.TempDir(), "spool")
runID := "20260516T010203Z-1a2b3c4d"
runWorkDir := artifacts.SessionRunRootForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID, runID)
otherRunDir := artifacts.SessionRunRootForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID, "20260516T010204Z-5e6f7a8b")
spoolAudioDir := artifacts.SessionSpoolAudioDir(cfg.Pipeline.Spool.Root, cfg.Session.Campaign, cfg.Session.SessionID, runID)
previousCachePath := artifacts.SessionPreviousArtifactPathForCampaign(
cfg.Pipeline.Workspace.Root,
cfg.Session.Campaign,
cfg.Session.SessionID,
"session_recap.md",
)
mustWriteFile(t, filepath.Join(runWorkDir, "manifest.json"), "{}\n")
mustWriteFile(t, filepath.Join(runWorkDir, "logs", "stage.log"), "log\n")
mustWriteFile(t, filepath.Join(otherRunDir, "logs", "stage.log"), "other\n")
mustWriteFile(t, filepath.Join(spoolAudioDir, "speaker.flac"), "flac\n")
mustWriteFile(t, previousCachePath, "# previous recap\n")
localSourceAudio := filepath.Join(filepath.Dir(cfg.SessionPath), "audio", "alice.flac")
mustWriteFile(t, localSourceAudio, "source\n")
seed := manifest.New(cfg.Session.SessionID, time.Now().UTC())
seed.Campaign = cfg.Session.Campaign
seed.RunID = runID
seed.LocalWorkDir = runWorkDir
seed.LocalSpoolDir = spoolAudioDir
seed.S3Bucket = "my-dnd-archive"
seed.S3SessionPrefix = "dnd/campaigns/sample-campaign/sessions/2026-05-03/"
seed.S3RunPrefix = seed.S3SessionPrefix + "runs/" + runID + "/"
store := &manifest.LocalStore{}
if err := os.MkdirAll(filepath.Dir(manifestPathFor(cfg)), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
if err := store.Save(context.Background(), manifestPathFor(cfg), seed); err != nil {
t.Fatalf("seed manifest save error = %v", err)
}
return cfg, cleanupSeed{
runWorkDir: runWorkDir,
otherRunDir: otherRunDir,
spoolAudioDir: spoolAudioDir,
localSourceAudio: localSourceAudio,
previousCachePath: previousCachePath,
sessionPrefix: seed.S3SessionPrefix,
}
}
func archiveStageCleanupFixture(t *testing.T) (*config.Config, cleanupSeed, string) {
t.Helper()
cfg, seed := cleanupFixtureConfig(t)
runID := "20260516T010203Z-1a2b3c4d"
cfg.Pipeline.Storage.S3 = &config.StorageS3Config{
Bucket: "my-dnd-archive",
RootPrefix: "dnd",
}
cfg.Pipeline.Archive = &config.ArchiveConfig{
Enabled: boolPtr(true),
UploadRun: boolPtr(true),
PromoteArtifacts: []config.ArchivePromotionRule{
{Source: "narratio.transcript.final_trimmed", Dest: "transcripts/final.trimmed.json", Required: boolPtr(true)},
{Source: "narratio.artifact.session_recap", Dest: "artifacts/session_recap.md", Required: boolPtr(true)},
},
}
cfg.Pipeline.Scriptorium = &config.ScriptoriumConfig{
Artifacts: map[string]config.ScriptoriumArtifactConfig{
"session_recap": {
OutputPath: "artifacts/session_recap.md",
},
},
}
writeArchiveFixtureRunFiles(
t,
seed.runWorkDir,
artifacts.SessionWorkDirForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID),
)
store := &manifest.LocalStore{}
seedManifest, err := store.Load(context.Background(), manifestPathFor(cfg))
if err != nil {
t.Fatalf("Load() error = %v", err)
}
for _, name := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "analyze"} {
seedManifest.MarkStageSucceeded(name, time.Now().UTC(), nil)
}
seedManifest.S3SessionPrefix = artifacts.S3SessionPrefix("dnd", cfg.Session.Campaign, cfg.Session.SessionID)
seedManifest.S3RunPrefix = artifacts.S3RunPrefix(seedManifest.S3SessionPrefix, runID)
if err := store.Save(context.Background(), manifestPathFor(cfg), seedManifest); err != nil {
t.Fatalf("Save() error = %v", err)
}
seed.sessionPrefix = seedManifest.S3SessionPrefix
return cfg, seed, runID
}
func writeArchiveFixtureRunFiles(t *testing.T, runWorkDir, sessionRoot string) {
t.Helper()
mustWriteFile(t, filepath.Join(runWorkDir, "prepare", "inputs", "session.yml"), "session_id: 2026-05-03\n")
mustWriteFile(t, filepath.Join(runWorkDir, "transcribe", "outputs", "transcripts", "raw", "speaker.json"), "{}\n")
mustWriteFile(t, filepath.Join(runWorkDir, "trim", "outputs", "transcripts", "final.trimmed.json"), "{\"segments\":[]}\n")
mustWriteFile(t, filepath.Join(runWorkDir, "analyze", "outputs", "artifacts", "session_recap.md"), "# recap\n")
mustWriteFile(t, filepath.Join(runWorkDir, "polish", "reports", "audita.report.json"), "{}\n")
mustWriteFile(t, filepath.Join(runWorkDir, "merge", "config", "seriatim.generated.yml"), "key: value\n")
mustWriteFile(t, filepath.Join(runWorkDir, "logs", "audita.stderr.log"), "stderr\n")
mustWriteFile(t, filepath.Join(runWorkDir, "manifest.json"), "{}\n")
mustWriteFile(t, filepath.Join(sessionRoot, "transcripts", "final.trimmed.json"), "{\"segments\":[]}\n")
mustWriteFile(t, filepath.Join(sessionRoot, "artifacts", "session_recap.md"), "# recap\n")
}
type failKeyStore struct {
delegate *storage.FakeBackend
failKey string
}
func (s *failKeyStore) List(ctx context.Context, prefix string) ([]storage.ObjectInfo, error) {
return s.delegate.List(ctx, prefix)
}
func (s *failKeyStore) Download(ctx context.Context, key, localPath string) error {
return s.delegate.Download(ctx, key, localPath)
}
func (s *failKeyStore) Upload(ctx context.Context, localPath, key string, opts storage.UploadOptions) (storage.ObjectInfo, error) {
if strings.TrimSpace(key) == strings.TrimSpace(s.failKey) {
return storage.ObjectInfo{}, errors.New("forced upload failure")
}
return s.delegate.Upload(ctx, localPath, key, opts)
}
func (s *failKeyStore) Exists(ctx context.Context, key string) (bool, error) {
return s.delegate.Exists(ctx, key)
}
func assertExists(t *testing.T, path string) {
t.Helper()
if _, err := os.Stat(path); err != nil {
t.Fatalf("expected path to exist %q: %v", path, err)
}
}
func assertMissing(t *testing.T, path string) {
t.Helper()
if _, err := os.Stat(path); !os.IsNotExist(err) {
t.Fatalf("expected path to be removed %q, stat err=%v", path, err)
}
}

View File

@@ -0,0 +1,157 @@
package app
import (
"context"
"fmt"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
type effectiveLocks struct {
Static []config.ArchiveLockRule
Remote []config.ArchiveLockRule
All []config.ArchiveLockRule
Key string
}
func remoteLocksKey(cfg *config.Config) (string, error) {
if cfg == nil || cfg.Pipeline == nil || cfg.Session == nil {
return "", fmt.Errorf("resolved config is required")
}
if cfg.Pipeline.Storage.S3 == nil {
return "", fmt.Errorf("pipeline.storage.s3 configuration is required")
}
sessionPrefix := artifacts.S3SessionPrefix(
cfg.Pipeline.Storage.S3.RootPrefix,
cfg.Session.Campaign,
cfg.Session.SessionID,
)
return artifacts.S3SessionLocksKey(sessionPrefix), nil
}
func loadRemoteLockStore(ctx context.Context, cfg *config.Config, store storage.ObjectStore) (*config.ArchiveLockStore, string, error) {
key, err := remoteLocksKey(cfg)
if err != nil {
return nil, "", err
}
exists, err := store.Exists(ctx, key)
if err != nil {
return nil, key, fmt.Errorf("check remote locks %q: %w", key, err)
}
if !exists {
return &config.ArchiveLockStore{}, key, nil
}
tmp, err := downloadObjectToTemp(ctx, store, key, "narratio-locks-*.yml")
if err != nil {
return nil, key, fmt.Errorf("download remote locks %q: %w", key, err)
}
defer func() { _ = os.Remove(tmp) }()
data, err := os.ReadFile(tmp)
if err != nil {
return nil, key, fmt.Errorf("read remote locks %q: %w", key, err)
}
lockStore, err := config.LoadArchiveLockStoreBytes("s3://"+s3BucketName(cfg.Pipeline)+"/"+key, data, cfg.Pipeline.Scriptorium)
if err != nil {
return nil, key, err
}
return lockStore, key, nil
}
func loadEffectiveLocks(ctx context.Context, cfg *config.Config, store storage.ObjectStore) (*effectiveLocks, error) {
staticLocks := staticArchiveLocks(cfg)
if store == nil {
return &effectiveLocks{
Static: staticLocks,
All: append([]config.ArchiveLockRule(nil), staticLocks...),
}, nil
}
lockStore, key, err := loadRemoteLockStore(ctx, cfg, store)
if err != nil {
return nil, err
}
remoteLocks := append([]config.ArchiveLockRule(nil), lockStore.Locks...)
return &effectiveLocks{
Static: staticLocks,
Remote: remoteLocks,
All: config.MergeArchiveLockRules(staticLocks, remoteLocks),
Key: key,
}, nil
}
func staticArchiveLocks(cfg *config.Config) []config.ArchiveLockRule {
if cfg == nil || cfg.Pipeline == nil || cfg.Pipeline.Archive == nil {
return nil
}
return append([]config.ArchiveLockRule(nil), cfg.Pipeline.Archive.Locks...)
}
func applyEffectiveLocks(cfg *config.Config, locks []config.ArchiveLockRule) {
if cfg == nil || cfg.Pipeline == nil {
return
}
if cfg.Pipeline.Archive == nil {
cfg.Pipeline.Archive = &config.ArchiveConfig{}
}
cfg.Pipeline.Archive.Locks = append([]config.ArchiveLockRule(nil), locks...)
}
func uploadRemoteLockStore(ctx context.Context, store storage.ObjectStore, key string, lockStore *config.ArchiveLockStore) error {
data, err := config.MarshalArchiveLockStore(lockStore)
if err != nil {
return err
}
tmp, err := os.CreateTemp("", "narratio-locks-upload-*.yml")
if err != nil {
return fmt.Errorf("create lock store temp file: %w", err)
}
tmpPath := tmp.Name()
defer func() { _ = os.Remove(tmpPath) }()
if _, err := tmp.Write(data); err != nil {
_ = tmp.Close()
return fmt.Errorf("write lock store temp file: %w", err)
}
if err := tmp.Close(); err != nil {
return fmt.Errorf("close lock store temp file: %w", err)
}
if _, err := store.Upload(ctx, tmpPath, key, storage.UploadOptions{ContentType: "application/x-yaml; charset=utf-8"}); err != nil {
return fmt.Errorf("upload remote locks %q: %w", key, err)
}
return nil
}
func lockSourceSet(locks []config.ArchiveLockRule) map[string]config.ArchiveLockRule {
out := make(map[string]config.ArchiveLockRule, len(locks))
for _, lock := range locks {
source := strings.TrimSpace(lock.Source)
if source == "" {
continue
}
lock.Source = source
lock.Reason = strings.TrimSpace(lock.Reason)
out[source] = lock
}
return out
}
func writeLocalFile(path string, data []byte, force bool) error {
cleaned := filepath.Clean(strings.TrimSpace(path))
if cleaned == "" || cleaned == "." {
return fmt.Errorf("output path is required")
}
if !force {
if _, err := os.Stat(cleaned); err == nil {
return fmt.Errorf("output file %q already exists; pass --force to overwrite", cleaned)
} else if err != nil && !os.IsNotExist(err) {
return fmt.Errorf("check output file %q: %w", cleaned, err)
}
}
if err := os.MkdirAll(filepath.Dir(cleaned), 0o755); err != nil {
return fmt.Errorf("create output directory: %w", err)
}
return os.WriteFile(cleaned, data, 0o644)
}

View File

@@ -0,0 +1,294 @@
package app
import (
"bytes"
"context"
"errors"
"fmt"
"os"
"path/filepath"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestExecuteRemoteSessionFallbackLoadsFromObjectStore(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
remoteKey := seedRemoteSessionConfig(t, fake, "2026-05-03", `session_id: 2026-05-03
inputs:
audio_s3:
prefix: audio/
`)
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{filepath.Join(t.TempDir(), "session.yml")})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "plan", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if storeInitCalls != 1 {
t.Fatalf("object store init calls = %d, want 1", storeInitCalls)
}
if !strings.Contains(stdout.String(), "narratio session plan: workdir prepared") {
t.Fatalf("stdout = %q, want plan output", stdout.String())
}
if _, ok := fake.Objects[remoteKey]; !ok {
t.Fatalf("remote session key %q was not seeded", remoteKey)
}
}
func TestExecuteRemoteSessionFallbackLoadsSecretsBeforeObjectStoreInit(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
accessKeyEnv := "NARRATIO_TEST_REMOTE_SESSION_KEY_ID"
secretKeyEnv := "NARRATIO_TEST_REMOTE_SESSION_SECRET"
restoreEnvAfterTest(t, accessKeyEnv, secretKeyEnv)
secretsDir := t.TempDir()
mustWriteTestFile(t, filepath.Join(secretsDir, accessKeyEnv), "remote-session-key-id\n")
mustWriteTestFile(t, filepath.Join(secretsDir, secretKeyEnv), "remote-session-secret\n")
addSecretsToPipelineConfig(t, pipelinePath, secretsDir, accessKeyEnv, secretKeyEnv)
fake := &storage.FakeBackend{}
seedRemoteSessionConfig(t, fake, "2026-05-03", `session_id: 2026-05-03
inputs:
audio_s3:
prefix: audio/
`)
origStoreFn := newObjectStoreFromConfigFn
origSessionDefaults := append([]string(nil), config.DefaultSessionConfigSearchPaths...)
config.DefaultSessionConfigSearchPaths = []string{filepath.Join(t.TempDir(), "session.yml")}
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
if os.Getenv(accessKeyEnv) != "remote-session-key-id" || os.Getenv(secretKeyEnv) != "remote-session-secret" {
return nil, fmt.Errorf("secrets were not loaded before remote session object store init")
}
return fake, nil
}
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
config.DefaultSessionConfigSearchPaths = origSessionDefaults
})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "plan", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stdout=%q stderr=%q", code, stdout.String(), stderr.String())
}
}
func TestExecuteExplicitLocalSessionPrecedenceSkipsRemote(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{filepath.Join(t.TempDir(), "session.yml")})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "plan", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if storeInitCalls != 0 {
t.Fatalf("object store init calls = %d, want 0", storeInitCalls)
}
}
func TestExecuteLocalSessionDiscoveryPrecedenceSkipsRemote(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{sessionPath})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "plan", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if storeInitCalls != 0 {
t.Fatalf("object store init calls = %d, want 0", storeInitCalls)
}
}
func TestExecuteRemoteSessionMissingObjectFailsClearly(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
var storeInitCalls int
missingSessionPath := filepath.Join(t.TempDir(), "session.yml")
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{missingSessionPath})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "plan", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "remote session") || !strings.Contains(stderr.String(), "session.yml") || !strings.Contains(stderr.String(), "not found") {
t.Fatalf("stderr = %q, want remote session not found context", stderr.String())
}
if !strings.Contains(stderr.String(), missingSessionPath) {
t.Fatalf("stderr = %q, want local searched path", stderr.String())
}
}
func TestExecuteRemoteSessionRequiresSessionID(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
var storeInitCalls int
restoreAppConfigTestGlobals(t, &storage.FakeBackend{}, &storeInitCalls, []string{filepath.Join(t.TempDir(), "session.yml")})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "plan", "--config", pipelinePath, "--campaign-file", campaignPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "plan: session_id is required") {
t.Fatalf("stderr = %q, want session_id guidance", stderr.String())
}
if storeInitCalls != 0 {
t.Fatalf("object store init calls = %d, want 0", storeInitCalls)
}
}
func TestExecuteRemoteSessionStorageInitErrorFailsClearly(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
origStoreFn := newObjectStoreFromConfigFn
origSessionDefaults := append([]string(nil), config.DefaultSessionConfigSearchPaths...)
config.DefaultSessionConfigSearchPaths = []string{filepath.Join(t.TempDir(), "session.yml")}
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
return nil, errors.New("storage unavailable")
}
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
config.DefaultSessionConfigSearchPaths = origSessionDefaults
})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "plan", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "storage unavailable") || !strings.Contains(stderr.String(), "remote session") {
t.Fatalf("stderr = %q, want remote storage context", stderr.String())
}
}
func TestExecuteRemoteSessionMalformedYAMLFailsStrictDecode(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
seedRemoteSessionConfig(t, fake, "2026-05-03", "session_id: 2026-05-03\nunknown: true\n")
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{filepath.Join(t.TempDir(), "session.yml")})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "plan", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "strict decode failed") {
t.Fatalf("stderr = %q, want strict decode context", stderr.String())
}
}
func TestExecuteRemoteSessionTemplateFailsConcreteSessionCheck(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
seedRemoteSessionConfig(t, fake, "2026-05-03", `session_id: "{{ session_id }}"
inputs:
audio_s3:
prefix: audio/
`)
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{filepath.Join(t.TempDir(), "session.yml")})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "plan", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "session.yml must be concrete") || !strings.Contains(stderr.String(), "run narratio session init") {
t.Fatalf("stderr = %q, want concrete session guidance", stderr.String())
}
}
func TestExecuteRemoteSessionMismatchFails(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
seedRemoteSessionConfig(t, fake, "2026-05-03", "session_id: 2026-05-04\ninputs:\n audio_s3:\n prefix: audio/\n")
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{filepath.Join(t.TempDir(), "session.yml")})
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "plan", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "session_id mismatch") {
t.Fatalf("stderr = %q, want session_id mismatch", stderr.String())
}
}
func restoreAppConfigTestGlobals(t *testing.T, fake *storage.FakeBackend, storeInitCalls *int, sessionDefaults []string) {
t.Helper()
origStoreFn := newObjectStoreFromConfigFn
origSessionDefaults := append([]string(nil), config.DefaultSessionConfigSearchPaths...)
config.DefaultSessionConfigSearchPaths = append([]string(nil), sessionDefaults...)
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
if storeInitCalls != nil {
(*storeInitCalls)++
}
return fake, nil
}
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
config.DefaultSessionConfigSearchPaths = origSessionDefaults
})
}
func seedRemoteSessionConfig(t *testing.T, fake *storage.FakeBackend, sessionID, content string) string {
t.Helper()
sessionPrefix := artifacts.S3SessionPrefix("dnd", "sample-campaign", sessionID)
remoteKey := artifacts.S3SessionConfigKey(sessionPrefix)
fake.SeedObject(storage.FakeObject{
Key: remoteKey,
Data: []byte(content),
ETag: "remote-session-etag",
})
return remoteKey
}
func addSecretsToPipelineConfig(t *testing.T, pipelinePath, secretsDir, accessKeyEnv, secretKeyEnv string) {
t.Helper()
pipelineData, err := os.ReadFile(pipelinePath)
if err != nil {
t.Fatalf("read pipeline: %v", err)
}
pipelineYAML := strings.Replace(
string(pipelineData),
"storage:\n backend: s3\n s3:\n bucket: test-bucket\n",
"storage:\n backend: s3\n s3:\n bucket: test-bucket\n access_key_id_env: "+accessKeyEnv+"\n secret_access_key_env: "+secretKeyEnv+"\nsecrets:\n env_dir: "+secretsDir+"\n",
1,
)
if err := os.WriteFile(pipelinePath, []byte(pipelineYAML), 0o644); err != nil {
t.Fatalf("write pipeline: %v", err)
}
}

159
internal/app/restore.go Normal file
View File

@@ -0,0 +1,159 @@
package app
import (
"context"
"errors"
"flag"
"fmt"
"io"
"log/slog"
"os"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/logging"
)
var newObjectStoreFromConfigFn = storage.NewObjectStoreFromConfig
var discoverRemoteCurrentStateFn = discoverRemoteCurrentState
var buildRestorePlanFn = buildRestorePlan
var executeRestorePlanFn = executeRestorePlan
// Restore validates restore CLI/config inputs and storage preflight for future restore phases.
func Restore(ctx context.Context, args []string, out io.Writer) error {
positionalSessionID, args := pullLeadingSessionID(args)
fs := flag.NewFlagSet("restore", flag.ContinueOnError)
fs.SetOutput(out)
var pipelinePath string
var campaignPath string
var campaignFilePath string
var sessionPath string
var sessionID string
var previousSessionID string
var dryRun bool
var force bool
var includeAudio bool
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
fs.StringVar(&campaignPath, "campaign", "", "campaign ID")
fs.StringVar(&campaignFilePath, "campaign-file", "", "path to campaign.yml")
fs.StringVar(&sessionPath, "session", "", "path to session.yml")
fs.StringVar(&previousSessionID, "previous-session-id", "", "expected previous session identifier")
fs.BoolVar(&dryRun, "dry-run", false, "plan restore actions without writing local files")
fs.BoolVar(&force, "force", false, "overwrite local conflicts with remote state")
fs.BoolVar(&includeAudio, "include-audio", false, "include archived session-level audio objects")
fs.Usage = func() {
_, _ = fmt.Fprintln(out, "Usage: narratio session restore <session_id> [--config <path>] [--campaign <id>] [--campaign-file <path>] [--session <path>] [--previous-session-id <value>] [--dry-run] [--force] [--include-audio]")
_, _ = fmt.Fprintln(out)
_, _ = fmt.Fprintln(out, "Flags:")
fs.PrintDefaults()
}
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
return nil
}
return fmt.Errorf("restore: invalid flags: %w", err)
}
if positionalSessionID == "" {
if err := applyParsedSessionIDArg("restore", fs, &sessionID); err != nil {
return err
}
} else {
if fs.NArg() != 0 {
return fmt.Errorf("restore: unexpected positional arguments")
}
if err := applyPositionalSessionID("restore", positionalSessionID, &sessionID); err != nil {
return err
}
}
if strings.TrimSpace(sessionID) == "" {
return fmt.Errorf("restore: session_id is required")
}
cfg, err := loadCommandConfig(ctx, pipelinePath, campaignPath, campaignFilePath, sessionPath, config.SessionLoadOptions{
SessionID: sessionID,
PreviousSessionID: previousSessionID,
})
if err != nil {
return fmt.Errorf("restore: %w", err)
}
if err := config.Validate(cfg); err != nil {
return fmt.Errorf("restore: %w", err)
}
objectStore, err := newCommandObjectStore(ctx, cfg, logging.NewLogger(os.Stderr, slog.LevelInfo))
if err != nil {
return fmt.Errorf("restore: %w", err)
}
current, err := discoverRemoteCurrentStateFn(ctx, cfg, objectStore)
if err != nil {
return fmt.Errorf("restore: %w", err)
}
plan, err := buildRestorePlanFn(ctx, cfg, current, objectStore, RestorePlanOptions{
IncludeAudio: includeAudio,
Force: force,
DryRun: dryRun,
})
if err != nil {
return fmt.Errorf("restore: %w", err)
}
report, err := newRestoreReport(current, plan, RestorePlanOptions{
IncludeAudio: includeAudio,
Force: force,
DryRun: dryRun,
})
if err != nil {
return fmt.Errorf("restore: %w", err)
}
if dryRun {
if err := writeRestoreDryRunSummary(out, report); err != nil {
return fmt.Errorf("restore: write plan output: %w", err)
}
return nil
}
artifactStore := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root)
if _, err := artifactStore.EnsureLayoutFor(cfg.Session.Campaign, cfg.Session.SessionID); err != nil {
return fmt.Errorf("restore: prepare workdir: %w", err)
}
lock, err := artifactStore.AcquireSessionLockFor(cfg.Session.Campaign, cfg.Session.SessionID)
if err != nil {
return fmt.Errorf("restore: acquire session lock: %w", err)
}
defer func() {
_ = artifactStore.ReleaseSessionLock(lock)
}()
if plan.ConflictCount > 0 && !force {
report.setFailed(fmt.Errorf("conflict: %d conflicting path(s)", plan.ConflictCount))
if _, reportErr := persistRestoreReport(artifactStore, cfg, report); reportErr != nil {
return fmt.Errorf("restore: report failure: %w", reportErr)
}
return fmt.Errorf(
"restore conflict: %d conflicting path(s); rerun with --force to overwrite (download=%d skip_same=%d conflicts=%d)",
plan.ConflictCount,
plan.DownloadCount,
plan.SkipSameCount,
plan.ConflictCount,
)
}
result, err := executeRestorePlanFn(ctx, cfg, current, plan, report, objectStore)
if err != nil {
report.setFailed(err)
if _, reportErr := persistRestoreReport(artifactStore, cfg, report); reportErr != nil {
return fmt.Errorf("restore: execute plan failed (%v) and report write failed (%v)", err, reportErr)
}
return fmt.Errorf("restore: execute plan: %w", err)
}
report.Execution.Downloaded = result.DownloadedCount
report.setSucceeded()
if _, err := persistRestoreReport(artifactStore, cfg, report); err != nil {
return fmt.Errorf("restore: write report: %w", err)
}
if err := writeRestoreSuccessSummary(out, report); err != nil {
return fmt.Errorf("restore: write summary: %w", err)
}
return nil
}

View File

@@ -0,0 +1,139 @@
package app
import (
"context"
"fmt"
"os"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
// RemoteCurrentState captures discovered committed remote archive state for one session.
type RemoteCurrentState struct {
Bucket string
SessionPrefix string
CurrentRunIDKey string
CurrentManifestKey string
RunID string
SessionID string
Campaign string
Manifest *manifest.Manifest
}
func discoverRemoteCurrentState(ctx context.Context, cfg *config.Config, store storage.ObjectStore) (*RemoteCurrentState, error) {
if cfg == nil || cfg.Pipeline == nil || cfg.Session == nil {
return nil, fmt.Errorf("resolved config with pipeline/session is required")
}
if store == nil {
return nil, fmt.Errorf("remote object store is required")
}
bucket := artifacts.ResolveArchiveBucket(cfg, nil)
if strings.TrimSpace(bucket) == "" {
return nil, fmt.Errorf("archive bucket is required")
}
sessionPrefix, err := artifacts.ResolveArchiveSessionPrefix(cfg, nil)
if err != nil {
return nil, fmt.Errorf("resolve archive session prefix: %w", err)
}
currentManifestKey, currentRunIDKey := artifacts.ResolveArchiveCurrentStateKeys(sessionPrefix)
exists, err := store.Exists(ctx, currentRunIDKey)
if err != nil {
return nil, fmt.Errorf("check remote current run pointer %q: %w", currentRunIDKey, err)
}
if !exists {
return nil, fmt.Errorf("remote current run pointer missing: %q", currentRunIDKey)
}
runIDPath, err := downloadObjectToTemp(ctx, store, currentRunIDKey, "narratio-restore-current-run-id-*.txt")
if err != nil {
return nil, fmt.Errorf("download remote current run pointer %q: %w", currentRunIDKey, err)
}
defer func() { _ = os.Remove(runIDPath) }()
runIDData, err := os.ReadFile(runIDPath)
if err != nil {
return nil, fmt.Errorf("read downloaded run pointer %q: %w", currentRunIDKey, err)
}
runID := strings.TrimSpace(string(runIDData))
if runID == "" {
return nil, fmt.Errorf("remote current run pointer %q is empty", currentRunIDKey)
}
exists, err = store.Exists(ctx, currentManifestKey)
if err != nil {
return nil, fmt.Errorf("check remote current manifest %q: %w", currentManifestKey, err)
}
if !exists {
return nil, fmt.Errorf("remote current manifest missing: %q", currentManifestKey)
}
manifestPath, err := downloadObjectToTemp(ctx, store, currentManifestKey, "narratio-restore-current-manifest-*.json")
if err != nil {
return nil, fmt.Errorf("download remote current manifest %q: %w", currentManifestKey, err)
}
defer func() { _ = os.Remove(manifestPath) }()
manifestStore := &manifest.LocalStore{}
remoteManifest, err := manifestStore.Load(ctx, manifestPath)
if err != nil {
return nil, fmt.Errorf("remote current manifest decode failed: %w", err)
}
requestedSession := strings.TrimSpace(cfg.Session.SessionID)
requestedCampaign := strings.TrimSpace(cfg.Session.Campaign)
manifestSession := strings.TrimSpace(remoteManifest.SessionID)
manifestCampaign := strings.TrimSpace(remoteManifest.Campaign)
if manifestSession != requestedSession {
return nil, fmt.Errorf(
"remote current manifest session_id %q does not match requested session_id %q",
manifestSession,
requestedSession,
)
}
if manifestCampaign == "" {
return nil, fmt.Errorf("remote current manifest campaign is required")
}
if manifestCampaign != requestedCampaign {
return nil, fmt.Errorf(
"remote current manifest campaign %q does not match requested campaign %q",
manifestCampaign,
requestedCampaign,
)
}
return &RemoteCurrentState{
Bucket: bucket,
SessionPrefix: sessionPrefix,
CurrentRunIDKey: currentRunIDKey,
CurrentManifestKey: currentManifestKey,
RunID: runID,
SessionID: manifestSession,
Campaign: manifestCampaign,
Manifest: remoteManifest,
}, nil
}
func downloadObjectToTemp(ctx context.Context, store storage.ObjectStore, key, pattern string) (string, error) {
tmp, err := os.CreateTemp("", pattern)
if err != nil {
return "", fmt.Errorf("create temp file: %w", err)
}
path := tmp.Name()
if err := tmp.Close(); err != nil {
_ = os.Remove(path)
return "", fmt.Errorf("close temp file: %w", err)
}
if err := store.Download(ctx, key, path); err != nil {
_ = os.Remove(path)
return "", err
}
return path, nil
}

View File

@@ -0,0 +1,239 @@
package app
import (
"context"
"encoding/json"
"fmt"
"strings"
"testing"
"time"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestDiscoverRemoteCurrentStateSuccess(t *testing.T) {
cfg := restoreDiscoveryConfig()
store := &storage.FakeBackend{}
sessionPrefix, manifestKey, runIDKey := restoreDiscoveryKeys(cfg)
store.SeedObject(storage.FakeObject{Key: runIDKey, Data: []byte("20260519T010203Z-a1b2c3d4\n")})
store.SeedObject(storage.FakeObject{Key: manifestKey, Data: restoreManifestJSON(t, cfg.Session.SessionID, cfg.Session.Campaign)})
state, err := discoverRemoteCurrentState(context.Background(), cfg, store)
if err != nil {
t.Fatalf("discoverRemoteCurrentState() error = %v", err)
}
if state.RunID != "20260519T010203Z-a1b2c3d4" {
t.Fatalf("run id = %q, want 20260519T010203Z-a1b2c3d4", state.RunID)
}
if state.SessionPrefix != sessionPrefix {
t.Fatalf("session prefix = %q, want %q", state.SessionPrefix, sessionPrefix)
}
if state.CurrentRunIDKey != runIDKey {
t.Fatalf("current run id key = %q, want %q", state.CurrentRunIDKey, runIDKey)
}
if state.CurrentManifestKey != manifestKey {
t.Fatalf("current manifest key = %q, want %q", state.CurrentManifestKey, manifestKey)
}
if state.Manifest == nil {
t.Fatal("manifest is nil")
}
}
func TestDiscoverRemoteCurrentStateMissingRunPointerFails(t *testing.T) {
cfg := restoreDiscoveryConfig()
store := &storage.FakeBackend{}
_, err := discoverRemoteCurrentState(context.Background(), cfg, store)
if err == nil || !strings.Contains(err.Error(), "remote current run pointer missing") {
t.Fatalf("error = %v, want missing run pointer failure", err)
}
}
func TestDiscoverRemoteCurrentStateEmptyRunPointerFails(t *testing.T) {
cfg := restoreDiscoveryConfig()
store := &storage.FakeBackend{}
_, manifestKey, runIDKey := restoreDiscoveryKeys(cfg)
store.SeedObject(storage.FakeObject{Key: runIDKey, Data: []byte(" \n\t")})
store.SeedObject(storage.FakeObject{Key: manifestKey, Data: restoreManifestJSON(t, cfg.Session.SessionID, cfg.Session.Campaign)})
_, err := discoverRemoteCurrentState(context.Background(), cfg, store)
if err == nil || !strings.Contains(err.Error(), "is empty") {
t.Fatalf("error = %v, want empty run pointer failure", err)
}
}
func TestDiscoverRemoteCurrentStateMissingManifestFails(t *testing.T) {
cfg := restoreDiscoveryConfig()
store := &storage.FakeBackend{}
_, _, runIDKey := restoreDiscoveryKeys(cfg)
store.SeedObject(storage.FakeObject{Key: runIDKey, Data: []byte("20260519T010203Z-a1b2c3d4\n")})
_, err := discoverRemoteCurrentState(context.Background(), cfg, store)
if err == nil || !strings.Contains(err.Error(), "remote current manifest missing") {
t.Fatalf("error = %v, want missing manifest failure", err)
}
}
func TestDiscoverRemoteCurrentStateInvalidManifestFails(t *testing.T) {
cfg := restoreDiscoveryConfig()
store := &storage.FakeBackend{}
_, manifestKey, runIDKey := restoreDiscoveryKeys(cfg)
store.SeedObject(storage.FakeObject{Key: runIDKey, Data: []byte("20260519T010203Z-a1b2c3d4\n")})
store.SeedObject(storage.FakeObject{Key: manifestKey, Data: []byte("{invalid json")})
_, err := discoverRemoteCurrentState(context.Background(), cfg, store)
if err == nil || !strings.Contains(err.Error(), "remote current manifest decode failed") {
t.Fatalf("error = %v, want manifest decode failure", err)
}
}
func TestDiscoverRemoteCurrentStateSessionMismatchFails(t *testing.T) {
cfg := restoreDiscoveryConfig()
store := &storage.FakeBackend{}
_, manifestKey, runIDKey := restoreDiscoveryKeys(cfg)
store.SeedObject(storage.FakeObject{Key: runIDKey, Data: []byte("20260519T010203Z-a1b2c3d4\n")})
store.SeedObject(storage.FakeObject{Key: manifestKey, Data: restoreManifestJSON(t, "wrong-session", cfg.Session.Campaign)})
_, err := discoverRemoteCurrentState(context.Background(), cfg, store)
if err == nil || !strings.Contains(err.Error(), "does not match requested session_id") {
t.Fatalf("error = %v, want session mismatch failure", err)
}
}
func TestDiscoverRemoteCurrentStateCampaignMismatchFails(t *testing.T) {
cfg := restoreDiscoveryConfig()
store := &storage.FakeBackend{}
_, manifestKey, runIDKey := restoreDiscoveryKeys(cfg)
store.SeedObject(storage.FakeObject{Key: runIDKey, Data: []byte("20260519T010203Z-a1b2c3d4\n")})
store.SeedObject(storage.FakeObject{Key: manifestKey, Data: restoreManifestJSON(t, cfg.Session.SessionID, "wrong-campaign")})
_, err := discoverRemoteCurrentState(context.Background(), cfg, store)
if err == nil || !strings.Contains(err.Error(), "does not match requested campaign") {
t.Fatalf("error = %v, want campaign mismatch failure", err)
}
}
func TestDiscoverRemoteCurrentStateEmptyCampaignFails(t *testing.T) {
cfg := restoreDiscoveryConfig()
store := &storage.FakeBackend{}
_, manifestKey, runIDKey := restoreDiscoveryKeys(cfg)
store.SeedObject(storage.FakeObject{Key: runIDKey, Data: []byte("20260519T010203Z-a1b2c3d4\n")})
store.SeedObject(storage.FakeObject{Key: manifestKey, Data: restoreManifestJSON(t, cfg.Session.SessionID, "")})
_, err := discoverRemoteCurrentState(context.Background(), cfg, store)
if err == nil || !strings.Contains(err.Error(), "campaign is required") {
t.Fatalf("error = %v, want empty campaign failure", err)
}
}
func TestDiscoverRemoteCurrentStateUsesCurrentKeysUnderSessionPrefix(t *testing.T) {
cfg := restoreDiscoveryConfig()
sessionPrefix, manifestKey, runIDKey := restoreDiscoveryKeys(cfg)
base := &storage.FakeBackend{}
store := &captureObjectStore{delegate: base}
base.SeedObject(storage.FakeObject{Key: runIDKey, Data: []byte("20260519T010203Z-a1b2c3d4\n")})
base.SeedObject(storage.FakeObject{Key: manifestKey, Data: restoreManifestJSON(t, cfg.Session.SessionID, cfg.Session.Campaign)})
_, err := discoverRemoteCurrentState(context.Background(), cfg, store)
if err != nil {
t.Fatalf("discoverRemoteCurrentState() error = %v", err)
}
expectedRunKey := fmt.Sprintf("%scurrent/run_id.txt", sessionPrefix)
expectedManifestKey := fmt.Sprintf("%scurrent/manifest.json", sessionPrefix)
if !containsString(store.existsKeys, expectedRunKey) {
t.Fatalf("exists keys = %#v, want run pointer key %q", store.existsKeys, expectedRunKey)
}
if !containsString(store.existsKeys, expectedManifestKey) {
t.Fatalf("exists keys = %#v, want manifest key %q", store.existsKeys, expectedManifestKey)
}
if !containsString(store.downloadKeys, expectedRunKey) {
t.Fatalf("download keys = %#v, want run pointer key %q", store.downloadKeys, expectedRunKey)
}
if !containsString(store.downloadKeys, expectedManifestKey) {
t.Fatalf("download keys = %#v, want manifest key %q", store.downloadKeys, expectedManifestKey)
}
}
type captureObjectStore struct {
delegate storage.ObjectStore
existsKeys []string
downloadKeys []string
}
func (s *captureObjectStore) List(ctx context.Context, prefix string) ([]storage.ObjectInfo, error) {
return s.delegate.List(ctx, prefix)
}
func (s *captureObjectStore) Download(ctx context.Context, key, localPath string) error {
s.downloadKeys = append(s.downloadKeys, key)
return s.delegate.Download(ctx, key, localPath)
}
func (s *captureObjectStore) Upload(ctx context.Context, localPath, key string, opts storage.UploadOptions) (storage.ObjectInfo, error) {
return s.delegate.Upload(ctx, localPath, key, opts)
}
func (s *captureObjectStore) Exists(ctx context.Context, key string) (bool, error) {
s.existsKeys = append(s.existsKeys, key)
return s.delegate.Exists(ctx, key)
}
func restoreDiscoveryConfig() *config.Config {
return &config.Config{
Pipeline: &config.PipelineConfig{
Storage: config.StorageConfig{
S3: &config.StorageS3Config{
Bucket: "my-dnd-archive",
RootPrefix: "dnd",
},
},
},
Session: &config.SessionConfig{
SessionID: "2026-05-03",
Campaign: "sample-campaign",
},
}
}
func restoreDiscoveryKeys(cfg *config.Config) (sessionPrefix, manifestKey, runIDKey string) {
sessionPrefix = artifacts.S3SessionPrefix(cfg.Pipeline.Storage.S3.RootPrefix, cfg.Session.Campaign, cfg.Session.SessionID)
manifestKey, runIDKey = artifacts.ResolveArchiveCurrentStateKeys(sessionPrefix)
return sessionPrefix, manifestKey, runIDKey
}
func restoreManifestJSON(t *testing.T, sessionID, campaign string) []byte {
t.Helper()
now := time.Date(2026, 5, 19, 23, 0, 0, 0, time.UTC).Format(time.RFC3339Nano)
payload := map[string]any{
"session_id": sessionID,
"campaign": campaign,
"created_at": now,
"updated_at": now,
"stages": map[string]any{},
}
data, err := json.Marshal(payload)
if err != nil {
t.Fatalf("marshal manifest payload: %v", err)
}
return append(data, '\n')
}
func containsString(values []string, target string) bool {
for _, value := range values {
if value == target {
return true
}
}
return false
}

View File

@@ -0,0 +1,217 @@
package app
import (
"context"
"fmt"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/audio"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
// RestoreExecutionResult captures concrete file-install results for one restore execution.
type RestoreExecutionResult struct {
DownloadedCount int
}
func executeRestorePlan(
ctx context.Context,
cfg *config.Config,
current *RemoteCurrentState,
plan *RestorePlan,
report *RestoreReport,
store storage.ObjectStore,
) (*RestoreExecutionResult, error) {
if cfg == nil || cfg.Pipeline == nil || cfg.Session == nil {
return nil, fmt.Errorf("resolved config with pipeline/session is required")
}
if current == nil {
return nil, fmt.Errorf("remote current state is required")
}
if plan == nil {
return nil, fmt.Errorf("restore plan is required")
}
if store == nil {
return nil, fmt.Errorf("remote object store is required")
}
sessionRoot := artifacts.SessionWorkDirForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID)
manifestActions := make([]RestoreAction, 0, 1)
actions := make([]RestoreAction, 0, len(plan.Actions))
for _, action := range plan.Actions {
if action.Kind != RestoreActionDownload {
continue
}
if action.LocalRelativePath == config.PathManifestFile {
manifestActions = append(manifestActions, action)
continue
}
actions = append(actions, action)
}
if len(manifestActions) > 1 {
return nil, fmt.Errorf("restore plan includes multiple manifest download actions")
}
if len(manifestActions) == 1 {
actions = append(actions, manifestActions[0])
}
result := &RestoreExecutionResult{}
for _, action := range actions {
if err := executeRestoreDownloadAction(ctx, cfg, sessionRoot, current, action, store); err != nil {
if report != nil {
report.markFailed(action, err)
}
return nil, fmt.Errorf("install %q from %q: %w", action.LocalRelativePath, action.RemoteKey, err)
}
if report != nil {
report.markDownloaded(action)
}
result.DownloadedCount++
}
return result, nil
}
func executeRestoreDownloadAction(
ctx context.Context,
cfg *config.Config,
sessionRoot string,
current *RemoteCurrentState,
action RestoreAction,
store storage.ObjectStore,
) error {
safeLocalPath, err := joinWithinSessionRoot(sessionRoot, action.LocalRelativePath)
if err != nil {
return fmt.Errorf("resolve safe local path: %w", err)
}
if strings.TrimSpace(action.LocalPath) != "" && filepath.Clean(action.LocalPath) != safeLocalPath {
return fmt.Errorf("restore plan local path mismatch for %q", action.LocalRelativePath)
}
if restoreActionIsAudio(action) {
return executeRestoreAudioAction(ctx, cfg, safeLocalPath, action, store)
}
tmpPath, err := downloadObjectToSiblingTemp(ctx, store, action.RemoteKey, safeLocalPath)
if err != nil {
return fmt.Errorf("download to temp file: %w", err)
}
removeTmp := true
defer func() {
if removeTmp {
_ = os.Remove(tmpPath)
}
}()
if action.LocalRelativePath == config.PathManifestFile {
if err := validateRestoredManifest(ctx, cfg, current, tmpPath); err != nil {
return err
}
}
if err := os.Chmod(tmpPath, 0o644); err != nil {
return fmt.Errorf("set file permissions: %w", err)
}
if err := os.Rename(tmpPath, safeLocalPath); err != nil {
return fmt.Errorf("install file atomically: %w", err)
}
removeTmp = false
return nil
}
func executeRestoreAudioAction(
ctx context.Context,
cfg *config.Config,
safeLocalPath string,
action RestoreAction,
store storage.ObjectStore,
) error {
if cfg == nil || cfg.Pipeline == nil || cfg.Pipeline.Storage.S3 == nil || cfg.Session == nil {
return fmt.Errorf("resolved s3 config and session are required")
}
spoolDir := artifacts.SessionSpoolRestoreAudioDir(cfg.Pipeline.Spool.Root, cfg.Session.Campaign, cfg.Session.SessionID)
spoolPath := filepath.Join(spoolDir, filepath.Base(safeLocalPath))
cacheEnabled := cfg.Pipeline.Cache.S3Audio == nil || *cfg.Pipeline.Cache.S3Audio
_, err := audio.MaterializeS3Audio(ctx, audio.S3MaterializeRequest{
Store: store,
Object: storage.ObjectInfo{
Key: action.RemoteKey,
Size: action.Size,
ETag: action.ETag,
},
Bucket: strings.TrimSpace(cfg.Pipeline.Storage.S3.Bucket),
CacheRoot: strings.TrimSpace(cfg.Pipeline.Cache.Root),
CacheEnabled: cacheEnabled,
SpoolPath: spoolPath,
DestPath: safeLocalPath,
})
if err != nil {
return fmt.Errorf("materialize audio: %w", err)
}
return nil
}
func downloadObjectToSiblingTemp(ctx context.Context, store storage.ObjectStore, remoteKey, destPath string) (string, error) {
if strings.TrimSpace(destPath) == "" {
return "", fmt.Errorf("destination path is required")
}
dir := filepath.Dir(destPath)
if err := os.MkdirAll(dir, 0o755); err != nil {
return "", fmt.Errorf("create destination directory: %w", err)
}
base := filepath.Base(destPath)
tmp, err := os.CreateTemp(dir, "."+base+".restore-*.tmp")
if err != nil {
return "", fmt.Errorf("create temp file: %w", err)
}
tmpPath := tmp.Name()
if err := tmp.Close(); err != nil {
_ = os.Remove(tmpPath)
return "", fmt.Errorf("close temp file: %w", err)
}
if err := store.Download(ctx, remoteKey, tmpPath); err != nil {
_ = os.Remove(tmpPath)
return "", err
}
return tmpPath, nil
}
func validateRestoredManifest(ctx context.Context, cfg *config.Config, current *RemoteCurrentState, path string) error {
manifestStore := &manifest.LocalStore{}
m, err := manifestStore.Load(ctx, path)
if err != nil {
return fmt.Errorf("validate manifest decode: %w", err)
}
requestedSession := strings.TrimSpace(cfg.Session.SessionID)
requestedCampaign := strings.TrimSpace(cfg.Session.Campaign)
manifestSession := strings.TrimSpace(m.SessionID)
manifestCampaign := strings.TrimSpace(m.Campaign)
if manifestSession != requestedSession {
return fmt.Errorf("manifest session_id %q does not match requested session_id %q", manifestSession, requestedSession)
}
if manifestCampaign == "" {
return fmt.Errorf("manifest campaign is required")
}
if manifestCampaign != requestedCampaign {
return fmt.Errorf("manifest campaign %q does not match requested campaign %q", manifestCampaign, requestedCampaign)
}
if current != nil {
if expected := strings.TrimSpace(current.SessionID); expected != "" && manifestSession != expected {
return fmt.Errorf("manifest session_id %q does not match discovered session_id %q", manifestSession, expected)
}
if expected := strings.TrimSpace(current.Campaign); expected != "" && manifestCampaign != expected {
return fmt.Errorf("manifest campaign %q does not match discovered campaign %q", manifestCampaign, expected)
}
}
return nil
}

View File

@@ -0,0 +1,539 @@
package app
import (
"bytes"
"context"
"encoding/json"
"fmt"
"os"
"path/filepath"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
func TestExecuteRestoreNonDryRunRestoresDurableFiles(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
cfg, sessionPrefix, manifestKey, runIDKey := seedRestoreCommittedState(t, fake, pipelinePath, campaignPath, sessionPath)
seedRestoreObject(fake, sessionPrefix+"transcripts/full.json", []byte(`{"segments":[1,2,3]}`))
seedRestoreObject(fake, sessionPrefix+"artifacts/session_recap.md", []byte("# recap\n"))
seedRestoreObject(fake, sessionPrefix+"audio/alice.flac", []byte("remote-audio"))
seedRestoreObject(fake, runIDKey, []byte("20260519T010203Z-a1b2c3d4\n"))
seedRestoreObject(fake, manifestKey, restoreManifestJSON(t, cfg.Session.SessionID, cfg.Session.Campaign))
restoreWithStoreAndRealPhases(t, fake)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if stderr.Len() != 0 {
t.Fatalf("stderr = %q, want empty", stderr.String())
}
if !strings.Contains(stdout.String(), "Restored session archive for sample-campaign/2026-05-03") {
t.Fatalf("stdout = %q, want completion summary", stdout.String())
}
sessionRoot := artifacts.SessionWorkDirForCampaign(workspaceRoot, cfg.Session.Campaign, cfg.Session.SessionID)
mustReadEquals(t, filepath.Join(sessionRoot, "transcripts", "full.json"), `{"segments":[1,2,3]}`)
mustReadEquals(t, filepath.Join(sessionRoot, "artifacts", "session_recap.md"), "# recap\n")
reportPath := filepath.Join(sessionRoot, "reports", "restore-latest.json")
report := mustReadRestoreReport(t, reportPath)
if report.Status != "succeeded" {
t.Fatalf("report status = %q, want succeeded", report.Status)
}
if report.Execution.Downloaded != 3 {
t.Fatalf("report execution.downloaded = %d, want 3", report.Execution.Downloaded)
}
if len(report.Actions) == 0 {
t.Fatal("report actions is empty")
}
if _, err := os.Stat(filepath.Join(sessionRoot, "audio", "alice.flac")); !os.IsNotExist(err) {
t.Fatalf("audio should not be restored by default; stat err=%v", err)
}
}
func TestExecuteRestoreIncludeAudioRestoresAudio(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
cfg, sessionPrefix, _, _ := seedRestoreCommittedState(t, fake, pipelinePath, campaignPath, sessionPath)
seedRestoreObject(fake, sessionPrefix+"audio/alice.flac", []byte("remote-audio"))
restoreWithStoreAndRealPhases(t, fake)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--include-audio"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
sessionRoot := artifacts.SessionWorkDirForCampaign(workspaceRoot, cfg.Session.Campaign, cfg.Session.SessionID)
mustReadEquals(t, filepath.Join(sessionRoot, "audio", "alice.flac"), "remote-audio")
report := mustReadRestoreReport(t, filepath.Join(sessionRoot, "reports", "restore-latest.json"))
if !report.IncludeAudio {
t.Fatalf("report include_audio = %v, want true", report.IncludeAudio)
}
}
func TestExecuteRestoreIncludeAudioUsesCacheAfterWorkspaceDeletion(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
cfg, sessionPrefix, _, _ := seedRestoreCommittedState(t, fake, pipelinePath, campaignPath, sessionPath)
audioKey := sessionPrefix + "audio/alice.flac"
seedRestoreObject(fake, audioKey, []byte("remote-audio"))
restoreWithStoreAndRealPhases(t, fake)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--include-audio"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("first restore exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if got := fakeDownloadCount(fake, audioKey); got != 1 {
t.Fatalf("audio downloads after first restore = %d, want 1", got)
}
sessionRoot := artifacts.SessionWorkDirForCampaign(workspaceRoot, cfg.Session.Campaign, cfg.Session.SessionID)
mustReadEquals(t, filepath.Join(sessionRoot, "audio", "alice.flac"), "remote-audio")
if err := os.RemoveAll(sessionRoot); err != nil {
t.Fatalf("remove session root: %v", err)
}
stdout.Reset()
stderr.Reset()
code = Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--include-audio"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("second restore exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if got := fakeDownloadCount(fake, audioKey); got != 1 {
t.Fatalf("audio downloads after cached restore = %d, want still 1", got)
}
mustReadEquals(t, filepath.Join(sessionRoot, "audio", "alice.flac"), "remote-audio")
}
func TestExecuteRestoreRestoresPreviousCacheWhenPresent(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
appendRestoreWorkflowPreviousInputConfig(t, pipelinePath, sessionPath)
fake := &storage.FakeBackend{}
cfg, _, _, _ := seedRestoreCommittedState(t, fake, pipelinePath, campaignPath, sessionPath)
seedRestorePreviousCurrent(t, fake, cfg, "# previous recap\n")
restoreWithStoreAndRealPhases(t, fake)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
sessionRoot := artifacts.SessionWorkDirForCampaign(workspaceRoot, cfg.Session.Campaign, cfg.Session.SessionID)
previousManifestBytes, err := os.ReadFile(filepath.Join(sessionRoot, "previous", "manifest.json"))
if err != nil {
t.Fatalf("read restored previous manifest: %v", err)
}
if !strings.Contains(string(previousManifestBytes), `"session_id":"2026-04-26"`) {
t.Fatalf("restored previous manifest = %q, want previous session id", string(previousManifestBytes))
}
mustReadEquals(t, filepath.Join(sessionRoot, "previous", "artifacts", "session_recap.md"), "# previous recap\n")
report := mustReadRestoreReport(t, filepath.Join(sessionRoot, "reports", "restore-latest.json"))
if report.Execution.Downloaded != 3 {
t.Fatalf("report execution.downloaded = %d, want 3", report.Execution.Downloaded)
}
}
func TestExecuteRestoreDryRunReportsPreviousCacheWithoutWriting(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
appendRestoreWorkflowPreviousInputConfig(t, pipelinePath, sessionPath)
fake := &storage.FakeBackend{}
cfg, _, _, _ := seedRestoreCommittedState(t, fake, pipelinePath, campaignPath, sessionPath)
seedRestorePreviousCurrent(t, fake, cfg, "# previous recap\n")
restoreWithStoreAndRealPhases(t, fake)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--dry-run"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "previous/artifacts/session_recap.md") {
t.Fatalf("stdout = %q, want planned previous-cache artifact", stdout.String())
}
sessionRoot := artifacts.SessionWorkDirForCampaign(workspaceRoot, cfg.Session.Campaign, cfg.Session.SessionID)
if _, err := os.Stat(filepath.Join(sessionRoot, "previous", "artifacts", "session_recap.md")); !os.IsNotExist(err) {
t.Fatalf("previous artifact should not be written during dry-run; stat err=%v", err)
}
}
func TestExecuteRestoreConflictWithoutForceDoesNotOverwrite(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
cfg, sessionPrefix, _, _ := seedRestoreCommittedState(t, fake, pipelinePath, campaignPath, sessionPath)
seedRestoreObject(fake, sessionPrefix+"transcripts/full.json", []byte("remote-transcript"))
sessionRoot := artifacts.SessionWorkDirForCampaign(workspaceRoot, cfg.Session.Campaign, cfg.Session.SessionID)
mustWriteTestFile(t, filepath.Join(sessionRoot, "transcripts", "full.json"), "local-transcript")
restoreWithStoreAndRealPhases(t, fake)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "conflicting path") {
t.Fatalf("stderr = %q, want conflict failure", stderr.String())
}
mustReadEquals(t, filepath.Join(sessionRoot, "transcripts", "full.json"), "local-transcript")
report := mustReadRestoreReport(t, filepath.Join(sessionRoot, "reports", "restore-latest.json"))
if report.Status != "failed" {
t.Fatalf("report status = %q, want failed", report.Status)
}
if report.Plan.Conflicts != 1 {
t.Fatalf("report plan.conflicts = %d, want 1", report.Plan.Conflicts)
}
}
func TestExecuteRestoreForceOverwritesDifferingFile(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
cfg, sessionPrefix, _, _ := seedRestoreCommittedState(t, fake, pipelinePath, campaignPath, sessionPath)
seedRestoreObject(fake, sessionPrefix+"transcripts/full.json", []byte("remote-transcript"))
sessionRoot := artifacts.SessionWorkDirForCampaign(workspaceRoot, cfg.Session.Campaign, cfg.Session.SessionID)
mustWriteTestFile(t, filepath.Join(sessionRoot, "transcripts", "full.json"), "local-transcript")
restoreWithStoreAndRealPhases(t, fake)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--force"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
mustReadEquals(t, filepath.Join(sessionRoot, "transcripts", "full.json"), "remote-transcript")
report := mustReadRestoreReport(t, filepath.Join(sessionRoot, "reports", "restore-latest.json"))
if !report.Force {
t.Fatalf("report force = %v, want true", report.Force)
}
}
func TestExecuteRestoreForceOverwritesDifferingPreviousCacheFile(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
appendRestoreWorkflowPreviousInputConfig(t, pipelinePath, sessionPath)
fake := &storage.FakeBackend{}
cfg, _, _, _ := seedRestoreCommittedState(t, fake, pipelinePath, campaignPath, sessionPath)
seedRestorePreviousCurrent(t, fake, cfg, "# remote previous recap\n")
sessionRoot := artifacts.SessionWorkDirForCampaign(workspaceRoot, cfg.Session.Campaign, cfg.Session.SessionID)
mustWriteTestFile(t, filepath.Join(sessionRoot, "previous", "artifacts", "session_recap.md"), "# local previous recap\n")
restoreWithStoreAndRealPhases(t, fake)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--force"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
mustReadEquals(t, filepath.Join(sessionRoot, "previous", "artifacts", "session_recap.md"), "# remote previous recap\n")
}
func TestExecuteRestoreLockConflictFailsAndWritesNothing(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
fake := &storage.FakeBackend{}
cfg, sessionPrefix, _, _ := seedRestoreCommittedState(t, fake, pipelinePath, campaignPath, sessionPath)
seedRestoreObject(fake, sessionPrefix+"transcripts/full.json", []byte("remote-transcript"))
store := artifacts.NewLocalStore(workspaceRoot)
lock, err := store.AcquireSessionLockFor(cfg.Session.Campaign, cfg.Session.SessionID)
if err != nil {
t.Fatalf("AcquireSessionLockFor() error = %v", err)
}
defer func() { _ = store.ReleaseSessionLock(lock) }()
restoreWithStoreAndRealPhases(t, fake)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "acquire session lock") {
t.Fatalf("stderr = %q, want lock failure", stderr.String())
}
sessionRoot := artifacts.SessionWorkDirForCampaign(workspaceRoot, cfg.Session.Campaign, cfg.Session.SessionID)
if _, err := os.Stat(filepath.Join(sessionRoot, "transcripts", "full.json")); !os.IsNotExist(err) {
t.Fatalf("transcript should not be restored when lock acquisition fails; stat err=%v", err)
}
}
func TestExecuteRestoreInvalidManifestDoesNotCorruptExistingManifest(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
base := &storage.FakeBackend{}
cfg, sessionPrefix, manifestKey, _ := seedRestoreCommittedState(t, base, pipelinePath, campaignPath, sessionPath)
seedRestoreObject(base, sessionPrefix+"transcripts/full.json", []byte("remote-transcript"))
toggled := &stagedManifestDownloadStore{
delegate: base,
manifestKey: manifestKey,
firstManifest: restoreManifestJSON(t, cfg.Session.SessionID, cfg.Session.Campaign),
secondManifest: []byte("{invalid json"),
manifestReads: 0,
}
sessionRoot := artifacts.SessionWorkDirForCampaign(workspaceRoot, cfg.Session.Campaign, cfg.Session.SessionID)
existing := manifest.New(cfg.Session.SessionID, nowUTC())
existing.Campaign = cfg.Session.Campaign
existingPath := filepath.Join(sessionRoot, "manifest.json")
manifestStore := &manifest.LocalStore{}
if err := manifestStore.Save(context.Background(), existingPath, existing); err != nil {
t.Fatalf("save existing local manifest: %v", err)
}
existingData, err := os.ReadFile(existingPath)
if err != nil {
t.Fatalf("read existing local manifest: %v", err)
}
restoreWithStoreAndRealPhases(t, toggled)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--force"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "validate manifest decode") {
t.Fatalf("stderr = %q, want manifest validation failure", stderr.String())
}
mustReadEquals(t, filepath.Join(sessionRoot, "transcripts", "full.json"), "remote-transcript")
report := mustReadRestoreReport(t, filepath.Join(sessionRoot, "reports", "restore-latest.json"))
if report.Status != "failed" {
t.Fatalf("report status = %q, want failed", report.Status)
}
if strings.TrimSpace(report.Error) == "" {
t.Fatal("report error is empty, want failure context")
}
afterData, err := os.ReadFile(existingPath)
if err != nil {
t.Fatalf("read local manifest after failure: %v", err)
}
if string(afterData) != string(existingData) {
t.Fatalf("local manifest changed after failed restore; before=%q after=%q", string(existingData), string(afterData))
}
}
func TestExecuteRestorePlanPathMismatchFails(t *testing.T) {
cfg := restorePlanConfig(t)
current := restorePlanCurrentState(t, cfg)
store := &storage.FakeBackend{}
seedRestoreObject(store, current.SessionPrefix+"transcripts/full.json", []byte("remote-transcript"))
plan := &RestorePlan{Actions: []RestoreAction{{
Kind: RestoreActionDownload,
RemoteKey: current.SessionPrefix + "transcripts/full.json",
LocalRelativePath: "transcripts/full.json",
LocalPath: "/tmp/escape.txt",
}}}
report, err := newRestoreReport(current, plan, RestorePlanOptions{})
if err != nil {
t.Fatalf("newRestoreReport() error = %v", err)
}
_, err = executeRestorePlan(context.Background(), cfg, current, plan, report, store)
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "local path mismatch") {
t.Fatalf("error = %v, want local path mismatch", err)
}
}
func mustReadRestoreReport(t *testing.T, path string) *RestoreReport {
t.Helper()
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("ReadFile(%q): %v", path, err)
}
var report RestoreReport
if err := json.Unmarshal(data, &report); err != nil {
t.Fatalf("Unmarshal restore report %q: %v", path, err)
}
return &report
}
func restoreWithStoreAndRealPhases(t *testing.T, objectStore storage.ObjectStore) {
t.Helper()
origStoreFn := newObjectStoreFromConfigFn
origDiscoverFn := discoverRemoteCurrentStateFn
origPlanFn := buildRestorePlanFn
origExecuteFn := executeRestorePlanFn
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
discoverRemoteCurrentStateFn = origDiscoverFn
buildRestorePlanFn = origPlanFn
executeRestorePlanFn = origExecuteFn
})
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
return objectStore, nil
}
discoverRemoteCurrentStateFn = discoverRemoteCurrentState
buildRestorePlanFn = buildRestorePlan
executeRestorePlanFn = executeRestorePlan
}
func seedRestoreCommittedState(t *testing.T, fake *storage.FakeBackend, pipelinePath, campaignPath, sessionPath string) (*config.Config, string, string, string) {
t.Helper()
cfg, err := config.LoadWithSessionOptions(pipelinePath, campaignPath, sessionPath, config.SessionLoadOptions{})
if err != nil {
t.Fatalf("LoadWithSessionOptions() error = %v", err)
}
if err := config.Validate(cfg); err != nil {
t.Fatalf("Validate() error = %v", err)
}
sessionPrefix := artifacts.S3SessionPrefix(cfg.Pipeline.Storage.S3.RootPrefix, cfg.Session.Campaign, cfg.Session.SessionID)
manifestKey, runIDKey := artifacts.ResolveArchiveCurrentStateKeys(sessionPrefix)
seedRestoreObject(fake, runIDKey, []byte("20260519T010203Z-a1b2c3d4\n"))
seedRestoreObject(fake, manifestKey, restoreManifestJSON(t, cfg.Session.SessionID, cfg.Session.Campaign))
return cfg, sessionPrefix, manifestKey, runIDKey
}
func appendRestoreWorkflowPreviousInputConfig(t *testing.T, pipelinePath, sessionPath string) {
t.Helper()
appendRestoreWorkflowScriptoriumConfig(t, pipelinePath, `
scriptorium:
binary: scriptorium
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
output_path: artifacts/session_recap.md
inputs:
previous_recap:
source: narratio.previous_session.artifact.session_recap
required: true
`)
appendRestoreWorkflowScriptoriumConfig(t, sessionPath, `
previous_session_id: 2026-04-26
`)
}
func seedRestorePreviousCurrent(t *testing.T, fake *storage.FakeBackend, cfg *config.Config, artifactBody string) {
t.Helper()
seedRestorePreviousCurrentManifestOnly(t, fake, cfg)
previousPrefix := artifacts.S3SessionPrefix(cfg.Pipeline.Storage.S3.RootPrefix, cfg.Session.Campaign, cfg.Session.PreviousSessionID)
seedRestoreObject(fake, previousPrefix+"artifacts/session_recap.md", []byte(artifactBody))
}
func seedRestorePreviousCurrentManifestOnly(t *testing.T, fake *storage.FakeBackend, cfg *config.Config) {
t.Helper()
previousPrefix := artifacts.S3SessionPrefix(cfg.Pipeline.Storage.S3.RootPrefix, cfg.Session.Campaign, cfg.Session.PreviousSessionID)
manifestKey, runIDKey := artifacts.ResolveArchiveCurrentStateKeys(previousPrefix)
previousRunID := "20260426T010203Z-a1b2c3d4"
seedRestoreObject(fake, runIDKey, []byte(previousRunID+"\n"))
m := manifest.New(cfg.Session.PreviousSessionID, nowUTC())
m.Campaign = cfg.Session.Campaign
m.RunID = previousRunID
data, err := json.Marshal(m)
if err != nil {
t.Fatalf("marshal previous restore manifest: %v", err)
}
seedRestoreObject(fake, manifestKey, append(data, '\n'))
}
func mustReadEquals(t *testing.T, path, want string) {
t.Helper()
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("ReadFile(%q): %v", path, err)
}
if string(data) != want {
t.Fatalf("file %q = %q, want %q", path, string(data), want)
}
}
func fakeDownloadCount(fake *storage.FakeBackend, key string) int {
count := 0
for _, call := range fake.Downloads {
if call.Key == key {
count++
}
}
return count
}
type stagedManifestDownloadStore struct {
delegate *storage.FakeBackend
manifestKey string
firstManifest []byte
secondManifest []byte
manifestReads int
}
func (s *stagedManifestDownloadStore) List(ctx context.Context, prefix string) ([]storage.ObjectInfo, error) {
return s.delegate.List(ctx, prefix)
}
func (s *stagedManifestDownloadStore) Download(ctx context.Context, key, localPath string) error {
if strings.TrimSpace(key) == strings.TrimSpace(s.manifestKey) {
s.manifestReads++
payload := s.secondManifest
if s.manifestReads <= 1 {
payload = s.firstManifest
}
if err := os.MkdirAll(filepath.Dir(localPath), 0o755); err != nil {
return fmt.Errorf("download staged manifest: create parent: %w", err)
}
if err := os.WriteFile(localPath, payload, 0o644); err != nil {
return fmt.Errorf("download staged manifest: write local file: %w", err)
}
return nil
}
return s.delegate.Download(ctx, key, localPath)
}
func (s *stagedManifestDownloadStore) Upload(ctx context.Context, localPath, key string, opts storage.UploadOptions) (storage.ObjectInfo, error) {
return s.delegate.Upload(ctx, localPath, key, opts)
}
func (s *stagedManifestDownloadStore) Exists(ctx context.Context, key string) (bool, error) {
return s.delegate.Exists(ctx, key)
}

View File

@@ -0,0 +1,421 @@
package app
import (
"context"
"fmt"
"io"
"os"
"path"
"path/filepath"
"sort"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/previouscache"
)
// RestoreActionKind identifies one restore planner action.
type RestoreActionKind string
const (
RestoreActionDownload RestoreActionKind = "download"
RestoreActionSkipSame RestoreActionKind = "skip_same"
RestoreActionConflict RestoreActionKind = "conflict"
)
// RestoreAction is one deterministic planner action.
type RestoreAction struct {
Kind RestoreActionKind
RemoteKey string
LocalRelativePath string
LocalPath string
Size int64
ETag string
ExistsLocal bool
SameLocal bool
Conflict bool
Reason string
}
// RestorePlan is the deterministic output of restore planning.
type RestorePlan struct {
Actions []RestoreAction
DownloadCount int
SkipSameCount int
ConflictCount int
}
// RestorePlanOptions control restore planning scope and classification.
type RestorePlanOptions struct {
IncludeAudio bool
Force bool
DryRun bool
}
func buildRestorePlan(ctx context.Context, cfg *config.Config, current *RemoteCurrentState, store storage.ObjectStore, opts RestorePlanOptions) (*RestorePlan, error) {
if cfg == nil || cfg.Pipeline == nil || cfg.Session == nil {
return nil, fmt.Errorf("resolved config with pipeline/session is required")
}
if current == nil {
return nil, fmt.Errorf("remote current state is required")
}
if store == nil {
return nil, fmt.Errorf("remote object store is required")
}
prefix := normalizeRemoteKey(current.SessionPrefix)
if strings.TrimSpace(prefix) == "" {
return nil, fmt.Errorf("remote session prefix is required")
}
if !strings.HasSuffix(prefix, "/") {
prefix += "/"
}
sessionPaths := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root).SessionPathsFor(cfg.Session.Campaign, cfg.Session.SessionID)
objects, err := store.List(ctx, prefix)
if err != nil {
return nil, fmt.Errorf("list remote session objects under %q: %w", prefix, err)
}
candidates := make(map[string]storage.ObjectInfo, len(objects)+1)
for _, obj := range objects {
key := normalizeRemoteKey(obj.Key)
if key == "" {
continue
}
obj.Key = key
candidates[key] = obj
}
if strings.TrimSpace(current.CurrentManifestKey) != "" {
key := normalizeRemoteKey(current.CurrentManifestKey)
if _, ok := candidates[key]; !ok {
candidates[key] = storage.ObjectInfo{Key: key}
}
}
actions := make([]RestoreAction, 0, len(candidates))
for key, obj := range candidates {
rel, include, err := restoreLocalRelativePathForKey(prefix, normalizeRemoteKey(current.CurrentManifestKey), key, opts.IncludeAudio)
if err != nil {
return nil, fmt.Errorf("map remote key %q: %w", key, err)
}
if !include {
continue
}
localPath, err := joinWithinSessionRoot(sessionPaths.Root, rel)
if err != nil {
return nil, fmt.Errorf("map remote key %q: %w", key, err)
}
action, err := classifyRestoreAction(ctx, store, obj, rel, localPath, opts.Force)
if err != nil {
return nil, fmt.Errorf("classify remote key %q: %w", key, err)
}
actions = append(actions, action)
}
previousActions, err := buildPreviousCacheRestoreActions(ctx, cfg, sessionPaths, store, opts.Force)
if err != nil {
return nil, err
}
actions = append(actions, previousActions...)
sort.Slice(actions, func(i, j int) bool {
if actions[i].LocalRelativePath == actions[j].LocalRelativePath {
return actions[i].RemoteKey < actions[j].RemoteKey
}
return actions[i].LocalRelativePath < actions[j].LocalRelativePath
})
plan := &RestorePlan{Actions: actions}
for _, action := range actions {
switch action.Kind {
case RestoreActionDownload:
plan.DownloadCount++
case RestoreActionSkipSame:
plan.SkipSameCount++
case RestoreActionConflict:
plan.ConflictCount++
}
}
_ = opts.DryRun
return plan, nil
}
func normalizeRemoteKey(v string) string {
return strings.Trim(strings.ReplaceAll(strings.TrimSpace(v), "\\", "/"), "/")
}
func restoreLocalRelativePathForKey(sessionPrefix, currentManifestKey, key string, includeAudio bool) (string, bool, error) {
if key == "" {
return "", false, nil
}
if key == currentManifestKey {
return config.PathManifestFile, true, nil
}
if !strings.HasPrefix(key, sessionPrefix) {
return "", false, fmt.Errorf("key is outside resolved session prefix %q", sessionPrefix)
}
rel := strings.TrimPrefix(key, sessionPrefix)
rel = strings.TrimSpace(rel)
if rel == "" {
return "", false, nil
}
cleanRel := path.Clean(rel)
if cleanRel == "." || cleanRel == "" {
return "", false, nil
}
if cleanRel == ".." || strings.HasPrefix(cleanRel, "../") || strings.HasPrefix(cleanRel, "/") {
return "", false, fmt.Errorf("key relative path %q escapes session scope", rel)
}
if cleanRel == config.PathManifestFile {
return config.PathManifestFile, true, nil
}
if strings.HasPrefix(cleanRel, config.S3CurrentSegment+"/") {
return "", false, nil
}
if strings.HasPrefix(cleanRel, config.S3RunsSegment+"/") {
return "", false, nil
}
excludedRoots := []string{
config.PathLogsDirSegment,
config.PathReportsDirSegment,
config.PathConfigDirSegment,
config.PathInputsDirSegment,
}
for _, root := range excludedRoots {
if cleanRel == root || strings.HasPrefix(cleanRel, root+"/") {
return "", false, nil
}
}
if cleanRel == config.PathTranscriptsSegment || strings.HasPrefix(cleanRel, config.PathTranscriptsSegment+"/") {
return cleanRel, true, nil
}
if cleanRel == config.PathArtifactsDirSegment || strings.HasPrefix(cleanRel, config.PathArtifactsDirSegment+"/") {
return cleanRel, true, nil
}
if cleanRel == config.PathPreviousDirSegment || strings.HasPrefix(cleanRel, config.PathPreviousDirSegment+"/") {
return "", false, nil
}
if includeAudio && (cleanRel == config.PathAudioDirSegment || strings.HasPrefix(cleanRel, config.PathAudioDirSegment+"/")) {
return cleanRel, true, nil
}
return "", false, nil
}
func joinWithinSessionRoot(sessionRoot, relative string) (string, error) {
if strings.TrimSpace(sessionRoot) == "" {
return "", fmt.Errorf("session root is required")
}
cleanRel := path.Clean(strings.TrimSpace(relative))
if cleanRel == "." || cleanRel == "" {
return "", fmt.Errorf("relative path is required")
}
if cleanRel == ".." || strings.HasPrefix(cleanRel, "../") || strings.HasPrefix(cleanRel, "/") {
return "", fmt.Errorf("relative path escapes session root")
}
abs := filepath.Clean(filepath.Join(sessionRoot, filepath.FromSlash(cleanRel)))
root := filepath.Clean(sessionRoot)
if abs != root && !strings.HasPrefix(abs, root+string(filepath.Separator)) {
return "", fmt.Errorf("resolved local path escapes session root")
}
return abs, nil
}
func buildPreviousCacheRestoreActions(
ctx context.Context,
cfg *config.Config,
sessionPaths artifacts.SessionPaths,
store storage.ObjectStore,
force bool,
) ([]RestoreAction, error) {
if cfg == nil || cfg.Pipeline == nil || cfg.Pipeline.Scriptorium == nil {
return nil, nil
}
requirements := artifacts.CollectPreviousArtifactRequirements(cfg.Pipeline.Scriptorium.Artifacts)
if len(requirements) == 0 {
return nil, nil
}
plan, err := previouscache.BuildPlan(ctx, cfg, sessionPaths, requirements, store)
if err != nil {
return nil, fmt.Errorf("plan previous-session cache restore: %w", err)
}
actions := make([]RestoreAction, 0, len(plan.Records))
for _, record := range plan.Records {
action, err := classifyRestoreAction(ctx, store, storage.ObjectInfo{Key: record.RemoteKey}, record.LocalRelativePath, record.LocalPath, force)
if err != nil {
return nil, fmt.Errorf("classify previous-session cache object %q: %w", record.RemoteKey, err)
}
actions = append(actions, action)
}
return actions, nil
}
func classifyRestoreAction(
ctx context.Context,
store storage.ObjectStore,
object storage.ObjectInfo,
localRelPath string,
localPath string,
force bool,
) (RestoreAction, error) {
action := RestoreAction{
RemoteKey: normalizeRemoteKey(object.Key),
LocalRelativePath: localRelPath,
LocalPath: localPath,
Size: object.Size,
ETag: object.ETag,
}
info, err := os.Stat(localPath)
if err != nil {
if os.IsNotExist(err) {
action.Kind = RestoreActionDownload
action.Reason = "local file missing"
return action, nil
}
return RestoreAction{}, fmt.Errorf("stat local file: %w", err)
}
action.ExistsLocal = true
if info.IsDir() {
action.Kind = RestoreActionConflict
action.Conflict = true
action.Reason = "local path is a directory"
return action, nil
}
if restoreRelativePathIsAudio(localRelPath) {
if object.Size > 0 {
if info.Size() == object.Size {
action.Kind = RestoreActionSkipSame
action.SameLocal = true
action.Reason = "local audio size matches remote content"
return action, nil
}
if force {
action.Kind = RestoreActionDownload
action.Reason = "local audio differs (size mismatch); overwrite with --force"
return action, nil
}
action.Kind = RestoreActionConflict
action.Conflict = true
action.Reason = "local audio differs (size mismatch)"
return action, nil
}
if force {
action.Kind = RestoreActionDownload
action.Reason = "local audio exists; remote size unavailable; overwrite with --force"
return action, nil
}
action.Kind = RestoreActionConflict
action.Conflict = true
action.Reason = "local audio exists; remote size unavailable"
return action, nil
}
if object.Size > 0 && info.Size() != object.Size {
if force {
action.Kind = RestoreActionDownload
action.Reason = "local file differs (size mismatch); overwrite with --force"
return action, nil
}
action.Kind = RestoreActionConflict
action.Conflict = true
action.Reason = "local file differs (size mismatch)"
return action, nil
}
localDigest, err := artifacts.SHA256File(localPath)
if err != nil {
return RestoreAction{}, fmt.Errorf("checksum local file: %w", err)
}
remotePath, err := downloadObjectToTemp(ctx, store, action.RemoteKey, "narratio-restore-plan-remote-*.tmp")
if err != nil {
return RestoreAction{}, fmt.Errorf("download remote object: %w", err)
}
defer func() { _ = os.Remove(remotePath) }()
remoteDigest, err := artifacts.SHA256File(remotePath)
if err != nil {
return RestoreAction{}, fmt.Errorf("checksum remote object: %w", err)
}
if remoteDigest == localDigest {
action.Kind = RestoreActionSkipSame
action.SameLocal = true
action.Reason = "local file matches remote content"
return action, nil
}
if force {
action.Kind = RestoreActionDownload
action.Reason = "local file differs; overwrite with --force"
return action, nil
}
action.Kind = RestoreActionConflict
action.Conflict = true
action.Reason = "local file differs"
return action, nil
}
func restoreActionIsAudio(action RestoreAction) bool {
return restoreRelativePathIsAudio(action.LocalRelativePath)
}
func restoreRelativePathIsAudio(rel string) bool {
cleanRel := path.Clean(strings.TrimSpace(rel))
return cleanRel == config.PathAudioDirSegment || strings.HasPrefix(cleanRel, config.PathAudioDirSegment+"/")
}
func writeRestorePlan(out io.Writer, current *RemoteCurrentState, plan *RestorePlan, opts RestorePlanOptions) error {
if out == nil {
return fmt.Errorf("output writer is required")
}
if current == nil {
return fmt.Errorf("remote current state is required")
}
if plan == nil {
return fmt.Errorf("restore plan is required")
}
if _, err := fmt.Fprintf(
out,
"restore plan: session %s/%s run=%s actions=%d download=%d skip_same=%d conflict=%d dry_run=%t force=%t include_audio=%t\n",
current.Campaign,
current.SessionID,
current.RunID,
len(plan.Actions),
plan.DownloadCount,
plan.SkipSameCount,
plan.ConflictCount,
opts.DryRun,
opts.Force,
opts.IncludeAudio,
); err != nil {
return err
}
for _, action := range plan.Actions {
if _, err := fmt.Fprintf(out, "%s %s <- %s", action.Kind, action.LocalRelativePath, action.RemoteKey); err != nil {
return err
}
if strings.TrimSpace(action.Reason) != "" {
if _, err := fmt.Fprintf(out, " (%s)", action.Reason); err != nil {
return err
}
}
if _, err := fmt.Fprintln(out); err != nil {
return err
}
}
return nil
}

View File

@@ -0,0 +1,354 @@
package app
import (
"context"
"path/filepath"
"reflect"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestRestorePlanDefaultScope(t *testing.T) {
cfg := restorePlanConfig(t)
current := restorePlanCurrentState(t, cfg)
store := &storage.FakeBackend{}
seedRestoreObject(store, current.CurrentManifestKey, []byte(`{"session_id":"2026-05-03"}`))
seedRestoreObject(store, current.SessionPrefix+"transcripts/full.json", []byte(`{"segments":[1]}`))
seedRestoreObject(store, current.SessionPrefix+"artifacts/session_recap.md", []byte("# recap\n"))
seedRestoreObject(store, current.SessionPrefix+"audio/alice.flac", []byte("audio"))
seedRestoreObject(store, current.SessionPrefix+"runs/20260519T010203Z-a1b2/manifest.json", []byte("{}"))
seedRestoreObject(store, current.SessionPrefix+"logs/archive.log", []byte("log"))
plan, err := buildRestorePlan(context.Background(), cfg, current, store, RestorePlanOptions{})
if err != nil {
t.Fatalf("buildRestorePlan() error = %v", err)
}
got := actionRelPaths(plan.Actions)
want := []string{"artifacts/session_recap.md", "manifest.json", "transcripts/full.json"}
if !reflect.DeepEqual(got, want) {
t.Fatalf("action local paths = %#v, want %#v", got, want)
}
if plan.DownloadCount != 3 || plan.SkipSameCount != 0 || plan.ConflictCount != 0 {
t.Fatalf("counts = download=%d skip_same=%d conflict=%d, want 3/0/0", plan.DownloadCount, plan.SkipSameCount, plan.ConflictCount)
}
}
func TestRestorePlanIncludeAudio(t *testing.T) {
cfg := restorePlanConfig(t)
current := restorePlanCurrentState(t, cfg)
store := &storage.FakeBackend{}
seedRestoreObject(store, current.CurrentManifestKey, []byte(`{"session_id":"2026-05-03"}`))
seedRestoreObject(store, current.SessionPrefix+"audio/alice.flac", []byte("audio"))
plan, err := buildRestorePlan(context.Background(), cfg, current, store, RestorePlanOptions{IncludeAudio: true})
if err != nil {
t.Fatalf("buildRestorePlan() error = %v", err)
}
got := actionRelPaths(plan.Actions)
want := []string{"audio/alice.flac", "manifest.json"}
if !reflect.DeepEqual(got, want) {
t.Fatalf("action local paths = %#v, want %#v", got, want)
}
}
func TestRestorePlanExistingAudioUsesSizeWithoutRemoteChecksumDownload(t *testing.T) {
cfg := restorePlanConfig(t)
current := restorePlanCurrentState(t, cfg)
store := &storage.FakeBackend{}
seedRestoreObject(store, current.CurrentManifestKey, []byte(`{"session_id":"2026-05-03"}`))
seedRestoreObject(store, current.SessionPrefix+"audio/alice.flac", []byte("audio"))
sessionRoot := artifacts.SessionWorkDirForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID)
mustWriteTestFile(t, filepath.Join(sessionRoot, "audio", "alice.flac"), "local")
plan, err := buildRestorePlan(context.Background(), cfg, current, store, RestorePlanOptions{IncludeAudio: true})
if err != nil {
t.Fatalf("buildRestorePlan() error = %v", err)
}
if len(store.Downloads) != 0 {
t.Fatalf("downloads = %d, want no remote checksum download for audio", len(store.Downloads))
}
actionByRel := map[string]RestoreAction{}
for _, action := range plan.Actions {
actionByRel[action.LocalRelativePath] = action
}
audioAction := actionByRel["audio/alice.flac"]
if audioAction.Kind != RestoreActionSkipSame {
t.Fatalf("audio action kind = %q, want %q", audioAction.Kind, RestoreActionSkipSame)
}
}
func TestRestorePlanIncludesPreviousCacheByDefault(t *testing.T) {
cfg := restorePlanConfig(t)
configureRestorePlanPreviousRequirement(cfg, true)
current := restorePlanCurrentState(t, cfg)
store := &storage.FakeBackend{}
seedRestoreObject(store, current.CurrentManifestKey, []byte(`{"session_id":"2026-05-03"}`))
seedRestorePreviousCurrent(t, store, cfg, "# previous recap\n")
plan, err := buildRestorePlan(context.Background(), cfg, current, store, RestorePlanOptions{})
if err != nil {
t.Fatalf("buildRestorePlan() error = %v", err)
}
got := actionRelPaths(plan.Actions)
want := []string{"manifest.json", "previous/artifacts/session_recap.md", "previous/manifest.json"}
if !reflect.DeepEqual(got, want) {
t.Fatalf("action local paths = %#v, want %#v", got, want)
}
}
func TestRestorePlanIgnoresCurrentSessionArchivedPreviousCache(t *testing.T) {
cfg := restorePlanConfig(t)
current := restorePlanCurrentState(t, cfg)
store := &storage.FakeBackend{}
seedRestoreObject(store, current.CurrentManifestKey, []byte(`{"session_id":"2026-05-03"}`))
seedRestoreObject(store, current.SessionPrefix+"previous/manifest.json", []byte(`{"session_id":"2026-04-26"}`))
seedRestoreObject(store, current.SessionPrefix+"previous/artifacts/session_recap.md", []byte("# previous recap\n"))
plan, err := buildRestorePlan(context.Background(), cfg, current, store, RestorePlanOptions{})
if err != nil {
t.Fatalf("buildRestorePlan() error = %v", err)
}
got := actionRelPaths(plan.Actions)
want := []string{"manifest.json"}
if !reflect.DeepEqual(got, want) {
t.Fatalf("action local paths = %#v, want %#v", got, want)
}
}
func TestRestorePlanMissingOptionalPreviousCacheSkipsArtifact(t *testing.T) {
cfg := restorePlanConfig(t)
configureRestorePlanPreviousRequirement(cfg, false)
current := restorePlanCurrentState(t, cfg)
store := &storage.FakeBackend{}
seedRestoreObject(store, current.CurrentManifestKey, []byte(`{"session_id":"2026-05-03"}`))
seedRestorePreviousCurrentManifestOnly(t, store, cfg)
plan, err := buildRestorePlan(context.Background(), cfg, current, store, RestorePlanOptions{})
if err != nil {
t.Fatalf("buildRestorePlan() error = %v", err)
}
got := actionRelPaths(plan.Actions)
want := []string{"manifest.json", "previous/manifest.json"}
if !reflect.DeepEqual(got, want) {
t.Fatalf("action local paths = %#v, want %#v", got, want)
}
}
func TestRestorePlanMissingRequiredPreviousCacheFails(t *testing.T) {
cfg := restorePlanConfig(t)
configureRestorePlanPreviousRequirement(cfg, true)
current := restorePlanCurrentState(t, cfg)
store := &storage.FakeBackend{}
seedRestoreObject(store, current.CurrentManifestKey, []byte(`{"session_id":"2026-05-03"}`))
seedRestorePreviousCurrentManifestOnly(t, store, cfg)
_, err := buildRestorePlan(context.Background(), cfg, current, store, RestorePlanOptions{})
if err == nil || !strings.Contains(err.Error(), "required previous-session artifact") {
t.Fatalf("buildRestorePlan() error = %v, want required previous artifact failure", err)
}
}
func TestRestorePlanPreviousCacheConflictRequiresForce(t *testing.T) {
cfg := restorePlanConfig(t)
configureRestorePlanPreviousRequirement(cfg, true)
current := restorePlanCurrentState(t, cfg)
store := &storage.FakeBackend{}
seedRestoreObject(store, current.CurrentManifestKey, []byte(`{"session_id":"2026-05-03"}`))
seedRestorePreviousCurrent(t, store, cfg, "# remote previous recap\n")
sessionRoot := artifacts.SessionWorkDirForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID)
mustWriteTestFile(t, filepath.Join(sessionRoot, "previous", "artifacts", "session_recap.md"), "# local previous recap\n")
plan, err := buildRestorePlan(context.Background(), cfg, current, store, RestorePlanOptions{})
if err != nil {
t.Fatalf("buildRestorePlan() error = %v", err)
}
if plan.ConflictCount != 1 {
t.Fatalf("ConflictCount = %d, want 1", plan.ConflictCount)
}
plan, err = buildRestorePlan(context.Background(), cfg, current, store, RestorePlanOptions{Force: true})
if err != nil {
t.Fatalf("buildRestorePlan(force) error = %v", err)
}
if plan.ConflictCount != 0 {
t.Fatalf("force ConflictCount = %d, want 0", plan.ConflictCount)
}
}
func TestRestorePlanClassifiesSameAndConflict(t *testing.T) {
cfg := restorePlanConfig(t)
current := restorePlanCurrentState(t, cfg)
store := &storage.FakeBackend{}
seedRestoreObject(store, current.CurrentManifestKey, []byte(`{"session_id":"2026-05-03"}`))
seedRestoreObject(store, current.SessionPrefix+"transcripts/full.json", []byte(`{"segments":[1]}`))
seedRestoreObject(store, current.SessionPrefix+"artifacts/session_recap.md", []byte("remote-content\n"))
sessionRoot := artifacts.SessionWorkDirForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID)
mustWriteTestFile(t, filepath.Join(sessionRoot, "transcripts", "full.json"), `{"segments":[1]}`)
mustWriteTestFile(t, filepath.Join(sessionRoot, "artifacts", "session_recap.md"), "different\n")
plan, err := buildRestorePlan(context.Background(), cfg, current, store, RestorePlanOptions{})
if err != nil {
t.Fatalf("buildRestorePlan() error = %v", err)
}
if plan.SkipSameCount != 1 {
t.Fatalf("SkipSameCount = %d, want 1", plan.SkipSameCount)
}
if plan.ConflictCount != 1 {
t.Fatalf("ConflictCount = %d, want 1", plan.ConflictCount)
}
actionByRel := map[string]RestoreAction{}
for _, action := range plan.Actions {
actionByRel[action.LocalRelativePath] = action
}
if actionByRel["transcripts/full.json"].Kind != RestoreActionSkipSame {
t.Fatalf("transcripts/full.json kind = %q, want %q", actionByRel["transcripts/full.json"].Kind, RestoreActionSkipSame)
}
if actionByRel["artifacts/session_recap.md"].Kind != RestoreActionConflict {
t.Fatalf("artifacts/session_recap.md kind = %q, want %q", actionByRel["artifacts/session_recap.md"].Kind, RestoreActionConflict)
}
}
func TestRestorePlanForceTurnsConflictsIntoDownloads(t *testing.T) {
cfg := restorePlanConfig(t)
current := restorePlanCurrentState(t, cfg)
store := &storage.FakeBackend{}
seedRestoreObject(store, current.CurrentManifestKey, []byte(`{"session_id":"2026-05-03"}`))
seedRestoreObject(store, current.SessionPrefix+"artifacts/session_recap.md", []byte("remote-content\n"))
sessionRoot := artifacts.SessionWorkDirForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID)
mustWriteTestFile(t, filepath.Join(sessionRoot, "artifacts", "session_recap.md"), "different\n")
plan, err := buildRestorePlan(context.Background(), cfg, current, store, RestorePlanOptions{Force: true})
if err != nil {
t.Fatalf("buildRestorePlan() error = %v", err)
}
actionByRel := map[string]RestoreAction{}
for _, action := range plan.Actions {
actionByRel[action.LocalRelativePath] = action
}
recap := actionByRel["artifacts/session_recap.md"]
if recap.Kind != RestoreActionDownload {
t.Fatalf("artifacts/session_recap.md kind = %q, want %q", recap.Kind, RestoreActionDownload)
}
if plan.ConflictCount != 0 {
t.Fatalf("ConflictCount = %d, want 0", plan.ConflictCount)
}
}
func TestRestorePlanTraversalUnsafeKeyFails(t *testing.T) {
cfg := restorePlanConfig(t)
current := restorePlanCurrentState(t, cfg)
store := &storage.FakeBackend{}
seedRestoreObject(store, current.CurrentManifestKey, []byte(`{"session_id":"2026-05-03"}`))
seedRestoreObject(store, current.SessionPrefix+"artifacts/../../escape.txt", []byte("bad"))
_, err := buildRestorePlan(context.Background(), cfg, current, store, RestorePlanOptions{})
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "escapes session scope") {
t.Fatalf("error = %v, want traversal safety failure", err)
}
}
func seedRestoreObject(store *storage.FakeBackend, key string, data []byte) {
store.SeedObject(storage.FakeObject{Key: key, Data: data})
}
func actionRelPaths(actions []RestoreAction) []string {
out := make([]string, 0, len(actions))
for _, action := range actions {
out = append(out, action.LocalRelativePath)
}
return out
}
func restorePlanConfig(t *testing.T) *config.Config {
t.Helper()
workspaceRoot := t.TempDir()
return &config.Config{
Pipeline: &config.PipelineConfig{
Workspace: config.WorkspaceConfig{Root: workspaceRoot},
Storage: config.StorageConfig{S3: &config.StorageS3Config{
Bucket: "test-bucket",
RootPrefix: "dnd",
}},
},
Session: &config.SessionConfig{
SessionID: "2026-05-03",
Campaign: "sample-campaign",
},
}
}
func configureRestorePlanPreviousRequirement(cfg *config.Config, required bool) {
cfg.Session.PreviousSessionID = "2026-04-26"
cfg.Pipeline.Scriptorium = &config.ScriptoriumConfig{
Artifacts: map[string]config.ScriptoriumArtifactConfig{
"session_recap": {
Enabled: true,
OutputPath: "artifacts/session_recap.md",
Inputs: map[string]config.ScriptoriumInputConfig{
"previous_recap": {
Source: "narratio.previous_session.artifact.session_recap",
Required: required,
},
},
},
},
}
}
func restorePlanCurrentState(t *testing.T, cfg *config.Config) *RemoteCurrentState {
t.Helper()
sessionPrefix := artifacts.S3SessionPrefix("dnd", cfg.Session.Campaign, cfg.Session.SessionID)
manifestKey, runIDKey := artifacts.ResolveArchiveCurrentStateKeys(sessionPrefix)
return &RemoteCurrentState{
Bucket: "test-bucket",
SessionPrefix: sessionPrefix,
CurrentManifestKey: manifestKey,
CurrentRunIDKey: runIDKey,
RunID: "20260519T010203Z-a1b2c3d4",
SessionID: cfg.Session.SessionID,
Campaign: cfg.Session.Campaign,
}
}
func TestWriteRestorePlan(t *testing.T) {
current := &RemoteCurrentState{Campaign: "sample-campaign", SessionID: "2026-05-03", RunID: "r-1"}
plan := &RestorePlan{Actions: []RestoreAction{{Kind: RestoreActionDownload, LocalRelativePath: "manifest.json", RemoteKey: "k", Reason: "local file missing"}}, DownloadCount: 1}
var out strings.Builder
if err := writeRestorePlan(&out, current, plan, RestorePlanOptions{DryRun: true}); err != nil {
t.Fatalf("writeRestorePlan() error = %v", err)
}
text := out.String()
if !strings.Contains(text, "restore plan: session sample-campaign/2026-05-03 run=r-1") {
t.Fatalf("output = %q, want plan summary", text)
}
if !strings.Contains(text, "download manifest.json <- k") {
t.Fatalf("output = %q, want action line", text)
}
}

View File

@@ -0,0 +1,243 @@
package app
import (
"encoding/json"
"fmt"
"io"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
// RestoreReport is the durable restore diagnostic model.
type RestoreReport struct {
GeneratedAt string `json:"generated_at"`
SessionID string `json:"session_id"`
Campaign string `json:"campaign"`
RunID string `json:"run_id"`
DryRun bool `json:"dry_run"`
Force bool `json:"force"`
IncludeAudio bool `json:"include_audio"`
Status string `json:"status"`
Error string `json:"error,omitempty"`
Plan RestorePlanSummary `json:"plan"`
Execution RestoreExecutionStats `json:"execution"`
Actions []RestoreReportAction `json:"actions"`
reportPathRel string
}
type RestorePlanSummary struct {
Actions int `json:"actions"`
Download int `json:"download"`
SkipSame int `json:"skip_same"`
Conflicts int `json:"conflicts"`
}
type RestoreExecutionStats struct {
Downloaded int `json:"downloaded"`
Failed int `json:"failed"`
}
type RestoreReportAction struct {
Kind string `json:"kind"`
LocalRelativePath string `json:"local_relative_path"`
RemoteKey string `json:"remote_key"`
Reason string `json:"reason,omitempty"`
Status string `json:"status"`
Error string `json:"error,omitempty"`
}
func newRestoreReport(current *RemoteCurrentState, plan *RestorePlan, opts RestorePlanOptions) (*RestoreReport, error) {
if current == nil {
return nil, fmt.Errorf("remote current state is required")
}
if plan == nil {
return nil, fmt.Errorf("restore plan is required")
}
r := &RestoreReport{
GeneratedAt: nowUTC().Format("2006-01-02T15:04:05.999999999Z07:00"),
SessionID: current.SessionID,
Campaign: current.Campaign,
RunID: current.RunID,
DryRun: opts.DryRun,
Force: opts.Force,
IncludeAudio: opts.IncludeAudio,
Status: "planned",
Plan: RestorePlanSummary{
Actions: len(plan.Actions),
Download: plan.DownloadCount,
SkipSame: plan.SkipSameCount,
Conflicts: plan.ConflictCount,
},
Actions: make([]RestoreReportAction, 0, len(plan.Actions)),
reportPathRel: filepath.ToSlash(filepath.Join(config.PathReportsDirSegment, "restore-latest.json")),
}
for _, action := range plan.Actions {
r.Actions = append(r.Actions, RestoreReportAction{
Kind: string(action.Kind),
LocalRelativePath: action.LocalRelativePath,
RemoteKey: action.RemoteKey,
Reason: action.Reason,
Status: initialRestoreActionStatus(action.Kind),
})
}
return r, nil
}
func initialRestoreActionStatus(kind RestoreActionKind) string {
switch kind {
case RestoreActionDownload:
return "planned_download"
case RestoreActionSkipSame:
return "skipped_same"
case RestoreActionConflict:
return "conflict"
default:
return "planned"
}
}
func (r *RestoreReport) markDownloaded(action RestoreAction) {
if r == nil {
return
}
if idx := r.findAction(action); idx >= 0 {
r.Actions[idx].Status = "downloaded"
r.Actions[idx].Error = ""
}
r.Execution.Downloaded++
}
func (r *RestoreReport) markFailed(action RestoreAction, err error) {
if r == nil {
return
}
if idx := r.findAction(action); idx >= 0 {
r.Actions[idx].Status = "failed"
if err != nil {
r.Actions[idx].Error = err.Error()
}
}
r.Execution.Failed++
}
func (r *RestoreReport) setFailed(err error) {
if r == nil {
return
}
r.Status = "failed"
if err != nil {
r.Error = err.Error()
}
}
func (r *RestoreReport) setSucceeded() {
if r == nil {
return
}
r.Status = "succeeded"
r.Error = ""
}
func (r *RestoreReport) findAction(action RestoreAction) int {
if r == nil {
return -1
}
for i := range r.Actions {
if r.Actions[i].LocalRelativePath == action.LocalRelativePath && r.Actions[i].RemoteKey == action.RemoteKey {
return i
}
}
return -1
}
func writeRestoreDryRunSummary(out io.Writer, report *RestoreReport) error {
if out == nil {
return fmt.Errorf("output writer is required")
}
if report == nil {
return fmt.Errorf("restore report is required")
}
if _, err := fmt.Fprintf(out, "Restore plan for %s/%s\n", report.Campaign, report.SessionID); err != nil {
return err
}
if _, err := fmt.Fprintf(out, "Remote run: %s\n", report.RunID); err != nil {
return err
}
if _, err := fmt.Fprintf(out, "Would download: %d\n", report.Plan.Download); err != nil {
return err
}
if _, err := fmt.Fprintf(out, "Would skip unchanged: %d\n", report.Plan.SkipSame); err != nil {
return err
}
if _, err := fmt.Fprintf(out, "Conflicts: %d\n", report.Plan.Conflicts); err != nil {
return err
}
for _, action := range report.Actions {
line := ""
switch action.Status {
case "planned_download":
line = "Would download: " + action.LocalRelativePath
case "skipped_same":
line = "Would skip unchanged: " + action.LocalRelativePath
case "conflict":
line = "Conflict: " + action.LocalRelativePath
default:
line = strings.TrimSpace(action.Kind) + ": " + action.LocalRelativePath
}
if _, err := fmt.Fprintln(out, line); err != nil {
return err
}
}
return nil
}
func writeRestoreSuccessSummary(out io.Writer, report *RestoreReport) error {
if out == nil {
return fmt.Errorf("output writer is required")
}
if report == nil {
return fmt.Errorf("restore report is required")
}
if _, err := fmt.Fprintf(out, "Restored session archive for %s/%s\n", report.Campaign, report.SessionID); err != nil {
return err
}
if _, err := fmt.Fprintf(out, "Remote run: %s\n", report.RunID); err != nil {
return err
}
if _, err := fmt.Fprintf(out, "Downloaded: %d\n", report.Execution.Downloaded); err != nil {
return err
}
if _, err := fmt.Fprintf(out, "Skipped unchanged: %d\n", report.Plan.SkipSame); err != nil {
return err
}
if _, err := fmt.Fprintf(out, "Conflicts: %d\n", report.Plan.Conflicts); err != nil {
return err
}
return nil
}
func persistRestoreReport(store artifacts.Store, cfg *config.Config, report *RestoreReport) (string, error) {
if store == nil {
return "", fmt.Errorf("artifact store is required")
}
if cfg == nil || cfg.Pipeline == nil || cfg.Session == nil {
return "", fmt.Errorf("resolved config with pipeline/session is required")
}
if report == nil {
return "", fmt.Errorf("restore report is required")
}
sessionRoot := artifacts.SessionWorkDirForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID)
reportPath := filepath.Join(sessionRoot, filepath.FromSlash(report.reportPathRel))
payload, err := json.MarshalIndent(report, "", " ")
if err != nil {
return "", fmt.Errorf("marshal restore report: %w", err)
}
payload = append(payload, '\n')
if err := store.WriteFileAtomic(reportPath, payload, 0o644); err != nil {
return "", fmt.Errorf("write restore report %q: %w", reportPath, err)
}
return reportPath, nil
}

View File

@@ -0,0 +1,413 @@
package app
import (
"bytes"
"context"
"fmt"
"os"
"path/filepath"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestExecuteRestoreHelp(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "--help"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0", code)
}
if stderr.Len() != 0 {
t.Fatalf("stderr = %q, want empty", stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "Usage: narratio session restore <session_id>") {
t.Fatalf("stdout = %q, want restore usage", out)
}
if !strings.Contains(out, "--include-audio") {
t.Fatalf("stdout = %q, want --include-audio flag", out)
}
if !strings.Contains(out, "--campaign") {
t.Fatalf("stdout = %q, want --campaign flag", out)
}
}
func TestExecuteRestoreRecognizedAndReturnsNYI(t *testing.T) {
origStoreFn := newObjectStoreFromConfigFn
origDiscoverFn := discoverRemoteCurrentStateFn
origPlanFn := buildRestorePlanFn
origExecuteFn := executeRestorePlanFn
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
discoverRemoteCurrentStateFn = origDiscoverFn
buildRestorePlanFn = origPlanFn
executeRestorePlanFn = origExecuteFn
})
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
return &storage.FakeBackend{}, nil
}
discoverRemoteCurrentStateFn = func(context.Context, *config.Config, storage.ObjectStore) (*RemoteCurrentState, error) {
return &RemoteCurrentState{
SessionID: "2026-05-03",
Campaign: "sample-campaign",
RunID: "20260519T010203Z-a1b2c3d4",
}, nil
}
buildRestorePlanFn = func(context.Context, *config.Config, *RemoteCurrentState, storage.ObjectStore, RestorePlanOptions) (*RestorePlan, error) {
return &RestorePlan{
Actions: []RestoreAction{
{
Kind: RestoreActionDownload,
LocalRelativePath: "manifest.json",
RemoteKey: "dnd/campaigns/sample-campaign/sessions/2026-05-03/current/manifest.json",
Reason: "local file missing",
},
},
DownloadCount: 1,
}, nil
}
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{
"session", "restore", "2026-05-03",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--dry-run",
"--force",
"--include-audio",
},
&stdout,
&stderr,
)
if code != 0 {
t.Fatalf("exit code = %d, want 0 for --dry-run restore planning; stderr=%q", code, stderr.String())
}
if stderr.Len() != 0 {
t.Fatalf("stderr = %q, want empty", stderr.String())
}
outText := stdout.String()
if !strings.Contains(outText, "Restore plan for sample-campaign/2026-05-03") {
t.Fatalf("stdout = %q, want restore plan summary", outText)
}
if !strings.Contains(outText, "Would download: 1") {
t.Fatalf("stdout = %q, want plan count output", outText)
}
if !strings.Contains(outText, "Would download: manifest.json") {
t.Fatalf("stdout = %q, want action output", outText)
}
manifestPath := artifacts.SessionManifestPathForCampaign(workspaceRoot, "sample-campaign", "2026-05-03")
if _, err := os.Stat(manifestPath); !os.IsNotExist(err) {
t.Fatalf("manifest should not be created during phase-4 restore planning; stat err=%v", err)
}
reportPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "reports", "restore-latest.json")
if _, err := os.Stat(reportPath); !os.IsNotExist(err) {
t.Fatalf("restore report should not be written during dry-run; stat err=%v", err)
}
}
func TestExecuteRestoreRejectsUnexpectedPositionalArguments(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "extra"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "restore: unexpected positional arguments") {
t.Fatalf("stderr = %q, want positional-args failure", stderr.String())
}
}
func TestExecuteRestoreFailsWhenStorageBackendNotConfigured(t *testing.T) {
origStoreFn := newObjectStoreFromConfigFn
origDiscoverFn := discoverRemoteCurrentStateFn
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
discoverRemoteCurrentStateFn = origDiscoverFn
})
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeRestoreConfigWithoutStorage(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "no remote object store backend is configured") {
t.Fatalf("stderr = %q, want storage backend preflight failure", stderr.String())
}
}
func TestExecuteRestoreDiscoveryErrorSurfaced(t *testing.T) {
origStoreFn := newObjectStoreFromConfigFn
origDiscoverFn := discoverRemoteCurrentStateFn
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
discoverRemoteCurrentStateFn = origDiscoverFn
})
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
return &storage.FakeBackend{}, nil
}
discoverRemoteCurrentStateFn = func(context.Context, *config.Config, storage.ObjectStore) (*RemoteCurrentState, error) {
return nil, fmt.Errorf("remote current run pointer missing: %q", "dnd/campaigns/sample-campaign/sessions/2026-05-03/current/run_id.txt")
}
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "remote current run pointer missing") {
t.Fatalf("stderr = %q, want discovery error context", stderr.String())
}
}
func TestExecuteRestoreLoadsSecretsBeforeObjectStoreInit(t *testing.T) {
origStoreFn := newObjectStoreFromConfigFn
origDiscoverFn := discoverRemoteCurrentStateFn
origPlanFn := buildRestorePlanFn
origExecuteFn := executeRestorePlanFn
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
discoverRemoteCurrentStateFn = origDiscoverFn
buildRestorePlanFn = origPlanFn
executeRestorePlanFn = origExecuteFn
})
const accessKeyEnv = "OBJECT_STORAGE_KEY_ID"
const secretKeyEnv = "OBJECT_STORAGE_KEY"
restoreEnv := func(name string) {
value, exists := os.LookupEnv(name)
_ = os.Unsetenv(name)
t.Cleanup(func() {
if exists {
_ = os.Setenv(name, value)
return
}
_ = os.Unsetenv(name)
})
}
restoreEnv(accessKeyEnv)
restoreEnv(secretKeyEnv)
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
secretsDir := filepath.Join(t.TempDir(), "secrets")
mustWriteTestFile(t, filepath.Join(secretsDir, accessKeyEnv), "test-access-key-id\n")
mustWriteTestFile(t, filepath.Join(secretsDir, secretKeyEnv), "test-secret-key\n")
f, err := os.OpenFile(pipelinePath, os.O_APPEND|os.O_WRONLY, 0)
if err != nil {
t.Fatalf("open pipeline config for append: %v", err)
}
defer f.Close()
if _, err := f.WriteString("\nsecrets:\n env_dir: " + secretsDir + "\n"); err != nil {
t.Fatalf("append secrets config: %v", err)
}
storeInitCalled := false
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
storeInitCalled = true
gotID, okID := os.LookupEnv(accessKeyEnv)
if !okID || gotID != "test-access-key-id" {
return nil, fmt.Errorf("missing or unexpected %s: %q (set=%t)", accessKeyEnv, gotID, okID)
}
gotSecret, okSecret := os.LookupEnv(secretKeyEnv)
if !okSecret || gotSecret != "test-secret-key" {
return nil, fmt.Errorf("missing or unexpected %s: %q (set=%t)", secretKeyEnv, gotSecret, okSecret)
}
return &storage.FakeBackend{}, nil
}
discoverRemoteCurrentStateFn = func(context.Context, *config.Config, storage.ObjectStore) (*RemoteCurrentState, error) {
return &RemoteCurrentState{
SessionID: "2026-05-03",
Campaign: "sample-campaign",
RunID: "20260519T010203Z-a1b2c3d4",
}, nil
}
buildRestorePlanFn = func(context.Context, *config.Config, *RemoteCurrentState, storage.ObjectStore, RestorePlanOptions) (*RestorePlan, error) {
return &RestorePlan{
Actions: []RestoreAction{
{
Kind: RestoreActionDownload,
LocalRelativePath: "manifest.json",
RemoteKey: "dnd/campaigns/sample-campaign/sessions/2026-05-03/current/manifest.json",
Reason: "local file missing",
},
},
DownloadCount: 1,
}, nil
}
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{
"session", "restore", "2026-05-03",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--dry-run",
},
&stdout,
&stderr,
)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !storeInitCalled {
t.Fatal("expected object store initialization to be called")
}
}
func TestExecuteRestoreNonDryRunConflictFailsBeforeNYI(t *testing.T) {
origStoreFn := newObjectStoreFromConfigFn
origDiscoverFn := discoverRemoteCurrentStateFn
origPlanFn := buildRestorePlanFn
origExecuteFn := executeRestorePlanFn
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
discoverRemoteCurrentStateFn = origDiscoverFn
buildRestorePlanFn = origPlanFn
executeRestorePlanFn = origExecuteFn
})
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
return &storage.FakeBackend{}, nil
}
discoverRemoteCurrentStateFn = func(context.Context, *config.Config, storage.ObjectStore) (*RemoteCurrentState, error) {
return &RemoteCurrentState{
SessionID: "2026-05-03",
Campaign: "sample-campaign",
RunID: "20260519T010203Z-a1b2c3d4",
}, nil
}
buildRestorePlanFn = func(context.Context, *config.Config, *RemoteCurrentState, storage.ObjectStore, RestorePlanOptions) (*RestorePlan, error) {
return &RestorePlan{
Actions: []RestoreAction{
{Kind: RestoreActionConflict, LocalRelativePath: "transcripts/full.json", RemoteKey: "k", Reason: "local file differs"},
},
ConflictCount: 1,
}, nil
}
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if stdout.Len() != 0 {
t.Fatalf("stdout = %q, want empty on conflict failure", stdout.String())
}
if !strings.Contains(stderr.String(), "restore conflict: 1 conflicting path(s); rerun with --force to overwrite") {
t.Fatalf("stderr = %q, want conflict failure", stderr.String())
}
if strings.Contains(stderr.String(), "phase 4: restore execution") {
t.Fatalf("stderr = %q, should fail before phase-4 NYI boundary", stderr.String())
}
}
func TestExecuteRestoreNonDryRunForceExecutesPlan(t *testing.T) {
origStoreFn := newObjectStoreFromConfigFn
origDiscoverFn := discoverRemoteCurrentStateFn
origPlanFn := buildRestorePlanFn
origExecuteFn := executeRestorePlanFn
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
discoverRemoteCurrentStateFn = origDiscoverFn
buildRestorePlanFn = origPlanFn
executeRestorePlanFn = origExecuteFn
})
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
return &storage.FakeBackend{}, nil
}
discoverRemoteCurrentStateFn = func(context.Context, *config.Config, storage.ObjectStore) (*RemoteCurrentState, error) {
return &RemoteCurrentState{
SessionID: "2026-05-03",
Campaign: "sample-campaign",
RunID: "20260519T010203Z-a1b2c3d4",
}, nil
}
buildRestorePlanFn = func(context.Context, *config.Config, *RemoteCurrentState, storage.ObjectStore, RestorePlanOptions) (*RestorePlan, error) {
return &RestorePlan{
Actions: []RestoreAction{
{Kind: RestoreActionDownload, LocalRelativePath: "transcripts/full.json", RemoteKey: "k", Reason: "local file differs; overwrite with --force"},
},
DownloadCount: 1,
}, nil
}
executeRestorePlanFn = func(context.Context, *config.Config, *RemoteCurrentState, *RestorePlan, *RestoreReport, storage.ObjectStore) (*RestoreExecutionResult, error) {
return &RestoreExecutionResult{DownloadedCount: 1}, nil
}
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "restore", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--force"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "Restored session archive for sample-campaign/2026-05-03") {
t.Fatalf("stdout = %q, want completion summary", stdout.String())
}
if stderr.Len() != 0 {
t.Fatalf("stderr = %q, want empty", stderr.String())
}
}
func writeRestoreConfigWithoutStorage(t *testing.T, workspaceRoot string) (string, string, string) {
t.Helper()
dir := t.TempDir()
pipelinePath := filepath.Join(dir, "pipeline.yml")
campaignPath := writeAppTestCampaignConfig(t, dir)
sessionPath := filepath.Join(dir, "session.yml")
pipelineYAML := `workspace:
root: ` + workspaceRoot + `
whisperx:
transcribe_url: https://example.com/transcribe
`
sessionYAML := `session_id: 2026-05-03
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
if err := os.WriteFile(pipelinePath, []byte(pipelineYAML), 0o644); err != nil {
t.Fatalf("write pipeline config: %v", err)
}
if err := os.WriteFile(sessionPath, []byte(sessionYAML), 0o644); err != nil {
t.Fatalf("write session config: %v", err)
}
mustWriteTestFile(t, filepath.Join(dir, "speakers.yml"), "alice: alice.flac\n")
mustWriteTestFile(t, filepath.Join(dir, "autocorrect.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(dir, "glossary.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(dir, "audio", "alice.flac"), "audio-bytes")
return pipelinePath, campaignPath, sessionPath
}

View File

@@ -0,0 +1,306 @@
package app
import (
"bytes"
"context"
"os"
"path/filepath"
"strings"
"testing"
"time"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/scriptorium"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
"gitea.maximumdirect.net/eric/narratio/internal/stage"
)
func TestRestoreThenRunStageForceAnalyzeUsesRestoredDurableState(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
fake := &storage.FakeBackend{}
cfg, sessionPrefix, manifestKey, runIDKey := seedRestoreCommittedState(t, fake, pipelinePath, campaignPath, sessionPath)
seedRestoreObject(fake, runIDKey, []byte("20260519T010203Z-a1b2c3d4\n"))
seedRestoreObject(fake, manifestKey, restoreWorkflowManifestJSON(t, cfg.Session.SessionID, cfg.Session.Campaign))
seedRestoreObject(fake, sessionPrefix+"transcripts/full.json", []byte(`{"segments":[1,2,3]}`+"\n"))
seedRestoreObject(fake, sessionPrefix+"artifacts/session_recap.md", []byte("# restored recap\n"))
restoreWithStoreAndRealPhases(t, fake)
origExecuteStagesFn := executeStagesFn
t.Cleanup(func() {
executeStagesFn = origExecuteStagesFn
})
executeStagesFn = func(ctx context.Context, cfg *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
if opts.Env == nil {
opts.Env = &Env{}
}
opts.Env.Scriptorium = &scriptorium.NoopRunner{}
return executeStages(ctx, cfg, stages, opts)
}
var stdout bytes.Buffer
var stderr bytes.Buffer
restoreCode := Execute(
[]string{
"session", "restore", cfg.Session.SessionID,
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
},
&stdout,
&stderr,
)
if restoreCode != 0 {
t.Fatalf("restore exit code = %d, want 0; stderr=%q", restoreCode, stderr.String())
}
if stderr.Len() != 0 {
t.Fatalf("restore stderr = %q, want empty", stderr.String())
}
sessionRoot := artifacts.SessionWorkDirForCampaign(workspaceRoot, cfg.Session.Campaign, cfg.Session.SessionID)
mustReadEquals(t, filepath.Join(sessionRoot, "transcripts", "full.json"), `{"segments":[1,2,3]}`+"\n")
mustReadEquals(t, filepath.Join(sessionRoot, "artifacts", "session_recap.md"), "# restored recap\n")
manifestStore := &manifest.LocalStore{}
sessionManifestPath := artifacts.SessionManifestPathForCampaign(workspaceRoot, cfg.Session.Campaign, cfg.Session.SessionID)
beforeAnalyze, err := manifestStore.Load(context.Background(), sessionManifestPath)
if err != nil {
t.Fatalf("load restored session manifest: %v", err)
}
upstreamCompletedAt := map[string]time.Time{}
for _, stageName := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim"} {
rec := beforeAnalyze.Stages[stageName]
if rec == nil || rec.Status != manifest.StatusSucceeded || rec.CompletedAt == nil {
t.Fatalf("restored manifest stage %q = %#v, want succeeded with completion timestamp", stageName, rec)
}
upstreamCompletedAt[stageName] = *rec.CompletedAt
}
stdout.Reset()
stderr.Reset()
runStageCode := Execute(
[]string{
"run-stage", "analyze", cfg.Session.SessionID,
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--force",
"--artifacts", "player_handout",
},
&stdout,
&stderr,
)
if runStageCode != 0 {
t.Fatalf("run-stage exit code = %d, want 0; stderr=%q", runStageCode, stderr.String())
}
if stderr.Len() != 0 {
t.Fatalf("run-stage stderr = %q, want empty", stderr.String())
}
if !strings.Contains(stdout.String(), "stage=analyze executed=1 skipped=0 force=true") {
t.Fatalf("run-stage stdout = %q, want analyze execution summary", stdout.String())
}
playerHandoutPath := filepath.Join(sessionRoot, "artifacts", "player_handout.md")
if _, err := os.Stat(playerHandoutPath); err != nil {
t.Fatalf("restored analyze output %q missing: %v", playerHandoutPath, err)
}
afterAnalyze, err := manifestStore.Load(context.Background(), sessionManifestPath)
if err != nil {
t.Fatalf("load session manifest after run-stage analyze: %v", err)
}
for _, stageName := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim"} {
rec := afterAnalyze.Stages[stageName]
if rec == nil || rec.Status != manifest.StatusSucceeded || rec.CompletedAt == nil {
t.Fatalf("post-analyze manifest stage %q = %#v, want succeeded with completion timestamp", stageName, rec)
}
if !rec.CompletedAt.Equal(upstreamCompletedAt[stageName]) {
t.Fatalf(
"stage %q completion changed: before=%s after=%s",
stageName,
upstreamCompletedAt[stageName].Format(time.RFC3339Nano),
rec.CompletedAt.Format(time.RFC3339Nano),
)
}
}
analyzeRec := afterAnalyze.Stages["analyze"]
if analyzeRec == nil || analyzeRec.Status != manifest.StatusSucceeded {
t.Fatalf("post-analyze stage record = %#v, want succeeded", analyzeRec)
}
runManifestPaths, err := filepath.Glob(filepath.Join(sessionRoot, "runs", "*", "manifest.json"))
if err != nil {
t.Fatalf("glob run manifests: %v", err)
}
if len(runManifestPaths) != 1 {
t.Fatalf("run manifest count = %d, want 1; paths=%v", len(runManifestPaths), runManifestPaths)
}
runManifest, err := manifestStore.LoadRun(context.Background(), runManifestPaths[0])
if err != nil {
t.Fatalf("load run manifest %q: %v", runManifestPaths[0], err)
}
if len(runManifest.RequestedStages) != 1 || runManifest.RequestedStages[0] != "analyze" {
t.Fatalf("run manifest requested_stages = %#v, want [analyze]", runManifest.RequestedStages)
}
if runManifest.Stages["analyze"] == nil || runManifest.Stages["analyze"].Status != manifest.StatusSucceeded {
t.Fatalf("run manifest analyze stage = %#v, want succeeded", runManifest.Stages["analyze"])
}
if runManifest.Stages["prepare"] != nil {
t.Fatalf("run manifest should not include upstream prepare stage, got %#v", runManifest.Stages["prepare"])
}
}
func TestRestoreThenAnalyzeUsesRestoredPreviousCacheWithoutObjectStore(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
appendRestoreWorkflowScriptoriumConfig(t, pipelinePath, `
scriptorium:
binary: scriptorium
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
output_path: artifacts/session_recap.md
inputs:
transcript:
source: narratio.transcript.final_trimmed
required: true
previous_recap:
source: narratio.previous_session.artifact.session_recap
required: true
`)
appendRestoreWorkflowScriptoriumConfig(t, sessionPath, `
previous_session_id: 2026-04-26
`)
fakeStore := &storage.FakeBackend{}
cfg, sessionPrefix, manifestKey, runIDKey := seedRestoreCommittedState(t, fakeStore, pipelinePath, campaignPath, sessionPath)
seedRestoreObject(fakeStore, runIDKey, []byte("20260519T010203Z-a1b2c3d4\n"))
seedRestoreObject(fakeStore, manifestKey, restoreWorkflowManifestJSON(t, cfg.Session.SessionID, cfg.Session.Campaign))
seedRestoreObject(fakeStore, sessionPrefix+"transcripts/final.trimmed.json", []byte(`{"segments":[]}`+"\n"))
seedRestorePreviousCurrent(t, fakeStore, cfg, "# previous recap\n")
restoreWithStoreAndRealPhases(t, fakeStore)
var stdout bytes.Buffer
var stderr bytes.Buffer
restoreCode := Execute(
[]string{
"session", "restore", cfg.Session.SessionID,
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
},
&stdout,
&stderr,
)
if restoreCode != 0 {
t.Fatalf("restore exit code = %d, want 0; stderr=%q", restoreCode, stderr.String())
}
if stderr.Len() != 0 {
t.Fatalf("restore stderr = %q, want empty", stderr.String())
}
sessionRoot := artifacts.SessionWorkDirForCampaign(workspaceRoot, cfg.Session.Campaign, cfg.Session.SessionID)
mustReadEquals(t, filepath.Join(sessionRoot, "transcripts", "final.trimmed.json"), `{"segments":[]}`+"\n")
previousManifestBytes, err := os.ReadFile(filepath.Join(sessionRoot, "previous", "manifest.json"))
if err != nil {
t.Fatalf("read restored previous manifest: %v", err)
}
if !strings.Contains(string(previousManifestBytes), `"session_id":"2026-04-26"`) {
t.Fatalf("restored previous manifest = %q, want previous session id", string(previousManifestBytes))
}
mustReadEquals(t, filepath.Join(sessionRoot, "previous", "artifacts", "session_recap.md"), "# previous recap\n")
scriptoriumFake := &scriptorium.FakeRunner{}
origExecuteStagesFn := executeStagesFn
origObjectStoreFn := newObjectStoreFromConfigFn
objectStoreConstructed := false
t.Cleanup(func() {
executeStagesFn = origExecuteStagesFn
newObjectStoreFromConfigFn = origObjectStoreFn
})
executeStagesFn = func(ctx context.Context, cfg *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
if opts.Env == nil {
opts.Env = &Env{}
}
opts.Env.Scriptorium = scriptoriumFake
return executeStages(ctx, cfg, stages, opts)
}
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
objectStoreConstructed = true
return nil, context.Canceled
}
stdout.Reset()
stderr.Reset()
runStageCode := Execute(
[]string{
"run-stage", "analyze", cfg.Session.SessionID,
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--force",
"--artifacts", "session_recap",
},
&stdout,
&stderr,
)
if runStageCode != 0 {
t.Fatalf("run-stage exit code = %d, want 0; stderr=%q", runStageCode, stderr.String())
}
if stderr.Len() != 0 {
t.Fatalf("run-stage stderr = %q, want empty", stderr.String())
}
if objectStoreConstructed {
t.Fatal("analyze run-stage should not construct object store for previous-session input resolution")
}
if len(scriptoriumFake.RunRequests) != 1 {
t.Fatalf("scriptorium run requests = %d, want 1", len(scriptoriumFake.RunRequests))
}
req := scriptoriumFake.RunRequests[0]
if got := req.InputPaths["transcript"]; got != filepath.Join(sessionRoot, "transcripts", "final.trimmed.json") {
t.Fatalf("transcript input = %q, want trimmed transcript path", got)
}
if got := req.InputPaths["previous_recap"]; got != filepath.Join(sessionRoot, "previous", "artifacts", "session_recap.md") {
t.Fatalf("previous_recap input = %q, want restored previous cache path", got)
}
}
func restoreWorkflowManifestJSON(t *testing.T, sessionID, campaign string) []byte {
t.Helper()
store := &manifest.LocalStore{}
now := time.Date(2026, 5, 19, 23, 0, 0, 0, time.UTC)
m := manifest.New(sessionID, now)
m.Campaign = campaign
m.RunID = "20260519T010203Z-a1b2c3d4"
stages := []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim"}
for i, stageName := range stages {
m.MarkStageSucceeded(stageName, now.Add(time.Duration(i+1)*time.Minute), nil)
}
path := filepath.Join(t.TempDir(), "manifest.json")
if err := store.Save(context.Background(), path, m); err != nil {
t.Fatalf("save workflow manifest fixture: %v", err)
}
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("read workflow manifest fixture: %v", err)
}
return data
}
func appendRestoreWorkflowScriptoriumConfig(t *testing.T, pipelinePath, extra string) {
t.Helper()
f, err := os.OpenFile(pipelinePath, os.O_APPEND|os.O_WRONLY, 0)
if err != nil {
t.Fatalf("open pipeline config for append: %v", err)
}
defer f.Close()
if _, err := f.WriteString(extra); err != nil {
t.Fatalf("append pipeline config: %v", err)
}
}

View File

@@ -5,45 +5,70 @@ import (
"flag" "flag"
"fmt" "fmt"
"io" "io"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config" "gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest" "gitea.maximumdirect.net/eric/narratio/internal/manifest"
) )
// Resume continues execution from the first non-succeeded stage in the manifest. // Resume continues execution from the first non-succeeded stage in the manifest.
func Resume(ctx context.Context, args []string, out io.Writer) error { func Resume(ctx context.Context, args []string, out io.Writer) error {
positionalSessionID, args := pullLeadingSessionID(args)
fs := flag.NewFlagSet("resume", flag.ContinueOnError) fs := flag.NewFlagSet("resume", flag.ContinueOnError)
fs.SetOutput(io.Discard) fs.SetOutput(io.Discard)
var pipelinePath string var pipelinePath string
var campaignPath string
var campaignFilePath string
var sessionPath string var sessionPath string
var sessionID string
var previousSessionID string
var force bool var force bool
var selectedArtifacts artifactSelectionFlag
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)") fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
fs.StringVar(&campaignPath, "campaign", "", "campaign ID")
fs.StringVar(&campaignFilePath, "campaign-file", "", "path to campaign.yml")
fs.StringVar(&sessionPath, "session", "", "path to session.yml") fs.StringVar(&sessionPath, "session", "", "path to session.yml")
fs.StringVar(&previousSessionID, "previous-session-id", "", "expected previous session identifier")
fs.BoolVar(&force, "force", false, "force stage execution") fs.BoolVar(&force, "force", false, "force stage execution")
fs.Var(&selectedArtifacts, "artifacts", "configured artifact names to execute and publish (comma-separated or repeatable)")
if err := fs.Parse(args); err != nil { if err := fs.Parse(args); err != nil {
return fmt.Errorf("resume: invalid flags: %w", err) return fmt.Errorf("resume: invalid flags: %w", err)
} }
if positionalSessionID == "" {
if err := applyParsedSessionIDArg("resume", fs, &sessionID); err != nil {
return err
}
} else {
if fs.NArg() != 0 { if fs.NArg() != 0 {
return fmt.Errorf("resume: unexpected positional arguments") return fmt.Errorf("resume: unexpected positional arguments")
} }
if sessionPath == "" { if err := applyPositionalSessionID("resume", positionalSessionID, &sessionID); err != nil {
return fmt.Errorf("resume: --session is required") return err
} }
resolvedPipelinePath, err := resolvePipelineConfigPath(pipelinePath)
if err != nil {
return fmt.Errorf("resume: %w", err)
} }
if strings.TrimSpace(sessionID) == "" {
cfg, err := config.Load(resolvedPipelinePath, sessionPath) return fmt.Errorf("resume: session_id is required")
}
cfg, err := loadCommandConfig(ctx, pipelinePath, campaignPath, campaignFilePath, sessionPath, config.SessionLoadOptions{
SessionID: sessionID,
PreviousSessionID: previousSessionID,
})
if err != nil { if err != nil {
return fmt.Errorf("resume: %w", err) return fmt.Errorf("resume: %w", err)
} }
if err := config.Validate(cfg); err != nil { if err := config.Validate(cfg); err != nil {
return fmt.Errorf("resume: %w", err) return fmt.Errorf("resume: %w", err)
} }
normalizedArtifacts, err := selectedArtifacts.Normalize()
if err != nil {
return fmt.Errorf("resume: invalid --artifacts: %w", err)
}
if err := validateSelectedArtifacts(cfg, normalizedArtifacts); err != nil {
return fmt.Errorf("resume: %w", err)
}
full := BuildFullPlan() full := BuildFullPlan()
selected := full selected := full
@@ -62,7 +87,10 @@ func Resume(ctx context.Context, args []string, out io.Writer) error {
} }
} }
summary, err := executeStages(ctx, cfg, selected, RunOptions{Force: force}) summary, err := executeStagesFn(ctx, cfg, selected, RunOptions{
Force: force,
SelectedArtifacts: normalizedArtifacts,
})
if err != nil { if err != nil {
return fmt.Errorf("resume: %w", err) return fmt.Errorf("resume: %w", err)
} }
@@ -79,7 +107,11 @@ func Resume(ctx context.Context, args []string, out io.Writer) error {
} }
func loadManifestIfPresent(ctx context.Context, cfg *config.Config) (*manifest.Manifest, error) { func loadManifestIfPresent(ctx context.Context, cfg *config.Config) (*manifest.Manifest, error) {
path := manifestPathFor(cfg) path := artifacts.SessionManifestPathForCampaign(
cfg.Pipeline.Workspace.Root,
cfg.Session.Campaign,
cfg.Session.SessionID,
)
exists, err := fileExists(path) exists, err := fileExists(path)
if err != nil { if err != nil {
return nil, fmt.Errorf("check manifest %q: %w", path, err) return nil, fmt.Errorf("check manifest %q: %w", path, err)

View File

@@ -15,8 +15,8 @@ import (
func TestResumeStartsAfterCompletedStages(t *testing.T) { func TestResumeStartsAfterCompletedStages(t *testing.T) {
workspaceRoot := t.TempDir() workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot) pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json") manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{} store := &manifest.LocalStore{}
m := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC)) m := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
@@ -25,14 +25,14 @@ func TestResumeStartsAfterCompletedStages(t *testing.T) {
if err := store.Save(context.Background(), manifestPath, m); err != nil { if err := store.Save(context.Background(), manifestPath, m); err != nil {
t.Fatalf("save manifest: %v", err) t.Fatalf("save manifest: %v", err)
} }
workRoot := filepath.Join(workspaceRoot, "work", "2026-05-03") workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "raw", "alice.json"), `{"segments":[]}`) mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "raw", "alice.json"), `{"segments":[]}`)
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "speakers.yml"), "match:\n - speaker: Alice\n match: [\"alice\"]\n") mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "speakers.yml"), "match:\n - speaker: Alice\n match: [\"alice\"]\n")
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "autocorrect.yml"), "[]\n") mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "autocorrect.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "terms: []\n") mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "terms: []\n")
var out bytes.Buffer var out bytes.Buffer
err := Resume(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath}, &out) err := Resume(context.Background(), []string{"2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &out)
if err != nil { if err != nil {
t.Fatalf("Resume() error = %v", err) t.Fatalf("Resume() error = %v", err)
} }
@@ -51,8 +51,8 @@ func TestResumeStartsAfterCompletedStages(t *testing.T) {
func TestResumeNoRemainingStages(t *testing.T) { func TestResumeNoRemainingStages(t *testing.T) {
workspaceRoot := t.TempDir() workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot) pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json") manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{} store := &manifest.LocalStore{}
m := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC)) m := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
@@ -64,7 +64,7 @@ func TestResumeNoRemainingStages(t *testing.T) {
} }
var out bytes.Buffer var out bytes.Buffer
err := Resume(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath}, &out) err := Resume(context.Background(), []string{"2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &out)
if err != nil { if err != nil {
t.Fatalf("Resume() error = %v", err) t.Fatalf("Resume() error = %v", err)
} }
@@ -80,8 +80,8 @@ func TestResumeForceRerunsSucceeded(t *testing.T) {
_, _ = w.Write([]byte(`{"source":"resume-force-test","segments":[{"speaker":"alice"}]}`)) _, _ = w.Write([]byte(`{"source":"resume-force-test","segments":[{"speaker":"alice"}]}`))
})) }))
defer srv.Close() defer srv.Close()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL) pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json") manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{} store := &manifest.LocalStore{}
m := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC)) m := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
@@ -93,7 +93,7 @@ func TestResumeForceRerunsSucceeded(t *testing.T) {
} }
var out bytes.Buffer var out bytes.Buffer
err := Resume(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath, "--force"}, &out) err := Resume(context.Background(), []string{"2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--force"}, &out)
if err != nil { if err != nil {
t.Fatalf("Resume() error = %v", err) t.Fatalf("Resume() error = %v", err)
} }
@@ -104,14 +104,14 @@ func TestResumeForceRerunsSucceeded(t *testing.T) {
func TestRunStageExecutesOnlySelectedStage(t *testing.T) { func TestRunStageExecutesOnlySelectedStage(t *testing.T) {
workspaceRoot := t.TempDir() workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot) pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json") manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "2026-05-03") workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "merged.json"), `{"segments":[]}`) mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "base.json"), `{"segments":[]}`)
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "terms: []\n") mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "terms: []\n")
var out bytes.Buffer var out bytes.Buffer
err := RunStage(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath, "polish"}, &out) err := RunStage(context.Background(), []string{"polish", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &out)
if err != nil { if err != nil {
t.Fatalf("RunStage() error = %v", err) t.Fatalf("RunStage() error = %v", err)
} }
@@ -134,10 +134,10 @@ func TestRunStageExecutesOnlySelectedStage(t *testing.T) {
func TestRunStageSkipAndForce(t *testing.T) { func TestRunStageSkipAndForce(t *testing.T) {
workspaceRoot := t.TempDir() workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot) pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json") manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "2026-05-03") workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "merged.json"), `{"segments":[]}`) mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "base.json"), `{"segments":[]}`)
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "terms: []\n") mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "terms: []\n")
store := &manifest.LocalStore{} store := &manifest.LocalStore{}
@@ -148,7 +148,7 @@ func TestRunStageSkipAndForce(t *testing.T) {
} }
var out bytes.Buffer var out bytes.Buffer
err := RunStage(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath, "polish"}, &out) err := RunStage(context.Background(), []string{"polish", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &out)
if err != nil { if err != nil {
t.Fatalf("RunStage() error = %v", err) t.Fatalf("RunStage() error = %v", err)
} }
@@ -157,7 +157,7 @@ func TestRunStageSkipAndForce(t *testing.T) {
} }
out.Reset() out.Reset()
err = RunStage(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath, "--force", "polish"}, &out) err = RunStage(context.Background(), []string{"polish", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--force"}, &out)
if err != nil { if err != nil {
t.Fatalf("RunStage(force) error = %v", err) t.Fatalf("RunStage(force) error = %v", err)
} }
@@ -166,15 +166,61 @@ func TestRunStageSkipAndForce(t *testing.T) {
} }
} }
func TestRunStageTrimExecutes(t *testing.T) { func TestRunStageForceMarksDownstreamStaleAndResumeContinuesFromStale(t *testing.T) {
workspaceRoot := t.TempDir() workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot) pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json") manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "2026-05-03") workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "normalized.json"), `{"segments":[{"id":1},{"id":2}]}`) mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "base.json"), `{"segments":[]}`)
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "terms: []\n")
store := &manifest.LocalStore{}
seed := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
for _, name := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "analyze", "archive", "notify"} {
seed.MarkStageSucceeded(name, time.Date(2026, 5, 3, 10, 1, 0, 0, time.UTC), nil)
}
if err := store.Save(context.Background(), manifestPath, seed); err != nil {
t.Fatalf("save manifest: %v", err)
}
var out bytes.Buffer var out bytes.Buffer
err := RunStage(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath, "trim"}, &out) err := RunStage(context.Background(), []string{"polish", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--force"}, &out)
if err != nil {
t.Fatalf("RunStage(force) error = %v", err)
}
if !strings.Contains(out.String(), "stage=polish executed=1 skipped=0 force=true") {
t.Fatalf("output = %q, want forced polish rerun", out.String())
}
afterForce, err := store.Load(context.Background(), manifestPath)
if err != nil {
t.Fatalf("load manifest after force: %v", err)
}
for _, name := range []string{"normalize", "trim", "analyze", "archive", "notify"} {
if afterForce.Stages[name] == nil || afterForce.Stages[name].Status != manifest.StatusStale {
t.Fatalf("stage %q = %#v, want stale", name, afterForce.Stages[name])
}
}
out.Reset()
err = Resume(context.Background(), []string{"2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &out)
if err != nil {
t.Fatalf("Resume() error = %v", err)
}
if !strings.Contains(out.String(), "executed=5 skipped=0") {
t.Fatalf("output = %q, want resume to execute normalize..notify", out.String())
}
}
func TestRunStageTrimExecutes(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "final.json"), `{"segments":[{"id":1},{"id":2}]}`)
var out bytes.Buffer
err := RunStage(context.Background(), []string{"trim", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &out)
if err != nil { if err != nil {
t.Fatalf("RunStage(trim) error = %v", err) t.Fatalf("RunStage(trim) error = %v", err)
} }
@@ -197,13 +243,13 @@ func TestRunStageTrimExecutes(t *testing.T) {
func TestRunStageNormalizeExecutes(t *testing.T) { func TestRunStageNormalizeExecutes(t *testing.T) {
workspaceRoot := t.TempDir() workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot) pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json") manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "2026-05-03") workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "processed.json"), `{"segments":[{"id":1},{"id":2}]}`) mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "polished.json"), `{"segments":[{"id":1},{"id":2}]}`)
var out bytes.Buffer var out bytes.Buffer
err := RunStage(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath, "normalize"}, &out) err := RunStage(context.Background(), []string{"normalize", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &out)
if err != nil { if err != nil {
t.Fatalf("RunStage(normalize) error = %v", err) t.Fatalf("RunStage(normalize) error = %v", err)
} }

View File

@@ -5,47 +5,74 @@ import (
"flag" "flag"
"fmt" "fmt"
"io" "io"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/config" "gitea.maximumdirect.net/eric/narratio/internal/config"
) )
// Run executes the pipeline plan and persists manifest state. // Run executes the pipeline plan and persists manifest state.
func Run(ctx context.Context, args []string, out io.Writer) error { func Run(ctx context.Context, args []string, out io.Writer) error {
positionalSessionID, args := pullLeadingSessionID(args)
fs := flag.NewFlagSet("run", flag.ContinueOnError) fs := flag.NewFlagSet("run", flag.ContinueOnError)
fs.SetOutput(io.Discard) fs.SetOutput(io.Discard)
var pipelinePath string var pipelinePath string
var campaignPath string
var campaignFilePath string
var sessionPath string var sessionPath string
var sessionID string
var previousSessionID string
var force bool var force bool
var selectedArtifacts artifactSelectionFlag
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)") fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
fs.StringVar(&campaignPath, "campaign", "", "campaign ID")
fs.StringVar(&campaignFilePath, "campaign-file", "", "path to campaign.yml")
fs.StringVar(&sessionPath, "session", "", "path to session.yml") fs.StringVar(&sessionPath, "session", "", "path to session.yml")
fs.StringVar(&previousSessionID, "previous-session-id", "", "expected previous session identifier")
fs.BoolVar(&force, "force", false, "force stage execution (reserved for future behavior)") fs.BoolVar(&force, "force", false, "force stage execution (reserved for future behavior)")
fs.Var(&selectedArtifacts, "artifacts", "configured artifact names to execute and publish (comma-separated or repeatable)")
if err := fs.Parse(args); err != nil { if err := fs.Parse(args); err != nil {
return fmt.Errorf("run: invalid flags: %w", err) return fmt.Errorf("run: invalid flags: %w", err)
} }
if positionalSessionID == "" {
if err := applyParsedSessionIDArg("run", fs, &sessionID); err != nil {
return err
}
} else {
if fs.NArg() != 0 { if fs.NArg() != 0 {
return fmt.Errorf("run: unexpected positional arguments") return fmt.Errorf("run: unexpected positional arguments")
} }
if sessionPath == "" { if err := applyPositionalSessionID("run", positionalSessionID, &sessionID); err != nil {
return fmt.Errorf("run: --session is required") return err
} }
resolvedPipelinePath, err := resolvePipelineConfigPath(pipelinePath)
if err != nil {
return fmt.Errorf("run: %w", err)
} }
if strings.TrimSpace(sessionID) == "" {
cfg, err := config.Load(resolvedPipelinePath, sessionPath) return fmt.Errorf("run: session_id is required")
}
cfg, err := loadCommandConfig(ctx, pipelinePath, campaignPath, campaignFilePath, sessionPath, config.SessionLoadOptions{
SessionID: sessionID,
PreviousSessionID: previousSessionID,
})
if err != nil { if err != nil {
return fmt.Errorf("run: %w", err) return fmt.Errorf("run: %w", err)
} }
if err := config.Validate(cfg); err != nil { if err := config.Validate(cfg); err != nil {
return fmt.Errorf("run: %w", err) return fmt.Errorf("run: %w", err)
} }
normalizedArtifacts, err := selectedArtifacts.Normalize()
if err != nil {
return fmt.Errorf("run: invalid --artifacts: %w", err)
}
if err := validateSelectedArtifacts(cfg, normalizedArtifacts); err != nil {
return fmt.Errorf("run: %w", err)
}
stages := BuildFullPlan() stages := BuildFullPlan()
summary, err := executeStages(ctx, cfg, stages, RunOptions{Force: force}) summary, err := executeStagesFn(ctx, cfg, stages, RunOptions{
Force: force,
SelectedArtifacts: normalizedArtifacts,
})
if err != nil { if err != nil {
return fmt.Errorf("run: %w", err) return fmt.Errorf("run: %w", err)
} }

Some files were not shown because too many files have changed in this diff Show More