288 Commits

Author SHA1 Message Date
545aa6893b Prepare the v1.6.0 release
All checks were successful
ci/woodpecker/push/verify Pipeline was successful
ci/woodpecker/tag/release Pipeline was successful
2026-08-30 20:40:04 +00:00
98139f7e8b Harden release validation 2026-08-30 20:36:23 +00:00
f8fa0a2623 Document release procedure and complete release audit 2026-08-30 19:27:36 +00:00
9af773491b Use shared release scripts in Woodpecker 2026-08-30 19:22:26 +00:00
2a656f0f11 Add guarded release tag command 2026-08-30 19:20:12 +00:00
aec35a2d9b Add release candidate checker 2026-08-30 19:16:02 +00:00
23dc4e2078 Add release asset builder 2026-08-30 19:10:26 +00:00
c812fe3655 Harden pipeline state and plan release upgrades 2026-08-30 18:51:20 +00:00
3da97ca50c Add assembled pipeline configuration regression coverage 2026-08-30 15:26:23 +00:00
4c203d8588 Document pipeline configuration migration workflow 2026-08-30 15:19:30 +00:00
fcb5f825e1 Add split pipeline configuration example bundle 2026-08-30 15:17:46 +00:00
4c57ace2f6 Add semantic configuration profile comparison 2026-08-30 15:10:38 +00:00
dde7f76ecb Add configuration source reporting 2026-08-30 14:56:22 +00:00
a102db36af Add read-only configuration inspection commands 2026-08-30 14:43:25 +00:00
b5b1d22011 Expand family publish policies into concrete rules 2026-08-30 14:30:07 +00:00
c91599ef36 Add family artifact selection and provenance 2026-08-30 14:27:42 +00:00
257f10c9fb Resolve corresponding artifact family dependencies 2026-08-30 14:20:03 +00:00
fb9a4d14f4 Expand canonical party artifact families 2026-08-30 14:15:57 +00:00
c4435b76c4 Prepare canonical party and derived players inputs 2026-08-30 14:08:35 +00:00
61000a9466 Resolve canonical parties with campaign configuration 2026-08-30 14:01:51 +00:00
7e4ceb3d48 Add canonical party domain and players projection 2026-08-30 13:51:12 +00:00
4e991fa21d Expose pipeline profile selection and provenance 2026-08-30 13:41:26 +00:00
f86b17045d Unify application configuration loading 2026-08-30 13:32:45 +00:00
f302488075 Add named pipeline profile composition 2026-08-30 13:25:02 +00:00
8c1171478d Complete downstream semantic resume coverage 2026-08-30 13:14:33 +00:00
7ee637803d Protect transcript refinement resume semantics 2026-08-30 13:03:13 +00:00
4d6086fefb Protect initial pipeline stage resume semantics 2026-08-30 12:54:38 +00:00
82cb53e107 Add semantic configuration resume evidence 2026-08-30 12:44:26 +00:00
b97b12da7f Add explicit pipeline configuration imports 2026-08-30 12:33:43 +00:00
49ea747b17 Add presence-aware configuration composition 2026-08-30 12:21:27 +00:00
3b5e9db41f Plan pipeline configuration improvements 2026-08-30 03:41:56 +00:00
8b4b328c4e Prepare the v1.5.0 release notes
All checks were successful
ci/woodpecker/push/verify Pipeline was successful
ci/woodpecker/tag/release Pipeline was successful
2026-08-29 23:13:39 +00:00
ee2b8e63e6 Report the embedded release version 2026-08-29 23:13:01 +00:00
51edd384c0 Remove completed feature roadmaps 2026-08-29 23:11:19 +00:00
b804d0f2c8 Share analysis dependent staling traversal 2026-08-29 20:50:43 +00:00
fd5ccc668b Use one execution plan throughout the runner 2026-08-29 20:49:43 +00:00
0dc8ff9b52 Exclude external paths from analysis fingerprints 2026-08-29 20:45:27 +00:00
8657a28bdb Preserve flag-based artifact regeneration arguments 2026-08-29 20:43:44 +00:00
a9c5e4ad4e Recheck bounded prerequisites under lock 2026-08-29 20:42:51 +00:00
2176b4371d Close out the artifact workflow roadmap 2026-08-29 20:19:19 +00:00
effc10d75b Finalize artifact workflow documentation 2026-08-29 20:14:28 +00:00
5887839aa1 Add assembled workflow compatibility coverage 2026-08-29 20:08:48 +00:00
3128bef20a Add artifact-aware analyze resume planning 2026-08-29 20:01:57 +00:00
4e4e2b7d96 Preserve incremental analysis state on failure 2026-08-29 19:49:48 +00:00
99f4f9a0db Execute incremental analysis artifact plans 2026-08-29 19:43:28 +00:00
6abdd67bb5 Add incremental analysis work planning 2026-08-29 19:26:10 +00:00
c32e0c401f Add versioned analysis fingerprint reconciliation 2026-08-29 19:18:28 +00:00
ab5751459a Add deterministic analysis input identities 2026-08-29 19:02:16 +00:00
62de6abdbf Make configured artifacts manifest authoritative 2026-08-29 18:46:40 +00:00
903dc70682 Persist partial analysis artifact results 2026-08-29 18:36:21 +00:00
23c714da66 Add versioned analysis artifact state 2026-08-29 18:27:44 +00:00
6639775d7d Add the artifact regeneration command 2026-08-29 18:19:54 +00:00
966b95b176 Enforce bounded run prerequisites 2026-08-29 18:17:05 +00:00
3bcf2c08dd Add bounded run and plan commands 2026-08-29 18:07:24 +00:00
700ab655ca Add shared bounded pipeline plans 2026-08-29 18:02:37 +00:00
85c5647385 Separate stage order from invalidation dependencies 2026-08-29 18:00:22 +00:00
2ef7c76d99 Add post-transcript implementation plan 2026-08-29 17:47:15 +00:00
9bc1b0feda Plan post-transcript artifact regeneration 2026-08-29 17:29:08 +00:00
5cec84a4a7 Remove redundant prepared input APIs
All checks were successful
ci/woodpecker/push/verify Pipeline was successful
ci/woodpecker/tag/release Pipeline was successful
2026-08-29 16:19:22 +00:00
abfbe42d61 Snapshot verified references for extraction 2026-08-29 16:18:40 +00:00
e7e3bef1e4 Reject special files before preparing inputs 2026-08-29 16:13:36 +00:00
a68e8e31a4 Complete Notarius reference compatibility audit 2026-08-29 15:49:36 +00:00
3a9e60cda9 Document prepared Notarius references 2026-08-29 15:40:23 +00:00
905ff03ccc Prove prepared reference extraction lifecycle 2026-08-29 15:31:33 +00:00
495f7bcde4 Bind prepared references to extraction identity 2026-08-29 15:24:31 +00:00
51e0e8c5d0 Pass reference bindings to Notarius 2026-08-29 15:16:06 +00:00
a2409a1fd1 Verify prepared inputs from manifest evidence 2026-08-29 15:11:59 +00:00
b3363f87d6 Prepare optional spell catalog inputs 2026-08-29 15:03:44 +00:00
5831c0c9e6 Add Notarius reference configuration vocabulary 2026-08-29 14:55:24 +00:00
42ed81cbe1 Update Notarius integration and plan references 2026-08-29 14:41:22 +00:00
e433c86203 Close out the completed audit 2026-08-11 12:53:31 +00:00
a2a144dffa Close diagnostic and restore coverage gaps 2026-08-11 12:28:55 +00:00
8ef6e99d69 Bound remote control object reads 2026-08-11 03:43:18 +00:00
2545faef6c Dispose subprocess descendants after leader exit 2026-08-11 03:22:21 +00:00
80be8be4d6 Confine subprocess diagnostics and retain redacted tails 2026-08-11 03:11:27 +00:00
801adb385d Reconcile lifecycle documentation 2026-08-10 23:18:29 +00:00
feba7b9d74 Enforce continuous validation 2026-08-10 23:08:52 +00:00
b89224bbde Simplify stage and Audita contracts 2026-08-10 22:43:22 +00:00
131ffd9887 Make analyze input resolution deterministic 2026-08-10 22:36:21 +00:00
af492c9e97 Centralize effective artifact selection 2026-08-10 22:27:06 +00:00
f39fc94610 Bind extraction reuse to transcript identity 2026-08-10 22:15:53 +00:00
4e4eff6ba7 Centralize extraction bundle evidence 2026-08-10 22:06:48 +00:00
8ff1b4fa66 Enforce requested adapter output paths 2026-08-10 21:57:54 +00:00
702f622e18 Harden prepare and transcribe transitions 2026-08-10 21:53:54 +00:00
9da2c1e144 Stream WhisperX uploads safely 2026-08-10 21:49:44 +00:00
32653f54f9 Make configuration truthful and clean remote session files 2026-08-10 21:41:00 +00:00
72a200968a Harden configuration validation 2026-08-10 21:32:15 +00:00
b39b68add7 Unify previous source resolution 2026-08-10 21:20:59 +00:00
d9fa1d9328 Bind audio cache reuse to remote identity 2026-08-10 21:09:21 +00:00
8375ad83f3 Serialize restore recovery and rebase manifest paths 2026-08-10 21:01:28 +00:00
4158394dcf Bind restore to committed remote snapshots 2026-08-10 20:42:43 +00:00
eac7e155a5 Persist retryable post-publish cleanup obligations 2026-08-10 20:29:00 +00:00
0cf2cbfeb3 Make remote publish locks generation-safe 2026-08-10 20:17:54 +00:00
361dbb4ca8 Publish immutable remote commits 2026-08-10 20:05:24 +00:00
d6deccf3e8 Add immutable remote commit reader 2026-08-10 19:47:12 +00:00
ee747243fe Terminalize handled invocation failures 2026-08-10 19:34:46 +00:00
a1ceb457e9 Unify invocation manifest identity 2026-08-10 19:25:03 +00:00
9900211fa4 Confine publish archive reads 2026-08-10 19:16:18 +00:00
60cebf0e4b Redact and cap subprocess diagnostics 2026-08-10 18:55:49 +00:00
7bd575187e Terminate owned subprocess trees 2026-08-10 18:44:10 +00:00
ab5a7e8e3d Bound external result file reads 2026-08-10 18:30:28 +00:00
99b2e1cd81 Harden API key file loading 2026-08-10 18:24:29 +00:00
363313d99c Confine cleanup and use held session locks 2026-08-10 18:14:09 +00:00
18ddf00d3d Confine local file installation paths 2026-08-10 17:59:29 +00:00
59f3fe3d1d Make atomic file replacement crash durable 2026-08-10 17:44:38 +00:00
1dccf5f140 Validate portable workspace identifiers 2026-08-10 17:35:52 +00:00
0b40cf8026 Make ordinary workspaces group shareable 2026-08-10 17:27:00 +00:00
a7ec195587 Prepare a staged implementation plan to address the audit findings. 2026-08-10 17:08:57 +00:00
13de820931 Finalize codebase audit findings 2026-08-10 15:14:54 +00:00
14ef59aaed Document test suite policy audit conclusions 2026-08-10 15:04:05 +00:00
f387222fce Document maintainability audit conclusions 2026-08-10 14:53:19 +00:00
9cb9008dfc Document analyze dependency audit findings 2026-08-10 14:38:49 +00:00
083decc5b4 Document extraction audit findings 2026-08-10 14:24:03 +00:00
57cac5d3f7 Document ordinary stage audit findings 2026-08-10 14:09:54 +00:00
0a772e03b4 Document external adapter audit findings 2026-08-10 13:53:49 +00:00
0920062a38 Document configuration and composition audit findings 2026-08-10 13:33:59 +00:00
39afe644eb Document restore and previous-state audit findings 2026-08-10 13:16:35 +00:00
e3ee3de10a Audit publish commit and cleanup behavior 2026-08-10 12:55:10 +00:00
9c72db56e9 Audit artifact paths and filesystem safety 2026-08-10 12:42:49 +00:00
bb2d606dbb Audit runner and manifest lifecycle behavior 2026-08-10 12:28:36 +00:00
9850767a8a Establish the codebase audit baseline and contract map 2026-08-10 12:12:48 +00:00
74e2d21de5 Close the completed roadmap documents 2026-08-10 03:24:49 +00:00
7cb18a1a40 Reconcile promotion and manifest documentation 2026-08-10 03:04:20 +00:00
b556fc2f4f Clear superseded session stage result details 2026-08-10 02:53:28 +00:00
b99bd38eb4 Harden bundle promotion against symlink replacement 2026-08-10 02:45:51 +00:00
701b6726d7 Reconcile Notarius extraction documentation 2026-08-10 02:10:37 +00:00
665039f4dc Support atomic directory promotion across platforms 2026-08-10 01:58:52 +00:00
ef8dae776e Enforce canonical Notarius bundle paths 2026-08-10 01:50:31 +00:00
d01775b68a Exclude staged Notarius bundles from publish uploads 2026-08-10 01:43:43 +00:00
0d6f2dd0ce Invalidate downstream results when stages are replaced 2026-08-10 01:38:24 +00:00
df40cbec6e Document and validate Notarius extraction workflows 2026-08-10 00:42:44 +00:00
0341e0c7c0 Publish and inspect configured extraction artifacts 2026-08-10 00:32:24 +00:00
39af7d4f3c Integrate extraction artifacts into analysis catalog 2026-08-10 00:24:02 +00:00
bba582b4ca Integrate extraction lifecycle and resume validation 2026-08-10 00:14:46 +00:00
1f16a85330 Implement direct Notarius extraction execution 2026-08-09 23:59:50 +00:00
f9482639d4 Add the Notarius subprocess adapter 2026-08-09 23:47:38 +00:00
dce721cdbd Add safe immutable directory promotion 2026-08-09 23:37:02 +00:00
98734644d6 Add Notarius configuration and extraction source policy 2026-08-09 23:30:32 +00:00
951383226c Add artifact provenance and stage skip outcomes 2026-08-09 23:20:32 +00:00
df58595d1e Remove completed documentation alignment roadmaps 2026-08-09 22:02:07 +00:00
c3c14e7468 Complete documentation alignment roadmap 2026-08-09 21:53:04 +00:00
e7319ea016 Align internal documentation and maintained examples 2026-08-09 21:50:21 +00:00
bd2d5e2496 Clarify user and integration documentation contracts 2026-08-09 21:41:23 +00:00
115a44f629 Align documentation entry points and internal overview 2026-08-09 21:32:11 +00:00
18411dc5b5 Move contributor guidance to its canonical location 2026-08-09 21:27:58 +00:00
e23dc1ab6e Refocus the Narratio architecture policy 2026-08-09 21:26:18 +00:00
e1359ea227 Adopt canonical documentation ownership policy 2026-08-09 21:23:50 +00:00
7fdd99ec27 Prepare roadmap for documentation policy update 2026-08-09 21:21:00 +00:00
a90231ce0c Implement support for passing a session_id variable to scriptorium to support sticky routing 2026-07-02 21:04:37 -05:00
ed879b8bb0 Clean up obsolete placeholder code 2026-07-02 20:47:08 -05:00
717451512a Implemented new campaign/session stable inputs and corresponding input source references
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-27 09:34:05 -05:00
3ddb3a947b Update pipeline defaults so trim is enabled when omitted 2026-05-27 08:35:02 -05:00
c6632d5576 Bugfix in the seriatim adapter
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-27 08:09:22 -05:00
ffc07922c7 Cleanup following the render stage implementation and remove the completed roadmap
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-25 08:35:18 -05:00
f3310d4d16 Finalize render documentation across operations, integrations, troubleshooting, and roadmap status 2026-05-25 00:48:15 +00:00
88cee96d8d Finish render rollout with markdown publish defaults, analyze guidance, and docs updates 2026-05-25 00:46:27 +00:00
2fece10215 Implement render stage runtime and integrate it into pipeline execution 2026-05-25 00:40:06 +00:00
0658f2f642 Add render artifact model, config, and Seriatim adapter contracts 2026-05-25 00:28:01 +00:00
a51228c803 Add a documentation roadmap for the upcoming render stage feature 2026-05-24 19:15:42 -05:00
4491fb5ccd Final documentation cleanup for v1.0.0 release
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-23 11:28:03 -05:00
30b905765c Remove the deprecated narratio resume command 2026-05-23 11:25:08 -05:00
03eac70881 Mark cleanup roadmap stages as implemented 2026-05-23 16:05:21 +00:00
0f7e6b979f Deduplicate locks add/remove session-id and source parsing 2026-05-23 16:03:20 +00:00
c366912586 Extract shared read-only session inspection checks 2026-05-23 16:00:31 +00:00
9fe44cd00d Centralize Scriptorium input source policy across config, analyze, and previous-cache 2026-05-23 15:50:29 +00:00
094b0d2532 Centralize path-safe root joins and atomic file operations 2026-05-23 15:45:31 +00:00
98649f4d81 Add a roadmap to implement the remaining items identfied by the code quality audit 2026-05-23 10:35:26 -05:00
8a559efd5b Audit code quality and deduplication opportunities 2026-05-23 10:10:21 -05:00
72deccb4e2 Implement final changes from the code quality and deduplication opportunity audit 2026-05-23 10:05:14 -05:00
5620fc5bcf Refresh CLI and internal restore documentation for current behavior 2026-05-23 14:07:11 +00:00
be57e675e0 Split operator helper implementations by command responsibility 2026-05-23 14:03:31 +00:00
3971443831 Centralize remote current-state loading and preserve caller policy 2026-05-23 13:56:53 +00:00
a6b0c33e9f Unify session-aware CLI parsing and add session-id compatibility 2026-05-23 13:47:39 +00:00
96b886e711 Align internal publish terminology across stage, app, and artifacts 2026-05-23 13:39:15 +00:00
7d584ee6cd Centralize artifact source and publish destination policy 2026-05-23 13:28:25 +00:00
572a112c31 Consolidate path safety, temp downloads, and cleanup validation helpers 2026-05-23 13:20:28 +00:00
ea87c335d6 Add a roadmap to implement the high-priority items revealed by the code quality audit 2026-05-23 08:10:00 -05:00
7169ff04df Audit code quality and deduplication opportunities 2026-05-23 08:08:24 -05:00
ef1f650bc0 Mark documentation roadmap complete after final validation sweep 2026-05-23 13:04:16 +00:00
0d02cb9fa0 Rewrite integration documentation and verify maintained examples 2026-05-23 13:01:17 +00:00
0299b128cf Rewrite internal documentation for current stage and state contracts 2026-05-23 12:57:59 +00:00
d723384888 Rewrite user and operator documentation for current CLI and config behavior 2026-05-23 12:50:46 +00:00
54228055c8 Audit Stage 1 documentation scope and fix broken references 2026-05-23 12:43:50 +00:00
23ed716450 Added a documentation update roadmap 2026-05-23 07:38:17 -05:00
ab59bab044 Initial documentation cleanup pass 2026-05-23 07:11:17 -05:00
71395bb076 Rewrite docs for the publish stage contract and current behavior 2026-05-23 04:51:16 +00:00
79737edf79 Rename publish runtime terminology to published outputs 2026-05-23 04:42:08 +00:00
df2c765b7f Rename archive config and stage contract to publish 2026-05-23 04:30:45 +00:00
f050b9dd54 Added roadmap documentation for the upcoming refactoring of the publish stage 2026-05-22 23:19:44 -05:00
9c9cb54339 Implemented multiple campaign support via a campaign directory registry with explicit campaign IDs 2026-05-22 23:01:27 -05:00
7657ec3ad6 CLI cleanup to consolidate session-related subcommands 2026-05-22 22:09:17 -05:00
cee52aa092 Updated transcript artifact names and canonical paths to use a consistent, role-based nomenclature 2026-05-22 19:05:23 -05:00
e920f3a8d5 Cleaned up and removed legacy configuration surfaces 2026-05-22 18:32:14 -05:00
591c529a09 Updated the analyze stage to accept --artifacts as a CLI flag 2026-05-22 18:01:05 -05:00
7324c5a686 Session configuration templates are now proceeded by narratio session init; all other commands require concrete configuration 2026-05-22 17:38:23 -05:00
d0936fb022 Implemented default config/campaign discovery for narratio session init 2026-05-22 11:36:57 -05:00
2aa074c5cf Implemented narratio publish as a shortcut to run the archive stage only 2026-05-22 11:28:38 -05:00
782d0cf3b9 Upgraded the restore command to download previous session artifcats when configured as inputs for the current session analyze stage 2026-05-21 23:31:01 -05:00
083c01cfa0 Implemented narratio analyze as a shortcut to run the analyze stage only
Some checks failed
ci/woodpecker/tag/release Pipeline failed
2026-05-21 23:02:19 -05:00
2937696024 Add clean command 2026-05-21 22:48:18 -05:00
b817a5b772 Implemented shared S3 audio caching for prepare and restore --include-audio 2026-05-21 22:22:08 -05:00
3022f20beb Simplified the output of narratio artifacts list --remote and narratio status --session-id 2026-05-21 21:20:41 -05:00
ca1ded1821 Bugfix for commands that list artifacts in the S3 backend 2026-05-21 21:00:51 -05:00
3752f3ed28 Added remote artifact listing to narratio status 2026-05-21 20:49:28 -05:00
870c2d69d5 Consolidated addition, removal, and listing of locks under a single narratio locks command 2026-05-21 19:36:14 -05:00
135407ba7c Implemented a centralized secret-backed object-store helper 2026-05-21 19:13:10 -05:00
228c348e42 Implemented operations helper commands for validation, locking, and status 2026-05-21 11:50:20 -05:00
a813bd5a50 Fixed a redundant path bug for previous session artifacts 2026-05-21 10:58:49 -05:00
d8f58dce31 Normalize the default configuration discovery paths for all three config files, and update documentation and tests accordingly 2026-05-21 09:55:56 -05:00
7111edeca4 Add archive promotion locks 2026-05-20 21:40:09 -05:00
3aae4bbb12 Add remote session loading 2026-05-20 20:55:13 -05:00
b29d8eeb50 Add campaign configuration support 2026-05-20 20:41:28 -05:00
dffb432537 Removed completed roadmap for previous_session artifacts 2026-05-20 20:15:47 -05:00
2dd38c7913 Refine campaign and remote session roadmap 2026-05-20 20:15:05 -05:00
bc2ade38d9 Finalize previous-session artifact documentation and restore-analyze continuity coverage 2026-05-20 15:17:04 +00:00
5be831eb13 Restore archived previous-session cache files with session state 2026-05-20 15:05:48 +00:00
cae4d99a89 Archive durable previous-session cache files with session state 2026-05-20 15:03:30 +00:00
e09dc0512d Add analyze integration coverage for previous-session inputs 2026-05-20 15:01:27 +00:00
ae82bc1ce0 Resolve canonical previous-session artifact sources from prepared previous cache 2026-05-20 14:59:28 +00:00
01eb7aa1aa Add prepare rerun guidance for unresolved previous-session analyze inputs 2026-05-20 14:55:44 +00:00
2ca700195c Integrate previous-session artifact hydration into prepare stage 2026-05-20 14:53:22 +00:00
2b08c34539 Add prepare helper to hydrate previous-session artifacts from archive 2026-05-20 14:49:22 +00:00
79f1fc1e09 Add helper to collect previous-session artifact input requirements 2026-05-20 14:37:02 +00:00
9c753270bd Add canonical previous-session artifact source parsing and validation 2026-05-20 14:34:23 +00:00
b907cb01aa Add previous-session workspace path helpers and layout support 2026-05-20 14:31:03 +00:00
7824afd4a5 Add previous session ID templating and CLI support 2026-05-20 14:26:32 +00:00
2a4e1e912c Update documentation to include a roadmap for previous session artifact support 2026-05-20 09:12:17 -05:00
dd03c09d75 Fixed a bug in the S3 credential loading for the restore command
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-19 22:51:26 -05:00
5bc8e8683f Documentation update for the restore subcommand 2026-05-19 22:32:55 -05:00
648001a8fe Add workflow integration tests for the restore command 2026-05-19 22:23:25 -05:00
6684774f52 Add restore report and operator summary 2026-05-19 22:15:40 -05:00
f3b63bd5e5 Implement restore execution for the restore subcommand 2026-05-19 22:06:48 -05:00
23d6470b0f Implement restore planning for the restore subcommand 2026-05-19 21:58:38 -05:00
128449040f Implement remote current-state discovery for the restore subcommand 2026-05-19 21:49:19 -05:00
02ab106ade Implement initial CLI command for narratio restore, and extract shared helper functions from the run stages 2026-05-19 21:39:13 -05:00
c128970f58 Updated documentation to remove the completed runtime artifacts roadmap and add a new restore subcommand roadmap 2026-05-19 21:21:06 -05:00
d001baa660 Use artifact source IDs for archive promotion 2026-05-19 20:05:24 -05:00
c5c35cd3b4 Moved example configuration from docs/examples/ to top-level examples/
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-19 19:46:40 -05:00
574b1cde6c Update documentation for the new analyze stage and artifact registry 2026-05-19 19:42:28 -05:00
ebb21b9201 Removed the legacy built-in session_recap from the analyze stage 2026-05-19 19:26:07 -05:00
958f446387 Add archive stage integration test for the new analyze stage features 2026-05-19 19:12:10 -05:00
86caf4b222 Update analyze-stage metadata and manifest output 2026-05-19 19:07:44 -05:00
e38ed8ba97 Refactor the analyze stage to actually produce the configured artifacts 2026-05-19 18:58:28 -05:00
3e79cf4724 Update artifact resolution so configured artifact IDs are resolved through the runtime catalog 2026-05-19 18:49:27 -05:00
859ae1ae10 Add abstractions for the internal artifact catalog 2026-05-19 18:43:55 -05:00
c63ecbab32 Add new configuration fields and CLI flags for the upcoming analyze stage enhancements 2026-05-19 18:36:30 -05:00
8480b74283 Updated the roadmap for configurable artifact generation 2026-05-19 11:21:28 -05:00
087869f7fa Added a roadmap for new work to support configurable artifacts defined at runtime 2026-05-19 09:26:46 -05:00
2b2a314d65 Move documentation for external integrations into the docs/integrations subfolder 2026-05-19 09:12:40 -05:00
08b0f4edc5 Removed legacy transcript artifact aliases 2026-05-19 09:00:41 -05:00
571a289296 Added new internal documentation 2026-05-19 08:47:45 -05:00
9f80635b42 Updated configuration docs to reflect the minimal pipeline config 2026-05-19 08:15:12 -05:00
11a3e174b6 Set default value for workspace.root and updated config documentation 2026-05-19 07:07:19 -05:00
9c5e5d6dc1 Simplified the reference tables in docs/config.md 2026-05-19 06:51:55 -05:00
c4e87f58c7 Complete documentation rebuild 2026-05-18 22:03:59 -05:00
37daab7857 Bugfix involving nested directory creation
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-18 11:52:40 +00:00
2356688cb9 Removed legacy interfaces and old documentation references to the previous on-disk layout 2026-05-18 03:02:22 +00:00
1054b64d9f Implement minimal downstream invalidation after forced upstream reruns 2026-05-18 01:50:51 +00:00
01fb02426c Update the analyze stage to utilize the new artifact package 2026-05-18 01:29:18 +00:00
7dc79e052f Aligned the archive stage with the new work directory layout 2026-05-18 01:13:51 +00:00
cb525c0f72 Implemented run-local stage execution + immediate promotion for core output-producing stages 2026-05-18 00:55:35 +00:00
622677d038 Added run manifest scaffolding and helpers 2026-05-17 21:15:51 +00:00
550288e008 Add campaign-aware workspace path foundation 2026-05-17 20:57:27 +00:00
e58e545686 Audit workspace architecture implementation plan
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-17 13:22:09 -05:00
6ff54c5a0f Documentation update and reorganization 2026-05-17 13:14:28 -05:00
924b5d15c6 Applied a more general bugfix to path-resolution issues in the archive stage 2026-05-17 11:08:11 -05:00
b065663180 Bugfix involving path resolution in the archive stage 2026-05-17 11:03:02 -05:00
3ba564b00f Bugfix involving directory creation during the merge stage 2026-05-17 08:18:57 -05:00
a3986cf0d6 Centralized defaults into internal/config/defaults.go 2026-05-17 07:53:51 -05:00
539601bd16 Updated the merge stage to normalize the per-speaker transcripts before merging them
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-16 23:30:13 -05:00
6ca1c8d6b0 The backend S3 client now resolves credentials from user-configurable environment variables 2026-05-16 23:22:21 -05:00
4b7b50981b Add .gocache to .gitignore and minor documentation cleanup 2026-05-16 23:21:45 -05:00
33f7ae8f2e Simplify downstream tool configuration
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-16 23:30:40 +00:00
d5a9ad38f8 Add post-archive cleanup policies 2026-05-16 23:09:39 +00:00
6fbefb9867 Add session discovery and template support 2026-05-16 22:57:42 +00:00
1665359486 Added a locally generated UX progress report 2026-05-16 20:11:20 +00:00
03f2543927 Created a UX status report 2026-05-16 20:09:53 +00:00
fe9c348092 Document and review S3 archive workflow 2026-05-16 15:24:44 +00:00
f7f8f1a949 Promote current session artifacts to storage 2026-05-16 15:01:01 +00:00
d40c91acde Upload successful run records to storage 2026-05-16 14:43:29 +00:00
ed4dcf1ef7 Updated go.mod 2026-05-16 09:34:43 -05:00
24cce49a70 Download S3 audio during prepare 2026-05-16 14:33:42 +00:00
1e6db89dd4 Add remote storage backend 2026-05-16 14:22:04 +00:00
0454296c81 Add archive storage path configuration 2026-05-16 14:11:59 +00:00
58c6ab2d54 Updated audita configuration to reflect the new audita public CLI
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-16 08:46:48 -05:00
7995c41675 Update the audita integration documentation reference
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-16 08:02:18 -05:00
62551d43a0 Added filesystem-based secrets loading configuration 2026-05-16 07:56:22 -05:00
8395c12dd3 Update documentation to include an implementation roadmap for the archive stage 2026-05-16 07:36:25 -05:00
421 changed files with 74873 additions and 5641 deletions

BIN
.DS_Store vendored

Binary file not shown.

5
.gitignore vendored
View File

@@ -2,6 +2,8 @@
.codex
AGENTS.md
.DS_Store
# ---> Go
# If you prefer the allow list template instead of the deny list, see community template:
# https://github.com/github/gitignore/blob/main/community/Golang/Go.AllowList.gitignore
@@ -22,6 +24,9 @@ AGENTS.md
# Dependency directories (remove the comment below to include it)
# vendor/
# Go cache
.gocache
# Go workspace file
go.work
go.work.sum

View File

@@ -2,39 +2,26 @@ when:
- event: tag
steps:
- name: build-release-assets
image: golang:1.25
validate-release:
image: golang:1.25.5
commands:
- ./scripts/check-release-candidate.sh "$CI_COMMIT_TAG"
build-release-assets:
image: golang:1.25.5
depends_on:
- validate-release
commands:
- |
set -eu
case "$PWD" in
/*) ;;
*) echo "release workspace must have an absolute path" >&2; exit 1 ;;
esac
./scripts/build-release-assets.sh "$CI_COMMIT_TAG" "$PWD/dist"
version="$CI_COMMIT_TAG"
dist="dist"
pkg="gitea.maximumdirect.net/eric/narratio/cmd/narratio"
rm -rf "$dist"
mkdir -p "$dist"
build_binary() {
goos="$1"
goarch="$2"
suffix="$3"
output="$dist/narratio-$version-$goos-$goarch$suffix"
CGO_ENABLED=0 GOOS="$goos" GOARCH="$goarch" \
go build -trimpath -ldflags "-s -w -X gitea.maximumdirect.net/eric/narratio/internal/buildinfo.Version=$version" \
-o "$output" "$pkg"
}
build_binary linux amd64 ""
build_binary linux arm64 ""
build_binary darwin amd64 ""
build_binary darwin arm64 ""
build_binary windows amd64 ".exe"
build_binary windows arm64 ".exe"
- name: publish-release
image: woodpeckerci/plugin-release
publish-release:
image: woodpeckerci/plugin-release:0.3.1
depends_on:
- build-release-assets
settings:
@@ -42,6 +29,8 @@ steps:
from_secret: GITEA_RELEASE_TOKEN
files:
- dist/narratio-*
title: Narratio ${CI_COMMIT_TAG}
note: docs/releases/${CI_COMMIT_TAG}.md
checksum: sha256
checksum-file: SHA256SUMS
checksum-flatten: true

8
.woodpecker/shuffle.yml Normal file
View File

@@ -0,0 +1,8 @@
when:
- event: cron
steps:
shuffled-race-tests:
image: golang:1.25
commands:
- go test -race -shuffle=on -count=3 ./...

47
.woodpecker/verify.yml Normal file
View File

@@ -0,0 +1,47 @@
when:
- event: [push, pull_request]
steps:
tests:
image: golang:1.25
commands:
- go test ./...
race-tests:
image: golang:1.25
depends_on: tests
commands:
- go test -race ./...
static-analysis:
image: golang:1.25
depends_on: tests
commands:
- go vet ./...
build:
image: golang:1.25
depends_on: tests
commands:
- go build ./...
documentation-and-examples:
image: golang:1.25
depends_on: tests
commands:
- go test ./internal/doccheck
- go test ./internal/config -run '^TestExamplesLoadAndValidate$'
cross-build:
image: golang:1.25
depends_on: [race-tests, static-analysis, build, documentation-and-examples]
commands:
- |
set -eu
output_dir="$(mktemp -d)"
trap 'rm -rf "$output_dir"' EXIT
for target in linux/amd64 linux/arm64 darwin/amd64 darwin/arm64 windows/amd64 windows/arm64; do
goos="${target%/*}"
goarch="${target#*/}"
CGO_ENABLED=0 GOOS="$goos" GOARCH="$goarch" go build -o "$output_dir/narratio-$goos-$goarch" ./cmd/narratio
done

310
README.md
View File

@@ -1,288 +1,42 @@
# narratio
`narratio` is a Go orchestration application for processing D&D session audio into transcripts and generated artifacts.
Narratio is a stage-driven Go orchestrator for turning D&D session audio into
polished transcripts, validated Notarius extraction lanes, and generated
artifacts.
## Current Implementation
It runs a deterministic workflow with manifest-driven continuation, remote
publish, and restore support.
Implemented now:
- strict config loading/validation (`pipeline.yml` and `session.yml`)
- local workspace/session layout, locking, and manifest persistence
- resumable stage control (`run`, `plan`, `resume`, `run-stage`, `status`)
- real `prepare`, `transcribe`, `merge`, `polish`, `normalize`, `trim`, and `analyze` stages
- real WhisperX, Seriatim, and Audita adapters
- real Scriptorium subprocess adapter
- optional Scriptorium render diagnostics (`render_debug`)
Not implemented yet:
- `archive` stage behavior
- `notify` stage behavior
- additional analyze artifacts beyond `session_recap`
- generic DAG orchestration
## Config Files
Narratio expects two YAML files:
- `pipeline.yml`: pipeline/workspace settings
- `session.yml`: per-session settings
Pipeline config lookup for CLI commands:
- if `--config <path>` is provided, Narratio uses that path
- if `--config` is omitted, Narratio searches in this order:
- `/usr/local/etc/narratio/pipeline.yml`
- `/etc/narratio/pipeline.yml`
YAML decoding is strict (`KnownFields(true)`), so unknown fields fail fast.
## Canonical Stage Order
1. `prepare`
2. `transcribe`
3. `merge`
4. `polish`
5. `normalize`
6. `trim`
7. `analyze`
8. `archive`
9. `notify`
## Transcript Tiers
- `transcripts/merged.json`: canonical deterministic merged transcript from Seriatim merge
- `transcripts/processed.json`: full raw Audita-polished transcript output
- `transcripts/normalized.json`: Seriatim-normalized transcript from the normalize stage
- `transcripts/trimmed.json`: gameplay-only normalized polished transcript from trim stage
## Normalize Configuration
`pipeline.normalize` is optional. When omitted, Narratio defaults to:
- `output_path: transcripts/normalized.json`
- `output_schema: seriatim-intermediate`
- `report: true`
Allowed `normalize.output_schema` values:
- `seriatim-minimal`
- `seriatim-intermediate`
- `seriatim-full`
`normalize.output_path` is treated as session-workdir-relative when not absolute.
Normalize stage behavior summary:
- normalize runs after `polish` and before `trim`
- normalize resolves `transcripts/processed.json`
- normalize runs Seriatim `normalize` to produce `transcripts/normalized.json`
- normalize diagnostics are written to:
- `artifacts/seriatim.normalize.report.json` (when enabled)
- `logs/seriatim.normalize.stdout.log`
- `logs/seriatim.normalize.stderr.log`
- `config/seriatim.normalize.generated.yml`
## Trim Configuration
`pipeline.trim` is optional. If omitted, no trim config is loaded. If `trim.enabled` is omitted, it defaults to `false`.
When `trim.enabled: true`:
- `trim.output_path` is required
- `trim.bounds.prompt_id` is required
- `trim.bounds.transcript_input_name` is required
- `trim.bounds.output_path` is required
- `trim.bounds.timeout` must be a valid Go duration when provided
- `trim.bounds.render_debug: true` requires `trim.bounds.render_output_path`
- `trim.bounds.profile_id` may be empty to use the prompt default profile
Trim paths are treated as session-workdir-relative when not absolute.
Example trim config:
```yaml
trim:
enabled: true
output_path: "transcripts/trimmed.json"
bounds:
prompt_id: "dnd_session.bounds"
profile_id: ""
transcript_input_name: "transcript"
output_path: "artifacts/session_bounds.json"
timeout: "10m"
render_debug: false
render_output_path: "artifacts/session_bounds.render.json"
seriatim:
report: false
```sh
narratio run 2026-04-04
```
Trim behavior summary:
This requires resolvable `pipeline.yml`, `campaign.yml`, and concrete
`session.yml` files or their explicit command-line alternatives.
- trim discovers and validates `transcripts/normalized.json`
- trim uses Scriptorium bounds (`dnd_session.bounds` by example config) to produce `artifacts/session_bounds.json`
- bounds IDs are validated against the same normalized transcript ID space that Seriatim trim will consume
- trim converts bounds to Seriatim keep selector (for example `10-868`) and runs Seriatim trim
- if trim is disabled, Narratio copies normalized transcript to trimmed transcript and records `trim_action=copy_disabled`
## Documentation
Trim outputs and diagnostics:
- [CLI reference](docs/cli.md) — commands, arguments, flags, and invocation
behavior.
- [Configuration](docs/config.md) — discovery, fields, defaults, and
validation.
- [Operations](docs/operations.md) — runtime workflow, state, publishing,
recovery, and cleanup.
- [Troubleshooting](docs/troubleshooting.md) — symptom-driven diagnosis and
safe remedies.
- [Integration contracts](docs/integrations/) — external tools, formats, and
compatibility expectations.
- [Maintained examples](examples/README.md) — complete copyable configuration
and input files, including the production/testing split bundle.
- `artifacts/session_bounds.json`
- `transcripts/trimmed.json`
- `logs/scriptorium.bounds.stdout.log`
- `logs/scriptorium.bounds.stderr.log`
- `config/scriptorium.bounds.generated.yml`
- `logs/seriatim.trim.stdout.log`
- `logs/seriatim.trim.stderr.log`
- `config/seriatim.trim.generated.yml`
- optional bounds render-debug outputs:
- `artifacts/session_bounds.render.json`
- `logs/scriptorium.bounds.render.stdout.log`
- `logs/scriptorium.bounds.render.stderr.log`
- `config/scriptorium.bounds.render.generated.yml`
## Maintainer Documentation
Render-debug files are diagnostics and are not treated as canonical stage output artifact refs.
## Scriptorium Configuration
`pipeline.scriptorium` is optional. When present, Narratio validates and uses it for analyze-stage artifact generation.
Key points:
- `scriptorium.binary` is required when section is present
- `scriptorium.config_path` is optional
- `scriptorium.timeout` defaults to `10m` when omitted
- `scriptorium.render_debug` enables render diagnostics globally
- artifacts are configured under `scriptorium.artifacts` (map shape supports multiple artifacts)
- enabled artifacts require `prompt_id` and `output_path`
- artifact `render_debug` may override global render setting
- `vars` currently support boolean and string values
Example `session_recap` artifact definition:
```yaml
scriptorium:
binary: "scriptorium"
config_path: "/etc/scriptorium/config.yml"
timeout: "10m"
render_debug: false
artifacts:
session_recap:
enabled: true
prompt_id: "dnd.session_recap"
profile_id: "local-quality" # optional
output_path: "artifacts/session_recap.md"
timeout: "10m"
# render_debug: true # optional per-artifact override
inputs:
transcript:
source: "trimmed_transcript"
required: true
previous_recap:
source: "previous_session_artifact"
artifact: "session_recap"
path: "" # optional; set when available
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: "session_recap"
```
Prompt IDs and profile IDs are configuration values. They are not hardcoded in analyze-stage logic.
Do not put secrets in `pipeline.yml`. If API-key behavior is configured, use env var names only.
## Scriptorium Runtime Behavior
Narratio integrates with Scriptorium through the public CLI subprocess contract:
- generation: `scriptorium run`
- diagnostics/testing: `scriptorium render --format json` when `render_debug` is enabled
For the initial implementation, only `session_recap` generation is supported.
Analyze-stage session recap behavior:
- available transcript input sources for configured artifacts: `processed_transcript`, `normalized_transcript`, `trimmed_transcript`
- session recap should use gameplay-only transcript input (`source: trimmed_transcript`)
- Narratio resolves `trimmed_transcript` from trim manifest output (`transcript_trimmed`) or fallback `transcripts/trimmed.json`
- Narratio resolves `normalized_transcript` from normalize manifest output (`transcript_normalized`) or fallback `transcripts/normalized.json`
- missing trimmed transcript fails clearly and advises running trim stage first
- `normalized_transcript` is the preferred full-transcript source for future table/meta-analysis artifacts
- `processed_transcript` remains supported for advanced/debug use cases
- optionally includes `previous_recap` when configured and resolvable
- omits optional previous recap when unavailable
- fails if required inputs are missing
- validates output file exists and is non-empty
Expected session output paths:
- `artifacts/session_recap.md`
- `logs/scriptorium.session_recap.stdout.log`
- `logs/scriptorium.session_recap.stderr.log`
- `config/scriptorium.session_recap.generated.yml`
- `artifacts/session_recap.render.json` when render diagnostics are enabled
## Examples
Starter files:
- `examples/pipeline.minimal.yml`
- `examples/session.minimal.yml`
- `examples/speakers.yml`
## Commands
Run tests:
```bash
go test ./...
```
Plan a run:
```bash
go run ./cmd/narratio plan --session examples/session.minimal.yml
```
Use `--config <path>` to override default pipeline lookup when needed.
Run full pipeline:
```bash
go run ./cmd/narratio run --config examples/pipeline.minimal.yml --session examples/session.minimal.yml
```
Run analyze only:
```bash
go run ./cmd/narratio run-stage --config examples/pipeline.minimal.yml --session examples/session.minimal.yml analyze
```
## Operational Note
Checksum-based stale detection is not implemented yet.
If prepared inputs or prompt/runtime config change, rerun the appropriate upstream stages before relying on downstream artifacts.
Examples:
- glossary/autocorrect/speaker-context changes: rerun at least `merge`, `polish`, `normalize`, `trim`, and `analyze`
- trim bounds prompt/profile/config changes: rerun at least `normalize`, `trim`, and `analyze`
- session recap prompt/profile/input-source changes: rerun `analyze`
## Roadmap
Near-term roadmap:
- extend analyze to additional configured artifacts
- support workflows where later artifacts consume earlier generated artifacts
- keep orchestration explicit without a generic DAG engine
- implement archive and notify backends
- [Development guide](docs/development.md) — first-read orientation and
task-specific reading routes.
- [Internal overview](docs/internal/overview.md) — implemented component map.
- [Architecture](docs/policy/architecture.md) — normative boundaries and
invariants.
- [Documentation policy](docs/policy/documentation.md) — canonical ownership
and maintenance rules.
- [Testing policy](docs/policy/testing.md) — test value, boundaries, and
sufficiency.

View File

@@ -1,323 +0,0 @@
# Narratio Architecture
## 1. Purpose
`narratio` is a Go orchestrator for D&D session processing. It runs a stage-based local pipeline from audio input through transcript processing and artifact generation, with manifest-based skip/force/resume behavior.
Narratio integrates with Scriptorium through the **public CLI** (`scriptorium run` and `scriptorium render`) via synchronous subprocess execution.
## 2. Current Status
Implemented:
- strict `pipeline.yml` + `session.yml` loading with strict YAML field checking (`KnownFields(true)`)
- local workspace/session layout, lock file handling, artifact path helpers, checksums, and atomic writes
- manifest store and stage status transitions for resumable runs
- real `prepare`, `transcribe`, `merge`, and `polish` stages
- real WhisperX HTTP adapter
- real Seriatim subprocess adapter
- real Audita subprocess adapter
- real Scriptorium subprocess adapter
- real `normalize` stage producing `transcripts/normalized.json`
- real `trim` stage producing `transcripts/trimmed.json`
- real `analyze` stage for initial `session_recap` generation
- optional Scriptorium render diagnostics (`render_debug`) before production run
Still placeholder/future:
- `archive` stage behavior
- `notify` stage behavior
- additional Scriptorium artifact types beyond `session_recap`
- artifact-to-artifact workflows beyond the initial single-artifact implementation
- generic stale detection based on input/config checksums
## 3. Pipeline and Stage Boundaries
Canonical stage order:
1. `prepare`
2. `transcribe`
3. `merge`
4. `polish`
5. `normalize`
6. `trim`
7. `analyze`
8. `archive`
9. `notify`
Boundary rules:
- orchestration logic lives in `internal/app`
- stage business logic lives in `internal/stage`
- external-tool CLI construction lives in adapter packages
- Scriptorium CLI details stay in `internal/adapters/scriptorium`
## 4. Scriptorium Integration Model
Integration mode:
- public CLI subprocesses only (no Scriptorium internal Go packages, no HTTP API)
- production generation uses `scriptorium run`
- diagnostics/testing render uses `scriptorium render --format json`
Run invocation shape used by adapter:
```bash
scriptorium run --prompt <prompt_id> --input name=path --out <output_path>
```
Optional flags passed when configured:
- `--config <path>`
- `--profile <profile_id>`
- repeated `--var name=value`
- repeated `--input name=path`
- `--timeout <duration>`
- `--api-key-env <ENV_NAME>` when configured
Render invocation shape used by adapter:
```bash
scriptorium render --prompt <prompt_id> --input name=path --format json --out <render_output_path>
```
Adapter behavior:
- always passes `--out`
- captures stdout/stderr separately
- writes generated invocation metadata YAML (redacted, no secrets)
- treats exit code `0` as success
- treats exit code `1` as failure
- treats exit code `2` as failure with `validation_failed=true` and preserves output metadata when available
- validates successful output files exist and are non-empty
- does not treat non-empty stderr as failure by itself
## 5. Configuration Contract
CLI pipeline config path resolution:
- when `--config <path>` is provided, that path is used
- when `--config` is omitted, Narratio searches defaults in order:
- `/usr/local/etc/narratio/pipeline.yml`
- `/etc/narratio/pipeline.yml`
`pipeline.scriptorium` is optional. Existing pipelines without Scriptorium continue to work.
`pipeline.trim` is optional. Existing pipelines without trim config continue to work.
`pipeline.normalize` is optional. Existing pipelines without normalize config continue to work.
When `pipeline.normalize` is omitted, defaults are applied:
- `output_path: transcripts/normalized.json`
- `output_schema: seriatim-intermediate`
- `report: true`
When `pipeline.normalize` is present:
- `output_path` must be non-empty
- `output_schema` must be one of `seriatim-minimal`, `seriatim-intermediate`, or `seriatim-full`
- relative `output_path` values are session-workdir-relative paths
- Seriatim binary settings still come from `pipeline.seriatim`
When `pipeline.trim` is present:
- `enabled` is optional and defaults to `false` when omitted
- relative `output_path`, `bounds.output_path`, and `bounds.render_output_path` values are session-workdir-relative paths
- do not store secrets in trim config values
When `pipeline.trim.enabled: true`:
- `output_path` is required and non-empty
- `bounds.prompt_id` is required and non-empty
- `bounds.transcript_input_name` is required and non-empty
- `bounds.output_path` is required and non-empty
- `bounds.timeout` must parse as a Go duration when provided
- `bounds.render_debug: true` requires non-empty `bounds.render_output_path`
- `bounds.profile_id` may be empty to use the prompt default profile
- prompt IDs are config values, not hardcoded stage logic
When `pipeline.scriptorium` is present:
- `binary` is required and non-empty
- `config_path` is optional; when provided it must be non-empty
- `timeout` is optional; when provided it must parse as a Go duration
- default `timeout` is `10m`
- unknown YAML fields fail strict decode
Artifacts are configured as a map under `pipeline.scriptorium.artifacts` so multiple artifacts are possible in the config shape.
For each artifact definition:
- `enabled: true` requires non-empty `prompt_id`
- `enabled: true` requires non-empty `output_path`
- `timeout` must parse as Go duration when present
- optional per-artifact `render_debug` may override global `scriptorium.render_debug`
- `inputs` are named and each input requires non-empty `source`
- inputs may be optional (`required: false`)
- `vars` values currently support `string` and `bool`
Prompt IDs and profile IDs are configuration values, not hardcoded stage logic.
Trim config shape:
```yaml
trim:
enabled: true
output_path: "transcripts/trimmed.json"
bounds:
prompt_id: "dnd_session.bounds"
profile_id: ""
transcript_input_name: "transcript"
output_path: "artifacts/session_bounds.json"
timeout: "10m"
render_debug: false
render_output_path: "artifacts/session_bounds.render.json"
seriatim:
report: false
```
## 6. Transcript Tiers
Narratio currently produces and uses four transcript tiers:
- `transcripts/merged.json`: canonical deterministic merged transcript from Seriatim merge
- `transcripts/processed.json`: full raw Audita-polished transcript output (includes pre/post-game content)
- `transcripts/normalized.json`: normalized transcript generated by Seriatim normalize
- `transcripts/trimmed.json`: gameplay-only normalized polished transcript from trim stage
Trim reads `transcripts/normalized.json`, validates bounds IDs against that same transcript ID space, and writes `transcripts/trimmed.json`.
## 7. Normalize Stage (Current Implementation)
Normalize stage behavior:
- stage order position: after `polish` and before `trim`
- discovers processed transcript from manifest polish outputs (`transcript_processed`) when present, else `work/<session_id>/transcripts/processed.json`
- validates processed transcript JSON shape (`segments` array required)
- runs Seriatim `normalize` to produce normalized transcript
- validates normalized transcript JSON shape (`segments` array required)
- validates normalize report JSON when enabled
Expected normalize outputs and diagnostics:
- `transcripts/normalized.json`
- `artifacts/seriatim.normalize.report.json` (when normalize report is enabled)
- `logs/seriatim.normalize.stdout.log`
- `logs/seriatim.normalize.stderr.log`
- `config/seriatim.normalize.generated.yml`
## 8. Trim Stage (Current Implementation)
Trim stage behavior:
- stage order position: after `normalize` and before `analyze`
- discovers normalized transcript from manifest normalize outputs (`transcript_normalized`) when present, else `work/<session_id>/transcripts/normalized.json`
- validates normalized transcript JSON shape (`segments` array required)
- when `trim.enabled: false` (or trim config omitted), deterministically copies normalized transcript to `transcripts/trimmed.json` and records `trim_action=copy_disabled`
- when `trim.enabled: true`:
- runs Scriptorium bounds prompt using configured `trim.bounds.prompt_id`
- writes bounds output to configured path (typically `artifacts/session_bounds.json`)
- parses and validates bounds output against the same normalized transcript being trimmed
- converts bounds range to Seriatim keep selector (for example `10-868`)
- runs Seriatim `trim` to produce `transcripts/trimmed.json`
- supports no-trim bounds actions (`none`/`copy`) by copying normalized transcript unchanged
- validates trimmed transcript JSON shape (`segments` array required)
Expected trim outputs and diagnostics:
- `artifacts/session_bounds.json`
- `transcripts/trimmed.json`
- `logs/scriptorium.bounds.stdout.log`
- `logs/scriptorium.bounds.stderr.log`
- `config/scriptorium.bounds.generated.yml`
- `logs/seriatim.trim.stdout.log`
- `logs/seriatim.trim.stderr.log`
- `config/seriatim.trim.generated.yml`
- optional bounds render-debug outputs when enabled:
- `artifacts/session_bounds.render.json`
- `logs/scriptorium.bounds.render.stdout.log`
- `logs/scriptorium.bounds.render.stderr.log`
- `config/scriptorium.bounds.render.generated.yml`
Render-debug files are diagnostics. They are recorded in stage metadata/log/config refs and are not treated as canonical stage output artifact refs.
## 9. Analyze Stage (Current Implementation)
The current real analyze implementation supports only `scriptorium.artifacts.session_recap`.
Behavior:
- if `pipeline.scriptorium` is missing, analyze returns a skipped result with metadata
- if no Scriptorium artifacts are enabled, analyze returns a skipped result with metadata
- if enabled artifacts exist but `session_recap` is not enabled, analyze fails clearly
- available transcript input sources for configured artifacts: `processed_transcript`, `normalized_transcript`, `trimmed_transcript`
- `session_recap` should use `trimmed_transcript` input (`transcripts/trimmed.json`) for in-universe recap generation
- `trimmed_transcript` input is resolved from manifest (`trim` output kind `transcript_trimmed`) when available, otherwise fallback path `work/<session_id>/transcripts/trimmed.json`
- `normalized_transcript` input is resolved from manifest (`normalize` output kind `transcript_normalized`) when available, otherwise fallback path `work/<session_id>/transcripts/normalized.json`
- `processed_transcript` input is resolved from manifest (`polish` output kind `transcript_processed`) when available, otherwise fallback path `work/<session_id>/transcripts/processed.json`
- `normalized_transcript` is the preferred full-transcript source for future table/meta-analysis artifacts
- `processed_transcript` remains available for advanced/debug use cases
- transcript inputs are validated as JSON with top-level `segments` array
- configured inputs are resolved by source
- optional `previous_recap` is omitted when unavailable
- required `previous_recap` fails before invocation when unavailable
- vars are built from config + session metadata
- `render_debug` controls pre-run `scriptorium render` diagnostics
- render failure stops stage before production run
- render output is validated as JSON
- production call uses Scriptorium adapter `RunArtifact`
- successful run output must exist and be non-empty
- missing `trimmed_transcript` input for configured `trimmed_transcript` source fails clearly with guidance to run trim stage first
- manifest records output refs, logs, generated config paths, and non-secret provenance metadata
## 10. Session Recap Paths
Current expected paths for `session_recap`:
- artifact output: `artifacts/session_recap.md`
- run stdout log: `logs/scriptorium.session_recap.stdout.log`
- run stderr log: `logs/scriptorium.session_recap.stderr.log`
- run generated invocation/config: `config/scriptorium.session_recap.generated.yml`
- render output (when enabled): `artifacts/session_recap.render.json`
- render stdout log: `logs/scriptorium.session_recap.render.stdout.log`
- render stderr log: `logs/scriptorium.session_recap.render.stderr.log`
- render generated invocation/config: `config/scriptorium.session_recap.render.generated.yml`
## 11. Security and Privacy
- do not store secrets in pipeline YAML, generated invocation YAML, logs, or manifest metadata
- if API-key integration is configured, pass env var names only (never raw key values)
- avoid logging transcript content or rendered prompt content by default
- treat generated artifacts and logs as potentially sensitive session material
## 12. Operational Caveat (Pre-Stale-Detection)
Checksum-based stale detection is not implemented yet.
If prepared inputs or prompt/runtime configuration change (for example glossary files, prompt IDs, profile IDs, or relevant pipeline settings), rerun the appropriate prior stages to refresh downstream artifacts.
Examples:
- glossary or autocorrect changes usually require rerunning at least `merge`, `polish`, `normalize`, `trim`, and `analyze`
- trim prompt/profile changes require rerunning at least `normalize`, `trim`, and `analyze`
- session recap prompt/profile/input-source changes require rerunning `analyze`
## 13. Roadmap
Planned next steps:
- extend analyze beyond `session_recap` to additional configured artifacts
- support artifact inputs that consume prior generated artifacts
- keep this composable without adding a generic DAG engine in the near term
- implement real `archive` backend behavior
- implement real `notify` backend behavior
- add checksum-based stale detection and stale transitions
Architectural invariants remain:
- strict config decoding/validation
- manifest-driven run control
- clear stage/adapter separation
- configuration-driven prompt/profile/input/vars/output mapping
- Scriptorium integration through public CLI subprocess contract

437
docs/cli.md Normal file
View File

@@ -0,0 +1,437 @@
# CLI Reference
## Shortest Useful Command
```bash
narratio run 2026-04-04
```
This runs the canonical full pipeline for session `2026-04-04`.
## Command Overview
Top-level commands:
- `version`: print the Narratio build version.
- `run <session_id>`: run all or one contiguous range of the canonical stage order.
- `regenerate-artifacts <session_id>`: force-run extraction through analysis.
- `run-stage <stage> <session_id>`: run one stage.
- `analyze <session_id>`: force-run analyze.
- `publish <session_id>`: force-run publish.
- `clean <session_id>` or `clean --all`: remove local work/spool state.
- `session <subcommand>`: session helper commands.
- `config <subcommand>`: validate, display, source-trace, or compare resolved pipeline configuration.
Session subcommands:
- `session init <session_id>`
- `session plan <session_id>`
- `session validate <session_id>`
- `session status <session_id>`
- `session restore <session_id>`
- `session artifacts <session_id>`
- `session locks <session_id>`
- `session locks add <session_id> <source>`
- `session locks remove <session_id> <source>`
## Common Config Flags
Most session-aware commands accept:
- `--config <pipeline.yml>`
- `--campaign <id>`
- `--campaign-file <campaign.yml>`
- `--session <session.yml>`
- `--session-id <session_id>`
- `--previous-session-id <session_id>`
- `--profile <name>`
Rules:
- `--campaign` and `--campaign-file` are mutually exclusive.
- `--session` is not used by `session init`.
- if both positional `<session_id>` and `--session-id` are provided, values must match.
- `--previous-session-id` is a strict expectation: the selected session file
must contain the same `previous_session_id`.
- `--profile` selects a declared pipeline profile. It may be supplied once;
an explicit empty or unknown value fails configuration resolution. When it is
omitted, a declared `default_profile` is used. The same selection applies to
all common-flag commands, including `regenerate-artifacts`.
- `clean --all` cannot be combined with campaign/session selectors.
- notification delivery is currently limited to the configured `noop` mode; see
the [configuration reference](./config.md#notifications).
## Session ID Input Rules
Session-aware commands accept one of these forms:
- positional session ID: `... <session_id>`
- compatibility flag: `... --session-id <session_id>`
When both are present, command parsing requires an exact match.
Commands with additional positionals keep their command-specific order:
- `run-stage <stage> <session_id>` or `run-stage <stage> --session-id <session_id>`
- `session locks add <session_id> <source>` or `session locks add --session-id <session_id> <source>`
- `session locks remove <session_id> <source>` or `session locks remove --session-id <session_id> <source>`
## Command Reference
### `config validate`, `config show`, `config sources`, and `config diff`
```bash
narratio config validate [--config <pipeline.yml>] [--campaign <id> | --campaign-file <campaign.yml>] [--profile <name>]
narratio config show [--config <pipeline.yml>] [--campaign <id> | --campaign-file <campaign.yml>] [--profile <name>]
narratio config sources [--config <pipeline.yml>] [--campaign <id> | --campaign-file <campaign.yml>] [--profile <name>]
narratio config diff <left-profile> <right-profile> [--config <pipeline.yml>] [--campaign <id> | --campaign-file <campaign.yml>]
```
These commands resolve the selected profile, defaults, ordinary paths, and—if
a campaign is selected—the campaign-owned party. They neither discover or load
a session nor create a workspace, manifest, run, lock, adapter, remote
connection, or credential environment.
Campaign selection is optional for a pipeline without party-driven artifact
families. A pipeline with `scriptorium.artifact_families` needs a selected or
configured default campaign so Narratio can expand its concrete artifacts and
publish rules. `--campaign` and `--campaign-file` remain mutually exclusive.
Session, range, force, and artifact-execution flags are not accepted.
`config validate` writes a concise root-path, selected-profile (or `none`), and
effective-digest summary after successful complete validation. `config show`
writes one deterministic, secret-free YAML document containing defaulted and
expanded concrete configuration. It omits composition declarations, artifact
family declarations, and runtime provenance.
`config sources` reports the same fully validated resolution without printing
effective values. Its header identifies the root, ordered imports, selected
profile and overlay, selected campaign, party mode/source, and digest. The
remaining tab-separated records are sorted as `path`, `role`, and `source`.
Roles distinguish root, import, profile, centralized default, campaign, party,
legacy-player, and generated family ownership. A generated party member has
one family record and one party record at the same logical path. The output
never reads or prints secret values.
`config diff` resolves both supplied profile names from one parsed root source
set and compares their fully resolved, secret-free effective mappings. It does
not accept `--profile`; the two positional names must be distinct, declared
profiles. When party-driven families are present, both profiles must resolve to
the same selected campaign and party. Use `--campaign-file` if profile-specific
campaign configuration would otherwise select different files.
Equal profiles print `no differences`. Otherwise, sorted tab-separated records
use one of these forms, with compact JSON values:
```text
added <path> <right-value>
removed <path> <left-value>
changed <path> <left-value> <right-value>
```
Mappings are flattened to their logical field paths; lists remain one atomic
value. The command compares defaulted concrete artifacts and publish rules, not
profile names, source-file layout, or formatting. It succeeds when differences
are found, making it suitable for review and migration checks.
### `version`
```bash
narratio version
```
Official release binaries report their exact Git tag. Binaries built directly
from source without release linker metadata report `dev`.
### `run`
```bash
narratio run <session_id> [--from <stage>] [--through <stage>] [--force] [--artifacts <name[,name...]>] [...common config flags]
```
Behavior:
- evaluates one inclusive contiguous range of the canonical stage order;
- defaults an omitted `--from` to `prepare` and an omitted `--through` to
`notify`, so omitting both retains full-pipeline behavior;
- rejects unknown endpoints and a `--from` endpoint after `--through`;
- runs `render` before `extract`; an omitted or disabled Notarius
configuration records an explicit `notarius_disabled` self-skip;
- skips already-succeeded stages unless `--force` is set or a stage-specific
resume check finds its durable result obsolete;
- applies `--force` only to stages in the selected range;
- rejects repeated `--from`, `--through`, or `--force` options, including
`--name=value` spellings;
- continues interrupted or partially completed sessions by running non-succeeded stages;
- writes session and run manifests.
- reports the resolved profile (or `none`) and effective configuration digest.
When `--artifacts` is present, the selected range must contain `analyze` or
`publish`. Either consumer is sufficient, including a one-stage range.
### `regenerate-artifacts`
```bash
narratio regenerate-artifacts <session_id> [--artifacts <name[,name...]>] [...common config flags]
```
Exactly equivalent to:
```bash
narratio run <session_id> --force --from extract --through analyze [caller options]
```
The command always reruns extraction. Analysis rebuilds the selected configured
artifacts and any prerequisites required by those targets; without
`--artifacts`, it uses the normal default analysis selection. Publish and notify
never run. Common session/configuration options and repeatable artifact values
pass through unchanged.
Because the expansion owns `--force`, `--from`, and `--through`, callers cannot
supply those options. The shared `run` parser reports them as duplicate
singleton flags. The alias has no private execution options or behavior, and
runtime diagnostics may identify the operation as `run`.
### `run-stage`
```bash
narratio run-stage <stage> <session_id> [--force] [--artifacts <name[,name...]>] [...common config flags]
```
Valid stage names:
- `prepare`
- `transcribe`
- `merge`
- `polish`
- `normalize`
- `trim`
- `render`
- `extract`
- `analyze`
- `publish`
- `notify`
Rules:
- `--artifacts` is accepted only for `analyze` and `publish` stage targets.
### `analyze`
```bash
narratio analyze <session_id> [--artifacts <name[,name...]>] [...common config flags]
```
Equivalent to:
```bash
narratio run-stage analyze <session_id> --force [...common config flags]
```
### `publish`
```bash
narratio publish <session_id> [--artifacts <name[,name...]>] [...common config flags]
```
Equivalent to:
```bash
narratio run-stage publish <session_id> --force [...common config flags]
```
### `clean`
```bash
narratio clean <session_id> [--dry-run] [--clear-cache] [...common config flags]
narratio clean --all [--dry-run] [--clear-cache] [--config <pipeline.yml>]
```
Behavior:
- session mode removes the selected session's local work and spool state;
- `--all` removes all local session work and spool state;
- cache remains unless `--clear-cache` is provided.
See [Operations: Cleanup](./operations.md#cleanup) for deletion scope and
post-publish cleanup behavior.
### `session plan`
```bash
narratio session plan <session_id> [--from <stage>] [--through <stage>] [--force] [--artifacts <name[,name...]>] [...common config flags]
```
Uses the same inclusive bounds, endpoint validation, force scope, and artifact
selection contract as `run`. It validates config and prints run/skip decisions
for selected stages only without creating the local workdir or changing the
manifest. Resume-capable selected stages are checked against durable evidence.
The output includes the resolved profile (or `none`) and effective configuration
digest without writing provenance or any manifest state.
For `analyze`, the preview also lists explicit targets, prerequisite-only work,
execution order, and reusable current artifacts with concise reasons. These
artifact decisions come from the same reconciliation and work planner used by
execution; the preview does not predict output identities.
### `session validate`
```bash
narratio session validate <session_id> [...common config flags]
```
Read-only preflight checks for config validity, required inputs, audio mode, previous-session requirements, publish outputs, and effective locks.
### `session status`
```bash
narratio session status <session_id> [...common config flags]
```
Prints local manifest state and, when storage is available, status for the
pointer-selected remote commit and its declared published outputs.
### `session init`
```bash
narratio session init <session_id> --output ./session.yml [options]
narratio session init <session_id> --remote [options]
```
Required target selection:
- exactly one of:
- `--output <path>`
- `--remote`
Options:
- `--config <pipeline.yml>`
- `--campaign <id>` or `--campaign-file <campaign.yml>`
- `--previous-session-id <id>`
- `--date <YYYY-MM-DD>`
- `--title <text>`
- `--audio-dir <path>`
- `--audio-s3-prefix <prefix>`
- `--force`
Rules:
- `--audio-dir` and `--audio-s3-prefix` are mutually exclusive.
- if campaign `session_template_file` is configured, `session init` renders it.
- generated session YAML must be concrete (no unresolved `{{ ... }}` placeholders).
### `session restore`
```bash
narratio session restore <session_id> [--dry-run] [--force] [--include-audio] [...common config flags]
```
Behavior:
- discovers committed remote current state;
- plans local restores;
- writes an execution report;
- blocks unresolved conflicts. `--force` permits replacement only of eligible
regular files.
See [Operations: Restore Workflow](./operations.md#restore-workflow) for the
default restore scope, report location, and conflict-handling workflow.
### `session artifacts`
```bash
narratio session artifacts <session_id> [--remote] [...common config flags]
```
Lists effective built-in, configured Scriptorium, and configured extraction
sources; reports planned, available, unavailable, and published state without
reading payload bodies; and includes publish rules, lock state, and optional
remote published-state availability.
### `session locks`
```bash
narratio session locks <session_id> [...common config flags]
narratio session locks add <session_id> <source> [--reason <text>] [--force] [...common config flags]
narratio session locks remove <session_id> <source> [...common config flags]
```
Behavior:
- list mode reports the effective merge of static and remote locks;
- add/remove mutate only remote locks;
- static locks from pipeline config cannot be removed by CLI commands.
See [Operations: Publish Locks](./operations.md#publish-locks) for lock storage
and precedence.
## `--artifacts` Selection Rules
An artifact-family key selects all of its concrete character members. A
concrete generated key selects only that member; mixed family and concrete
selection is deduplicated and executed as concrete keys. The resulting plan
and command output identify both the concrete key and, where applicable, its
family and character ID.
- accepted on `run`, `session plan`, `run-stage`, `analyze`, and `publish`;
- repeatable and comma-separated values are combined, surrounding whitespace
is removed, and duplicate names are collapsed;
- names must exist in `pipeline.scriptorium.artifacts`;
- empty entries are invalid;
- on `run-stage`, only `analyze` and `publish` accept the option.
Effects:
- selects explicit analyze targets; required configured prerequisites may be
reused or rebuilt before them;
- filters publish rules that source `narratio.artifact.<name>`;
- does not filter built-in transcript/bounds or explicitly configured
`narratio.extraction.<name>` publish sources; and
- does not select or filter Notarius lanes.
## Common Workflows
Run full pipeline:
```bash
narratio run 2026-04-04
```
Dry-run restore plan:
```bash
narratio session restore 2026-04-04 --dry-run
```
Generate a concrete session file from template/default structure:
```bash
narratio session init 2026-04-04 --output ./session.yml --date 2026-04-04 --title "Session 12"
```
Force publish only:
```bash
narratio publish 2026-04-04
```
Regenerate post-transcript artifacts without publishing:
```bash
narratio regenerate-artifacts 2026-04-04 --artifacts session_recap,player_handout
```
## Output And Exit Behavior
- Successful commands write their result or summary to standard output and
exit with status `0`.
- Command failures and invalid invocations write an error to standard error and
exit with status `1`.
- An unknown top-level command also prints the top-level usage summary to
standard error.
- `session restore --help` prints its command-specific usage and exits with
status `0`.
Output is intended for operator inspection. Narratio does not currently offer
a machine-readable CLI output mode; durable machine-readable state is recorded
in manifests and reports described in [Operations](./operations.md).

572
docs/config.md Normal file
View File

@@ -0,0 +1,572 @@
# Configuration Reference
## Purpose
Narratio resolves three YAML documents:
- `pipeline.yml`: pipeline/runtime settings
- `campaign.yml`: campaign identity and stable input defaults
- `session.yml`: session identity, metadata, and audio source selection
## Discovery and Selection
### `pipeline.yml`
When `--config` is omitted, search order is:
1. `/usr/local/etc/narratio/pipeline.yml`
2. `/etc/narratio/pipeline.yml`
### `campaign.yml`
Selection rules:
- if `--campaign-file` is set, use that path;
- else if `--campaign <id>` is set, use `{pipeline.campaigns.root}/{id}/campaign.yml`;
- else use `{pipeline.campaigns.root}/{pipeline.campaigns.default_campaign_id}/campaign.yml`.
### `session.yml`
When `--session` is omitted, local search order is:
1. `/usr/local/etc/narratio/session.yml`
2. `/etc/narratio/session.yml`
If local session discovery fails and a `session_id` is known, Narratio attempts remote session loading from:
- `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/session.yml`
using configured object storage.
The downloaded remote session file is command-scoped: Narratio removes it after
the command finishes and records only the remote object provenance alongside
the durable copied session input.
### Read-only effective pipeline inspection
`narratio config validate`, `narratio config show`, and `narratio config
sources` use the same `--config`, `--campaign`, `--campaign-file`, and
`--profile` selection rules as pipeline commands, but do not select, discover,
or load a session. They do not read credential values or create runtime state.
`narratio config diff <left-profile> <right-profile>` uses the same pipeline and
campaign selectors, resolves each named profile independently from one parsed
root source set, and does not accept a separate `--profile` flag.
Campaign selection is optional only when the resolved pipeline has no
`scriptorium.artifact_families`. When families are declared, Narratio selects a
campaign through an explicit flag or `pipeline.campaigns.default_campaign_id`,
then parses the campaign-owned party and expands concrete artifacts and any
family publish rules before validation. `config validate` prints the resulting
root, profile, and effective digest. `config show` emits the normalized
effective pipeline YAML, with defaults and concrete expansion included but
composition and family declarations omitted. `config sources` prints a stable
source projection instead of effective values: root/import/profile/default
ownership plus campaign/party and generated-family records. Canonical derived
players trace to the party; a legacy configured players file is explicitly
marked as a legacy player source. The [CLI reference](cli.md#config-validate-config-show-config-sources-and-config-diff)
owns command syntax and output conventions.
`config diff` compares normalized field values rather than YAML text or source
ownership. It emits sorted `added`, `removed`, and `changed` records, uses
compact deterministic JSON values, treats lists atomically, and reports `no
differences` when the complete effective configurations are equal. Concrete
family members and generated publish rules participate after expansion; moving
an equal value between eligible root/import sources does not create a
difference.
### Migrating to the maintained bundle
Use the [production/testing bundle](../examples/production-testing/pipeline.yml)
as the complete copyable migration reference. Split stable pipeline settings
into explicit additive imports, place production/testing differences in one
selected overlay, and retain a production `default_profile`. Convert campaign
rosters to [canonical party input](integrations/party.md), remove a separate
`players_file`, then express character work as families. Inspect the result
with `config validate`, `config show`, and `config sources`; use `config diff`
to review profiles before running a session. Unversioned parties and their
`players_file` remain a clearly bounded legacy compatibility path.
### Identity segments
Campaign IDs (`campaign_id` and `default_campaign_id`), session IDs, previous
session IDs, and Narratio run IDs are opaque portable segments. They must use
only ASCII letters, digits, `.`, `_`, and `-`; empty values, `.`/`..`, path
separators, drive forms, whitespace, control characters, and non-ASCII text are
rejected. Narratio does not trim or rewrite these values. Existing manifests or
remote state with an unsafe legacy identity must be migrated before use.
## Validation and Merge Rules
- YAML decode is strict (`KnownFields(true)`) and accepts exactly one document:
unknown fields or trailing documents fail load.
- A pipeline file may explicitly import additive YAML fragments through the
root-only `composition.imports` list. Imported files contribute fields to one
logical pipeline document; they do not override fields supplied by the root
or another import.
- A root pipeline may declare named profiles. Exactly one profile is selected
by an option-aware caller or by `composition.default_profile`; a caller's
explicit selection takes precedence. Declaring profiles without either form
of selection is an error.
- Configured timeout and retry-delay durations must be positive. An omitted
artifact timeout continues to inherit its configured Scriptorium timeout.
- Session files must be concrete; unresolved `{{ ... }}` placeholders fail load.
- Pipeline defaults are applied before validation.
- Campaign and session identities must agree.
- Required stable files (`speakers_file`, `autocorrect_file`, `glossary_file`,
`party_file`) and the optional `spell_catalog_file` resolve from session
overrides when provided, otherwise from campaign defaults. An empty or
omitted session spell-catalog value inherits the campaign value.
- `party_file` is classified when pipeline and campaign configuration are
combined. A versioned [canonical party](integrations/party.md) is
campaign-owned, derives the players input internally, and forbids both a
separate `players_file` and a session `party_file` override. An unversioned
party remains a bounded legacy input and requires `players_file`; its normal
campaign/session overrides continue to apply.
- Exactly one audio mode must be configured in session input:
- local (`audio_dir` or `audio_files`), or
- S3 (`audio_s3.prefix`).
### Pipeline composition
Large pipeline configurations may be split into explicitly named fragments and
may declare one overlay per selectable profile:
```yaml
composition:
imports:
- config/storage.yml
- config/integrations.yaml
default_profile: production
profiles:
production:
overlay: profiles/production.yml
testing:
overlay: profiles/testing.yml
campaigns:
root: /usr/local/share/narratio/campaigns
```
The maintained [production/testing bundle](../examples/production-testing/pipeline.yml)
is a complete copyable example of this structure, including canonical-party
artifact families.
Imports are resolved relative to the directory containing the root pipeline
file and are loaded in declaration order. Narratio does not scan directories or
infer fragments. Each import must be a confined regular `.yml` or `.yaml` file:
absolute paths, traversal, symlinks, directories, duplicate files, and an
import of the root pipeline itself are rejected. Only the root pipeline may
contain `composition`; nested composition is rejected.
Composition is additive. A map may be extended by multiple files when every
leaf is distinct, but a scalar, list, or map/list/scalar kind cannot be claimed
more than once, even when the repeated values are identical. Conflict errors
name the full field path and every source that claimed it. The assembled YAML is
then decoded against the normal strict pipeline schema and defaults are applied
once.
Profile names are case-sensitive, non-empty, trimmed, and cannot contain
control characters. If `profiles` is present, it must contain at least one
entry and every entry must contain only an `overlay` path. An explicit profile
selection overrides `default_profile`; unknown and explicitly empty selections
fail. Narratio never selects the first profile implicitly and does not read a
profile selection from the environment.
Every declared overlay is resolved relative to the root pipeline directory and
must satisfy the same confined regular-YAML-file rules as an import. Narratio
parses every declared overlay even when it is not selected, then applies only
the selected one. Maps merge recursively, overlay scalars replace base scalars,
and overlay lists replace base lists completely. Explicit `false`, zero, empty
lists, and empty maps remain meaningful. YAML null cannot delete a value, and
kind changes are rejected. Profiles cannot inherit from or stack with other
profiles, and overlays cannot import files or declare profiles.
After composition, Narratio strictly decodes the result, applies centralized
defaults once, resolves ordinary paths, and computes a deterministic effective
configuration digest. The digest represents the normalized, secret-free
runtime pipeline mapping; it excludes composition declarations, source
provenance, profile identity, and raw environment secret values. Equivalent
effective mappings therefore have the same digest regardless of how fields are
split among the root and imports.
An imported field has the same meaning it would have in a monolithic root
pipeline. In particular, ordinary relative pipeline paths continue to resolve
from the root pipeline directory, not from the importing fragment's directory.
## Minimal Working Configuration
`pipeline.yml`
```yaml
campaigns:
root: /usr/local/share/narratio/campaigns
default_campaign_id: sample-campaign
whisperx:
transcribe_url: https://transcription.example.com/transcribe
```
`campaign.yml`
```yaml
campaign_id: sample-campaign
inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
players_file: ./players.yml
party_file: ./party.yml
```
`session.yml` (local audio)
```yaml
session_id: 2026-05-03
inputs:
audio_dir: ./audio
```
## Secrets Handling
- Do not place raw secrets in YAML.
- Use env var names in config (for example `pipeline.audita.llm_api_key_env`).
- Optionally load credential files from `pipeline.secrets.env_dir`. Each valid
environment-variable filename supplies one value; trailing CR/LF is removed.
- An existing process environment value takes precedence over a credential file.
- Credential directories and files must not be symlinks and must be regular,
bounded files (at most 8 KiB per value). On POSIX, provision the directory
with no group/other access (normally `0700`) and files with no group/other
access (normally `0600`).
- Commands that need storage/auth load filesystem secrets before constructing adapters.
## Publish Configuration Summary
Publish rules live under `pipeline.publish`.
```yaml
publish:
enabled: true
upload_run: true
outputs:
- source: narratio.transcript.final_trimmed
dest: transcripts/final.trimmed.json
required: true
- source: narratio.transcript.final_markdown
dest: transcripts/final.md
required: true
- source: narratio.transcript.final_trimmed_markdown
dest: transcripts/final.trimmed.md
required: true
- source: narratio.artifact.session_recap
dest: artifacts/session_recap.md
required: true
locks:
- source: narratio.artifact.session_recap
reason: manual post-publish edits
```
Rules:
- `outputs[].source` is required.
- `outputs[].dest` may be omitted when derivable from source.
- extraction sources require an explicit `outputs[].dest` and publish only when
a rule names that source; the Notarius index and complete bundle are not
publish sources.
- `outputs[].required` defaults to `true`.
- static locks (`pipeline.publish.locks`) merge with remote locks (`{session_prefix}/locks.yml`), with static locks taking precedence on duplicates.
## Full Schema
### Pipeline
| Field | Type | Required | Default / Rule |
| --- | --- | --- | --- |
| `composition.imports[]` | list of strings | No | explicit additive pipeline fragments relative to the root pipeline directory; `.yml` or `.yaml` regular files only |
| `composition.default_profile` | string | Conditional | selected when profiles exist and no caller explicitly selects one; must name a declared profile |
| `composition.profiles.<name>.overlay` | string | Conditional | required for every declared profile; one confined `.yml` or `.yaml` overlay relative to the root pipeline directory |
| `pipeline.workspace.root` | string | No | `/var/lib/narratio` |
| `pipeline.workspace.cleanup_after_publish` | bool | No | `false` |
| `pipeline.campaigns.root` | string | No | `/usr/local/share/narratio/campaigns` |
| `pipeline.campaigns.default_campaign_id` | string | No | empty |
| `pipeline.secrets.env_dir` | string | No | empty |
| `pipeline.storage.backend` | string | No | `local`; supported values are `local` and `s3` (case-insensitive) |
| `pipeline.storage.s3.bucket` | string | Conditional | required when backend is `s3` and S3 session-audio or publish upload is enabled |
| `pipeline.storage.s3.root_prefix` | string | No | `dnd` |
| `pipeline.storage.s3.region` | string | No | empty |
| `pipeline.storage.s3.endpoint` | string | No | empty |
| `pipeline.storage.s3.force_path_style` | bool | No | `false` |
| `pipeline.storage.s3.access_key_id_env` | string | No | `OBJECT_STORAGE_KEY_ID` |
| `pipeline.storage.s3.secret_access_key_env` | string | No | `OBJECT_STORAGE_KEY` |
| `pipeline.spool.root` | string | No | `/var/spool/narratio` |
| `pipeline.spool.delete_audio_after_publish` | bool | No | `false` |
| `pipeline.cache.root` | string | No | `/var/cache/narratio` |
| `pipeline.cache.s3_audio` | bool | No | `true` |
| `pipeline.publish.enabled` | bool | No | `true` |
| `pipeline.publish.upload_run` | bool | No | `true` |
| `pipeline.publish.outputs[]` | list | No | defaults to final trimmed JSON plus final and final-trimmed Markdown outputs |
| `pipeline.publish.outputs[].source` | string | Yes (per rule) | must reference built-in or configured artifact source |
| `pipeline.publish.outputs[].dest` | string | Conditional | derived if omitted and source supports derivation |
| `pipeline.publish.outputs[].required` | bool | No | `true` |
| `pipeline.publish.locks[]` | list | No | empty |
| `pipeline.publish.locks[].source` | string | Yes (per lock) | must reference supported publish source |
| `pipeline.publish.locks[].reason` | string | No | empty |
| `pipeline.whisperx.transcribe_url` | string | Yes | absolute `http` or `https` URL |
| `pipeline.whisperx.language` | string | No | `en` |
| `pipeline.whisperx.timeout` | duration | No | `30m` |
| `pipeline.whisperx.retries` | int | No | `3` |
| `pipeline.whisperx.retry_delay` | duration | No | `2s` |
| `pipeline.whisperx.concurrency` | int | No | `2` |
| `pipeline.seriatim.binary` | string | No | `seriatim` |
| `pipeline.seriatim.timeout` | duration | No | `10m` |
| `pipeline.seriatim.output_schema` | string | No | `seriatim-intermediate` |
| `pipeline.seriatim.coalesce_gap` | float | No | `3.0` |
| `pipeline.seriatim.report` | bool | No | `true` |
| `pipeline.seriatim.env.overlap_word_run_gap` | float | No | unset |
| `pipeline.seriatim.env.overlap_word_run_reorder_window` | float | No | unset |
| `pipeline.seriatim.env.backchannel_max_duration` | float | No | unset |
| `pipeline.seriatim.env.filler_max_duration` | float | No | unset |
| `pipeline.audita.binary` | string | No | `audita` |
| `pipeline.audita.timeout` | duration | No | `3h` |
| `pipeline.audita.llm_api_key_env` | string | No | empty |
| `pipeline.audita.modules[]` | list[string] | No | empty |
| `pipeline.audita.base_url` | string | No | empty |
| `pipeline.audita.model` | string | No | empty |
| `pipeline.audita.total_llm_concurrency` | int | No | unset |
| `pipeline.audita.proposal_llm_concurrency` | int | No | unset |
| `pipeline.audita.validation_model` | string | No | empty |
| `pipeline.audita.validation_llm_concurrency` | int | No | unset |
| `pipeline.audita.transcript_description` | string | No | empty |
| `pipeline.audita.config_path` | string | No | empty |
| `pipeline.audita.output_schema` | string | No | empty |
| `pipeline.audita.work_dir_retention` | string | No | empty |
| `pipeline.audita.report` | bool | No | `true` |
| `pipeline.normalize.output_path` | string | No | `transcripts/final.json` |
| `pipeline.normalize.output_schema` | string | No | `seriatim-intermediate` |
| `pipeline.normalize.report` | bool | No | `true` |
| `pipeline.trim.enabled` | bool | No | `true` |
| `pipeline.trim.output_path` | string | No | `transcripts/final.trimmed.json` |
| `pipeline.trim.bounds.prompt_id` | string | No | `dnd.session_bounds` |
| `pipeline.trim.bounds.profile_id` | string | No | empty |
| `pipeline.trim.bounds.transcript_input_name` | string | No | `transcript` |
| `pipeline.trim.bounds.output_path` | string | No | `artifacts/session_bounds.json` |
| `pipeline.trim.bounds.timeout` | duration | No | `10m` |
| `pipeline.trim.bounds.render_debug` | bool | No | `false` |
| `pipeline.trim.bounds.render_output_path` | string | Conditional | required when `render_debug` is true |
| `pipeline.trim.seriatim.report` | bool | No | `false` |
| `pipeline.notarius.enabled` | bool | No | `false` |
| `pipeline.notarius.binary` | string | No | `notarius` |
| `pipeline.notarius.config_path` | string | Conditional | required when enabled; relative paths resolve from the pipeline file directory |
| `pipeline.notarius.pipeline_id` | string | Conditional | required when enabled |
| `pipeline.notarius.timeout` | duration | No | `3h`; must be positive |
| `pipeline.notarius.working_directory` | string | No | directory containing resolved `config_path`; relative paths resolve from the pipeline file directory |
| `pipeline.notarius.references` | map[string]string | No | empty; maps normalized Notarius selectors to supported prepared Narratio source IDs; maximum 256 entries |
| `pipeline.notarius.outputs` | map | Conditional | at least one entry when enabled |
| `pipeline.render.enabled` | bool | No | `true` |
| `pipeline.render.format` | string | No | `markdown` (only supported value) |
| `pipeline.render.title` | string | No | empty (falls back to `session.title` when set) |
| `pipeline.render.include_timestamps` | bool | No | `true` |
| `pipeline.render.include_segment_ids` | bool | No | `true` |
| `pipeline.render.include_metadata` | bool | No | `false` |
| `pipeline.scriptorium.binary` | string | No | `scriptorium` |
| `pipeline.scriptorium.config_path` | string | No | empty |
| `pipeline.scriptorium.timeout` | duration | No | `10m` |
| `pipeline.scriptorium.render_debug` | bool | No | `false` |
| `pipeline.scriptorium.artifacts` | map | No | empty |
| `pipeline.scriptorium.artifact_families` | map | No | empty; expands one ordinary artifact per canonical party character |
| `pipeline.notification.mode` | string | No | `noop`; the only supported notification mode until a provider is implemented |
### Notarius Reference Bindings
`pipeline.notarius.references` maps a Notarius CLI selector to a prepared
Narratio source, not to a filesystem path:
```yaml
notarius:
references:
glossary: narratio.input.glossary
party: narratio.input.party
players: narratio.input.players
spell_catalog: narratio.input.spell_catalog
```
Supported sources are `narratio.input.party`, `narratio.input.players`,
`narratio.input.glossary`, and `narratio.input.spell_catalog`. Each map entry is
required by its presence: omit a binding when the selected Notarius pipeline
does not need it. A spell-catalog binding additionally requires an effective
campaign or session `spell_catalog_file`.
Selectors accept Notarius's `slot`, `chunk.slot`, `lane.slot`,
`lane.extract.slot`, `lane.merge.slot`, and `lane.normalize.slot` forms.
Narratio trims whitespace around
selectors and their dot-separated components, rejects empty components and
`=`, rejects duplicate normalized selectors, and limits the map to 256 entries.
It validates only selector structure and the prepared source vocabulary;
Notarius owns target-slot declarations and media compatibility.
Before extraction, Narratio resolves every binding from the current prepared
session manifest and streams it into a verified invocation-local snapshot whose
absolute path is passed to Notarius. Missing, unsafe, empty,
changed-during-copy, or checksum-inconsistent prepared evidence fails with
guidance to force `prepare`. Bindings are sorted by normalized selector and are
part of extraction fingerprint and resume identity. See the
[Notarius integration contract](./integrations/notarius.md) for the subprocess
boundary and the [complete example](../examples/pipeline.full.annotated.yml)
for a copyable configuration.
### Notarius Output Entries
For each `pipeline.notarius.outputs.<name>`:
| Field | Type | Required | Rule |
| --- | --- | --- | --- |
| `lane_id` | string | Yes | unique Notarius lane ID |
| `media_type` | string | Yes | exact accepted descriptor media type |
| `schema_id` | string | Yes | exact accepted descriptor schema ID |
| `schema_version` | string | Yes | exact accepted descriptor schema version |
| `module_key` | string | No | exact accepted module key when set |
Output names must match `^[a-z][a-z0-9_]*$` and become selectable sources named
`narratio.extraction.<name>`. Lane IDs must be unique. Every declared output is
required from a successful Notarius result; a missing, rejected, duplicate, or
contract-incompatible lane fails extraction. See the
[complete maintained example](../examples/pipeline.full.annotated.yml) for the
current ten-lane D&D mapping and the [Notarius contract](./integrations/notarius.md)
for compatibility ownership.
### Scriptorium Artifact Entries
For each `pipeline.scriptorium.artifacts.<name>`:
| Field | Type | Required | Rule |
| --- | --- | --- | --- |
| `enabled` | bool | No | `false` if omitted |
| `depends_on[]` | list[string] | No | must reference configured artifact keys; no self-reference; configured graph must be acyclic |
| `render_debug` | bool | No | per-artifact override |
| `prompt_id` | string | Conditional | required when artifact is enabled |
| `profile_id` | string | No | empty |
| `output_path` | string | Conditional | required when enabled; also required when referenced by publish/output/input rules |
| `timeout` | duration | No | artifact override |
| `inputs` | map | No | input key names must be non-empty |
| `vars` | map | No | values must be string or bool; `session_id` is reserved and overwritten by Narratio |
Narratio adds `session_id=narratio-session-<session_id>` to every Scriptorium request for sticky upstream LLM routing. If an artifact config sets `vars.session_id`, Narratio replaces that value before invoking Scriptorium. Use a different variable name if a prompt needs the raw Narratio session ID as content.
Without `--artifacts`, analyze executes enabled configured artifacts. With an
explicit `--artifacts` list, the exact named configured artifacts are the
one-invocation targets even if their `enabled` values are false. Analyze closes
those targets over `depends_on`: a current prerequisite is reused, while a
stale, missing, failed, or legacy prerequisite is rebuilt before its dependent.
Unrelated artifacts are not executed. Named targets and any prerequisite that
may require rebuilding must therefore have valid executable fields. This
override affects analyze planning only; publish uses the list only to filter
configured `narratio.artifact.<name>` output rules.
For each artifact input `pipeline.scriptorium.artifacts.<name>.inputs.<input_name>`:
| Field | Type | Required | Rule |
| --- | --- | --- | --- |
| `source` | string | Yes | built-in runtime source, prepared input source, `narratio.extraction.<name>`, `narratio.artifact.<name>`, or `narratio.previous_session.artifact.<name>` |
| `required` | bool | No | optional input requirement |
`artifact` and `path` are obsolete and rejected by strict configuration
loading. Use the canonical `source` identifier to select the input; Narratio
does not provide adapter-specific input passthrough fields.
### Scriptorium Artifact Families
`pipeline.scriptorium.artifact_families` declares a shared artifact template
for every canonical campaign character. Configuration resolution expands each
family into ordinary `pipeline.scriptorium.artifacts` entries before analyze
planning or Scriptorium invocation. A legacy party cannot be used for a family.
For each `pipeline.scriptorium.artifact_families.<name>`:
| Field | Type | Required | Rule |
| --- | --- | --- | --- |
| `enabled`, `prompt_id`, `profile_id`, `timeout`, `render_debug`, `depends_on`, `inputs`, `vars` | ordinary artifact fields | No | copied to each generated artifact under the corresponding ordinary rules |
| `for_each` | string | Yes | exactly `party.characters` |
| `output_path_pattern` | string | Yes | safe path beneath `artifacts/` with exactly one `{character_id}` token and no other brace syntax |
| `member_vars` | map | No | maps an ordinary Scriptorium variable name to a supported canonical character selector |
| `member_dependencies` | list | No | unique family keys; each generated member depends on the corresponding generated member of each listed family |
| `publish` | map | No | typed family publish policy (`enabled`, `required`, `dest_pattern`) expanded into concrete publish outputs when enabled |
Generated keys are `<family>_<character_id>` and generated output paths must
not collide with explicit artifacts or another generated artifact. Families
expand even when disabled; normal analyze selection still omits disabled
artifacts unless they are explicitly selected by their concrete key.
Supported `member_vars` selectors are `character_id`, `player.name`,
`character.name`, `character.class_summary`, and `character.alias_summary`.
Their resolved values are strings. A member variable may not reuse a static
`vars` name; `session_id` remains owned and overwritten by Narratio as for any
other Scriptorium artifact.
Within a family only, an input source may use
`narratio.member_artifact.<family>`. The referenced family must be named in
that family's `member_dependencies`; resolution rewrites the source to the
corresponding ordinary `narratio.artifact.<family>_<character_id>` source.
This syntax is rejected in explicit artifacts and never reaches runtime stages
or Scriptorium.
### Notifications
Narratio currently supports only `notification.mode: noop`, which is also the
default when the section is omitted. The notify stage performs no delivery in
this mode. Backend, recipient, timeout, and other provider settings are
rejected by strict configuration loading until Narratio has a provider
integration.
### Campaign
| Field | Type | Required | Notes |
| --- | --- | --- | --- |
| `campaign_id` | string | Yes | canonical opaque campaign identity |
| `session_template_file` | string | No | used by `session init` when set |
| `inputs.speakers_file` | string | Yes | stable input default |
| `inputs.autocorrect_file` | string | Yes | stable input default |
| `inputs.glossary_file` | string | Yes | stable input default |
| `inputs.players_file` | string | Conditional | required only with an unversioned legacy `party_file`; forbidden for a canonical party |
| `inputs.party_file` | string | Yes | stable campaign party source; relative paths resolve from `campaign.yml` |
| `inputs.spell_catalog_file` | string | No | optional spell-catalog overlay default; required when a Notarius reference selects `narratio.input.spell_catalog` |
### Session
| Field | Type | Required in session file | Notes |
| --- | --- | --- | --- |
| `session_id` | string | Yes | opaque identity; must match CLI session target when provided |
| `previous_session_id` | string | No | opaque identity; must not equal `session_id` |
| `campaign` | string | No | opaque identity; filled from `campaign_id` during resolve if omitted |
| `date` | string | No | metadata |
| `title` | string | No | metadata |
| `inputs.speakers_file` | string | No | overrides campaign stable input |
| `inputs.autocorrect_file` | string | No | overrides campaign stable input |
| `inputs.glossary_file` | string | No | overrides campaign stable input |
| `inputs.players_file` | string | No | legacy-party override; forbidden for a canonical party |
| `inputs.party_file` | string | No | legacy-party override; forbidden for a canonical campaign party |
| `inputs.spell_catalog_file` | string | No | overrides the optional campaign spell catalog; empty or omitted inherits the campaign value |
| `inputs.audio_dir` | string | Conditional | local audio mode |
| `inputs.audio_files[]` | list[string] | Conditional | local audio mode |
| `inputs.audio_s3.prefix` | string | Conditional | S3 audio mode |
Audio rules:
- configure local mode (`audio_dir` or `audio_files`) or S3 mode (`audio_s3.prefix`), not both.
- `audio_s3` requires `pipeline.storage.backend: s3` and a configured S3 bucket.
### Storage backend selection
`local` is the default and disables remote object-store operations. Configure
`s3` explicitly before supplying `storage.s3`; a populated S3 block does not
select a backend on its own. Unknown backend names and an S3 block paired with
`local` are rejected during configuration validation.
### Previous-session expectation
`previous_session_id` is optional in a session file. When a command supplies
`--previous-session-id`, however, the session file must contain the same value;
an omitted or different value is rejected before the command performs work.
## Maintained Examples
See the [maintained examples index](../examples/README.md) for complete pipeline,
campaign, session, template, and input fixtures. Keep complete copyable files
there rather than duplicating them in this reference.

59
docs/development.md Normal file
View File

@@ -0,0 +1,59 @@
# Development
This is the first-read landing page for people and LLM coding agents working on
Narratio. It provides a concise repository orientation and routes each kind of
change to its canonical documentation.
Narratio is a stage-driven Go orchestrator for turning D&D session audio into
polished transcripts and generated artifacts. Start with the
[README](../README.md) for product context,
[Architecture](policy/architecture.md) for normative system boundaries, and the
[Internal Overview](internal/overview.md) for implemented component ownership.
## What To Read
| When working on | Read | Why |
| --- | --- | --- |
| Finding the package or component that owns current behavior | [Internal Overview](internal/overview.md) | It is the implemented component inventory and routes to focused internal documents. |
| Application shape, boundaries, dependency direction, runtime invariants, safety properties, or dependencies | [Architecture](policy/architecture.md) | It defines the intended system shape, ownership, and non-goals. |
| Any documentation addition or revision | [Documentation Policy](policy/documentation.md) | It defines canonical owners, audiences, current-behavior rules, and maintenance requirements. |
| Adding, changing, reviewing, rewriting, or deleting tests | [Testing Policy](policy/testing.md) | It defines risk-based sufficiency, durable test boundaries, test-double guidance, and test lifecycle decisions. |
| CLI composition or command behavior | [Internal Overview](internal/overview.md) and [CLI Reference](cli.md) | The overview routes to command ownership; the reference owns public syntax and invocation behavior. |
| Configuration loading, resolution, or user-visible configuration | [Internal Overview](internal/overview.md) and [Configuration](config.md) | The overview routes to implementation ownership; the reference owns fields, defaults, discovery, and validation. |
| Session workflow, status, restore, cleanup, or object storage | [Restore Internals](internal/command-restore.md), [Workspace Internals](internal/workspace.md), [Storage Internals](internal/storage.md), [Operations](operations.md), and [Troubleshooting](troubleshooting.md) | These separate implementation mechanics, operator procedures, and symptom-driven recovery. |
| Pipeline sequencing or the behavior of a stage | [Internal Overview](internal/overview.md) and its focused stage documents | The overview owns the implemented stage inventory and routes to each stage contract. |
| Adapters or external tool contracts | [Adapter Internals](internal/adapters.md) and [Integration Contracts](integrations/README.md) | The internal guide owns adapter composition and mechanics; integration documents own external formats and protocols. |
| Manifests, artifacts, workspace paths, or publish behavior | [Manifest Internals](internal/manifest.md), [Artifact Internals](internal/artifacts.md), [Workspace Internals](internal/workspace.md), [Publish Internals](internal/stage-publish.md), and [Operations](operations.md) | These separate implementation state and resolution from operator-visible layout and lifecycle. |
| Maintained configuration or input examples | [Configuration](config.md) and [Examples](../examples/README.md) | The reference owns field meanings; the examples directory owns complete copyable files. |
| Preparing, validating, or publishing a release | [Release Procedure](release.md) | The maintainer procedure owns version selection, candidate validation, guarded tag publication, and optional later CI inspection. |
| Proposed or unimplemented behavior | `docs/roadmap/` | Future work belongs only in roadmap documentation until implemented. |
For an existing subsystem, also inspect its focused tests and package-level
contracts before changing behavior.
## Validation
Use focused package tests while iterating. Every pull request and push runs the
following repository-wide checks before it can be accepted:
```sh
go test ./...
go test -race ./...
go vet ./...
go build ./...
go test ./internal/doccheck
go test ./internal/config -run '^TestExamplesLoadAndValidate$'
```
The documentation check verifies local Markdown links and the dependency graph
of the Woodpecker workflows. The configuration check loads every maintained
pipeline and session example. Tag CI reuses this validation path before its
asynchronous asset publication; the maintainer release boundary is documented
in the [Release Procedure](release.md).
Woodpecker also runs `go test -race -shuffle=on -count=3 ./...` on its scheduled
job to expose ordering and repeatability defects. Current runners cross-compile
for macOS and Windows, but do not provide native macOS or Windows execution.
Those cross-builds establish compilation only, not platform-equivalent runtime
evidence. Add native checks only when official runner labels and successful
native-run evidence are available.

View File

@@ -0,0 +1,36 @@
# Integrations Index
## Audience
Operators, developers, and coding agents who need to understand Narratio's
externally observable integration boundaries.
## Scope
`docs/integrations/` is the canonical reference for protocols, invocation and
data contracts, logical outputs, and compatibility behavior at external tool
boundaries.
These documents describe what Narratio sends or invokes, what it accepts in
return, and how failures are surfaced. Internal composition and stage mechanics
belong in [the adapter implementation guide](../internal/adapters.md) and the
focused stage documents.
## Integration Contracts
- [Audita](./audita.md): transcript polishing (`audita process`).
- [Notarius](./notarius.md): complete pipeline execution and safe JSON bundle
discovery (`notarius run`).
- [Party](./party.md): canonical campaign roster input.
- [Seriatim](./seriatim.md): merge, normalize, trim, and render operations.
- [Scriptorium](./scriptorium.md): artifact generation and debug rendering
(`scriptorium run|render`).
- [WhisperX](./whisperx.md): speaker-audio transcription over HTTP.
## Related Canonical Docs
- [Configuration](../config.md): operator-facing configuration reference.
- [Adapter implementation](../internal/adapters.md): shared adapter boundary and
runner wiring.
- [Internal documentation](../internal/overview.md): stage-specific integration
usage and component ownership.

View File

@@ -1,147 +1,67 @@
# Audita
# Integration: Audita
Audita is a framework-first transcript correction application. The public `audita` package provides:
## Purpose
Define the Audita adapter contract used by the `polish` stage.
- deterministic transcript normalization
- token-batched module orchestration
- concrete `glossary`, `homophones`, `spoken_word`, and `grammar` modules built on reusable proposal / validator contracts
- structured run reporting and work-dir diagnostics
## External Boundary
The previous working implementation has been preserved as `audita_prototype` inside this repository. Its full regression suite lives under `tests/audita_prototype`.
Narratio invokes `audita process` as a subprocess for each polish operation.
The configured timeout and parent cancellation bound the invocation. Internal
runner composition is documented in
[the adapter implementation guide](../internal/adapters.md).
## Development
## Request Contract
`PolishRequest` carries:
- required transcript/glossary/output/work-dir paths;
- optional report path (required when report mode is enabled);
- generated config and stdout/stderr log paths;
- optional per-invocation module override.
This project is set up for `uv`.
The constructed runner owns static Audita settings: binary, timeout,
credentials, default modules, model and endpoint settings, validation and output
settings, report mode, and concurrency. The `polish` stage supplies only
invocation-specific paths and may override modules for that invocation.
```sh
uv sync --extra dev
uv run pytest
```
## Result Contract
`PolishResult` returns:
- processed transcript path;
- optional report path;
- work dir and generated-config/log paths;
- exit code, duration, binary provenance;
- adapter metadata map.
## Usage
## Validation and Failure Semantics
Construction fails for invalid static config values, including:
- empty binary;
- non-positive timeout;
- invalid base URL;
- invalid output schema;
- invalid work-dir retention value;
- invalid concurrency values.
Process a transcript with the current framework implementation:
Run fails for:
- missing required request paths;
- missing required credential env var when configured (`llm_api_key_env`);
- subprocess execution failure;
- invalid processed transcript JSON (`segments` array required);
- invalid report JSON when reporting is enabled.
```sh
uv run audita process transcript.json --glossary glossary.yaml --output corrected.json
```
Processed transcript JSON is limited to 64 MiB and optional report JSON to 16
MiB. Both must be regular files without symlinked path components.
The framework currently runs this default module sequence:
Failure results still include output/log/config/exit metadata for diagnostics.
1. `glossary`
2. `homophones`
3. `glossary`
4. `spoken_word`
5. `grammar`
## Deterministic Behavior
- CLI args are built from runner config + request in a fixed order.
- Generated invocation YAML (`audita.generated.v1`) is emitted when requested.
- Manifest writes are stage-owned; adapter itself is stateless.
Resolved run instance names are auto-numbered for repeats, so the default report pipeline is:
## Configuration
1. `glossary_1`
2. `homophones`
3. `glossary_2`
4. `spoken_word`
5. `grammar`
Operator-selected values are defined under `pipeline.audita.*` in the
[configuration reference](../config.md#pipeline).
The default module sequence is fully implemented today:
Maintained example with Audita config:
- `glossary` proposes glossary-supported acoustic corrections
- `homophones` proposes conservative homophone and mistranscription corrections
- `spoken_word` proposes conservative dysfluency cleanup
- `grammar` proposes punctuation, capitalization, and spacing cleanup only
To run a custom module sequence, pass `--modules`:
```sh
uv run audita process transcript.json --glossary glossary.yaml --modules grammar --output corrected.json
```
To also write a structured JSON report:
```sh
uv run audita process transcript.json --glossary glossary.yaml --output corrected.json --report-json report.json
```
From a checked-out repository, you can also use the root launcher:
```sh
./audita process transcript.json --glossary glossary.yaml --output corrected.json
```
For a system-wide command, install the source tree under `/usr/local/src/audita`, sync dependencies there, and symlink the root launcher into your `PATH`:
```sh
cd /usr/local/src/audita
uv sync --extra dev
ln -s /usr/local/src/audita/audita /usr/local/bin/audita
audita process transcript.json --glossary glossary.yaml --output corrected.json
```
Without `--output`, Audita writes the corrected transcript JSON to stdout and progress logs to stderr.
`--report-json` writes a separate machine-readable run report and never mixes report data into stdout.
Useful configuration can be supplied by CLI flag or environment variable. CLI flags take precedence over environment variables. Normal runs now require LLM API credentials, because the `glossary`, `homophones`, `spoken_word`, and `grammar` modules make real LLM calls. `AUDITA_LLM_API_KEY` and `--llm-api-key` are the preferred provider-neutral credential surfaces, while `OPENROUTER_API_KEY` remains supported as a backward-compatible fallback.
| Environment variable | CLI flag | Default | Purpose |
| --- | --- | --- | --- |
| `AUDITA_MODULES` | `--modules` | `glossary,homophones,glossary,spoken_word,grammar` | Comma-separated logical module keys to run; CLI overrides the environment value |
| `AUDITA_LLM_API_KEY` | `--llm-api-key` | unset | Preferred provider-neutral LLM API credential; CLI overrides both environment-key variants |
| `AUDITA_VALIDATION_LLM_API_KEY` | `--validation-llm-api-key` | unset | Validation-phase LLM API credential; defaults to the primary LLM API key |
| `AUDITA_MODEL` | `--model` | `openrouter/google/gemma-4-31b-it` | LLM model name sent to the configured OpenAI-compatible endpoint |
| `AUDITA_VALIDATION_MODEL` | `--validation-model` | unset | Validation-phase LLM model; defaults to `AUDITA_MODEL` |
| `AUDITA_BASE_URL` | `--base-url` | `https://openrouter.ai/api/v1` | OpenAI-compatible API base URL |
| `AUDITA_VALIDATION_BASE_URL` | `--validation-base-url` | unset | Validation-phase OpenAI-compatible API base URL; defaults to `AUDITA_BASE_URL` |
| `AUDITA_LLM_TIMEOUT_SECONDS` | `--llm-timeout-seconds` | `600` | Per-request timeout in seconds for LLM calls to the configured OpenAI-compatible endpoint |
| `AUDITA_VALIDATION_LLM_TIMEOUT_SECONDS` | `--validation-llm-timeout-seconds` | unset | Validation-phase per-request timeout in seconds; defaults to `AUDITA_LLM_TIMEOUT_SECONDS` |
| `AUDITA_VALIDATION_MAX_PROMPT_TOKENS` | `--validation-max-prompt-tokens` | `2048` | Maximum estimated tokens per validation-phase LLM prompt batch |
| `AUDITA_TARGET_SECTIONS` | `--target-sections` | unset | Exact number of contiguous proposal-stage transcript sections; errors if min/max token bounds cannot be satisfied |
| `AUDITA_MAX_RETRIES` | `--max-retries` | `3` | Maximum Instructor retries for structured responses |
| `AUDITA_VALIDATION_MAX_RETRIES` | `--validation-max-retries` | unset | Validation-phase structured-output retries; defaults to `AUDITA_MAX_RETRIES` |
| `AUDITA_VALIDATION_LLM_CONCURRENCY` | `--validation-llm-concurrency` | unset | Validation-phase LLM concurrency; defaults to `AUDITA_LLM_CONCURRENCY` |
| `AUDITA_MAX_SECTION_TOKENS` | `--max-section-tokens` | `8192` | Maximum estimated tokens per proposal-stage transcript section |
| `AUDITA_MIN_SECTION_TOKENS` | `--min-section-tokens` | `2048` | Minimum estimated tokens per proposal-stage transcript section when balancing for concurrency |
| `AUDITA_GLOSSARY_CONFIDENCE_THRESHOLD` | `--glossary-confidence-threshold` | `0.8` | Minimum confidence required for glossary proposals to survive validation |
| `AUDITA_GRAMMAR_CONFIDENCE_THRESHOLD` | `--grammar-confidence-threshold` | `0.8` | Minimum confidence required for grammar proposals to survive validation |
| `AUDITA_HOMOPHONES_CONFIDENCE_THRESHOLD` | `--homophones-confidence-threshold` | `0.8` | Minimum confidence required for homophone proposals to survive validation |
| `AUDITA_SPOKEN_WORD_CONFIDENCE_THRESHOLD` | `--spoken-word-confidence-threshold` | `0.8` | Minimum confidence required for spoken-word proposals to survive validation |
| `AUDITA_NORMALIZE_MAX_SEGMENT_GAP` | `--normalize-max-segment-gap` | `4.0` | Same-speaker gaps eligible for deterministic merging |
| `AUDITA_NORMALIZE_ELLIPSIS_GAP` | `--normalize-ellipsis-gap` | `3.5` | Same-speaker gaps above this value are joined with ` ... ` |
| `AUDITA_NORMALIZE_MAX_SEGMENT_DURATION` | `--normalize-max-segment-duration` | `60.0` | Maximum merged segment duration |
| `AUDITA_NORMALIZE_MAX_SEGMENT_TOKENS` | `--normalize-max-segment-tokens` | `2048` | Maximum merged segment prompt payload size |
| `AUDITA_WORK_DIR` | `--work-dir` | `/tmp/audita` | Per-run scratch diagnostics directory |
| `AUDITA_WORK_DIR_RETENTION` | `--work-dir-retention` | `auto` | Whether to retain the per-run work directory: `auto`, `always`, or `never` |
Set `AUDITA_MODULES=grammar` to run only the grammar module by default, or override it per command with `--modules`.
Validation-phase LLM settings inherit from the primary `AUDITA_*` LLM settings by default. Set any of the `AUDITA_VALIDATION_*` values only when you want LLM-backed validators to use a different model, endpoint, credential, timeout, retry budget, or concurrency level.
OpenRouter remains the default out of the box:
```sh
export AUDITA_LLM_API_KEY=your-openrouter-key
audita process transcript.json --glossary glossary.yaml --output corrected.json
```
You can point Audita at any OpenAI-compatible endpoint by changing `AUDITA_BASE_URL` and, if needed, `AUDITA_MODEL`. For example, a local vLLM server:
```sh
export AUDITA_LLM_API_KEY=local-dev-key
export AUDITA_BASE_URL=http://localhost:8000/v1
export AUDITA_MODEL=meta-llama/Llama-3.1-8B-Instruct
audita process transcript.json --glossary glossary.yaml --output corrected.json
```
Or the actual OpenAI API:
```sh
export AUDITA_LLM_API_KEY=your-openai-key
export AUDITA_BASE_URL=https://api.openai.com/v1
export AUDITA_MODEL=gpt-4.1-mini
audita process transcript.json --glossary glossary.yaml --output corrected.json
```
`AUDITA_WORK_DIR` stores per-run diagnostics while processing. Under the default `AUDITA_WORK_DIR_RETENTION=auto`, clean successful runs are removed, while failed runs and successful runs with final skipped corrections are preserved. Use `always` to keep every run directory and `never` to remove successful run directories even when skips remain.
Failed runs always preserve the run directory and include an authoritative `report.json` alongside normalization and prompt/response diagnostics.
## Prototype Archive
The archived prototype remains importable as `audita_prototype` and is still covered by its original regression suite. This is intentional: the new `audita` package is a framework-oriented rewrite, not a thin wrapper around the old code.
- [Full annotated pipeline](../../examples/pipeline.full.annotated.yml)
- [Production-shaped pipeline](../../examples/pipeline.production.yml)

View File

@@ -0,0 +1,132 @@
# Notarius Integration Contract
## Boundary
Narratio uses Notarius as a subprocess to extract configured structured JSON
lanes from the final trimmed Seriatim transcript. Narratio owns invocation,
safe bundle discovery, lane selection, and its own artifact metadata. Notarius
owns pipeline definitions, lane schemas, the receipt, and bundle formats.
Canonical Notarius v0.6.0 references:
- [CLI reference](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/cli.md)
- [Subprocess consumer contract](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/consumers/subprocess.md)
- [D&D pipeline and lane contracts](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/consumers/dnd-pipeline.md)
- [Run-result receipt](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/integrations/run-result.md)
- [JSON output bundle](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/integrations/json-output.md)
- [D&D spell-catalog overlay](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/integrations/dnd-spell-catalog-overlays.md)
The [complete Narratio example](../../examples/pipeline.full.annotated.yml)
records the exact current constraints for all ten D&D lanes. Treat the linked
Notarius documents as canonical when changing those values; Narratio does not
duplicate the complete schemas.
## Invocation
When `pipeline.notarius.enabled` is true, Narratio resolves the executable,
configuration path, input path, output directory, and working directory to
absolute paths. Narratio requires the Notarius v0.6.0 CLI contract when
references are configured and invokes each binding as a separate argument
before `--json`:
```text
notarius run <pipeline_id> --config <config_path> --input <trimmed_json> --output-dir <staging_dir> [--reference <selector>=<verified_snapshot_path>]... --json
```
Reference paths are absolute invocation-local snapshots streamed from the
manifest-verified canonical files prepared inside the current Narratio session
workspace. Narratio verifies snapshot checksum and size before and after the
subprocess, and passes only configured bindings, ordered lexically by normalized
selector, as direct argument-vector entries without shell interpretation. A CLI
binding takes precedence over a matching external path in Notarius
configuration. Narratio never emits `--without-reference`.
For canonical party configuration, the `party` binding is the unchanged,
validated authored roster and the `players` binding is its generated
projection. Both retain their established `narratio.input.party` and
`narratio.input.players` source IDs, and both are resolved from the prepared
manifest rather than from campaign configuration at extraction time.
The maintained D&D boundary binds only the four campaign-owned external slots:
```text
notarius run dnd-session \
--config <absolute config path> \
--input <absolute trimmed transcript path> \
--output-dir <absolute staging directory> \
--reference glossary=<absolute verified glossary snapshot> \
--reference party=<absolute verified party snapshot> \
--reference players=<absolute verified players snapshot> \
--reference spell_catalog=<absolute verified spell catalog snapshot> \
--json
```
The spell-catalog binding is omitted when the campaign does not maintain that
optional overlay. Registry, scene-description, combat-turn, and occurrence
handoffs generated during the same Notarius run remain in Notarius pipeline
composition and must not be emitted as CLI references. The linked CLI and D&D
consumer documents own selector targeting, declared slots, media compatibility,
and generated-handoff collision rules.
Standard output is reserved for the JSON receipt. Standard error is captured
separately as diagnostic output. Narratio applies the configured timeout and
does not interpret stdout as a receipt unless the subprocess exits successfully.
It does not pass a Narratio session ID or run `notarius config validate`
automatically; the configured working directory and Narratio's minimal child
environment apply to the subprocess.
## Accepted Result
Narratio's supported invocation baseline is Notarius v0.6.0. The accepted
receipt remains `notarius.run-result.v2`; reference flags do not change the
receipt or ten-lane output contract. The receipt
must identify the configured pipeline, and its `index_file` must be exactly
`index.json` beneath the reported bundle root. The production index must name
the management files exactly as `manifest.json`, `rejected.json`,
`warnings.json`, and `diagnostics.json`. All receipt, index, and lane paths must
stay inside that bundle; symlinks and non-regular lane payloads are rejected.
Supported receipt and index shapes tolerate unknown fields for forward
compatibility, while required identity, validation, count, manifest,
rejection, warning, diagnostic, and lane-list fields remain mandatory.
Narratio applies bounded reads to the receipt, index, rejection, warning, and
diagnostic documents. Warning and diagnostic envelopes, group counts,
occurrence counts, truncation state, framework-owned origins, and
receipt-to-bundle counts must be internally consistent. Optional chunk-map and
evidence-context descriptors must carry their complete generic contract
metadata when present.
For every entry in `pipeline.notarius.outputs`, Narratio requires exactly one
index descriptor with the configured lane ID, media type, schema ID, schema
version, and, when configured, module key. Missing, duplicate, rejected, or
incompatible required lanes fail extraction even if Notarius exited zero. A
configured lane whose v2 validation summary is `rejected` or `incomplete` also
fails extraction.
Unconfigured lanes may remain in the preserved bundle but do not become
selectable Narratio sources.
Each accepted configured lane is registered as
`narratio.extraction.<output_key>`. The bundle index is retained for audit and
resume validation but is not selectable. Scriptorium and publish rules consume
only explicitly named lane sources; `--artifacts` never selects Notarius lanes.
## Failure And Compatibility Behavior
- Startup and nonzero-exit errors fail extraction and retain captured diagnostics.
- Invalid receipt JSON or an unsupported receipt schema fails before bundle use.
- Unsafe or incompatible index data and required-lane rejection fail before the
staged bundle is promoted to durable storage.
- Contract and external provenance metadata are preserved on lane artifact
records and through explicit publication.
- Undeclared selectors, incompatible reference files, and external/generated
reference collisions are Notarius errors and fail extraction normally.
Rejection, validation, warning, and diagnostic summaries retain bounded stable
identity, category, origin, reason-code, status, and occurrence fields without
copying free-form external messages into Narratio manifest metadata or reading
lane payload bodies.
Configuration fields and defaults are in [Configuration](../config.md).
Operator paths, rerun procedures, and bundle retention are in
[Operations](../operations.md). See [Troubleshooting](../troubleshooting.md)
for failure recovery.

View File

@@ -0,0 +1,73 @@
# Canonical Party Input
`party.yml` is a campaign-owned roster input. Narratio recognizes the
versioned `narratio.party.v1` document below when it resolves a pipeline,
campaign, and session together.
```yaml
schema_version: narratio.party.v1
characters:
arannis:
player:
name: Eric
character:
name: Arannis
alias:
- Ari
- The Grey Owl
classes:
- name: wizard
level: 8
```
`characters` is a non-empty mapping. Each key is a stable character ID using
the configured-artifact key grammar: a lowercase ASCII letter followed by zero
or more lowercase ASCII letters, digits, or underscores. Character order is
preserved where roster order matters.
Every entry has `player.name`, `character.name`, and a non-empty
`character.classes` list. Class entries require a non-empty `name` and may
include a positive integer `level`. The optional, intentionally singular
`character.alias` field is a list. Names, aliases, and class names must be
non-empty, trimmed display strings without control characters. Character names
and aliases must be unique across the full roster under Unicode-aware
case-insensitive comparison; player names may repeat.
The document has exactly one YAML document and accepts no unknown fields. A
wrong or malformed `schema_version` is an error.
## Legacy migration boundary
An unversioned party input remains supported only as opaque legacy reference
material while campaigns migrate. It requires a separate `players_file` and
retains the existing session override behavior. It cannot be mixed with a
canonical party: canonical campaigns must omit `players_file`, and sessions
must not override their party or players inputs.
Use the canonical document for new campaigns. The configuration rules and
source-relative path behavior are defined in the [Configuration Reference](../config.md).
## Derived players document
During `prepare`, Narratio copies the canonical party source bytes unchanged
to `inputs/party.yml` and writes this deterministic players-only projection to
`inputs/players.yml`:
```yaml
schema_version: narratio.players.v1
players:
- name: Eric
character:
id: arannis
name: Arannis
alias:
- Ari
- The Grey Owl
```
There is one entry per character, sorted by stable character ID. Repeated
player names remain separate entries. The optional `alias` list retains its
declared order and is omitted when empty. The projection carries no class
data. Its prepared manifest record is marked `derived_from_party`; it is not a
separate user-provided `players_file`.

View File

@@ -1,339 +1,71 @@
# Narratio -> Scriptorium CLI Integration
# Integration: Scriptorium
## 1. Purpose
## Purpose
Define the Scriptorium adapter contract used by `analyze` and trim-bounds generation in `trim`.
This document defines how Narratio should invoke Scriptorium through the **public CLI**.
## External Boundary
This is a **subprocess integration contract**, not an internal Go API contract.
## 2. Assumptions
- `scriptorium` is installed and available on `PATH`.
- Scriptorium is configured with `config.yml`.
- `config.yml` provides `prompt_dir`, `profile_dir`, and `schema_dir` as needed.
- Prompt and profile libraries are already deployed for the environment.
- Narratio provides prepared artifact files (for example polished transcript, glossary, previous recap, campaign notes).
- Initial integration is synchronous subprocess execution.
- Narratio remains the orchestrator.
In normal operation, Narratio does not need to pass `--prompt-dir` and `--profile-dir` if they are supplied by Scriptorium config.
Narratio may pass `--config <PATH>` when it must use a non-default Scriptorium config file.
## 3. Core Commands Narratio May Call
Primary commands for subprocess integration:
Narratio invokes Scriptorium as a subprocess in these modes:
- `scriptorium run`
- `scriptorium render`
For production generation, use `scriptorium run`.
`scriptorium render` is for debugging, dry-runs, test assertions, and validating command construction without LLM execution.
Note: `scriptorium serve` and HTTP API exist, but they are not the initial integration path.
## 4. Command Selection Guidance
- Use `run` to generate an output artifact.
- Use `render` to inspect the prepared prompt and effective settings without calling the LLM.
- Use `render --format json` when Narratio/tests need structured prepare output.
## 5. Recommended `run` Invocation Shape
Production shape:
```bash
scriptorium run \
--prompt <prompt_id> \
--input transcript=<processed-transcript-path> \
--out <output-artifact-path>
```
Common optional additions:
- `--config <path>`: use a specific Scriptorium config file.
- `--profile <profile_id>`: override prompt default profile.
- `--var name=value` (repeatable): small metadata values.
- `--input name=path` (repeatable): additional named artifacts.
- `--timeout <duration>`: per-run timeout override.
- Runtime model override flags (`--llm-base-url`, `--model`, etc.) only for exceptional/operator-directed cases.
## 6. Recommended `render` Invocation Shape
Human-readable debug shape:
```bash
scriptorium render \
--prompt <prompt_id> \
--input transcript=<processed-transcript-path> \
--format text
```
Structured debug/test shape:
```bash
scriptorium render \
--prompt <prompt_id> \
--input transcript=<processed-transcript-path> \
--format json \
--out <render-debug-path>
```
`render` does **not** call the LLM, does **not** validate model output, and does **not** perform repair.
## 7. Inputs
- Pass inputs as repeated `--input name=path` flags.
- `name` must match the Prompt Definition input name.
- Prefer absolute paths, or paths relative to a working directory controlled by Narratio.
- Pass Audita output as the primary transcript input.
- Additional inputs may include glossary, previous recap, campaign notes, event logs, final state maps, or other prompt-specific artifacts.
- Scriptorium reads input files directly; Narratio does not need to inline file content for CLI use.
## 8. Variables
Use repeated `--var name=value` for small metadata values.
Typical examples:
- `session_date`
- `session_id`
- `campaign_name`
- `previous_session_id`
- `output_kind`
Large content belongs in input files, not `--var` values.
## 9. Prompt IDs and Output Artifact Types
Narratio should treat prompt IDs as configuration, not hardcoded business logic.
Narratio config may map stage/output names to prompt IDs, for example:
- session recap prompt
- structured event extraction prompt
- glossary suggestion prompt
- player-facing summary prompt
Prompt IDs used by Narratio should come from the deployed Scriptorium prompt library.
## 10. Profiles
- Prompts may declare `default_profile`.
- Narratio may omit `--profile` to use prompt default profile.
- Narratio may pass `--profile` to force profile selection.
- This enables environment/profile selection like `local-fast`, `local-quality`, `frontier`, `batch`, or test profiles.
- Profile names should generally be Narratio configuration values.
## 11. Runtime Overrides
Supported runtime override flags:
- `--llm-base-url`
- `--model`
- `--api-key-env`
- `--temperature`
- `--max-tokens`
- `--top-p`
- `--timeout`
Guidance:
- Keep normal model/runtime settings in Execution Profiles.
- Use runtime overrides only for explicit per-run exceptions, tests, or operator overrides.
- Never pass raw API keys on the command line.
- `--api-key-env` names an environment variable; Narratio must ensure that variable is set in subprocess environment.
## 12. Config Behavior
- Default config path: `/etc/scriptorium/config.yml`.
- `--config <PATH>` overrides default path.
- Missing default config is allowed by Scriptorium.
- If `--config` is provided explicitly, the file must exist and be valid.
- CLI flags override `config.yml`.
- `config.yml` overrides built-in application defaults.
Narratio can either:
- rely on system default config path, or
- carry an explicit config path and pass `--config`.
## 13. Environment Handling
Subprocess environment recommendations:
- Pass through required API-key environment variables referenced by `api_key_env`.
- Do not pass raw API keys as CLI arguments.
- Avoid logging full environment dumps.
- Capture stdout and stderr separately.
- Use a controlled working directory.
- Prefer absolute artifact paths.
## 14. Output Handling
For `scriptorium run`:
- Use `--out` when Narratio needs durable artifact files.
- Without `--out`, artifact content is written to stdout.
- Preferred orchestration pattern: always use `--out`, then treat the file as stage output artifact.
- Capture stderr for diagnostics.
For `scriptorium render`:
- Use `--out` to store render diagnostics.
- Use `--format json` when tests need to inspect selected profile, effective runtime settings, input hashes, prompt hash, and rendered messages.
## 15. Exit Status and Errors
Current CLI behavior (verified from implementation/tests):
- `0`: success.
- `1`: runtime/parse/config/load/render/generation/IO error.
- `2`: run completed but output validation failed (`ValidationFailed`).
Additional details:
- On `run`, output artifact write happens before exit code selection. If validation fails, artifact may still be written and exit code is `2`.
- `stderr` carries both errors and normal run summary output; non-empty stderr alone does not imply failure.
- `render` returns `0` on success and `1` on failures.
Narratio should treat non-zero exit codes as failed stage execution, but may record generated artifact paths if a run exited `2` and output file exists.
## 16. Recommended Narratio Integration Pattern
1. Build CLI args from Narratio stage configuration.
2. Use subprocess context cancellation/timeout.
3. Pass absolute input paths.
4. Pass `--out` to a session-scoped artifact path.
5. Add `--var` metadata values.
6. Optionally add `--config`.
7. Optionally add `--profile`.
8. Ensure required API-key env vars are present.
9. Run subprocess synchronously.
10. Capture stdout/stderr separately.
11. On success, store output artifact path and invocation metadata in stage artifacts.
12. On failure, store exit code and stderr diagnostics in stage status.
## 17. Suggested Narratio Configuration Shape
Illustrative `pipeline.yml` shape:
```yaml
scriptorium:
binary: scriptorium
config_path: /etc/scriptorium/config.yml
timeout: 10m
render_debug: false
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
profile_id: local-quality # optional
output_path: artifacts/session_recap.md
timeout: 10m
render_debug: false # optional artifact override
inputs:
transcript:
source: trimmed_transcript
required: true
previous_recap:
source: previous_session_artifact
artifact: session_recap
path: "" # optional
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: session_recap
```
The key idea: map Narratio artifact names to prompt ID, optional profile, expected inputs, vars, and output destination.
## 18. Testing Strategy for Narratio Integration
- Use `scriptorium render --format json` to verify command construction without LLM calls.
- Use dedicated test prompt/profile libraries for integration tests.
- Use small fixture transcripts.
- Verify missing-input failure behavior.
- Verify prompt `default_profile` behavior.
- Verify explicit `--profile` override behavior.
- Verify `--config` behavior (default and explicit).
- Verify output file creation when `--out` is used.
- Verify stderr capture on failures.
- Avoid real API keys in tests.
## 19. Security and Privacy Notes
- Never pass raw API keys on command line.
- Do not log full rendered prompts by default; transcripts may contain sensitive content.
- Avoid logging prompt content unless explicit debug mode is enabled.
- Treat generated artifacts as potentially sensitive.
- Use session-scoped, access-controlled output paths.
- `api_key_env` names should come from environment management, not embedded secrets.
## 20. Initial D&D Artifact Generation Examples
These are examples only. Use prompt IDs from the deployed prompt library.
Session recap:
```bash
scriptorium run \
--prompt dnd.session_recap \
--input transcript=/work/session-42/transcript.polished.md \
--input glossary=/work/session-42/glossary.yml \
--out /work/session-42/artifacts/session_recap.md
```
Structured events:
```bash
scriptorium run \
--prompt dnd.structured_events \
--input transcript=/work/session-42/transcript.polished.md \
--out /work/session-42/artifacts/structured_events.json
```
Glossary suggestions:
```bash
scriptorium run \
--prompt dnd.glossary_suggestions \
--input transcript=/work/session-42/transcript.polished.md \
--input previous_recap=/work/session-41/artifacts/session_recap.md \
--out /work/session-42/artifacts/glossary_suggestions.md
```
Player-facing summary:
```bash
scriptorium run \
--prompt dnd.player_summary \
--input transcript=/work/session-42/transcript.polished.md \
--input structured_events=/work/session-42/artifacts/structured_events.json \
--out /work/session-42/artifacts/player_summary.md
```
## 21. Non-Goals
Initial Narratio integration should not:
- call Scriptorium internal Go packages
- use HTTP API as the primary path
- expect Scriptorium to read S3 refs directly
- make Scriptorium responsible for Narratio stage state
- make Scriptorium responsible for notification
- require Scriptorium to understand D&D workflow semantics beyond prompt definitions
## 22. Future Extension Notes
Possible later extensions:
- HTTP API integration
- S3 artifact references if Scriptorium adds S3 reader support
- richer render diagnostics and policy controls
- token budgeting/prompt-size checks
- batch execution if Scriptorium later adds batch support
The request timeout and parent cancellation bound each invocation. Internal
runner composition is documented in
[the adapter implementation guide](../internal/adapters.md).
## Request Contract
Both request types carry:
- binary/config/prompt/profile IDs;
- input map and vars map;
- output path;
- timeout;
- generated config + stdout/stderr log paths;
- optional API-key env var name;
- optional working directory.
## Result Contract
`ArtifactResult` returns:
- output/log/generated-config paths;
- exit code and duration;
- command mode (`run` or `render`);
- prompt/profile provenance;
- `ValidationFailed` marker;
- metadata map.
## Validation and Failure Semantics
Request validation fails for:
- missing binary, prompt id, or output path;
- non-positive timeout;
- empty input/var names;
- empty input path values;
- missing required credential env var when `APIKeyEnv` is set.
Run behavior:
- subprocess errors propagate with context;
- `run` exit code `2` is mapped to `ValidationFailed=true`;
- successful subprocess still fails if output file is missing or empty.
Each artifact result is limited to 64 MiB and must be a regular file without
symlinked path components.
Render behavior:
- subprocess errors propagate;
- output file must exist and be non-empty.
## Deterministic Behavior
- input and var maps are sorted into deterministic `--input` and `--var` CLI args.
- stage wiring adds `session_id=narratio-session-<session_id>` to every Scriptorium request for sticky upstream routing, overriding any configured `vars.session_id`.
- generated invocation YAML (`scriptorium.generated.v1`) is emitted when requested.
- adapter is stateless and does not own artifact-selection policy.
## Configuration
Operator-selected values are defined under `pipeline.scriptorium.*`, including
per-artifact settings under `pipeline.scriptorium.artifacts.*`, in the
[configuration reference](../config.md#pipeline).
Maintained examples with Scriptorium config:
- [Full annotated pipeline](../../examples/pipeline.full.annotated.yml)
- [Production-shaped pipeline](../../examples/pipeline.production.yml)

View File

@@ -1,403 +1,62 @@
# seriatim
# Integration: Seriatim
`seriatim` merges per-speaker WhisperX-style JSON transcripts into a single JSON transcript that preserves speaker identity and chronological order.
## Purpose
Define the Seriatim adapter contract used by `merge`, `normalize`, `trim`, and `render`.
The current implementation supports the `merge` command. It reads one or more input JSON files, optionally maps each input file to a canonical speaker using `speakers.yml`, sorts all segments by timestamp, detects and resolves overlaps when word-level timing is available, assigns consecutive numeric `id` values, and writes a merged JSON artifact.
## External Boundary
## Usage
Narratio invokes Seriatim as a subprocess in these modes:
Run from source:
- `seriatim merge`
- `seriatim normalize`
- `seriatim trim`
- `seriatim render`
```sh
go run ./cmd/seriatim merge \
--input-file samples/raw/2026-04-19-Eric_Rakestraw.json \
--input-file samples/raw/2026-04-19-Mike_Brown.json \
--output-file merged.json
```
The configured timeout and parent cancellation bound each invocation. Internal
runner composition is documented in
[the adapter implementation guide](../internal/adapters.md).
Optional report output:
## Request/Result Contracts
- `MergeRequest`/`MergeResult`: multi-input merge to base transcript, optional report.
- `NormalizeRequest`/`NormalizeResult`: transcript normalization with explicit schema.
- `TrimRequest`/`TrimResult`: transcript trimming with required keep selector.
- `RenderRequest`/`RenderResult`: transcript-to-markdown rendering with explicit format and render booleans.
```sh
go run ./cmd/seriatim merge \
--input-file eric.json \
--input-file mike.json \
--output-file merged.json \
--report-file report.json
```
Results include output/log/config paths, timing, exit code, and metadata.
## CLI
## Validation and Failure Semantics
Runner construction validates:
- binary presence;
- timeout > 0;
- supported output schema (`seriatim-minimal|seriatim-intermediate|seriatim-full`);
- non-negative coalesce gap.
```text
seriatim merge [flags]
```
Invocation fails on:
- missing required request paths/inputs;
- invalid normalize schema override;
- unsupported render format;
- subprocess failure;
- invalid JSON outputs for merge/normalize/trim;
- missing `segments` array for normalize/trim transcript outputs;
- empty render output files.
Global flags:
When report paths are provided/enabled, report files must parse as JSON.
Each Seriatim JSON or rendered-text result is limited to 64 MiB and must be a
regular file without symlinked path components.
| Flag | Description |
| --- | --- |
| `--help` | Show command help. |
| `--version` | Show application version. Local builds default to `dev`; release builds inject the release version. |
## Deterministic Behavior
- argument ordering is deterministic per command construction.
- merge env overrides are explicit (`SERIATIM_*`) and only emitted when configured.
- generated invocation YAML (`seriatim.generated.v1`) is emitted when requested.
- adapter does not write manifests or choose stage inputs.
`merge` flags:
## Configuration
| Flag | Required | Default | Description |
| --- | --- | --- | --- |
| `--input-file` | Yes | none | Input transcript JSON file. Repeat once per speaker/input file. |
| `--output-file` | Yes | none | Merged transcript JSON output path. |
| `--report-file` | No | none | Optional report JSON output path. |
| `--speakers` | No | none | Speaker map YAML file. When omitted, input file basenames are used as speaker labels. |
| `--autocorrect` | No | none | Autocorrect rules YAML file. When omitted, the default `autocorrect` module leaves text unchanged. |
| `--input-reader` | No | `json-files` | Input reader module. |
| `--output-modules` | No | `json` | Comma-separated output modules. |
| `--output-schema` | No | `seriatim-intermediate` | JSON output contract. Allowed values are `seriatim-minimal`, `seriatim-intermediate`, and `seriatim-full`. If omitted, the runtime default is used; consumers that depend on a specific shape should set this explicitly. |
| `--preprocessing-modules` | No | `validate-raw,normalize-speakers,trim-text` | Comma-separated preprocessing modules, evaluated in order. |
| `--postprocessing-modules` | No | `detect-overlaps,resolve-overlaps,backchannel,filler,resolve-danglers,coalesce,detect-overlaps,autocorrect,assign-ids,validate-output` | Comma-separated postprocessing modules, evaluated in order. |
| `--coalesce-gap` | No | `3.0` | Maximum same-speaker gap in seconds for `coalesce`; also used as the `resolve-overlaps` context window. Must be a non-negative float. |
Operator-selected values are defined under `pipeline.seriatim.*` and
`pipeline.render.*` in the
[configuration reference](../config.md#pipeline).
Environment variables:
Maintained examples with Seriatim config:
| Environment Variable | Default | Description |
| --- | --- | --- |
| `SERIATIM_OUTPUT_SCHEMA` | `seriatim-intermediate` | Output schema used when `--output-schema` is not explicitly provided. Allowed values are `seriatim-minimal`, `seriatim-intermediate`, and `seriatim-full`. The CLI flag takes precedence. |
| `SERIATIM_OVERLAP_WORD_RUN_GAP` | `1.0` | Maximum gap in seconds between adjacent timed words when `resolve-overlaps` builds word-run replacement segments. Must be a positive float. |
| `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW` | `1.0` | Near-start window in seconds for ordering replacement word runs shortest-first. Must be a positive float. |
| `SERIATIM_BACKCHANNEL_MAX_DURATION` | `2.0` | Maximum duration in seconds for `backchannel` classification. Must be a positive float. |
| `SERIATIM_FILLER_MAX_DURATION` | `1.25` | Maximum duration in seconds for `filler` classification. Must be a positive float. |
## Input JSON Format
Each input file must be valid JSON with a top-level `segments` array. The current parser accepts the WhisperX segment subset needed for merging:
```json
{
"segments": [
{
"start": 1.25,
"end": 3.5,
"text": "Hello there.",
"words": [
{"word": "Hello", "start": 1.25, "end": 1.55, "score": 0.98},
{"word": "there.", "start": 1.7, "end": 2.0}
]
}
]
}
```
Required segment fields:
- `start`: number, must be `>= 0`.
- `end`: number, must be `>= start`.
- `text`: string.
Optional word fields:
- `words`: array of word timing objects.
- `words[].word`: string.
- `words[].start`: optional number, must be `>= 0` when present.
- `words[].end`: optional number, must be `>= start` when present with `start`.
- `words[].score`: optional number.
- `words[].speaker`: optional raw speaker label string.
Word-level timing is preserved internally for overlap resolution. If a word is missing `start` or `end`, seriatim keeps the word text, emits a warning in the optional report, and does not use that word as a timing anchor. Word timing is not emitted in the final JSON artifact.
## Speaker Map Format
`speakers.yml` maps input files to canonical speaker names using ordered substring rules:
This file is optional. If `--speakers` is omitted, `seriatim` uses each input file basename as the segment speaker label.
```yaml
match:
- speaker: "Eric Rakestraw"
match:
- "Eric_Rakestraw"
- "Eric"
- speaker: "Mike Brown"
match:
- "Mike_Brown"
- "mb"
```
For each `--input-file`, `seriatim` takes the file basename and evaluates the rules in order. The first rule with a matching substring wins, and no later rules are evaluated.
For example, this input:
```text
samples/raw/2026-04-19-Eric_Rakestraw.json
```
matches this rule because the basename contains `Eric_Rakestraw`:
```yaml
- speaker: "Eric Rakestraw"
match:
- "Eric_Rakestraw"
```
Important details:
- Matching is against the input file basename, not the full path.
- Matching is case-insensitive.
- Rules are evaluated from first to last.
- Each rule must have a non-empty `speaker`.
- Each rule must have at least one non-empty `match` string.
- Duplicate speaker names are invalid.
- Every input file must match at least one rule or the command fails.
Deprecated old format:
```yaml
inputs:
eric.json:
speaker: "Eric Rakestraw"
```
The old `inputs:` direct mapping format is no longer supported.
## Output JSON Format
`--output-modules json` controls the writer. `--output-schema` controls the JSON contract that writer serializes.
The named schemas are stable public contracts. If a consumer depends on a specific shape, it should request that schema explicitly at runtime. The runtime default selection may change in a future release.
The `seriatim-intermediate` schema is the current default selection when neither `--output-schema` nor `SERIATIM_OUTPUT_SCHEMA` is set. It stays close to the minimal schema, but adds optional `categories` on each segment:
```json
{
"metadata": {
"application": "seriatim",
"version": "dev",
"output_schema": "seriatim-intermediate"
},
"segments": [
{
"id": 1,
"start": 1.25,
"end": 3.5,
"speaker": "Eric Rakestraw",
"text": "Hello there.",
"categories": ["backchannel"]
}
]
}
```
The `seriatim-full` schema uses the full seriatim envelope:
```json
{
"metadata": {
"application": "seriatim",
"version": "dev",
"input_reader": "json-files",
"input_files": ["eric.json", "mike.json"],
"preprocessing_modules": ["validate-raw", "normalize-speakers", "trim-text"],
"postprocessing_modules": ["detect-overlaps", "resolve-overlaps", "backchannel", "filler", "resolve-danglers", "coalesce", "detect-overlaps", "autocorrect", "assign-ids", "validate-output"],
"output_modules": ["json"]
},
"segments": [
{
"id": 1,
"source": "eric.json",
"source_segment_index": 0,
"speaker": "Eric Rakestraw",
"start": 1.25,
"end": 3.5,
"text": "Hello there.",
"overlap_group_id": 1
},
{
"id": 2,
"source": "eric.json",
"source_ref": "word-run:1:1:1",
"derived_from": ["eric.json#0"],
"speaker": "Eric Rakestraw",
"start": 2.0,
"end": 2.5,
"text": "Resolved word run",
"categories": ["backchannel"]
}
],
"overlap_groups": [
{
"id": 1,
"start": 1.25,
"end": 4.0,
"segments": ["eric.json#0", "mike.json#0"],
"speakers": ["Eric Rakestraw", "Mike Brown"],
"class": "unknown",
"resolution": "unresolved"
}
]
}
```
The `seriatim-minimal` schema emits minimal metadata and compact ordered segments:
```json
{
"metadata": {
"application": "seriatim",
"version": "dev",
"output_schema": "seriatim-minimal"
},
"segments": [
{
"id": 1,
"start": 1.25,
"end": 3.5,
"speaker": "Eric Rakestraw",
"text": "Hello there."
}
]
}
```
Minimal output intentionally omits categories, overlap groups, source/provenance fields, and pipeline configuration metadata.
Intermediate output intentionally omits overlap groups and source/provenance fields, but keeps optional `categories` and minimal metadata.
Segments are sorted deterministically by:
```text
(start, end, source, source_segment_index/source_ref, speaker)
```
Final segment IDs are assigned after sorting and start at `1`.
The public Go output contract is available from:
```go
import "gitea.maximumdirect.net/eric/seriatim/schema"
```
The same package embeds machine-readable JSON Schemas in `schema/full-output.schema.json`, `schema/intermediate-output.schema.json`, and `schema/minimal-output.schema.json`. The default `validate-output` postprocessor validates the selected output shape and verifies final segment IDs are present, sequential, and start at `1`.
## Overlap Detection
The default postprocessing pipeline detects overlapping segment groups.
Overlap behavior:
- A strict timing overlap is required: `next.start < current_group_end`.
- Segments that only touch at a boundary are not grouped.
- Groups require at least two distinct speakers.
- Transitive overlaps are grouped together.
- Segments in detected groups receive `overlap_group_id`.
- `overlap_groups[].segments` contains stable references in `source#source_segment_index` format.
- `class` is currently `unknown`.
- `resolution` is `unresolved` until `resolve-overlaps` replaces the group.
## Overlap Resolution
The default postprocessing pipeline runs `detect-overlaps`, then `resolve-overlaps`, then `backchannel`, then `filler`, then `resolve-danglers`, then `coalesce`, then a second `detect-overlaps` pass.
For each detected overlap group, `resolve-overlaps` uses preserved WhisperX word timing to build smaller word-run replacement segments:
- The resolution window expands the detected overlap group by `--coalesce-gap` seconds on both sides.
- Nearby same-speaker context segments are included when they intersect the expanded window and their start or end is within `--coalesce-gap` of the original overlap boundary.
- Once a segment is selected for replacement, all timed words from that segment participate in word-run construction; the window controls segment selection, not per-word clipping.
- Context segments that are part of another detected overlap group are not pulled into the current group.
- Untimed words are included in replacement text in original word order when nearby timed words create a replacement run.
- Untimed words do not affect replacement segment start/end times or word-run gap splitting.
- Words for the same speaker are merged into one run when the gap between adjacent words is no greater than `SERIATIM_OVERLAP_WORD_RUN_GAP`.
- The default word-run gap is `1.0` seconds.
- Set `SERIATIM_OVERLAP_WORD_RUN_GAP` to a positive number of seconds to override the default.
- Near-start replacement word runs are reordered so shorter segments come first when adjacent starts are within `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW`.
- The default word-run reorder window is `1.0` seconds.
- Set `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW` to a positive number of seconds to override the default.
- Replacement segment text is built by joining word text with single spaces.
- Replacement segments include `source_ref` and `derived_from`.
- Replacement segments omit `source_segment_index` because they are derived from one or more original segments.
- Resolved overlap groups are removed before the second detection pass.
- Replacement segments are left without `overlap_group_id` until the second detection pass annotates any remaining overlap.
- If a speaker has no usable word timing in a group, that speaker's original segment is kept.
- If no speakers in a group have usable word timing, the original group and annotations remain unchanged.
## Backchannels
The default pipeline runs `backchannel` before `coalesce`. It tags short acknowledgement segments with:
```json
"categories": ["backchannel"]
```
Backchannel matching is case-insensitive, ignores punctuation for matching and word-count purposes, trims surrounding whitespace, and requires a matching acknowledgement phrase, no more than three whitespace-delimited words, and duration no greater than `SERIATIM_BACKCHANNEL_MAX_DURATION` seconds. The default maximum duration is `2.0` seconds.
## Fillers
The default pipeline runs `filler` after `backchannel` and before `coalesce`. It tags short filler utterances with:
```json
"categories": ["filler"]
```
Filler matching is case-insensitive, ignores punctuation for matching and word-count purposes, trims surrounding whitespace, and requires only filler tokens such as `um`, `uh`, `er`, `erm`, `ah`, `eh`, `hmm`, `mm`, or repeated combinations of those tokens. Matching segments must contain no more than three whitespace-delimited words and have duration no greater than `SERIATIM_FILLER_MAX_DURATION` seconds. The default maximum duration is `1.25` seconds.
## Dangler Resolution
The default pipeline runs `resolve-danglers` before `coalesce` and before the second overlap detection pass. It repairs short derived fragments when they share provenance with a nearby segment:
- Dangling-end fragments have no more than two words and end in punctuation.
- Dangling-start fragments have no more than two words.
- Matching uses same-speaker segments with any shared `derived_from` value.
- Merged segments use `source_ref` values such as `resolve-danglers:1`, keep the target segment's transcript position, and union `derived_from`.
## Coalescing
The default pipeline runs `coalesce` after `resolve-danglers` and before the second overlap detection pass. It merges adjacent same-speaker segments in the transcript's current order when `next.start - current.end <= --coalesce-gap`.
Coalesced segments use `source_ref` values such as `coalesce:1`, include `derived_from`, and omit `source_segment_index`.
Different-speaker backchannel and filler segments do not block coalescing of surrounding same-speaker segments. Same-speaker backchannel and filler segments are merged normally when they are within `--coalesce-gap`. When same-speaker segments are coalesced, any `backchannel` or `filler` category from the merged inputs is dropped from the coalesced segment.
## Autocorrect
Autocorrect is included in the default postprocessing pipeline. If `--autocorrect` is omitted, the module leaves transcript text unchanged and records a skip event in the optional report.
Enable corrections by passing `--autocorrect`:
```sh
go run ./cmd/seriatim merge \
--input-file input.json \
--autocorrect autocorrect.yml \
--output-file merged.json
```
`autocorrect.yml` format:
```yaml
autocorrect:
- target: "Hrank"
match:
- "hrank"
- "Frank"
- target: "Mike Brown"
match:
- "Mike Pat"
```
Matching behavior:
- Matching is case-sensitive.
- Matches apply only to whole tokens, not substrings inside larger words.
- Punctuation and whitespace can surround a match.
- Multi-word and hyphenated matches are supported.
- Duplicate match strings are invalid, including duplicates across separate rules.
## Current Limitations
- Only JSON input is supported.
- Overlap resolution depends on WhisperX word timing; groups without usable word timing remain unresolved.
- Alternate output formats are not implemented yet.
## Release Builds
Local builds record version metadata as `dev`. Release builds should inject the release version with `ldflags`:
```sh
go build -ldflags "-X gitea.maximumdirect.net/eric/seriatim/internal/buildinfo.Version=v1.0.0" ./cmd/seriatim
```
- [Full annotated pipeline](../../examples/pipeline.full.annotated.yml)
- [Production-shaped pipeline](../../examples/pipeline.production.yml)

View File

@@ -0,0 +1,71 @@
# Integration: WhisperX
## Purpose
WhisperX transcribes each prepared speaker audio file for Narratio's
`transcribe` stage. Narratio uses an HTTP boundary and installs each successful
response as that speaker's raw transcript JSON.
## HTTP Boundary
Narratio sends an HTTP `POST` to the configured transcription URL using
`multipart/form-data` with:
- `file`: the audio file, retaining its base filename; and
- `language`: the configured language string.
The server must return a `2xx` response whose body is valid JSON. Narratio does
not currently require a more specific response schema at this boundary.
The transcription URL must be an absolute `http` or `https` URL. The audio body
is streamed through a fresh multipart writer for every attempt, so its memory
use is bounded by the transport buffer rather than by the complete audio file.
WhisperX response acquisition is capped at 10 MiB.
## Request And Result Contract
Each adapter request identifies a speaker, a readable audio file, and the
destination for the raw transcript. The HTTP request carries the audio and
language; the speaker identifier remains Narratio orchestration metadata.
On success, Narratio atomically writes the response body to the requested
destination. The adapter result reports that logical output together with the
attempt count, final HTTP status when available, elapsed duration, and adapter
identity metadata. A failed or invalid response is not installed as the
transcript output.
## Retry, Timeout, And Cancellation
- The configured timeout applies independently to each HTTP attempt.
- `retries` means additional attempts after the first.
- HTTP `429`, HTTP `5xx`, attempt timeouts, and network errors are retryable.
- Other HTTP `4xx` responses and explicit cancellation are not retryable.
- Narratio waits the configured retry delay between attempts and aborts that
wait when the parent context is canceled.
## Validation And Failure Semantics
Client construction rejects a missing or non-HTTP(S) absolute transcription URL,
a missing language, a non-positive timeout, negative retries, or a negative
retry delay. A request fails before transmission when its audio or output path
is missing.
Non-`2xx` status, transport failure, response-size overflow, invalid JSON, or
failure to install the output causes the transcription to fail. Errors include
attempt context, and the result retains attempts, final status when available,
and elapsed duration for diagnostics.
## Determinism And Concurrency
Each audio request has stable multipart field names, and successful bytes are
installed atomically. The transcribe stage may process speaker files in
parallel, bounded by the configured concurrency. It records results in stable
speaker order after all work completes; any speaker failure fails the stage.
## Related Canonical Docs
- [Configuration](../config.md#pipeline) defines the operator-selected
WhisperX URL, language, timeouts, retry policy, and concurrency.
- [Adapter implementation](../internal/adapters.md) describes internal wiring.
- [Transcribe stage](../internal/stage-transcribe.md) describes stage mechanics,
durable artifacts, and manifests.

105
docs/internal/adapters.md Normal file
View File

@@ -0,0 +1,105 @@
# Internal: Adapters
## Purpose
Explain the adapter interfaces and production composition used by application
and stage orchestration. Externally observable protocols and formats belong in
the [integration contracts](../integrations/).
## Adapter Boundaries
Narratio stage logic depends on adapter interfaces, not transport-specific details.
Primary adapters:
- `whisperx.Client`
- `seriatim.Runner`
- `audita.Runner`
- `scriptorium.Runner`
- `notarius.Runner`
- `storage.ObjectStore`
- `notify.Sender`
## Ownership
Adapters own:
- HTTP/subprocess/SDK argument and transport details.
- Backend-specific request/response mapping.
Adapters do not own:
- stage ordering/skip/force logic;
- manifest transitions;
- canonical path policy.
## Default Wiring
`internal/app/runner.go` initializes default adapters when not injected and
only when the selected execution plan needs them:
- WhisperX HTTP client for `transcribe`.
- Seriatim subprocess runner for `merge`, `normalize`, `trim`, or `render`.
- Audita subprocess runner for `polish`.
- Scriptorium subprocess runner for `trim` or `analyze`.
- Notarius subprocess runner for `extract` when extraction is enabled.
- Noop notifier (`notify.NoopSender`) for `notify`.
- Object store only when required by selected stages/config.
Remote publish locks are loaded only for a selected, enabled publish that
uploads a run. Shared session lifecycle setup still applies to every selected
range, but an unselected integration is neither initialized nor validated by
runner composition. Each selected stage retains its own fail-fast configuration
and input validation.
`session plan` is outside production adapter composition. It performs
resume validation and models selected transitions against cloned manifest
state without constructing or invoking stage-execution adapters. The shared
command configuration loader may still use object storage to retrieve a missing
remote session file before planning begins.
Notarius is composed only when extraction is enabled; the extract stage owns
prepared reference resolution, receipt, bundle, and configured-lane policy.
The adapter validates the ordered selector/absolute-path pairs and is the sole
owner of serializing them as repeated `--reference` arguments before `--json`.
Object-store construction goes through `newCommandObjectStore`, which loads
configured filesystem secrets before adapter initialization.
## Failure Semantics
- Constructor errors fail stage execution setup early.
- Runtime adapter errors propagate to stage code and then manifest failure handling.
- Subprocess adapters persist stage logs/generated configs through stage-managed paths.
- Shared subprocess execution starts an owned process group on Linux/macOS or a
kill-on-close job object on Windows. Every terminal path disposes of that
owned tree before returning. After a natural leader exit, Unix checks for
remaining group members and uses bounded graceful then forceful termination;
Windows closes the job so kill-on-close applies. Cancellation, deadlines, and
diagnostic limits use the same terminal disposal path without losing their
original result classification. Child environments contain only the execution
baseline and adapter-specified values; configured credentials are explicit
sensitive values. Stdout and stderr are redacted while streaming into separate
8 MiB diagnostic captures; a bounded wait closes a stream retained by a
departed leader's descendant. Unsupported platforms reject owned command
execution.
## Implementation And Tests
- Composition: `internal/app/runner.go`, `internal/app/object_store.go`
- Shared subprocess mechanics: `internal/adapters/subprocess`
- Focused adapters: `internal/adapters/{whisperx,seriatim,audita,scriptorium,notarius,storage,notify}`
- `internal/adapters/whisperx/http_test.go`
- `internal/adapters/seriatim/subprocess_test.go`
- `internal/adapters/audita/subprocess_test.go`
- `internal/adapters/scriptorium/subprocess_test.go`
- `internal/adapters/notarius/subprocess_test.go`
- `internal/adapters/storage/*_test.go`
- `internal/app/runner_test.go`
See the [WhisperX](../integrations/whisperx.md),
[Seriatim](../integrations/seriatim.md), [Audita](../integrations/audita.md),
[Scriptorium](../integrations/scriptorium.md), and
[Notarius](../integrations/notarius.md) contracts before changing an
externally visible boundary. Operator-selected values belong in
[Configuration](../config.md).

264
docs/internal/artifacts.md Normal file
View File

@@ -0,0 +1,264 @@
# Internal: Artifacts
## Purpose
Explain the artifact registry, runtime catalog, resolver, previous-input
requirements, and shared remote current-state mechanics implemented by
`internal/artifacts`. Configuration fields that accept source IDs belong in
[Configuration](../config.md); physical placement belongs in
[Operations](../operations.md).
## Built-in Source IDs
The internal registry recognizes these stable built-in source IDs:
- `narratio.transcript.base`
- `narratio.transcript.polished`
- `narratio.transcript.final`
- `narratio.transcript.final_trimmed`
- `narratio.transcript.final_markdown`
- `narratio.transcript.final_trimmed_markdown`
- `narratio.bounds.session`
Registry entries bind each ID to its producer, output kind, canonical fallback,
and content validator. The focused stage documents own their input/output flow;
[Configuration](../config.md) owns where operators may select these IDs.
## Configured, Extraction, And Previous-Session Sources
- configured source ID format: `narratio.artifact.<artifact_key>`
- extraction source ID format: `narratio.extraction.<output_key>`
- previous-session source ID format: `narratio.previous_session.artifact.<artifact_key>`
All formats are validated by strict source-policy rules. Configured artifact and
extraction keys use `^[a-z][a-z0-9_]*$`; source parsers never normalize an
unrecognized token into a valid source. Extraction sources are registered only
from `pipeline.notarius.outputs`; the Notarius index has no selectable source
ID.
Prepared stable source IDs are `narratio.input.players`,
`narratio.input.party`, `narratio.input.glossary`, and
`narratio.input.spell_catalog`. Artifact policy owns their canonical manifest
kind and prepared filename vocabulary. Canonical party mode preserves the
party source bytes in the party record and supplies the players record from the
deterministic `derived_from_party` projection; both remain ordinary prepared
source IDs for consumers.
## Runtime Catalog
`ArtifactCatalog` tracks:
- `planned`: source registered for run context;
- `executable`: included in the effective analyze artifact set;
- `available`: the source's canonical evidence owner validates its current
manifest record and durable bytes;
- `provenance`: availability source.
Configured definitions are always registered. Without an explicit selection,
the effective analyze set contains enabled definitions. With `--artifacts`, the
exact named configured definitions become the effective set for that invocation,
regardless of their `enabled` value. The effective-set resolver itself does not
expand dependencies; the analyze work planner closes those targets over their
configured prerequisite graph. Availability is separate from executability.
Configuration may normalize a family selection into its concrete generated
members before this resolver runs. The effective set retains optional family
and character origin metadata, but its keys, catalog sources, and runtime
lookups remain concrete configured-artifact identities.
Family publish policies are likewise expanded into ordinary configured-source
publish rules during configuration resolution.
Configured outputs, including non-executable prerequisites, become available
only when the versioned analyze state identifies a current result whose source,
contract, canonical configured path, size, and checksum match a confined
no-follow regular file. An incidental canonical file and a legacy aggregate
analyze output are unavailable.
Extraction entries are registered from configuration and become available only
after compatible extraction evidence is hydrated.
During an analyze invocation, a newly validated and atomically materialized
configured output is marked available with its producer run ID, contract,
checksum, and size. Later scheduled dependents therefore observe the same
semantic identity whether their prerequisite was reused from current manifest
evidence or produced earlier in the invocation.
Current provenance values:
- `generated.current_analyze_run`
- `manifest.current_analyze_artifact`
- `manifest.inputs.previous_cache`
- `current_session.previous_cache`
## Resolution Rules
Built-ins:
1. manifest producer outputs (when present)
2. canonical session-path fallback
Configured sources (`narratio.artifact.*`):
- resolve only through runtime catalog availability;
- use the shared typed analyze-evidence inspection in
`analyze_evidence.go` for prior current-session results;
- require the supported analyze-state and fingerprint versions, a `current`
record for the exact configured key and source ID, a complete contract, the
configured canonical relative path, positive stored size, and stored
checksum matching bytes read from a confined no-follow regular file; and
- treat non-current statuses, legacy or malformed records, removed keys,
unsafe or missing files, and size/checksum mismatches as unavailable without
rewriting manifest state. Catalog construction iterates current
configuration, so removed or renamed records are not advertised.
`narratio.member_artifact.*` is not a runtime source family. Configuration
resolution accepts it only in an artifact-family declaration and rewrites it
to the corresponding configured source before this catalog is built.
Prepared stable sources (`narratio.input.*`):
- resolve only from the current manifest's exact prepared-input record;
- require the policy-owned canonical path below the session root, a confined
non-symlink regular file, a non-empty payload, and a matching SHA-256
checksum; and
- return an immutable source/path/checksum/size identity shared by extract and
analyze rather than falling back to campaign/session source paths.
Extraction sources (`narratio.extraction.*`):
- use the shared typed bundle evidence inspection in `extraction_evidence.go`;
- require a current successful extract record with the exact configured source,
compatible contract and Notarius provenance, a confined regular durable
payload, matching checksum, and the current resolved trimmed-transcript
identity;
- remain unavailable unless catalog hydration receives valid evidence. Resume
treats absent or obsolete evidence as a rerun decision and unsafe evidence as
an error; and
- are never inferred by scanning the Notarius bundle directory.
Previous-session sources (`narratio.previous_session.artifact.*`):
- resolve only from local `previous/` cache state;
- prefer manifest-backed previous-input paths;
- fallback to existing previous-cache filesystem paths.
Source absence is evaluated by the consuming artifact input. An optional input
is omitted from that invocation; a required input fails resolution. This is
separate from a stage's lifecycle outcome.
Validation by content type:
- transcript JSON built-ins: JSON with top-level `segments` array;
- transcript Markdown built-ins: non-empty text file;
- bounds built-in: valid JSON;
- configured/previous-session artifact files: non-empty text file.
## Previous Requirement Collection
`CollectPreviousArtifactRequirements`:
- scans the effective configured artifact set;
- extracts only canonical previous-session sources;
- deduplicates by artifact key;
- merges required and optional references (required wins);
- returns deterministic ordering and source locations.
## Current-State Helpers
Artifacts package owns shared remote current-state loading mechanics used by
restore, status and validation checks, and previous-cache planning.
For a new-protocol current state, the pointer-selected immutable commit is the
complete restore authority. Callers receive its declared object identities and
must not supplement them by listing mutable session prefixes. The legacy reader
is intentionally separate and remains migration-only support.
The reader opens each small control object directly and enforces owner-specific
limits before decoding: 64 KiB for the mutable commit pointer, 4 MiB for the
immutable commit manifest, and 8 MiB for the selected session manifest. Legacy
compatibility applies a 4 KiB limit to `current/run_id.txt` and the same 8 MiB
manifest limit to `current/manifest.json`. These are exposed as
`MaxCurrentCommitPointerBytes`, `MaxRemoteCommitManifestBytes`,
`MaxRemoteSessionManifestBytes`, `MaxLegacyCurrentRunPointerBytes`, and
`MaxLegacyCurrentManifestBytes`.
Each read uses the generation and size metadata returned with its opened body.
Actual bytes remain subject to a limit-plus-one read even if size metadata is
absent or inaccurate. Immutable selections then retain their declared-size,
checksum, generation, and identity checks. No current-state control object is
downloaded through a temporary file.
Core helpers:
- `LoadCurrentState`
- `ValidateCurrentStateIdentity`
- `RemoteCommitManifest` and `CurrentCommitPointer`
Typed missing-state errors:
- `CurrentRunPointerMissingError` (`ErrCurrentRunPointerMissing`)
- `CurrentManifestMissingError` (`ErrCurrentManifestMissing`)
Identity validation supports caller-provided expectations:
- expected campaign;
- expected session ID;
- expected run ID, or pointer/manifest run-ID consistency check.
Caller policy is intentionally outside artifacts helpers:
- some callers fail on missing current state;
- some callers downgrade missing state to status/findings;
- some callers skip optional behavior when state is missing.
## Key Path Helpers
`internal/artifacts/paths.go` and S3-key helpers define canonical helpers for:
- session/work/run paths;
- previous-cache paths;
- spool/cache paths;
- S3 session/run/current-state key layout.
New publication creates run-scoped immutable objects, including
`runs/{run_id}/commit.json` and `runs/{run_id}/session-manifest.json`. The sole
mutable selector is `current/commit-pointer.json`; readers verify its selected
commit and declared object generations/checksums. Legacy current-pair loading
is confined to `current_state_legacy.go` for migration only.
Campaign, session, and Narratio run IDs are validated as portable opaque
segments at configuration and artifact boundaries before they can be used in a
workspace or S3 namespace. Previous-artifact destinations remain typed,
multi-segment relative paths and are confined beneath `previous/artifacts`; they
are not treated as opaque identifiers.
See [Workspace Internals](workspace.md) for how callers consume local helpers
and [Operations](../operations.md#local-state-layout) for the authoritative
physical layout.
## Invariants
- source ID formats are stable contracts;
- artifact resolution is deterministic and manifest-aware;
- extraction sources are available only from a compatible successful manifest
record;
- previous-session source resolution in `analyze` is local-only;
- remote current-state key construction remains centralized in artifacts helpers.
## Implementation And Tests
- Registry and resolution: `internal/artifacts/artifact_resolver.go`,
`internal/artifacts/catalog.go`, `internal/artifacts/transcripts.go`,
`internal/artifacts/extraction_catalog.go`,
`internal/artifacts/extraction_evidence.go`,
`internal/artifacts/extraction_input.go`,
`internal/artifacts/prepared_input.go`
- Current state: `internal/artifacts/current_state.go`,
`internal/artifacts/current_state_commit.go`,
`internal/artifacts/current_state_legacy.go`
- Paths and keys: `internal/artifacts/paths.go`,
`internal/artifacts/s3_keys.go`
- Previous requirements: `internal/artifacts/previous_requirements.go`
- Tests: `internal/artifacts/artifact_resolver_test.go`,
`internal/artifacts/catalog_test.go`,
`internal/artifacts/extraction_catalog_test.go`,
`internal/artifacts/current_state_test.go`,
`internal/artifacts/paths_model_test.go`,
`internal/artifacts/previous_requirements_test.go`

View File

@@ -0,0 +1,110 @@
# Internal: Command Restore
## Purpose
Explain the implemented restore discovery, planning, installation, and
reporting flow in `internal/app`. User invocation belongs in
[CLI](../cli.md#session-restore), and the operator recovery procedure and
physical restore scope belong in
[Operations](../operations.md#restore-workflow).
Restore separates remote authority, local conflict policy, and filesystem
mutation so each remains testable independently.
## Discovery Contract
Discovery delegates current-state pointer and manifest loading to
`internal/artifacts`, then validates the result against the resolved request:
- campaign must match;
- session ID must match.
- run ID must match the pointer-selected committed run.
Restore treats any missing or invalid remote current state as a command error.
## Planning Contract
Restore planner action kinds:
- `download`;
- `skip_same`;
- `conflict`.
Planner behavior:
- a new-protocol restore uses only the selected commit's declared artifact set;
each action carries that artifact's immutable key, checksum, size, and
generation. Coherent legacy state remains on the isolated compatibility path;
- remote-to-local mapping is traversal-safe;
- actions are sorted by local relative path and then remote key;
- force converts differing eligible regular files from conflicts to downloads;
directories and other non-regular targets remain conflicts.
For a non-dry-run restore, planning/classification happens only after acquiring
the session lock. Runner manifest/reuse checks acquire that same lock first.
Previous-cache readiness is resolved through `previouscache.Resolve` for restore,
prepare, status, and validation. A committed source is selected only by its
exact source identity; legacy fallback remains isolated and rejects ambiguity.
## Execution Contract
Execution order and safety:
- non-manifest downloads happen before manifest install;
- `manifest.json` installs last;
- downloads use sibling temp files plus atomic rename;
- manifest replacement is validated before rename;
- each committed object is verified against its declared checksum, size, and
generation before installation;
- a committed manifest already verified during discovery is retained for the
matching restore action and revalidated before installation, avoiding a
second body transfer;
- failed installs do not roll back files already written in the same execution.
- a durable `.restore-incomplete.json` marker is written before installation.
It blocks runners until a restore retry completes all verified installs and
the local manifest replacement, at which point it is removed.
- restored manifest local references are rebased beneath the selected local
session root. Unsafe relative references and producer-machine absolute paths
outside the manifest's producer session root are rejected; producer-local
spool/cache and cleanup locations are not restored as authority.
Audio restore path:
- uses `audio.MaterializeS3Audio`;
- integrates spool and S3 audio cache paths;
- reuses cached audio only when its no-follow regular file, content digest, and
identity sidecar all match the selected remote object version; otherwise it
refreshes through the durable download path.
## Reporting Contract
- dry-run mode prints a summary, performs no durable session writes, and may
read remote current-state or object-identity data to produce that summary;
- execution mode persists the canonical restore report described in
[Operations](../operations.md#restore-workflow);
- report includes plan counts, per-action status, and execution failures.
## Invariants
- restore uses committed remote current state as authority;
- one restore or status inspection observes the single pointer-selected commit
loaded at discovery; later pointer changes cannot add objects or substitute a
different run into its plan;
- a verified `current/commit-pointer.json` and its selected immutable commit
establish new-protocol remote commitment; coherent legacy
`current/run_id.txt` plus `current/manifest.json` remains read-only migration
support;
- restore does not execute pipeline stages.
## Implementation And Tests
- Discovery: `internal/app/restore_discovery.go`
- Planning: `internal/app/restore_plan.go`, `internal/previouscache`
- Execution: `internal/app/restore_execute.go`
- Reporting and command coordination: `internal/app/restore_report.go`,
`internal/app/restore.go`
- Tests: `internal/app/restore_discovery_test.go`,
`internal/app/restore_plan_test.go`,
`internal/app/restore_execution_test.go`,
`internal/app/restore_workflow_test.go`

View File

@@ -0,0 +1,180 @@
# Configuration Internals
User-visible fields, defaults, and selection behavior belong in the
[Configuration Reference](../config.md). This document describes the internal
pipeline-loading boundary implemented by `internal/config`.
## Pipeline Loading
`LoadPipeline` assembles and validates a pipeline in this order:
1. Parse the root YAML into a presence-aware composition tree. The tree retains
source names, full field paths, node kinds, declaration order, and explicit
zero, false, empty-map, and empty-list values.
2. Remove the root-only `composition` envelope and validate its explicit
`imports`, `default_profile`, and named `profiles` declarations. A load
option retains the difference between omitted and explicitly empty profile
selection.
3. Open each import relative to the root pipeline directory through the
confined regular-file boundary. Imports must use a `.yml` or `.yaml`
extension and cannot traverse, use symlinks, repeat a file, import the root,
or contain another composition envelope.
4. Resolve and structurally parse every declared profile overlay through the
same confined regular-file boundary. Missing or malformed unselected
overlays fail the load. Overlays cannot contain a composition envelope.
5. Additively merge the root body and imports. Distinct map leaves compose;
repeated scalar or list paths and node-kind disagreements are conflicts.
6. Select exactly one declared profile from an explicit option or the default,
then recursively merge its overlay. Overlay leaves replace base leaves,
lists are atomic replacements, and null or kind changes fail.
7. Emit deterministic canonical YAML and strictly decode it into
`PipelineConfig`.
8. Apply pipeline defaults once, resolve ordinary relative pipeline paths from
the root pipeline file, and digest the normalized effective mapping.
This ordering preserves monolithic configuration behavior. Moving a field to
an imported fragment changes its source ownership, not its path base, default,
or schema semantics.
## Loaded Context Resolution
`LoadedPipelineCampaign` carries one already composed pipeline and its selected
campaign into session resolution. `LoadSessionWithPipelineCampaignOptions`
loads a local session against that context, while
`ResolveLoadedPipelineCampaign` also accepts an already loaded remote session
or no session while a caller retrieves one. Compatibility loaders route through
these functions after their initial pipeline and campaign reads.
Application commands own pipeline and campaign discovery, campaign-file versus
registry selection, and the corresponding mutual-exclusion rules. Once they
have a `LoadedPipelineCampaign`, local session discovery and remote-session
download retain that exact pipeline object and its private provenance. Removing
a temporary downloaded session file therefore cannot invalidate the resolved
pipeline or campaign context.
The application also has a separate read-only inspection resolver for `config
validate`, `config show`, and `config sources`. It uses the same production root/profile and
campaign selection functions, but never routes through session discovery,
remote-session download, secret loading, adapter composition, workspace
initialization, manifest access, or cleanup. A pipeline with retained artifact
family declarations must resolve its selected campaign before ordinary pipeline
validation, which expands its canonical-party members and generated publish
rules. A pipeline without those declarations may be validated by itself.
`MarshalEffectivePipeline` is the configuration-owned projection for `config
show`. It serializes the typed, defaulted effective mapping through the
deterministic composition renderer, then removes resolution-only artifact
family declarations. The result contains no composition envelope or private
provenance fields and has one trailing newline; commands do not marshal runtime
objects directly.
`EffectivePipelineSources` and `EffectiveCampaignSources` provide the separate
safe provenance projection for `config sources`. Pipeline ownership begins with
the complete logical field paths retained during composition and classifies
each contributor as root, import, profile, or centralized default. The
projection replaces generated concrete member paths with paired family and
canonical-party records, and does the same for generated publish rules.
Campaign records identify campaign-owned fields and party inputs; canonical
derived players point to the party source, while legacy players retain a
dedicated legacy-player role. The application command only joins these sorted
records with selection metadata and never reparses configuration files.
`config diff` uses a paired profile loader that parses the root, imports, and
declared overlays once, then clones the additive base before independently
selecting, decoding, defaulting, and finalizing each profile. When campaign
resolution is needed, the command loads one selected campaign and party and
expands both effective pipelines from that same party value. The configuration
owner projects each normalized effective mapping into sorted logical paths;
mapping leaves are compared individually while sequence values remain atomic.
Values are compact deterministic JSON representations for command output, not
raw YAML fragments, ownership records, or secret material. A differing digest
with no projected difference is treated as an internal consistency error.
Campaign context construction also reads and classifies the campaign-owned
party source through `ParseParty`. A canonical party retains its raw bytes and
normalized roster in runtime-only `ResolvedParty` provenance, while a legacy
party remains opaque. Canonical resolution creates a virtual
`derived_from_party` players input and rejects competing campaign or session
players files and session party overrides. The compact legacy compatibility
path resolves the effective campaign/session party and players files together.
## Canonical Party Domain
`ParseParty` is the package-owned boundary for classifying a party source.
When a top-level `schema_version` is present, it strictly validates the
`narratio.party.v1` contract into ordered character domain values. The
canonical value retains a separate exact byte copy of its source so consumers
can materialize the authored party document without reserializing it. Its
`PlayersYAML` method deterministically derives the versioned players-only
projection.
An unversioned source is classified by the small legacy compatibility boundary
in `party_legacy.go`; it deliberately exposes no parsed roster information.
That boundary exists solely to isolate removable compatibility behavior from
the canonical parser.
## Diagnostics And Runtime Metadata
Syntax, duplicate-key, composition, conflict, and schema failures include the
relevant source name and full field path. Additive conflicts report every
claiming source so operators can repair the split without repeatedly
rediscovering additional conflicts.
The loaded pipeline retains private runtime metadata for the absolute root
path, ordered imports, selected profile name and selection source, selected
overlay, contributing sources, effective digest, and leaf ownership. Base
leaves retain their root/import owners, replaced leaves belong to the selected
overlay, and centrally supplied values use the synthetic `default` owner. This
metadata does not participate in YAML decoding or alter the public
configuration model.
The effective digest is SHA-256 over deterministic canonical YAML produced from
the defaulted `PipelineConfig`. Runtime Notarius paths remain absolute for
execution, but the digest substitutes their normalized logical values captured
before root-relative resolution, so relocating an equivalent configuration
bundle does not change provenance. Because composition and resolution metadata
are private, the digest excludes source layout, profile name, and ownership.
Configuration stores environment variable names rather than resolving raw
credentials, so raw secret values are neither loaded nor hashed.
`recomputePipelineEffectiveDigest` is the single package-owned refresh point
for later runtime expansion.
## Test Surfaces
`composition_test.go` protects the presence and merge algebra independently of
the public schema. `pipeline_composition_test.go` exercises explicit imports,
confinement, conflicts, strict decoding, metadata, and root-relative path
behavior through `LoadPipeline`. `pipeline_profiles_test.go` covers selection,
all-overlay validation, overlay behavior, provenance, option propagation, and
effective-digest stability. Application configuration-loader tests protect the
single-read boundary by changing the pipeline file after its initial load and
confirming local session resolution retains the original pipeline. Other
configuration tests continue to protect defaults and validation after assembly.
`party_test.go` protects the versioned party schema, domain invariants, and
deterministic players projection without involving campaign or runtime wiring.
`party_resolution_test.go` protects campaign-owned party loading, canonical
input restrictions, legacy overrides, source provenance, and virtual players
input selection.
## Artifact Family Resolution
Pipeline loading retains `scriptorium.artifact_families` as a resolution-only
declaration. Once campaign party resolution establishes a canonical roster,
configuration expands families in sorted family-key and character-ID order
into ordinary `ScriptoriumArtifactConfig` values. The expansion owns the narrow
`{character_id}` output substitution, closed member-variable selectors, key and
output collision checks, and the runtime-only family-origin catalog. It then
removes family declarations from `ScriptoriumConfig`, runs ordinary Scriptorium
validation, and refreshes the effective pipeline digest. Stages and adapters
therefore receive only concrete artifact maps.
The catalog retains sorted family member keys plus family/character/source
origins and the typed dependency/publish declarations for their later owners.
`member_dependencies` add corresponding ordinary concrete dependencies, while
the family-only `narratio.member_artifact.<family>` input form is rewritten to
the matching ordinary configured-artifact source. The catalog records those
resolved dependency and input identities with their declaring family and party
member. No member-artifact source is registered as a runtime policy source.
An enabled family publish declaration expands to ordinary configured-artifact
publish rules before the existing publish and lock validators run. Runtime
publication consequently receives no family wildcard or special matcher.

71
docs/internal/fileops.md Normal file
View File

@@ -0,0 +1,71 @@
# Internal: File Operations
`internal/fileops` owns the narrow mechanics for durable replacement of one
byte file. Callers keep ownership of serialization, validation, cancellation,
and destination-directory policy.
## Destination Confinement
Before it creates, replaces, or installs a destination file, `fileops` opens
each ancestor from the filesystem root and rejects symbolic links or components
that change during traversal. The resulting parent-directory handle is retained
for sibling temporary-file creation and rename, so a later pathname swap cannot
redirect the replacement. Existing destination symlinks are replaced as leaf
entries; their targets are never followed.
Remote object acquisition uses a writer supplied by the storage owner. The
writer receives a `fileops`-owned, already-open sibling temporary file rather
than a mutable destination path. Callers still own remote object selection,
validation, conflict handling, and final mode.
Directory promotion keeps the verified destination parent open while it creates
the temporary tree, copies regular source entries, and performs the platform
no-replace rename. Platforms without a verified handle-relative atomic
no-replace primitive reject promotion before writing a temporary tree.
## Cleanup Contract
`RemoveAllUnderRoot` accepts an explicit root and a proper descendant. It opens
the root and each target ancestor without following symlinks, then removes the
tree through those directory handles. It rejects root deletion and any symlink
encountered in the target path or tree; repeated removal of a missing target is
successful. Command and post-publish policy remains owned by `internal/app`.
## Confined Reads
`ReadRegularFileUnderRoot` is the no-follow, bounded read primitive for a
caller-selected root and relative file path; `ReadRegularFile` is its
path-based convenience wrapper. They verify every ancestor through directory
handles and admit only a stable regular-file handle. Callers enforce their own
byte limits and access policy. Credential mode policy and environment
precedence remain owned by `internal/app`.
## Replacement Contract
`ReplaceFileAtomic` requires an existing destination directory. It creates a
sibling temporary file, writes the complete byte sequence, applies the
caller-supplied mode, syncs and closes the file, runs an optional pre-rename
check, replaces the destination with a rename, then syncs the containing
directory.
The pre-rename check is the last point at which a caller can cancel without
installing a new destination. A failure before the rename leaves the old
destination unchanged and removes the temporary file; any cleanup failure is
returned alongside the primary failure. A failure after the rename may leave
the new file visible, but it is not reported as crash-durable.
Replacement follows the operating system's same-filesystem rename semantics.
If a platform cannot replace an existing destination, the operation returns an
error and never removes the old file as an emulation step.
## Directory-Sync Support
Linux and macOS attempt to sync the destination directory. Windows opens the
directory with backup semantics and flushes its buffers. If either operation
is unavailable for the platform, directory handle, or filesystem,
`ErrDirectorySyncUnsupported` is returned. Narratio does not treat that result
as successful crash-durable replacement.
`WriteFileAtomic`, copy helpers, and downloaded temporary-file installation
retain their compatibility behavior of creating the destination parent with
the repository's workspace permissions before using this contract.

316
docs/internal/manifest.md Normal file
View File

@@ -0,0 +1,316 @@
# Internal: Manifest
## Purpose
Explain the session-progress and invocation-audit models implemented by
`internal/manifest`. Physical manifest placement belongs in
[Operations](../operations.md#local-state-layout).
## Session Manifest
`manifest.Manifest` records:
- identity (`session_id`, `campaign`, `run_id`)
- local path metadata (`local_workdir`, `local_spool_dir`)
- remote identity metadata (`s3_bucket`, `s3_session_prefix`, `s3_run_prefix`)
- `inputs` records
- durable `artifacts` records
- per-stage `stages` map
- an optional `post_publish_cleanup` obligation, which binds a committed run,
remote commit identity, and each exact root-confined local target to its
completion evidence
Session, campaign, and run identities in local and downloaded manifests must be
portable opaque segments. Unsafe legacy identities are rejected with migration
guidance rather than being normalized into a different workspace or remote
namespace.
Prepare records independent `party` and `players` input checksums. In canonical
party mode, the party record retains its campaign source identity while the
players record uses `derived_from_party`; raw roster content is never embedded
in manifest metadata. Both records remain the durable authority for consumers
of their prepared input source IDs.
The model admits these stage states:
- `pending`
- `running`
- `succeeded`
- `failed`
- `skipped`
- `stale`
- `interrupted`
### Analyze-owned artifact state
The `analyze` stage record may carry `analyze_state_version: 1` and an
`analyze_artifacts` map keyed by normalized configured artifact key. The
version is the authority marker: version 1 with no entries is a valid evaluated
empty set, while an absent version is legacy aggregate-only state and provides
no current configured-artifact evidence.
Each analyze artifact record has one disposition:
- `current`: the configured artifact is available and carries a versioned
fingerprint plus a complete output record and separate output size;
- `stale`: the recorded semantic identity is no longer current;
- `missing`: no validated current result exists;
- `failed`: the attempted work failed and carries a bounded diagnostic; or
- `unselected`: the artifact was intentionally outside the evaluated set.
Records bind their normalized key and dependencies, fingerprint contract when
evaluated, canonical session-relative output identity when current, producing
Narratio run, update time, and bounded non-secret Scriptorium provenance and
diagnostic paths. A current output includes its configured source ID, contract,
checksum, and positive byte size. Non-current records cannot carry an output,
so an older file is not advertised through stale, missing, failed, or
unselected state.
Family-produced records additionally retain optional `family` and
`character_id` provenance supplied by configuration resolution. These fields
do not replace the concrete configured key or infer family membership from a
name, so older records without them remain valid.
The session-stage collection is the reconciled authority across invocations.
The corresponding collection on an invocation's `analyze` stage record is an
audit of only the artifacts evaluated or attempted by that run. These records
remain analyze-owned data inside the fixed stage; they are not dynamic stages
or generic subtasks.
The stage result contract has one analyze-specific projection boundary. On
success, the runner validates and deep-copies the complete reconciled session
collection and the invocation subset. Aggregate session outputs are rebuilt in
configured-key order from current session records only; invocation outputs are
limited to current records produced by that invocation's run ID. Ordinary
stage outputs cannot accompany this projection, so there is one source of
artifact authority.
Successful incremental execution replaces only evaluated artifact records and
preserves valid unrelated current records. Rebuilt outputs are compared by
bytes and contract: an unchanged identity permits an unselected dependent with
the same recomputed fingerprint to remain current, while a changed identity
removes output authority from every unselected transitive dependent by marking
it stale. A partial analyze invocation can therefore succeed while unrelated
configured records remain stale. Existing canonical files never create current
records without validated execution and projection.
Aggregate analyze status is deliberately coarser than this collection. Resume
validation may skip a succeeded aggregate record when the selected artifact
closure is current even if unrelated records are stale. Conversely, a stale
aggregate record may cross the ordinary runner boundary and perform zero
Scriptorium calls when reconciliation proves every selected artifact current;
the successful projection then restores the aggregate status.
Analyze may return a projection together with an error. That restricted result
cannot carry ordinary outputs, skip state, aggregate logs, generated configs,
or metadata. The runner persists only the validated per-artifact collections,
then marks the aggregate analyze and run state failed and invalidates delivery
dependents conservatively. Unrelated current records survive because the
session projection is complete. A malformed projection is not applied, and a
failed session projection save restores the prior per-artifact authority before
terminal failure persistence.
The incremental executor constructs this restricted projection at each
scheduled artifact boundary. The active record is failed without output,
current transitive dependents are stale, unrelated current records survive, and
only earlier validated and materialized completions remain current in the
invocation subset. Session failure state is persisted before invocation failure
state. If either terminal save fails, its persistence error is joined with the
original adapter, validation, or filesystem cause; a failed projection save
does not turn incidental canonical bytes into manifest authority.
## Run Manifest
`manifest.RunManifest` is created for each invocation and records:
- invocation identity and `force` flag
- the selected profile (when any) and secret-free effective configuration digest
- requested stages
- per-stage action (`run` or `skip`)
- per-stage status
- overall run status (`running`, `succeeded`, `failed`)
## Remote Commit Manifest
`artifacts.RemoteCommitManifest` is a separate, versioned remote snapshot
contract. It is not a serialized session manifest and contains no local
post-publication assertion such as `current_pointer_written`. A remote commit
identifies one campaign, session, and run and declares its immutable artifact
set. Each artifact has a typed source, immutable destination key, SHA-256
checksum, size, and storage generation.
`current/commit-pointer.json` is the sole mutable selector for the new
contract. It identifies exactly one run-scoped `runs/{run_id}/commit.json` and
binds that object by checksum, size, and generation. Readers strictly reject
unknown fields, version mismatches, pointer/commit identity mismatches, and
objects that do not match their declaration.
The reader retains a temporary, clearly isolated compatibility path for a
coherent legacy `current/manifest.json` plus `current/run_id.txt` pair. That
path is removable after migration and is never used to write new state.
## Persistence Semantics
`manifest.LocalStore`:
- validates loaded documents;
- normalizes missing maps/stage records;
- writes through a sibling temporary file, syncing the completed file and
destination directory after atomic replacement;
- updates `updated_at` on save.
If the operating system or filesystem cannot sync a directory, save returns an
explicit error instead of claiming crash-durable replacement. A returned error
after the rename can therefore leave the new manifest visible but not confirmed
durable; callers must reload it before retrying.
## Execution Semantics
The application runner marks an executing stage running and then succeeded or
failed in both manifests, persisting each transition. On success it records
outputs, logs, generated configuration references, metadata, and—when the
stage implements the optional contract—a versioned semantic-configuration
fingerprint. Artifact
records may include optional contract and external provenance objects; old
manifests remain compatible when those fields are absent. A successful forced
rerun marks only succeeded transitive dependent session-stage records stale.
The application owns a fixed dependency relation distinct from execution order;
dependents are returned in canonical order. Render and extract therefore never
stale one another, while either can stale analyze, publish, and notify.
Starting an execution clears the current session-stage record's prior outputs,
logs, generated configuration references, metadata, and semantic fingerprint.
Failed and skipped
transitions enforce the same clearing rule directly, while success repopulates
only fields returned by the new result. Marking a record stale does not clear
those details because resume validation and diagnosis may still require them
before execution begins. Invocation run manifests remain immutable audit
records of their own outcomes.
Aggregate lifecycle clearing deliberately preserves the analyze-owned
per-artifact collection. This lets later reconciliation replace only evaluated
entries without erasing unrelated current results. Other stages retain their
existing aggregate-only lifecycle behavior and are forbidden from carrying the
analyze-specific fields.
A stage may explicitly return a skipped disposition and stable reason. The
runner persists that outcome in both manifests, clears older outputs for the
session-stage record along with older logs, generated configuration references,
and metadata, then applies any bounded details from the current skip and
continues. This self-skip is distinct from deciding not to execute an
already-succeeded stage and is reconsidered on later runs. Skipped results
cannot contain outputs. An intentional self-skip records the current semantic
fingerprint because it is a completed, reusable stage result; failed or
interrupted work never promotes one.
When an already-succeeded stage is skipped, the invocation run manifest records
the `skip` action and reason. The session manifest deliberately retains its
existing succeeded record because it remains the cross-invocation progress
authority. If a stage supplies semantic configuration evidence, reuse first
requires the persisted positive schema version and lowercase SHA-256 digest to
match the current resolved stage semantics. Missing legacy evidence, malformed
evidence, or a mismatch makes the stage and its fixed transitive dependents
stale. The invocation skip copies the matched fingerprint for provenance but
does not rewrite session authority. The existing stage-specific resume
validator runs only after this semantic check succeeds; both checks are
required. Extraction and analyze have resume validators and may reject an
otherwise eligible skip when their selected durable evidence is obsolete; the
runner marks the aggregate record stale and executes it. Analyze's validator
can still accept a partial selection when only unrelated artifact records are
stale.
Implemented reuse coverage is deliberately split between aggregate semantic
evidence and focused durable validators:
| Work | Reuse authority | Focused owners |
| --- | --- | --- |
| prepare | aggregate semantic fingerprint | [prepare](stage-prepare.md) |
| transcribe | aggregate semantic fingerprint | [transcribe](stage-transcribe.md), [WhisperX](../integrations/whisperx.md) |
| merge | aggregate semantic fingerprint | [merge](stage-merge.md), [Seriatim](../integrations/seriatim.md) |
| polish | aggregate semantic fingerprint | [polish](stage-polish.md), [Audita](../integrations/audita.md) |
| normalize | aggregate semantic fingerprint | [normalize](stage-normalize.md), [Seriatim](../integrations/seriatim.md) |
| trim | aggregate semantic fingerprint | [trim](stage-trim.md), [Scriptorium](../integrations/scriptorium.md), [Seriatim](../integrations/seriatim.md) |
| render | aggregate semantic fingerprint | [render](stage-render.md), [Seriatim](../integrations/seriatim.md) |
| extract | aggregate semantic fingerprint plus reference/output validator | [extract](stage-extract.md), [Notarius](../integrations/notarius.md) |
| analyze artifacts | per-artifact fingerprint, reconciliation, and output validator | [analyze](stage-analyze.md), [Scriptorium](../integrations/scriptorium.md) |
| publish | aggregate semantic fingerprint plus immediate lock/commit checks | [publish](stage-publish.md), [storage adapter](adapters.md) |
| notify | aggregate delivery-mode fingerprint | [pipeline overview](overview.md), [configuration](../config.md#notifications) |
These contracts record resolved choices Narratio can observe, not operational
runner tuning. External model, module, prompt, profile, and configuration-file
contents that a tool privately loads remain outside the contract when their
configured identifier is unchanged; operators must force the affected work
after such a private content change.
Session manifest is the authoritative stage-progress ledger across invocations.
Run manifest is invocation-scoped audit state.
Both manifests retain the most recently resolved invocation's bounded
configuration provenance. It identifies the selected profile name and source
(`default` or `cli`) plus the effective configuration digest, but never a raw
secret or profile content. This provenance is informational: it does not
participate in stage resume or cache decisions. A profile change therefore
invalidates only stages whose semantic configuration changed. When a private
external-tool model, module, prompt, or profile changes behind an unchanged
configured identifier, use `--force` for the affected work.
`session plan` computes the same current fingerprint and applies the same
comparison and invalidation rules to a cloned manifest. It predicts the runner
decision without persisting session or invocation state. The shared helper
hashes deterministic JSON from stage-owned typed structs; stage providers must
exclude secrets, complete effective-configuration dumps, and operational
values that cannot affect canonical results. Concrete coverage is owned by the
focused stage and integration documents linked above.
Before an explicitly bounded execution starts after `prepare`, the application
reads the session manifest and accepts only `succeeded` or `skipped` for every
excluded canonical prefix stage. The first other status or absent record fails
the request before layout mutation, adapter initialization, session-manifest
writes, or run-manifest creation. Excluded prefix records are not passed to
resume validators. Records after the selected end are not prerequisites and
may be made stale by selected work without being scheduled.
After a publish commits remotely, any configured local cleanup is first recorded
as a session-manifest obligation before deletion begins. Each target becomes
complete only after its confined deletion (or safe absence check) and a
successful manifest save. An incomplete obligation is retried when publish
executes again and retains the committed run and remote identity that authorized
it; an invocation that does not execute publish does not perform cleanup.
Each invocation derives campaign, session, run, local-path, and remote-prefix
metadata from the validated resolved configuration as one projection. A persisted
session manifest must agree on campaign and session identity before execution;
the current projection is refreshed for every invocation while stage progress,
inputs, and durable artifacts remain session history.
For handled failures after an invocation record is created, the runner records
the failure on the session ledger and persists it before persisting the failed
run audit record. This preserves the resume authority while making a partial
persistence disagreement visible. Abrupt process death remains an accepted case
where a durable running record can require operator interpretation.
## Invariants
- stage resume/skip decisions are session-manifest driven.
- semantic fingerprint comparison precedes stage-specific resume validation.
- only successful and intentional-skipped results promote current semantic
evidence; invocation reuse copies evidence without replacing session state.
- running, failed, and self-skipped stages do not retain result payloads from
an earlier success.
- stale stages retain prior details until replacement execution starts.
- force reruns stale succeeded stages in the fixed dependency relation.
- run manifest does not replace session manifest as progress authority.
- remote commitment is established by a verified current pointer and remote
commit relationship, never by a mutable session-manifest boolean.
## Implementation And Tests
- Models and transitions: `internal/manifest/manifest.go`,
`internal/manifest/run_manifest.go`
- Remote commit model and readers: `internal/artifacts/remote_commit.go`,
`internal/artifacts/current_state_commit.go`,
`internal/artifacts/current_state_legacy.go`
- Persistence and validation: `internal/manifest/store.go`
- Package tests: `internal/manifest/*_test.go`
- Assembled execution behavior: `internal/app/runner_test.go`,
`internal/app/run_stage_test.go`

118
docs/internal/overview.md Normal file
View File

@@ -0,0 +1,118 @@
# Internal Overview
This document is the implemented component map for Narratio. Normative system
boundaries and dependency direction belong in
[Architecture](../policy/architecture.md). User and operator contracts belong
in the [CLI](../cli.md), [Configuration](../config.md),
[Operations](../operations.md), and [Troubleshooting](../troubleshooting.md).
Externally observable tool and format contracts belong under
[Integrations](../integrations/).
## Execution Path
```text
cmd/narratio -> internal/app -> configuration and production composition
-> internal/stage -> adapters and external systems
-> manifests and artifact resolution -> durable local/remote output
```
The executable delegates process behavior to the application boundary. The
application resolves configuration, composes concrete collaborators, acquires
session safety controls, and runs commands. Pipeline commands execute the
canonical stage sequence through adapter interfaces, while manifests record
progress and artifact services resolve durable inputs and outputs.
## Components
| Area | Implemented owners | Responsibility |
| --- | --- | --- |
| Executable | `cmd/narratio` | Process entry, standard stream wiring, argument handoff, and exit status. |
| Application orchestration | `internal/app` | Command dispatch, configuration selection, secret-file environment loading, production composition, session locking, planning, execution, restore, cleanup gates, and user-facing reporting. |
| Configuration | [`internal/config`](configuration.md) | Presence-aware root/import/profile composition, canonical party and family expansion, strict YAML loading, defaults, normalization, session templating, and validation. |
| Pipeline stages | `internal/stage` | Canonical stage registry, shared stage contract, execution dependencies, and implemented stage behavior. |
| External boundaries | `internal/adapters`, `internal/audio` | WhisperX HTTP, downstream subprocesses, notification, object storage, and S3 audio materialization behind Narratio contracts. |
| Manifests | `internal/manifest` | Durable session progress, invocation audit state, stage transitions, validation, and atomic persistence. |
| Artifacts and paths | `internal/artifacts`, `internal/pathsafe` | Artifact identities and resolution, local and remote path/key models, current-state discovery, and confined relative destinations. |
| Previous-session cache | `internal/previouscache` | Deterministic planning and materialization requirements for configured previous-session inputs. |
| Artifact policy | `internal/artifactpolicy` | Source and destination policy, configured artifact identity validation, and publish destination safety. |
| Shared models and file operations | `internal/artifactmodel`, `internal/contracts`, [`internal/fileops`](fileops.md) | Transcript and artifact data contracts plus durable single-file replacement helpers; unsupported directory syncing is reported explicitly. |
| Logging | `internal/logging` | Application logger construction and shared structured logging behavior. |
The application boundary composes concrete implementations. Stages depend on
Narratio-level contracts; external transport and SDK details remain in
adapters. The normative rules for these relationships remain in
[Architecture](../policy/architecture.md).
Pipeline execution and `session plan` share the same inclusive contiguous-range
model. Planning clones session state and applies selected-stage transitions and
resume validation in memory; it does not create invocation state or initialize
stage-execution adapters. Command configuration loading can still retrieve a
missing session file through configured remote storage. It retains the initially
composed pipeline and selected campaign while resolving either a local or
downloaded remote session, so one invocation cannot mix pipeline revisions.
Analyze planning additionally exposes the artifact closure's targets,
prerequisite rebuilds, execution order, and current reuse.
## Pipeline Stage Set
The implemented canonical order is:
1. [`prepare`](stage-prepare.md)
2. [`transcribe`](stage-transcribe.md)
3. [`merge`](stage-merge.md)
4. [`polish`](stage-polish.md)
5. [`normalize`](stage-normalize.md)
6. [`trim`](stage-trim.md)
7. [`render`](stage-render.md)
8. [`extract`](stage-extract.md)
9. [`analyze`](stage-analyze.md)
10. [`publish`](stage-publish.md)
11. `notify` (no-op)
`notify` currently has no persisted pipeline outputs and uses the explicit
`noop` notification mode. Its versioned semantic evidence records that delivery
mode and excludes adapter credentials and response data. The focused stage
documents own implementation mechanics. The
[CLI](../cli.md) and [Operations](../operations.md) own user-visible invocation
and execution semantics.
Execution order and invalidation are separate application contracts. The stage
registry owns the flat execution sequence. The application orchestration owner
uses a fixed, validated dependency relation to find transitive dependents in
canonical order. In particular, `render` and `extract` are sibling consumers of
trimmed transcript state: neither invalidates the other, while either can stale
`analyze`, `publish`, and `notify`.
## Focused Documentation
- [Configuration Internals](configuration.md): pipeline composition, import
confinement, field ownership, decoding, and root-relative path semantics.
- [Adapter Internals](adapters.md): external adapter boundaries, composition,
failure behavior, and test surfaces.
- [Artifact Internals](artifacts.md): source identities, runtime catalog,
resolution, previous requirements, and current-state helpers.
- [Manifest Internals](manifest.md): session and run records, persistence, and
execution transitions.
- [Storage Internals](storage.md): object-store interface and S3 behavior.
- [Workspace Internals](workspace.md): local layout, locking, and cleanup
guardrails.
- [Restore Internals](command-restore.md): discovery, planning, execution, and
reporting.
- [`prepare`](stage-prepare.md)
- [`transcribe`](stage-transcribe.md)
- [`merge`](stage-merge.md)
- [`polish`](stage-polish.md)
- [`normalize`](stage-normalize.md)
- [`trim`](stage-trim.md)
- [`render`](stage-render.md)
- [`extract`](stage-extract.md)
- [`analyze`](stage-analyze.md)
- [`publish`](stage-publish.md)
Use this map to find an owner, then read its focused documentation and tests
before changing behavior.
The stage registry is implemented in `internal/stage/placeholders.go` and its
ordering is protected by `internal/app/planner_test.go`. Cross-invocation skip,
force, failure, and invalidation behavior is exercised in
`internal/app/runner_test.go` and `internal/app/run_stage_test.go`.

View File

@@ -0,0 +1,185 @@
# Stage: analyze
## Purpose
Reconcile configured Scriptorium artifacts, execute only required work in
dependency order, and safely materialize validated outputs.
## Inputs
- ordinary configured artifacts from `pipeline.scriptorium.artifacts`; canonical
party artifact families have already expanded into this map during
configuration resolution, including corresponding member dependencies and
rewritten member-artifact input sources
- optional selected artifact keys supplied through the stage environment
- built-in, configured, extraction, and previous-session source references in
artifact inputs
Supported source families:
- built-ins: `narratio.transcript.*`, `narratio.bounds.session`
- prepared stable inputs: `narratio.input.players`, `narratio.input.party`,
`narratio.input.glossary`, `narratio.input.spell_catalog`
- configured artifacts: `narratio.artifact.<key>`
- extraction lanes: `narratio.extraction.<key>`
- previous-session cache: `narratio.previous_session.artifact.<key>`
## Outputs
- one current per-artifact manifest record per validated materialized output
- stage metadata describing selected/generated/reused artifacts
## Key Behavior
- when `pipeline.scriptorium` is absent or no configured artifact is
executable, completes successfully with no outputs and records explanatory
metadata. This is not an explicit self-skip: both manifests record success,
satisfy publish's prerequisite, and an ordinary later run reuses the result
while the effective set remains empty. Enabling or selecting an artifact
later makes missing versioned evidence non-resumable and schedules it without
requiring force.
- builds a runtime artifact catalog containing built-ins, configured artifacts,
and configured extraction lanes. Extraction availability is hydrated only
from compatible successful extraction evidence.
- uses enabled configured artifacts by default. An explicit `--artifacts`
selection is a one-invocation override that makes exactly the named
configured artifacts explicit targets even when disabled. The work planner
adds required configured prerequisites, reuses current ones, and schedules
stale, missing, or otherwise non-current prerequisites before dependents.
- makes a non-executable configured artifact reusable only when its current
manifest record and durable output pass the configured-artifact evidence
contract; an incidental or stale canonical file is unavailable.
- validates selected artifact dependency order (cycle-safe topo ordering).
- resolves required/optional inputs per artifact source definition into an
ordered semantic identity. Each identity records the configured input name,
canonical source ID, required policy, explicit presence, source contract,
checksum, size, and a source-based logical identity. Workspace paths and
producer run IDs are excluded.
- orders input identities by configured input name independently of Go map
iteration. Runtime adapter paths remain a separate execution-only map.
- omits an unavailable optional input from the adapter request while retaining
explicit absence in its semantic identity; an unavailable required input
fails.
- resolves prepared stable input sources through the shared manifest-authoritative
identity resolver; it does not accept incidental files or fall back to
campaign/session source paths.
- reuses checksums and sizes from validated prepared, extraction, and current
configured-artifact evidence. Other resolved inputs are hashed as confined
regular files with streaming reads and the central resolved-artifact size
limit.
- owns a versioned SHA-256 fingerprint contract with one fixed-field canonical
JSON payload and no map serialization. Configured artifacts are fingerprinted
in deterministic dependency order.
- fingerprints the normalized artifact key, prompt and profile identifiers,
effective render-debug behavior, session-relative output identity, sorted
dependency keys, ordered input declarations and semantic identities,
validated current dependency-output identities, and sorted effective
Scriptorium variables (including Narratio's sticky session variable).
- provides read-only reconciliation that classifies each configured record as
current, stale, missing, failed, legacy, or otherwise non-resumable, and
separately identifies manifest records removed from current configuration.
A record is current only when its fingerprint version and value match and its
configured output still passes manifest-authoritative evidence validation.
- owns a read-only typed work planner. Its explicit targets are enabled
artifacts by default or the exact normalized `--artifacts` selection when
supplied. It closes targets over configured prerequisites, orders the closure
topologically, reuses current members, and schedules every non-current member
before its dependents.
- force applies only to explicit targets. A current prerequisite is reused
unless it is itself an explicit forced target; disabled prerequisites may be
rebuilt when required, while unrelated disabled artifacts are excluded.
- the work plan carries explicit targets, prerequisite-only work, deterministic
execution and reuse lists, invalidated and removed records, and a cloned
projected record collection. Valid unrelated configured records survive the
projection, removed records are omitted, and legacy files never become
current without regeneration.
- implements aggregate resume validation by running the same read-only catalog,
fingerprint reconciliation, and work planner used by execution. A succeeded
aggregate record is reusable exactly when the selected closure schedules no
artifact work; stale unrelated records do not block a partial selection.
- exposes the typed artifact decision to `session plan`. Planning applies it to
a cloned manifest after modeling earlier selected stage transitions, so
aggregate run/skip and artifact execute/reuse decisions match the ordinary
runner without creating durable state or invoking Scriptorium.
- executes only the work plan's scheduled entries. Manifest-validated current
prerequisites remain available through the runtime catalog without invoking
Scriptorium; newly produced prerequisites enter that catalog with the same
contract, checksum, and size identity used for persisted current evidence.
- keeps adapter output in the invocation's run-local analyze directory until
it is a safe, non-empty, bounded regular file with a calculated checksum and
complete output contract. Canonical replacement uses the shared atomic file
operation boundary and verifies that the installed checksum matches the
validated run-local bytes.
- records each successful artifact's freshly computed fingerprint, canonical
relative output path, contract, checksum, size, producer run ID, bounded
Scriptorium provenance, logs, and generated configuration references in the
analyze-owned projection.
- preserves valid unrelated current records during partial execution. If a
rebuilt output's bytes and contract are unchanged, unselected dependents may
remain current. If that semantic identity changes, unselected transitive
dependents become stale without being executed; dependents included in the
invocation are evaluated in dependency order instead.
- reports all evaluated targets and prerequisites in invocation state. The
runner reconstructs aggregate session outputs from every current session
record and invocation outputs from only records produced by the current run.
Unrelated stale records do not make an otherwise successful partial
invocation fail.
- resolves previous-session sources from local `previous/` cache only.
- runs optional render-debug, then artifact execution.
- validates non-empty output files and materializes canonical outputs.
## Failure Semantics
- required missing configured/previous-session inputs fail.
- missing required prepared stable input source includes prepare rerun guidance.
- missing required previous-session source includes prepare rerun guidance.
- missing required `narratio.transcript.final_markdown` or
`narratio.transcript.final_trimmed_markdown` inputs includes render rerun
guidance.
- dependency cycles or unavailable required dependencies fail.
- adapter validation failures fail stage.
- a scheduled artifact failure returns the restricted analyze-state projection
with the active artifact marked `failed`, a bounded error, and no output
authority. Current transitive dependents become stale without execution.
- earlier artifacts from the invocation remain current only after their
run-local output passed validation and canonical materialization. They remain
in invocation history; unattempted later artifacts do not appear there.
- unrelated current records survive a partial failure. Old canonical bytes for
the failed artifact and newly materialized bytes whose projection cannot be
persisted are incidental, not current evidence.
- the runner persists a valid partial projection before it marks aggregate
analyze failed and invalidates publish and notify through the application
dependency relation. Projection-persistence errors retain the last durable
per-artifact authority and are joined with the original failure context.
## Invariants
- `analyze` performs no remote storage calls for previous-session source resolution.
- input-identity resolution is read-only: it does not invoke adapters,
materialize outputs, update status, or create run records.
- fingerprints exclude timeouts, retries, timestamps, producer and Narratio run
IDs, executable and config paths, workspace roots, diagnostic locations, and
executable or private transitive configuration contents. A change that is
visible only inside Scriptorium—such as a file privately loaded by its config
path—requires an explicit forced regeneration.
- output provenance and metadata are deterministic per execution.
- a canonical file without current per-artifact manifest evidence is never
promoted to current state.
## Related Contracts And Tests
- [Configuration](../config.md#scriptorium-artifact-entries) owns artifact
fields and source-selection rules, including
[artifact families](../config.md#scriptorium-artifact-families).
- [CLI](../cli.md) owns user-visible artifact selection.
- [Scriptorium](../integrations/scriptorium.md) owns the subprocess contract.
- Implementation and tests: `internal/stage/analyze.go`,
`internal/stage/analyze_input_identity.go`, `internal/stage/analyze_test.go`,
`internal/stage/analyze_input_identity_test.go`,
`internal/stage/analyze_fingerprint.go`,
`internal/stage/analyze_fingerprint_test.go`,
`internal/stage/analyze_reconciliation.go`, and
`internal/stage/analyze_reconciliation_test.go`,
`internal/stage/analyze_plan.go`, `internal/stage/analyze_plan_test.go`, and
`internal/stage/analyze_incremental_execution_test.go`, and
`internal/stage/analyze_failure_test.go`,
`internal/stage/analyze_resume.go`, and `internal/stage/analyze_resume_test.go`

View File

@@ -0,0 +1,118 @@
# Internal: Extract Stage
## Responsibility
`extract` runs after `render` and before `analyze`. It converts the canonical
`narratio.transcript.final_trimmed` JSON into configured Notarius lane artifacts.
An omitted or disabled Notarius section makes the stage explicitly self-skip
with reason `notarius_disabled`, no outputs, and no Notarius runner.
The external protocol is documented in the
[Notarius integration contract](../integrations/notarius.md). Configuration
fields belong in [Configuration](../config.md), and physical paths and force
procedures belong in [Operations](../operations.md).
## Lifecycle
`internal/stage/extract.go`:
1. resolves the final trimmed transcript from the shared artifact catalog;
2. resolves every configured prepared reference through the shared
manifest-authoritative identity resolver before creating run-local output;
3. streams each verified reference into an invocation-local snapshot and
rejects any source change observed while copying;
4. fingerprints the byte- and provenance-bearing Notarius invocation evidence,
including sorted reference identities;
5. creates a run-local staging directory and invokes the injected
`notarius.Runner`;
6. revalidates the reference snapshots, then validates the v2 successful
receipt, confined index, management documents, configured required lane
descriptors, validation summaries, and regular payload files;
7. atomically promotes the complete bundle to its immutable durable location;
8. records one non-selectable `notarius_index` output and one selectable
`notarius_lane` output per configured lane; and
9. registers each lane as `narratio.extraction.<output_key>` for downstream
Scriptorium and publish resolution.
Lane records retain checksum, contract, producer run ID, and Notarius system,
run, pipeline, and lane provenance. Stage metadata retains the durable bundle
root, receipt, diagnostic paths, rejection/warning summaries, producing
Narratio run ID, the resolved trimmed-input identity, and invocation
fingerprint. The input identity binds the exact transcript bytes, canonical
source ID, producer stage/output/run identity, and resolution provenance.
Reference metadata contains only selector, source ID, canonical session-relative
path, checksum, and size; adapter requests receive selector and absolute
invocation-local snapshot path, never payload contents. Snapshot bytes must
match the prepared identity both before and after Notarius runs, so a concurrent
prepared-file replacement cannot make recorded provenance describe different
bytes from those supplied to Notarius.
Validation completes before
promotion, so a rejected result cannot expose a partial durable bundle.
Any executed extraction outcome that replaces a different effective outcome
marks succeeded analysis and delivery dependents stale. Render is an independent
sibling and remains current. Repeating the same disabled self-skip with no
outputs is stable and does not repeatedly invalidate dependent stages.
## Resume Validation
Before the focused validator runs, the application compares extract's versioned
semantic fingerprint. It covers enablement, Notarius pipeline identity, sorted
reference selector/source mappings, sorted declared output contracts, and each
canonical `narratio.extraction.<key>` output identity. It excludes executable,
timeout, working directory, config path, and private Notarius config contents.
`internal/stage/extract_resume.go` then permits a skip only when the existing
stage record still matches the current byte- and provenance-bearing invocation
evidence. That evidence covers the current direct trimmed-transcript identity,
sorted prepared-reference identities, pipeline identity, and configured output
contracts. The same reference helper and transcript identity are resolved again
for artifact evidence, so changing current transcript bytes, reference bytes,
or producer identity makes the prior extraction obsolete. Operational runner
settings do not invalidate otherwise current durable evidence.
A valid prepared-reference change makes extraction non-resumable. Missing,
unsafe, or checksum-inconsistent prepared evidence is a hard validation error
with prepare-force guidance because an immediate extract rerun cannot succeed.
The validator then checks the producing run identity, canonical immutable
bundle root, path confinement and absence of symlink components, receipt
identity, exactly one canonical index, the exact configured source set,
contracts and provenance, regular-file status, and stored checksums. Missing or
obsolete results are non-resumable and run again; unsafe filesystem conditions
return an error rather than silently accepting or replacing data.
Neither contract can observe files imported by Notarius configuration, profile
contents, prompt/module definitions, or other transitive inputs. Operators must
force extraction after changing any such private input behind a stable
identifier.
## Failure Behavior
Adapter startup, timeout, nonzero exit, receipt decoding, path confinement,
index compatibility, inconsistent warning or diagnostic envelopes,
required-lane rejection or incomplete validation, payload inspection,
checksum, or promotion errors fail the stage through ordinary manifest
transition handling.
Stdout receipt and stderr diagnostics remain separate. Downstream stages are
not given selectable extraction sources unless the complete configured result
has passed validation and promotion.
When a replacement attempt begins, the current session-stage record no longer
advertises payload from the previous success. A failed replacement therefore
has no current outputs, logs, generated configuration references, or metadata,
while the earlier invocation manifest and immutable promoted bundle remain
available for audit and recovery.
## Implementation And Focused Tests
- Stage execution, selection, and resume validation: `internal/stage/extract.go`,
`internal/stage/extract_resume.go`,
`internal/stage/extract_test.go`,
`internal/stage/semantic_contracts_delivery.go`
- Subprocess boundary: `internal/adapters/notarius/subprocess.go`,
`internal/adapters/notarius/subprocess_test.go`
- Catalog hydration: `internal/artifacts/extraction_catalog.go`,
`internal/artifacts/extraction_catalog_test.go`
- Composition and downstream behavior: `internal/app/runner_test.go`,
`internal/stage/analyze_test.go`, `internal/stage/publish_test.go`

View File

@@ -0,0 +1,51 @@
# Stage: merge
## Purpose
Normalize raw transcript inputs and merge into base transcript via Seriatim.
## Inputs
- `transcripts/raw/*.json`
- `inputs/speakers.yml`
- `inputs/autocorrect.yml`
## Outputs
- `transcripts/base.json`
- optional `artifacts/seriatim.report.json`
## Key Behavior
- discovers and validates raw transcript inputs.
- normalizes each raw transcript (`seriatim.Normalize`) into run-local scratch output.
- merges normalized inputs (`seriatim.Run`) into base transcript.
- validates merged transcript and optional report JSON.
- materializes canonical outputs and records stage logs/generated configs.
## Invariants
- merge always consumes normalized forms of raw inputs.
- base transcript must validate before stage success.
- report output is config-gated.
## Resume Evidence
Merge records a versioned semantic-configuration fingerprint for the Seriatim
merge operation, output schema, coalesce gap, and every configured advanced
merge transformation. A change reruns merge and stales only its fixed
descendants; prepare and transcribe remain reusable. Binary path, timeout,
report emission, logs, and diagnostic retention are operational exclusions.
Configuration or resources loaded privately inside Seriatim are outside
Narratio's observable contract and require `--force` when changed. An existing
successful merge record without evidence reruns once when selected.
## Related Contracts And Tests
- [Seriatim](../integrations/seriatim.md) owns subprocess and output semantics.
- [Configuration](../config.md#pipeline) owns operator-selected Seriatim values.
- Implementation and tests: `internal/stage/merge.go`,
`internal/stage/merge_test.go`,
`internal/stage/semantic_contracts_initial.go`, and
`internal/stage/semantic_contracts_initial_test.go`

View File

@@ -0,0 +1,42 @@
# Stage: normalize
## Purpose
Normalize polished transcript into final transcript using Seriatim.
## Inputs
- `transcripts/polished.json`
## Outputs
- `transcripts/final.json` (or configured normalize output path)
- optional `artifacts/seriatim.normalize.report.json`
## Key Behavior
- resolves polished transcript from manifest outputs/canonical fallback.
- applies `pipeline.normalize` config or default normalize config.
- runs Seriatim normalize with configured timeout/binary.
- validates normalized transcript and optional report.
- materializes canonical outputs and records logs/generated configs.
## Invariants
- final transcript must validate as processed transcript JSON (`segments` array).
- normalize defaults are applied when `pipeline.normalize` is unset.
## Resume Semantics
The versioned semantic fingerprint covers the Seriatim normalize operation,
output schema and canonical output identity, plus the configured transcript
transformations. Seriatim's executable and timeout and optional report
generation are operational and do not invalidate the normalized transcript.
## Related Contracts And Tests
- [Seriatim](../integrations/seriatim.md) owns subprocess and output semantics.
- [Configuration](../config.md#pipeline) owns normalize fields and defaults.
- Implementation and tests: `internal/stage/normalize.go`,
`internal/stage/normalize_test.go`,
`internal/stage/semantic_contracts_refinement.go`

View File

@@ -0,0 +1,53 @@
# Stage: polish
## Purpose
Run Audita polishing on base transcript and produce polished transcript.
## Inputs
- `transcripts/base.json`
- `inputs/glossary.yml`
## Outputs
- `transcripts/polished.json`
- optional `artifacts/audita.report.json`
## Key Behavior
- resolves base transcript from merge outputs/canonical fallback.
- invokes an Audita runner configured with static model/runtime options; the
invocation supplies paths and modules.
- validates processed transcript structure (`segments` array required).
- validates optional report JSON.
- materializes canonical outputs; records logs/generated config and adapter metadata.
## Invariants
- polished transcript schema validation is mandatory.
- report output is config-gated.
## Resume Semantics
The versioned semantic fingerprint covers the Audita service endpoint, model,
validation model, module set, transcript description, output schema, selected
external configuration path, and canonical polished-transcript identity. Module
ordering is normalized because the configured modules form a set. Audita's
executable, timeouts, concurrency, report and debug behavior, work retention,
and credential environment name are operational and do not invalidate a
successful result.
Narratio can fingerprint a selected model, module, or configuration identifier,
but it cannot inspect content that Audita privately resolves behind that stable
identifier. Force `polish` after changing such private content without changing
its identifier.
## Related Contracts And Tests
- [Audita](../integrations/audita.md) owns subprocess, validation, and failure
semantics.
- [Configuration](../config.md#pipeline) owns operator-selected Audita values.
- Implementation and tests: `internal/stage/polish.go`,
`internal/stage/polish_test.go`,
`internal/stage/semantic_contracts_refinement.go`

View File

@@ -0,0 +1,98 @@
# Stage: prepare
## Purpose
Materialize canonical current-session inputs before processing stages.
## Inputs
- resolved campaign, session, and pipeline configuration
- stable input files (`speakers`, `autocorrect`, `glossary`, `players`, `party`)
- optional spell-catalog overlay
- one resolved local or S3 audio source
- enabled configured artifact input requirements for previous-session sources
## Outputs
- `inputs/campaign.yml`
- `inputs/session.yml`
- `inputs/pipeline.resolved.yml`
- `inputs/speakers.yml`
- `inputs/autocorrect.yml`
- `inputs/glossary.yml`
- `inputs/players.yml`
- `inputs/party.yml`
- optional `inputs/spell_catalog.json`
- `audio/*.flac`
- optional `previous/manifest.json`
- optional `previous/artifacts/**`
- deterministic `manifest.inputs` entries (checksums + provenance)
## Key Behavior
- validates required config/store state.
- enforces local audio vs S3 audio mutual exclusivity.
- rejects duplicate explicit local audio sources after resolution.
- gives distinct local source paths with the same basename deterministic unique
prepared filenames so neither source is overwritten.
- materializes S3 audio through spool/cache-aware logic.
- materializes a configured spell catalog with checksum and provenance, or
safely removes an obsolete canonical spell catalog and its manifest record
when the effective input is omitted.
- in canonical party mode, copies the validated raw party bytes unchanged and
deterministically generates the prepared players projection; legacy mode
continues to copy its opaque party and explicit players sources.
- scans enabled configured artifact inputs for `narratio.previous_session.artifact.*` requirements.
- clears managed `previous/` state on every invocation, then, when requirements exist:
- resolves the pointer-selected previous source through the shared resolver;
- downloads previous manifest/artifacts;
- records previous inputs in `manifest.inputs`.
Required previous-session inputs fail when unavailable; optional missing inputs
are typed skipped results. Committed sources use their exact source-to-destination
mapping, while the isolated legacy reader rejects ambiguous fallback matches.
## Invariants
- only `prepare` hydrates canonical `previous/` cache state.
- managed previous artifacts are stored under `previous/artifacts/**` without
duplicate `artifacts/artifacts/` nesting.
- managed `previous/` state represents only the current requirement set.
- `manifest.inputs` ordering is deterministic (`kind`, `path`).
## Resume Evidence
Prepare records a versioned semantic-configuration fingerprint for the
resolved campaign/session selection, local-versus-S3 audio mode and canonical
audio names, stable-input ownership/presence, previous-session identity, and
the party mode plus canonical players projection version, and the effective
previous-artifact requirement set. A change reruns prepare and
stales its fixed descendants. Existing successful records without this
evidence rerun once when selected.
Workspace, spool, and cache placement and absolute source relocation are not
semantic when logical selection, canonical names, and bytes are equivalent.
The fingerprint deliberately does not read or rehash large audio. Prepared
input checksums remain the content provenance. Before reusing success, prepare
validates every durable prepared copy and compares current stable-input bytes,
canonical party and derived-player bytes, local audio membership/checksums, or
S3 key/size/entity-tag identity with that provenance. Source relocation with
equivalent names and bytes remains reusable; changed or unavailable evidence
causes a normal prepare rerun.
## Related Contracts And Tests
- [Configuration](../config.md) owns audio selection, stable input fields, and
previous-session settings.
- [Operations](../operations.md) owns physical input, audio, spool, cache, and
previous-state layout.
- [Storage Internals](storage.md) and [Artifact Internals](artifacts.md) explain
the internal collaborators.
- Implementation and tests: `internal/stage/prepare.go`,
`internal/stage/prepare_test.go`,
`internal/stage/prepare_resume.go`,
`internal/stage/prepare_resume_test.go`,
`internal/stage/semantic_contracts_initial.go`,
`internal/stage/semantic_contracts_initial_test.go`,
`internal/audio/s3_audio_test.go`,
`internal/previouscache/*_test.go`

View File

@@ -0,0 +1,117 @@
# Stage: publish
## Purpose
Upload run/session outputs to object storage and atomically advance remote current state.
## Inputs
- successful preceding stages from the [canonical stage set](overview.md#pipeline-stage-set)
- invocation-scoped run files
- resolved publish output rules
- effective publish locks (static + remote merged lock set), revalidated at the
remote commit boundary
- durable previous-session cache files when present
## Outputs
- uploaded invocation record and selected publish outputs;
- uploaded durable previous-session cache files when present;
- immutable run-scoped commit manifest; and
- current commit pointer, written last.
Exact remote placement and the operator workflow belong in
[Operations](../operations.md#publish-workflow).
## Key Behavior
- when publishing or run upload is disabled, completes successfully with no
outputs and records explanatory metadata. This is not an explicit self-skip:
both manifests record success. Enablement and upload policy are fingerprinted,
so changing either automatically makes the prior result non-resumable.
- validates prerequisite stage success and object-store availability.
- derives a deterministic run-archive allowlist from the validated run
`manifest.json`: declared run-local outputs, logs, generated configs, and the
manifest itself. Unlisted workspace files are not archive candidates.
- opens each archive candidate beneath its archive root without following
symlinked ancestors or leaf entries, verifies that it is a regular file and
checks a declared checksum when present, then streams the opened descriptor.
- derives the durable previous-cache archive from its validated manifest using
the same confinement and regular-file checks.
- resolves publish output sources through runtime artifact catalog and
manifest-aware resolution. Configured Scriptorium outputs are publishable
only from validated `current` per-artifact analyze evidence; an incidental
canonical file, legacy aggregate output, stale/failed/unselected record, or
mismatched path, size, or checksum remains unavailable. This does not change
the explicit compatibility policies owned by built-in, extraction, or
previous-session sources.
- publishes extraction lanes only through explicit configured output rules;
neither run-local nor durable Notarius bundles are scanned or uploaded wholesale.
- selected artifact filter applies to configured artifact sources only.
- locked outputs are skipped intentionally (including required ones).
- optional missing outputs are skipped; required missing unlocked outputs fail.
- creates one complete immutable source-to-destination mapping before upload;
- uploads and verifies every declared immutable object and the commit manifest;
- updates `current/commit-pointer.json` exactly once, last; and
- does not write the legacy `current/manifest.json` or `current/run_id.txt` pair.
- rechecks remote lock state immediately before the pointer update. A newly
committed lock aborts selection, leaving any uploaded immutable attempt
unselected.
- reads the mutable remote lock document through a direct limit-plus-one read
capped by `MaxRemoteLockStoreBytes` (1 MiB), retaining the generation returned
with the opened body for conditional updates. Oversized lock documents fail
before YAML decoding; published artifact payloads do not use this limit.
## Metadata Signals
Includes counts/lists for:
- run uploads
- published output uploads
- previous uploads
- skipped optional outputs
- skipped unselected outputs
- locked outputs
- remote commit and current-pointer key paths
- the run identifier selected by the commit
## Invariants
- `current/commit-pointer.json` is the remote commit marker and is written last.
- run files, selected outputs, previous-cache files, and the committed session
manifest are all declared by an immutable commit under the run prefix.
- run and previous uploads contain only manifest-declared regular files opened
from verified descriptors; symlinks, special files, replacement races, and
undeclared entries are rejected or ignored before uploads begin.
- run-local diagnostics, including Notarius receipt and stderr files, are
archived only when recorded by the run manifest.
- publish locks are not overridden by `--force`; remote locks are revalidated
immediately before current-state selection.
- post-commit local cleanup is authorized by the committed publish metadata and
is durably recorded by the application lifecycle before any local deletion.
## Resume Semantics
The versioned semantic fingerprint covers enabled behavior, run-upload policy,
normalized source/destination/required output rules, static lock policy, and
the remote backend, bucket, region, endpoint, and root-prefix identity. Rule
and lock ordering is canonicalized. Credential environment names,
path-addressing transport mode, local workspace placement, and run identifiers
are excluded. Remote locks remain mutable state and are still revalidated at
the commit boundary; semantic evidence does not replace that safety check.
The commit boundary and cleanup gate are normative architecture invariants; see
[Architecture](../policy/architecture.md#publish-commit-boundary).
## Related Contracts And Tests
- [Configuration](../config.md#publish-configuration-summary) owns output and
static-lock fields.
- [Operations](../operations.md#publish-locks) owns remote lock lifecycle and
physical remote state.
- [Artifact Internals](artifacts.md) explains source resolution and current-state
helpers.
- Implementation and tests: `internal/stage/publish.go`,
`internal/stage/publish_test.go`,
`internal/stage/semantic_contracts_delivery.go`,
`internal/app/operator_helpers_test.go`, and
`internal/app/post_publish_cleanup_test.go`

View File

@@ -0,0 +1,60 @@
# Stage: render
## Purpose
Render Markdown transcript artifacts from normalized JSON transcripts via Seriatim.
It runs after `trim` and before `extract` in the canonical sequence. Render and
extract are independent sibling consumers: replacing render output does not
invalidate extraction, but it does invalidate succeeded analysis and delivery
records that may consume rendered transcripts.
## Inputs
- `narratio.transcript.final` (`transcripts/final.json`)
- `narratio.transcript.final_trimmed` (`transcripts/final.trimmed.json`)
## Outputs
- `narratio.transcript.final_markdown` -> `transcripts/final.md`
- `narratio.transcript.final_trimmed_markdown` -> `transcripts/final.trimmed.md`
## Key Behavior
- uses `pipeline.render` settings (enabled/format/title/booleans).
- resolves inputs manifest-first, then canonical fallback.
- writes run-local outputs first, then materializes canonical session outputs.
- records input provenance, output paths, adapter metadata, logs, and generated config refs.
- when `pipeline.render.enabled=false`, completes successfully with no outputs
and records explanatory metadata. This is not an explicit self-skip: both
manifests record success. Because enablement is fingerprinted, enabling
render later automatically makes the prior result non-resumable.
## Failure Semantics
- missing normalized input fails with normalize rerun guidance.
- missing trimmed input fails with trim rerun guidance.
- adapter/subprocess failure fails stage.
- empty render output files fail validation.
## Invariants
- only `format: markdown` is supported.
- render stage owns production of built-in Markdown transcript sources.
## Resume Semantics
The versioned semantic fingerprint covers enablement, final format, resolved
title (including the session-title fallback), timestamp, segment-ID and
metadata inclusion, both canonical input identities, and both Markdown output
identities. Seriatim's executable, timeout, and report behavior are operational
and do not invalidate rendered transcripts. A render-only change leaves the
independent `extract` sibling reusable while invalidating their shared
downstream consumers.
## Related Contracts And Tests
- [Seriatim](../integrations/seriatim.md) owns render subprocess behavior.
- [Configuration](../config.md#pipeline) owns render fields and defaults.
- Implementation and tests: `internal/stage/render.go`,
`internal/stage/render_test.go`,
`internal/stage/semantic_contracts_refinement.go`

View File

@@ -0,0 +1,56 @@
# Stage: transcribe
## Purpose
Generate raw per-speaker transcripts from prepared audio using WhisperX.
## Inputs
- `audio/*.flac` from `prepare`
## Outputs
- `transcripts/raw/<speaker>.json`
## Key Behavior
- discovers prepared audio from manifest inputs or canonical audio directory.
- derives the transcript identity from the prepared `.flac` filename.
- dispatches WhisperX requests through a bounded worker pool.
- validates each output as JSON.
- writes run-local outputs then materializes canonical transcript outputs only
after every planned request succeeds.
## Invariants
- prepared audio identities must be unique; prepare disambiguates distinct
source paths that share a basename.
- output path returned by adapter must match requested output path.
- an empty adapter result path means the requested path; adapters cannot select
an alternate destination.
- each successful output is validated before stage success, and cancellation or
incomplete dispatch cannot be reported as a successful result.
## Resume Evidence
Transcribe records a versioned semantic-configuration fingerprint containing
the Narratio-visible WhisperX service URL and recognition language. Changes to
either rerun transcription and stale its fixed descendants while leaving
prepare reusable. Retry count/delay, concurrency, timeout, credentials, and
diagnostic locations are operational and do not change this evidence.
WhisperX models or private service configuration not exposed by Narratio's
adapter contract cannot be fingerprinted; use `--force` after changing them.
An existing successful transcribe record without evidence reruns once when
selected.
## Related Contracts And Tests
- [WhisperX](../integrations/whisperx.md) owns HTTP, retry, timeout, and
cancellation semantics.
- [Configuration](../config.md#pipeline) owns concurrency and other
operator-selected values.
- Implementation and tests: `internal/stage/transcribe.go`,
`internal/stage/transcribe_test.go`,
`internal/stage/semantic_contracts_initial.go`, and
`internal/stage/semantic_contracts_initial_test.go`

View File

@@ -0,0 +1,57 @@
# Stage: trim
## Purpose
Produce a final-trimmed transcript. By default, the stage generates bounds and
applies a bounds-driven trim.
## Inputs
- `transcripts/final.json`
## Outputs
- `transcripts/final.trimmed.json` (or configured trim output path)
- when trim enabled: `artifacts/session_bounds.json`
## Key Behavior
When `trim.enabled=true`:
- runs Scriptorium bounds artifact generation;
- optionally runs render-debug output generation;
- validates bounds payload against transcript;
- derives keep selector;
- either copies unchanged transcript or runs Seriatim trim;
- validates trimmed transcript and materializes bounds output.
When `trim.enabled=false`:
- copies normalized transcript to trimmed output.
## Invariants
- normalized transcript is required input.
- bounds output exists only in enabled trim path.
- render-debug output is diagnostic and not a declared stage output.
## Resume Semantics
The versioned semantic fingerprint covers enablement, the bounds prompt and
profile identifiers, the Scriptorium configuration identity, transcript input
name, sticky session variable, bounds and trimmed output identities, and the
Seriatim trim operation. Diagnostic bounds rendering, diagnostic output paths,
timeouts, executable paths, and optional reports are operational and do not
invalidate the canonical trimmed transcript.
Narratio cannot inspect prompt, profile, or configuration content that
Scriptorium or Seriatim privately resolves behind a stable identifier. Force
`trim` after changing such private content without changing its identifier.
## Related Contracts And Tests
- [Scriptorium](../integrations/scriptorium.md) owns bounds generation and
debug-render subprocess behavior.
- [Seriatim](../integrations/seriatim.md) owns transcript trimming behavior.
- [Configuration](../config.md#pipeline) owns trim fields and defaults.
- Implementation and tests: `internal/stage/trim.go`,
`internal/stage/trim_test.go`,
`internal/stage/semantic_contracts_refinement.go`

64
docs/internal/storage.md Normal file
View File

@@ -0,0 +1,64 @@
# Internal: Storage
## Purpose
Explain the object-store interface and S3 implementation used by Narratio.
Remote key layout and lifecycle belong in [Operations](../operations.md), while
operator-selected storage fields and credential mechanisms belong in
[Configuration](../config.md).
## Primary Contract
`storage.ObjectStore` interface:
- `List(ctx, prefix)`
- `Read(ctx, key)` returns an object body and the generation observed with it
- `Download(ctx, key, localPath)`
- `Upload(ctx, localPath, key, opts)`
- `UploadConditional(ctx, source, key, opts, condition)`
- `Exists(ctx, key)`
Key invariant:
- callers pass full bucket-relative keys;
- storage implementations do not infer campaign/session/run prefixes.
`ReadObjectBounded` is the shared mechanism for small control objects. It opens
one object version, returns the metadata observed with that body, rejects an
oversized known size before transfer, and still performs a context-aware
limit-plus-one read. It closes the body on every exit. Callers own the policy
limit and add the control-object category to errors; this helper is not used for
large artifact payloads.
## Composition
`NewObjectStoreFromConfig` constructs the S3-backed implementation from
resolved configuration. The application loads configured filesystem secrets
before calling it. The storage adapter consumes already-resolved values; it does
not own discovery, defaults, or configuration validation.
## S3 Backend Behavior
- normalizes object keys.
- `List` paginates and returns normalized `ObjectInfo`.
- A truncated S3 listing must supply a new, non-empty continuation token;
otherwise listing fails with bucket and prefix context instead of looping.
- `Download` writes local files with parent directory creation.
- `Upload` streams local file and returns remote metadata.
- `Read` binds a returned body to its S3 ETag. `UploadConditional` maps an ETag
match or absence precondition directly to the provider request and reports a
failed precondition without performing a local check-then-write replacement.
- `Exists` maps not-found responses to `false`.
## Invariants
- storage layer is stateless regarding manifest/stage progression.
- bounded reads never retain more than the caller's limit plus one byte and do
not replace owner-specific size policy.
- publish ordering semantics are owned by stage/app code, not storage adapters.
## Implementation And Tests
- Contract and S3 adapter: `internal/adapters/storage`
- Composition: `internal/app/object_store.go`
- Tests: `internal/adapters/storage/*_test.go`,
`internal/app/object_store_test.go`

View File

@@ -0,0 +1,94 @@
# Internal: Workspace
## Purpose
Explain the helpers that construct local session and run paths, coordinate
single-writer access, and confine cleanup. The authoritative physical layout and
retention workflow belong in [Operations](../operations.md#local-state-layout).
## Path Ownership
`internal/artifacts` owns canonical session, run, spool, cache, and
previous-cache path construction. `SessionPathsFor` provides the session-scoped
path model, and layout creation goes through `EnsureLayoutFor`. Callers should
consume those helpers instead of rebuilding relative paths.
`internal/pathsafe` validates relative destinations. `internal/fileops` opens
cleanup roots and their descendants through no-follow directory handles before
removing them.
`internal/fileops` owns the ordinary workspace mode contract. On POSIX,
`WorkspaceDirectoryMode` is setgid `02775` and `WorkspaceFileMode` is `0664`.
`EnsureWorkspaceDirectory` reapplies the directory mode after creation so a
restrictive umask cannot remove group access, while retaining existing ownership
and group. Credential paths are outside this contract; the platform-specific
operational requirements are in [Operations](../operations.md#workspace-permissions).
## Run-Local Stage Layout
`internal/stage/run_local.go` maps stage outputs and diagnostics into an
invocation-scoped layout. Successful outputs are validated and atomically
materialized into canonical session paths before stage success. Managed
previous-session cache paths remain session-durable and are never redirected
into run-local output space.
Extraction uses run-local receipt, stderr, and output-root helpers, then
promotes the validated external bundle to the unique immutable Notarius bundle
path supplied by `internal/artifacts`. `internal/fileops.PromoteDirectory`
copies only regular files and directories to a same-filesystem temporary
sibling. Source traversal uses confined directory handles and identity checks
so replacing an inspected root, directory, or file is rejected rather than
followed. The completed tree is atomically renamed without replacing an
existing destination. Exact physical paths belong in
[Operations](../operations.md#extraction-workflow).
## Locking
`artifacts.LocalStore` enforces the single-writer session lock via an
operating-system lock held on `.lock` (`ErrLockConflict` on contention). The
file retains owner metadata after release or process death; its existence is
not evidence that a lock is active. Command and restore flows wait for this
lock only while their context remains active, and report a release failure.
## Cleanup Semantics
Automatic post-publish cleanup:
- is created only after a successful publish commit with complete publish
metadata, then is persisted before any deletion;
- requires `uploaded=true`, a remote commit key, and a current commit-pointer
key in publish metadata;
- consumes the resolved cleanup policy described in
[Configuration](../config.md);
- refuses unsafe deletes (root delete, out-of-root delete, and symlinked
ancestors or entries);
- retries any recorded incomplete target on later invocations even when no
publish work is selected. Missing targets are a successful, idempotent
cleanup result only after the completion evidence is saved.
Manual cleanup uses the same root-confined deletion mechanism. Invocation
syntax and exact deletion scope belong in [CLI](../cli.md#clean) and
[Operations](../operations.md#cleanup).
## Invariants
- campaign-aware session root is mandatory.
- manifest-driven stage state is durable across runs.
- cleanup guardrails prevent destructive root/out-of-scope deletion.
- ordinary workspace paths retain group-writable directory and file modes across
nested creation, replacement, and Notarius promotion.
## Implementation And Tests
- Path model and local store: `internal/artifacts/paths.go`,
`internal/artifacts/local.go`
- Run-local materialization: `internal/stage/run_local.go`
- Immutable bundle promotion: `internal/fileops/directory.go`
- Workspace modes: `internal/fileops/modes.go`
- Cleanup confinement: `internal/fileops/cleanup.go`,
`internal/app/cleanup_targets.go`, `internal/app/post_publish_cleanup.go`
- Tests: `internal/artifacts/paths_model_test.go`,
`internal/artifacts/local_test.go`, `internal/stage/run_local_test.go`,
`internal/fileops/directory_test.go`, `internal/fileops/modes_posix_test.go`,
`internal/fileops/cleanup_test.go`, `internal/app/cleanup_targets_test.go`,
`internal/app/post_publish_cleanup_test.go`

574
docs/operations.md Normal file
View File

@@ -0,0 +1,574 @@
# Operations Guide
Operator workflow for running, recovering, and publishing Narratio sessions.
For command syntax, see [docs/cli.md](./cli.md). For field-level config, see [docs/config.md](./config.md).
## Campaign and Session Selection
Campaign selection priority:
- `--campaign-file`
- `--campaign`
- `pipeline.campaigns.default_campaign_id`
Session source priority:
- `--session`
- local default search paths
- remote session object (S3) when local session file is not found and storage is configured
## Session Initialization
Use `session init` to generate a concrete session file for local or remote use.
Local file:
```bash
narratio session init 2026-04-04 --output ./session.yml --date 2026-04-04 --title "Session 12"
```
Remote session object:
```bash
narratio session init 2026-04-04 --remote --force
```
If `campaign.yml` sets `session_template_file`, `session init` renders it. Template variables must resolve to concrete values.
Campaigns must provide stable input files for speakers, autocorrect, glossary,
players, and party, and may provide an optional spell-catalog overlay. Session
files may override those paths for one session. The `prepare` stage
materializes them under `inputs/`; configured consumers use the prepared files,
never the original campaign or session source paths. Field definitions and
source IDs are in [Configuration](./config.md#notarius-reference-bindings).
## Standard Session Workflow
1. Select pipeline/campaign/session config.
2. Validate session readiness:
```bash
narratio session validate 2026-04-04
```
3. (Optional) inspect stage decisions:
```bash
narratio session plan 2026-04-04
```
4. Run the pipeline:
```bash
narratio run 2026-04-04
```
5. Check state:
```bash
narratio session status 2026-04-04
```
Run, plan, and status output identify the resolved pipeline profile (or `none`)
and effective configuration digest. Status distinguishes the current resolved
value from the last value persisted in the session manifest, which helps
diagnose profile switches without changing resume authority.
Before switching an operational profile, compare its effective meaning with the
current selection through `narratio config diff <left-profile> <right-profile>`.
The command is read-only and succeeds whether it finds differences or not. Its
sorted records describe defaulted, expanded concrete configuration—not source
file layout—so it can be used to review model, artifact, and publish changes
without creating a session or run. Select the same campaign explicitly when
profiles could resolve different campaign paths; see the [CLI reference](cli.md#config-validate-config-show-config-sources-and-config-diff) for syntax and record format.
## Stage Execution and Continuation Behavior
Canonical stage order:
1. `prepare`
2. `transcribe`
3. `merge`
4. `polish`
5. `normalize`
6. `trim`
7. `render`
8. `extract`
9. `analyze`
10. `publish`
11. `notify`
Execution rules:
- succeeded stages are skipped unless `--force` is set; stages with semantic
configuration contracts additionally require matching versioned evidence,
and missing legacy evidence causes a safe one-time rerun;
- `run` continues interrupted or partially completed sessions by running non-succeeded stages;
- forcing a stage marks succeeded transitive dependents as `stale` before the
replacement runs; render and extract are independent siblings; and
- an executed failure, changed self-skip, or success that replaces a different
effective outcome uses the same fixed dependency relation. A
repeated self-skip with the same reason and no outputs is stable and does not
perpetually rerun dependent work.
Every aggregate stage except analyze currently provides semantic-configuration
evidence; analyze retains its more precise per-artifact fingerprints and
validator.
Prepare additionally validates current stable/local/S3 source identity and the
checksums of its durable prepared copies before reuse. Changed bytes, audio
membership, S3 object identity, or missing/tampered copies rerun prepare and
its fixed descendants without requiring `--force`. Changing prepare selection
semantics likewise reruns all fixed descendants; changing
WhisperX language/service identity reuses prepare; changing a Seriatim merge
transformation reuses prepare and transcribe; and changing an Audita model
reuses prepare, transcribe, and merge while rebuilding transcript refinement.
A trim change invalidates both render and extract through the fixed dependency
relation, while a render-only change preserves the extract sibling.
Operational timeouts, retry/concurrency tuning, executable paths,
workspace/cache/spool placement, reports, diagnostics, and secret values are
excluded. Configuration, models, prompts, modules, or resources loaded
privately inside external tools remain unobservable to Narratio. If their
contents change behind the same configured identifier, explicitly force the
affected stage.
An explicit self-skip is a durable `skipped` stage outcome that later runs
reconsider. It differs from successful no-output execution: disabled `render`
and `publish`, and absent or no-executable `analyze`, record `succeeded` with
metadata and no outputs. Ordinary later runs reuse those successful results;
force the affected stage after enabling or configuring it. Optional artifact
inputs are omitted only from the consuming artifact invocation and do not make
the stage self-skip.
Single-stage execution:
```bash
narratio run-stage normalize 2026-04-04 --force
```
Contiguous bounded execution uses inclusive canonical endpoints:
```bash
narratio session plan 2026-04-04 --from extract --through analyze --force
narratio run 2026-04-04 --from extract --through analyze --force
```
Omitting `--from` selects from `prepare`; omitting `--through` selects through
`notify`. Force applies only within the selected range. Repeating `--from`,
`--through`, or `--force` is rejected instead of resolving by argument order.
The plan command uses the same selection contract and prints only the selected
range. Planning is read-only: it clones the loaded manifest, models selected
stage transitions and invalidation in memory, and invokes resume validation
without writing the manifest, creating run directories, materializing files,
or invoking pipeline adapters. Analyze detail separates explicit targets,
prerequisite rebuilds, scheduled execution, and current reuse. This lets a
coarsely stale aggregate analyze stage show zero artifact executions when its
selected artifact evidence is still semantically current.
Before a bounded run or plan whose range starts after `prepare`, every excluded
prefix stage must already have a session-manifest status of `succeeded` or
`skipped`. Narratio reports the first absent, pending, running, failed, stale,
or interrupted prerequisite without creating a run record or changing session
state. Widen `--from` to include that stage, or recover it explicitly before
retrying. Excluded prefix stages are not resume-validated or repaired as part
of the bounded invocation; selected stages still reject missing, unsafe, or
manifest-inconsistent inputs at their owning boundary.
Stages after `--through` are not prerequisites and are never scheduled by the
bounded invocation. A selected forced stage can mark one of those succeeded
dependents stale through the fixed invalidation relation, but the dependent
does not execute until a later invocation selects it. Production composition
likewise initializes only collaborators needed by the selected range and
shared session lifecycle. In particular, render does not require Notarius or
Scriptorium, extract does not require Scriptorium, and analyze does not require
the transcription, Seriatim, Audita, or Notarius adapters.
For the common post-transcript development loop, use:
```bash
narratio regenerate-artifacts 2026-04-04
narratio regenerate-artifacts 2026-04-04 --artifacts session_recap,player_handout
```
This command is a transparent expansion to a forced bounded `run` from
`extract` through `analyze`. Extraction always rebuilds its complete configured
bundle. Analysis rebuilds the selected targets and their required analysis
prerequisites, or uses the normal default selection when no artifact names are
given. The command does not run publish or notify; delivery remains a separate
operator action.
Inspect current artifact evidence, then publish explicitly when the regenerated
set is ready:
```bash
narratio session artifacts 2026-04-04
narratio publish 2026-04-04
```
If planning or execution reports stale, missing, failed, legacy, or tampered
analysis evidence, regenerate the affected target instead of copying an older
canonical file into place or editing the manifest. See
[Troubleshooting: Analysis artifact evidence is not current](./troubleshooting.md#analysis-artifact-evidence-is-not-current).
## Artifact Selection
For a configured artifact family, selecting its family key expands to every
concrete character artifact. Select a concrete generated key to operate on one
member only. Manifests and plan output retain the concrete key as the durable
identity and include the family and character ID as optional provenance.
`--artifacts` can be used on `run`, `session plan`, `run-stage`, `analyze`, and
`publish`. For a bounded run or plan, the selected range must contain `analyze`
or `publish`.
Selection behavior:
- validates names against `pipeline.scriptorium.artifacts`;
- selects explicit analyze targets and permits their required configured
prerequisites to be reused or rebuilt first;
- filters publish rules for `narratio.artifact.<name>` sources only;
- does not suppress built-in transcript, bounds, or explicitly configured
`narratio.extraction.<name>` publish sources; and
- never partially selects Notarius lanes.
## Extraction Workflow
When Notarius is omitted or disabled, `extract` records an explicit skipped
outcome with reason `notarius_disabled` and no outputs. A later invocation
reconsiders the skipped stage, so enabling Notarius does not require force.
When Notarius extraction is enabled, the stage consumes the final trimmed JSON
and preserves the complete validated Notarius bundle at:
- `artifacts/notarius/{narratio_run_id}/`
The directory is immutable once promoted. Configured lanes become
`narratio.extraction.<name>` sources for Scriptorium and explicit publish rules;
the bundle and `index.json` are retained for audit and resume validation but
are not selectable or published implicitly.
Configured Notarius references resolve only from the current manifest-backed
prepared inputs. Their canonical locations are `inputs/party.yml`,
`inputs/players.yml`, `inputs/glossary.yml`, and, when configured,
`inputs/spell_catalog.json`. Extraction supplies Notarius with verified copies
under `runs/<run_id>/extract/references/` so a concurrent refresh of canonical
prepared files cannot change the bytes consumed by an in-flight invocation.
For a canonical party, preparation retains the validated authored party bytes
at `inputs/party.yml` and generates `inputs/players.yml` from that roster.
The manifest records their checksums separately, with the players input marked
as derived from the party; refresh preparation after changing the roster rather
than editing either prepared file.
Inspect the effective stable-input inventory and
prepared-file readiness with:
```bash
narratio session status 2026-04-04
narratio session validate 2026-04-04
```
Reference metadata records selector, source ID, session-relative path,
checksum, and byte size, but never payload contents. Changing a prepared
reference changes extraction identity: ordinary continuation rejects the old
result, reruns Notarius, and marks successful downstream stages stale. If the
prepared file is missing or inconsistent with its manifest checksum, repair
the source configuration and refresh prepared state first:
```bash
narratio run-stage prepare 2026-04-04 --force
```
Starting a replacement clears the previous extraction payload from the current
session-stage record. If that replacement fails or self-skips, the current
record does not fall back to the earlier outputs. The earlier run manifest and
immutable bundle remain available for inspection, but downstream resolution
requires a new current successful extraction record.
Atomic Notarius bundle promotion is supported on Linux and macOS. On Windows
and other operating systems, extraction fails before copying the bundle into a
temporary promotion tree because Narratio has no verified atomic no-replace
directory primitive there. This is an extraction limitation, not a broader
platform-support guarantee for every Narratio workflow.
## External Command Lifecycle
When an external command is cancelled or times out, Narratio terminates its
owned descendants as well as the command itself. Cancellation first requests
termination where the platform supports it, then force terminates after a
bounded wait. A command is not considered finished until its leader has been
reaped, and descendants that keep standard output or error open cannot keep
the invocation blocked. Other operating systems fail closed rather than launch
a command without tree ownership.
Subprocess stdout and stderr diagnostics are separately redacted and capped at
8 MiB per invocation. Narratio does not retain configured credential values in
these logs or their error tails; reaching a capture limit terminates the command
tree and reports which stream exceeded the limit.
Run-local diagnostics are:
- `runs/{run_id}/extract/notarius.receipt.json`
- `runs/{run_id}/extract/notarius.stderr.log`
- `runs/{run_id}/extract/notarius-output/` before durable promotion
The run-record upload is an allowlist derived from the validated run manifest,
not a workspace scan. Each declared source is opened without following
symlinked ancestors or the leaf, verified as a regular file, and streamed from
that verified descriptor. Unlisted files and unsafe entries are never uploaded.
The durable bundle is never scanned for implicit publication; only lanes named
by explicit `pipeline.publish.outputs` rules are uploaded.
To intentionally replace the current extraction result, run:
```bash
narratio run-stage extract 2026-04-04 --force
```
Narratio automatically reruns extraction when its recorded invocation contract,
prepared Narratio reference identities, or durable output validation changes.
The semantic portion covers Notarius enablement, pipeline identity, declared
reference mapping, and output contracts. Executable, timeout, working directory,
and private config-file paths are operational and do not invalidate a current
result.
It cannot fingerprint configuration files, profiles, prompts, modules, or
other references loaded transitively by Notarius itself. Force extraction after
changing any of those inputs, even when the top-level Narratio and Notarius
config paths remain the same. A forced extract
marks successful downstream stages stale. Ordinary extraction failures or
outcome changes also stale affected downstream stages, while an identical
repeated `notarius_disabled` self-skip does not repeatedly invalidate them.
Publish reuse additionally tracks enabled/run-upload behavior, normalized
output rules, static locks, and remote backend/bucket/region/endpoint/root
identity. Credential environment names, local workspace placement, and run IDs
are excluded. Regardless of semantic reuse evidence, executing publish still
revalidates mutable remote locks immediately before commit selection.
## Publish Workflow
Run publish only:
```bash
narratio publish 2026-04-04
```
Equivalent:
```bash
narratio run-stage publish 2026-04-04 --force
```
Publish commit model:
- uploads eligible run files under `{session_prefix}/runs/{run_id}/`, excluding
audio and the run-local Notarius staging bundle;
- uploads configured published outputs and `previous/**` cache files into the
same immutable run scope, including only explicitly configured extraction
lanes;
- writes `{session_prefix}/runs/{run_id}/commit.json` after all declared
immutable objects are uploaded and verified; and
- writes `{session_prefix}/current/commit-pointer.json` once, last.
`current/commit-pointer.json` is the remote current-state commit marker. It
selects exactly one immutable commit, which declares the complete object set.
## Remote Commit Migration
The immutable remote commit contract uses
`runs/{run_id}/commit.json` to declare a run's complete object set and a small
`current/commit-pointer.json` to select it. The pointer binds the selected
commit by version, checksum, size, and storage generation; committed artifacts
are also checksum- and generation-bound. Readers accept this contract now and
strictly reject mismatched or unknown data.
Legacy reads are limited to a coherent `current/manifest.json` and
`current/run_id.txt` pair; a torn pair is rejected. New publication does not
write that pair and remote commit state does not carry local
`current_pointer_written` metadata.
## Publish Locks
Lock sources:
- static locks in `pipeline.publish.locks`
- mutable remote locks in `{session_prefix}/locks.yml`
Effective lock rules:
- static and remote locks are merged;
- static locks win on source collisions;
- locked outputs are intentional skips;
- lock add/remove commands mutate only remote lock state through generation-bound
conditional writes. A command retries a bounded number of concurrent
conflicts while its invocation context remains active, so it never replaces a
different lock-document generation; and
- a publish re-reads remote locks immediately before it writes the current
commit pointer. A lock committed before that recheck prevents selecting the
new snapshot, even though its already-uploaded immutable objects may remain
available for a later retry.
Examples:
```bash
narratio session locks 2026-04-04
narratio session locks add 2026-04-04 narratio.artifact.session_recap --reason "manual edits" --force
narratio session locks remove 2026-04-04 narratio.artifact.session_recap
```
## Restore Workflow
Use restore when local durable session state is missing or stale and remote committed current state is authoritative.
Dry run:
```bash
narratio session restore 2026-04-04 --dry-run
```
Apply:
```bash
narratio session restore 2026-04-04
```
`--dry-run` does not write durable session files. It still reads the selected
remote current state and may read object identity/content needed to classify the
plan, so it is not a network-free operation.
Default restore scope:
- the committed session manifest and the committed transcript/artifact objects
declared by the selected remote commit
- `previous/**` when needed by configured previous-session artifact inputs
Optional:
- `--include-audio` to include `audio/**`
- `--force` to overwrite eligible conflicting regular files; it never replaces
directories or other non-regular local targets
Restore writes an execution report at `reports/restore-latest.json`.
If restore fails after beginning installation, it leaves a durable
`.restore-incomplete.json` marker in the session root. Pipeline runs will stop
until you rerun the same restore command and it completes. Restore intentionally
does not try to roll back files already installed; retrying the selected remote
snapshot is the recovery procedure.
## Local State Layout
Session root:
- `{workspace.root}/work/{campaign}/{session_id}`
Durable session paths:
- `manifest.json`
- `inputs/**`
- `audio/**`
- `transcripts/**`
- `artifacts/**`
- `previous/**`
- `reports/**`
- `logs/**`
- `config/**`
- `runs/**`
Validated Notarius bundles live below `artifacts/notarius/{run_id}/`; receipt,
stderr, and pre-promotion output remain in the producing run's `extract`
directory as described in [Extraction Workflow](#extraction-workflow).
Run-local layout:
- `runs/{run_id}/{stage}/outputs`
- `runs/{run_id}/{stage}/logs`
- `runs/{run_id}/{stage}/reports`
- `runs/{run_id}/{stage}/config`
- `runs/{run_id}/{stage}/scratch`
Spool layout (runtime/transient):
- `{spool.root}/{campaign}/{session_id}/{run_id}/...`
- restore audio spool under `{spool.root}/{campaign}/{session_id}/restore/audio`
Cache layout (durable S3 audio cache):
- `{cache.root}/s3/{bucket}/...`
Each cached audio file has an adjacent managed identity record. It binds the
file to its remote object version and verified digest; deleting or altering the
record simply causes Narratio to download and verify the object again.
### Workspace Permissions
Ordinary Narratio workspace content is intentionally shareable with the
workspace group. On POSIX systems, Narratio-created workspace, spool, and cache
directories converge on setgid `02775`; ordinary files, including manifests,
transcripts, generated configuration, logs, reports, and Notarius artifacts,
converge on `0664`. Narratio explicitly applies these modes so a restrictive
caller umask does not remove group write or setgid. It does not change file or
directory ownership: the configured workspace's existing group is inherited.
Windows does not implement POSIX mode bits or setgid semantics. Configure the
workspace, spool, and cache locations with an ACL that grants the collaborating
group read/write access, and configure credential locations with an ACL limited
to the intended credential owner. Do not use POSIX mode displays as evidence of
Windows access control.
API keys are credentials, not ordinary workspace data. Store them outside the
shared workspace or in a separately restricted credential location; ordinary
workspace group access must never be treated as authorization to read keys.
On POSIX, provision a credential directory as `0700` and credential files as
`0600`; Narratio rejects group- or other-readable configured credential paths.
On Windows, restrict the directory and files with ACLs to the credential owner.
External adapter results are individually bounded before Narratio validates or
materializes them. These per-file limits do not reserve disk space: prevent hard
disk exhaustion with filesystem, service, container, or volume quotas sized for
the session workload.
## Cleanup
Session-scoped cleanup:
```bash
narratio clean 2026-04-04
```
Global cleanup:
```bash
narratio clean --all
```
Dry-run and cache variants:
```bash
narratio clean 2026-04-04 --dry-run --clear-cache
narratio clean --all --dry-run --clear-cache
```
Rules:
- `clean` deletes work/spool session state;
- cache is preserved unless `--clear-cache` is set;
- each deletion is confined beneath its configured workspace, spool, or cache
root and refuses symlinked paths;
- automatic post-publish cleanup is gated by successful publish commit plus:
- `pipeline.spool.delete_audio_after_publish=true`
- `pipeline.workspace.cleanup_after_publish=true`
- Narratio first records the exact run-scoped cleanup obligation. If cleanup
reports incomplete, the remote committed snapshot remains current; rerun
publish to retry only the outstanding confined local cleanup.
Post-publish cleanup is evaluated only when `publish` actually executes in the
current invocation. A bounded range that excludes publish does not replay a
cleanup obligation as an unrelated side effect.
## Operational Caveats
- Local and S3 audio modes are mutually exclusive.
- Publish requires prerequisite stages through `render` and `analyze` to be succeeded.
- Markdown publish defaults require render outputs (`transcripts/final.md` and `transcripts/final.trimmed.md`).
- Restore requires configured object storage and committed remote current state.
- Storage-backed commands load filesystem secrets before object-store initialization.

286
docs/policy/architecture.md Normal file
View File

@@ -0,0 +1,286 @@
# Architecture
This document defines Narratio's intended high-level architecture and the
invariants that changes must preserve. Implemented component details belong in
the [Internal Overview](../internal/overview.md) and its linked documents.
Significant architectural decision history belongs under `docs/adr/` when such
records exist.
## System Shape
Narratio is a small Go application that turns D&D session audio into polished
transcripts and generated session artifacts. It is an explicit, stage-driven
orchestrator, not a general workflow engine.
Narratio coordinates specialized external systems rather than reimplementing
their domains:
- WhisperX performs transcription;
- Seriatim performs deterministic transcript processing and rendering;
- Audita performs transcript correction and polishing;
- Notarius extracts validated structured artifact bundles; and
- Scriptorium executes prompts and produces configured artifacts.
Narratio owns orchestration, configuration resolution, session and run state,
artifact and path modeling, manifest persistence, stage sequencing, resume,
restore, cleanup gates, and publish semantics. External contracts are defined
in the [integration documentation](../integrations/).
The pipeline has one canonical ordered stage set. Configuration may enable,
disable, or parameterize supported behavior, but it must not turn that sequence
into an arbitrary DAG or hide orchestration in generic workflow abstractions.
An invocation selects either the full sequence or one inclusive contiguous
range of it. Execution remains flat and canonical even though invalidation is
dependency-aware: the application owns a separate fixed relation used only to
stale transitive dependents, including dependents outside a selected range.
The implemented stage inventory belongs in the
[Internal Overview](../internal/overview.md).
Narratio is contract-first without being abstraction-heavy. Interfaces and
extension points should protect demonstrated boundaries. New abstraction is not
itself an architectural goal.
## Ownership And Dependency Direction
The application boundary owns command dispatch, configuration selection,
production composition, session locking, and top-level lifecycle. It may depend
on concrete implementations to assemble a run.
Stage orchestration expresses intent in Narratio-level data and interfaces.
Stages may depend on configuration, manifest, artifact, path, and adapter
contracts, but they must not depend on transport-specific request types,
subprocess argument construction, cloud SDK types, or downstream tool internals.
Adapters translate between Narratio contracts and external systems. They own
HTTP, subprocess, notification, and object-storage mechanics, including command
construction, transport behavior, provider response handling, and external
error adaptation. External dependency types must remain inside the adapter that
owns them unless that dependency is the adapter's explicit public contract.
WhisperX HTTP behavior, Seriatim, Audita, Notarius, and Scriptorium command
construction, notification transport, and object-storage SDK details remain
behind these boundaries.
State and path services must not infer stage policy. Storage implementations
receive explicit bucket-relative keys and do not infer campaign, session, run,
or root-prefix semantics. Manifest persistence records transitions but does not
choose orchestration policy. Artifact resolution identifies and validates
artifacts but does not execute producers.
Dependencies should remain narrow and point toward Narratio-owned contracts.
Prefer the Go standard library. Add an external dependency only when it provides
a clear correctness, security, interoperability, or complexity benefit, and
confine it to the boundary that needs it.
## Stage Boundaries
Each stage has one explicit responsibility and declares:
- required input state;
- produced output state;
- configuration it consumes;
- external adapters it uses;
- manifest references and metadata it reads or writes;
- skip, force, invalidation, and resume behavior; and
- failure behavior.
Stages write and validate run-local results before materializing canonical
outputs where that distinction applies. A stage is complete only after its
required outputs have been written, validated, and recorded in durable manifest
state. Later stages depend on recorded success and artifact resolution, not
merely on incidental files existing on disk.
A failed or interrupted stage must not be presented as successful. Failure
should preserve enough local state and diagnostics for inspection, recovery,
and resume. Forcing a stage invalidates succeeded transitive dependents
according to a fixed application-owned relation that is separate from canonical
execution order. The relation is validated against the stage inventory and is
not configurable.
A stage may explicitly self-skip with a stable reason and no outputs. That
outcome is persisted, clears older outputs owned by the stage, and is
reconsidered on a later invocation. A stage may also validate whether an
otherwise successful recorded result is still resumable; an obsolete result
is staled and rerun, while an unsafe condition that prevents a sound decision
stops execution.
Shared behavior should live behind a narrow service or helper with one clear
owner. Stages must not reach across boundaries or reproduce adapter, manifest,
artifact, or path policy ad hoc.
## Manifest, Resume, And Restore
The session manifest is the durable ledger for progress across invocations. It
records session and run identity, stage state, input and output references,
diagnostic references, checksums or provenance where useful, and non-secret
adapter and publish metadata.
Resume and skip decisions are manifest-driven. Filesystem state may be
inspected and validated, but file presence alone does not replace recorded
stage state. Invocation-scoped run records provide an audit of one execution;
they do not replace the session manifest as progress authority.
Restore treats committed remote current state as its authority. It must plan
deterministically, confine remote-to-local paths, protect local conflicts, and
install the validated session manifest after other restored durable files. The
physical workflow and recovery procedures belong in
[Operations](../operations.md).
Restore and runner transitions for one session use the same local lock. A
durable incomplete-restore marker blocks runner reuse after a partial restore;
safe retry, rather than rollback of arbitrary local effects, is the recovery
mechanism. Restored manifest-local references must be confined to the selected
local session root, never trusted as producer-machine absolute paths.
For the immutable remote-commit protocol, a restore or status operation binds
to one pointer-selected commit and only its declared object identities. A force
flag may replace an eligible regular managed file, but never turns a directory
or other non-regular conflict into a successful restore.
## Configuration
Configuration is strict, explicit, centralized, and operator-oriented.
- YAML decoding rejects unknown fields.
- Defaults are centralized and testable.
- Empty configured values do not silently replace meaningful defaults.
- Validation rejects invalid composition before stage execution where
practical.
- Root-owned imports and one selected profile resolve deterministically through
the configuration owner; commands do not implement their own merge rules.
- Canonical party rosters are campaign-owned. Their derived players projection
and concrete character-family artifacts are resolved before runtime stages
or adapters receive configuration.
- Resume uses stage- or artifact-owned semantic evidence for observable
result-affecting configuration; profile identity and an effective digest are
provenance, never blanket cache keys.
- Session templating remains narrow and deterministic rather than becoming a
general configuration language.
- Secret values are supplied indirectly and are not persisted in ordinary
configuration.
Narratio must not become a second configuration system for downstream tools.
External systems own their runtime defaults wherever practical; Narratio passes
the paths required by its stage contracts and explicit operator overrides. The
field-level contract and credential-supply mechanisms belong in
[Configuration](../config.md).
## Artifacts, Paths, And Storage
Artifact identities and local and remote paths are application contracts.
Canonical helpers own workspace, spool, cache, session, run, input, transcript,
artifact, log, report, configuration, and publish-current paths. Callers must
not reconstruct canonical paths through scattered string concatenation.
Reusable audio cache entries require a typed record that binds a confined,
no-follow regular file and its digest to the selected remote object identity.
Size alone and unqualified multipart ETags are not content-integrity evidence.
Artifact resolution is deterministic and manifest-aware. Producers materialize
canonical outputs before reporting success, and consumers resolve declared
artifact identities rather than infer files from unrelated directory contents.
External artifact bundles become current only through validated immutable
promotion and manifest records; directory presence alone never establishes
availability.
Writes, moves, replacements, and deletions must use narrow, explicit,
root-confined destinations. Symlinks, traversal, broad roots, and ambiguous
relative destinations must not expand the scope of an operation. Cleanup is
permitted only through explicit operator action or configured post-publish
gates, and it must preserve durable cache unless cache removal is explicitly
requested.
Physical layout, retention, and operational lifecycle belong in
[Operations](../operations.md). Logical external formats and durable integration
contracts belong under [Integrations](../integrations/).
## Publish Commit Boundary
Publish has one explicit remote commit boundary. A remote run becomes current
only after Narratio has successfully uploaded its immutable run-scoped objects,
the immutable commit manifest, and finally the current commit pointer.
`current/commit-pointer.json` is the sole mutable selector and must be written
exactly once, last. Failed, incomplete, skipped, or uncommitted publish attempts
must not be presented as current remote state. Publish locks remain authoritative
and are not bypassed by a forced run. Mutable remote locks use provider-enforced
generation preconditions and are revalidated immediately before pointer
selection; loss of that check leaves the prior committed snapshot current.
Automatic local cleanup is permitted only after a successful publish commit,
only when explicitly configured, and only through the path-safety guardrails.
It is a durable local obligation bound to that committed run and its exact
targets, not an inferred side effect of the current stage list. A cleanup
failure makes the invocation incomplete while leaving the committed remote
snapshot authoritative; later invocations resume the recorded obligation.
## Security, Privacy, And Diagnostics
Narratio distinguishes ordinary workspace data from credentials. Campaign and
session material—including manifests, transcripts, prompts, generated
configuration, logs, reports, diagnostics, and Notarius artifacts—is
intentionally shareable with the configured workspace group. API-key material
is sensitive and is not covered by the ordinary workspace-sharing policy.
On POSIX systems, Narratio-created ordinary workspace directories converge on
setgid `02775` and ordinary workspace files on `0664`, even when the caller's
umask is restrictive. This preserves the existing workspace group for nested
creation and atomic replacements without changing ownership. API-key storage
uses a separate restrictive contract. On Windows, POSIX mode bits and setgid
are not authoritative; operators must provide the equivalent shared-group and
credential-restricted ACLs described in [Operations](../operations.md#workspace-permissions).
Raw secrets must not be stored in pipeline, campaign, or session YAML or written
to manifests, logs, generated configuration, reports, publish metadata,
documentation, or examples. Secrets enter through configured environment
variable names or secret-file references. Diagnostics should avoid transcript
and prompt content unless a deliberate, bounded inspection mechanism requires
it.
Logs, reports, generated invocation files, generated configuration, and render
debug files are diagnostics, not canonical pipeline products. They should be
durable and discoverable where configured, and manifest references must preserve
the distinction between diagnostics and artifacts.
Documentation security rules belong in the
[Documentation Policy](documentation.md). Credential supply belongs in
[Configuration](../config.md), while permissions, sensitive runtime-artifact
handling, and recovery belong in [Operations](../operations.md).
## Determinism And Testability
Narratio prefers deterministic behavior where practical, including stable local
and remote layouts, sorted operation order, predictable generated
configuration, repeatable command construction, deterministic artifact
resolution, and reproducible planning.
Run IDs and timestamps may be intentionally variable, but surrounding behavior
must remain controllable in tests. Core behavior should be testable without live
external services; expensive, nondeterministic, destructive, or external
boundaries should be replaceable with focused test doubles. General testing
philosophy and sufficiency rules belong in the [Testing Policy](testing.md).
## Documentation And Decision Records
Documentation follows the [Documentation Policy](documentation.md). Current
behavior belongs in its canonical user, operator, integration, architecture, or
internal owner. Proposed behavior and implementation status belong under
`docs/roadmap/`.
Significant architectural decisions may be recorded under `docs/adr/` using the
format and lifecycle defined by the documentation policy. ADR acceptance does
not establish that a decision has been implemented.
## Architectural Non-Goals
Narratio does not aim to provide:
- a generic DAG or workflow engine;
- a replacement configuration layer for WhisperX, Seriatim, Audita,
Scriptorium, or other downstream tools;
- a storage abstraction broader than the needs of this pipeline;
- stage logic coupled directly to cloud SDKs, transports, subprocess details,
or downstream implementation internals;
- raw-secret persistence;
- implicit cross-stage behavior that bypasses manifest and artifact contracts;
or
- a prompt-authoring system.

View File

@@ -0,0 +1,150 @@
# Documentation Policy
## Purpose
This policy assigns each documentation topic to one canonical owner. Its goal is
to keep Narratio documentation accurate, concise, discoverable, and resistant
to drift for users, operators, developers, integrators, and LLM coding agents.
## Core Rules
### One Canonical Owner
Each authoritative fact belongs in one document. A non-owning document may give
a short, stable summary for orientation, but it must link to the canonical owner
instead of repeating volatile details.
Volatile details include commands, flags, configuration fields and defaults,
stage or integration keys, schemas, file names, paths, status codes, retry
behavior, and runtime guarantees. If readers could reasonably treat a statement
as a contract, maintain it only in the owning document.
### Current And Future Behavior
Outside `docs/roadmap/`, documentation describes implemented behavior only.
Partial features may be described only to their implemented boundary.
ADRs are the narrow exception: an ADR may record an accepted architectural
decision before implementation, but acceptance must not be presented as proof
that the behavior exists. The roadmap owns implementation status and sequencing
until the decision is implemented. Current architecture, user, operator,
integration, and internal documentation are updated when the behavior lands.
### Audience And Detail
Write for the document's stated audience and include only the detail needed for
its owned topic. User and operator docs should not expose implementation detail.
Developer docs should link to user-facing and external contracts rather than
restate them.
### Examples
Complete copyable files belong in `examples/`. Documentation may use the
smallest illustrative snippet needed to explain its owned topic, but should link
to maintained examples instead of embedding a second complete copy.
Examples must be valid, secret-free, and tested where practical. Commands and
configuration used in documentation should match the application.
### Security And Privacy
Documentation and examples must not contain real credentials, private keys,
private environment dumps, sensitive source material, or private infrastructure
details unless intentionally public. Document secret-handling mechanisms, not
secret values.
## Canonical Ownership
| Topic | Canonical owner | Owned content | Content owned elsewhere |
| --- | --- | --- | --- |
| Product orientation and minimal end-to-end quickstart | `README.md` | What Narratio is, why it is useful, one shortest successful invocation, and links onward. | Complete command reference, configuration reference, operational procedures, implementation detail. |
| Contributor entry point | `docs/development.md` | Task-oriented reading guide, minimal contributor orientation, baseline validation commands, and links to canonical docs. | Package inventory, architecture rules, subsystem behavior, detailed change recipes. |
| Maintainer release procedure | `docs/release.md` | Version selection, candidate preparation and validation, guarded tag publication, release completion boundary, failure recovery, and optional asynchronous inspection. | Script implementation mechanics, current application contracts, and historical release summaries. |
| Historical release summary | `docs/releases/<tag>.md` | Immutable summary, compatibility, upgrade, and changes for one released version. | Current maintainer procedure and current application contract details. |
| Current application architecture | `docs/policy/architecture.md` | System shape, normative ownership, dependency direction, architectural boundaries, invariants, safety properties, and non-goals. | Concrete package inventory, implementation mechanics, contributor procedures, decision history, future work. |
| Documentation organization | `docs/policy/documentation.md` | Documentation ownership, audience boundaries, maintenance rules, and ADR/document lifecycle. | Application architecture or product behavior. |
| Testing policy | `docs/policy/testing.md` | Test philosophy, risk-based sufficiency, test boundaries, doubles, coverage guidance, regression-test policy, and criteria for adding, rewriting, or deleting tests. | Subsystem behavior, application contracts, subsystem-specific test inventories, and implementation plans. |
| CLI contract | `docs/cli.md` | Commands, arguments, flags, invocation semantics, output conventions, and exit behavior. | End-to-end operating procedures, configuration field definitions, runtime filesystem layout, stage implementation details. |
| Configuration contract | `docs/config.md` | Discovery and precedence, file schemas, fields, defaults, environment overrides, validation rules, and user-selectable stage or integration settings. | Complete example files, CLI syntax, runtime state lifecycle, implementation details. |
| Operations | `docs/operations.md` | Runtime workflows, physical filesystem and remote-state layout, output and diagnostic handling, resume, cleanup, permissions, recovery, and operational limits. | CLI flag syntax, configuration field definitions, logical artifact schemas, implementation mechanics. |
| Troubleshooting | `docs/troubleshooting.md` | Symptom-driven diagnosis, likely causes, safe inspection steps and remedies, and links to relevant contracts. | CLI syntax, configuration definitions, operational procedures, integration contracts, implementation mechanics. |
| Public HTTP contract, if introduced | `docs/api.md` | Routes, authentication, media types, request and response schemas, status codes, pagination, caching, idempotency, rate limits, and HTTP retry semantics. | Client walkthroughs, upstream or downstream integration internals, implementation detail. |
| Consumer guidance, if a public package or API is introduced | `docs/consumers/` | Task-oriented use of the public interface, minimal client examples, and consumer responsibilities. | HTTP wire semantics, external protocol contracts, internal implementation detail. |
| External and durable integration contracts | `docs/integrations/` | External file formats and protocols, upstream and downstream contracts, logical artifact paths and schemas, media types, and compatibility behavior. | Physical runtime placement and lifecycle, internal transformations, CLI syntax, configuration defaults. |
| Implemented component inventory | `docs/internal/overview.md` | Current packages and components, their implemented responsibilities, and links to focused internal docs. | Normative architecture, contributor reading policy, external contracts. |
| Internal component behavior | Other files under `docs/internal/` | Implementation flow, internal collaborators and state transitions, package-local guarantees and failures, and relevant tests. | Global architecture invariants, configuration definitions and defaults, external schemas, operator procedures. |
| Architectural decision history | `docs/adr/` | Significant decisions, context, alternatives, rationale, consequences, and supersession history. | Current behavior reference, implementation status, task sequencing. |
| Future work and implementation status | `docs/roadmap/` | Proposed, accepted, deferred, or rejected work; implementation status; sequencing; and task breakdowns. | Implemented behavior reference and architectural decision rationale. |
| Complete copyable artifacts | `examples/` | Maintained configuration, inputs, and other files intended to be copied or run. | Field-by-field reference, command reference, prose explanation. |
Documents that do not exist are required only when the corresponding interface
or responsibility exists. Do not create placeholder API, consumer, integration,
or operations documents for behavior the application does not have.
## Boundary Rules
### Orientation
The README owns product orientation. The developer guide routes contributors.
Architecture owns normative structure. Internal overview owns the current
concrete component map. These documents may link to one another but should not
maintain parallel package or behavior descriptions.
### Commands, Configuration, Operations, And Troubleshooting
CLI documentation answers how to invoke the application. Configuration
documentation answers what settings mean. Operations answers what happens to
runtime state and how to operate or recover the application. Troubleshooting
starts from observable symptoms and links readers to the owning command,
configuration, operational, or integration contract. When a workflow crosses
these topics, choose the document that owns the task and link to the other
contracts.
### Contracts And Implementation
Integration and API documents define externally observable shapes and
semantics. Internal documents explain how Narratio implements or consumes those
contracts. Internal docs may name a field, file, or protocol to identify a
dependency, but must link to its canonical contract for the definition.
### Security Topics
This policy owns what documentation and examples may contain. Architecture owns
application security invariants. Configuration owns credential-supply
mechanisms. Operations owns permissions and handling of sensitive runtime
artifacts. Troubleshooting owns safe diagnostic and remediation guidance.
Internal docs own implementation mechanisms only.
## Architecture Decision Records
Use sequentially numbered ADR filenames such as
`0001-record-architecture-decisions.md`. Follow the lightweight Nygard format:
1. title;
2. status;
3. date;
4. context;
5. decision;
6. alternatives considered;
7. consequences.
Treat the decision content of an accepted ADR as immutable. When a decision
changes, create a new ADR and update the earlier ADR's status to superseded.
Rejected architectural alternatives belong in the ADR; rejected product ideas
belong in the roadmap.
## Maintenance
When behavior changes, update its canonical owner in the same change. If
ownership moves, remove the old definition and replace it with a link where
navigation remains useful.
Before completing documentation work:
- verify affected behavior and examples;
- check commands, flags, fields, defaults, schemas, and paths against their
implementation;
- keep unimplemented behavior in the roadmap, subject to the ADR exception;
- remove stale references and validate links;
- confirm that non-owning documents summarize and link rather than redefine;
- confirm that no secrets or sensitive private data were added.

296
docs/policy/testing.md Normal file
View File

@@ -0,0 +1,296 @@
# Testing Policy
## Purpose
Our tests exist to make **incorrect changes expensive and correct changes cheap**.
We do not optimize for test count, line coverage, exhaustive isolation, or the fewest possible tests. We optimize for sufficient confidence in important behavior while imposing as little unnecessary friction as possible on future development.
## Every test has a cost
Testing is not an unqualified good. Every test imposes both an immediate cost and a continuing lifetime cost.
A test must be:
- written and reviewed;
- understood by future maintainers and coding agents;
- executed in local and CI workflows;
- diagnosed when it fails;
- updated when legitimate behavior changes;
- maintained as fixtures, APIs, and dependencies evolve; and
- removed or rewritten when it becomes redundant, brittle, misleading, or obsolete.
Tests also create cognitive and architectural friction. They can constrain refactoring, duplicate policy, slow feedback loops, add noise to failures, and cause harmless implementation changes to require unrelated edits across the suite.
A test is warranted only when the confidence it provides justifies these costs.
Apply this cost-benefit analysis at two levels:
1. **Per test:** What realistic defect does this test detect, how consequential would that defect be, and is that protection worth the test's lifetime cost?
2. **Across the suite:** Does this collection provide materially more confidence than a smaller, simpler suite would?
The preferred test suite is a **lean suite that provides sufficient confidence in the risks that matter, without redundant or low-value tests**. We seek sufficient confidence with the least unnecessary testing friction, not the fewest possible tests.
Some friction is intentional. Tests should make dangerous changes—such as breaking compatibility, corrupting data, violating security boundaries, or reintroducing subtle bugs—require deliberate review. They should not make ordinary internal changes needlessly expensive.
The cost of a test is not a reason to omit testing by default. Do not cite maintenance cost abstractly. When omitting a plausible test, be able to state why the protected failure is low-risk, already covered, obvious, reversible, or cheaper to detect elsewhere. For consequential, subtle, or difficult-to-observe behavior, the presumption should favor testing.
## Default testing style
Use a **classical/Detroit-style** approach:
- Test observable behavior, resulting state, contracts, and invariants.
- Use real internal collaborators when they are fast and deterministic.
- Use fakes, stubs, or mocks primarily at expensive, nondeterministic, destructive, or external boundaries.
- Prefer package-level behavioral tests over tests coupled to private helpers or internal call sequences.
- Treat exact collaborator interactions as testable behavior only when the interaction itself is a requirement.
Examples of appropriate seams include clocks, randomness, subprocesses, remote APIs, object storage, email, and paid LLM calls.
## Test execution requirements
Tests in the default suite must be deterministic, offline, and independent of real credentials. They must not invoke paid APIs or depend on mutable external services. Tests that require live infrastructure must be explicitly opt-in and clearly separated from the default suite.
Control clocks, randomness, environment variables, and other process-global or machine-specific state when they affect behavior. Tests should be safe to run repeatedly and alongside other tests without depending on execution order or state left by an earlier test.
## What deserves tests
Prioritize tests for:
1. Public and package-level contracts.
2. Domain rules and important invariants.
3. Boundary conditions and malformed input.
4. Failure handling, cancellation, retries, recovery, and partial success.
5. Serialization, schemas, compatibility, and round trips.
6. Previously observed or plausible regressions.
7. Representative integration and end-to-end workflows.
A package-level contract is behavior relied upon by another package or major collaborator, not every observable detail of a package implementation.
For behavior involving **data integrity, destructive operations, compatibility, security, concurrency, idempotency, or recovery**, presume that durable tests are required unless the behavior is already credibly protected at another layer.
Do not add tests merely because a function, branch, or line exists. Do not add a test when the same meaningful risk is already adequately protected elsewhere.
## Choose the right test boundary
Test through the narrowest stable boundary that expresses the behavior clearly.
This is often the package API, but it may instead be:
- a smaller pure function when dense domain logic is most clearly isolated there;
- a package-level operation when several internal collaborators jointly produce the behavior; or
- a larger integration boundary when correctness emerges from interaction with a real dependency.
Do not force all behavior through oversized end-to-end tests. Do not test every private helper merely because it exists. Choose the boundary that gives durable confidence with the least incidental coupling.
## Test behavior, not implementation
A test should protect a decision, contract, or invariant—not memorialize the current implementation.
Before adding or retaining a test, ask:
> What realistic defect would this test catch?
A test is suspect when its main purpose is to detect that someone:
- changed an internal constant;
- renamed or split a private helper;
- reordered equivalent internal operations;
- changed incidental formatting;
- replaced one correct algorithm with another; or
- refactored internal object structure without changing behavior.
Refactoring should normally require no test edits unless the refactored structure is itself part of the contract.
A test can be factually correct and still have negative value. Accurately describing current behavior is not enough; the protected behavior must be important enough to justify the future friction.
## Expected effects of different changes
Use the following expectations when evaluating test failures and test maintenance:
| Change | Expected effect on tests |
|---|---|
| Internal refactor that preserves behavior | Existing tests should normally remain unchanged and continue to pass. |
| Change to an internal default with no contractual significance | Behavioral tests should normally remain unchanged; tests should derive expectations from configuration or relationships rather than duplicate the old value. |
| Intentional change to public behavior, policy, schema, or compatibility guarantees | The relevant tests should be reviewed and changed deliberately. |
| Accidental violation of a contract or invariant | Tests should fail; fix the production code rather than rewriting the tests to accept the defect. |
A test failing is not the same as a test needing to be edited. Many tests may correctly fail because of one production defect. The maintenance smell is a correct internal change that requires unrelated expectation updates throughout the suite.
## Separate mechanism from policy
Configurable thresholds and defaults must not be duplicated throughout the test suite.
For example, do not encode an internal concurrency limit indirectly:
```go
// Production policy:
const maxConcurrency = 4
// Brittle test:
err := startProcesses(5)
require.Error(t, err)
```
Instead, test the mechanism relationally:
```go
const limit = 2
runner := NewRunner(limit)
require.NoError(t, runner.Start(limit))
require.ErrorIs(t, runner.Start(limit+1), ErrTooMuchConcurrency)
```
The test should prove:
- the configured limit is accepted; and
- one beyond the configured limit is rejected.
The production default should be tested exactly only when its literal value is itself a public, operational, safety, protocol, or compatibility requirement.
Apply the same rule to limits, timeouts, capacities, retry counts, and ranges: test relationships and behavior, not duplicated literals.
For concurrency limits, test both kinds of behavior when relevant:
1. **Configuration enforcement:** invalid or excessive requested values are handled correctly.
2. **Runtime enforcement:** observed peak concurrency never exceeds the configured limit.
Use a test-controlled limit and measure the behavior relative to that limit. Do not merely assert today's default value.
## Avoid semantic duplication across layers
Each behavior should have a clear test owner.
- Parser tests own parsing cases.
- Validator tests own validation rules.
- Domain tests own transformations and invariants.
- Adapter tests own external integration behavior.
- Orchestrator tests own coordination and failure propagation.
- CLI tests own argument and configuration mapping.
- End-to-end tests prove that representative assembled workflows work.
Higher-level tests should not repeat every lower-level case. A single intentional policy change should not require unrelated edits across many test files.
Tests that are individually reasonable may still be collectively redundant. Evaluate the marginal value of each additional test in light of the protection already provided by the rest of the suite.
## Use test doubles deliberately
Choose the least elaborate test double that provides the required control or observation.
As a default:
1. Prefer real collaborators when they are fast and deterministic.
2. Use small in-memory fakes when realistic stateful behavior is helpful.
3. Use stubs when a dependency only needs to provide controlled responses.
4. Use mocks when the interaction itself is contractual.
Mocks are appropriate when the contract includes facts such as:
- a notification is sent exactly once;
- a transaction is committed only after successful writes;
- cancellation reaches a subprocess;
- an expensive API is called no more than once; or
- a security audit event is emitted.
Do not use mocks merely to isolate every object or reproduce the implementation's call graph.
## Go-specific guidance
Use:
- table-driven tests for meaningful behavioral categories and boundaries;
- `t.TempDir()` for real filesystem behavior;
- `httptest.Server` for realistic HTTP interactions;
- fuzz tests for parsers, normalization, path handling, and broad input spaces;
- golden files only when the complete output is intentionally stable;
- integration tests where correctness depends on component interaction; and
- a small number of representative end-to-end tests.
Avoid exact error-string assertions unless the wording is itself contractual. Prefer `errors.Is`, `errors.As`, typed errors, or structured error fields.
At CLI boundaries, prefer exit classifications, structured output, and the smallest stable semantic fragment needed to identify the error. Do not snapshot complete diagnostic wording unless it is contractual.
Golden-file updates must require an explicit local flag. CI must not update golden files automatically, and reviewers must inspect the semantic diff before accepting an update.
Keep tests readable and direct. Test helpers and fixture frameworks must earn their own maintenance cost; do not build elaborate test infrastructure for small or isolated needs.
## Coverage
Coverage is a diagnostic, not a target.
Use it to find untested critical branches and unexpectedly weak packages. Do not write low-value tests solely to increase a percentage, and do not infer test quality from coverage alone.
Pure domain logic will often warrant higher coverage than CLI wiring or external adapters. Uneven coverage is acceptable when it reflects risk.
Increasing coverage is valuable only when the newly covered behavior protects a meaningful risk at an acceptable cost.
## Regression tests
A bug fix should normally include a regression test that fails before the fix and passes afterward.
Retain the test when the defect could realistically recur and its consequences justify the ongoing cost. Prefer the narrowest durable test of the violated contract or invariant; do not preserve accidental implementation details from the original bug.
Not every historical bug requires a permanent test. If the underlying design has made recurrence impossible, the test has become redundant, or a stronger invariant test now subsumes it, remove or consolidate it.
## Deleting or rewriting tests
Tests are maintained code, not permanent historical artifacts.
Delete or rewrite a test when its maintenance cost exceeds the confidence it provides.
Strong candidates include tests that:
- require updates after harmless internal changes;
- directly assert private constants without protecting a real contract;
- duplicate the same policy across several layers;
- verify mock choreography rather than outcomes;
- snapshot large amounts of incidental output;
- test trivial private helpers already exercised through stable package behavior;
- protect risks already covered more effectively elsewhere;
- are flaky, misleading, obsolete, or disproportionately expensive to diagnose; or
- no longer correspond to a plausible failure mode.
Several brittle tests may encode one genuine requirement. Replace them with one durable behavior-level or invariant test rather than preserving all of them.
Deleting a low-value test can improve the quality of the suite by reducing noise, maintenance burden, and friction around legitimate change.
## Reviewing a proposed test
Use the following questions when the value, boundary, or durability of a proposed test is not self-evident. Significant test additions should be reviewable against them, but written answers are not required for every routine test.
1. What realistic defect would it catch?
2. How likely is that defect?
3. How consequential would it be?
4. Is the behavior already protected elsewhere?
5. At which layer should this behavior be owned?
6. Does the test assert a durable contract or an incidental implementation detail?
7. Could the implementation be refactored without changing the behavior and without editing this test?
8. What should cause this test to fail?
9. What legitimate changes should not cause this test to fail?
10. What ongoing maintenance, execution, and diagnostic cost will the test impose?
11. Is there a smaller or more direct test that protects the same risk?
Do not add the test when its expected lifetime cost exceeds its expected protective value.
When deciding not to test plausible behavior, record or be able to explain why the risk is low, already protected, obvious, reversible, or cheaper to detect elsewhere.
## Definition of sufficient
A test suite is sufficient when:
- important contracts and invariants are protected;
- meaningful boundaries and failure modes are exercised;
- realistic and consequential regressions are credibly protected against silent recurrence;
- behavior involving data integrity, destructive operations, compatibility, security, concurrency, idempotency, and recovery is credibly protected;
- important external boundaries have realistic integration coverage;
- representative complete workflows are tested;
- failures provide useful signal rather than redundant noise;
- legitimate internal changes usually do not require test edits; and
- additional tests would mostly repeat existing protection or preserve inconsequential implementation details.
Sufficiency is a risk judgment, not a coverage percentage or test count. Reassess it as the application, its users, and the consequences of failure evolve.
The governing rule is:
> Test heavily where failure is consequential, subtle, or difficult to detect after the fact. Test lightly where failure is obvious, reversible, and inexpensive—and retain no test whose lifetime cost exceeds the confidence it provides.

97
docs/release.md Normal file
View File

@@ -0,0 +1,97 @@
# Releasing Narratio
This document is the maintainer procedure for creating a Narratio source and
binary release. The synchronous release boundary is a successful push of one
new tag to `origin`; Woodpecker and Gitea publication happen later and do not
change that result.
## Choose a version and write its note
Narratio is past `v1.0.0`. Use an unused stable tag in the exact form
`vMAJOR.MINOR.PATCH`:
- increment `MINOR` for backward-compatible features;
- increment `PATCH` for backward-compatible fixes; and
- reserve a new `MAJOR` for an intentional breaking documented contract.
Before preparing the candidate, create
`docs/releases/vMAJOR.MINOR.PATCH.md` with this structure:
```markdown
# Narratio vMAJOR.MINOR.PATCH
This release ...
## Summary
## Compatibility
## Upgrade
## Changes
```
The compatibility section identifies relevant CLI, configuration, artifact,
integration, or operating-contract changes. The upgrade section states the
required operator action, or explicitly says that no special action is
required. Release notes are immutable historical summaries; link to the
current canonical documentation for detailed behavior.
Commit the note and all candidate changes, then use the ordinary development
workflow to push that commit to `main`. Do not create a release tag before the
candidate is committed and `origin/main` contains the exact same commit.
## Validate the candidate
Run the shared checker from any directory:
```sh
scripts/check-release-candidate.sh vMAJOR.MINOR.PATCH
```
It validates the version and matching note, module hygiene, formatting,
whitespace, uncached tests, race tests, static checks, documentation, examples,
and six official cross-build assets. It uses `GOWORK=off`, does not contact
application services, CI, or Gitea, and does not create tags or modify tracked
source. Fix any failure on `main`, commit it, push it normally, and rerun the
checker.
The checker builds Linux, macOS, and Windows assets for `amd64` and `arm64`.
Cross-builds prove compilation; they are not native macOS or Windows runtime
evidence.
## Publish the tag
From a clean checkout on `main` whose `HEAD` equals `origin/main`, run:
```sh
scripts/release.sh vMAJOR.MINOR.PATCH
```
The command fetches and checks `origin/main`, re-runs candidate validation,
then fetches and checks again before creating an explicitly unsigned lightweight
tag for the originally recorded commit. It refuses dirty, divergent, changed,
or already-tagged candidates. It pushes only:
```text
refs/tags/vMAJOR.MINOR.PATCH:refs/tags/vMAJOR.MINOR.PATCH
```
It never commits changes, pushes `main`, force-pushes, moves a tag, or pushes
all tags. A successful push of that exact ref completes the release command;
the command prints the tag and commit, then returns without waiting for CI,
querying Gitea, downloading assets, or checking checksums.
If a failure occurs before the tag is created, correct the candidate on `main`
and repeat validation. If the push fails after local tag creation, the local tag
is intentionally retained for inspection and the command must not be retried
blindly. Once the upstream tag has been pushed, it is immutable. Correct any
defect or failed asynchronous publication with a new patch version, a new
release note, and the complete procedure again.
## Optional asynchronous inspection
After a successful tag push, a human may later inspect the tag-triggered
Woodpecker run and the corresponding Gitea release for binaries and checksums.
This is optional follow-up only. Automated releasers must not wait for, poll,
or treat CI/Gitea completion as a condition of the successful tag push.

26
docs/releases/README.md Normal file
View File

@@ -0,0 +1,26 @@
# Release Notes
This directory contains immutable historical release notes for Narratio.
Future notes are created with the matching stable version and use this minimum
structure:
```markdown
# Narratio vMAJOR.MINOR.PATCH
This release ...
## Summary
## Compatibility
## Upgrade
## Changes
```
See the [release procedure](../release.md) for creating a candidate and tag.
When asynchronous publication succeeds, the corresponding Gitea release is the
canonical source for downloadable binaries and checksums.
- [v1.6.0](v1.6.0.md)
- [v1.5.0](v1.5.0.md)

47
docs/releases/v1.5.0.md Normal file
View File

@@ -0,0 +1,47 @@
# Narratio v1.5.0
Narratio v1.5.0 makes repeated post-transcript artifact development faster and
more explicit while retaining the fixed, stage-driven pipeline model.
## Highlights
- The canonical pipeline now completes deterministic rendering before
extraction, cleanly separating transcript-generating stages from
artifact-generating stages.
- `narratio run` and `narratio session plan` accept inclusive `--from` and
`--through` bounds. Excluded transcript stages are not executed or
invalidated by a bounded artifact-regeneration run.
- `narratio regenerate-artifacts SESSION` is an exact convenience alias for a
forced run from `extract` through `analyze`, including focused
`--artifacts` selections.
- Configured Scriptorium artifacts now have independent,
manifest-authoritative freshness. Narratio reuses validated current work,
rebuilds stale prerequisites in dependency order, and persists successful,
failed, and newly stale artifact state when an analysis invocation only
partially succeeds.
- Publish consumes only configured artifacts backed by current manifest
evidence; incidental or tampered files are not promoted as current output.
## Reliability And Administration
- Bounded prerequisites are checked again under the session lock before any
run mutation, closing a concurrent-run race.
- Analysis fingerprints are stable across executable and configuration path
changes and continue to cover only Narratio-observable semantic inputs.
- Runner composition now carries one validated execution plan from command
parsing through prerequisite validation, adapter composition, manifest
recording, and stage execution.
- `narratio version` reports the exact tag embedded in official release
binaries; ordinary source builds report `dev`.
## Upgrade Notes
- Existing unbounded commands and direct `run-stage`, `analyze`, and `publish`
workflows retain their meanings.
- Manifests written before artifact-level analysis state remain readable.
Legacy aggregate analysis success is not sufficient freshness evidence, so
the first analysis evaluation after upgrading may regenerate configured
artifacts once.
- Narratio cannot observe executable contents or configuration, prompt,
profile, module, and other files loaded privately by Scriptorium. Explicitly
force affected artifacts after changing those private inputs.

74
docs/releases/v1.6.0.md Normal file
View File

@@ -0,0 +1,74 @@
# Narratio v1.6.0
Narratio v1.6.0 makes large pipeline configurations easier to organize,
inspect, and vary while adding character-oriented artifact generation and a
guarded, reproducible release procedure.
## Summary
Pipeline configuration can now be assembled from explicit additive imports and
a selected production or testing profile. Campaigns can own a canonical,
versioned party roster, and Scriptorium artifact families can expand one
definition into concrete per-character artifacts, dependencies, variables, and
publish rules.
New read-only configuration commands expose the fully resolved pipeline,
source provenance, semantic digest, and profile differences before a session is
run. Stage reuse now records configuration-sensitive semantic evidence so
profile or configuration changes cannot silently reuse incompatible work.
## Compatibility
This is a backward-compatible feature release. Existing monolithic pipeline
files, concrete Scriptorium artifacts, publish rules, and configurations
without profiles remain supported. When profiles are declared and no explicit
profile is selected, Narratio uses the configured production default.
An unversioned party file plus a separate players file remains available as an
isolated legacy compatibility path, but it cannot drive artifact families. New
campaigns and new party-oriented features should use the `narratio.party.v1`
schema. Existing manifests remain readable; missing legacy semantic evidence
is treated as stale rather than trusted.
Narratio continues to consume the documented Notarius D&D pipeline contract.
The release workflow still cross-compiles Linux, macOS, and Windows binaries
for `amd64` and `arm64`; cross-compilation is not native runtime evidence for
macOS or Windows.
## Upgrade
No special action is required for existing monolithic configurations that do
not adopt profiles or artifact families. On the first run after upgrading,
stages recorded by older manifests may regenerate once because those records do
not contain the new semantic configuration evidence.
To adopt the new configuration model, use the maintained
`examples/production-testing` bundle as a migration reference: split stable
settings into explicit imports, define a production default and optional
testing profile, convert campaign party data to `narratio.party.v1`, remove the
separate players file, and then introduce character artifact families. Review
the result with `narratio config validate`, `config show`, `config sources`, and
`config diff` before running a session.
Campaign, session, previous-session, and run identifiers must satisfy the
documented portable identity grammar. Existing manifests or remote state with
unsafe legacy identifiers must be migrated before use.
## Changes
- Added root-owned, non-recursive additive pipeline imports with strict,
source-aware conflict detection.
- Added named pipeline profiles with an explicit production default and
deliberate command-line selection.
- Added `config validate`, `config show`, `config sources`, and `config diff`
for read-only inspection of resolved configuration and provenance.
- Added the strict `narratio.party.v1` campaign roster, including stable
character IDs, player and character names, optional aliases, and classes.
- Added deterministic players derivation and canonical party delivery to
downstream integrations.
- Added character-oriented artifact families, corresponding member
dependencies, family selection, and generated publish policies.
- Added semantic configuration fingerprints and stage-specific resume checks
across transcript and artifact stages.
- Added shared release candidate, asset build, and guarded tag-publication
scripts, with tag-triggered publication remaining asynchronous.

704
docs/troubleshooting.md Normal file
View File

@@ -0,0 +1,704 @@
# Troubleshooting
Operational diagnosis guide for common Narratio failures.
## Config file not found
Symptom:
- command fails to resolve `pipeline.yml`, `campaign.yml`, or `session.yml`.
Likely causes:
- missing files in default search paths;
- wrong campaign selection;
- omitted explicit flags.
Diagnostics:
```bash
narratio session plan 2026-04-04
```
Safe fix:
- pass explicit `--config`, `--campaign` or `--campaign-file`, and `--session`.
Relevant reference: [Configuration discovery](./config.md#discovery-and-selection).
## Session template placeholders rejected
Symptom:
- load error says session file must be concrete or contains `{{ ... }}` placeholders.
Likely cause:
- using template content as runtime session config.
Diagnostics:
```bash
narratio session validate 2026-04-04 --session /path/session.yml
```
Safe fix:
- generate concrete session YAML with `narratio session init`.
Relevant reference: [Operations: Session Initialization](./operations.md#session-initialization).
## Strict decode or schema validation failure
Symptom:
- unknown field / invalid value error during config load.
Likely cause:
- stale field name, typo, invalid enum, or invalid duration/path format.
Diagnostics:
```bash
narratio session plan 2026-04-04 --config /path/pipeline.yml --campaign-file /path/campaign.yml --session /path/session.yml
```
Safe fix:
- align config with [Configuration](./config.md) and the
[maintained examples](../examples/README.md).
Relevant reference: [Configuration](./config.md).
## Unexpected imported or profile value
Symptom:
- an effective configuration value differs from the root file, or a duplicate
ownership/configuration error is hard to locate.
Diagnostics:
```bash
narratio config sources --config /path/pipeline.yml --profile testing
```
Add `--campaign` or `--campaign-file` when the pipeline has party-driven
artifact families. The output identifies each effective logical field's root,
import, profile, default, campaign, party, or family source without printing
the field value or credential contents.
Safe fix:
- move a duplicated base field so it has one owner;
- correct the selected profile or its overlay; or
- correct the campaign party/family declaration that owns generated values.
To review what would actually change before switching profiles, run `config
diff` with the same pipeline and campaign selectors. It compares normalized
effective values rather than YAML formatting or source-file layout.
Relevant reference: [Configuration inspection](./config.md#read-only-effective-pipeline-inspection).
## Audio mode conflict
Symptom:
- validation fails on session audio configuration.
Likely cause:
- configured both local and S3 session audio inputs.
Diagnostics:
```bash
narratio session validate 2026-04-04
```
Safe fix:
- use local mode (`audio_dir` or `audio_files`) or S3 mode (`audio_s3.prefix`), not both.
Relevant reference: [Session configuration](./config.md#session).
## `--artifacts` selection error
Symptom:
- unknown artifact key or invalid `--artifacts` usage.
Likely causes:
- key not defined in `pipeline.scriptorium.artifacts`;
- empty list entry (for example trailing comma);
- `run-stage` used with non-`analyze`/`publish` target.
Diagnostics:
```bash
narratio session artifacts 2026-04-04
```
Safe fix:
- provide only configured keys and use `--artifacts` with supported commands/stages.
Relevant reference: [CLI artifact selection](./cli.md).
## Bounded run prerequisite is unusable
Symptom:
- `run` or `session plan` reports that a prerequisite stage is absent or has a
pending, running, failed, stale, or interrupted status before the selected
start.
Likely cause:
- `--from` excludes upstream work that has not reached the terminal
`succeeded` or `skipped` state in the session manifest.
Diagnostics:
```bash
narratio session status 2026-04-04
narratio session plan 2026-04-04 --from render --through analyze
```
Safe fix:
- widen the bounded range to include the first reported stage, or recover that
stage explicitly with `run-stage` before retrying. The failed check does not
create a run record or modify the manifest. Narratio does not resume-validate
excluded prefix stages, and stages after `--through` are not prerequisites.
If prerequisite statuses are terminal but a selected stage reports a missing,
unsafe, or checksum-inconsistent artifact, repair the artifact at the stage
that owns it; do not edit the manifest to bypass the selected stage's concrete
input validation.
Relevant reference: [Operations: Stage Execution and Continuation Behavior](./operations.md#stage-execution-and-continuation-behavior).
## Notarius executable missing
Symptom:
- extraction fails while resolving or starting the Notarius executable.
Likely causes:
- `pipeline.notarius.binary` is not installed, executable, or on `PATH`;
- a configured executable path is wrong.
Safe fix:
- install a compatible Notarius release or correct the binary setting, then
rerun extraction.
Relevant references: [Notarius configuration](./config.md#notarius-output-entries)
and [Notarius integration](./integrations/notarius.md).
## Notarius exits nonzero
Symptom:
- extraction reports a Notarius exit error instead of a receipt.
Diagnostics:
- inspect `runs/{run_id}/extract/notarius.stderr.log`; stdout is reserved for
the receipt and is not merged with diagnostics.
Safe fix:
- correct the reported Notarius pipeline, input, provider, or configuration
failure and rerun extraction. Do not edit a staged output bundle into place.
After a failed replacement, an older immutable bundle may still exist even
though the current session manifest has no successful extraction payload. This
is expected audit state, not a signal to relink the old bundle manually.
Relevant reference: [Operations: Extraction Workflow](./operations.md#extraction-workflow).
## Prepared Notarius reference missing or inconsistent
Symptom:
- extraction or resume validation reports that a configured reference source is
unavailable, unsafe, empty, or checksum-inconsistent and recommends
`prepare --force`.
Likely causes:
- `prepare` has not run since the campaign/session stable input changed;
- the configured source file is missing;
- a prepared `inputs/` file or its manifest record was modified independently;
- a spell-catalog binding exists without an effective `spell_catalog_file`.
Diagnostics:
```bash
narratio session status 2026-04-04
narratio session validate 2026-04-04
```
Safe fix:
- correct the campaign/session input path, then refresh canonical prepared
evidence before extraction:
```bash
narratio run-stage prepare 2026-04-04 --force
```
Do not point Notarius directly at the original source path or edit the manifest
checksum. Relevant references: [Notarius reference configuration](./config.md#notarius-reference-bindings)
and [Operations: Extraction Workflow](./operations.md#extraction-workflow).
## Notarius reference selector or generated-handoff collision
Symptom:
- Notarius exits nonzero with an undeclared reference-slot, incompatible media,
or external/generated reference collision error.
Likely causes:
- a selector does not identify a slot declared by the selected Notarius target;
- a prepared file does not satisfy that slot's Notarius media contract; or
- a CLI binding attempts to replace a same-run generated D&D handoff.
Safe fix:
- compare external bindings with the selected Notarius pipeline's canonical
consumer documentation;
- keep only campaign-owned external slots on the CLI; and
- leave registry, scene, combat, and occurrence handoffs to Notarius pipeline
composition.
Narratio validates selector structure and prepared evidence, while Notarius
owns slot declarations, media compatibility, and generated-handoff conflicts.
Relevant reference: [Notarius integration](./integrations/notarius.md).
## Atomic Notarius promotion unsupported
Symptom:
- extraction fails with `atomic no-replace directory promotion is unsupported`
before a durable bundle or temporary promotion tree is created.
Likely cause:
- Narratio is running on an operating system other than Linux, macOS, or
Windows, where the required atomic no-replace directory primitive has not
been implemented and verified.
Safe fix:
- run extraction on Linux, macOS, or Windows. Do not replace the atomic commit
with a manual copy or move; the session manifest must never observe a partial
or overwritten bundle.
This is an extraction-specific platform boundary, not a support statement for
unrelated Narratio workflows. See
[Operations: Extraction Workflow](./operations.md#extraction-workflow).
## Notarius receipt or index incompatible
Symptom:
- extraction rejects the receipt schema, pipeline identity, bundle/index path,
lane descriptor, or payload path even though Notarius exited successfully.
Likely causes:
- Narratio and Notarius versions disagree on their consumer contract;
- the configured pipeline or lane constraints are stale;
- output paths escape the bundle or traverse symlinks.
Safe fix:
- compare installed Notarius output with the canonical Notarius contracts,
including receipt `index_file: index.json` and index management names
`manifest.json`, `rejected.json`, `warnings.json`, and `diagnostics.json`; align
`pipeline.notarius` constraints and rerun. Do not bypass confinement or schema
checks.
Relevant reference: [Notarius integration](./integrations/notarius.md).
## Required Notarius lane rejected or missing
Symptom:
- extraction fails because a configured lane is rejected, missing, duplicated,
or incompatible, including after a zero exit.
Safe fix:
- inspect the Notarius diagnostic log and bundle rejection/warning information;
- correct the Notarius module or the exact declared lane contract;
- remove an output declaration only if downstream consumers genuinely no longer
require that source, then rerun extraction.
Every configured output is required. Narratio does not promote a partial result.
## Extraction resume invalidated
Symptom:
- a previously successful extraction runs again during ordinary continuation.
Likely causes:
- the executable/config path, pipeline ID, timeout, working directory, or
configured output contracts changed;
- a configured prepared reference selector, source, path, checksum, or byte
size changed;
- the durable bundle, index, lane set, provenance, regular-file status, or
checksum no longer validates.
Safe fix:
- allow the automatic rerun after verifying the current configuration. Treat
an unsafe path or symlink error as filesystem corruption or tampering and
investigate it rather than replacing files manually.
## Notarius transitive configuration changed
Symptom:
- Notarius profiles, prompts, modules, imported files, or references changed,
but Narratio still considers the previous extraction resumable.
Safe fix:
```bash
narratio run-stage extract 2026-04-04 --force
```
Narratio fingerprints its invocation contract and prepared Narratio reference
identities, not the contents of other transitive Notarius inputs. Always force
extraction after changing those external inputs; downstream
successful stages are then marked stale normally.
Relevant reference: [Operations: Extraction Workflow](./operations.md#extraction-workflow).
## Analysis artifact evidence is not current
Symptom:
- ordinary continuation or `session plan` schedules one or more configured
artifacts even though a canonical output file exists; or
- publish reports a configured artifact source unavailable.
Likely causes:
- the per-artifact record is stale, missing, failed, unselected, malformed, or
from the legacy aggregate-only manifest contract;
- a configured prompt/profile, dependency, input identity, output path, or
effective variable changed; or
- the recorded output is missing, unsafe, empty, or has a size/checksum that no
longer matches its manifest evidence.
Diagnostics:
```bash
narratio session status 2026-04-04
narratio session artifacts 2026-04-04
narratio session plan 2026-04-04 --from analyze --through analyze
```
Safe fix:
- investigate unexpected path or checksum changes as possible tampering;
- otherwise let the selected analyze work rerun, or explicitly regenerate only
the affected targets; and
- never edit the fingerprint/checksum in the manifest or copy an old file into
the canonical path as a substitute for current evidence.
```bash
narratio analyze 2026-04-04 --artifacts session_recap
```
Relevant references: [Operations: Artifact Selection](./operations.md#artifact-selection)
and [Artifact Internals](./internal/artifacts.md#resolution-rules).
## Legacy aggregate analysis requires regeneration
Symptom:
- a manifest from an older Narratio version reports aggregate analyze success
and the old files are present, but configured artifact sources remain
unavailable.
Likely cause:
- the manifest has no supported per-artifact analyze state. Aggregate output
lists do not establish current configured-artifact authority.
Safe fix:
- regenerate the required artifacts. A partial selection makes only its
targets and prerequisites eligible for current state; unselected legacy
files intentionally remain unavailable. Run full analysis later when every
enabled configured artifact must become current.
```bash
narratio analyze 2026-04-04 --artifacts session_recap
narratio analyze 2026-04-04
```
After current records exist, inspect them and publish explicitly. Do not delete
the legacy files merely to influence selection; availability is manifest-owned.
Relevant references: [Operations: Stage Execution and Continuation Behavior](./operations.md#stage-execution-and-continuation-behavior)
and [Manifest Internals](./internal/manifest.md#analyze-owned-artifact-state).
## Scriptorium private input changed without a rerun
Symptom:
- a prompt, profile, imported configuration file, executable, or other input
loaded privately by Scriptorium changed, but Narratio still considers an
artifact current.
Likely cause:
- analysis fingerprints cover Narratio-observable semantic identities, not
executable contents or arbitrary files and transitive configuration that
Scriptorium loads behind its configured paths and identifiers.
Safe fix:
- explicitly force the affected target after changing an unobserved private
input. Force applies to explicit targets; current prerequisites remain
reusable unless selected themselves.
```bash
narratio analyze 2026-04-04 --artifacts session_recap
```
Relevant reference: [Analyze Internals](./internal/stage-analyze.md#invariants).
## Previous-session artifact input missing
Symptom:
- prepare/analyze fails due to missing required previous-session artifact cache input.
Likely causes:
- missing `session.previous_session_id`;
- previous artifact not restored/published for source session.
Diagnostics:
```bash
narratio session validate 2026-04-04
narratio session status 2026-04-04
```
Safe fix:
```bash
narratio session restore 2026-04-04
```
or rerun prepare after correcting session config:
```bash
narratio run-stage prepare 2026-04-04 --force
```
Relevant reference: [Operations: Restore Workflow](./operations.md#restore-workflow).
## Session lock conflict (`.lock`)
Symptom:
- command fails acquiring session lock.
Likely causes:
- another process is running for the same session;
- a process still holds the operating-system lock while it is shutting down.
Diagnostics:
```bash
ls -l {workspace.root}/work/{campaign}/{session_id}/.lock
ps aux | grep narratio
```
Safe fix:
- wait for active process completion;
- retry after an interrupted holder has exited; the kernel releases its lock
even though the `.lock` metadata file remains for inspection.
Relevant reference: [Operations: Local State Layout](./operations.md#local-state-layout).
## Restore conflict without `--force`
Symptom:
- restore fails with conflict count.
Likely cause:
- local durable files differ from remote restore sources.
Diagnostics:
```bash
narratio session restore 2026-04-04 --dry-run
```
Safe fix:
- review conflicts;
- rerun with `--force` only when remote state should overwrite local.
Relevant reference: [Operations: Restore Workflow](./operations.md#restore-workflow).
## Restore current-state discovery failure
Symptom:
- restore cannot find current pointer or current manifest.
Likely causes:
- no committed publish current state;
- storage credentials or connectivity failure.
Diagnostics:
```bash
narratio session status 2026-04-04
narratio session restore 2026-04-04 --dry-run
```
Safe fix:
- resolve storage/auth issue;
- republish from healthy local state if current pointer is missing.
Relevant reference: [Operations: Publish Workflow](./operations.md#publish-workflow).
## Publish output failure
Symptom:
- publish fails on missing required source, upload error, or commit write.
Likely causes:
- required source file not produced;
- lock/state expectations mismatch;
- remote storage failure.
Diagnostics:
```bash
narratio session artifacts 2026-04-04 --remote
narratio session status 2026-04-04
narratio run-stage publish 2026-04-04 --force
```
Safe fix:
- regenerate missing sources by rerunning prerequisite stages;
- correct publish source/destination rules;
- retry after storage failure is resolved.
Relevant reference: [Publish configuration](./config.md#publish-configuration-summary).
## Render markdown source missing
Symptom:
- analyze or publish fails because `narratio.transcript.final_markdown` or `narratio.transcript.final_trimmed_markdown` is unavailable.
Likely causes:
- render stage was not executed after transcript changes;
- render stage failed before producing canonical markdown outputs.
Diagnostics:
```bash
narratio session status 2026-04-04
```
Safe fix:
- rerun render and then retry downstream stage(s):
```bash
narratio run-stage render 2026-04-04 --force
narratio run-stage analyze 2026-04-04 --force
```
Relevant reference: [Operations: Stage Execution](./operations.md#stage-execution-and-continuation-behavior).
## Secrets or storage credential failure
Symptom:
- object-store command fails at initialization/auth.
Likely causes:
- invalid `pipeline.secrets.env_dir`;
- missing credential environment variables;
- invalid S3 endpoint/bucket settings.
Diagnostics:
```bash
ls -la /path/to/secrets_dir
env | sed 's/=.*//' | grep -E 'OBJECT_STORAGE|AWS|AUDITA|SCRIPTORIUM'
```
Safe fix:
- correct secret-file path and permissions;
- provide required env vars;
- keep secret values out of YAML.
Relevant reference: [Secrets](./config.md#secrets-handling).
## S3 audio prepare failure
Symptom:
- prepare fails listing/downloading session S3 audio.
Likely causes:
- incorrect `session.inputs.audio_s3.prefix`;
- no matching `.flac` objects;
- storage connectivity or permissions failure.
Diagnostics:
```bash
narratio session validate 2026-04-04
```
Safe fix:
- verify prefix contents and storage access;
- keep session audio mode consistent.
Relevant reference: [Operations](./operations.md).
## References
- [docs/cli.md](./cli.md)
- [docs/config.md](./config.md)
- [docs/operations.md](./operations.md)
- [docs/internal/stage-publish.md](./internal/stage-publish.md)

62
examples/README.md Normal file
View File

@@ -0,0 +1,62 @@
# Maintained Examples
These files are safe, copyable starting points for Narratio configuration and
input structure. Replace placeholder identifiers, storage names, integration
URLs, and paths for the target environment. Field meanings and defaults belong
in the [configuration reference](../docs/config.md).
## Pipeline Configuration
- [Minimal pipeline](pipeline.minimal.yml): campaign discovery plus the required
WhisperX URL.
- [Production-shaped pipeline](pipeline.production.yml): S3 storage, publish,
external tools, and configured Scriptorium artifacts.
- [Production/testing split bundle](production-testing/pipeline.yml): explicit
`conf.d` imports, a production default, and selectable production/testing
overlays. It also demonstrates canonical-party artifact families and a
testing-only disabled artifact. Validate it with:
```sh
narratio config validate --config examples/production-testing/pipeline.yml --campaign-file examples/campaigns/sample-campaign/campaign.yml
narratio config diff production testing --config examples/production-testing/pipeline.yml --campaign-file examples/campaigns/sample-campaign/campaign.yml
```
`config show` and `config sources` accept the same selectors and remain
read-only.
- [Full annotated pipeline](pipeline.full.annotated.yml): every implemented
pipeline section with explanatory comments.
- [Extraction subset pipeline](pipeline.extraction-subset.yml): a focused
Scriptorium artifact consuming only three declared Notarius lanes.
The existing `internal/config` example test loads and validates each pipeline
with the sample campaign and a compatible local- or S3-audio session.
## Campaign And Session Configuration
- [Sample campaign](campaigns/sample-campaign/campaign.yml), its
[session template](campaigns/sample-campaign/session.template.yml), and its
adjacent stable inputs provide a complete campaign directory shape.
- [Local-audio session](session.local-audio.yml) and
[S3-audio session](session.s3-audio.yml) are concrete session files.
- [Session template](session.template.yml) and the campaign-local equivalent
demonstrate the narrow placeholder syntax consumed by `session init`; they
are templates, not runtime session files.
## Input Fixtures
- [Speakers](speakers.yml), [autocorrect](autocorrect.yml), and
[glossary](glossary.yml) show the standalone input shapes.
- The sample campaign references its local
[speakers](campaigns/sample-campaign/speakers.yml),
[autocorrect](campaigns/sample-campaign/autocorrect.yml),
[glossary](campaigns/sample-campaign/glossary.yml), and canonical
[party](campaigns/sample-campaign/party.yml) fixture, plus an optional
[spell-catalog overlay](campaigns/sample-campaign/spell_catalog.json) that
follows the Notarius v0.6 contract. Narratio derives the players projection
from this party source; the campaign deliberately has no `players_file`.
- [Sample speaker audio](audio/sample-speaker.flac) is a text placeholder that
reserves the expected filename and directory shape. Replace it with a real
FLAC file before running transcription.
The examples contain environment-variable names but no credential values. They
use fictional campaign content and reserved example domains.

View File

@@ -0,0 +1 @@
[]

View File

@@ -0,0 +1,8 @@
campaign_id: sample-campaign
session_template_file: ./session.template.yml
inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
party_file: ./party.yml
spell_catalog_file: ./spell_catalog.json

View File

@@ -0,0 +1 @@
[]

View File

@@ -0,0 +1,26 @@
schema_version: narratio.party.v1
characters:
arannis:
player:
name: Rowan Hale
character:
name: Arannis
alias:
- Ari
- The Grey Owl
classes:
- name: wizard
level: 8
brenna:
player:
name: Rowan Hale
character:
name: Brenna
alias:
- Shield of Dawn
classes:
- name: paladin
level: 6
- name: warlock
level: 2

View File

@@ -0,0 +1,3 @@
session_id: "{{ session_id }}"
inputs:
audio_dir: ./audio

View File

@@ -0,0 +1,5 @@
match:
- speaker: "Example Speaker"
match:
- "Example_Speaker"
- "Example"

View File

@@ -0,0 +1,18 @@
{
"schema_version": "notarius.dnd.spell-catalog-overlay.v1",
"catalogs": [
{
"id": "narratio.sample-campaign",
"ruleset": "dnd-5e-2014",
"source": {
"title": "Narratio sample campaign spell names"
},
"spells": [
{
"name": "Aegis of Emberfall",
"aliases": ["Emberfall Aegis"]
}
]
}
]
}

View File

@@ -0,0 +1,59 @@
# Purpose-specific extraction example: a Scriptorium session brief consumes
# only the three Notarius lanes it needs.
campaigns:
root: /usr/local/share/narratio/campaigns
default_campaign_id: sample-campaign
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"
notarius:
enabled: true
binary: notarius
config_path: /usr/local/etc/notarius/config.yml
pipeline_id: dnd-session
timeout: 3h
references:
glossary: narratio.input.glossary
party: narratio.input.party
players: narratio.input.players
spell_catalog: narratio.input.spell_catalog
outputs:
npc_registry:
lane_id: npc-registry
media_type: application/json
schema_id: notarius.dnd.npc_registry
schema_version: v1
module_key: dnd/npc-registry
location_registry:
lane_id: location-registry
media_type: application/json
schema_id: notarius.dnd.location_registry
schema_version: v1
module_key: dnd/location-registry
scene_descriptions:
lane_id: scene-descriptions
media_type: application/json
schema_id: notarius.dnd.scene_descriptions
schema_version: v1
module_key: dnd/scene-descriptions
scriptorium:
binary: scriptorium
config_path: /usr/local/etc/scriptorium/config.yml
artifacts:
session_brief:
enabled: true
prompt_id: dnd.session_brief
output_path: artifacts/session_brief.md
inputs:
npcs:
source: narratio.extraction.npc_registry
required: true
locations:
source: narratio.extraction.location_registry
required: true
scenes:
source: narratio.extraction.scene_descriptions
required: true

View File

@@ -0,0 +1,271 @@
# Full annotated pipeline example for implemented Narratio config fields.
# Values are safe placeholders and must be adapted per environment.
workspace:
# Optional: defaults to /var/lib/narratio.
root: /var/lib/narratio/workspace
# Optional: remove run-scoped workdir after successful publish commit.
cleanup_after_publish: false
# Optional: local secret file loader (directory of ENV_VAR_NAME files).
# secrets:
# env_dir: ./secrets
storage:
# Defaults to "local". Use "s3" explicitly for publish + S3 audio workflows.
backend: s3
s3:
# Required when using S3 audio or S3 publish uploads.
bucket: my-dnd-archive
# Optional; defaults to "dnd".
root_prefix: dnd
# Optional region/endpoint settings.
region: us-east-1
endpoint: ""
force_path_style: false
# Optional; defaults shown explicitly.
access_key_id_env: OBJECT_STORAGE_KEY_ID
secret_access_key_env: OBJECT_STORAGE_KEY
campaigns:
# Optional; defaults to /usr/local/share/narratio/campaigns.
root: /usr/local/share/narratio/campaigns
# Optional command default when --campaign is omitted.
default_campaign_id: sample-campaign
spool:
# Optional; defaults to /var/spool/narratio.
root: /var/spool/narratio
# Optional cleanup of run-scoped spool audio after successful publish commit.
delete_audio_after_publish: false
publish:
# Optional booleans; defaults are true.
enabled: true
upload_run: true
# Optional publish output rules; sources use Narratio artifact source IDs.
outputs:
- source: narratio.transcript.final_trimmed
dest: transcripts/final.trimmed.json
required: true
- source: narratio.transcript.final_markdown
dest: transcripts/final.md
required: true
- source: narratio.transcript.final_trimmed_markdown
dest: transcripts/final.trimmed.md
required: true
- source: narratio.artifact.session_recap
dest: artifacts/session_recap.md
required: true
- source: narratio.artifact.player_handout
dest: artifacts/player_handout.md
required: false
# Extraction lanes publish only when named explicitly; the bundle and index
# are never implicit publish sources.
- source: narratio.extraction.npc_registry
dest: artifacts/extraction/npc-registry.json
required: true
whisperx:
# Required.
transcribe_url: "https://transcription.example.com/transcribe"
# Optional overrides; defaults shown explicitly.
language: en
timeout: 30m
retries: 3
retry_delay: 2s
concurrency: 2
seriatim:
# Optional overrides; defaults shown explicitly.
binary: seriatim
timeout: 10m
output_schema: seriatim-intermediate
coalesce_gap: 3.0
report: true
env:
# Optional advanced tuning; set only when needed.
overlap_word_run_gap: 1.0
overlap_word_run_reorder_window: 1.0
backchannel_max_duration: 2.0
filler_max_duration: 1.25
audita:
# Optional overrides; defaults shown explicitly where applicable.
binary: audita
timeout: 3h
llm_api_key_env: AUDITA_LLM_API_KEY
modules: [glossary, homophones, spoken_word, grammar]
base_url: ""
model: ""
total_llm_concurrency: 2
proposal_llm_concurrency: 1
validation_model: ""
validation_llm_concurrency: 1
transcript_description: ""
config_path: /usr/local/etc/audita/config.yml
output_schema: audita-v1
work_dir_retention: auto
report: true
normalize:
# Optional; defaults shown explicitly.
output_path: transcripts/final.json
output_schema: seriatim-intermediate
report: true
trim:
# Optional; defaults shown explicitly.
enabled: true
output_path: transcripts/final.trimmed.json
bounds:
prompt_id: dnd.session_bounds
profile_id: ""
transcript_input_name: transcript
output_path: artifacts/session_bounds.json
timeout: 10m
render_debug: false
seriatim:
report: false
notarius:
# Optional structured extraction between trim and render.
enabled: true
binary: notarius
config_path: /usr/local/etc/notarius/config.yml
pipeline_id: dnd-session
timeout: 3h
working_directory: /usr/local/etc/notarius
# External campaign references use prepared Narratio source IDs. Omit an
# optional binding when the selected Notarius pipeline does not need it.
references:
glossary: narratio.input.glossary
party: narratio.input.party
players: narratio.input.players
spell_catalog: narratio.input.spell_catalog
# Each key creates source narratio.extraction.<key>. These constraints match
# the current Notarius D&D lane contracts; update them with Notarius.
outputs:
item_registry:
lane_id: item-registry
media_type: application/json
schema_id: notarius.dnd.item_registry
schema_version: v1
module_key: dnd/item-registry
npc_registry:
lane_id: npc-registry
media_type: application/json
schema_id: notarius.dnd.npc_registry
schema_version: v1
module_key: dnd/npc-registry
location_registry:
lane_id: location-registry
media_type: application/json
schema_id: notarius.dnd.location_registry
schema_version: v1
module_key: dnd/location-registry
scene_descriptions:
lane_id: scene-descriptions
media_type: application/json
schema_id: notarius.dnd.scene_descriptions
schema_version: v1
module_key: dnd/scene-descriptions
item_occurrences:
lane_id: item-occurrences
media_type: application/json
schema_id: notarius.dnd.item_occurrences
schema_version: v1
module_key: dnd/item-occurrences
spells:
lane_id: spells
media_type: application/json
schema_id: notarius.dnd.spells
schema_version: v1
module_key: dnd/spells
combat_turns:
lane_id: combat-turns
media_type: application/json
schema_id: notarius.dnd.combat_turns
schema_version: v1
module_key: dnd/combat-turns
npc_occurrences:
lane_id: npc-occurrences
media_type: application/json
schema_id: notarius.dnd.npc_occurrences
schema_version: v1
module_key: dnd/npc-occurrences
location_occurrences:
lane_id: location-occurrences
media_type: application/json
schema_id: notarius.dnd.location_occurrences
schema_version: v1
module_key: dnd/location-occurrences
enemy_events:
lane_id: enemy-events
media_type: application/json
schema_id: notarius.dnd.enemy_events
schema_version: v1
module_key: dnd/enemy-events
scriptorium:
binary: scriptorium
config_path: /usr/local/etc/scriptorium/config.yml
timeout: 10m
render_debug: false
artifacts:
# Configured artifact keys map to source IDs narratio.artifact.<key>.
session_recap:
enabled: true
prompt_id: dnd.session_recap
profile_id: local-fast
output_path: artifacts/session_recap.md
timeout: 10m
inputs:
transcript:
source: narratio.transcript.final_trimmed
required: true
previous_recap:
source: narratio.previous_session.artifact.session_recap
required: false
players:
source: narratio.input.players
required: true
party:
source: narratio.input.party
required: true
glossary:
source: narratio.input.glossary
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: session_recap
# Example dependent artifact:
# - depends_on entries use artifact keys.
# - narratio.artifact.<key> sources require matching depends_on membership.
player_handout:
enabled: true
depends_on:
- session_recap
prompt_id: dnd.player_handout
profile_id: local-fast
output_path: artifacts/player_handout.md
timeout: 10m
inputs:
recap:
source: narratio.artifact.session_recap
required: true
transcript:
source: narratio.transcript.final_trimmed
required: true
vars:
session_id: true
campaign_name: true
output_kind: player_handout
notification:
# No delivery provider is currently implemented.
mode: noop

View File

@@ -1,109 +1,6 @@
workspace:
root: ./tmp/narratio-workspace
storage:
backend: local
campaigns:
root: /usr/local/share/narratio/campaigns
default_campaign_id: sample-campaign
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"
language: "en"
timeout: "30m"
retries: 3
retry_delay: "2s"
concurrency: 2
seriatim:
binary: "seriatim"
timeout: "10m"
output_schema: "seriatim-intermediate"
coalesce_gap: 3.0
report: true
env:
overlap_word_run_gap: 1.0
overlap_word_run_reorder_window: 1.0
backchannel_max_duration: 2.0
filler_max_duration: 1.25
audita:
binary: "audita"
timeout: "3h"
llm_api_key_env: "AUDITA_LLM_API_KEY"
modules:
- glossary
- homophones
- glossary
- spoken_word
- grammar
- homophones
- glossary
base_url: "https://openrouter.ai/api/v1"
model: "openrouter/google/gemma-4-31b-it"
llm_concurrency: 1
validation_model: ""
validation_llm_concurrency: 1
report: true
normalize:
# Session-workdir-relative when not absolute.
output_path: "transcripts/normalized.json"
output_schema: "seriatim-intermediate"
report: true
trim:
enabled: true
# Session-workdir-relative when not absolute.
output_path: "transcripts/trimmed.json"
bounds:
prompt_id: "dnd_session.bounds"
# Empty means use prompt default profile.
profile_id: ""
transcript_input_name: "transcript"
output_path: "artifacts/session_bounds.json"
timeout: "10m"
render_debug: false
render_output_path: "artifacts/session_bounds.render.json"
seriatim:
report: false
scriptorium:
binary: "scriptorium"
config_path: "/etc/scriptorium/config.yml"
timeout: "10m"
render_debug: false
artifacts:
session_recap:
enabled: true
prompt_id: "dnd.session_recap"
profile_id: "local-quality"
output_path: "artifacts/session_recap.md"
timeout: "10m"
# Optional per-artifact override of global scriptorium.render_debug.
# render_debug: true
inputs:
transcript:
# Available transcript sources:
# - trimmed_transcript (recommended for session_recap)
# - normalized_transcript (recommended for future full-session analysis)
# - processed_transcript (raw Audita-polished output)
source: "trimmed_transcript"
required: true
previous_recap:
source: "previous_session_artifact"
artifact: "session_recap"
# Optional: set when previous recap is available.
path: ""
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: "session_recap"
analyzer:
timeout: 20m
artifacts:
output_dir: artifacts
notification:
timeout: 10s

View File

@@ -0,0 +1,128 @@
workspace:
root: /var/lib/narratio/workspace
cleanup_after_publish: true
storage:
backend: s3
s3:
bucket: my-dnd-archive
root_prefix: dnd
region: us-east-1
access_key_id_env: OBJECT_STORAGE_KEY_ID
secret_access_key_env: OBJECT_STORAGE_KEY
campaigns:
root: /usr/local/share/narratio/campaigns
default_campaign_id: sample-campaign
spool:
root: /var/spool/narratio
delete_audio_after_publish: true
publish:
enabled: true
upload_run: true
outputs:
- source: narratio.transcript.final_trimmed
dest: transcripts/final.trimmed.json
required: true
- source: narratio.transcript.final_markdown
dest: transcripts/final.md
required: true
- source: narratio.transcript.final_trimmed_markdown
dest: transcripts/final.trimmed.md
required: true
- source: narratio.artifact.session_recap
dest: artifacts/session_recap.md
required: true
- source: narratio.artifact.player_handout
dest: artifacts/player_handout.md
required: false
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"
language: en
timeout: 45m
retries: 3
retry_delay: 3s
concurrency: 2
seriatim:
binary: seriatim
timeout: 10m
output_schema: seriatim-intermediate
coalesce_gap: 3.0
report: true
audita:
binary: audita
timeout: 3h
llm_api_key_env: AUDITA_LLM_API_KEY
modules: [glossary, homophones, spoken_word, grammar]
output_schema: audita-v1
work_dir_retention: auto
total_llm_concurrency: 2
proposal_llm_concurrency: 1
validation_llm_concurrency: 1
report: true
normalize:
output_path: transcripts/final.json
output_schema: seriatim-intermediate
report: true
scriptorium:
binary: scriptorium
config_path: /usr/local/etc/scriptorium/config.yml
timeout: 10m
render_debug: false
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
profile_id: local-fast
output_path: artifacts/session_recap.md
timeout: 10m
inputs:
transcript:
source: narratio.transcript.final_trimmed
required: true
previous_recap:
source: narratio.previous_session.artifact.session_recap
required: false
players:
source: narratio.input.players
required: true
party:
source: narratio.input.party
required: true
glossary:
source: narratio.input.glossary
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: session_recap
player_handout:
enabled: true
depends_on:
- session_recap
prompt_id: dnd.player_handout
profile_id: local-fast
output_path: artifacts/player_handout.md
timeout: 10m
inputs:
recap:
source: narratio.artifact.session_recap
required: true
transcript:
source: narratio.transcript.final_trimmed
required: true
vars:
session_id: true
output_kind: player_handout
notification:
mode: noop

View File

@@ -0,0 +1,32 @@
scriptorium:
binary: scriptorium
config_path: ./scriptorium/config.yml
artifact_families:
character_meta:
enabled: true
for_each: party.characters
prompt_id: dnd.character_meta
output_path_pattern: artifacts/characters/{character_id}/meta.md
member_vars:
character_id: character_id
character_name: character.name
player_name: player.name
class_summary: character.class_summary
character_items:
enabled: true
for_each: party.characters
prompt_id: dnd.character_items
output_path_pattern: artifacts/characters/{character_id}/items.md
member_dependencies: [character_meta]
inputs:
character_meta:
source: narratio.member_artifact.character_meta
required: true
member_vars:
character_id: character_id
character_name: character.name
aliases: character.alias_summary
publish:
enabled: true
required: false
dest_pattern: artifacts/characters/{character_id}/items.md

View File

@@ -0,0 +1,12 @@
workspace:
root: ./workspace
campaigns:
root: ../../campaigns
default_campaign_id: sample-campaign
cache:
root: ./cache
spool:
root: ./spool

View File

@@ -0,0 +1,7 @@
publish:
enabled: true
upload_run: false
outputs:
- source: narratio.transcript.final_markdown
dest: transcripts/final.md
required: true

View File

@@ -0,0 +1,2 @@
storage:
backend: local

View File

@@ -0,0 +1,16 @@
whisperx:
transcribe_url: https://transcription.example.com/transcribe
language: en
seriatim:
binary: seriatim
output_schema: seriatim-intermediate
audita:
binary: audita
modules: [glossary, grammar]
output_schema: audita-v1
normalize:
output_path: transcripts/final.json
output_schema: seriatim-intermediate

View File

@@ -0,0 +1,15 @@
# Copyable production/testing pipeline entry point. Every fragment is named
# explicitly; Narratio never scans conf.d automatically.
composition:
imports:
- conf.d/platform.yml
- conf.d/storage.yml
- conf.d/transcript.yml
- conf.d/artifacts.yml
- conf.d/publish.yml
default_profile: production
profiles:
production:
overlay: profiles/production.yml
testing:
overlay: profiles/testing.yml

View File

@@ -0,0 +1,10 @@
audita:
model: narratio-production-model-placeholder
validation_model: narratio-production-validator-placeholder
scriptorium:
artifact_families:
character_meta:
profile_id: production-placeholder
character_items:
profile_id: production-placeholder

View File

@@ -0,0 +1,14 @@
audita:
model: narratio-testing-model-placeholder
validation_model: narratio-testing-validator-placeholder
scriptorium:
artifacts:
testing_notes:
enabled: false
output_path: artifacts/testing-notes.md
artifact_families:
character_meta:
profile_id: testing-placeholder
character_items:
profile_id: testing-placeholder

View File

@@ -0,0 +1,5 @@
session_id: 2026-05-03
date: 2026-05-03
title: Sample Session
inputs:
audio_dir: ./audio

View File

@@ -1,10 +0,0 @@
session_id: 2026-05-03
campaign: sample-campaign
date: 2026-05-03
title: Sample Session
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml

View File

@@ -0,0 +1,6 @@
session_id: 2026-05-03
date: 2026-05-03
title: Sample Session
inputs:
audio_s3:
prefix: audio/

View File

@@ -0,0 +1,3 @@
session_id: "{{ session_id }}"
inputs:
audio_dir: ./audio

View File

@@ -1,5 +1,5 @@
match:
- speaker: "Eric Rakestraw"
- speaker: "Example Speaker"
match:
- "Eric_Rakestraw"
- "Eric"
- "Example_Speaker"
- "Example"

26
go.mod
View File

@@ -2,4 +2,28 @@ module gitea.maximumdirect.net/eric/narratio
go 1.25.0
require gopkg.in/yaml.v3 v3.0.1
require (
github.com/aws/aws-sdk-go-v2/config v1.32.17
github.com/aws/aws-sdk-go-v2/credentials v1.19.16
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0
github.com/aws/smithy-go v1.25.1
golang.org/x/sys v0.47.0
gopkg.in/yaml.v3 v3.0.1
)
require (
github.com/aws/aws-sdk-go-v2 v1.41.7 // indirect
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10 // indirect
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.23 // indirect
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23 // indirect
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23 // indirect
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.24 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.9 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.15 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.23 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.23 // indirect
github.com/aws/aws-sdk-go-v2/service/signin v1.0.11 // indirect
github.com/aws/aws-sdk-go-v2/service/sso v1.30.17 // indirect
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.35.21 // indirect
github.com/aws/aws-sdk-go-v2/service/sts v1.42.1 // indirect
)

38
go.sum
View File

@@ -1,3 +1,41 @@
github.com/aws/aws-sdk-go-v2 v1.41.7 h1:DWpAJt66FmnnaRIOT/8ASTucrvuDPZASqhhLey6tLY8=
github.com/aws/aws-sdk-go-v2 v1.41.7/go.mod h1:4LAfZOPHNVNQEckOACQx60Y8pSRjIkNZQz1w92xpMJc=
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10 h1:gx1AwW1Iyk9Z9dD9F4akX5gnN3QZwUB20GGKH/I+Rho=
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10/go.mod h1:qqY157uZoqm5OXq/amuaBJyC9hgBCBQnsaWnPe905GY=
github.com/aws/aws-sdk-go-v2/config v1.32.17 h1:FpL4/758/diKwqbytU0prpuiu60fgXKUWCpDJtApclU=
github.com/aws/aws-sdk-go-v2/config v1.32.17/go.mod h1:OXqUMzgXytfoF9JaKkhrOYsyh72t9G+MJH8mMRaexOE=
github.com/aws/aws-sdk-go-v2/credentials v1.19.16 h1:r3RJBuU7X9ibt8RHbMjWE6y60QbKBiII6wSrXnapxSU=
github.com/aws/aws-sdk-go-v2/credentials v1.19.16/go.mod h1:6cx7zqDENJDbBIIWX6P8s0h6hqHC8Avbjh9Dseo27ug=
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.23 h1:UuSfcORqNSz/ey3VPRS8TcVH2Ikf0/sC+Hdj400QI6U=
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.23/go.mod h1:+G/OSGiOFnSOkYloKj/9M35s74LgVAdJBSD5lsFfqKg=
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23 h1:GpT/TrnBYuE5gan2cZbTtvP+JlHsutdmlV2YfEyNde0=
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23/go.mod h1:xYWD6BS9ywC5bS3sz9Xh04whO/hzK2plt2Zkyrp4JuA=
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23 h1:bpd8vxhlQi2r1hiueOw02f/duEPTMK59Q4QMAoTTtTo=
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23/go.mod h1:15DfR2nw+CRHIk0tqNyifu3G1YdAOy68RftkhMDDwYk=
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.24 h1:OQqn11BtaYv1WLUowvcA30MpzIu8Ti4pcLPIIyoKZrA=
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.24/go.mod h1:X5ZJyfwVrWA96GzPmUCWFQaEARPR7gCrpq2E92PJwAE=
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.9 h1:FLudkZLt5ci0ozzgkVo8BJGwvqNaZbTWb3UcucAateA=
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.9/go.mod h1:w7wZ/s9qK7c8g4al+UyoF1Sp/Z45UwMGcqIzLWVQHWk=
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.15 h1:ieLCO1JxUWuxTZ1cRd0GAaeX7O6cIxnwk7tc1LsQhC4=
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.15/go.mod h1:e3IzZvQ3kAWNykvE0Tr0RDZCMFInMvhku3qNpcIQXhM=
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.23 h1:pbrxO/kuIwgEsOPLkaHu0O+m4fNgLU8B3vxQ+72jTPw=
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.23/go.mod h1:/CMNUqoj46HpS3MNRDEDIwcgEnrtZlKRaHNaHxIFpNA=
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.23 h1:03xatSQO4+AM1lTAbnRg5OK528EUg744nW7F73U8DKw=
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.23/go.mod h1:M8l3mwgx5ToK7wot2sBBce/ojzgnPzZXUV445gTSyE8=
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0 h1:etqBTKY581iwLL/H/S2sVgk3C9lAsTJFeXWFDsDcWOU=
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0/go.mod h1:L2dcoOgS2VSgbPLvpak2NyUPsO1TBN7M45Z4H7DlRc4=
github.com/aws/aws-sdk-go-v2/service/signin v1.0.11 h1:TdJ+HdzOBhU8+iVAOGUTU63VXopcumCOF1paFulHWZc=
github.com/aws/aws-sdk-go-v2/service/signin v1.0.11/go.mod h1:R82ZRExE/nheo0N+T8zHPcLRTcH8MGsnR3BiVGX0TwI=
github.com/aws/aws-sdk-go-v2/service/sso v1.30.17 h1:7byT8HUWrgoRp6sXjxtZwgOKfhss5fW6SkLBtqzgRoE=
github.com/aws/aws-sdk-go-v2/service/sso v1.30.17/go.mod h1:xNWknVi4Ezm1vg1QsB/5EWpAJURq22uqd38U8qKvOJc=
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.35.21 h1:+1Kl1zx6bWi4X7cKi3VYh29h8BvsCoHQEQ6ST9X8w7w=
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.35.21/go.mod h1:4vIRDq+CJB2xFAXZ+YgGUTiEft7oAQlhIs71xcSeuVg=
github.com/aws/aws-sdk-go-v2/service/sts v1.42.1 h1:F/M5Y9I3nwr2IEpshZgh1GeHpOItExNM9L1euNuh/fk=
github.com/aws/aws-sdk-go-v2/service/sts v1.42.1/go.mod h1:mTNxImtovCOEEuD65mKW7DCsL+2gjEH+RPEAexAzAio=
github.com/aws/smithy-go v1.25.1 h1:J8ERsGSU7d+aCmdQur5Txg6bVoYelvQJgtZehD12GkI=
github.com/aws/smithy-go v1.25.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc=
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=

View File

@@ -1,40 +0,0 @@
package analyzer
import "context"
// NoopRunner is a deterministic no-op analyzer adapter.
type NoopRunner struct{}
// Run returns the requested output path with placeholder metadata.
func (n *NoopRunner) Run(ctx context.Context, req AnalyzeRequest) (AnalyzeResult, error) {
if err := ctx.Err(); err != nil {
return AnalyzeResult{}, err
}
return AnalyzeResult{ArtifactPath: req.OutputPath, Metadata: map[string]any{"placeholder": true}}, nil
}
// FakeRunner captures analyze requests and returns deterministic responses.
type FakeRunner struct {
Requests []AnalyzeRequest
Err error
Result AnalyzeResult
}
// Run records request and returns configured response.
func (f *FakeRunner) Run(ctx context.Context, req AnalyzeRequest) (AnalyzeResult, error) {
if err := ctx.Err(); err != nil {
return AnalyzeResult{}, err
}
f.Requests = append(f.Requests, req)
if f.Err != nil {
return AnalyzeResult{}, f.Err
}
res := f.Result
if res.ArtifactPath == "" {
res.ArtifactPath = req.OutputPath
}
if res.Metadata == nil {
res.Metadata = map[string]any{"fake": true}
}
return res, nil
}

View File

@@ -1,31 +0,0 @@
package analyzer
import (
"context"
"errors"
"testing"
)
func TestFakeRunnerCapturesRequestAndReturnsPath(t *testing.T) {
fake := &FakeRunner{}
req := AnalyzeRequest{ArtifactType: "session-log", OutputPath: "artifacts/session-log.md"}
res, err := fake.Run(context.Background(), req)
if err != nil {
t.Fatalf("Run() error = %v", err)
}
if len(fake.Requests) != 1 || fake.Requests[0].ArtifactType != "session-log" {
t.Fatalf("requests = %#v, want captured request", fake.Requests)
}
if res.ArtifactPath != req.OutputPath {
t.Fatalf("artifact path = %q, want %q", res.ArtifactPath, req.OutputPath)
}
}
func TestFakeRunnerError(t *testing.T) {
fake := &FakeRunner{Err: errors.New("boom")}
_, err := fake.Run(context.Background(), AnalyzeRequest{})
if err == nil {
t.Fatal("expected error, got nil")
}
}

View File

@@ -1,28 +0,0 @@
// Package analyzer declares the adapter contract for artifact analysis generation.
package analyzer
import "context"
// TODO: implement analyzer integration once the analyzer contract is finalized.
// Runner is the adapter boundary for analyzer invocations.
type Runner interface {
Run(ctx context.Context, req AnalyzeRequest) (AnalyzeResult, error)
}
// AnalyzeRequest describes one analyzer artifact generation request.
type AnalyzeRequest struct {
ArtifactType string
ProcessedTranscriptPath string
ContextReferences []string
OutputPath string
GeneratedConfigPath string
StdoutLogPath string
StderrLogPath string
}
// AnalyzeResult describes analyzer output.
type AnalyzeResult struct {
ArtifactPath string
Metadata map[string]any
}

View File

@@ -4,10 +4,10 @@ import (
"context"
"encoding/json"
"fmt"
"os"
"path/filepath"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
)
// NoopRunner is a deterministic no-op audita adapter.
@@ -84,17 +84,17 @@ func materializePlaceholders(req PolishRequest) error {
"merged_transcript_path": req.MergedTranscriptPath,
"output_path": req.OutputProcessedPath,
}
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
}
}
if req.StdoutLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("audita noop/fake stdout placeholder\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("audita noop/fake stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
}
}
if req.StderrLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("audita noop/fake stderr placeholder\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("audita noop/fake stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
}
}
@@ -117,14 +117,14 @@ func writeJSONIfRequested(path string, payload any) error {
if path == "" {
return nil
}
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
if err := fileops.EnsureWorkspaceDirectory(filepath.Dir(path)); err != nil {
return fmt.Errorf("create parent directory %q: %w", filepath.Dir(path), err)
}
data, err := json.Marshal(payload)
if err != nil {
return fmt.Errorf("marshal placeholder json for %q: %w", path, err)
}
if err := subprocess.WriteFileAtomic(path, data, 0o644); err != nil {
if err := subprocess.WriteFileAtomic(path, data, fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write placeholder json %q: %w", path, err)
}
return nil

View File

@@ -14,7 +14,7 @@ func TestFakeRunnerCapturesRequestAndReturnsPath(t *testing.T) {
dir := t.TempDir()
req := PolishRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "audita.yml"),
OutputProcessedPath: filepath.Join(dir, "transcripts", "processed.json"),
OutputProcessedPath: filepath.Join(dir, "transcripts", "polished.json"),
StdoutLogPath: filepath.Join(dir, "logs", "audita.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "audita.stderr.log"),
}

View File

@@ -6,8 +6,6 @@ import (
"time"
)
// TODO: implement a real Audita subprocess/service adapter.
// Runner is the adapter boundary for audita polish invocations.
type Runner interface {
Run(ctx context.Context, req PolishRequest) (PolishResult, error)
@@ -15,19 +13,15 @@ type Runner interface {
// PolishRequest describes an audita invocation.
type PolishRequest struct {
GeneratedConfigPath string
MergedTranscriptPath string
OutputProcessedPath string
GlossaryPath string
ReportPath string
WorkDir string
Modules []string
BaseURL string
Model string
ValidationModel string
ValidationLLMConcurrency *int
StdoutLogPath string
StderrLogPath string
GeneratedConfigPath string
MergedTranscriptPath string
OutputProcessedPath string
GlossaryPath string
ReportPath string
WorkDir string
Modules []string
StdoutLogPath string
StderrLogPath string
}
// PolishResult describes a polish output.

View File

@@ -11,8 +11,15 @@ import (
"time"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
)
// MaxProcessedOutputBytes bounds Audita's processed-transcript JSON result.
const MaxProcessedOutputBytes int64 = 64 * 1024 * 1024
// MaxReportOutputBytes bounds Audita's optional report JSON result.
const MaxReportOutputBytes int64 = 16 * 1024 * 1024
// SubprocessRunnerConfig defines deterministic settings for Audita CLI execution.
type SubprocessRunnerConfig struct {
Binary string
@@ -21,7 +28,12 @@ type SubprocessRunnerConfig struct {
Modules []string
BaseURL string
Model string
LLMConcurrency *int
TranscriptDescription string
ConfigPath string
OutputSchema string
WorkDirRetention string
TotalLLMConcurrency *int
ProposalLLMConcurrency *int
ValidationModel string
ValidationLLMConcurrency *int
Report bool
@@ -35,7 +47,12 @@ type SubprocessRunner struct {
modules []string
baseURL string
model string
llmConcurrency *int
transcriptDescription string
configPath string
outputSchema string
workDirRetention string
totalLLMConcurrency *int
proposalLLMConcurrency *int
validationModel string
validationLLMConcurrency *int
report bool
@@ -49,7 +66,12 @@ func NewSubprocessRunnerFromConfigValues(
modules []string,
baseURL string,
model string,
llmConcurrency *int,
transcriptDescription string,
configPath string,
outputSchema string,
workDirRetention string,
totalLLMConcurrency *int,
proposalLLMConcurrency *int,
validationModel string,
validationLLMConcurrency *int,
report bool,
@@ -68,7 +90,12 @@ func NewSubprocessRunnerFromConfigValues(
Modules: modules,
BaseURL: baseURL,
Model: model,
LLMConcurrency: llmConcurrency,
TranscriptDescription: transcriptDescription,
ConfigPath: configPath,
OutputSchema: outputSchema,
WorkDirRetention: workDirRetention,
TotalLLMConcurrency: totalLLMConcurrency,
ProposalLLMConcurrency: proposalLLMConcurrency,
ValidationModel: validationModel,
ValidationLLMConcurrency: validationLLMConcurrency,
Report: report,
@@ -83,33 +110,39 @@ func NewSubprocessRunner(cfg SubprocessRunnerConfig) (*SubprocessRunner, error)
if cfg.Timeout <= 0 {
return nil, fmt.Errorf("audita timeout must be > 0")
}
if len(cfg.Modules) == 0 {
return nil, fmt.Errorf("audita modules must include at least one module")
}
for i, module := range cfg.Modules {
if strings.TrimSpace(module) == "" {
return nil, fmt.Errorf("audita module at index %d is empty", i)
}
}
if strings.TrimSpace(cfg.BaseURL) == "" {
return nil, fmt.Errorf("audita base url is required")
}
u, err := url.Parse(cfg.BaseURL)
if err != nil || u.Scheme == "" || u.Host == "" {
if err != nil {
return nil, fmt.Errorf("audita base url %q is invalid: %w", cfg.BaseURL, err)
if strings.TrimSpace(cfg.BaseURL) != "" {
u, err := url.Parse(cfg.BaseURL)
if err != nil || u.Scheme == "" || u.Host == "" {
if err != nil {
return nil, fmt.Errorf("audita base url %q is invalid: %w", cfg.BaseURL, err)
}
return nil, fmt.Errorf("audita base url %q is invalid", cfg.BaseURL)
}
return nil, fmt.Errorf("audita base url %q is invalid", cfg.BaseURL)
}
if strings.TrimSpace(cfg.Model) == "" {
return nil, fmt.Errorf("audita model is required")
if cfg.TotalLLMConcurrency != nil && *cfg.TotalLLMConcurrency <= 0 {
return nil, fmt.Errorf("audita total llm concurrency must be > 0 when provided")
}
if cfg.LLMConcurrency != nil && *cfg.LLMConcurrency <= 0 {
return nil, fmt.Errorf("audita llm concurrency must be > 0 when provided")
if cfg.ProposalLLMConcurrency != nil && *cfg.ProposalLLMConcurrency <= 0 {
return nil, fmt.Errorf("audita proposal llm concurrency must be > 0 when provided")
}
if cfg.ValidationLLMConcurrency != nil && *cfg.ValidationLLMConcurrency <= 0 {
return nil, fmt.Errorf("audita validation llm concurrency must be > 0 when provided")
}
switch strings.TrimSpace(cfg.OutputSchema) {
case "", "bare-segments", "audita-v1":
default:
return nil, fmt.Errorf("audita output schema must be one of: bare-segments, audita-v1")
}
switch strings.TrimSpace(cfg.WorkDirRetention) {
case "", "always", "auto", "never":
default:
return nil, fmt.Errorf("audita work dir retention must be one of: always, auto, never")
}
modules := make([]string, len(cfg.Modules))
for i, m := range cfg.Modules {
@@ -123,7 +156,12 @@ func NewSubprocessRunner(cfg SubprocessRunnerConfig) (*SubprocessRunner, error)
modules: modules,
baseURL: strings.TrimSpace(cfg.BaseURL),
model: strings.TrimSpace(cfg.Model),
llmConcurrency: cfg.LLMConcurrency,
transcriptDescription: strings.TrimSpace(cfg.TranscriptDescription),
configPath: strings.TrimSpace(cfg.ConfigPath),
outputSchema: strings.TrimSpace(cfg.OutputSchema),
workDirRetention: strings.TrimSpace(cfg.WorkDirRetention),
totalLLMConcurrency: cfg.TotalLLMConcurrency,
proposalLLMConcurrency: cfg.ProposalLLMConcurrency,
validationModel: strings.TrimSpace(cfg.ValidationModel),
validationLLMConcurrency: cfg.ValidationLLMConcurrency,
report: cfg.Report,
@@ -152,7 +190,7 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
}
reqModules := req.Modules
if len(reqModules) == 0 {
if reqModules == nil {
reqModules = append([]string(nil), r.modules...)
}
args := r.buildArgs(req, reqModules)
@@ -168,25 +206,21 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
env["AUDITA_LLM_API_KEY"] = credential
credentialPresent = true
}
primaryConcurrencyViaEnv := false
if r.llmConcurrency != nil {
env["AUDITA_LLM_CONCURRENCY"] = strconv.Itoa(*r.llmConcurrency)
primaryConcurrencyViaEnv = true
}
if req.GeneratedConfigPath != "" {
if err := r.writeInvocationConfig(req, args, reqModules, credentialPresent, primaryConcurrencyViaEnv); err != nil {
if err := r.writeInvocationConfig(req, args, reqModules, credentialPresent); err != nil {
return PolishResult{}, fmt.Errorf("write audita invocation config %q: %w", req.GeneratedConfigPath, err)
}
}
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
Executable: r.binary,
Args: args,
Timeout: r.timeout,
EnvOverrides: env,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
Executable: r.binary,
Args: args,
Timeout: r.timeout,
EnvOverrides: env,
DiagnosticOwner: "audita",
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
})
if err != nil {
wrappedMessage := fmt.Sprintf(
@@ -196,7 +230,7 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
req.StderrLogPath,
)
wrappedMessage = addSubprocessStreamHint(wrappedMessage, err)
return r.failureResult(req, reqModules, runRes, credentialPresent, primaryConcurrencyViaEnv), fmt.Errorf(
return r.failureResult(req, reqModules, runRes, credentialPresent), fmt.Errorf(
"%s: %w",
wrappedMessage,
err,
@@ -204,11 +238,11 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
}
if err := validateProcessedOutput(req.OutputProcessedPath); err != nil {
return r.failureResult(req, reqModules, runRes, credentialPresent, primaryConcurrencyViaEnv), fmt.Errorf("validate audita processed output %q: %w", req.OutputProcessedPath, err)
return r.failureResult(req, reqModules, runRes, credentialPresent), fmt.Errorf("validate audita processed output %q: %w", req.OutputProcessedPath, err)
}
if r.report {
if err := validateJSONFile(req.ReportPath); err != nil {
return r.failureResult(req, reqModules, runRes, credentialPresent, primaryConcurrencyViaEnv), fmt.Errorf("validate audita report output %q: %w", req.ReportPath, err)
return r.failureResult(req, reqModules, runRes, credentialPresent), fmt.Errorf("validate audita report output %q: %w", req.ReportPath, err)
}
}
@@ -223,21 +257,25 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
Duration: runRes.Duration,
InvokedBinary: r.binary,
Metadata: map[string]any{
"adapter": "audita_subprocess",
"modules": reqModules,
"base_url": r.baseURL,
"model": r.model,
"validation_model": r.validationModel,
"validation_llm_concurrency": r.validationLLMConcurrency,
"credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent,
"primary_llm_concurrency_via_env": primaryConcurrencyViaEnv,
"primary_llm_concurrency_env_name": "AUDITA_LLM_CONCURRENCY",
"adapter": "audita_subprocess",
"modules": reqModules,
"base_url": r.baseURL,
"model": r.model,
"transcript_description": r.transcriptDescription,
"config_path": r.configPath,
"output_schema": r.outputSchema,
"work_dir_retention": r.workDirRetention,
"validation_model": r.validationModel,
"total_llm_concurrency": r.totalLLMConcurrency,
"proposal_llm_concurrency": r.proposalLLMConcurrency,
"validation_llm_concurrency": r.validationLLMConcurrency,
"credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent,
},
}, nil
}
func (r *SubprocessRunner) failureResult(req PolishRequest, modules []string, runRes subprocess.RunResult, credentialPresent bool, primaryConcurrencyViaEnv bool) PolishResult {
func (r *SubprocessRunner) failureResult(req PolishRequest, modules []string, runRes subprocess.RunResult, credentialPresent bool) PolishResult {
return PolishResult{
ProcessedTranscriptPath: req.OutputProcessedPath,
ReportPath: req.ReportPath,
@@ -249,16 +287,20 @@ func (r *SubprocessRunner) failureResult(req PolishRequest, modules []string, ru
Duration: runRes.Duration,
InvokedBinary: r.binary,
Metadata: map[string]any{
"adapter": "audita_subprocess",
"modules": modules,
"base_url": r.baseURL,
"model": r.model,
"validation_model": r.validationModel,
"validation_llm_concurrency": r.validationLLMConcurrency,
"credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent,
"primary_llm_concurrency_via_env": primaryConcurrencyViaEnv,
"primary_llm_concurrency_env_name": "AUDITA_LLM_CONCURRENCY",
"adapter": "audita_subprocess",
"modules": modules,
"base_url": r.baseURL,
"model": r.model,
"transcript_description": r.transcriptDescription,
"config_path": r.configPath,
"output_schema": r.outputSchema,
"work_dir_retention": r.workDirRetention,
"validation_model": r.validationModel,
"total_llm_concurrency": r.totalLLMConcurrency,
"proposal_llm_concurrency": r.proposalLLMConcurrency,
"validation_llm_concurrency": r.validationLLMConcurrency,
"credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent,
},
}
}
@@ -269,14 +311,38 @@ func (r *SubprocessRunner) buildArgs(req PolishRequest, modules []string) []stri
req.MergedTranscriptPath,
"--glossary", req.GlossaryPath,
"--output", req.OutputProcessedPath,
"--modules", strings.Join(modules, ","),
"--base-url", r.baseURL,
"--model", r.model,
"--work-dir", req.WorkDir,
}
if r.baseURL != "" {
args = append(args, "--base-url", r.baseURL)
}
if r.model != "" {
args = append(args, "--model", r.model)
}
if len(modules) > 0 {
args = append(args, "--modules", strings.Join(modules, ","))
}
if r.report {
args = append(args, "--report-json", req.ReportPath)
}
if r.transcriptDescription != "" {
args = append(args, "--transcript-description", r.transcriptDescription)
}
if r.configPath != "" {
args = append(args, "--config", r.configPath)
}
if r.outputSchema != "" {
args = append(args, "--output-schema", r.outputSchema)
}
if r.workDirRetention != "" {
args = append(args, "--work-dir-retention", r.workDirRetention)
}
if r.totalLLMConcurrency != nil {
args = append(args, "--total-llm-concurrency", strconv.Itoa(*r.totalLLMConcurrency))
}
if r.proposalLLMConcurrency != nil {
args = append(args, "--proposal-llm-concurrency", strconv.Itoa(*r.proposalLLMConcurrency))
}
if r.validationModel != "" {
args = append(args, "--validation-model", r.validationModel)
}
@@ -286,37 +352,39 @@ func (r *SubprocessRunner) buildArgs(req PolishRequest, modules []string) []stri
return args
}
func (r *SubprocessRunner) writeInvocationConfig(req PolishRequest, args []string, modules []string, credentialPresent bool, primaryConcurrencyViaEnv bool) error {
func (r *SubprocessRunner) writeInvocationConfig(req PolishRequest, args []string, modules []string, credentialPresent bool) error {
payload := map[string]any{
"schema": "audita.generated.v1",
"binary": r.binary,
"args": args,
"timeout": r.timeout.String(),
"modules": modules,
"base_url": r.baseURL,
"model": r.model,
"validation_model": r.validationModel,
"validation_llm_concurrency": r.validationLLMConcurrency,
"report_enabled": r.report,
"merged_transcript_path": req.MergedTranscriptPath,
"glossary_path": req.GlossaryPath,
"output_path": req.OutputProcessedPath,
"report_path": req.ReportPath,
"work_dir": req.WorkDir,
"credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent,
"primary_llm_concurrency_via_env": primaryConcurrencyViaEnv,
"schema": "audita.generated.v1",
"binary": r.binary,
"args": args,
"timeout": r.timeout.String(),
"modules": modules,
"base_url": r.baseURL,
"model": r.model,
"transcript_description": r.transcriptDescription,
"config_path": r.configPath,
"output_schema": r.outputSchema,
"work_dir_retention": r.workDirRetention,
"validation_model": r.validationModel,
"total_llm_concurrency": r.totalLLMConcurrency,
"proposal_llm_concurrency": r.proposalLLMConcurrency,
"validation_llm_concurrency": r.validationLLMConcurrency,
"report_enabled": r.report,
"merged_transcript_path": req.MergedTranscriptPath,
"glossary_path": req.GlossaryPath,
"output_path": req.OutputProcessedPath,
"report_path": req.ReportPath,
"work_dir": req.WorkDir,
"credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent,
}
if r.llmConcurrency != nil {
payload["llm_concurrency"] = *r.llmConcurrency
}
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
}
func validateProcessedOutput(path string) error {
data, err := os.ReadFile(path)
data, err := readAuditaResult(path, MaxProcessedOutputBytes, "processed transcript")
if err != nil {
return fmt.Errorf("read file: %w", err)
return err
}
var payload map[string]any
@@ -345,9 +413,9 @@ func addSubprocessStreamHint(message string, runErr error) string {
}
func validateJSONFile(path string) error {
data, err := os.ReadFile(path)
data, err := readAuditaResult(path, MaxReportOutputBytes, "report")
if err != nil {
return fmt.Errorf("read file: %w", err)
return err
}
var v any
if err := json.Unmarshal(data, &v); err != nil {
@@ -355,3 +423,11 @@ func validateJSONFile(path string) error {
}
return nil
}
func readAuditaResult(path string, limit int64, category string) ([]byte, error) {
data, err := fileops.ReadRegularFile(path, limit)
if err != nil {
return nil, fmt.Errorf("audita %s result exceeds or cannot be read within %d-byte limit: %w", category, limit, err)
}
return data, nil
}

View File

@@ -25,7 +25,8 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath)
wrapper := writeAuditaHelperWrapper(t)
llmConcurrency := 1
totalLLMConcurrency := 3
proposalLLMConcurrency := 2
validationLLMConcurrency := 2
runner, err := NewSubprocessRunner(SubprocessRunnerConfig{
Binary: wrapper,
@@ -34,7 +35,12 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
Modules: []string{"glossary", "homophones", "glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
TranscriptDescription: "Campaign Session 42",
ConfigPath: "/etc/audita/config.yml",
OutputSchema: "audita-v1",
WorkDirRetention: "auto",
TotalLLMConcurrency: &totalLLMConcurrency,
ProposalLLMConcurrency: &proposalLLMConcurrency,
ValidationModel: "openrouter/google/gemma-4-31b-it",
ValidationLLMConcurrency: &validationLLMConcurrency,
Report: true,
@@ -46,9 +52,9 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
dir := t.TempDir()
req := PolishRequest{
GeneratedConfigPath: filepath.Join(dir, "audita.generated.yml"),
MergedTranscriptPath: filepath.Join(dir, "merged.json"),
MergedTranscriptPath: filepath.Join(dir, "base.json"),
GlossaryPath: filepath.Join(dir, "glossary.yml"),
OutputProcessedPath: filepath.Join(dir, "processed.json"),
OutputProcessedPath: filepath.Join(dir, "polished.json"),
ReportPath: filepath.Join(dir, "audita.report.json"),
WorkDir: filepath.Join(dir, "artifacts", "audita-work"),
StdoutLogPath: filepath.Join(dir, "audita.stdout.log"),
@@ -93,11 +99,17 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
"process", req.MergedTranscriptPath,
"--glossary", req.GlossaryPath,
"--output", req.OutputProcessedPath,
"--modules", "glossary,homophones,glossary",
"--work-dir", req.WorkDir,
"--base-url", "https://openrouter.ai/api/v1",
"--model", "openrouter/google/gemma-4-31b-it",
"--work-dir", req.WorkDir,
"--modules", "glossary,homophones,glossary",
"--report-json", req.ReportPath,
"--transcript-description", "Campaign Session 42",
"--config", "/etc/audita/config.yml",
"--output-schema", "audita-v1",
"--work-dir-retention", "auto",
"--total-llm-concurrency", "3",
"--proposal-llm-concurrency", "2",
"--validation-model", "openrouter/google/gemma-4-31b-it",
"--validation-llm-concurrency", "2",
}
@@ -107,8 +119,8 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
if rec.Env["AUDITA_LLM_API_KEY"] != "super-secret" {
t.Fatalf("AUDITA_LLM_API_KEY = %q, want propagated secret", rec.Env["AUDITA_LLM_API_KEY"])
}
if rec.Env["AUDITA_LLM_CONCURRENCY"] != "1" {
t.Fatalf("AUDITA_LLM_CONCURRENCY = %q, want 1", rec.Env["AUDITA_LLM_CONCURRENCY"])
if rec.Env["AUDITA_LLM_CONCURRENCY"] != "" {
t.Fatalf("AUDITA_LLM_CONCURRENCY = %q, want empty/omitted", rec.Env["AUDITA_LLM_CONCURRENCY"])
}
cfgData, err := os.ReadFile(req.GeneratedConfigPath)
@@ -124,16 +136,14 @@ func TestSubprocessRunnerMissingConfiguredCredentialFails(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "MISSING_AUDITA_KEY",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "MISSING_AUDITA_KEY",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: false,
})
req := auditaReqForTest(t, false)
@@ -155,16 +165,14 @@ func TestSubprocessRunnerUnconfiguredCredentialEnvOmitsCredential(t *testing.T)
recordPath := filepath.Join(t.TempDir(), "record.json")
t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath)
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: false,
})
req := auditaReqForTest(t, false)
@@ -181,7 +189,7 @@ func TestSubprocessRunnerUnconfiguredCredentialEnvOmitsCredential(t *testing.T)
}
}
func TestSubprocessRunnerInheritsParentEnvironment(t *testing.T) {
func TestSubprocessRunnerOmitsUnspecifiedParentEnvironment(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
@@ -206,8 +214,67 @@ func TestSubprocessRunnerInheritsParentEnvironment(t *testing.T) {
}
rec := readAuditaHelperRecord(t, recordPath)
if rec.Env["AUDITA_INHERITED_MARKER"] != "inherited-from-parent" {
t.Fatalf("AUDITA_INHERITED_MARKER = %q, want inherited-from-parent", rec.Env["AUDITA_INHERITED_MARKER"])
if rec.Env["AUDITA_INHERITED_MARKER"] != "" {
t.Fatalf("AUDITA_INHERITED_MARKER = %q, want omitted from the child environment", rec.Env["AUDITA_INHERITED_MARKER"])
}
}
func TestSubprocessRunnerOmitsModulesFlagWhenNotConfigured(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_AUDITA_HELPER", "1")
t.Setenv("AUDITA_HELPER_MODE", "success")
recordPath := filepath.Join(t.TempDir(), "record.json")
t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath)
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "",
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: false,
})
req := auditaReqForTest(t, false)
if _, err := runner.Run(context.Background(), req); err != nil {
t.Fatalf("Run() error = %v", err)
}
rec := readAuditaHelperRecord(t, recordPath)
for i := 0; i < len(rec.Args); i++ {
if rec.Args[i] == "--modules" {
t.Fatalf("args contained --modules unexpectedly: %#v", rec.Args)
}
}
}
func TestSubprocessRunnerOmitsBaseURLAndModelFlagsWhenNotConfigured(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_AUDITA_HELPER", "1")
t.Setenv("AUDITA_HELPER_MODE", "success")
recordPath := filepath.Join(t.TempDir(), "record.json")
t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath)
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "",
Report: false,
})
req := auditaReqForTest(t, false)
if _, err := runner.Run(context.Background(), req); err != nil {
t.Fatalf("Run() error = %v", err)
}
rec := readAuditaHelperRecord(t, recordPath)
for i := 0; i < len(rec.Args); i++ {
if rec.Args[i] == "--base-url" {
t.Fatalf("args contained --base-url unexpectedly: %#v", rec.Args)
}
if rec.Args[i] == "--model" {
t.Fatalf("args contained --model unexpectedly: %#v", rec.Args)
}
}
}
@@ -220,16 +287,14 @@ func TestSubprocessRunnerSubprocessFailure(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: true,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: true,
})
req := auditaReqForTest(t, true)
_, err := runner.Run(context.Background(), req)
@@ -256,16 +321,14 @@ func TestSubprocessRunnerSubprocessFailureAddsStderrDescriptorHint(t *testing.T)
t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: true,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: true,
})
req := auditaReqForTest(t, true)
_, err := runner.Run(context.Background(), req)
@@ -286,16 +349,14 @@ func TestSubprocessRunnerMissingOutputFails(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: false,
})
req := auditaReqForTest(t, false)
_, err := runner.Run(context.Background(), req)
@@ -316,16 +377,14 @@ func TestSubprocessRunnerInvalidOutputJSONFails(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: false,
})
req := auditaReqForTest(t, false)
_, err := runner.Run(context.Background(), req)
@@ -346,16 +405,14 @@ func TestSubprocessRunnerSegmentsMissingFails(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: false,
})
req := auditaReqForTest(t, false)
_, err := runner.Run(context.Background(), req)
@@ -376,16 +433,14 @@ func TestSubprocessRunnerInvalidReportJSONFails(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: true,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: true,
})
req := auditaReqForTest(t, true)
_, err := runner.Run(context.Background(), req)
@@ -398,11 +453,11 @@ func TestSubprocessRunnerInvalidReportJSONFails(t *testing.T) {
}
func TestSubprocessRunnerConstructorValidation(t *testing.T) {
_, err := NewSubprocessRunnerFromConfigValues("", "3h", "AUDITA_LLM_API_KEY", []string{"glossary"}, "https://openrouter.ai/api/v1", "openrouter/google/gemma-4-31b-it", nil, "", nil, true)
_, err := NewSubprocessRunnerFromConfigValues("", "3h", "AUDITA_LLM_API_KEY", []string{"glossary"}, "https://openrouter.ai/api/v1", "openrouter/google/gemma-4-31b-it", "", "", "", "", nil, nil, "", nil, true)
if err == nil {
t.Fatal("expected binary validation error")
}
_, err = NewSubprocessRunnerFromConfigValues("audita", "bad", "AUDITA_LLM_API_KEY", []string{"glossary"}, "https://openrouter.ai/api/v1", "openrouter/google/gemma-4-31b-it", nil, "", nil, true)
_, err = NewSubprocessRunnerFromConfigValues("audita", "bad", "AUDITA_LLM_API_KEY", []string{"glossary"}, "https://openrouter.ai/api/v1", "openrouter/google/gemma-4-31b-it", "", "", "", "", nil, nil, "", nil, true)
if err == nil {
t.Fatal("expected timeout parse error")
}
@@ -516,7 +571,7 @@ func mustAuditaRunner(t *testing.T, cfg SubprocessRunnerConfig) *SubprocessRunne
func auditaReqForTest(t *testing.T, withReport bool) PolishRequest {
t.Helper()
dir := t.TempDir()
merged := filepath.Join(dir, "merged.json")
merged := filepath.Join(dir, "base.json")
glossary := filepath.Join(dir, "glossary.yml")
writeAuditaTestFile(t, merged, `{"segments":[]}`)
writeAuditaTestFile(t, glossary, "terms: []\n")
@@ -524,7 +579,7 @@ func auditaReqForTest(t *testing.T, withReport bool) PolishRequest {
GeneratedConfigPath: filepath.Join(dir, "audita.generated.yml"),
MergedTranscriptPath: merged,
GlossaryPath: glossary,
OutputProcessedPath: filepath.Join(dir, "processed.json"),
OutputProcessedPath: filepath.Join(dir, "polished.json"),
WorkDir: filepath.Join(dir, "artifacts", "audita-work"),
StdoutLogPath: filepath.Join(dir, "audita.stdout.log"),
StderrLogPath: filepath.Join(dir, "audita.stderr.log"),

View File

@@ -0,0 +1,24 @@
package notarius
import "context"
// FakeRunner is a configurable in-memory runner for stage tests.
type FakeRunner struct {
Requests []RunRequest
Result RunResult
Err error
}
// Run records the request and returns the configured result or error.
func (f *FakeRunner) Run(ctx context.Context, req RunRequest) (RunResult, error) {
if err := ctx.Err(); err != nil {
return RunResult{}, err
}
copyRequest := req
copyRequest.References = append([]ReferenceBinding(nil), req.References...)
f.Requests = append(f.Requests, copyRequest)
if f.Err != nil {
return RunResult{}, f.Err
}
return f.Result, nil
}

View File

@@ -0,0 +1,160 @@
// Package notarius declares the adapter contract for Notarius CLI invocations.
package notarius
import (
"context"
"time"
)
const ReceiptSchemaVersion = "notarius.run-result.v2"
// Runner is the adapter boundary for a complete Notarius pipeline invocation.
type Runner interface {
Run(ctx context.Context, req RunRequest) (RunResult, error)
}
// ReferenceBinding maps one normalized Notarius selector to an absolute
// external reference path.
type ReferenceBinding struct {
Selector string
Path string
}
// RunRequest contains the resolved inputs and diagnostic destinations for one invocation.
type RunRequest struct {
Binary string
ConfigPath string
PipelineID string
InputPath string
OutputRoot string
WorkingDirectory string
ReceiptPath string
LogPath string
Timeout time.Duration
References []ReferenceBinding
}
// Receipt is the transport-neutral successful run receipt.
type Receipt struct {
SchemaVersion string
RunID string
PipelineID string
OutputDirectory string
IndexFile string
NormalizedOutputCount int
RejectedOutputCount int
WarningGroupCount int
WarningOccurrenceCount int
DiagnosticGroupCount int
DiagnosticOccurrenceCount int
DiagnosticsTruncated bool
ValidationStatus string
ValidationSummaries []ValidationSummary
DebugDirectory string
}
// ValidationSummary retains the bounded outcome of one Notarius producer result.
type ValidationSummary struct {
Stage string
StepID string
LaneID string
ModuleKey string
ChunkID string
Status string
RejectingValidators []string
ReasonCodes []string
IncompleteValidators []string
ProducerAttemptCount int
TerminalAction string
}
// LaneDescriptor identifies one normalized lane payload discovered through the index.
type LaneDescriptor struct {
LaneID string
File string
Path string
MediaType string
ModuleKey string
SchemaID string
SchemaName string
SchemaVersion string
}
// PipelineDescriptor identifies a pipeline-wide artifact discovered through the index.
type PipelineDescriptor struct {
ArtifactKind string
File string
Path string
MediaType string
SchemaID string
SchemaName string
SchemaVersion string
}
// Index describes the validated bundle-management and artifact paths.
type Index struct {
Path string
ManifestFile string
ManifestPath string
RejectedFile string
RejectedPath string
WarningsFile string
WarningsPath string
DiagnosticsFile string
DiagnosticsPath string
Lanes []LaneDescriptor
ChunkMap *PipelineDescriptor
EvidenceContext *PipelineDescriptor
}
// RejectionSummary retains structured rejection identity without free-form messages.
type RejectionSummary struct {
Stage string
StepID string
LaneID string
ModuleKey string
ChunkID string
ValidatorName string
ReasonCode string
}
// WarningSummary retains structured warning identity without free-form messages.
type WarningSummary struct {
Disposition string
Category string
ReasonCode string
Origin DiagnosticOrigin
OccurrenceCount int
}
// DiagnosticOrigin identifies the framework-owned pipeline location of a finding.
type DiagnosticOrigin struct {
Stage string
StepID string
LaneID string
ModuleKey string
ValidatorKey string
}
// DiagnosticSummary retains bounded advisory or observation group metadata.
type DiagnosticSummary struct {
Disposition string
Category string
ReasonCode string
Origin DiagnosticOrigin
OccurrenceCount int
}
// RunResult describes a successfully decoded and validated Notarius bundle.
type RunResult struct {
Receipt Receipt
Index Index
BundleRoot string
ReceiptPath string
LogPath string
ExitCode int
Duration time.Duration
Rejections []RejectionSummary
Warnings []WarningSummary
Diagnostics []DiagnosticSummary
}

View File

@@ -0,0 +1,784 @@
package notarius
import (
"context"
"encoding/json"
"errors"
"fmt"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
"gitea.maximumdirect.net/eric/narratio/internal/notariusref"
"gitea.maximumdirect.net/eric/narratio/internal/pathsafe"
)
const (
maxReceiptBytes = 1 << 20
maxIndexBytes = 4 << 20
maxSummaryBytes = 4 << 20
canonicalIndexFile = "index.json"
canonicalManifestFile = "manifest.json"
canonicalRejectedFile = "rejected.json"
canonicalWarningsFile = "warnings.json"
canonicalDiagnosticsFile = "diagnostics.json"
warningsSchemaVersion = "notarius.warnings.v2"
diagnosticsSchemaVersion = "notarius.diagnostics.v1"
maxWarningGroups = 128
maxDiagnosticGroups = 256
maxFindingSamples = 3
)
type subprocessRun func(context.Context, subprocess.RunRequest) (subprocess.RunResult, error)
// SubprocessRunner invokes Notarius through its public CLI.
type SubprocessRunner struct {
run subprocessRun
}
// NewSubprocessRunner constructs a production Notarius subprocess runner.
func NewSubprocessRunner() *SubprocessRunner {
return &SubprocessRunner{run: subprocess.Run}
}
// Run executes a complete Notarius pipeline and discovers its published bundle.
func (r *SubprocessRunner) Run(ctx context.Context, req RunRequest) (RunResult, error) {
if r == nil || r.run == nil {
return RunResult{}, fmt.Errorf("notarius subprocess runner is nil")
}
references, err := validateRunRequest(req)
if err != nil {
return RunResult{}, err
}
args := []string{
"run", req.PipelineID,
"--config", req.ConfigPath,
"--input", req.InputPath,
"--output-dir", req.OutputRoot,
}
for _, reference := range references {
args = append(args, "--reference", reference.Selector+"="+reference.Path)
}
args = append(args, "--json")
processResult, err := r.run(ctx, subprocess.RunRequest{
Executable: req.Binary,
Args: args,
WorkingDir: req.WorkingDirectory,
Timeout: req.Timeout,
DiagnosticOwner: "notarius",
StdoutLogPath: req.ReceiptPath,
StderrLogPath: req.LogPath,
})
baseResult := RunResult{
ReceiptPath: req.ReceiptPath,
LogPath: req.LogPath,
ExitCode: processResult.ExitCode,
Duration: processResult.Duration,
}
if err != nil {
return baseResult, fmt.Errorf("run notarius pipeline %q: %w", req.PipelineID, err)
}
receipt, err := loadReceipt(req.ReceiptPath, req.PipelineID)
if err != nil {
return baseResult, err
}
bundleRoot, err := validateBundleRoot(req.OutputRoot, receipt.OutputDirectory)
if err != nil {
return baseResult, err
}
indexPath, err := resolveRegularFile(bundleRoot, receipt.IndexFile)
if err != nil {
return baseResult, fmt.Errorf("resolve receipt index file: %w", err)
}
index, err := loadIndex(bundleRoot, indexPath)
if err != nil {
return baseResult, err
}
rejections, err := loadRejections(index.RejectedPath)
if err != nil {
return baseResult, err
}
warnings, err := loadWarnings(index.WarningsPath)
if err != nil {
return baseResult, err
}
diagnostics, diagnosticOccurrences, diagnosticsTruncated, err := loadDiagnostics(index.DiagnosticsPath)
if err != nil {
return baseResult, err
}
if receipt.NormalizedOutputCount != len(index.Lanes) || receipt.RejectedOutputCount != len(rejections) ||
receipt.WarningGroupCount != len(warnings) || receipt.DiagnosticGroupCount != len(diagnostics) {
return baseResult, fmt.Errorf("notarius receipt counts do not match published bundle")
}
warningOccurrences, err := sumWarningOccurrences(warnings)
if err != nil {
return baseResult, err
}
if receipt.WarningOccurrenceCount != warningOccurrences ||
receipt.DiagnosticOccurrenceCount != diagnosticOccurrences ||
receipt.DiagnosticsTruncated != diagnosticsTruncated {
return baseResult, fmt.Errorf("notarius receipt occurrence counts do not match published bundle")
}
baseResult.Receipt = receipt
baseResult.Index = index
baseResult.BundleRoot = bundleRoot
baseResult.Rejections = rejections
baseResult.Warnings = warnings
baseResult.Diagnostics = diagnostics
return baseResult, nil
}
func validateRunRequest(req RunRequest) ([]ReferenceBinding, error) {
if strings.TrimSpace(req.Binary) == "" {
return nil, fmt.Errorf("notarius binary is required")
}
if strings.TrimSpace(req.PipelineID) == "" {
return nil, fmt.Errorf("notarius pipeline id is required")
}
if req.Timeout <= 0 {
return nil, fmt.Errorf("notarius timeout must be positive")
}
for label, path := range map[string]string{
"config": req.ConfigPath,
"input": req.InputPath,
"output root": req.OutputRoot,
"working directory": req.WorkingDirectory,
"receipt": req.ReceiptPath,
"log": req.LogPath,
} {
if strings.TrimSpace(path) == "" {
return nil, fmt.Errorf("notarius %s path is required", label)
}
if !filepath.IsAbs(path) {
return nil, fmt.Errorf("notarius %s path must be absolute", label)
}
}
if filepath.Clean(req.ReceiptPath) == filepath.Clean(req.LogPath) {
return nil, fmt.Errorf("notarius receipt and log paths must be different")
}
references := make([]ReferenceBinding, 0, len(req.References))
selectors := make(map[string]struct{}, len(req.References))
for index, binding := range req.References {
selector, err := notariusref.NormalizeSelector(binding.Selector)
if err != nil {
return nil, fmt.Errorf("notarius reference %d selector: %w", index, err)
}
if _, duplicate := selectors[selector]; duplicate {
return nil, fmt.Errorf("notarius reference selector %q is duplicated", selector)
}
selectors[selector] = struct{}{}
if strings.TrimSpace(binding.Path) == "" {
return nil, fmt.Errorf("notarius reference %q path is required", selector)
}
if !filepath.IsAbs(binding.Path) {
return nil, fmt.Errorf("notarius reference %q path must be absolute", selector)
}
references = append(references, ReferenceBinding{Selector: selector, Path: binding.Path})
}
if err := requireRegularFile(req.ConfigPath); err != nil {
return nil, fmt.Errorf("validate notarius config path: %w", err)
}
if err := requireRegularFile(req.InputPath); err != nil {
return nil, fmt.Errorf("validate notarius input path: %w", err)
}
if err := requireDirectory(req.OutputRoot); err != nil {
return nil, fmt.Errorf("validate notarius output root: %w", err)
}
if err := requireDirectory(req.WorkingDirectory); err != nil {
return nil, fmt.Errorf("validate notarius working directory: %w", err)
}
if err := validateLogDestination(req.ReceiptPath); err != nil {
return nil, fmt.Errorf("validate notarius receipt path: %w", err)
}
if err := validateLogDestination(req.LogPath); err != nil {
return nil, fmt.Errorf("validate notarius log path: %w", err)
}
return references, nil
}
type receiptDocument struct {
SchemaVersion string `json:"schema_version"`
RunID string `json:"run_id"`
PipelineID string `json:"pipeline_id"`
OutputDirectory string `json:"output_directory"`
IndexFile string `json:"index_file"`
NormalizedOutputCount *int `json:"normalized_output_count"`
RejectedOutputCount *int `json:"rejected_output_count"`
WarningGroupCount *int `json:"warning_group_count"`
WarningOccurrenceCount *int `json:"warning_occurrence_count"`
DiagnosticGroupCount *int `json:"diagnostic_group_count"`
DiagnosticOccurrenceCount *int `json:"diagnostic_occurrence_count"`
DiagnosticsTruncated *bool `json:"diagnostics_truncated"`
ValidationStatus string `json:"validation_status"`
ValidationSummaries []validationSummaryDocument `json:"validation_summaries"`
DebugDirectory string `json:"debug_directory"`
}
type validationSummaryDocument struct {
Stage string `json:"stage"`
StepID string `json:"step_id"`
LaneID string `json:"lane_id"`
ModuleKey string `json:"module_key"`
ChunkID string `json:"chunk_id"`
Status string `json:"status"`
RejectingValidators []string `json:"rejecting_validators"`
ReasonCodes []string `json:"reason_codes"`
IncompleteValidators []string `json:"incomplete_validators"`
ProducerAttemptCount *int `json:"producer_attempt_count"`
TerminalAction string `json:"terminal_action"`
}
func validValidationStatus(value string) bool {
switch value {
case "approved", "rejected", "incomplete":
return true
default:
return false
}
}
func validateValidationSummaries(documents []validationSummaryDocument) ([]ValidationSummary, error) {
summaries := make([]ValidationSummary, 0, len(documents))
for _, document := range documents {
if document.Status != "complete" && document.Status != "rejected" && document.Status != "incomplete" {
return nil, fmt.Errorf("notarius validation summary status %q is invalid", document.Status)
}
if document.ProducerAttemptCount == nil || *document.ProducerAttemptCount <= 0 || !validTerminalAction(document.TerminalAction) {
return nil, fmt.Errorf("notarius validation summary is missing required fields")
}
summaries = append(summaries, ValidationSummary{
Stage: document.Stage, StepID: document.StepID, LaneID: document.LaneID,
ModuleKey: document.ModuleKey, ChunkID: document.ChunkID, Status: document.Status,
RejectingValidators: append([]string(nil), document.RejectingValidators...),
ReasonCodes: append([]string(nil), document.ReasonCodes...),
IncompleteValidators: append([]string(nil), document.IncompleteValidators...),
ProducerAttemptCount: *document.ProducerAttemptCount, TerminalAction: document.TerminalAction,
})
}
return summaries, nil
}
func validTerminalAction(value string) bool {
switch value {
case "accepted", "reject_output", "warn_continue", "fail_run":
return true
default:
return false
}
}
func loadReceipt(path, pipelineID string) (Receipt, error) {
var document receiptDocument
if err := decodeBoundedJSON(path, maxReceiptBytes, &document); err != nil {
return Receipt{}, fmt.Errorf("decode notarius receipt: %w", err)
}
if document.SchemaVersion != ReceiptSchemaVersion {
return Receipt{}, fmt.Errorf("unsupported notarius receipt schema version %q", document.SchemaVersion)
}
if strings.TrimSpace(document.RunID) == "" || strings.TrimSpace(document.PipelineID) == "" ||
strings.TrimSpace(document.OutputDirectory) == "" || strings.TrimSpace(document.ValidationStatus) == "" ||
document.NormalizedOutputCount == nil ||
document.RejectedOutputCount == nil || document.WarningGroupCount == nil ||
document.WarningOccurrenceCount == nil || document.DiagnosticGroupCount == nil ||
document.DiagnosticOccurrenceCount == nil || document.DiagnosticsTruncated == nil {
return Receipt{}, fmt.Errorf("notarius receipt is missing required fields")
}
if document.IndexFile != canonicalIndexFile {
return Receipt{}, fmt.Errorf("notarius receipt index_file %q is incompatible; want %q", document.IndexFile, canonicalIndexFile)
}
if document.PipelineID != pipelineID {
return Receipt{}, fmt.Errorf("notarius receipt pipeline id %q does not match requested pipeline %q", document.PipelineID, pipelineID)
}
if *document.NormalizedOutputCount < 0 || *document.RejectedOutputCount < 0 ||
*document.WarningGroupCount < 0 || *document.WarningOccurrenceCount < 0 ||
*document.DiagnosticGroupCount < 0 || *document.DiagnosticOccurrenceCount < 0 {
return Receipt{}, fmt.Errorf("notarius receipt counts must be non-negative")
}
if !validValidationStatus(document.ValidationStatus) {
return Receipt{}, fmt.Errorf("notarius receipt validation_status %q is invalid", document.ValidationStatus)
}
validationSummaries, err := validateValidationSummaries(document.ValidationSummaries)
if err != nil {
return Receipt{}, err
}
if !filepath.IsAbs(document.OutputDirectory) {
return Receipt{}, fmt.Errorf("notarius receipt output directory must be absolute")
}
if document.DebugDirectory != "" && !filepath.IsAbs(document.DebugDirectory) {
return Receipt{}, fmt.Errorf("notarius receipt debug directory must be absolute when present")
}
return Receipt{
SchemaVersion: document.SchemaVersion,
RunID: document.RunID,
PipelineID: document.PipelineID,
OutputDirectory: filepath.Clean(document.OutputDirectory),
IndexFile: document.IndexFile,
NormalizedOutputCount: *document.NormalizedOutputCount,
RejectedOutputCount: *document.RejectedOutputCount,
WarningGroupCount: *document.WarningGroupCount,
WarningOccurrenceCount: *document.WarningOccurrenceCount,
DiagnosticGroupCount: *document.DiagnosticGroupCount,
DiagnosticOccurrenceCount: *document.DiagnosticOccurrenceCount,
DiagnosticsTruncated: *document.DiagnosticsTruncated,
ValidationStatus: document.ValidationStatus,
ValidationSummaries: validationSummaries,
DebugDirectory: document.DebugDirectory,
}, nil
}
type indexDocument struct {
ManifestFile string `json:"manifest_file"`
OutputFiles *[]laneDocument `json:"output_files"`
RejectedFile string `json:"rejected_file"`
WarningsFile string `json:"warnings_file"`
DiagnosticsFile string `json:"diagnostics_file"`
ChunkMap *pipelineDocument `json:"chunk_map"`
EvidenceContext *pipelineDocument `json:"evidence_context"`
}
type laneDocument struct {
LaneID string `json:"lane_id"`
File string `json:"file"`
MediaType string `json:"media_type"`
ModuleKey string `json:"module_key"`
SchemaID string `json:"schema_id"`
SchemaName string `json:"schema_name"`
SchemaVersion string `json:"schema_version"`
}
type pipelineDocument struct {
ArtifactKind string `json:"artifact_kind"`
File string `json:"file"`
MediaType string `json:"media_type"`
SchemaID string `json:"schema_id"`
SchemaName string `json:"schema_name"`
SchemaVersion string `json:"schema_version"`
}
func loadIndex(bundleRoot, indexPath string) (Index, error) {
var document indexDocument
if err := decodeBoundedJSON(indexPath, maxIndexBytes, &document); err != nil {
return Index{}, fmt.Errorf("decode notarius index: %w", err)
}
for _, field := range []struct {
name string
got string
want string
}{
{name: "manifest_file", got: document.ManifestFile, want: canonicalManifestFile},
{name: "rejected_file", got: document.RejectedFile, want: canonicalRejectedFile},
{name: "warnings_file", got: document.WarningsFile, want: canonicalWarningsFile},
{name: "diagnostics_file", got: document.DiagnosticsFile, want: canonicalDiagnosticsFile},
} {
if field.got != field.want {
return Index{}, fmt.Errorf("notarius index %s %q is incompatible; want %q", field.name, field.got, field.want)
}
}
if document.OutputFiles == nil {
return Index{}, fmt.Errorf("notarius index is missing required output_files")
}
index := Index{
Path: indexPath,
ManifestFile: document.ManifestFile,
RejectedFile: document.RejectedFile,
WarningsFile: document.WarningsFile,
DiagnosticsFile: document.DiagnosticsFile,
}
var err error
if index.ManifestPath, err = resolveRegularFile(bundleRoot, index.ManifestFile); err != nil {
return Index{}, fmt.Errorf("resolve notarius manifest file: %w", err)
}
if index.RejectedPath, err = resolveRegularFile(bundleRoot, index.RejectedFile); err != nil {
return Index{}, fmt.Errorf("resolve notarius rejection file: %w", err)
}
if index.WarningsPath, err = resolveRegularFile(bundleRoot, index.WarningsFile); err != nil {
return Index{}, fmt.Errorf("resolve notarius warning file: %w", err)
}
if index.DiagnosticsPath, err = resolveRegularFile(bundleRoot, index.DiagnosticsFile); err != nil {
return Index{}, fmt.Errorf("resolve notarius diagnostics file: %w", err)
}
seenLanes := make(map[string]struct{}, len(*document.OutputFiles))
for _, lane := range *document.OutputFiles {
if strings.TrimSpace(lane.LaneID) == "" || strings.TrimSpace(lane.File) == "" {
return Index{}, fmt.Errorf("notarius lane descriptors require lane_id and file")
}
if _, exists := seenLanes[lane.LaneID]; exists {
return Index{}, fmt.Errorf("notarius index contains duplicate lane id %q", lane.LaneID)
}
seenLanes[lane.LaneID] = struct{}{}
path, err := resolveRegularFile(bundleRoot, lane.File)
if err != nil {
return Index{}, fmt.Errorf("resolve notarius lane %q file: %w", lane.LaneID, err)
}
index.Lanes = append(index.Lanes, LaneDescriptor{
LaneID: lane.LaneID, File: lane.File, Path: path, MediaType: lane.MediaType,
ModuleKey: lane.ModuleKey, SchemaID: lane.SchemaID, SchemaName: lane.SchemaName,
SchemaVersion: lane.SchemaVersion,
})
}
if document.ChunkMap != nil {
index.ChunkMap, err = resolvePipelineDescriptor(bundleRoot, "chunk_map", *document.ChunkMap)
if err != nil {
return Index{}, err
}
}
if document.EvidenceContext != nil {
index.EvidenceContext, err = resolvePipelineDescriptor(bundleRoot, "evidence_context", *document.EvidenceContext)
if err != nil {
return Index{}, err
}
}
return index, nil
}
func resolvePipelineDescriptor(bundleRoot, label string, document pipelineDocument) (*PipelineDescriptor, error) {
if strings.TrimSpace(document.ArtifactKind) == "" || strings.TrimSpace(document.File) == "" ||
strings.TrimSpace(document.MediaType) == "" || strings.TrimSpace(document.SchemaID) == "" ||
strings.TrimSpace(document.SchemaName) == "" || strings.TrimSpace(document.SchemaVersion) == "" {
return nil, fmt.Errorf("notarius %s descriptor is missing required fields", label)
}
path, err := resolveRegularFile(bundleRoot, document.File)
if err != nil {
return nil, fmt.Errorf("resolve notarius %s file: %w", label, err)
}
return &PipelineDescriptor{
ArtifactKind: document.ArtifactKind, File: document.File, Path: path,
MediaType: document.MediaType, SchemaID: document.SchemaID,
SchemaName: document.SchemaName, SchemaVersion: document.SchemaVersion,
}, nil
}
type rejectionDocument struct {
Rejected *[]struct {
Stage string `json:"stage"`
StepID string `json:"step_id"`
LaneID string `json:"lane_id"`
ModuleKey string `json:"module_key"`
ChunkID string `json:"chunk_id"`
ValidatorName string `json:"validator_name"`
ReasonCode string `json:"reason_code"`
Message string `json:"message"`
} `json:"rejected"`
}
func loadRejections(path string) ([]RejectionSummary, error) {
var document rejectionDocument
if err := decodeBoundedJSON(path, maxSummaryBytes, &document); err != nil {
return nil, fmt.Errorf("decode notarius rejections: %w", err)
}
if document.Rejected == nil {
return nil, fmt.Errorf("notarius rejection document is missing rejected array")
}
summaries := make([]RejectionSummary, 0, len(*document.Rejected))
for _, item := range *document.Rejected {
if strings.TrimSpace(item.Stage) == "" || strings.TrimSpace(item.Message) == "" {
return nil, fmt.Errorf("notarius rejection entries require stage and message")
}
summaries = append(summaries, RejectionSummary{
Stage: item.Stage, StepID: item.StepID, LaneID: item.LaneID,
ModuleKey: item.ModuleKey, ChunkID: item.ChunkID,
ValidatorName: item.ValidatorName, ReasonCode: item.ReasonCode,
})
}
return summaries, nil
}
type findingGroupDocument struct {
Disposition string `json:"disposition"`
Category string `json:"category"`
ReasonCode string `json:"reason_code"`
Origin diagnosticOriginDocument `json:"origin"`
OccurrenceCount *int `json:"occurrence_count"`
Samples *[]struct {
Scope string `json:"scope"`
Message string `json:"message"`
ChunkID string `json:"chunk_id"`
ChunkIndex *int `json:"chunk_index"`
} `json:"samples"`
OmittedSampleCount *int `json:"omitted_sample_count"`
}
type diagnosticOriginDocument struct {
Stage string `json:"stage"`
StepID string `json:"step_id"`
LaneID string `json:"lane_id"`
ModuleKey string `json:"module_key"`
ValidatorKey string `json:"validator_key"`
}
type warningDocument struct {
SchemaVersion string `json:"schema_version"`
GroupCount *int `json:"group_count"`
OccurrenceCount *int `json:"occurrence_count"`
Groups *[]findingGroupDocument `json:"groups"`
}
func loadWarnings(path string) ([]WarningSummary, error) {
var document warningDocument
if err := decodeBoundedJSON(path, maxSummaryBytes, &document); err != nil {
return nil, fmt.Errorf("decode notarius warnings: %w", err)
}
if document.SchemaVersion != warningsSchemaVersion || document.GroupCount == nil ||
document.OccurrenceCount == nil || document.Groups == nil {
return nil, fmt.Errorf("notarius warning document is missing or incompatible required fields")
}
if *document.GroupCount < 0 || *document.GroupCount > maxWarningGroups || *document.OccurrenceCount < 0 ||
*document.GroupCount != len(*document.Groups) {
return nil, fmt.Errorf("notarius warning document counts are inconsistent")
}
summaries := make([]WarningSummary, 0, len(*document.Groups))
occurrences := 0
for _, group := range *document.Groups {
if err := validateFindingGroup(group); err != nil {
return nil, fmt.Errorf("notarius warning group: %w", err)
}
if group.Disposition != "warning" {
return nil, fmt.Errorf("notarius warning group disposition %q is invalid", group.Disposition)
}
if *group.OccurrenceCount > int(^uint(0)>>1)-occurrences {
return nil, fmt.Errorf("notarius warning occurrence count overflows")
}
occurrences += *group.OccurrenceCount
summaries = append(summaries, WarningSummary{
Disposition: group.Disposition, Category: group.Category, ReasonCode: group.ReasonCode,
Origin: diagnosticOrigin(group.Origin), OccurrenceCount: *group.OccurrenceCount,
})
}
if occurrences != *document.OccurrenceCount {
return nil, fmt.Errorf("notarius warning document occurrence count is inconsistent")
}
return summaries, nil
}
type diagnosticDocument struct {
SchemaVersion string `json:"schema_version"`
GroupCount *int `json:"group_count"`
OccurrenceCount *int `json:"occurrence_count"`
Truncated *bool `json:"truncated"`
UnrepresentedOccurrenceCount *int `json:"unrepresented_occurrence_count"`
Groups *[]findingGroupDocument `json:"groups"`
}
func loadDiagnostics(path string) ([]DiagnosticSummary, int, bool, error) {
var document diagnosticDocument
if err := decodeBoundedJSON(path, maxSummaryBytes, &document); err != nil {
return nil, 0, false, fmt.Errorf("decode notarius diagnostics: %w", err)
}
if document.SchemaVersion != diagnosticsSchemaVersion || document.GroupCount == nil ||
document.OccurrenceCount == nil || document.Truncated == nil ||
document.UnrepresentedOccurrenceCount == nil || document.Groups == nil {
return nil, 0, false, fmt.Errorf("notarius diagnostics document is missing or incompatible required fields")
}
if *document.GroupCount < 0 || *document.GroupCount > maxDiagnosticGroups || *document.OccurrenceCount < 0 ||
*document.UnrepresentedOccurrenceCount < 0 || *document.GroupCount != len(*document.Groups) {
return nil, 0, false, fmt.Errorf("notarius diagnostics document counts are inconsistent")
}
if !*document.Truncated && *document.UnrepresentedOccurrenceCount != 0 {
return nil, 0, false, fmt.Errorf("notarius diagnostics document has unrepresented occurrences without truncation")
}
summaries := make([]DiagnosticSummary, 0, len(*document.Groups))
representedOccurrences := 0
for _, group := range *document.Groups {
if err := validateFindingGroup(group); err != nil {
return nil, 0, false, fmt.Errorf("notarius diagnostic group: %w", err)
}
if group.Disposition != "advisory" && group.Disposition != "observation" {
return nil, 0, false, fmt.Errorf("notarius diagnostic group disposition %q is invalid", group.Disposition)
}
if *group.OccurrenceCount > int(^uint(0)>>1)-representedOccurrences {
return nil, 0, false, fmt.Errorf("notarius diagnostic occurrence count overflows")
}
representedOccurrences += *group.OccurrenceCount
summaries = append(summaries, DiagnosticSummary{
Disposition: group.Disposition, Category: group.Category, ReasonCode: group.ReasonCode,
Origin: diagnosticOrigin(group.Origin), OccurrenceCount: *group.OccurrenceCount,
})
}
if *document.UnrepresentedOccurrenceCount > int(^uint(0)>>1)-representedOccurrences ||
representedOccurrences+*document.UnrepresentedOccurrenceCount != *document.OccurrenceCount {
return nil, 0, false, fmt.Errorf("notarius diagnostics document occurrence count is inconsistent")
}
return summaries, *document.OccurrenceCount, *document.Truncated, nil
}
func validateFindingGroup(group findingGroupDocument) error {
if strings.TrimSpace(group.Disposition) == "" || strings.TrimSpace(group.Category) == "" ||
strings.TrimSpace(group.ReasonCode) == "" || !validDiagnosticOriginStage(group.Origin.Stage) ||
!validDiagnosticCategory(group.Disposition, group.Category) ||
group.OccurrenceCount == nil || *group.OccurrenceCount <= 0 || group.Samples == nil ||
group.OmittedSampleCount == nil || *group.OmittedSampleCount < 0 {
return fmt.Errorf("missing required fields")
}
if len(*group.Samples) == 0 || len(*group.Samples) > maxFindingSamples ||
*group.OmittedSampleCount != *group.OccurrenceCount-len(*group.Samples) {
return fmt.Errorf("sample counts are inconsistent")
}
for _, sample := range *group.Samples {
if strings.TrimSpace(sample.Scope) == "" || strings.TrimSpace(sample.Message) == "" ||
(sample.ChunkIndex != nil && *sample.ChunkIndex < 0) {
return fmt.Errorf("samples require scope and message")
}
}
return nil
}
func validDiagnosticCategory(disposition, category string) bool {
switch disposition {
case "warning":
return category == "configuration" || category == "degradation" ||
category == "validation_incomplete" || category == "fallback"
case "advisory":
return category == "data_quality"
case "observation":
return category == "normalization"
default:
return false
}
}
func validDiagnosticOriginStage(stage string) bool {
switch stage {
case "references", "chunk", "extract", "merge", "normalize":
return true
default:
return false
}
}
func diagnosticOrigin(document diagnosticOriginDocument) DiagnosticOrigin {
return DiagnosticOrigin{
Stage: document.Stage, StepID: document.StepID, LaneID: document.LaneID,
ModuleKey: document.ModuleKey, ValidatorKey: document.ValidatorKey,
}
}
func sumWarningOccurrences(values []WarningSummary) (int, error) {
total := 0
for _, value := range values {
if value.OccurrenceCount > int(^uint(0)>>1)-total {
return 0, fmt.Errorf("notarius warning occurrence count overflows")
}
total += value.OccurrenceCount
}
return total, nil
}
func decodeBoundedJSON(path string, limit int64, destination any) error {
data, err := fileops.ReadRegularFile(path, limit)
if err != nil {
return fmt.Errorf("notarius JSON result exceeds or cannot be read within %d-byte limit: %w", limit, err)
}
if err := json.Unmarshal(data, destination); err != nil {
return err
}
return nil
}
func validateBundleRoot(outputRoot, bundleRoot string) (string, error) {
root := filepath.Clean(outputRoot)
bundle := filepath.Clean(bundleRoot)
relative, err := filepath.Rel(root, bundle)
if err != nil {
return "", fmt.Errorf("compare notarius output paths: %w", err)
}
if relative == "." || relative == ".." || strings.HasPrefix(relative, ".."+string(filepath.Separator)) {
return "", fmt.Errorf("notarius output directory %q is not beneath output root %q", bundleRoot, outputRoot)
}
if err := requireDirectoryTree(root, relative); err != nil {
return "", fmt.Errorf("validate notarius output directory: %w", err)
}
return bundle, nil
}
func resolveRegularFile(root, logicalPath string) (string, error) {
resolved, err := pathsafe.JoinSlashRelativeUnderRoot(root, logicalPath)
if err != nil {
return "", err
}
relative, err := filepath.Rel(root, resolved)
if err != nil {
return "", err
}
if err := requireRegularFileTree(root, relative); err != nil {
return "", err
}
return resolved, nil
}
func requireDirectoryTree(root, relative string) error {
if err := requireDirectory(root); err != nil {
return err
}
current := root
for _, component := range strings.Split(relative, string(filepath.Separator)) {
current = filepath.Join(current, component)
if err := requireDirectory(current); err != nil {
return err
}
}
return nil
}
func requireRegularFileTree(root, relative string) error {
components := strings.Split(relative, string(filepath.Separator))
if len(components) == 0 {
return fmt.Errorf("regular file path is required")
}
if err := requireDirectory(root); err != nil {
return err
}
current := root
for _, component := range components[:len(components)-1] {
current = filepath.Join(current, component)
if err := requireDirectory(current); err != nil {
return err
}
}
return requireRegularFile(filepath.Join(current, components[len(components)-1]))
}
func requireDirectory(path string) error {
info, err := os.Lstat(path)
if err != nil {
return err
}
if info.Mode()&os.ModeSymlink != 0 || !info.IsDir() {
return fmt.Errorf("path %q must be a directory without symlinks", path)
}
return nil
}
func requireRegularFile(path string) error {
info, err := os.Lstat(path)
if err != nil {
return err
}
if info.Mode()&os.ModeSymlink != 0 || !info.Mode().IsRegular() {
return fmt.Errorf("path %q must be a regular file without symlinks", path)
}
return nil
}
func validateLogDestination(path string) error {
if err := requireDirectory(filepath.Dir(path)); err != nil {
return err
}
info, err := os.Lstat(path)
if errors.Is(err, os.ErrNotExist) {
return nil
}
if err != nil {
return err
}
if info.Mode()&os.ModeSymlink != 0 || !info.Mode().IsRegular() {
return fmt.Errorf("path %q must be absent or a regular file without symlinks", path)
}
return nil
}

View File

@@ -0,0 +1,716 @@
package notarius
import (
"context"
"encoding/json"
"errors"
"os"
"path/filepath"
"reflect"
"strings"
"testing"
"time"
sharedsubprocess "gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
)
func TestSubprocessRunnerBuildsExactInvocationAndDiscoversBundle(t *testing.T) {
req := validRunRequest(t)
var captured sharedsubprocess.RunRequest
runner := &SubprocessRunner{run: func(_ context.Context, processReq sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
captured = processReq
writeValidBundleAndReceipt(t, req, true)
return sharedsubprocess.RunResult{ExitCode: 0, Duration: 2 * time.Second}, nil
}}
result, err := runner.Run(context.Background(), req)
if err != nil {
t.Fatalf("Run() error = %v", err)
}
wantArgs := []string{
"run", "dnd-session", "--config", req.ConfigPath, "--input", req.InputPath,
"--output-dir", req.OutputRoot, "--json",
}
if !reflect.DeepEqual(captured.Args, wantArgs) {
t.Fatalf("subprocess args = %#v, want %#v", captured.Args, wantArgs)
}
if captured.Executable != req.Binary || captured.WorkingDir != req.WorkingDirectory || captured.Timeout != req.Timeout {
t.Fatalf("subprocess request = %#v", captured)
}
if captured.StdoutLogPath != req.ReceiptPath || captured.StderrLogPath != req.LogPath {
t.Fatalf("stream paths = stdout %q stderr %q", captured.StdoutLogPath, captured.StderrLogPath)
}
if captured.EnvOverrides != nil {
t.Fatalf("environment overrides = %#v, want inherited environment only", captured.EnvOverrides)
}
for _, arg := range captured.Args {
if arg == "--session-id" {
t.Fatal("subprocess args unexpectedly contain --session-id")
}
}
if result.Receipt.SchemaVersion != ReceiptSchemaVersion || result.Receipt.RunID != "notarius-run-1" {
t.Fatalf("receipt = %#v", result.Receipt)
}
if len(result.Receipt.ValidationSummaries) != 1 || result.Receipt.ValidationSummaries[0].LaneID != "npc-registry" ||
result.Receipt.ValidationSummaries[0].Status != "complete" {
t.Fatalf("validation summaries = %#v", result.Receipt.ValidationSummaries)
}
if len(result.Index.Lanes) != 1 || result.Index.Lanes[0].LaneID != "npc-registry" {
t.Fatalf("lanes = %#v", result.Index.Lanes)
}
if result.Index.ChunkMap == nil || result.Index.ChunkMap.ArtifactKind != "chunk_map" {
t.Fatalf("chunk map = %#v", result.Index.ChunkMap)
}
if result.Index.EvidenceContext == nil || result.Index.EvidenceContext.ArtifactKind != "evidence_context" {
t.Fatalf("evidence context = %#v", result.Index.EvidenceContext)
}
if len(result.Rejections) != 1 || result.Rejections[0].LaneID != "spells" || result.Rejections[0].ReasonCode != "invalid_spell" {
t.Fatalf("rejections = %#v", result.Rejections)
}
if len(result.Warnings) != 1 || result.Warnings[0].Category != "degradation" || result.Warnings[0].ReasonCode != "normalized_name" {
t.Fatalf("warnings = %#v", result.Warnings)
}
if len(result.Diagnostics) != 1 || result.Diagnostics[0].Category != "data_quality" || result.Diagnostics[0].ReasonCode != "low_confidence" {
t.Fatalf("diagnostics = %#v", result.Diagnostics)
}
}
func TestSubprocessRunnerBuildsOrderedReferenceArguments(t *testing.T) {
req := validRunRequest(t)
referenceRoot := t.TempDir()
req.References = []ReferenceBinding{
{Selector: " party ", Path: filepath.Join(referenceRoot, "party context=primary.json")},
{Selector: " npc-registry . extract . glossary ", Path: filepath.Join(referenceRoot, "glossary.json")},
}
originalReferences := append([]ReferenceBinding(nil), req.References...)
var captured sharedsubprocess.RunRequest
runner := &SubprocessRunner{run: func(_ context.Context, processReq sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
captured = processReq
writeValidBundleAndReceipt(t, req, false)
return sharedsubprocess.RunResult{ExitCode: 0}, nil
}}
if _, err := runner.Run(context.Background(), req); err != nil {
t.Fatalf("Run() error = %v", err)
}
wantArgs := []string{
"run", req.PipelineID,
"--config", req.ConfigPath,
"--input", req.InputPath,
"--output-dir", req.OutputRoot,
"--reference", "party=" + req.References[0].Path,
"--reference", "npc-registry.extract.glossary=" + req.References[1].Path,
"--json",
}
if !reflect.DeepEqual(captured.Args, wantArgs) {
t.Fatalf("subprocess args = %#v, want %#v", captured.Args, wantArgs)
}
if !reflect.DeepEqual(req.References, originalReferences) {
t.Fatalf("Run() mutated caller references = %#v, want %#v", req.References, originalReferences)
}
}
func TestSubprocessRunnerRejectsInvalidReferencesBeforeLaunch(t *testing.T) {
tests := []struct {
name string
references func(string) []ReferenceBinding
wantErr string
}{
{
name: "invalid selector",
references: func(root string) []ReferenceBinding {
return []ReferenceBinding{{Selector: "lane.prepare.party", Path: filepath.Join(root, "party.json")}}
},
wantErr: "selector",
},
{
name: "duplicate normalized selector",
references: func(root string) []ReferenceBinding {
return []ReferenceBinding{
{Selector: "lane.party", Path: filepath.Join(root, "party.json")},
{Selector: " lane . party ", Path: filepath.Join(root, "party-2.json")},
}
},
wantErr: "duplicated",
},
{
name: "empty path",
references: func(string) []ReferenceBinding {
return []ReferenceBinding{{Selector: "party", Path: " "}}
},
wantErr: "path is required",
},
{
name: "relative path",
references: func(string) []ReferenceBinding {
return []ReferenceBinding{{Selector: "party", Path: "references/party.json"}}
},
wantErr: "path must be absolute",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
req := validRunRequest(t)
req.References = tt.references(t.TempDir())
started := false
runner := &SubprocessRunner{run: func(context.Context, sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
started = true
return sharedsubprocess.RunResult{}, nil
}}
_, err := runner.Run(context.Background(), req)
if err == nil || !strings.Contains(err.Error(), tt.wantErr) {
t.Fatalf("Run() error = %v, want containing %q", err, tt.wantErr)
}
if started {
t.Fatal("subprocess started after request validation failure")
}
})
}
}
func TestSubprocessRunnerUsesMinimalEnvironmentAndSeparatesStreams(t *testing.T) {
req := validRunRequest(t)
writeValidBundleAndReceipt(t, req, false)
receiptFixture := req.ReceiptPath + ".fixture"
data, err := os.ReadFile(req.ReceiptPath)
if err != nil {
t.Fatalf("ReadFile(receipt) error = %v", err)
}
if err := os.WriteFile(receiptFixture, data, 0o644); err != nil {
t.Fatalf("WriteFile(receipt fixture) error = %v", err)
}
if err := os.Remove(req.ReceiptPath); err != nil {
t.Fatalf("Remove(receipt) error = %v", err)
}
captureDir := filepath.Join(filepath.Dir(req.ReceiptPath), "capture")
if err := os.Mkdir(captureDir, 0o755); err != nil {
t.Fatalf("Mkdir(capture) error = %v", err)
}
script := writeShellScript(t, `#!/bin/sh
pwd > "$NOTARIUS_CAPTURE_DIR/working-directory"
printf '%s' "$NOTARIUS_INHERITED_VALUE" > "$NOTARIUS_CAPTURE_DIR/environment"
printf 'diagnostic stream\n' >&2
cat "$NOTARIUS_RECEIPT_FIXTURE"
`)
req.Binary = script
t.Setenv("NOTARIUS_CAPTURE_DIR", captureDir)
t.Setenv("NOTARIUS_INHERITED_VALUE", "inherited-value")
t.Setenv("NOTARIUS_RECEIPT_FIXTURE", receiptFixture)
if _, err := NewSubprocessRunner().Run(context.Background(), req); err != nil {
t.Fatalf("Run() error = %v", err)
}
assertTextFile(t, filepath.Join(captureDir, "working-directory"), req.WorkingDirectory+"\n")
assertTextFile(t, filepath.Join(captureDir, "environment"), "")
assertTextFile(t, req.LogPath, "diagnostic stream\n")
receiptBytes, err := os.ReadFile(req.ReceiptPath)
if err != nil {
t.Fatalf("ReadFile(receipt) error = %v", err)
}
if strings.Contains(string(receiptBytes), "diagnostic stream") {
t.Fatal("receipt contains stderr output")
}
}
func TestSubprocessRunnerReturnsProcessFailuresWithoutParsingStdout(t *testing.T) {
tests := []struct {
name string
scriptBody string
timeout time.Duration
cancel bool
want string
}{
{name: "nonzero", scriptBody: "printf '{malformed receipt'; printf 'failed\\n' >&2; exit 7\n", timeout: time.Second, want: "exit code 7"},
{name: "timeout", scriptBody: "sleep 5\n", timeout: 20 * time.Millisecond, want: "timed out"},
{name: "cancellation", scriptBody: "sleep 5\n", timeout: time.Second, cancel: true, want: "canceled"},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
req := validRunRequest(t)
req.Binary = writeShellScript(t, "#!/bin/sh\n"+test.scriptBody)
req.Timeout = test.timeout
ctx := context.Background()
if test.cancel {
cancelCtx, cancel := context.WithCancel(ctx)
ctx = cancelCtx
time.AfterFunc(20*time.Millisecond, cancel)
}
_, err := NewSubprocessRunner().Run(ctx, req)
if err == nil || !strings.Contains(err.Error(), test.want) {
t.Fatalf("Run() error = %v, want fragment %q", err, test.want)
}
if strings.Contains(err.Error(), "decode notarius receipt") {
t.Fatalf("Run() parsed stdout after process failure: %v", err)
}
})
}
}
func TestSubprocessRunnerReturnsSharedSubprocessErrorWithoutReadingReceipt(t *testing.T) {
req := validRunRequest(t)
if err := os.WriteFile(req.ReceiptPath, []byte("not json"), 0o644); err != nil {
t.Fatalf("WriteFile(receipt) error = %v", err)
}
wantErr := errors.New("process failed")
runner := &SubprocessRunner{run: func(context.Context, sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
return sharedsubprocess.RunResult{ExitCode: 9}, wantErr
}}
_, err := runner.Run(context.Background(), req)
if !errors.Is(err, wantErr) {
t.Fatalf("Run() error = %v, want wrapped process error", err)
}
if strings.Contains(err.Error(), "decode") {
t.Fatalf("Run() parsed receipt after failure: %v", err)
}
}
func TestLoadReceiptValidation(t *testing.T) {
root := t.TempDir()
valid := map[string]any{
"schema_version": ReceiptSchemaVersion, "run_id": "run-1", "pipeline_id": "pipeline-1",
"output_directory": filepath.Join(root, "outputs", "run-1"), "index_file": "index.json",
"normalized_output_count": 1, "rejected_output_count": 0,
"warning_group_count": 0, "warning_occurrence_count": 0,
"diagnostic_group_count": 0, "diagnostic_occurrence_count": 0,
"diagnostics_truncated": false, "validation_status": "approved", "future_field": true,
}
tests := []struct {
name string
mutate func(map[string]any)
raw []byte
wantOK bool
wantError string
}{
{name: "unknown fields tolerated", wantOK: true},
{name: "malformed", raw: []byte("{")},
{name: "unsupported version", mutate: func(v map[string]any) { v["schema_version"] = "notarius.run-result.v1" }},
{name: "missing field", mutate: func(v map[string]any) { delete(v, "run_id") }},
{name: "pipeline mismatch", mutate: func(v map[string]any) { v["pipeline_id"] = "other" }},
{name: "relative output", mutate: func(v map[string]any) { v["output_directory"] = "run-1" }},
{name: "negative count", mutate: func(v map[string]any) { v["warning_group_count"] = -1 }},
{name: "invalid validation status", mutate: func(v map[string]any) { v["validation_status"] = "valid" }},
{
name: "nested index", mutate: func(v map[string]any) { v["index_file"] = "nested/index.json" },
wantError: `index_file "nested/index.json"`,
},
{
name: "cleanable index", mutate: func(v map[string]any) { v["index_file"] = "./index.json" },
wantError: `index_file "./index.json"`,
},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
path := filepath.Join(root, strings.ReplaceAll(test.name, " ", "-")+".json")
values := cloneMap(valid)
if test.mutate != nil {
test.mutate(values)
}
if test.raw != nil {
if err := os.WriteFile(path, test.raw, 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
} else {
writeJSONFile(t, path, values)
}
_, err := loadReceipt(path, "pipeline-1")
if test.wantOK && err != nil {
t.Fatalf("loadReceipt() error = %v", err)
}
if !test.wantOK && err == nil {
t.Fatal("loadReceipt() error = nil, want validation failure")
}
if test.wantError != "" && !strings.Contains(err.Error(), test.wantError) {
t.Fatalf("loadReceipt() error = %v, want fragment %q", err, test.wantError)
}
})
}
oversized := filepath.Join(root, "oversized.json")
if err := os.WriteFile(oversized, []byte(strings.Repeat("x", maxReceiptBytes+1)), 0o644); err != nil {
t.Fatalf("WriteFile(oversized) error = %v", err)
}
if _, err := loadReceipt(oversized, "pipeline-1"); err == nil || !strings.Contains(err.Error(), "exceeds") {
t.Fatalf("loadReceipt(oversized) error = %v", err)
}
}
func TestValidateBundleRootRejectsEscapesAndSymlinks(t *testing.T) {
root := t.TempDir()
outputRoot := filepath.Join(root, "output")
if err := os.Mkdir(outputRoot, 0o755); err != nil {
t.Fatalf("Mkdir(output root) error = %v", err)
}
validBundle := filepath.Join(outputRoot, "run-1")
if err := os.Mkdir(validBundle, 0o755); err != nil {
t.Fatalf("Mkdir(bundle) error = %v", err)
}
if _, err := validateBundleRoot(outputRoot, validBundle); err != nil {
t.Fatalf("validateBundleRoot(valid) error = %v", err)
}
outside := filepath.Join(root, "output-other")
if err := os.Mkdir(outside, 0o755); err != nil {
t.Fatalf("Mkdir(outside) error = %v", err)
}
for name, candidate := range map[string]string{"equal root": outputRoot, "escape": root, "prefix confusion": outside} {
t.Run(name, func(t *testing.T) {
if _, err := validateBundleRoot(outputRoot, candidate); err == nil {
t.Fatalf("validateBundleRoot(%q) error = nil", candidate)
}
})
}
symlink := filepath.Join(outputRoot, "linked")
if err := os.Symlink(outside, symlink); err != nil {
t.Skipf("Symlink() unavailable: %v", err)
}
if _, err := validateBundleRoot(outputRoot, symlink); err == nil {
t.Fatal("validateBundleRoot(symlink) error = nil")
}
}
func TestLoadIndexRejectsMalformedUnsafeAndUnsupportedDocuments(t *testing.T) {
tests := []struct {
name string
indexValue any
prepare func(*testing.T, string)
wantError string
}{
{name: "malformed", indexValue: json.RawMessage(`{"manifest_file":`)},
{name: "unsupported output shape", indexValue: map[string]any{"manifest_file": "manifest.json", "output_files": map[string]any{}, "rejected_file": "rejected.json", "warnings_file": "warnings.json"}},
{name: "missing management path", indexValue: map[string]any{"output_files": []any{}, "rejected_file": "rejected.json", "warnings_file": "warnings.json"}},
{name: "renamed manifest", indexValue: func() any {
value := validIndexValue([]any{})
value["manifest_file"] = "metadata.json"
return value
}(), wantError: `manifest_file "metadata.json"`},
{name: "cleanable manifest", indexValue: func() any {
value := validIndexValue([]any{})
value["manifest_file"] = "./manifest.json"
return value
}(), wantError: `manifest_file "./manifest.json"`},
{name: "renamed rejections", indexValue: func() any {
value := validIndexValue([]any{})
value["rejected_file"] = "rejections.json"
return value
}(), wantError: `rejected_file "rejections.json"`},
{name: "renamed warnings", indexValue: func() any {
value := validIndexValue([]any{})
value["warnings_file"] = "diagnostics/warnings.json"
return value
}(), wantError: `warnings_file "diagnostics/warnings.json"`},
{name: "duplicate lane", indexValue: validIndexValue([]any{
map[string]any{"lane_id": "npc", "file": "lanes/npc.json"},
map[string]any{"lane_id": "npc", "file": "lanes/npc.json"},
})},
{name: "absolute logical path", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "/tmp/npc.json"}})},
{name: "lexical traversal", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "../outside.json"}})},
{name: "root prefix confusion", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "../bundle-other/npc.json"}})},
{name: "file symlink", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "lanes/npc.json"}}), prepare: func(t *testing.T, bundle string) {
if err := os.Symlink(filepath.Join(bundle, "manifest.json"), filepath.Join(bundle, "lanes", "npc.json")); err != nil {
t.Skipf("Symlink() unavailable: %v", err)
}
}},
{name: "directory symlink", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "linked/npc.json"}}), prepare: func(t *testing.T, bundle string) {
if err := os.Symlink(filepath.Join(bundle, "lanes"), filepath.Join(bundle, "linked")); err != nil {
t.Skipf("Symlink() unavailable: %v", err)
}
}},
{name: "missing management file", indexValue: validIndexValue([]any{}), prepare: func(t *testing.T, bundle string) {
if err := os.Remove(filepath.Join(bundle, "manifest.json")); err != nil {
t.Fatalf("Remove(manifest) error = %v", err)
}
}},
{name: "incomplete pipeline descriptor", indexValue: func() any {
value := validIndexValue([]any{})
value["chunk_map"] = map[string]any{"artifact_kind": "chunk_map", "file": "chunk-map.json"}
return value
}()},
{name: "pipeline descriptor escape", indexValue: func() any {
value := validIndexValue([]any{})
value["evidence_context"] = map[string]any{
"artifact_kind": "evidence_context", "file": "../evidence.json", "media_type": "application/json",
"schema_id": "evidence", "schema_name": "Evidence", "schema_version": "v1",
}
return value
}()},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
bundle := createBundleSkeleton(t)
indexPath := filepath.Join(bundle, "index.json")
if raw, ok := test.indexValue.(json.RawMessage); ok {
if err := os.WriteFile(indexPath, raw, 0o644); err != nil {
t.Fatalf("WriteFile(index) error = %v", err)
}
} else {
writeJSONFile(t, indexPath, test.indexValue)
}
if test.prepare != nil {
test.prepare(t, bundle)
}
if _, err := loadIndex(bundle, indexPath); err == nil {
t.Fatal("loadIndex() error = nil, want failure")
} else if test.wantError != "" && !strings.Contains(err.Error(), test.wantError) {
t.Fatalf("loadIndex() error = %v, want fragment %q", err, test.wantError)
}
})
}
bundle := createBundleSkeleton(t)
oversizedIndex := filepath.Join(bundle, "index.json")
if err := os.WriteFile(oversizedIndex, []byte(strings.Repeat("x", maxIndexBytes+1)), 0o644); err != nil {
t.Fatalf("WriteFile(oversized index) error = %v", err)
}
if _, err := loadIndex(bundle, oversizedIndex); err == nil || !strings.Contains(err.Error(), "exceeds") {
t.Fatalf("loadIndex(oversized) error = %v", err)
}
}
func TestLoadDiagnosticSummariesValidateBoundsAndTolerateUnknownFields(t *testing.T) {
root := t.TempDir()
rejectedPath := filepath.Join(root, "rejected.json")
warningsPath := filepath.Join(root, "warnings.json")
diagnosticsPath := filepath.Join(root, "diagnostics.json")
writeJSONFile(t, rejectedPath, map[string]any{"rejected": []any{map[string]any{
"stage": "validate", "lane_id": "spells", "reason_code": "invalid", "message": "do not retain this", "future": true,
}}, "future": true})
writeJSONFile(t, warningsPath, findingEnvelope(warningsSchemaVersion, []any{findingGroup("warning", "degradation", "bounded", "normalize", 2)}, 2, false, 0))
writeJSONFile(t, diagnosticsPath, findingEnvelope(diagnosticsSchemaVersion, []any{findingGroup("advisory", "data_quality", "low_confidence", "normalize", 3)}, 4, true, 1))
rejections, err := loadRejections(rejectedPath)
if err != nil || len(rejections) != 1 || rejections[0].ReasonCode != "invalid" {
t.Fatalf("loadRejections() = %#v, %v", rejections, err)
}
warnings, err := loadWarnings(warningsPath)
if err != nil || len(warnings) != 1 || warnings[0].Category != "degradation" || warnings[0].OccurrenceCount != 2 {
t.Fatalf("loadWarnings() = %#v, %v", warnings, err)
}
diagnostics, occurrences, truncated, err := loadDiagnostics(diagnosticsPath)
if err != nil || len(diagnostics) != 1 || occurrences != 4 || !truncated || diagnostics[0].Category != "data_quality" {
t.Fatalf("loadDiagnostics() = %#v, %d, %t, %v", diagnostics, occurrences, truncated, err)
}
for name, path := range map[string]string{"rejections": rejectedPath, "warnings": warningsPath, "diagnostics": diagnosticsPath} {
t.Run("malformed "+name, func(t *testing.T) {
if err := os.WriteFile(path, []byte("{"), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
var err error
if name == "rejections" {
_, err = loadRejections(path)
} else if name == "warnings" {
_, err = loadWarnings(path)
} else {
_, _, _, err = loadDiagnostics(path)
}
if err == nil {
t.Fatal("summary decoder error = nil")
}
})
}
oversized := filepath.Join(root, "oversized.json")
if err := os.WriteFile(oversized, []byte(strings.Repeat("x", maxSummaryBytes+1)), 0o644); err != nil {
t.Fatalf("WriteFile(oversized) error = %v", err)
}
if _, err := loadWarnings(oversized); err == nil || !strings.Contains(err.Error(), "exceeds") {
t.Fatalf("loadWarnings(oversized) error = %v", err)
}
if _, err := loadRejections(oversized); err == nil || !strings.Contains(err.Error(), "exceeds") {
t.Fatalf("loadRejections(oversized) error = %v", err)
}
if _, _, _, err := loadDiagnostics(oversized); err == nil || !strings.Contains(err.Error(), "exceeds") {
t.Fatalf("loadDiagnostics(oversized) error = %v", err)
}
}
func TestFakeRunnerCapturesRequestsAndHonorsContextAndError(t *testing.T) {
req := RunRequest{PipelineID: "pipeline", References: []ReferenceBinding{{Selector: "party", Path: "/references/party.json"}}}
want := RunResult{BundleRoot: "/bundle"}
fake := &FakeRunner{Result: want}
got, err := fake.Run(context.Background(), req)
if err != nil || !reflect.DeepEqual(got, want) || !reflect.DeepEqual(fake.Requests, []RunRequest{req}) {
t.Fatalf("Run() = %#v, %v; requests = %#v", got, err, fake.Requests)
}
req.References[0].Path = "/references/changed.json"
if fake.Requests[0].References[0].Path != "/references/party.json" {
t.Fatalf("fake retained aliased request references: %#v", fake.Requests[0].References)
}
wantErr := errors.New("configured failure")
fake.Err = wantErr
if _, err := fake.Run(context.Background(), req); !errors.Is(err, wantErr) {
t.Fatalf("Run(configured error) = %v", err)
}
canceled, cancel := context.WithCancel(context.Background())
cancel()
before := len(fake.Requests)
if _, err := fake.Run(canceled, req); !errors.Is(err, context.Canceled) || len(fake.Requests) != before {
t.Fatalf("Run(canceled) error = %v; requests = %d", err, len(fake.Requests))
}
}
func validRunRequest(t *testing.T) RunRequest {
t.Helper()
root := t.TempDir()
configPath := filepath.Join(root, "notarius.yml")
inputPath := filepath.Join(root, "input.json")
outputRoot := filepath.Join(root, "outputs")
workingDirectory := filepath.Join(root, "work")
diagnostics := filepath.Join(root, "diagnostics")
for _, directory := range []string{outputRoot, workingDirectory, diagnostics} {
if err := os.Mkdir(directory, 0o755); err != nil {
t.Fatalf("Mkdir(%q) error = %v", directory, err)
}
}
if err := os.WriteFile(configPath, []byte("pipelines: {}\n"), 0o644); err != nil {
t.Fatalf("WriteFile(config) error = %v", err)
}
if err := os.WriteFile(inputPath, []byte("{}\n"), 0o644); err != nil {
t.Fatalf("WriteFile(input) error = %v", err)
}
return RunRequest{
Binary: "notarius", ConfigPath: configPath, PipelineID: "dnd-session", InputPath: inputPath,
OutputRoot: outputRoot, WorkingDirectory: workingDirectory,
ReceiptPath: filepath.Join(diagnostics, "receipt.json"), LogPath: filepath.Join(diagnostics, "stderr.log"),
Timeout: time.Second,
}
}
func writeValidBundleAndReceipt(t *testing.T, req RunRequest, includeUnknown bool) {
t.Helper()
bundle := filepath.Join(req.OutputRoot, "notarius-run-1")
if err := os.MkdirAll(filepath.Join(bundle, "lanes"), 0o755); err != nil {
t.Fatalf("MkdirAll(bundle) error = %v", err)
}
for path, data := range map[string]string{
"manifest.json": `{}`,
"lanes/npc.json": `{}`,
"chunk-map.json": `{}`,
"evidence-context.json": `{}`,
} {
if err := os.WriteFile(filepath.Join(bundle, filepath.FromSlash(path)), []byte(data), 0o644); err != nil {
t.Fatalf("WriteFile(%q) error = %v", path, err)
}
}
rejection := map[string]any{"stage": "validate", "lane_id": "spells", "reason_code": "invalid_spell", "message": strings.Repeat("external detail", 20)}
if includeUnknown {
rejection["future"] = true
}
writeJSONFile(t, filepath.Join(bundle, "rejected.json"), map[string]any{"rejected": []any{rejection}, "future": true})
writeJSONFile(t, filepath.Join(bundle, "warnings.json"), findingEnvelope(warningsSchemaVersion, []any{findingGroup("warning", "degradation", "normalized_name", "normalize", 2)}, 2, false, 0))
writeJSONFile(t, filepath.Join(bundle, "diagnostics.json"), findingEnvelope(diagnosticsSchemaVersion, []any{findingGroup("advisory", "data_quality", "low_confidence", "normalize", 3)}, 4, true, 1))
index := validIndexValue([]any{map[string]any{
"lane_id": "npc-registry", "file": "lanes/npc.json", "media_type": "application/json",
"module_key": "dnd/npc-registry", "schema_id": "notarius.dnd.npc_registry",
"schema_name": "NPCRegistry", "schema_version": "v1", "future": true,
}})
index["chunk_map"] = map[string]any{
"artifact_kind": "chunk_map", "file": "chunk-map.json", "media_type": "application/json",
"schema_id": "notarius.chunk_map", "schema_name": "ChunkMap", "schema_version": "v1", "future": true,
}
index["evidence_context"] = map[string]any{
"artifact_kind": "evidence_context", "file": "evidence-context.json", "media_type": "application/json",
"schema_id": "notarius.evidence_context", "schema_name": "EvidenceContext", "schema_version": "v1", "future": true,
}
index["future"] = true
writeJSONFile(t, filepath.Join(bundle, "index.json"), index)
receipt := map[string]any{
"schema_version": ReceiptSchemaVersion, "run_id": "notarius-run-1", "pipeline_id": req.PipelineID,
"output_directory": bundle, "index_file": "index.json", "normalized_output_count": 1,
"rejected_output_count": 1, "warning_group_count": 1, "warning_occurrence_count": 2,
"diagnostic_group_count": 1, "diagnostic_occurrence_count": 4,
"diagnostics_truncated": true, "validation_status": "rejected",
"validation_summaries": []any{map[string]any{
"stage": "normalize", "lane_id": "npc-registry", "status": "complete",
"producer_attempt_count": 1, "terminal_action": "accepted",
}},
}
if includeUnknown {
receipt["future"] = true
}
writeJSONFile(t, req.ReceiptPath, receipt)
}
func createBundleSkeleton(t *testing.T) string {
t.Helper()
bundle := filepath.Join(t.TempDir(), "bundle")
if err := os.MkdirAll(filepath.Join(bundle, "lanes"), 0o755); err != nil {
t.Fatalf("MkdirAll(bundle) error = %v", err)
}
for _, name := range []string{"manifest.json", "rejected.json", "warnings.json", "diagnostics.json", "lanes/npc.json", "chunk-map.json"} {
if err := os.WriteFile(filepath.Join(bundle, filepath.FromSlash(name)), []byte("{}"), 0o644); err != nil {
t.Fatalf("WriteFile(%q) error = %v", name, err)
}
}
return bundle
}
func validIndexValue(lanes []any) map[string]any {
return map[string]any{
"manifest_file": "manifest.json", "output_files": lanes,
"rejected_file": "rejected.json", "warnings_file": "warnings.json", "diagnostics_file": "diagnostics.json",
}
}
func findingGroup(disposition, category, reasonCode, origin string, occurrences int) map[string]any {
return map[string]any{
"disposition": disposition, "category": category, "reason_code": reasonCode,
"origin": map[string]any{"stage": origin, "lane_id": "npc-registry"}, "occurrence_count": occurrences,
"samples": []any{map[string]any{"scope": "lane:npc-registry", "message": "external detail"}},
"omitted_sample_count": occurrences - 1,
}
}
func findingEnvelope(schema string, groups []any, occurrences int, truncated bool, unrepresented int) map[string]any {
value := map[string]any{
"schema_version": schema, "group_count": len(groups), "occurrence_count": occurrences,
"groups": groups,
}
if schema == diagnosticsSchemaVersion {
value["truncated"] = truncated
value["unrepresented_occurrence_count"] = unrepresented
}
return value
}
func writeJSONFile(t *testing.T, path string, value any) {
t.Helper()
data, err := json.Marshal(value)
if err != nil {
t.Fatalf("json.Marshal() error = %v", err)
}
if err := os.WriteFile(path, data, 0o644); err != nil {
t.Fatalf("WriteFile(%q) error = %v", path, err)
}
}
func writeShellScript(t *testing.T, body string) string {
t.Helper()
path := filepath.Join(t.TempDir(), "notarius-helper")
if err := os.WriteFile(path, []byte(body), 0o755); err != nil {
t.Fatalf("WriteFile(script) error = %v", err)
}
return path
}
func assertTextFile(t *testing.T, path, want string) {
t.Helper()
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("ReadFile(%q) error = %v", path, err)
}
if string(data) != want {
t.Fatalf("ReadFile(%q) = %q, want %q", path, string(data), want)
}
}
func cloneMap(source map[string]any) map[string]any {
result := make(map[string]any, len(source))
for key, value := range source {
result[key] = value
}
return result
}

View File

@@ -5,6 +5,7 @@ import (
"fmt"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
)
// NoopRunner is a deterministic no-op scriptorium adapter.
@@ -142,7 +143,7 @@ func (f *FakeRunner) RenderArtifact(ctx context.Context, req RenderArtifactReque
func materializeRunPlaceholders(req RunArtifactRequest) error {
if req.OutputPath != "" {
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("scriptorium noop/fake run artifact\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("scriptorium noop/fake run artifact\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write run output %q: %w", req.OutputPath, err)
}
}
@@ -154,17 +155,17 @@ func materializeRunPlaceholders(req RunArtifactRequest) error {
"prompt_id": req.PromptID,
"output_path": req.OutputPath,
}
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
}
}
if req.StdoutLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake run stdout placeholder\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake run stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
}
}
if req.StderrLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake run stderr placeholder\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake run stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
}
}
@@ -173,7 +174,7 @@ func materializeRunPlaceholders(req RunArtifactRequest) error {
func materializeRenderPlaceholders(req RenderArtifactRequest) error {
if req.OutputPath != "" {
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("{\"schema\":\"scriptorium.render.v1\",\"placeholder\":true}\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("{\"schema\":\"scriptorium.render.v1\",\"placeholder\":true}\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write render output %q: %w", req.OutputPath, err)
}
}
@@ -185,17 +186,17 @@ func materializeRenderPlaceholders(req RenderArtifactRequest) error {
"prompt_id": req.PromptID,
"output_path": req.OutputPath,
}
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
}
}
if req.StdoutLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake render stdout placeholder\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake render stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
}
}
if req.StderrLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake render stderr placeholder\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake render stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
}
}

View File

@@ -9,8 +9,12 @@ import (
"time"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
)
// MaxOutputFileBytes bounds one Scriptorium artifact result.
const MaxOutputFileBytes int64 = 64 * 1024 * 1024
// SubprocessRunner invokes Scriptorium through its public CLI.
type SubprocessRunner struct{}
@@ -52,13 +56,17 @@ func (r *SubprocessRunner) RunArtifact(ctx context.Context, req RunArtifactReque
}
}
envOverrides, sensitiveNames := credentialEnvironment(req.APIKeyEnv)
runRes, runErr := subprocess.Run(ctx, subprocess.RunRequest{
Executable: req.Binary,
Args: args,
WorkingDir: req.WorkingDir,
Timeout: req.Timeout,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
Executable: req.Binary,
Args: args,
WorkingDir: req.WorkingDir,
Timeout: req.Timeout,
EnvOverrides: envOverrides,
SensitiveEnvNames: sensitiveNames,
DiagnosticOwner: "scriptorium",
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
})
result := ArtifactResult{
@@ -134,13 +142,17 @@ func (r *SubprocessRunner) RenderArtifact(ctx context.Context, req RenderArtifac
}
}
envOverrides, sensitiveNames := credentialEnvironment(req.APIKeyEnv)
runRes, runErr := subprocess.Run(ctx, subprocess.RunRequest{
Executable: req.Binary,
Args: args,
WorkingDir: req.WorkingDir,
Timeout: req.Timeout,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
Executable: req.Binary,
Args: args,
WorkingDir: req.WorkingDir,
Timeout: req.Timeout,
EnvOverrides: envOverrides,
SensitiveEnvNames: sensitiveNames,
DiagnosticOwner: "scriptorium",
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
})
result := ArtifactResult{
@@ -223,6 +235,15 @@ func validateCommonRunRequest(
return true, nil
}
func credentialEnvironment(apiKeyEnv string) (map[string]string, []string) {
name := strings.TrimSpace(apiKeyEnv)
if name == "" {
return nil, nil
}
value, _ := os.LookupEnv(name)
return map[string]string{name: value}, []string{name}
}
func buildRunArgs(req RunArtifactRequest) []string {
args := []string{"run", "--prompt", strings.TrimSpace(req.PromptID)}
if cfgPath := strings.TrimSpace(req.ConfigPath); cfgPath != "" {
@@ -321,18 +342,15 @@ func writeInvocationConfig(path string, payload invocationPayload) error {
"render_format": payload.RenderFormat,
"render_prompt_logged": payload.RenderPromptStore,
}
return subprocess.WriteYAMLAtomic(path, data, 0o644)
return subprocess.WriteYAMLAtomic(path, data, fileops.WorkspaceFileMode)
}
func validateNonEmptyOutput(path string) error {
info, err := os.Stat(path)
data, err := fileops.ReadRegularFile(path, MaxOutputFileBytes)
if err != nil {
return fmt.Errorf("stat file: %w", err)
return fmt.Errorf("scriptorium artifact output exceeds or cannot be read within %d-byte limit: %w", MaxOutputFileBytes, err)
}
if info.IsDir() {
return fmt.Errorf("path is a directory")
}
if info.Size() <= 0 {
if len(data) == 0 {
return fmt.Errorf("file is empty")
}
return nil

View File

@@ -31,7 +31,7 @@ func TestSubprocessRunnerRunSuccessBuildsDeterministicArgsAndCapturesLogs(t *tes
ConfigPath: "/etc/scriptorium/config.yml",
PromptID: "dnd.session_recap",
ProfileID: "local-quality",
InputPaths: map[string]string{"transcript": filepath.Join(dir, "processed.json"), "other": filepath.Join(dir, "other.md")},
InputPaths: map[string]string{"transcript": filepath.Join(dir, "polished.json"), "other": filepath.Join(dir, "other.md")},
Vars: map[string]string{"session_id": "2026-05-03", "campaign_name": "Icewind Dale"},
OutputPath: filepath.Join(dir, "artifacts", "session_recap.md"),
StdoutLogPath: filepath.Join(dir, "logs", "scriptorium.run.stdout.log"),
@@ -180,7 +180,7 @@ func TestSubprocessRunnerRenderSuccess(t *testing.T) {
req := RenderArtifactRequest{
Binary: wrapper,
PromptID: "dnd.session_recap",
InputPaths: map[string]string{"transcript": filepath.Join(dir, "processed.json")},
InputPaths: map[string]string{"transcript": filepath.Join(dir, "polished.json")},
OutputPath: filepath.Join(dir, "artifacts", "session_recap.render.json"),
StdoutLogPath: filepath.Join(dir, "logs", "scriptorium.render.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "scriptorium.render.stderr.log"),
@@ -285,7 +285,7 @@ type scriptoriumHelperRecord struct {
func runReqForTest(t *testing.T, binary string) RunArtifactRequest {
t.Helper()
dir := t.TempDir()
transcriptPath := filepath.Join(dir, "processed.json")
transcriptPath := filepath.Join(dir, "polished.json")
writeScriptoriumFile(t, transcriptPath, `{"segments":[]}`)
return RunArtifactRequest{
Binary: binary,

View File

@@ -5,6 +5,7 @@ import (
"fmt"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
)
// NoopRunner is a deterministic no-op seriatim adapter.
@@ -68,6 +69,26 @@ func (n *NoopRunner) Normalize(ctx context.Context, req NormalizeRequest) (Norma
}, nil
}
// Render returns the requested output path with placeholder metadata.
func (n *NoopRunner) Render(ctx context.Context, req RenderRequest) (RenderResult, error) {
if err := ctx.Err(); err != nil {
return RenderResult{}, err
}
if err := materializeRenderPlaceholders(req); err != nil {
return RenderResult{}, err
}
return RenderResult{
OutputRenderedPath: req.OutputRenderedPath,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
GeneratedConfigPath: req.GeneratedConfigPath,
InvokedBinary: "noop",
Format: req.Format,
Title: req.Title,
Metadata: map[string]any{"placeholder": true},
}, nil
}
// FakeRunner captures merge requests and returns deterministic responses.
type FakeRunner struct {
Requests []MergeRequest
@@ -79,6 +100,9 @@ type FakeRunner struct {
TrimRequests []TrimRequest
TrimErr error
TrimResult TrimResult
RenderRequests []RenderRequest
RenderErr error
RenderResult RenderResult
}
// Run records request and returns configured response.
@@ -195,9 +219,49 @@ func (f *FakeRunner) Normalize(ctx context.Context, req NormalizeRequest) (Norma
return res, nil
}
// Render records request and returns configured response.
func (f *FakeRunner) Render(ctx context.Context, req RenderRequest) (RenderResult, error) {
if err := ctx.Err(); err != nil {
return RenderResult{}, err
}
f.RenderRequests = append(f.RenderRequests, req)
if f.RenderErr != nil {
return RenderResult{}, f.RenderErr
}
if err := materializeRenderPlaceholders(req); err != nil {
return RenderResult{}, err
}
res := f.RenderResult
if res.OutputRenderedPath == "" {
res.OutputRenderedPath = req.OutputRenderedPath
}
if res.StdoutLogPath == "" {
res.StdoutLogPath = req.StdoutLogPath
}
if res.StderrLogPath == "" {
res.StderrLogPath = req.StderrLogPath
}
if res.GeneratedConfigPath == "" {
res.GeneratedConfigPath = req.GeneratedConfigPath
}
if res.InvokedBinary == "" {
res.InvokedBinary = "fake"
}
if res.Format == "" {
res.Format = req.Format
}
if res.Title == "" {
res.Title = req.Title
}
if res.Metadata == nil {
res.Metadata = map[string]any{"fake": true}
}
return res, nil
}
func materializePlaceholders(req MergeRequest) error {
if req.OutputMergedTranscriptPath != "" {
if err := subprocess.WriteFileAtomic(req.OutputMergedTranscriptPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.OutputMergedTranscriptPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write merged transcript %q: %w", req.OutputMergedTranscriptPath, err)
}
}
@@ -208,22 +272,22 @@ func materializePlaceholders(req MergeRequest) error {
"input_transcript_paths": req.InputTranscriptPaths,
"output_path": req.OutputMergedTranscriptPath,
}
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
}
}
if req.StdoutLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake stdout placeholder\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
}
}
if req.StderrLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake stderr placeholder\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
}
}
if req.ReportPath != "" {
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write report %q: %w", req.ReportPath, err)
}
}
@@ -232,7 +296,7 @@ func materializePlaceholders(req MergeRequest) error {
func materializeTrimPlaceholders(req TrimRequest) error {
if req.OutputTrimmedPath != "" {
if err := subprocess.WriteFileAtomic(req.OutputTrimmedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.OutputTrimmedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write trimmed transcript %q: %w", req.OutputTrimmedPath, err)
}
}
@@ -245,17 +309,17 @@ func materializeTrimPlaceholders(req TrimRequest) error {
"output_path": req.OutputTrimmedPath,
"keep_selector": req.KeepSelector,
}
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
}
}
if req.StdoutLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake trim stdout placeholder\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake trim stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
}
}
if req.StderrLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake trim stderr placeholder\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake trim stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
}
}
@@ -264,7 +328,7 @@ func materializeTrimPlaceholders(req TrimRequest) error {
func materializeNormalizePlaceholders(req NormalizeRequest) error {
if req.OutputNormalizedPath != "" {
if err := subprocess.WriteFileAtomic(req.OutputNormalizedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.OutputNormalizedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write normalized transcript %q: %w", req.OutputNormalizedPath, err)
}
}
@@ -280,24 +344,60 @@ func materializeNormalizePlaceholders(req NormalizeRequest) error {
if req.ReportPath != "" {
payload["report_path"] = req.ReportPath
}
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
}
}
if req.StdoutLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake normalize stdout placeholder\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake normalize stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
}
}
if req.StderrLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake normalize stderr placeholder\n"), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake normalize stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
}
}
if req.ReportPath != "" {
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), 0o644); err != nil {
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write report %q: %w", req.ReportPath, err)
}
}
return nil
}
func materializeRenderPlaceholders(req RenderRequest) error {
if req.OutputRenderedPath != "" {
if err := subprocess.WriteFileAtomic(req.OutputRenderedPath, []byte("# Transcript\n\nRendered markdown placeholder.\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write rendered transcript %q: %w", req.OutputRenderedPath, err)
}
}
if req.GeneratedConfigPath != "" {
payload := map[string]any{
"schema": "seriatim.generated.v1",
"placeholder": true,
"command": "render",
"input_path": req.InputTranscriptPath,
"output_path": req.OutputRenderedPath,
"format": req.Format,
"title": req.Title,
"include_timestamps": req.IncludeTimestamps,
"include_segment_ids": req.IncludeSegmentIDs,
"include_metadata": req.IncludeMetadata,
}
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
}
}
if req.StdoutLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake render stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
}
}
if req.StderrLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake render stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
}
}
return nil
}

View File

@@ -14,7 +14,7 @@ func TestFakeRunnerCapturesRequestAndReturnsPath(t *testing.T) {
dir := t.TempDir()
req := MergeRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.yml"),
OutputMergedTranscriptPath: filepath.Join(dir, "transcripts", "merged.json"),
OutputMergedTranscriptPath: filepath.Join(dir, "transcripts", "base.json"),
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "seriatim.stderr.log"),
}
@@ -57,8 +57,8 @@ func TestFakeRunnerTrimCapturesRequestAndReturnsPath(t *testing.T) {
dir := t.TempDir()
req := TrimRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.trim.yml"),
InputTranscriptPath: filepath.Join(dir, "transcripts", "processed.json"),
OutputTrimmedPath: filepath.Join(dir, "transcripts", "trimmed.json"),
InputTranscriptPath: filepath.Join(dir, "transcripts", "polished.json"),
OutputTrimmedPath: filepath.Join(dir, "transcripts", "final.trimmed.json"),
KeepSelector: "1-10",
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.trim.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "seriatim.trim.stderr.log"),
@@ -105,8 +105,8 @@ func TestFakeRunnerNormalizeCapturesRequestAndReturnsPath(t *testing.T) {
dir := t.TempDir()
req := NormalizeRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.normalize.yml"),
InputTranscriptPath: filepath.Join(dir, "transcripts", "processed.json"),
OutputNormalizedPath: filepath.Join(dir, "transcripts", "normalized.json"),
InputTranscriptPath: filepath.Join(dir, "transcripts", "polished.json"),
OutputNormalizedPath: filepath.Join(dir, "transcripts", "final.json"),
OutputSchema: "seriatim-intermediate",
ReportPath: filepath.Join(dir, "artifacts", "seriatim.normalize.report.json"),
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.normalize.stdout.log"),
@@ -148,3 +148,58 @@ func TestFakeRunnerNormalizeError(t *testing.T) {
t.Fatal("expected error, got nil")
}
}
func TestFakeRunnerRenderCapturesRequestAndReturnsPath(t *testing.T) {
fake := &FakeRunner{}
dir := t.TempDir()
req := RenderRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.render.yml"),
InputTranscriptPath: filepath.Join(dir, "transcripts", "final.trimmed.json"),
OutputRenderedPath: filepath.Join(dir, "transcripts", "final.trimmed.md"),
Format: "markdown",
Title: "Session render",
IncludeTimestamps: true,
IncludeSegmentIDs: false,
IncludeMetadata: true,
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.render.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "seriatim.render.stderr.log"),
}
res, err := fake.Render(context.Background(), req)
if err != nil {
t.Fatalf("Render() error = %v", err)
}
if len(fake.RenderRequests) != 1 || fake.RenderRequests[0].GeneratedConfigPath == "" {
t.Fatalf("render requests = %#v, want captured request", fake.RenderRequests)
}
if res.OutputRenderedPath != req.OutputRenderedPath {
t.Fatalf("rendered path = %q, want %q", res.OutputRenderedPath, req.OutputRenderedPath)
}
if res.Format != req.Format {
t.Fatalf("format = %q, want %q", res.Format, req.Format)
}
if res.Title != req.Title {
t.Fatalf("title = %q, want %q", res.Title, req.Title)
}
cfgData, err := os.ReadFile(req.GeneratedConfigPath)
if err != nil {
t.Fatalf("read generated config: %v", err)
}
if !strings.Contains(string(cfgData), "command: render") {
t.Fatalf("generated config = %q, want render command marker", string(cfgData))
}
for _, path := range []string{req.StdoutLogPath, req.StderrLogPath, req.OutputRenderedPath} {
if _, err := os.Stat(path); err != nil {
t.Fatalf("expected file %q to exist: %v", path, err)
}
}
}
func TestFakeRunnerRenderError(t *testing.T) {
fake := &FakeRunner{RenderErr: errors.New("boom")}
_, err := fake.Render(context.Background(), RenderRequest{})
if err == nil {
t.Fatal("expected error, got nil")
}
}

View File

@@ -1,4 +1,4 @@
// Package seriatim declares the adapter contract for transcript merge/normalize/trim execution.
// Package seriatim declares the adapter contract for transcript merge/normalize/trim/render execution.
package seriatim
import (
@@ -6,11 +6,12 @@ import (
"time"
)
// Runner is the adapter boundary for seriatim merge/normalize/trim invocations.
// Runner is the adapter boundary for seriatim merge/normalize/trim/render invocations.
type Runner interface {
Run(ctx context.Context, req MergeRequest) (MergeResult, error)
Normalize(ctx context.Context, req NormalizeRequest) (NormalizeResult, error)
Trim(ctx context.Context, req TrimRequest) (TrimResult, error)
Render(ctx context.Context, req RenderRequest) (RenderResult, error)
}
// MergeRequest describes a seriatim merge invocation.
@@ -90,3 +91,33 @@ type TrimResult struct {
KeepSelector string
Metadata map[string]any
}
// RenderRequest describes a seriatim render invocation.
type RenderRequest struct {
Binary string
InputTranscriptPath string
OutputRenderedPath string
Format string
Title string
IncludeTimestamps bool
IncludeSegmentIDs bool
IncludeMetadata bool
StdoutLogPath string
StderrLogPath string
GeneratedConfigPath string
Timeout time.Duration
}
// RenderResult describes a render output.
type RenderResult struct {
OutputRenderedPath string
StdoutLogPath string
StderrLogPath string
GeneratedConfigPath string
ExitCode int
Duration time.Duration
InvokedBinary string
Format string
Title string
Metadata map[string]any
}

View File

@@ -4,14 +4,18 @@ import (
"context"
"encoding/json"
"fmt"
"os"
"strconv"
"strings"
"time"
"unicode/utf8"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
)
// MaxOutputFileBytes bounds each Seriatim JSON or rendered-text result.
const MaxOutputFileBytes int64 = 64 * 1024 * 1024
// EnvConfig defines optional Seriatim environment tuning values.
type EnvConfig struct {
OverlapWordRunGap *float64
@@ -128,12 +132,13 @@ func (r *SubprocessRunner) Run(ctx context.Context, req MergeRequest) (MergeResu
}
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
Executable: r.binary,
Args: args,
Timeout: r.timeout,
EnvOverrides: env,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
Executable: r.binary,
Args: args,
Timeout: r.timeout,
EnvOverrides: env,
DiagnosticOwner: "seriatim",
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
})
if err != nil {
return MergeResult{
@@ -231,11 +236,12 @@ func (r *SubprocessRunner) Trim(ctx context.Context, req TrimRequest) (TrimResul
}
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
Executable: binary,
Args: args,
Timeout: timeout,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
Executable: binary,
Args: args,
Timeout: timeout,
DiagnosticOwner: "seriatim",
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
})
if err != nil {
return TrimResult{
@@ -319,11 +325,12 @@ func (r *SubprocessRunner) Normalize(ctx context.Context, req NormalizeRequest)
}
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
Executable: binary,
Args: args,
Timeout: timeout,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
Executable: binary,
Args: args,
Timeout: timeout,
DiagnosticOwner: "seriatim",
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
})
if err != nil {
return NormalizeResult{
@@ -384,6 +391,97 @@ func (r *SubprocessRunner) Normalize(ctx context.Context, req NormalizeRequest)
}, nil
}
// Render executes Seriatim render with deterministic flags and validates non-empty text output.
func (r *SubprocessRunner) Render(ctx context.Context, req RenderRequest) (RenderResult, error) {
if r == nil {
return RenderResult{}, fmt.Errorf("seriatim subprocess runner is nil")
}
if strings.TrimSpace(req.InputTranscriptPath) == "" {
return RenderResult{}, fmt.Errorf("seriatim render input path is required")
}
if strings.TrimSpace(req.OutputRenderedPath) == "" {
return RenderResult{}, fmt.Errorf("seriatim render output path is required")
}
format := strings.TrimSpace(req.Format)
if format == "" {
format = "markdown"
}
if format != "markdown" {
return RenderResult{}, fmt.Errorf("seriatim render format %q is unsupported", req.Format)
}
binary := r.binary
if strings.TrimSpace(req.Binary) != "" {
binary = strings.TrimSpace(req.Binary)
}
timeout := r.timeout
if req.Timeout < 0 {
return RenderResult{}, fmt.Errorf("seriatim render timeout must be >= 0")
}
if req.Timeout > 0 {
timeout = req.Timeout
}
args := buildRenderArgs(req, format)
if req.GeneratedConfigPath != "" {
if err := writeRenderInvocationConfig(req, args, binary, timeout, format); err != nil {
return RenderResult{}, fmt.Errorf("write seriatim render invocation config %q: %w", req.GeneratedConfigPath, err)
}
}
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
Executable: binary,
Args: args,
Timeout: timeout,
DiagnosticOwner: "seriatim",
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
})
if err != nil {
return RenderResult{
OutputRenderedPath: req.OutputRenderedPath,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
GeneratedConfigPath: req.GeneratedConfigPath,
ExitCode: runRes.ExitCode,
Duration: runRes.Duration,
InvokedBinary: binary,
Format: format,
Title: req.Title,
}, fmt.Errorf("run seriatim render (binary=%q): %w", binary, err)
}
if err := validateNonEmptyTextFile(req.OutputRenderedPath); err != nil {
return RenderResult{
OutputRenderedPath: req.OutputRenderedPath,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
GeneratedConfigPath: req.GeneratedConfigPath,
ExitCode: runRes.ExitCode,
Duration: runRes.Duration,
InvokedBinary: binary,
Format: format,
Title: req.Title,
}, fmt.Errorf("validate seriatim rendered output %q: %w", req.OutputRenderedPath, err)
}
return RenderResult{
OutputRenderedPath: req.OutputRenderedPath,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
GeneratedConfigPath: req.GeneratedConfigPath,
ExitCode: runRes.ExitCode,
Duration: runRes.Duration,
InvokedBinary: binary,
Format: format,
Title: req.Title,
Metadata: map[string]any{
"adapter": "seriatim_subprocess",
},
}, nil
}
func (r *SubprocessRunner) buildMergeArgs(req MergeRequest) []string {
args := []string{"merge"}
@@ -455,7 +553,7 @@ func (r *SubprocessRunner) writeMergeInvocationConfig(req MergeRequest, args []s
payload["coalesce_gap"] = *r.coalesceGap
}
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
}
func buildTrimArgs(req TrimRequest) []string {
@@ -480,6 +578,22 @@ func buildNormalizeArgs(req NormalizeRequest, outputSchema string) []string {
return args
}
func buildRenderArgs(req RenderRequest, format string) []string {
args := []string{
"render",
"--input-file", req.InputTranscriptPath,
"--output-file", req.OutputRenderedPath,
"--format", format,
"--include-timestamps=" + strconv.FormatBool(req.IncludeTimestamps),
"--include-segment-ids=" + strconv.FormatBool(req.IncludeSegmentIDs),
"--include-metadata=" + strconv.FormatBool(req.IncludeMetadata),
}
if strings.TrimSpace(req.Title) != "" {
args = append(args, "--title", req.Title)
}
return args
}
func writeTrimInvocationConfig(req TrimRequest, args []string, binary string, timeout time.Duration) error {
payload := map[string]any{
"schema": "seriatim.generated.v1",
@@ -491,7 +605,7 @@ func writeTrimInvocationConfig(req TrimRequest, args []string, binary string, ti
"output_path": req.OutputTrimmedPath,
"keep_selector": req.KeepSelector,
}
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
}
func writeNormalizeInvocationConfig(req NormalizeRequest, args []string, binary string, timeout time.Duration, outputSchema string) error {
@@ -506,13 +620,31 @@ func writeNormalizeInvocationConfig(req NormalizeRequest, args []string, binary
"output_schema": outputSchema,
"report_path": req.ReportPath,
}
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
}
func writeRenderInvocationConfig(req RenderRequest, args []string, binary string, timeout time.Duration, format string) error {
payload := map[string]any{
"schema": "seriatim.generated.v1",
"command": "render",
"binary": binary,
"args": args,
"timeout": timeout.String(),
"input_path": req.InputTranscriptPath,
"output_path": req.OutputRenderedPath,
"format": format,
"title": req.Title,
"include_timestamps": req.IncludeTimestamps,
"include_segment_ids": req.IncludeSegmentIDs,
"include_metadata": req.IncludeMetadata,
}
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
}
func validateJSONFile(path string) error {
data, err := os.ReadFile(path)
data, err := readSeriatimResult(path, "JSON output")
if err != nil {
return fmt.Errorf("read file: %w", err)
return err
}
var v any
if err := json.Unmarshal(data, &v); err != nil {
@@ -522,9 +654,9 @@ func validateJSONFile(path string) error {
}
func validateJSONFileWithSegments(path string) error {
data, err := os.ReadFile(path)
data, err := readSeriatimResult(path, "transcript JSON output")
if err != nil {
return fmt.Errorf("read file: %w", err)
return err
}
var payload map[string]any
@@ -541,3 +673,28 @@ func validateJSONFileWithSegments(path string) error {
}
return nil
}
func validateNonEmptyTextFile(path string) error {
data, err := readSeriatimResult(path, "rendered text output")
if err != nil {
return err
}
if len(data) == 0 {
return fmt.Errorf("file is empty")
}
if !utf8.Valid(data) {
return fmt.Errorf("file is not valid utf-8 text")
}
if strings.TrimSpace(string(data)) == "" {
return fmt.Errorf("file has no non-whitespace content")
}
return nil
}
func readSeriatimResult(path, category string) ([]byte, error) {
data, err := fileops.ReadRegularFile(path, MaxOutputFileBytes)
if err != nil {
return nil, fmt.Errorf("seriatim %s exceeds or cannot be read within %d-byte limit: %w", category, MaxOutputFileBytes, err)
}
return data, nil
}

View File

@@ -50,7 +50,7 @@ func TestSubprocessRunnerSuccessWithReportArgsAndEnv(t *testing.T) {
req := MergeRequest{
GeneratedConfigPath: filepath.Join(dir, "seriatim.generated.yml"),
InputTranscriptPaths: []string{filepath.Join(dir, "a.json"), filepath.Join(dir, "b.json")},
OutputMergedTranscriptPath: filepath.Join(dir, "merged.json"),
OutputMergedTranscriptPath: filepath.Join(dir, "base.json"),
ReportPath: filepath.Join(dir, "seriatim.report.json"),
SpeakersPath: filepath.Join(dir, "speakers.yml"),
AutocorrectPath: filepath.Join(dir, "autocorrect.yml"),
@@ -569,6 +569,156 @@ func TestSubprocessRunnerNormalizeInvalidReportJSONFails(t *testing.T) {
}
}
func TestSubprocessRunnerRenderSuccessInvocationAndProvenance(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
t.Setenv("SERIATIM_HELPER_MODE", "render_success")
recordPath := filepath.Join(t.TempDir(), "record.json")
t.Setenv("SERIATIM_HELPER_RECORD_PATH", recordPath)
wrapper := writeHelperWrapper(t)
runner := mustRunner(t, wrapper, false)
req := renderReqForTest(t)
res, err := runner.Render(context.Background(), req)
if err != nil {
t.Fatalf("Render() error = %v", err)
}
if res.OutputRenderedPath != req.OutputRenderedPath {
t.Fatalf("OutputRenderedPath = %q, want %q", res.OutputRenderedPath, req.OutputRenderedPath)
}
if res.Format != req.Format {
t.Fatalf("Format = %q, want %q", res.Format, req.Format)
}
if res.Title != req.Title {
t.Fatalf("Title = %q, want %q", res.Title, req.Title)
}
if res.InvokedBinary != wrapper {
t.Fatalf("InvokedBinary = %q, want %q", res.InvokedBinary, wrapper)
}
if res.ExitCode != 0 {
t.Fatalf("ExitCode = %d, want 0", res.ExitCode)
}
if res.Duration <= 0 {
t.Fatalf("Duration = %s, want >0", res.Duration)
}
if res.Metadata == nil || res.Metadata["adapter"] != "seriatim_subprocess" {
t.Fatalf("Metadata = %#v, want adapter marker", res.Metadata)
}
if _, err := os.Stat(req.OutputRenderedPath); err != nil {
t.Fatalf("rendered output missing: %v", err)
}
if _, err := os.Stat(req.StdoutLogPath); err != nil {
t.Fatalf("stdout log missing: %v", err)
}
if _, err := os.Stat(req.StderrLogPath); err != nil {
t.Fatalf("stderr log missing: %v", err)
}
if _, err := os.Stat(req.GeneratedConfigPath); err != nil {
t.Fatalf("generated config missing: %v", err)
}
rec := readHelperRecord(t, recordPath)
wantArgs := []string{
"render",
"--input-file", req.InputTranscriptPath,
"--output-file", req.OutputRenderedPath,
"--format", req.Format,
"--include-timestamps=true",
"--include-segment-ids=true",
"--include-metadata=false",
"--title", req.Title,
}
if strings.Join(rec.Args, "\n") != strings.Join(wantArgs, "\n") {
t.Fatalf("args = %#v, want %#v", rec.Args, wantArgs)
}
}
func TestSubprocessRunnerRenderWithoutTitleOmitsTitleArg(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
t.Setenv("SERIATIM_HELPER_MODE", "render_success")
recordPath := filepath.Join(t.TempDir(), "record.json")
t.Setenv("SERIATIM_HELPER_RECORD_PATH", recordPath)
runner := mustRunner(t, writeHelperWrapper(t), false)
req := renderReqForTest(t)
req.Title = ""
if _, err := runner.Render(context.Background(), req); err != nil {
t.Fatalf("Render() error = %v", err)
}
rec := readHelperRecord(t, recordPath)
for i := 0; i < len(rec.Args); i++ {
if rec.Args[i] == "--title" {
t.Fatalf("args = %#v, did not expect --title", rec.Args)
}
}
}
func TestSubprocessRunnerRenderSubprocessFailure(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
t.Setenv("SERIATIM_HELPER_MODE", "fail")
t.Setenv("SERIATIM_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
runner := mustRunner(t, writeHelperWrapper(t), false)
req := renderReqForTest(t)
_, err := runner.Render(context.Background(), req)
if err == nil {
t.Fatal("Render() error = nil, want non-nil")
}
if !strings.Contains(err.Error(), "run seriatim render") {
t.Fatalf("error = %q, want subprocess context", err.Error())
}
}
func TestSubprocessRunnerRenderMissingOutputFails(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
t.Setenv("SERIATIM_HELPER_MODE", "missing_output")
t.Setenv("SERIATIM_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
runner := mustRunner(t, writeHelperWrapper(t), false)
req := renderReqForTest(t)
_, err := runner.Render(context.Background(), req)
if err == nil {
t.Fatal("Render() error = nil, want non-nil")
}
if !strings.Contains(err.Error(), "validate seriatim rendered output") {
t.Fatalf("error = %q, want output validation context", err.Error())
}
}
func TestSubprocessRunnerRenderEmptyOutputFails(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
t.Setenv("SERIATIM_HELPER_MODE", "render_empty_output")
t.Setenv("SERIATIM_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
runner := mustRunner(t, writeHelperWrapper(t), false)
req := renderReqForTest(t)
_, err := runner.Render(context.Background(), req)
if err == nil {
t.Fatal("Render() error = nil, want non-nil")
}
if !strings.Contains(err.Error(), "file is empty") {
t.Fatalf("error = %q, want empty-file validation", err.Error())
}
}
func TestSubprocessRunnerConstructorValidation(t *testing.T) {
_, err := NewSubprocessRunnerFromConfigValues("", "10m", "seriatim-intermediate", nil, true, EnvConfig{})
if err == nil {
@@ -702,6 +852,14 @@ func TestSeriatimSubprocessHelper(t *testing.T) {
case "normalize_report_missing":
writeSeriatimHelperFile(outputPath, `{"schema":"seriatim.intermediate.v1","segments":[]}`)
os.Exit(0)
case "render_success":
writeSeriatimHelperFile(outputPath, "# Rendered transcript\n\nHello.\n")
_, _ = os.Stdout.WriteString("seriatim helper render stdout\n")
_, _ = os.Stderr.WriteString("seriatim helper render stderr\n")
os.Exit(0)
case "render_empty_output":
writeSeriatimHelperFile(outputPath, "")
os.Exit(0)
default:
_, _ = os.Stderr.WriteString(fmt.Sprintf("unknown helper mode %q\n", mode))
os.Exit(2)
@@ -732,7 +890,7 @@ func mergeReqForTest(t *testing.T, withReport bool) MergeRequest {
req := MergeRequest{
GeneratedConfigPath: filepath.Join(dir, "seriatim.generated.yml"),
InputTranscriptPaths: []string{in1, in2},
OutputMergedTranscriptPath: filepath.Join(dir, "merged.json"),
OutputMergedTranscriptPath: filepath.Join(dir, "base.json"),
StdoutLogPath: filepath.Join(dir, "seriatim.stdout.log"),
StderrLogPath: filepath.Join(dir, "seriatim.stderr.log"),
}
@@ -745,11 +903,11 @@ func mergeReqForTest(t *testing.T, withReport bool) MergeRequest {
func trimReqForTest(t *testing.T) TrimRequest {
t.Helper()
dir := t.TempDir()
input := filepath.Join(dir, "processed.json")
input := filepath.Join(dir, "polished.json")
writeSeriatimFile(t, input, `{"schema":"seriatim.intermediate.v1","segments":[]}`)
return TrimRequest{
InputTranscriptPath: input,
OutputTrimmedPath: filepath.Join(dir, "trimmed.json"),
OutputTrimmedPath: filepath.Join(dir, "final.trimmed.json"),
KeepSelector: "5-12",
GeneratedConfigPath: filepath.Join(dir, "seriatim.trim.generated.yml"),
StdoutLogPath: filepath.Join(dir, "seriatim.trim.stdout.log"),
@@ -760,12 +918,12 @@ func trimReqForTest(t *testing.T) TrimRequest {
func normalizeReqForTest(t *testing.T, withReport bool) NormalizeRequest {
t.Helper()
dir := t.TempDir()
input := filepath.Join(dir, "processed.json")
input := filepath.Join(dir, "polished.json")
writeSeriatimFile(t, input, `{"schema":"audita.processed.v1","segments":[]}`)
req := NormalizeRequest{
InputTranscriptPath: input,
OutputNormalizedPath: filepath.Join(dir, "normalized.json"),
OutputNormalizedPath: filepath.Join(dir, "final.json"),
OutputSchema: "seriatim-intermediate",
GeneratedConfigPath: filepath.Join(dir, "seriatim.normalize.generated.yml"),
StdoutLogPath: filepath.Join(dir, "seriatim.normalize.stdout.log"),
@@ -777,6 +935,25 @@ func normalizeReqForTest(t *testing.T, withReport bool) NormalizeRequest {
return req
}
func renderReqForTest(t *testing.T) RenderRequest {
t.Helper()
dir := t.TempDir()
input := filepath.Join(dir, "final.trimmed.json")
writeSeriatimFile(t, input, `{"schema":"seriatim.intermediate.v1","segments":[]}`)
return RenderRequest{
InputTranscriptPath: input,
OutputRenderedPath: filepath.Join(dir, "final.trimmed.md"),
Format: "markdown",
Title: "Session 42",
IncludeTimestamps: true,
IncludeSegmentIDs: true,
IncludeMetadata: false,
GeneratedConfigPath: filepath.Join(dir, "seriatim.render.generated.yml"),
StdoutLogPath: filepath.Join(dir, "seriatim.render.stdout.log"),
StderrLogPath: filepath.Join(dir, "seriatim.render.stderr.log"),
}
}
func mustRunner(t *testing.T, binary string, report bool) *SubprocessRunner {
t.Helper()
coalesce := 3.0

View File

@@ -1,31 +0,0 @@
// Package storage declares archive/storage backend adapter boundaries.
package storage
import "context"
// TODO: implement remote storage/archive backends (S3/SFTP/etc.).
// Backend is the adapter boundary for archive/storage operations.
type Backend interface {
Archive(ctx context.Context, req ArchiveRequest) (ArchiveResult, error)
}
// ArchiveItem describes one item to archive.
type ArchiveItem struct {
Kind string
LocalPath string
RemoteKey string
}
// ArchiveRequest describes one archive operation.
type ArchiveRequest struct {
SessionID string
ManifestPath string
Items []ArchiveItem
}
// ArchiveResult describes archive operation output.
type ArchiveResult struct {
Archived []ArchiveItem
Metadata map[string]any
}

View File

@@ -0,0 +1,90 @@
package storage
import (
"context"
"errors"
"fmt"
"io"
"math"
"strings"
)
// ReadLimitError reports that a remote object exceeded its caller-owned read
// limit. The limit is enforced against both available object metadata and the
// bytes returned by the opened object body.
type ReadLimitError struct {
Key string
Limit int64
Observed int64
}
func (e *ReadLimitError) Error() string {
return fmt.Sprintf("object %q exceeds %d-byte read limit (observed at least %d bytes)", e.Key, e.Limit, e.Observed)
}
// ReadObjectBounded opens one object version and retains at most maxBytes of
// its content. Object metadata may reject an oversized body early, but a
// limit-plus-one read always enforces the boundary when transfer begins.
func ReadObjectBounded(ctx context.Context, store ObjectStore, key string, maxBytes int64) (info ObjectInfo, data []byte, err error) {
key = strings.TrimSpace(key)
if store == nil {
return ObjectInfo{}, nil, fmt.Errorf("read bounded object: store is required")
}
if key == "" {
return ObjectInfo{}, nil, fmt.Errorf("read bounded object: key is required")
}
if maxBytes <= 0 || maxBytes == math.MaxInt64 {
return ObjectInfo{}, nil, fmt.Errorf("read bounded object %q: limit must be between 1 and %d bytes", key, int64(math.MaxInt64-1))
}
if err := ctx.Err(); err != nil {
return ObjectInfo{}, nil, err
}
info, body, err := store.Read(ctx, key)
if err != nil {
return ObjectInfo{}, nil, err
}
if body == nil {
return ObjectInfo{}, nil, fmt.Errorf("read bounded object %q: store returned no body", key)
}
defer func() {
if closeErr := body.Close(); closeErr != nil {
data = nil
err = errors.Join(err, fmt.Errorf("close object %q: %w", key, closeErr))
}
}()
if info.Size > maxBytes {
return info, nil, &ReadLimitError{Key: key, Limit: maxBytes, Observed: info.Size}
}
data, err = io.ReadAll(io.LimitReader(contextReader{ctx: ctx, reader: body}, maxBytes+1))
if err != nil {
return info, nil, err
}
if err := ctx.Err(); err != nil {
return info, nil, err
}
if int64(len(data)) > maxBytes {
return info, nil, &ReadLimitError{Key: key, Limit: maxBytes, Observed: int64(len(data))}
}
return info, data, nil
}
type contextReader struct {
ctx context.Context
reader io.Reader
}
func (r contextReader) Read(p []byte) (int, error) {
if err := r.ctx.Err(); err != nil {
return 0, err
}
n, err := r.reader.Read(p)
if err == nil {
if contextErr := r.ctx.Err(); contextErr != nil {
return n, contextErr
}
}
return n, err
}

View File

@@ -0,0 +1,135 @@
package storage
import (
"bytes"
"context"
"errors"
"io"
"testing"
)
func TestReadObjectBoundedAcceptsExactLimitWithAbsentSizeMetadata(t *testing.T) {
body := &trackingReadCloser{reader: bytes.NewReader([]byte("12345678")), chunkSize: 2}
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
return ObjectInfo{Key: "control.json", ETag: "generation"}, body, nil
}}
info, data, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
if err != nil {
t.Fatalf("ReadObjectBounded() error = %v", err)
}
if string(data) != "12345678" || info.ETag != "generation" {
t.Fatalf("ReadObjectBounded() = (%#v, %q), want opened object metadata and bytes", info, data)
}
if !body.closed {
t.Fatal("object body was not closed")
}
}
func TestReadObjectBoundedRejectsLimitPlusOneDespiteMissingOrInaccurateMetadata(t *testing.T) {
tests := []struct {
name string
metadataSize int64
}{
{name: "missing", metadataSize: 0},
{name: "inaccurate", metadataSize: 2},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
body := &trackingReadCloser{reader: bytes.NewReader([]byte("123456789")), chunkSize: 1}
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
return ObjectInfo{Key: "control.json", Size: test.metadataSize}, body, nil
}}
_, data, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
var limitErr *ReadLimitError
if !errors.As(err, &limitErr) {
t.Fatalf("ReadObjectBounded() error = %v, want ReadLimitError", err)
}
if data != nil || body.bytesRead != 9 || !body.closed {
t.Fatalf("data=%q bytes read=%d closed=%t, want nil, 9, true", data, body.bytesRead, body.closed)
}
})
}
}
func TestReadObjectBoundedRejectsOversizedMetadataBeforeTransfer(t *testing.T) {
body := &trackingReadCloser{reader: bytes.NewReader([]byte("small"))}
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
return ObjectInfo{Key: "control.json", Size: 9}, body, nil
}}
_, _, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
var limitErr *ReadLimitError
if !errors.As(err, &limitErr) {
t.Fatalf("ReadObjectBounded() error = %v, want ReadLimitError", err)
}
if body.bytesRead != 0 || !body.closed {
t.Fatalf("bytes read=%d closed=%t, want zero-byte transfer and closed body", body.bytesRead, body.closed)
}
}
func TestReadObjectBoundedPropagatesCancellationAndClosesBody(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
body := &trackingReadCloser{reader: bytes.NewReader([]byte("12345678")), chunkSize: 1, afterRead: cancel}
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
return ObjectInfo{Key: "control.json"}, body, nil
}}
_, data, err := ReadObjectBounded(ctx, store, "control.json", 8)
if !errors.Is(err, context.Canceled) {
t.Fatalf("ReadObjectBounded() error = %v, want context cancellation", err)
}
if data != nil || body.bytesRead != 1 || !body.closed {
t.Fatalf("data=%q bytes read=%d closed=%t, want nil, 1, true", data, body.bytesRead, body.closed)
}
}
func TestReadObjectBoundedReturnsCloseFailure(t *testing.T) {
closeErr := errors.New("close failed")
body := &trackingReadCloser{reader: bytes.NewReader([]byte("ok")), closeErr: closeErr}
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
return ObjectInfo{Key: "control.json", Size: 2}, body, nil
}}
_, data, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
if !errors.Is(err, closeErr) || data != nil || !body.closed {
t.Fatalf("data=%q error=%v closed=%t, want close failure and no retained data", data, err, body.closed)
}
}
type boundedReadStore struct {
ObjectStore
read func(context.Context, string) (ObjectInfo, io.ReadCloser, error)
}
func (s *boundedReadStore) Read(ctx context.Context, key string) (ObjectInfo, io.ReadCloser, error) {
return s.read(ctx, key)
}
type trackingReadCloser struct {
reader io.Reader
chunkSize int
afterRead func()
closeErr error
bytesRead int
closed bool
}
func (r *trackingReadCloser) Read(p []byte) (int, error) {
if r.chunkSize > 0 && len(p) > r.chunkSize {
p = p[:r.chunkSize]
}
n, err := r.reader.Read(p)
r.bytesRead += n
if n > 0 && r.afterRead != nil {
r.afterRead()
r.afterRead = nil
}
return n, err
}
func (r *trackingReadCloser) Close() error {
r.closed = true
return r.closeErr
}

View File

@@ -0,0 +1,23 @@
package storage
import (
"context"
"fmt"
"io"
)
// WriterDownloader is implemented by storage backends that stream an object
// into a caller-owned file handle.
type WriterDownloader interface {
DownloadTo(ctx context.Context, key string, destination io.Writer) error
}
// DownloadTo streams one object into destination. Destination-confined callers
// require this capability rather than granting a backend a mutable pathname.
func DownloadTo(ctx context.Context, store ObjectStore, key string, destination io.Writer) error {
writer, ok := store.(WriterDownloader)
if !ok {
return fmt.Errorf("object store does not support handle-confined downloads")
}
return writer.DownloadTo(ctx, key, destination)
}

View File

@@ -0,0 +1,28 @@
package storage
import (
"context"
"fmt"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
// NewObjectStoreFromConfig constructs a remote object store from resolved config.
func NewObjectStoreFromConfig(ctx context.Context, cfg *config.Config) (ObjectStore, error) {
if cfg == nil || cfg.Pipeline == nil {
return nil, fmt.Errorf("pipeline config is required")
}
switch strings.ToLower(strings.TrimSpace(cfg.Pipeline.Storage.Backend)) {
case config.StorageBackendS3:
if cfg.Pipeline.Storage.S3 == nil {
return nil, fmt.Errorf("pipeline.storage.s3 is required when pipeline.storage.backend is s3")
}
return NewS3BackendFromConfig(ctx, *cfg.Pipeline.Storage.S3)
case "", config.StorageBackendLocal:
return nil, fmt.Errorf("no remote object store backend is configured")
default:
return nil, fmt.Errorf("unsupported pipeline.storage.backend %q", cfg.Pipeline.Storage.Backend)
}
}

View File

@@ -0,0 +1,88 @@
package storage
import (
"context"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestNewObjectStoreFromConfigBuildsS3WhenBackendIsS3(t *testing.T) {
original := newS3Client
t.Cleanup(func() { newS3Client = original })
newS3Client = func(_ context.Context, _ s3ClientOptions) (s3API, error) {
return &fakeS3API{}, nil
}
store, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
Pipeline: &config.PipelineConfig{
Storage: config.StorageConfig{
Backend: "s3",
S3: &config.StorageS3Config{
Bucket: "my-archive",
},
},
},
})
if err != nil {
t.Fatalf("NewObjectStoreFromConfig() error = %v", err)
}
if _, ok := store.(*S3Backend); !ok {
t.Fatalf("store type = %T, want *S3Backend", store)
}
}
func TestNewObjectStoreFromConfigRequiresS3ConfigWhenBackendIsS3(t *testing.T) {
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
Pipeline: &config.PipelineConfig{
Storage: config.StorageConfig{Backend: "s3"},
},
})
if err == nil || !strings.Contains(err.Error(), "pipeline.storage.s3 is required") {
t.Fatalf("NewObjectStoreFromConfig() error = %v, want missing storage.s3 error", err)
}
}
func TestNewObjectStoreFromConfigNoRemoteBackendConfigured(t *testing.T) {
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
Pipeline: &config.PipelineConfig{
Storage: config.StorageConfig{Backend: "local"},
},
})
if err == nil || !strings.Contains(err.Error(), "no remote object store backend is configured") {
t.Fatalf("NewObjectStoreFromConfig() error = %v, want no-backend error", err)
}
}
func TestNewObjectStoreFromConfigDoesNotInferS3FromProviderFields(t *testing.T) {
called := false
original := newS3Client
t.Cleanup(func() { newS3Client = original })
newS3Client = func(_ context.Context, _ s3ClientOptions) (s3API, error) {
called = true
return &fakeS3API{}, nil
}
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
Pipeline: &config.PipelineConfig{Storage: config.StorageConfig{
Backend: config.StorageBackendLocal,
S3: &config.StorageS3Config{Bucket: "my-archive"},
}},
})
if err == nil || !strings.Contains(err.Error(), "no remote object store backend is configured") {
t.Fatalf("NewObjectStoreFromConfig() error = %v, want no-backend error", err)
}
if called {
t.Fatal("NewObjectStoreFromConfig() constructed S3 from incidental provider fields")
}
}
func TestNewObjectStoreFromConfigRejectsUnknownBackend(t *testing.T) {
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
Pipeline: &config.PipelineConfig{Storage: config.StorageConfig{Backend: "s33"}},
})
if err == nil || !strings.Contains(err.Error(), "unsupported pipeline.storage.backend") {
t.Fatalf("NewObjectStoreFromConfig() error = %v, want unsupported-backend error", err)
}
}

View File

@@ -1,40 +1,341 @@
package storage
import "context"
import (
"bytes"
"context"
"crypto/sha256"
"encoding/hex"
"fmt"
"io"
"os"
"path/filepath"
"sort"
"strings"
"sync"
"time"
)
// NoopBackend is a deterministic no-op archive/storage adapter.
type NoopBackend struct{}
// Archive returns the requested items as archived with placeholder metadata.
func (n *NoopBackend) Archive(ctx context.Context, req ArchiveRequest) (ArchiveResult, error) {
if err := ctx.Err(); err != nil {
return ArchiveResult{}, err
}
return ArchiveResult{Archived: append([]ArchiveItem(nil), req.Items...), Metadata: map[string]any{"placeholder": true}}, nil
}
// FakeBackend captures archive requests and returns deterministic responses.
// FakeBackend provides a deterministic in-memory object store for tests.
type FakeBackend struct {
Requests []ArchiveRequest
Err error
Result ArchiveResult
mu sync.RWMutex
Objects map[string]FakeObject
Uploads []FakeUploadCall
Downloads []FakeDownloadCall
Reads []FakeReadCall
ListErr error
DownloadErr error
UploadErr error
ExistsErr error
UploadHook func(FakeUploadCall) error
DownloadHook func(FakeDownloadCall) error
}
// Archive records request and returns configured response.
func (f *FakeBackend) Archive(ctx context.Context, req ArchiveRequest) (ArchiveResult, error) {
if err := ctx.Err(); err != nil {
return ArchiveResult{}, err
}
f.Requests = append(f.Requests, req)
if f.Err != nil {
return ArchiveResult{}, f.Err
}
res := f.Result
if res.Archived == nil {
res.Archived = append([]ArchiveItem(nil), req.Items...)
}
if res.Metadata == nil {
res.Metadata = map[string]any{"fake": true}
}
return res, nil
// FakeUploadCall captures one upload invocation in call order.
type FakeUploadCall struct {
LocalPath string
Key string
Options UploadOptions
}
// FakeDownloadCall captures one download invocation in call order.
type FakeDownloadCall struct {
Key string
LocalPath string
Bytes int64
}
// FakeReadCall captures one opened object in call order.
type FakeReadCall struct {
Key string
}
// FakeObject is a deterministic fake object-store record.
type FakeObject struct {
Key string
Data []byte
Metadata map[string]string
ETag string
LastModified *time.Time
}
// SeedObject inserts or replaces an object in the fake object store.
func (f *FakeBackend) SeedObject(obj FakeObject) {
f.mu.Lock()
defer f.mu.Unlock()
f.seedObject(obj)
}
func (f *FakeBackend) seedObject(obj FakeObject) {
if f.Objects == nil {
f.Objects = map[string]FakeObject{}
}
key := normalizeObjectKey(obj.Key)
obj.Key = key
obj.Data = append([]byte(nil), obj.Data...)
obj.Metadata = copyMetadata(obj.Metadata)
if obj.ETag == "" {
obj.ETag = fakeObjectETag(obj.Data)
}
f.Objects[key] = obj
}
// List returns deterministic prefix-filtered objects.
func (f *FakeBackend) List(ctx context.Context, prefix string) ([]ObjectInfo, error) {
if err := ctx.Err(); err != nil {
return nil, err
}
if f.ListErr != nil {
return nil, f.ListErr
}
f.mu.RLock()
defer f.mu.RUnlock()
normalizedPrefix := normalizeObjectKey(prefix)
keys := make([]string, 0, len(f.Objects))
for key := range f.Objects {
if strings.HasPrefix(key, normalizedPrefix) {
keys = append(keys, key)
}
}
sort.Strings(keys)
out := make([]ObjectInfo, 0, len(keys))
for _, key := range keys {
obj := f.Objects[key]
out = append(out, ObjectInfo{
Key: obj.Key,
Size: int64(len(obj.Data)),
ETag: obj.ETag,
LastModified: obj.LastModified,
})
}
return out, nil
}
// Read returns a stable object body and the generation observed with it.
func (f *FakeBackend) Read(ctx context.Context, key string) (ObjectInfo, io.ReadCloser, error) {
if err := ctx.Err(); err != nil {
return ObjectInfo{}, nil, err
}
if f.DownloadErr != nil {
return ObjectInfo{}, nil, f.DownloadErr
}
normalizedKey := normalizeObjectKey(key)
f.mu.RLock()
obj, ok := f.Objects[normalizedKey]
if ok {
obj.Data = append([]byte(nil), obj.Data...)
obj.Metadata = copyMetadata(obj.Metadata)
}
f.mu.RUnlock()
if !ok {
return ObjectInfo{}, nil, fmt.Errorf("read object %q: %w", normalizedKey, os.ErrNotExist)
}
f.mu.Lock()
f.Reads = append(f.Reads, FakeReadCall{Key: normalizedKey})
f.mu.Unlock()
return ObjectInfo{Key: obj.Key, Size: int64(len(obj.Data)), ETag: obj.ETag, LastModified: obj.LastModified}, io.NopCloser(bytes.NewReader(obj.Data)), nil
}
// DownloadTo writes one object to a caller-owned destination writer.
func (f *FakeBackend) DownloadTo(ctx context.Context, key string, destination io.Writer) error {
if err := ctx.Err(); err != nil {
return err
}
if f.DownloadErr != nil {
return f.DownloadErr
}
if destination == nil {
return fmt.Errorf("download object: destination writer is required")
}
if f.DownloadHook != nil {
if err := f.DownloadHook(FakeDownloadCall{Key: normalizeObjectKey(key)}); err != nil {
return err
}
}
_, source, err := f.Read(ctx, key)
if err != nil {
return err
}
defer source.Close()
count, err := io.Copy(destination, source)
if err != nil {
return fmt.Errorf("download object %q: write destination: %w", key, err)
}
f.mu.Lock()
f.Downloads = append(f.Downloads, FakeDownloadCall{Key: normalizeObjectKey(key), Bytes: count})
f.mu.Unlock()
return nil
}
// Download writes one object to a local path.
func (f *FakeBackend) Download(ctx context.Context, key, localPath string) error {
if strings.TrimSpace(localPath) == "" {
return fmt.Errorf("download object: local path is required")
}
if err := os.MkdirAll(filepath.Dir(localPath), 0o755); err != nil {
return fmt.Errorf("download object %q: create parent directory: %w", key, err)
}
destination, err := os.Create(localPath)
if err != nil {
return fmt.Errorf("download object %q: create local file: %w", key, err)
}
defer destination.Close()
return f.DownloadTo(ctx, key, destination)
}
// Upload reads a local file and stores it under key.
func (f *FakeBackend) Upload(ctx context.Context, localPath, key string, opts UploadOptions) (ObjectInfo, error) {
if err := ctx.Err(); err != nil {
return ObjectInfo{}, err
}
if f.UploadErr != nil {
return ObjectInfo{}, f.UploadErr
}
if strings.TrimSpace(localPath) == "" {
return ObjectInfo{}, fmt.Errorf("upload object: local path is required")
}
if strings.TrimSpace(key) == "" {
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
}
file, err := os.Open(localPath)
if err != nil {
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", key, localPath, err)
}
defer file.Close()
return f.uploadReader(ctx, file, key, opts, localPath)
}
// UploadReader stores content provided by a caller-owned reader.
func (f *FakeBackend) UploadReader(ctx context.Context, source io.Reader, key string, opts UploadOptions) (ObjectInfo, error) {
return f.uploadReader(ctx, source, key, opts, "reader")
}
func (f *FakeBackend) uploadReader(ctx context.Context, source io.Reader, key string, opts UploadOptions, localPath string) (ObjectInfo, error) {
if err := ctx.Err(); err != nil {
return ObjectInfo{}, err
}
if f.UploadErr != nil {
return ObjectInfo{}, f.UploadErr
}
if source == nil {
return ObjectInfo{}, fmt.Errorf("upload object: source is required")
}
if strings.TrimSpace(key) == "" {
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
}
data, err := io.ReadAll(source)
if err != nil {
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", key, localPath, err)
}
normalizedKey := normalizeObjectKey(key)
call := FakeUploadCall{
LocalPath: localPath,
Key: normalizedKey,
Options: UploadOptions{
Metadata: copyMetadata(opts.Metadata),
ContentType: opts.ContentType,
},
}
f.mu.Lock()
f.Uploads = append(f.Uploads, call)
f.mu.Unlock()
if f.UploadHook != nil {
if err := f.UploadHook(call); err != nil {
return ObjectInfo{}, err
}
}
return f.storeUploadedObject(normalizedKey, data, opts), nil
}
// UploadConditional atomically checks and replaces one mutable object.
func (f *FakeBackend) UploadConditional(ctx context.Context, source io.Reader, key string, opts UploadOptions, condition WriteCondition) (ObjectInfo, error) {
if err := ctx.Err(); err != nil {
return ObjectInfo{}, err
}
if err := validateWriteCondition(condition); err != nil {
return ObjectInfo{}, err
}
if f.UploadErr != nil {
return ObjectInfo{}, f.UploadErr
}
if source == nil {
return ObjectInfo{}, fmt.Errorf("upload object: source is required")
}
normalizedKey := normalizeObjectKey(key)
if normalizedKey == "" {
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
}
data, err := io.ReadAll(source)
if err != nil {
return ObjectInfo{}, fmt.Errorf("upload object %q: %w", normalizedKey, err)
}
call := FakeUploadCall{Key: normalizedKey, Options: UploadOptions{Metadata: copyMetadata(opts.Metadata), ContentType: opts.ContentType}}
f.mu.Lock()
f.Uploads = append(f.Uploads, call)
f.mu.Unlock()
if f.UploadHook != nil {
if err := f.UploadHook(call); err != nil {
return ObjectInfo{}, err
}
}
f.mu.Lock()
defer f.mu.Unlock()
existing, found := f.Objects[normalizedKey]
if condition.RequireAbsent && found {
return ObjectInfo{}, ErrConditionNotMet
}
if expected := strings.TrimSpace(condition.MatchETag); expected != "" && (!found || existing.ETag != expected) {
return ObjectInfo{}, ErrConditionNotMet
}
return f.storeUploadedObjectLocked(normalizedKey, data, opts), nil
}
func (f *FakeBackend) storeUploadedObject(key string, data []byte, opts UploadOptions) ObjectInfo {
f.mu.Lock()
defer f.mu.Unlock()
return f.storeUploadedObjectLocked(key, data, opts)
}
func (f *FakeBackend) storeUploadedObjectLocked(key string, data []byte, opts UploadOptions) ObjectInfo {
now := time.Now().UTC()
obj := FakeObject{Key: key, Data: append([]byte(nil), data...), Metadata: copyMetadata(opts.Metadata), LastModified: &now}
f.seedObject(obj)
return ObjectInfo{Key: key, Size: int64(len(data)), ETag: fakeObjectETag(data), LastModified: &now}
}
func fakeObjectETag(data []byte) string {
sum := sha256.Sum256(data)
return hex.EncodeToString(sum[:])
}
// Exists checks object presence.
func (f *FakeBackend) Exists(ctx context.Context, key string) (bool, error) {
if err := ctx.Err(); err != nil {
return false, err
}
if f.ExistsErr != nil {
return false, f.ExistsErr
}
f.mu.RLock()
_, ok := f.Objects[normalizeObjectKey(key)]
f.mu.RUnlock()
return ok, nil
}
func copyMetadata(in map[string]string) map[string]string {
if len(in) == 0 {
return nil
}
out := make(map[string]string, len(in))
for k, v := range in {
out[k] = v
}
return out
}

View File

@@ -3,29 +3,104 @@ package storage
import (
"context"
"errors"
"os"
"path/filepath"
"strings"
"testing"
)
func TestFakeBackendCapturesRequestAndReturnsItems(t *testing.T) {
func TestFakeBackendListPrefixFiltering(t *testing.T) {
fake := &FakeBackend{}
req := ArchiveRequest{SessionID: "s1", Items: []ArchiveItem{{Kind: "artifact", LocalPath: "artifacts/log.md"}}}
fake.SeedObject(FakeObject{Key: "dnd/campaigns/forsaken/audio/a.flac", Data: []byte("a")})
fake.SeedObject(FakeObject{Key: "dnd/campaigns/forsaken/audio/b.flac", Data: []byte("b")})
fake.SeedObject(FakeObject{Key: "dnd/campaigns/other/audio/c.flac", Data: []byte("c")})
res, err := fake.Archive(context.Background(), req)
items, err := fake.List(context.Background(), "dnd/campaigns/forsaken/audio/")
if err != nil {
t.Fatalf("Archive() error = %v", err)
t.Fatalf("List() error = %v", err)
}
if len(fake.Requests) != 1 || fake.Requests[0].SessionID != "s1" {
t.Fatalf("requests = %#v, want captured request", fake.Requests)
if len(items) != 2 {
t.Fatalf("List() len = %d, want 2", len(items))
}
if len(res.Archived) != 1 {
t.Fatalf("archived len = %d, want 1", len(res.Archived))
if items[0].Key != "dnd/campaigns/forsaken/audio/a.flac" || items[1].Key != "dnd/campaigns/forsaken/audio/b.flac" {
t.Fatalf("List() keys = %#v", items)
}
}
func TestFakeBackendError(t *testing.T) {
fake := &FakeBackend{Err: errors.New("boom")}
_, err := fake.Archive(context.Background(), ArchiveRequest{})
if err == nil {
t.Fatal("expected error, got nil")
func TestFakeBackendDownload(t *testing.T) {
fake := &FakeBackend{}
fake.SeedObject(FakeObject{Key: "audio/a.flac", Data: []byte("audio-a")})
dst := filepath.Join(t.TempDir(), "nested", "a.flac")
if err := fake.Download(context.Background(), `audio\a.flac`, dst); err != nil {
t.Fatalf("Download() error = %v", err)
}
data, err := os.ReadFile(dst)
if err != nil {
t.Fatalf("ReadFile() error = %v", err)
}
if string(data) != "audio-a" {
t.Fatalf("downloaded content = %q, want %q", string(data), "audio-a")
}
}
func TestFakeBackendUploadAndExists(t *testing.T) {
fake := &FakeBackend{}
local := filepath.Join(t.TempDir(), "upload.txt")
if err := os.WriteFile(local, []byte("payload"), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
info, err := fake.Upload(context.Background(), local, `runs\id\artifact.txt`, UploadOptions{
Metadata: map[string]string{"kind": "artifact"},
})
if err != nil {
t.Fatalf("Upload() error = %v", err)
}
if info.Key != "runs/id/artifact.txt" {
t.Fatalf("Upload() key = %q, want normalized key", info.Key)
}
ok, err := fake.Exists(context.Background(), "runs/id/artifact.txt")
if err != nil {
t.Fatalf("Exists() error = %v", err)
}
if !ok {
t.Fatal("Exists() = false, want true")
}
}
func TestFakeBackendConditionalUploadRejectsStaleGeneration(t *testing.T) {
fake := &FakeBackend{}
fake.SeedObject(FakeObject{Key: "locks.yml", Data: []byte("old")})
old := fake.Objects["locks.yml"].ETag
if _, err := fake.UploadConditional(context.Background(), strings.NewReader("new"), "locks.yml", UploadOptions{}, WriteCondition{MatchETag: old}); err != nil {
t.Fatalf("UploadConditional() error = %v", err)
}
if _, err := fake.UploadConditional(context.Background(), strings.NewReader("lost"), "locks.yml", UploadOptions{}, WriteCondition{MatchETag: old}); !errors.Is(err, ErrConditionNotMet) {
t.Fatalf("UploadConditional() error = %v, want ErrConditionNotMet", err)
}
if got := string(fake.Objects["locks.yml"].Data); got != "new" {
t.Fatalf("locks object = %q, want successful replacement preserved", got)
}
}
func TestFakeBackendObjectErrors(t *testing.T) {
fake := &FakeBackend{DownloadErr: errors.New("download fail"), UploadErr: errors.New("upload fail"), ListErr: errors.New("list fail"), ExistsErr: errors.New("exists fail")}
if _, err := fake.List(context.Background(), "x"); err == nil || !strings.Contains(err.Error(), "list fail") {
t.Fatalf("List() error = %v, want list fail", err)
}
if err := fake.Download(context.Background(), "x", filepath.Join(t.TempDir(), "x")); err == nil || !strings.Contains(err.Error(), "download fail") {
t.Fatalf("Download() error = %v, want download fail", err)
}
local := filepath.Join(t.TempDir(), "x.txt")
_ = os.WriteFile(local, []byte("x"), 0o644)
if _, err := fake.Upload(context.Background(), local, "x", UploadOptions{}); err == nil || !strings.Contains(err.Error(), "upload fail") {
t.Fatalf("Upload() error = %v, want upload fail", err)
}
if _, err := fake.Exists(context.Background(), "x"); err == nil || !strings.Contains(err.Error(), "exists fail") {
t.Fatalf("Exists() error = %v, want exists fail", err)
}
}

Some files were not shown because too many files have changed in this diff Show More