Compare commits
108 Commits
13de820931
...
v1.6.0
| Author | SHA1 | Date | |
|---|---|---|---|
| 545aa6893b | |||
| 98139f7e8b | |||
| f8fa0a2623 | |||
| 9af773491b | |||
| 2a656f0f11 | |||
| aec35a2d9b | |||
| 23dc4e2078 | |||
| c812fe3655 | |||
| 3da97ca50c | |||
| 4c203d8588 | |||
| fcb5f825e1 | |||
| 4c57ace2f6 | |||
| dde7f76ecb | |||
| a102db36af | |||
| b5b1d22011 | |||
| c91599ef36 | |||
| 257f10c9fb | |||
| fb9a4d14f4 | |||
| c4435b76c4 | |||
| 61000a9466 | |||
| 7e4ceb3d48 | |||
| 4e991fa21d | |||
| f86b17045d | |||
| f302488075 | |||
| 8c1171478d | |||
| 7ee637803d | |||
| 4d6086fefb | |||
| 82cb53e107 | |||
| b97b12da7f | |||
| 49ea747b17 | |||
| 3b5e9db41f | |||
| 8b4b328c4e | |||
| ee2b8e63e6 | |||
| 51edd384c0 | |||
| b804d0f2c8 | |||
| fd5ccc668b | |||
| 0dc8ff9b52 | |||
| 8657a28bdb | |||
| a9c5e4ad4e | |||
| 2176b4371d | |||
| effc10d75b | |||
| 5887839aa1 | |||
| 3128bef20a | |||
| 4e4e2b7d96 | |||
| 99f4f9a0db | |||
| 6abdd67bb5 | |||
| c32e0c401f | |||
| ab5751459a | |||
| 62de6abdbf | |||
| 903dc70682 | |||
| 23c714da66 | |||
| 6639775d7d | |||
| 966b95b176 | |||
| 3bcf2c08dd | |||
| 700ab655ca | |||
| 85c5647385 | |||
| 2ef7c76d99 | |||
| 9bc1b0feda | |||
| 5cec84a4a7 | |||
| abfbe42d61 | |||
| e7e3bef1e4 | |||
| a68e8e31a4 | |||
| 3a9e60cda9 | |||
| 905ff03ccc | |||
| 495f7bcde4 | |||
| 51e0e8c5d0 | |||
| a2409a1fd1 | |||
| b3363f87d6 | |||
| 5831c0c9e6 | |||
| 42ed81cbe1 | |||
| e433c86203 | |||
| a2a144dffa | |||
| 8ef6e99d69 | |||
| 2545faef6c | |||
| 80be8be4d6 | |||
| 801adb385d | |||
| feba7b9d74 | |||
| b89224bbde | |||
| 131ffd9887 | |||
| af492c9e97 | |||
| f39fc94610 | |||
| 4e4eff6ba7 | |||
| 8ff1b4fa66 | |||
| 702f622e18 | |||
| 9da2c1e144 | |||
| 32653f54f9 | |||
| 72a200968a | |||
| b39b68add7 | |||
| d9fa1d9328 | |||
| 8375ad83f3 | |||
| 4158394dcf | |||
| eac7e155a5 | |||
| 0cf2cbfeb3 | |||
| 361dbb4ca8 | |||
| d6deccf3e8 | |||
| ee747243fe | |||
| a1ceb457e9 | |||
| 9900211fa4 | |||
| 60cebf0e4b | |||
| 7bd575187e | |||
| ab5a7e8e3d | |||
| 99b2e1cd81 | |||
| 363313d99c | |||
| 18ddf00d3d | |||
| 59f3fe3d1d | |||
| 1dccf5f140 | |||
| 0b40cf8026 | |||
| a7ec195587 |
@@ -2,39 +2,26 @@ when:
|
||||
- event: tag
|
||||
|
||||
steps:
|
||||
- name: build-release-assets
|
||||
image: golang:1.25
|
||||
validate-release:
|
||||
image: golang:1.25.5
|
||||
commands:
|
||||
- ./scripts/check-release-candidate.sh "$CI_COMMIT_TAG"
|
||||
|
||||
build-release-assets:
|
||||
image: golang:1.25.5
|
||||
depends_on:
|
||||
- validate-release
|
||||
commands:
|
||||
- |
|
||||
set -eu
|
||||
case "$PWD" in
|
||||
/*) ;;
|
||||
*) echo "release workspace must have an absolute path" >&2; exit 1 ;;
|
||||
esac
|
||||
./scripts/build-release-assets.sh "$CI_COMMIT_TAG" "$PWD/dist"
|
||||
|
||||
version="$CI_COMMIT_TAG"
|
||||
dist="dist"
|
||||
pkg="gitea.maximumdirect.net/eric/narratio/cmd/narratio"
|
||||
|
||||
rm -rf "$dist"
|
||||
mkdir -p "$dist"
|
||||
|
||||
build_binary() {
|
||||
goos="$1"
|
||||
goarch="$2"
|
||||
suffix="$3"
|
||||
output="$dist/narratio-$version-$goos-$goarch$suffix"
|
||||
|
||||
CGO_ENABLED=0 GOOS="$goos" GOARCH="$goarch" \
|
||||
go build -trimpath -ldflags "-s -w -X gitea.maximumdirect.net/eric/narratio/internal/buildinfo.Version=$version" \
|
||||
-o "$output" "$pkg"
|
||||
}
|
||||
|
||||
build_binary linux amd64 ""
|
||||
build_binary linux arm64 ""
|
||||
build_binary darwin amd64 ""
|
||||
build_binary darwin arm64 ""
|
||||
build_binary windows amd64 ".exe"
|
||||
build_binary windows arm64 ".exe"
|
||||
|
||||
- name: publish-release
|
||||
image: woodpeckerci/plugin-release
|
||||
publish-release:
|
||||
image: woodpeckerci/plugin-release:0.3.1
|
||||
depends_on:
|
||||
- build-release-assets
|
||||
settings:
|
||||
@@ -42,6 +29,8 @@ steps:
|
||||
from_secret: GITEA_RELEASE_TOKEN
|
||||
files:
|
||||
- dist/narratio-*
|
||||
title: Narratio ${CI_COMMIT_TAG}
|
||||
note: docs/releases/${CI_COMMIT_TAG}.md
|
||||
checksum: sha256
|
||||
checksum-file: SHA256SUMS
|
||||
checksum-flatten: true
|
||||
|
||||
8
.woodpecker/shuffle.yml
Normal file
8
.woodpecker/shuffle.yml
Normal file
@@ -0,0 +1,8 @@
|
||||
when:
|
||||
- event: cron
|
||||
|
||||
steps:
|
||||
shuffled-race-tests:
|
||||
image: golang:1.25
|
||||
commands:
|
||||
- go test -race -shuffle=on -count=3 ./...
|
||||
47
.woodpecker/verify.yml
Normal file
47
.woodpecker/verify.yml
Normal file
@@ -0,0 +1,47 @@
|
||||
when:
|
||||
- event: [push, pull_request]
|
||||
|
||||
steps:
|
||||
tests:
|
||||
image: golang:1.25
|
||||
commands:
|
||||
- go test ./...
|
||||
|
||||
race-tests:
|
||||
image: golang:1.25
|
||||
depends_on: tests
|
||||
commands:
|
||||
- go test -race ./...
|
||||
|
||||
static-analysis:
|
||||
image: golang:1.25
|
||||
depends_on: tests
|
||||
commands:
|
||||
- go vet ./...
|
||||
|
||||
build:
|
||||
image: golang:1.25
|
||||
depends_on: tests
|
||||
commands:
|
||||
- go build ./...
|
||||
|
||||
documentation-and-examples:
|
||||
image: golang:1.25
|
||||
depends_on: tests
|
||||
commands:
|
||||
- go test ./internal/doccheck
|
||||
- go test ./internal/config -run '^TestExamplesLoadAndValidate$'
|
||||
|
||||
cross-build:
|
||||
image: golang:1.25
|
||||
depends_on: [race-tests, static-analysis, build, documentation-and-examples]
|
||||
commands:
|
||||
- |
|
||||
set -eu
|
||||
output_dir="$(mktemp -d)"
|
||||
trap 'rm -rf "$output_dir"' EXIT
|
||||
for target in linux/amd64 linux/arm64 darwin/amd64 darwin/arm64 windows/amd64 windows/arm64; do
|
||||
goos="${target%/*}"
|
||||
goarch="${target#*/}"
|
||||
CGO_ENABLED=0 GOOS="$goos" GOARCH="$goarch" go build -o "$output_dir/narratio-$goos-$goarch" ./cmd/narratio
|
||||
done
|
||||
@@ -27,7 +27,7 @@ This requires resolvable `pipeline.yml`, `campaign.yml`, and concrete
|
||||
- [Integration contracts](docs/integrations/) — external tools, formats, and
|
||||
compatibility expectations.
|
||||
- [Maintained examples](examples/README.md) — complete copyable configuration
|
||||
and input files.
|
||||
and input files, including the production/testing split bundle.
|
||||
|
||||
## Maintainer Documentation
|
||||
|
||||
|
||||
160
docs/cli.md
160
docs/cli.md
@@ -12,12 +12,15 @@ This runs the canonical full pipeline for session `2026-04-04`.
|
||||
|
||||
Top-level commands:
|
||||
|
||||
- `run <session_id>`: run full stage order.
|
||||
- `version`: print the Narratio build version.
|
||||
- `run <session_id>`: run all or one contiguous range of the canonical stage order.
|
||||
- `regenerate-artifacts <session_id>`: force-run extraction through analysis.
|
||||
- `run-stage <stage> <session_id>`: run one stage.
|
||||
- `analyze <session_id>`: force-run analyze.
|
||||
- `publish <session_id>`: force-run publish.
|
||||
- `clean <session_id>` or `clean --all`: remove local work/spool state.
|
||||
- `session <subcommand>`: session helper commands.
|
||||
- `config <subcommand>`: validate, display, source-trace, or compare resolved pipeline configuration.
|
||||
|
||||
Session subcommands:
|
||||
|
||||
@@ -41,13 +44,22 @@ Most session-aware commands accept:
|
||||
- `--session <session.yml>`
|
||||
- `--session-id <session_id>`
|
||||
- `--previous-session-id <session_id>`
|
||||
- `--profile <name>`
|
||||
|
||||
Rules:
|
||||
|
||||
- `--campaign` and `--campaign-file` are mutually exclusive.
|
||||
- `--session` is not used by `session init`.
|
||||
- if both positional `<session_id>` and `--session-id` are provided, values must match.
|
||||
- `--previous-session-id` is a strict expectation: the selected session file
|
||||
must contain the same `previous_session_id`.
|
||||
- `--profile` selects a declared pipeline profile. It may be supplied once;
|
||||
an explicit empty or unknown value fails configuration resolution. When it is
|
||||
omitted, a declared `default_profile` is used. The same selection applies to
|
||||
all common-flag commands, including `regenerate-artifacts`.
|
||||
- `clean --all` cannot be combined with campaign/session selectors.
|
||||
- notification delivery is currently limited to the configured `noop` mode; see
|
||||
the [configuration reference](./config.md#notifications).
|
||||
|
||||
## Session ID Input Rules
|
||||
|
||||
@@ -66,21 +78,119 @@ Commands with additional positionals keep their command-specific order:
|
||||
|
||||
## Command Reference
|
||||
|
||||
### `config validate`, `config show`, `config sources`, and `config diff`
|
||||
|
||||
```bash
|
||||
narratio config validate [--config <pipeline.yml>] [--campaign <id> | --campaign-file <campaign.yml>] [--profile <name>]
|
||||
narratio config show [--config <pipeline.yml>] [--campaign <id> | --campaign-file <campaign.yml>] [--profile <name>]
|
||||
narratio config sources [--config <pipeline.yml>] [--campaign <id> | --campaign-file <campaign.yml>] [--profile <name>]
|
||||
narratio config diff <left-profile> <right-profile> [--config <pipeline.yml>] [--campaign <id> | --campaign-file <campaign.yml>]
|
||||
```
|
||||
|
||||
These commands resolve the selected profile, defaults, ordinary paths, and—if
|
||||
a campaign is selected—the campaign-owned party. They neither discover or load
|
||||
a session nor create a workspace, manifest, run, lock, adapter, remote
|
||||
connection, or credential environment.
|
||||
|
||||
Campaign selection is optional for a pipeline without party-driven artifact
|
||||
families. A pipeline with `scriptorium.artifact_families` needs a selected or
|
||||
configured default campaign so Narratio can expand its concrete artifacts and
|
||||
publish rules. `--campaign` and `--campaign-file` remain mutually exclusive.
|
||||
Session, range, force, and artifact-execution flags are not accepted.
|
||||
|
||||
`config validate` writes a concise root-path, selected-profile (or `none`), and
|
||||
effective-digest summary after successful complete validation. `config show`
|
||||
writes one deterministic, secret-free YAML document containing defaulted and
|
||||
expanded concrete configuration. It omits composition declarations, artifact
|
||||
family declarations, and runtime provenance.
|
||||
|
||||
`config sources` reports the same fully validated resolution without printing
|
||||
effective values. Its header identifies the root, ordered imports, selected
|
||||
profile and overlay, selected campaign, party mode/source, and digest. The
|
||||
remaining tab-separated records are sorted as `path`, `role`, and `source`.
|
||||
Roles distinguish root, import, profile, centralized default, campaign, party,
|
||||
legacy-player, and generated family ownership. A generated party member has
|
||||
one family record and one party record at the same logical path. The output
|
||||
never reads or prints secret values.
|
||||
|
||||
`config diff` resolves both supplied profile names from one parsed root source
|
||||
set and compares their fully resolved, secret-free effective mappings. It does
|
||||
not accept `--profile`; the two positional names must be distinct, declared
|
||||
profiles. When party-driven families are present, both profiles must resolve to
|
||||
the same selected campaign and party. Use `--campaign-file` if profile-specific
|
||||
campaign configuration would otherwise select different files.
|
||||
|
||||
Equal profiles print `no differences`. Otherwise, sorted tab-separated records
|
||||
use one of these forms, with compact JSON values:
|
||||
|
||||
```text
|
||||
added <path> <right-value>
|
||||
removed <path> <left-value>
|
||||
changed <path> <left-value> <right-value>
|
||||
```
|
||||
|
||||
Mappings are flattened to their logical field paths; lists remain one atomic
|
||||
value. The command compares defaulted concrete artifacts and publish rules, not
|
||||
profile names, source-file layout, or formatting. It succeeds when differences
|
||||
are found, making it suitable for review and migration checks.
|
||||
|
||||
### `version`
|
||||
|
||||
```bash
|
||||
narratio version
|
||||
```
|
||||
|
||||
Official release binaries report their exact Git tag. Binaries built directly
|
||||
from source without release linker metadata report `dev`.
|
||||
|
||||
### `run`
|
||||
|
||||
```bash
|
||||
narratio run <session_id> [--force] [--artifacts <name[,name...]>] [...common config flags]
|
||||
narratio run <session_id> [--from <stage>] [--through <stage>] [--force] [--artifacts <name[,name...]>] [...common config flags]
|
||||
```
|
||||
|
||||
Behavior:
|
||||
|
||||
- evaluates full stage order;
|
||||
- runs `extract` between `trim` and `render`; an omitted or disabled Notarius
|
||||
- evaluates one inclusive contiguous range of the canonical stage order;
|
||||
- defaults an omitted `--from` to `prepare` and an omitted `--through` to
|
||||
`notify`, so omitting both retains full-pipeline behavior;
|
||||
- rejects unknown endpoints and a `--from` endpoint after `--through`;
|
||||
- runs `render` before `extract`; an omitted or disabled Notarius
|
||||
configuration records an explicit `notarius_disabled` self-skip;
|
||||
- skips already-succeeded stages unless `--force` is set or a stage-specific
|
||||
resume check finds its durable result obsolete;
|
||||
- applies `--force` only to stages in the selected range;
|
||||
- rejects repeated `--from`, `--through`, or `--force` options, including
|
||||
`--name=value` spellings;
|
||||
- continues interrupted or partially completed sessions by running non-succeeded stages;
|
||||
- writes session and run manifests.
|
||||
- reports the resolved profile (or `none`) and effective configuration digest.
|
||||
|
||||
When `--artifacts` is present, the selected range must contain `analyze` or
|
||||
`publish`. Either consumer is sufficient, including a one-stage range.
|
||||
|
||||
### `regenerate-artifacts`
|
||||
|
||||
```bash
|
||||
narratio regenerate-artifacts <session_id> [--artifacts <name[,name...]>] [...common config flags]
|
||||
```
|
||||
|
||||
Exactly equivalent to:
|
||||
|
||||
```bash
|
||||
narratio run <session_id> --force --from extract --through analyze [caller options]
|
||||
```
|
||||
|
||||
The command always reruns extraction. Analysis rebuilds the selected configured
|
||||
artifacts and any prerequisites required by those targets; without
|
||||
`--artifacts`, it uses the normal default analysis selection. Publish and notify
|
||||
never run. Common session/configuration options and repeatable artifact values
|
||||
pass through unchanged.
|
||||
|
||||
Because the expansion owns `--force`, `--from`, and `--through`, callers cannot
|
||||
supply those options. The shared `run` parser reports them as duplicate
|
||||
singleton flags. The alias has no private execution options or behavior, and
|
||||
runtime diagnostics may identify the operation as `run`.
|
||||
|
||||
### `run-stage`
|
||||
|
||||
@@ -96,8 +206,8 @@ Valid stage names:
|
||||
- `polish`
|
||||
- `normalize`
|
||||
- `trim`
|
||||
- `extract`
|
||||
- `render`
|
||||
- `extract`
|
||||
- `analyze`
|
||||
- `publish`
|
||||
- `notify`
|
||||
@@ -149,10 +259,19 @@ post-publish cleanup behavior.
|
||||
### `session plan`
|
||||
|
||||
```bash
|
||||
narratio session plan <session_id> [--force] [...common config flags]
|
||||
narratio session plan <session_id> [--from <stage>] [--through <stage>] [--force] [--artifacts <name[,name...]>] [...common config flags]
|
||||
```
|
||||
|
||||
Validates config, prepares local workdir layout, and prints run/skip decisions for each stage.
|
||||
Uses the same inclusive bounds, endpoint validation, force scope, and artifact
|
||||
selection contract as `run`. It validates config and prints run/skip decisions
|
||||
for selected stages only without creating the local workdir or changing the
|
||||
manifest. Resume-capable selected stages are checked against durable evidence.
|
||||
The output includes the resolved profile (or `none`) and effective configuration
|
||||
digest without writing provenance or any manifest state.
|
||||
For `analyze`, the preview also lists explicit targets, prerequisite-only work,
|
||||
execution order, and reusable current artifacts with concise reasons. These
|
||||
artifact decisions come from the same reconciliation and work planner used by
|
||||
execution; the preview does not predict output identities.
|
||||
|
||||
### `session validate`
|
||||
|
||||
@@ -168,7 +287,8 @@ Read-only preflight checks for config validity, required inputs, audio mode, pre
|
||||
narratio session status <session_id> [...common config flags]
|
||||
```
|
||||
|
||||
Prints local manifest state and, when storage is available, remote current-state and published-output status.
|
||||
Prints local manifest state and, when storage is available, status for the
|
||||
pointer-selected remote commit and its declared published outputs.
|
||||
|
||||
### `session init`
|
||||
|
||||
@@ -211,7 +331,8 @@ Behavior:
|
||||
- discovers committed remote current state;
|
||||
- plans local restores;
|
||||
- writes an execution report;
|
||||
- blocks conflicting overwrites unless `--force` is set.
|
||||
- blocks unresolved conflicts. `--force` permits replacement only of eligible
|
||||
regular files.
|
||||
|
||||
See [Operations: Restore Workflow](./operations.md#restore-workflow) for the
|
||||
default restore scope, report location, and conflict-handling workflow.
|
||||
@@ -246,14 +367,23 @@ and precedence.
|
||||
|
||||
## `--artifacts` Selection Rules
|
||||
|
||||
- accepted on `run`, `run-stage`, `analyze`, and `publish`;
|
||||
An artifact-family key selects all of its concrete character members. A
|
||||
concrete generated key selects only that member; mixed family and concrete
|
||||
selection is deduplicated and executed as concrete keys. The resulting plan
|
||||
and command output identify both the concrete key and, where applicable, its
|
||||
family and character ID.
|
||||
|
||||
- accepted on `run`, `session plan`, `run-stage`, `analyze`, and `publish`;
|
||||
- repeatable and comma-separated values are combined, surrounding whitespace
|
||||
is removed, and duplicate names are collapsed;
|
||||
- names must exist in `pipeline.scriptorium.artifacts`;
|
||||
- empty entries are invalid;
|
||||
- repeated names are deduplicated.
|
||||
- on `run-stage`, only `analyze` and `publish` accept the option.
|
||||
|
||||
Effects:
|
||||
|
||||
- filters analyze execution to selected configured artifacts;
|
||||
- selects explicit analyze targets; required configured prerequisites may be
|
||||
reused or rebuilt before them;
|
||||
- filters publish rules that source `narratio.artifact.<name>`;
|
||||
- does not filter built-in transcript/bounds or explicitly configured
|
||||
`narratio.extraction.<name>` publish sources; and
|
||||
@@ -285,6 +415,12 @@ Force publish only:
|
||||
narratio publish 2026-04-04
|
||||
```
|
||||
|
||||
Regenerate post-transcript artifacts without publishing:
|
||||
|
||||
```bash
|
||||
narratio regenerate-artifacts 2026-04-04 --artifacts session_recap,player_handout
|
||||
```
|
||||
|
||||
## Output And Exit Behavior
|
||||
|
||||
- Successful commands write their result or summary to standard output and
|
||||
|
||||
303
docs/config.md
303
docs/config.md
@@ -38,17 +38,161 @@ If local session discovery fails and a `session_id` is known, Narratio attempts
|
||||
|
||||
using configured object storage.
|
||||
|
||||
The downloaded remote session file is command-scoped: Narratio removes it after
|
||||
the command finishes and records only the remote object provenance alongside
|
||||
the durable copied session input.
|
||||
|
||||
### Read-only effective pipeline inspection
|
||||
|
||||
`narratio config validate`, `narratio config show`, and `narratio config
|
||||
sources` use the same `--config`, `--campaign`, `--campaign-file`, and
|
||||
`--profile` selection rules as pipeline commands, but do not select, discover,
|
||||
or load a session. They do not read credential values or create runtime state.
|
||||
`narratio config diff <left-profile> <right-profile>` uses the same pipeline and
|
||||
campaign selectors, resolves each named profile independently from one parsed
|
||||
root source set, and does not accept a separate `--profile` flag.
|
||||
|
||||
Campaign selection is optional only when the resolved pipeline has no
|
||||
`scriptorium.artifact_families`. When families are declared, Narratio selects a
|
||||
campaign through an explicit flag or `pipeline.campaigns.default_campaign_id`,
|
||||
then parses the campaign-owned party and expands concrete artifacts and any
|
||||
family publish rules before validation. `config validate` prints the resulting
|
||||
root, profile, and effective digest. `config show` emits the normalized
|
||||
effective pipeline YAML, with defaults and concrete expansion included but
|
||||
composition and family declarations omitted. `config sources` prints a stable
|
||||
source projection instead of effective values: root/import/profile/default
|
||||
ownership plus campaign/party and generated-family records. Canonical derived
|
||||
players trace to the party; a legacy configured players file is explicitly
|
||||
marked as a legacy player source. The [CLI reference](cli.md#config-validate-config-show-config-sources-and-config-diff)
|
||||
owns command syntax and output conventions.
|
||||
|
||||
`config diff` compares normalized field values rather than YAML text or source
|
||||
ownership. It emits sorted `added`, `removed`, and `changed` records, uses
|
||||
compact deterministic JSON values, treats lists atomically, and reports `no
|
||||
differences` when the complete effective configurations are equal. Concrete
|
||||
family members and generated publish rules participate after expansion; moving
|
||||
an equal value between eligible root/import sources does not create a
|
||||
difference.
|
||||
|
||||
### Migrating to the maintained bundle
|
||||
|
||||
Use the [production/testing bundle](../examples/production-testing/pipeline.yml)
|
||||
as the complete copyable migration reference. Split stable pipeline settings
|
||||
into explicit additive imports, place production/testing differences in one
|
||||
selected overlay, and retain a production `default_profile`. Convert campaign
|
||||
rosters to [canonical party input](integrations/party.md), remove a separate
|
||||
`players_file`, then express character work as families. Inspect the result
|
||||
with `config validate`, `config show`, and `config sources`; use `config diff`
|
||||
to review profiles before running a session. Unversioned parties and their
|
||||
`players_file` remain a clearly bounded legacy compatibility path.
|
||||
|
||||
### Identity segments
|
||||
|
||||
Campaign IDs (`campaign_id` and `default_campaign_id`), session IDs, previous
|
||||
session IDs, and Narratio run IDs are opaque portable segments. They must use
|
||||
only ASCII letters, digits, `.`, `_`, and `-`; empty values, `.`/`..`, path
|
||||
separators, drive forms, whitespace, control characters, and non-ASCII text are
|
||||
rejected. Narratio does not trim or rewrite these values. Existing manifests or
|
||||
remote state with an unsafe legacy identity must be migrated before use.
|
||||
|
||||
## Validation and Merge Rules
|
||||
|
||||
- YAML decode is strict (`KnownFields(true)`): unknown fields fail load.
|
||||
- YAML decode is strict (`KnownFields(true)`) and accepts exactly one document:
|
||||
unknown fields or trailing documents fail load.
|
||||
- A pipeline file may explicitly import additive YAML fragments through the
|
||||
root-only `composition.imports` list. Imported files contribute fields to one
|
||||
logical pipeline document; they do not override fields supplied by the root
|
||||
or another import.
|
||||
- A root pipeline may declare named profiles. Exactly one profile is selected
|
||||
by an option-aware caller or by `composition.default_profile`; a caller's
|
||||
explicit selection takes precedence. Declaring profiles without either form
|
||||
of selection is an error.
|
||||
- Configured timeout and retry-delay durations must be positive. An omitted
|
||||
artifact timeout continues to inherit its configured Scriptorium timeout.
|
||||
- Session files must be concrete; unresolved `{{ ... }}` placeholders fail load.
|
||||
- Pipeline defaults are applied before validation.
|
||||
- Campaign and session identities must agree.
|
||||
- Stable files (`speakers_file`, `autocorrect_file`, `glossary_file`, `players_file`, `party_file`) resolve from session overrides when provided, otherwise from campaign defaults.
|
||||
- Required stable files (`speakers_file`, `autocorrect_file`, `glossary_file`,
|
||||
`party_file`) and the optional `spell_catalog_file` resolve from session
|
||||
overrides when provided, otherwise from campaign defaults. An empty or
|
||||
omitted session spell-catalog value inherits the campaign value.
|
||||
- `party_file` is classified when pipeline and campaign configuration are
|
||||
combined. A versioned [canonical party](integrations/party.md) is
|
||||
campaign-owned, derives the players input internally, and forbids both a
|
||||
separate `players_file` and a session `party_file` override. An unversioned
|
||||
party remains a bounded legacy input and requires `players_file`; its normal
|
||||
campaign/session overrides continue to apply.
|
||||
- Exactly one audio mode must be configured in session input:
|
||||
- local (`audio_dir` or `audio_files`), or
|
||||
- S3 (`audio_s3.prefix`).
|
||||
|
||||
### Pipeline composition
|
||||
|
||||
Large pipeline configurations may be split into explicitly named fragments and
|
||||
may declare one overlay per selectable profile:
|
||||
|
||||
```yaml
|
||||
composition:
|
||||
imports:
|
||||
- config/storage.yml
|
||||
- config/integrations.yaml
|
||||
default_profile: production
|
||||
profiles:
|
||||
production:
|
||||
overlay: profiles/production.yml
|
||||
testing:
|
||||
overlay: profiles/testing.yml
|
||||
|
||||
campaigns:
|
||||
root: /usr/local/share/narratio/campaigns
|
||||
```
|
||||
|
||||
The maintained [production/testing bundle](../examples/production-testing/pipeline.yml)
|
||||
is a complete copyable example of this structure, including canonical-party
|
||||
artifact families.
|
||||
|
||||
Imports are resolved relative to the directory containing the root pipeline
|
||||
file and are loaded in declaration order. Narratio does not scan directories or
|
||||
infer fragments. Each import must be a confined regular `.yml` or `.yaml` file:
|
||||
absolute paths, traversal, symlinks, directories, duplicate files, and an
|
||||
import of the root pipeline itself are rejected. Only the root pipeline may
|
||||
contain `composition`; nested composition is rejected.
|
||||
|
||||
Composition is additive. A map may be extended by multiple files when every
|
||||
leaf is distinct, but a scalar, list, or map/list/scalar kind cannot be claimed
|
||||
more than once, even when the repeated values are identical. Conflict errors
|
||||
name the full field path and every source that claimed it. The assembled YAML is
|
||||
then decoded against the normal strict pipeline schema and defaults are applied
|
||||
once.
|
||||
|
||||
Profile names are case-sensitive, non-empty, trimmed, and cannot contain
|
||||
control characters. If `profiles` is present, it must contain at least one
|
||||
entry and every entry must contain only an `overlay` path. An explicit profile
|
||||
selection overrides `default_profile`; unknown and explicitly empty selections
|
||||
fail. Narratio never selects the first profile implicitly and does not read a
|
||||
profile selection from the environment.
|
||||
|
||||
Every declared overlay is resolved relative to the root pipeline directory and
|
||||
must satisfy the same confined regular-YAML-file rules as an import. Narratio
|
||||
parses every declared overlay even when it is not selected, then applies only
|
||||
the selected one. Maps merge recursively, overlay scalars replace base scalars,
|
||||
and overlay lists replace base lists completely. Explicit `false`, zero, empty
|
||||
lists, and empty maps remain meaningful. YAML null cannot delete a value, and
|
||||
kind changes are rejected. Profiles cannot inherit from or stack with other
|
||||
profiles, and overlays cannot import files or declare profiles.
|
||||
|
||||
After composition, Narratio strictly decodes the result, applies centralized
|
||||
defaults once, resolves ordinary paths, and computes a deterministic effective
|
||||
configuration digest. The digest represents the normalized, secret-free
|
||||
runtime pipeline mapping; it excludes composition declarations, source
|
||||
provenance, profile identity, and raw environment secret values. Equivalent
|
||||
effective mappings therefore have the same digest regardless of how fields are
|
||||
split among the root and imports.
|
||||
|
||||
An imported field has the same meaning it would have in a monolithic root
|
||||
pipeline. In particular, ordinary relative pipeline paths continue to resolve
|
||||
from the root pipeline directory, not from the importing fragment's directory.
|
||||
|
||||
## Minimal Working Configuration
|
||||
|
||||
`pipeline.yml`
|
||||
@@ -85,7 +229,13 @@ inputs:
|
||||
|
||||
- Do not place raw secrets in YAML.
|
||||
- Use env var names in config (for example `pipeline.audita.llm_api_key_env`).
|
||||
- Optionally load env files from `pipeline.secrets.env_dir`.
|
||||
- Optionally load credential files from `pipeline.secrets.env_dir`. Each valid
|
||||
environment-variable filename supplies one value; trailing CR/LF is removed.
|
||||
- An existing process environment value takes precedence over a credential file.
|
||||
- Credential directories and files must not be symlinks and must be regular,
|
||||
bounded files (at most 8 KiB per value). On POSIX, provision the directory
|
||||
with no group/other access (normally `0700`) and files with no group/other
|
||||
access (normally `0600`).
|
||||
- Commands that need storage/auth load filesystem secrets before constructing adapters.
|
||||
|
||||
## Publish Configuration Summary
|
||||
@@ -130,13 +280,16 @@ Rules:
|
||||
|
||||
| Field | Type | Required | Default / Rule |
|
||||
| --- | --- | --- | --- |
|
||||
| `composition.imports[]` | list of strings | No | explicit additive pipeline fragments relative to the root pipeline directory; `.yml` or `.yaml` regular files only |
|
||||
| `composition.default_profile` | string | Conditional | selected when profiles exist and no caller explicitly selects one; must name a declared profile |
|
||||
| `composition.profiles.<name>.overlay` | string | Conditional | required for every declared profile; one confined `.yml` or `.yaml` overlay relative to the root pipeline directory |
|
||||
| `pipeline.workspace.root` | string | No | `/var/lib/narratio` |
|
||||
| `pipeline.workspace.cleanup_after_publish` | bool | No | `false` |
|
||||
| `pipeline.campaigns.root` | string | No | `/usr/local/share/narratio/campaigns` |
|
||||
| `pipeline.campaigns.default_campaign_id` | string | No | empty |
|
||||
| `pipeline.secrets.env_dir` | string | No | empty |
|
||||
| `pipeline.storage.backend` | string | No | empty |
|
||||
| `pipeline.storage.s3.bucket` | string | Conditional | required for S3 session-audio and for publish upload when backend is `s3` |
|
||||
| `pipeline.storage.backend` | string | No | `local`; supported values are `local` and `s3` (case-insensitive) |
|
||||
| `pipeline.storage.s3.bucket` | string | Conditional | required when backend is `s3` and S3 session-audio or publish upload is enabled |
|
||||
| `pipeline.storage.s3.root_prefix` | string | No | `dnd` |
|
||||
| `pipeline.storage.s3.region` | string | No | empty |
|
||||
| `pipeline.storage.s3.endpoint` | string | No | empty |
|
||||
@@ -156,7 +309,7 @@ Rules:
|
||||
| `pipeline.publish.locks[]` | list | No | empty |
|
||||
| `pipeline.publish.locks[].source` | string | Yes (per lock) | must reference supported publish source |
|
||||
| `pipeline.publish.locks[].reason` | string | No | empty |
|
||||
| `pipeline.whisperx.transcribe_url` | string | Yes | valid URL |
|
||||
| `pipeline.whisperx.transcribe_url` | string | Yes | absolute `http` or `https` URL |
|
||||
| `pipeline.whisperx.language` | string | No | `en` |
|
||||
| `pipeline.whisperx.timeout` | duration | No | `30m` |
|
||||
| `pipeline.whisperx.retries` | int | No | `3` |
|
||||
@@ -205,6 +358,7 @@ Rules:
|
||||
| `pipeline.notarius.pipeline_id` | string | Conditional | required when enabled |
|
||||
| `pipeline.notarius.timeout` | duration | No | `3h`; must be positive |
|
||||
| `pipeline.notarius.working_directory` | string | No | directory containing resolved `config_path`; relative paths resolve from the pipeline file directory |
|
||||
| `pipeline.notarius.references` | map[string]string | No | empty; maps normalized Notarius selectors to supported prepared Narratio source IDs; maximum 256 entries |
|
||||
| `pipeline.notarius.outputs` | map | Conditional | at least one entry when enabled |
|
||||
| `pipeline.render.enabled` | bool | No | `true` |
|
||||
| `pipeline.render.format` | string | No | `markdown` (only supported value) |
|
||||
@@ -217,9 +371,46 @@ Rules:
|
||||
| `pipeline.scriptorium.timeout` | duration | No | `10m` |
|
||||
| `pipeline.scriptorium.render_debug` | bool | No | `false` |
|
||||
| `pipeline.scriptorium.artifacts` | map | No | empty |
|
||||
| `pipeline.notification.backend` | string | No | empty |
|
||||
| `pipeline.notification.recipient` | string | No | empty |
|
||||
| `pipeline.notification.timeout` | duration | No | empty |
|
||||
| `pipeline.scriptorium.artifact_families` | map | No | empty; expands one ordinary artifact per canonical party character |
|
||||
| `pipeline.notification.mode` | string | No | `noop`; the only supported notification mode until a provider is implemented |
|
||||
|
||||
### Notarius Reference Bindings
|
||||
|
||||
`pipeline.notarius.references` maps a Notarius CLI selector to a prepared
|
||||
Narratio source, not to a filesystem path:
|
||||
|
||||
```yaml
|
||||
notarius:
|
||||
references:
|
||||
glossary: narratio.input.glossary
|
||||
party: narratio.input.party
|
||||
players: narratio.input.players
|
||||
spell_catalog: narratio.input.spell_catalog
|
||||
```
|
||||
|
||||
Supported sources are `narratio.input.party`, `narratio.input.players`,
|
||||
`narratio.input.glossary`, and `narratio.input.spell_catalog`. Each map entry is
|
||||
required by its presence: omit a binding when the selected Notarius pipeline
|
||||
does not need it. A spell-catalog binding additionally requires an effective
|
||||
campaign or session `spell_catalog_file`.
|
||||
|
||||
Selectors accept Notarius's `slot`, `chunk.slot`, `lane.slot`,
|
||||
`lane.extract.slot`, `lane.merge.slot`, and `lane.normalize.slot` forms.
|
||||
Narratio trims whitespace around
|
||||
selectors and their dot-separated components, rejects empty components and
|
||||
`=`, rejects duplicate normalized selectors, and limits the map to 256 entries.
|
||||
It validates only selector structure and the prepared source vocabulary;
|
||||
Notarius owns target-slot declarations and media compatibility.
|
||||
|
||||
Before extraction, Narratio resolves every binding from the current prepared
|
||||
session manifest and streams it into a verified invocation-local snapshot whose
|
||||
absolute path is passed to Notarius. Missing, unsafe, empty,
|
||||
changed-during-copy, or checksum-inconsistent prepared evidence fails with
|
||||
guidance to force `prepare`. Bindings are sorted by normalized selector and are
|
||||
part of extraction fingerprint and resume identity. See the
|
||||
[Notarius integration contract](./integrations/notarius.md) for the subprocess
|
||||
boundary and the [complete example](../examples/pipeline.full.annotated.yml)
|
||||
for a copyable configuration.
|
||||
|
||||
### Notarius Output Entries
|
||||
|
||||
@@ -248,7 +439,7 @@ For each `pipeline.scriptorium.artifacts.<name>`:
|
||||
| Field | Type | Required | Rule |
|
||||
| --- | --- | --- | --- |
|
||||
| `enabled` | bool | No | `false` if omitted |
|
||||
| `depends_on[]` | list[string] | No | must reference configured artifact keys; no self-reference; enabled graph must be acyclic |
|
||||
| `depends_on[]` | list[string] | No | must reference configured artifact keys; no self-reference; configured graph must be acyclic |
|
||||
| `render_debug` | bool | No | per-artifact override |
|
||||
| `prompt_id` | string | Conditional | required when artifact is enabled |
|
||||
| `profile_id` | string | No | empty |
|
||||
@@ -259,41 +450,99 @@ For each `pipeline.scriptorium.artifacts.<name>`:
|
||||
|
||||
Narratio adds `session_id=narratio-session-<session_id>` to every Scriptorium request for sticky upstream LLM routing. If an artifact config sets `vars.session_id`, Narratio replaces that value before invoking Scriptorium. Use a different variable name if a prompt needs the raw Narratio session ID as content.
|
||||
|
||||
Without `--artifacts`, analyze executes enabled configured artifacts. With an
|
||||
explicit `--artifacts` list, the exact named configured artifacts are the
|
||||
one-invocation targets even if their `enabled` values are false. Analyze closes
|
||||
those targets over `depends_on`: a current prerequisite is reused, while a
|
||||
stale, missing, failed, or legacy prerequisite is rebuilt before its dependent.
|
||||
Unrelated artifacts are not executed. Named targets and any prerequisite that
|
||||
may require rebuilding must therefore have valid executable fields. This
|
||||
override affects analyze planning only; publish uses the list only to filter
|
||||
configured `narratio.artifact.<name>` output rules.
|
||||
|
||||
For each artifact input `pipeline.scriptorium.artifacts.<name>.inputs.<input_name>`:
|
||||
|
||||
| Field | Type | Required | Rule |
|
||||
| --- | --- | --- | --- |
|
||||
| `source` | string | Yes | built-in runtime source, prepared input source, `narratio.extraction.<name>`, `narratio.artifact.<name>`, or `narratio.previous_session.artifact.<name>` |
|
||||
| `artifact` | string | No | optional passthrough adapter field |
|
||||
| `path` | string | No | optional passthrough adapter field |
|
||||
| `required` | bool | No | optional input requirement |
|
||||
|
||||
`artifact` and `path` are obsolete and rejected by strict configuration
|
||||
loading. Use the canonical `source` identifier to select the input; Narratio
|
||||
does not provide adapter-specific input passthrough fields.
|
||||
|
||||
### Scriptorium Artifact Families
|
||||
|
||||
`pipeline.scriptorium.artifact_families` declares a shared artifact template
|
||||
for every canonical campaign character. Configuration resolution expands each
|
||||
family into ordinary `pipeline.scriptorium.artifacts` entries before analyze
|
||||
planning or Scriptorium invocation. A legacy party cannot be used for a family.
|
||||
|
||||
For each `pipeline.scriptorium.artifact_families.<name>`:
|
||||
|
||||
| Field | Type | Required | Rule |
|
||||
| --- | --- | --- | --- |
|
||||
| `enabled`, `prompt_id`, `profile_id`, `timeout`, `render_debug`, `depends_on`, `inputs`, `vars` | ordinary artifact fields | No | copied to each generated artifact under the corresponding ordinary rules |
|
||||
| `for_each` | string | Yes | exactly `party.characters` |
|
||||
| `output_path_pattern` | string | Yes | safe path beneath `artifacts/` with exactly one `{character_id}` token and no other brace syntax |
|
||||
| `member_vars` | map | No | maps an ordinary Scriptorium variable name to a supported canonical character selector |
|
||||
| `member_dependencies` | list | No | unique family keys; each generated member depends on the corresponding generated member of each listed family |
|
||||
| `publish` | map | No | typed family publish policy (`enabled`, `required`, `dest_pattern`) expanded into concrete publish outputs when enabled |
|
||||
|
||||
Generated keys are `<family>_<character_id>` and generated output paths must
|
||||
not collide with explicit artifacts or another generated artifact. Families
|
||||
expand even when disabled; normal analyze selection still omits disabled
|
||||
artifacts unless they are explicitly selected by their concrete key.
|
||||
|
||||
Supported `member_vars` selectors are `character_id`, `player.name`,
|
||||
`character.name`, `character.class_summary`, and `character.alias_summary`.
|
||||
Their resolved values are strings. A member variable may not reuse a static
|
||||
`vars` name; `session_id` remains owned and overwritten by Narratio as for any
|
||||
other Scriptorium artifact.
|
||||
|
||||
Within a family only, an input source may use
|
||||
`narratio.member_artifact.<family>`. The referenced family must be named in
|
||||
that family's `member_dependencies`; resolution rewrites the source to the
|
||||
corresponding ordinary `narratio.artifact.<family>_<character_id>` source.
|
||||
This syntax is rejected in explicit artifacts and never reaches runtime stages
|
||||
or Scriptorium.
|
||||
|
||||
### Notifications
|
||||
|
||||
Narratio currently supports only `notification.mode: noop`, which is also the
|
||||
default when the section is omitted. The notify stage performs no delivery in
|
||||
this mode. Backend, recipient, timeout, and other provider settings are
|
||||
rejected by strict configuration loading until Narratio has a provider
|
||||
integration.
|
||||
|
||||
### Campaign
|
||||
|
||||
| Field | Type | Required | Notes |
|
||||
| --- | --- | --- | --- |
|
||||
| `campaign_id` | string | Yes | canonical campaign identity |
|
||||
| `campaign_id` | string | Yes | canonical opaque campaign identity |
|
||||
| `session_template_file` | string | No | used by `session init` when set |
|
||||
| `inputs.speakers_file` | string | Yes | stable input default |
|
||||
| `inputs.autocorrect_file` | string | Yes | stable input default |
|
||||
| `inputs.glossary_file` | string | Yes | stable input default |
|
||||
| `inputs.players_file` | string | Yes | stable input default |
|
||||
| `inputs.party_file` | string | Yes | stable input default |
|
||||
| `inputs.players_file` | string | Conditional | required only with an unversioned legacy `party_file`; forbidden for a canonical party |
|
||||
| `inputs.party_file` | string | Yes | stable campaign party source; relative paths resolve from `campaign.yml` |
|
||||
| `inputs.spell_catalog_file` | string | No | optional spell-catalog overlay default; required when a Notarius reference selects `narratio.input.spell_catalog` |
|
||||
|
||||
### Session
|
||||
|
||||
| Field | Type | Required in session file | Notes |
|
||||
| --- | --- | --- | --- |
|
||||
| `session_id` | string | Yes | must match CLI session target when provided |
|
||||
| `previous_session_id` | string | No | must not equal `session_id` |
|
||||
| `campaign` | string | No | filled from `campaign_id` during resolve if omitted |
|
||||
| `session_id` | string | Yes | opaque identity; must match CLI session target when provided |
|
||||
| `previous_session_id` | string | No | opaque identity; must not equal `session_id` |
|
||||
| `campaign` | string | No | opaque identity; filled from `campaign_id` during resolve if omitted |
|
||||
| `date` | string | No | metadata |
|
||||
| `title` | string | No | metadata |
|
||||
| `inputs.speakers_file` | string | No | overrides campaign stable input |
|
||||
| `inputs.autocorrect_file` | string | No | overrides campaign stable input |
|
||||
| `inputs.glossary_file` | string | No | overrides campaign stable input |
|
||||
| `inputs.players_file` | string | No | overrides campaign stable input |
|
||||
| `inputs.party_file` | string | No | overrides campaign stable input |
|
||||
| `inputs.players_file` | string | No | legacy-party override; forbidden for a canonical party |
|
||||
| `inputs.party_file` | string | No | legacy-party override; forbidden for a canonical campaign party |
|
||||
| `inputs.spell_catalog_file` | string | No | overrides the optional campaign spell catalog; empty or omitted inherits the campaign value |
|
||||
| `inputs.audio_dir` | string | Conditional | local audio mode |
|
||||
| `inputs.audio_files[]` | list[string] | Conditional | local audio mode |
|
||||
| `inputs.audio_s3.prefix` | string | Conditional | S3 audio mode |
|
||||
@@ -301,6 +550,20 @@ For each artifact input `pipeline.scriptorium.artifacts.<name>.inputs.<input_nam
|
||||
Audio rules:
|
||||
|
||||
- configure local mode (`audio_dir` or `audio_files`) or S3 mode (`audio_s3.prefix`), not both.
|
||||
- `audio_s3` requires `pipeline.storage.backend: s3` and a configured S3 bucket.
|
||||
|
||||
### Storage backend selection
|
||||
|
||||
`local` is the default and disables remote object-store operations. Configure
|
||||
`s3` explicitly before supplying `storage.s3`; a populated S3 block does not
|
||||
select a backend on its own. Unknown backend names and an S3 block paired with
|
||||
`local` are rejected during configuration validation.
|
||||
|
||||
### Previous-session expectation
|
||||
|
||||
`previous_session_id` is optional in a session file. When a command supplies
|
||||
`--previous-session-id`, however, the session file must contain the same value;
|
||||
an omitted or different value is rejected before the command performs work.
|
||||
|
||||
## Maintained Examples
|
||||
|
||||
|
||||
@@ -25,19 +25,35 @@ polished transcripts and generated artifacts. Start with the
|
||||
| Adapters or external tool contracts | [Adapter Internals](internal/adapters.md) and [Integration Contracts](integrations/README.md) | The internal guide owns adapter composition and mechanics; integration documents own external formats and protocols. |
|
||||
| Manifests, artifacts, workspace paths, or publish behavior | [Manifest Internals](internal/manifest.md), [Artifact Internals](internal/artifacts.md), [Workspace Internals](internal/workspace.md), [Publish Internals](internal/stage-publish.md), and [Operations](operations.md) | These separate implementation state and resolution from operator-visible layout and lifecycle. |
|
||||
| Maintained configuration or input examples | [Configuration](config.md) and [Examples](../examples/README.md) | The reference owns field meanings; the examples directory owns complete copyable files. |
|
||||
| Proposed or unimplemented behavior | [Roadmap](roadmap/) | Future work belongs only in roadmap documentation until implemented. |
|
||||
| Preparing, validating, or publishing a release | [Release Procedure](release.md) | The maintainer procedure owns version selection, candidate validation, guarded tag publication, and optional later CI inspection. |
|
||||
| Proposed or unimplemented behavior | `docs/roadmap/` | Future work belongs only in roadmap documentation until implemented. |
|
||||
|
||||
For an existing subsystem, also inspect its focused tests and package-level
|
||||
contracts before changing behavior.
|
||||
|
||||
## Validation
|
||||
|
||||
Use focused package tests while iterating. Run the repository-wide checks when a
|
||||
change affects shared contracts, application behavior, or maintained
|
||||
documentation examples:
|
||||
Use focused package tests while iterating. Every pull request and push runs the
|
||||
following repository-wide checks before it can be accepted:
|
||||
|
||||
```sh
|
||||
go test ./...
|
||||
go test -race ./...
|
||||
go vet ./...
|
||||
go build ./cmd/narratio
|
||||
go build ./...
|
||||
go test ./internal/doccheck
|
||||
go test ./internal/config -run '^TestExamplesLoadAndValidate$'
|
||||
```
|
||||
|
||||
The documentation check verifies local Markdown links and the dependency graph
|
||||
of the Woodpecker workflows. The configuration check loads every maintained
|
||||
pipeline and session example. Tag CI reuses this validation path before its
|
||||
asynchronous asset publication; the maintainer release boundary is documented
|
||||
in the [Release Procedure](release.md).
|
||||
|
||||
Woodpecker also runs `go test -race -shuffle=on -count=3 ./...` on its scheduled
|
||||
job to expose ordering and repeatability defects. Current runners cross-compile
|
||||
for macOS and Windows, but do not provide native macOS or Windows execution.
|
||||
Those cross-builds establish compilation only, not platform-equivalent runtime
|
||||
evidence. Add native checks only when official runner labels and successful
|
||||
native-run evidence are available.
|
||||
|
||||
@@ -21,6 +21,7 @@ focused stage documents.
|
||||
- [Audita](./audita.md): transcript polishing (`audita process`).
|
||||
- [Notarius](./notarius.md): complete pipeline execution and safe JSON bundle
|
||||
discovery (`notarius run`).
|
||||
- [Party](./party.md): canonical campaign roster input.
|
||||
- [Seriatim](./seriatim.md): merge, normalize, trim, and render operations.
|
||||
- [Scriptorium](./scriptorium.md): artifact generation and debug rendering
|
||||
(`scriptorium run|render`).
|
||||
|
||||
@@ -15,7 +15,12 @@ runner composition is documented in
|
||||
- required transcript/glossary/output/work-dir paths;
|
||||
- optional report path (required when report mode is enabled);
|
||||
- generated config and stdout/stderr log paths;
|
||||
- optional module/model/base-url/config/output-schema/concurrency settings.
|
||||
- optional per-invocation module override.
|
||||
|
||||
The constructed runner owns static Audita settings: binary, timeout,
|
||||
credentials, default modules, model and endpoint settings, validation and output
|
||||
settings, report mode, and concurrency. The `polish` stage supplies only
|
||||
invocation-specific paths and may override modules for that invocation.
|
||||
|
||||
## Result Contract
|
||||
`PolishResult` returns:
|
||||
@@ -41,6 +46,9 @@ Run fails for:
|
||||
- invalid processed transcript JSON (`segments` array required);
|
||||
- invalid report JSON when reporting is enabled.
|
||||
|
||||
Processed transcript JSON is limited to 64 MiB and optional report JSON to 16
|
||||
MiB. Both must be regular files without symlinked path components.
|
||||
|
||||
Failure results still include output/log/config/exit metadata for diagnostics.
|
||||
|
||||
## Deterministic Behavior
|
||||
|
||||
@@ -7,12 +7,14 @@ lanes from the final trimmed Seriatim transcript. Narratio owns invocation,
|
||||
safe bundle discovery, lane selection, and its own artifact metadata. Notarius
|
||||
owns pipeline definitions, lane schemas, the receipt, and bundle formats.
|
||||
|
||||
Canonical Notarius references:
|
||||
Canonical Notarius v0.6.0 references:
|
||||
|
||||
- [Subprocess consumer contract](https://gitea.maximumdirect.net/eric/notarius/src/branch/main/docs/consumers/subprocess.md)
|
||||
- [D&D pipeline and lane contracts](https://gitea.maximumdirect.net/eric/notarius/src/branch/main/docs/consumers/dnd-pipeline.md)
|
||||
- [Run-result receipt](https://gitea.maximumdirect.net/eric/notarius/src/branch/main/docs/integrations/run-result.md)
|
||||
- [JSON output bundle](https://gitea.maximumdirect.net/eric/notarius/src/branch/main/docs/integrations/json-output.md)
|
||||
- [CLI reference](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/cli.md)
|
||||
- [Subprocess consumer contract](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/consumers/subprocess.md)
|
||||
- [D&D pipeline and lane contracts](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/consumers/dnd-pipeline.md)
|
||||
- [Run-result receipt](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/integrations/run-result.md)
|
||||
- [JSON output bundle](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/integrations/json-output.md)
|
||||
- [D&D spell-catalog overlay](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/integrations/dnd-spell-catalog-overlays.md)
|
||||
|
||||
The [complete Narratio example](../../examples/pipeline.full.annotated.yml)
|
||||
records the exact current constraints for all ten D&D lanes. Treat the linked
|
||||
@@ -23,39 +25,83 @@ duplicate the complete schemas.
|
||||
|
||||
When `pipeline.notarius.enabled` is true, Narratio resolves the executable,
|
||||
configuration path, input path, output directory, and working directory to
|
||||
absolute paths and invokes:
|
||||
absolute paths. Narratio requires the Notarius v0.6.0 CLI contract when
|
||||
references are configured and invokes each binding as a separate argument
|
||||
before `--json`:
|
||||
|
||||
```text
|
||||
notarius run <pipeline_id> --config <config_path> --input <trimmed_json> --output-dir <staging_dir> --json
|
||||
notarius run <pipeline_id> --config <config_path> --input <trimmed_json> --output-dir <staging_dir> [--reference <selector>=<verified_snapshot_path>]... --json
|
||||
```
|
||||
|
||||
Reference paths are absolute invocation-local snapshots streamed from the
|
||||
manifest-verified canonical files prepared inside the current Narratio session
|
||||
workspace. Narratio verifies snapshot checksum and size before and after the
|
||||
subprocess, and passes only configured bindings, ordered lexically by normalized
|
||||
selector, as direct argument-vector entries without shell interpretation. A CLI
|
||||
binding takes precedence over a matching external path in Notarius
|
||||
configuration. Narratio never emits `--without-reference`.
|
||||
|
||||
For canonical party configuration, the `party` binding is the unchanged,
|
||||
validated authored roster and the `players` binding is its generated
|
||||
projection. Both retain their established `narratio.input.party` and
|
||||
`narratio.input.players` source IDs, and both are resolved from the prepared
|
||||
manifest rather than from campaign configuration at extraction time.
|
||||
|
||||
The maintained D&D boundary binds only the four campaign-owned external slots:
|
||||
|
||||
```text
|
||||
notarius run dnd-session \
|
||||
--config <absolute config path> \
|
||||
--input <absolute trimmed transcript path> \
|
||||
--output-dir <absolute staging directory> \
|
||||
--reference glossary=<absolute verified glossary snapshot> \
|
||||
--reference party=<absolute verified party snapshot> \
|
||||
--reference players=<absolute verified players snapshot> \
|
||||
--reference spell_catalog=<absolute verified spell catalog snapshot> \
|
||||
--json
|
||||
```
|
||||
|
||||
The spell-catalog binding is omitted when the campaign does not maintain that
|
||||
optional overlay. Registry, scene-description, combat-turn, and occurrence
|
||||
handoffs generated during the same Notarius run remain in Notarius pipeline
|
||||
composition and must not be emitted as CLI references. The linked CLI and D&D
|
||||
consumer documents own selector targeting, declared slots, media compatibility,
|
||||
and generated-handoff collision rules.
|
||||
|
||||
Standard output is reserved for the JSON receipt. Standard error is captured
|
||||
separately as diagnostic output. Narratio applies the configured timeout and
|
||||
does not interpret stdout as a receipt unless the subprocess exits successfully.
|
||||
It does not pass a Narratio session ID or run `notarius config validate`
|
||||
automatically; the configured working directory and inherited environment
|
||||
apply to the subprocess.
|
||||
automatically; the configured working directory and Narratio's minimal child
|
||||
environment apply to the subprocess.
|
||||
|
||||
## Accepted Result
|
||||
|
||||
Narratio currently accepts receipt schema `notarius.run-result.v1`. The receipt
|
||||
Narratio's supported invocation baseline is Notarius v0.6.0. The accepted
|
||||
receipt remains `notarius.run-result.v2`; reference flags do not change the
|
||||
receipt or ten-lane output contract. The receipt
|
||||
must identify the configured pipeline, and its `index_file` must be exactly
|
||||
`index.json` beneath the reported bundle root. The production index must name
|
||||
the management files exactly as `manifest.json`, `rejected.json`, and
|
||||
`warnings.json`. All receipt, index, and lane paths must stay inside that
|
||||
bundle; symlinks and non-regular lane payloads are rejected.
|
||||
the management files exactly as `manifest.json`, `rejected.json`,
|
||||
`warnings.json`, and `diagnostics.json`. All receipt, index, and lane paths must
|
||||
stay inside that bundle; symlinks and non-regular lane payloads are rejected.
|
||||
|
||||
Supported receipt and index shapes tolerate unknown fields for forward
|
||||
compatibility, while required identity, validation, count, manifest,
|
||||
rejection, warning, and lane-list fields remain mandatory. Narratio applies
|
||||
bounded reads to the receipt, index, rejection, and warning documents. Optional
|
||||
chunk-map and evidence-context descriptors must carry their complete generic
|
||||
contract metadata when present.
|
||||
rejection, warning, diagnostic, and lane-list fields remain mandatory.
|
||||
Narratio applies bounded reads to the receipt, index, rejection, warning, and
|
||||
diagnostic documents. Warning and diagnostic envelopes, group counts,
|
||||
occurrence counts, truncation state, framework-owned origins, and
|
||||
receipt-to-bundle counts must be internally consistent. Optional chunk-map and
|
||||
evidence-context descriptors must carry their complete generic contract
|
||||
metadata when present.
|
||||
|
||||
For every entry in `pipeline.notarius.outputs`, Narratio requires exactly one
|
||||
index descriptor with the configured lane ID, media type, schema ID, schema
|
||||
version, and, when configured, module key. Missing, duplicate, rejected, or
|
||||
incompatible required lanes fail extraction even if Notarius exited zero.
|
||||
incompatible required lanes fail extraction even if Notarius exited zero. A
|
||||
configured lane whose v2 validation summary is `rejected` or `incomplete` also
|
||||
fails extraction.
|
||||
Unconfigured lanes may remain in the preserved bundle but do not become
|
||||
selectable Narratio sources.
|
||||
|
||||
@@ -72,10 +118,13 @@ only explicitly named lane sources; `--artifacts` never selects Notarius lanes.
|
||||
staged bundle is promoted to durable storage.
|
||||
- Contract and external provenance metadata are preserved on lane artifact
|
||||
records and through explicit publication.
|
||||
- Undeclared selectors, incompatible reference files, and external/generated
|
||||
reference collisions are Notarius errors and fail extraction normally.
|
||||
|
||||
Rejection and warning summaries retain structured stage, scope, lane, and
|
||||
reason-code fields for diagnostics without exposing free-form external messages
|
||||
or reading lane payload bodies.
|
||||
Rejection, validation, warning, and diagnostic summaries retain bounded stable
|
||||
identity, category, origin, reason-code, status, and occurrence fields without
|
||||
copying free-form external messages into Narratio manifest metadata or reading
|
||||
lane payload bodies.
|
||||
|
||||
Configuration fields and defaults are in [Configuration](../config.md).
|
||||
Operator paths, rerun procedures, and bundle retention are in
|
||||
|
||||
73
docs/integrations/party.md
Normal file
73
docs/integrations/party.md
Normal file
@@ -0,0 +1,73 @@
|
||||
# Canonical Party Input
|
||||
|
||||
`party.yml` is a campaign-owned roster input. Narratio recognizes the
|
||||
versioned `narratio.party.v1` document below when it resolves a pipeline,
|
||||
campaign, and session together.
|
||||
|
||||
```yaml
|
||||
schema_version: narratio.party.v1
|
||||
|
||||
characters:
|
||||
arannis:
|
||||
player:
|
||||
name: Eric
|
||||
character:
|
||||
name: Arannis
|
||||
alias:
|
||||
- Ari
|
||||
- The Grey Owl
|
||||
classes:
|
||||
- name: wizard
|
||||
level: 8
|
||||
```
|
||||
|
||||
`characters` is a non-empty mapping. Each key is a stable character ID using
|
||||
the configured-artifact key grammar: a lowercase ASCII letter followed by zero
|
||||
or more lowercase ASCII letters, digits, or underscores. Character order is
|
||||
preserved where roster order matters.
|
||||
|
||||
Every entry has `player.name`, `character.name`, and a non-empty
|
||||
`character.classes` list. Class entries require a non-empty `name` and may
|
||||
include a positive integer `level`. The optional, intentionally singular
|
||||
`character.alias` field is a list. Names, aliases, and class names must be
|
||||
non-empty, trimmed display strings without control characters. Character names
|
||||
and aliases must be unique across the full roster under Unicode-aware
|
||||
case-insensitive comparison; player names may repeat.
|
||||
|
||||
The document has exactly one YAML document and accepts no unknown fields. A
|
||||
wrong or malformed `schema_version` is an error.
|
||||
|
||||
## Legacy migration boundary
|
||||
|
||||
An unversioned party input remains supported only as opaque legacy reference
|
||||
material while campaigns migrate. It requires a separate `players_file` and
|
||||
retains the existing session override behavior. It cannot be mixed with a
|
||||
canonical party: canonical campaigns must omit `players_file`, and sessions
|
||||
must not override their party or players inputs.
|
||||
|
||||
Use the canonical document for new campaigns. The configuration rules and
|
||||
source-relative path behavior are defined in the [Configuration Reference](../config.md).
|
||||
|
||||
## Derived players document
|
||||
|
||||
During `prepare`, Narratio copies the canonical party source bytes unchanged
|
||||
to `inputs/party.yml` and writes this deterministic players-only projection to
|
||||
`inputs/players.yml`:
|
||||
|
||||
```yaml
|
||||
schema_version: narratio.players.v1
|
||||
players:
|
||||
- name: Eric
|
||||
character:
|
||||
id: arannis
|
||||
name: Arannis
|
||||
alias:
|
||||
- Ari
|
||||
- The Grey Owl
|
||||
```
|
||||
|
||||
There is one entry per character, sorted by stable character ID. Repeated
|
||||
player names remain separate entries. The optional `alias` list retains its
|
||||
declared order and is omitted when empty. The projection carries no class
|
||||
data. Its prepared manifest record is marked `derived_from_party`; it is not a
|
||||
separate user-provided `players_file`.
|
||||
@@ -46,6 +46,9 @@ Run behavior:
|
||||
- `run` exit code `2` is mapped to `ValidationFailed=true`;
|
||||
- successful subprocess still fails if output file is missing or empty.
|
||||
|
||||
Each artifact result is limited to 64 MiB and must be a regular file without
|
||||
symlinked path components.
|
||||
|
||||
Render behavior:
|
||||
- subprocess errors propagate;
|
||||
- output file must exist and be non-empty.
|
||||
|
||||
@@ -41,6 +41,8 @@ Invocation fails on:
|
||||
- empty render output files.
|
||||
|
||||
When report paths are provided/enabled, report files must parse as JSON.
|
||||
Each Seriatim JSON or rendered-text result is limited to 64 MiB and must be a
|
||||
regular file without symlinked path components.
|
||||
|
||||
## Deterministic Behavior
|
||||
- argument ordering is deterministic per command construction.
|
||||
|
||||
@@ -17,6 +17,11 @@ Narratio sends an HTTP `POST` to the configured transcription URL using
|
||||
The server must return a `2xx` response whose body is valid JSON. Narratio does
|
||||
not currently require a more specific response schema at this boundary.
|
||||
|
||||
The transcription URL must be an absolute `http` or `https` URL. The audio body
|
||||
is streamed through a fresh multipart writer for every attempt, so its memory
|
||||
use is bounded by the transport buffer rather than by the complete audio file.
|
||||
WhisperX response acquisition is capped at 10 MiB.
|
||||
|
||||
## Request And Result Contract
|
||||
|
||||
Each adapter request identifies a speaker, a readable audio file, and the
|
||||
@@ -40,7 +45,7 @@ transcript output.
|
||||
|
||||
## Validation And Failure Semantics
|
||||
|
||||
Client construction rejects a missing or invalid absolute transcription URL,
|
||||
Client construction rejects a missing or non-HTTP(S) absolute transcription URL,
|
||||
a missing language, a non-positive timeout, negative retries, or a negative
|
||||
retry delay. A request fails before transmission when its audio or output path
|
||||
is missing.
|
||||
|
||||
@@ -35,18 +35,33 @@ Adapters do not own:
|
||||
|
||||
## Default Wiring
|
||||
|
||||
`internal/app/runner.go` initializes default adapters when not injected:
|
||||
`internal/app/runner.go` initializes default adapters when not injected and
|
||||
only when the selected execution plan needs them:
|
||||
|
||||
- WhisperX HTTP client from pipeline config.
|
||||
- Seriatim subprocess runner.
|
||||
- Audita subprocess runner.
|
||||
- Scriptorium subprocess runner.
|
||||
- Notarius subprocess runner when extraction is enabled.
|
||||
- Noop notifier (`notify.NoopSender`).
|
||||
- WhisperX HTTP client for `transcribe`.
|
||||
- Seriatim subprocess runner for `merge`, `normalize`, `trim`, or `render`.
|
||||
- Audita subprocess runner for `polish`.
|
||||
- Scriptorium subprocess runner for `trim` or `analyze`.
|
||||
- Notarius subprocess runner for `extract` when extraction is enabled.
|
||||
- Noop notifier (`notify.NoopSender`) for `notify`.
|
||||
- Object store only when required by selected stages/config.
|
||||
|
||||
Remote publish locks are loaded only for a selected, enabled publish that
|
||||
uploads a run. Shared session lifecycle setup still applies to every selected
|
||||
range, but an unselected integration is neither initialized nor validated by
|
||||
runner composition. Each selected stage retains its own fail-fast configuration
|
||||
and input validation.
|
||||
|
||||
`session plan` is outside production adapter composition. It performs
|
||||
resume validation and models selected transitions against cloned manifest
|
||||
state without constructing or invoking stage-execution adapters. The shared
|
||||
command configuration loader may still use object storage to retrieve a missing
|
||||
remote session file before planning begins.
|
||||
|
||||
Notarius is composed only when extraction is enabled; the extract stage owns
|
||||
receipt, bundle, and configured-lane policy rather than the adapter.
|
||||
prepared reference resolution, receipt, bundle, and configured-lane policy.
|
||||
The adapter validates the ordered selector/absolute-path pairs and is the sole
|
||||
owner of serializing them as repeated `--reference` arguments before `--json`.
|
||||
|
||||
Object-store construction goes through `newCommandObjectStore`, which loads
|
||||
configured filesystem secrets before adapter initialization.
|
||||
@@ -56,6 +71,18 @@ configured filesystem secrets before adapter initialization.
|
||||
- Constructor errors fail stage execution setup early.
|
||||
- Runtime adapter errors propagate to stage code and then manifest failure handling.
|
||||
- Subprocess adapters persist stage logs/generated configs through stage-managed paths.
|
||||
- Shared subprocess execution starts an owned process group on Linux/macOS or a
|
||||
kill-on-close job object on Windows. Every terminal path disposes of that
|
||||
owned tree before returning. After a natural leader exit, Unix checks for
|
||||
remaining group members and uses bounded graceful then forceful termination;
|
||||
Windows closes the job so kill-on-close applies. Cancellation, deadlines, and
|
||||
diagnostic limits use the same terminal disposal path without losing their
|
||||
original result classification. Child environments contain only the execution
|
||||
baseline and adapter-specified values; configured credentials are explicit
|
||||
sensitive values. Stdout and stderr are redacted while streaming into separate
|
||||
8 MiB diagnostic captures; a bounded wait closes a stream retained by a
|
||||
departed leader's descendant. Unsupported platforms reject owned command
|
||||
execution.
|
||||
|
||||
## Implementation And Tests
|
||||
|
||||
|
||||
@@ -30,23 +30,60 @@ and content validator. The focused stage documents own their input/output flow;
|
||||
- extraction source ID format: `narratio.extraction.<output_key>`
|
||||
- previous-session source ID format: `narratio.previous_session.artifact.<artifact_key>`
|
||||
|
||||
All formats are validated by strict source-policy rules. Extraction sources are
|
||||
registered only from `pipeline.notarius.outputs`; the Notarius index has no
|
||||
selectable source ID.
|
||||
All formats are validated by strict source-policy rules. Configured artifact and
|
||||
extraction keys use `^[a-z][a-z0-9_]*$`; source parsers never normalize an
|
||||
unrecognized token into a valid source. Extraction sources are registered only
|
||||
from `pipeline.notarius.outputs`; the Notarius index has no selectable source
|
||||
ID.
|
||||
|
||||
Prepared stable source IDs are `narratio.input.players`,
|
||||
`narratio.input.party`, `narratio.input.glossary`, and
|
||||
`narratio.input.spell_catalog`. Artifact policy owns their canonical manifest
|
||||
kind and prepared filename vocabulary. Canonical party mode preserves the
|
||||
party source bytes in the party record and supplies the players record from the
|
||||
deterministic `derived_from_party` projection; both remain ordinary prepared
|
||||
source IDs for consumers.
|
||||
|
||||
## Runtime Catalog
|
||||
|
||||
`ArtifactCatalog` tracks:
|
||||
|
||||
- `planned`: source registered for run context;
|
||||
- `executable`: selected and enabled for analyze execution;
|
||||
- `available`: local file exists and validates;
|
||||
- `executable`: included in the effective analyze artifact set;
|
||||
- `available`: the source's canonical evidence owner validates its current
|
||||
manifest record and durable bytes;
|
||||
- `provenance`: availability source.
|
||||
|
||||
Configured definitions are always registered. Without an explicit selection,
|
||||
the effective analyze set contains enabled definitions. With `--artifacts`, the
|
||||
exact named configured definitions become the effective set for that invocation,
|
||||
regardless of their `enabled` value. The effective-set resolver itself does not
|
||||
expand dependencies; the analyze work planner closes those targets over their
|
||||
configured prerequisite graph. Availability is separate from executability.
|
||||
Configuration may normalize a family selection into its concrete generated
|
||||
members before this resolver runs. The effective set retains optional family
|
||||
and character origin metadata, but its keys, catalog sources, and runtime
|
||||
lookups remain concrete configured-artifact identities.
|
||||
Family publish policies are likewise expanded into ordinary configured-source
|
||||
publish rules during configuration resolution.
|
||||
Configured outputs, including non-executable prerequisites, become available
|
||||
only when the versioned analyze state identifies a current result whose source,
|
||||
contract, canonical configured path, size, and checksum match a confined
|
||||
no-follow regular file. An incidental canonical file and a legacy aggregate
|
||||
analyze output are unavailable.
|
||||
Extraction entries are registered from configuration and become available only
|
||||
after compatible extraction evidence is hydrated.
|
||||
|
||||
During an analyze invocation, a newly validated and atomically materialized
|
||||
configured output is marked available with its producer run ID, contract,
|
||||
checksum, and size. Later scheduled dependents therefore observe the same
|
||||
semantic identity whether their prerequisite was reused from current manifest
|
||||
evidence or produced earlier in the invocation.
|
||||
|
||||
Current provenance values:
|
||||
|
||||
- `generated.current_analyze_run`
|
||||
- `filesystem.disabled_artifact_output`
|
||||
- `manifest.current_analyze_artifact`
|
||||
- `manifest.inputs.previous_cache`
|
||||
- `current_session.previous_cache`
|
||||
|
||||
@@ -59,15 +96,41 @@ Built-ins:
|
||||
|
||||
Configured sources (`narratio.artifact.*`):
|
||||
|
||||
- resolve only through runtime catalog availability.
|
||||
- resolve only through runtime catalog availability;
|
||||
- use the shared typed analyze-evidence inspection in
|
||||
`analyze_evidence.go` for prior current-session results;
|
||||
- require the supported analyze-state and fingerprint versions, a `current`
|
||||
record for the exact configured key and source ID, a complete contract, the
|
||||
configured canonical relative path, positive stored size, and stored
|
||||
checksum matching bytes read from a confined no-follow regular file; and
|
||||
- treat non-current statuses, legacy or malformed records, removed keys,
|
||||
unsafe or missing files, and size/checksum mismatches as unavailable without
|
||||
rewriting manifest state. Catalog construction iterates current
|
||||
configuration, so removed or renamed records are not advertised.
|
||||
|
||||
`narratio.member_artifact.*` is not a runtime source family. Configuration
|
||||
resolution accepts it only in an artifact-family declaration and rewrites it
|
||||
to the corresponding configured source before this catalog is built.
|
||||
|
||||
Prepared stable sources (`narratio.input.*`):
|
||||
|
||||
- resolve only from the current manifest's exact prepared-input record;
|
||||
- require the policy-owned canonical path below the session root, a confined
|
||||
non-symlink regular file, a non-empty payload, and a matching SHA-256
|
||||
checksum; and
|
||||
- return an immutable source/path/checksum/size identity shared by extract and
|
||||
analyze rather than falling back to campaign/session source paths.
|
||||
|
||||
Extraction sources (`narratio.extraction.*`):
|
||||
|
||||
- use the shared registration and manifest hydration path in
|
||||
`extraction_catalog.go`;
|
||||
- use the shared typed bundle evidence inspection in `extraction_evidence.go`;
|
||||
- require a current successful extract record with the exact configured source,
|
||||
compatible contract and Notarius provenance, a confined regular durable
|
||||
payload, and matching checksum; and
|
||||
payload, matching checksum, and the current resolved trimmed-transcript
|
||||
identity;
|
||||
- remain unavailable unless catalog hydration receives valid evidence. Resume
|
||||
treats absent or obsolete evidence as a rerun decision and unsafe evidence as
|
||||
an error; and
|
||||
- are never inferred by scanning the Notarius bundle directory.
|
||||
|
||||
Previous-session sources (`narratio.previous_session.artifact.*`):
|
||||
@@ -76,6 +139,10 @@ Previous-session sources (`narratio.previous_session.artifact.*`):
|
||||
- prefer manifest-backed previous-input paths;
|
||||
- fallback to existing previous-cache filesystem paths.
|
||||
|
||||
Source absence is evaluated by the consuming artifact input. An optional input
|
||||
is omitted from that invocation; a required input fails resolution. This is
|
||||
separate from a stage's lifecycle outcome.
|
||||
|
||||
Validation by content type:
|
||||
|
||||
- transcript JSON built-ins: JSON with top-level `segments` array;
|
||||
@@ -87,7 +154,7 @@ Validation by content type:
|
||||
|
||||
`CollectPreviousArtifactRequirements`:
|
||||
|
||||
- scans enabled configured artifacts only;
|
||||
- scans the effective configured artifact set;
|
||||
- extracts only canonical previous-session sources;
|
||||
- deduplicates by artifact key;
|
||||
- merges required and optional references (required wins);
|
||||
@@ -98,12 +165,31 @@ Validation by content type:
|
||||
Artifacts package owns shared remote current-state loading mechanics used by
|
||||
restore, status and validation checks, and previous-cache planning.
|
||||
|
||||
For a new-protocol current state, the pointer-selected immutable commit is the
|
||||
complete restore authority. Callers receive its declared object identities and
|
||||
must not supplement them by listing mutable session prefixes. The legacy reader
|
||||
is intentionally separate and remains migration-only support.
|
||||
|
||||
The reader opens each small control object directly and enforces owner-specific
|
||||
limits before decoding: 64 KiB for the mutable commit pointer, 4 MiB for the
|
||||
immutable commit manifest, and 8 MiB for the selected session manifest. Legacy
|
||||
compatibility applies a 4 KiB limit to `current/run_id.txt` and the same 8 MiB
|
||||
manifest limit to `current/manifest.json`. These are exposed as
|
||||
`MaxCurrentCommitPointerBytes`, `MaxRemoteCommitManifestBytes`,
|
||||
`MaxRemoteSessionManifestBytes`, `MaxLegacyCurrentRunPointerBytes`, and
|
||||
`MaxLegacyCurrentManifestBytes`.
|
||||
|
||||
Each read uses the generation and size metadata returned with its opened body.
|
||||
Actual bytes remain subject to a limit-plus-one read even if size metadata is
|
||||
absent or inaccurate. Immutable selections then retain their declared-size,
|
||||
checksum, generation, and identity checks. No current-state control object is
|
||||
downloaded through a temporary file.
|
||||
|
||||
Core helpers:
|
||||
|
||||
- `LoadCurrentRunPointer`
|
||||
- `LoadCurrentManifest`
|
||||
- `LoadCurrentState`
|
||||
- `ValidateCurrentStateIdentity`
|
||||
- `RemoteCommitManifest` and `CurrentCommitPointer`
|
||||
|
||||
Typed missing-state errors:
|
||||
|
||||
@@ -131,6 +217,18 @@ Caller policy is intentionally outside artifacts helpers:
|
||||
- spool/cache paths;
|
||||
- S3 session/run/current-state key layout.
|
||||
|
||||
New publication creates run-scoped immutable objects, including
|
||||
`runs/{run_id}/commit.json` and `runs/{run_id}/session-manifest.json`. The sole
|
||||
mutable selector is `current/commit-pointer.json`; readers verify its selected
|
||||
commit and declared object generations/checksums. Legacy current-pair loading
|
||||
is confined to `current_state_legacy.go` for migration only.
|
||||
|
||||
Campaign, session, and Narratio run IDs are validated as portable opaque
|
||||
segments at configuration and artifact boundaries before they can be used in a
|
||||
workspace or S3 namespace. Previous-artifact destinations remain typed,
|
||||
multi-segment relative paths and are confined beneath `previous/artifacts`; they
|
||||
are not treated as opaque identifiers.
|
||||
|
||||
See [Workspace Internals](workspace.md) for how callers consume local helpers
|
||||
and [Operations](../operations.md#local-state-layout) for the authoritative
|
||||
physical layout.
|
||||
@@ -148,8 +246,13 @@ physical layout.
|
||||
|
||||
- Registry and resolution: `internal/artifacts/artifact_resolver.go`,
|
||||
`internal/artifacts/catalog.go`, `internal/artifacts/transcripts.go`,
|
||||
`internal/artifacts/extraction_catalog.go`
|
||||
- Current state: `internal/artifacts/current_state.go`
|
||||
`internal/artifacts/extraction_catalog.go`,
|
||||
`internal/artifacts/extraction_evidence.go`,
|
||||
`internal/artifacts/extraction_input.go`,
|
||||
`internal/artifacts/prepared_input.go`
|
||||
- Current state: `internal/artifacts/current_state.go`,
|
||||
`internal/artifacts/current_state_commit.go`,
|
||||
`internal/artifacts/current_state_legacy.go`
|
||||
- Paths and keys: `internal/artifacts/paths.go`,
|
||||
`internal/artifacts/s3_keys.go`
|
||||
- Previous requirements: `internal/artifacts/previous_requirements.go`
|
||||
|
||||
@@ -8,8 +8,8 @@ reporting flow in `internal/app`. User invocation belongs in
|
||||
physical restore scope belong in
|
||||
[Operations](../operations.md#restore-workflow).
|
||||
|
||||
Restore is split into explicit phases so remote authority, local conflict
|
||||
policy, and filesystem mutation can be tested independently.
|
||||
Restore separates remote authority, local conflict policy, and filesystem
|
||||
mutation so each remains testable independently.
|
||||
|
||||
## Discovery Contract
|
||||
|
||||
@@ -18,6 +18,7 @@ Discovery delegates current-state pointer and manifest loading to
|
||||
|
||||
- campaign must match;
|
||||
- session ID must match.
|
||||
- run ID must match the pointer-selected committed run.
|
||||
|
||||
Restore treats any missing or invalid remote current state as a command error.
|
||||
|
||||
@@ -31,13 +32,20 @@ Restore planner action kinds:
|
||||
|
||||
Planner behavior:
|
||||
|
||||
- remote list scope is the resolved session prefix;
|
||||
- a new-protocol restore uses only the selected commit's declared artifact set;
|
||||
each action carries that artifact's immutable key, checksum, size, and
|
||||
generation. Coherent legacy state remains on the isolated compatibility path;
|
||||
- remote-to-local mapping is traversal-safe;
|
||||
- actions are sorted by local relative path and then remote key;
|
||||
- force converts differing local targets from conflicts to downloads.
|
||||
- force converts differing eligible regular files from conflicts to downloads;
|
||||
directories and other non-regular targets remain conflicts.
|
||||
|
||||
Previous-cache files are planned separately through `previouscache.BuildPlan`
|
||||
when configured previous-session requirements exist.
|
||||
For a non-dry-run restore, planning/classification happens only after acquiring
|
||||
the session lock. Runner manifest/reuse checks acquire that same lock first.
|
||||
|
||||
Previous-cache readiness is resolved through `previouscache.Resolve` for restore,
|
||||
prepare, status, and validation. A committed source is selected only by its
|
||||
exact source identity; legacy fallback remains isolated and rejects ambiguity.
|
||||
|
||||
## Execution Contract
|
||||
|
||||
@@ -47,17 +55,32 @@ Execution order and safety:
|
||||
- `manifest.json` installs last;
|
||||
- downloads use sibling temp files plus atomic rename;
|
||||
- manifest replacement is validated before rename;
|
||||
- each committed object is verified against its declared checksum, size, and
|
||||
generation before installation;
|
||||
- a committed manifest already verified during discovery is retained for the
|
||||
matching restore action and revalidated before installation, avoiding a
|
||||
second body transfer;
|
||||
- failed installs do not roll back files already written in the same execution.
|
||||
- a durable `.restore-incomplete.json` marker is written before installation.
|
||||
It blocks runners until a restore retry completes all verified installs and
|
||||
the local manifest replacement, at which point it is removed.
|
||||
- restored manifest local references are rebased beneath the selected local
|
||||
session root. Unsafe relative references and producer-machine absolute paths
|
||||
outside the manifest's producer session root are rejected; producer-local
|
||||
spool/cache and cleanup locations are not restored as authority.
|
||||
|
||||
Audio restore path:
|
||||
|
||||
- uses `audio.MaterializeS3Audio`;
|
||||
- integrates spool and S3 audio cache paths;
|
||||
- supports cache-hit reuse without object redownload.
|
||||
- reuses cached audio only when its no-follow regular file, content digest, and
|
||||
identity sidecar all match the selected remote object version; otherwise it
|
||||
refreshes through the durable download path.
|
||||
|
||||
## Reporting Contract
|
||||
|
||||
- dry-run mode prints a summary and performs no local writes;
|
||||
- dry-run mode prints a summary, performs no durable session writes, and may
|
||||
read remote current-state or object-identity data to produce that summary;
|
||||
- execution mode persists the canonical restore report described in
|
||||
[Operations](../operations.md#restore-workflow);
|
||||
- report includes plan counts, per-action status, and execution failures.
|
||||
@@ -65,7 +88,13 @@ Audio restore path:
|
||||
## Invariants
|
||||
|
||||
- restore uses committed remote current state as authority;
|
||||
- `current/run_id.txt` is the remote publish commit marker;
|
||||
- one restore or status inspection observes the single pointer-selected commit
|
||||
loaded at discovery; later pointer changes cannot add objects or substitute a
|
||||
different run into its plan;
|
||||
- a verified `current/commit-pointer.json` and its selected immutable commit
|
||||
establish new-protocol remote commitment; coherent legacy
|
||||
`current/run_id.txt` plus `current/manifest.json` remains read-only migration
|
||||
support;
|
||||
- restore does not execute pipeline stages.
|
||||
|
||||
## Implementation And Tests
|
||||
|
||||
180
docs/internal/configuration.md
Normal file
180
docs/internal/configuration.md
Normal file
@@ -0,0 +1,180 @@
|
||||
# Configuration Internals
|
||||
|
||||
User-visible fields, defaults, and selection behavior belong in the
|
||||
[Configuration Reference](../config.md). This document describes the internal
|
||||
pipeline-loading boundary implemented by `internal/config`.
|
||||
|
||||
## Pipeline Loading
|
||||
|
||||
`LoadPipeline` assembles and validates a pipeline in this order:
|
||||
|
||||
1. Parse the root YAML into a presence-aware composition tree. The tree retains
|
||||
source names, full field paths, node kinds, declaration order, and explicit
|
||||
zero, false, empty-map, and empty-list values.
|
||||
2. Remove the root-only `composition` envelope and validate its explicit
|
||||
`imports`, `default_profile`, and named `profiles` declarations. A load
|
||||
option retains the difference between omitted and explicitly empty profile
|
||||
selection.
|
||||
3. Open each import relative to the root pipeline directory through the
|
||||
confined regular-file boundary. Imports must use a `.yml` or `.yaml`
|
||||
extension and cannot traverse, use symlinks, repeat a file, import the root,
|
||||
or contain another composition envelope.
|
||||
4. Resolve and structurally parse every declared profile overlay through the
|
||||
same confined regular-file boundary. Missing or malformed unselected
|
||||
overlays fail the load. Overlays cannot contain a composition envelope.
|
||||
5. Additively merge the root body and imports. Distinct map leaves compose;
|
||||
repeated scalar or list paths and node-kind disagreements are conflicts.
|
||||
6. Select exactly one declared profile from an explicit option or the default,
|
||||
then recursively merge its overlay. Overlay leaves replace base leaves,
|
||||
lists are atomic replacements, and null or kind changes fail.
|
||||
7. Emit deterministic canonical YAML and strictly decode it into
|
||||
`PipelineConfig`.
|
||||
8. Apply pipeline defaults once, resolve ordinary relative pipeline paths from
|
||||
the root pipeline file, and digest the normalized effective mapping.
|
||||
|
||||
This ordering preserves monolithic configuration behavior. Moving a field to
|
||||
an imported fragment changes its source ownership, not its path base, default,
|
||||
or schema semantics.
|
||||
|
||||
## Loaded Context Resolution
|
||||
|
||||
`LoadedPipelineCampaign` carries one already composed pipeline and its selected
|
||||
campaign into session resolution. `LoadSessionWithPipelineCampaignOptions`
|
||||
loads a local session against that context, while
|
||||
`ResolveLoadedPipelineCampaign` also accepts an already loaded remote session
|
||||
or no session while a caller retrieves one. Compatibility loaders route through
|
||||
these functions after their initial pipeline and campaign reads.
|
||||
|
||||
Application commands own pipeline and campaign discovery, campaign-file versus
|
||||
registry selection, and the corresponding mutual-exclusion rules. Once they
|
||||
have a `LoadedPipelineCampaign`, local session discovery and remote-session
|
||||
download retain that exact pipeline object and its private provenance. Removing
|
||||
a temporary downloaded session file therefore cannot invalidate the resolved
|
||||
pipeline or campaign context.
|
||||
|
||||
The application also has a separate read-only inspection resolver for `config
|
||||
validate`, `config show`, and `config sources`. It uses the same production root/profile and
|
||||
campaign selection functions, but never routes through session discovery,
|
||||
remote-session download, secret loading, adapter composition, workspace
|
||||
initialization, manifest access, or cleanup. A pipeline with retained artifact
|
||||
family declarations must resolve its selected campaign before ordinary pipeline
|
||||
validation, which expands its canonical-party members and generated publish
|
||||
rules. A pipeline without those declarations may be validated by itself.
|
||||
|
||||
`MarshalEffectivePipeline` is the configuration-owned projection for `config
|
||||
show`. It serializes the typed, defaulted effective mapping through the
|
||||
deterministic composition renderer, then removes resolution-only artifact
|
||||
family declarations. The result contains no composition envelope or private
|
||||
provenance fields and has one trailing newline; commands do not marshal runtime
|
||||
objects directly.
|
||||
|
||||
`EffectivePipelineSources` and `EffectiveCampaignSources` provide the separate
|
||||
safe provenance projection for `config sources`. Pipeline ownership begins with
|
||||
the complete logical field paths retained during composition and classifies
|
||||
each contributor as root, import, profile, or centralized default. The
|
||||
projection replaces generated concrete member paths with paired family and
|
||||
canonical-party records, and does the same for generated publish rules.
|
||||
Campaign records identify campaign-owned fields and party inputs; canonical
|
||||
derived players point to the party source, while legacy players retain a
|
||||
dedicated legacy-player role. The application command only joins these sorted
|
||||
records with selection metadata and never reparses configuration files.
|
||||
|
||||
`config diff` uses a paired profile loader that parses the root, imports, and
|
||||
declared overlays once, then clones the additive base before independently
|
||||
selecting, decoding, defaulting, and finalizing each profile. When campaign
|
||||
resolution is needed, the command loads one selected campaign and party and
|
||||
expands both effective pipelines from that same party value. The configuration
|
||||
owner projects each normalized effective mapping into sorted logical paths;
|
||||
mapping leaves are compared individually while sequence values remain atomic.
|
||||
Values are compact deterministic JSON representations for command output, not
|
||||
raw YAML fragments, ownership records, or secret material. A differing digest
|
||||
with no projected difference is treated as an internal consistency error.
|
||||
|
||||
Campaign context construction also reads and classifies the campaign-owned
|
||||
party source through `ParseParty`. A canonical party retains its raw bytes and
|
||||
normalized roster in runtime-only `ResolvedParty` provenance, while a legacy
|
||||
party remains opaque. Canonical resolution creates a virtual
|
||||
`derived_from_party` players input and rejects competing campaign or session
|
||||
players files and session party overrides. The compact legacy compatibility
|
||||
path resolves the effective campaign/session party and players files together.
|
||||
|
||||
## Canonical Party Domain
|
||||
|
||||
`ParseParty` is the package-owned boundary for classifying a party source.
|
||||
When a top-level `schema_version` is present, it strictly validates the
|
||||
`narratio.party.v1` contract into ordered character domain values. The
|
||||
canonical value retains a separate exact byte copy of its source so consumers
|
||||
can materialize the authored party document without reserializing it. Its
|
||||
`PlayersYAML` method deterministically derives the versioned players-only
|
||||
projection.
|
||||
|
||||
An unversioned source is classified by the small legacy compatibility boundary
|
||||
in `party_legacy.go`; it deliberately exposes no parsed roster information.
|
||||
That boundary exists solely to isolate removable compatibility behavior from
|
||||
the canonical parser.
|
||||
|
||||
## Diagnostics And Runtime Metadata
|
||||
|
||||
Syntax, duplicate-key, composition, conflict, and schema failures include the
|
||||
relevant source name and full field path. Additive conflicts report every
|
||||
claiming source so operators can repair the split without repeatedly
|
||||
rediscovering additional conflicts.
|
||||
|
||||
The loaded pipeline retains private runtime metadata for the absolute root
|
||||
path, ordered imports, selected profile name and selection source, selected
|
||||
overlay, contributing sources, effective digest, and leaf ownership. Base
|
||||
leaves retain their root/import owners, replaced leaves belong to the selected
|
||||
overlay, and centrally supplied values use the synthetic `default` owner. This
|
||||
metadata does not participate in YAML decoding or alter the public
|
||||
configuration model.
|
||||
|
||||
The effective digest is SHA-256 over deterministic canonical YAML produced from
|
||||
the defaulted `PipelineConfig`. Runtime Notarius paths remain absolute for
|
||||
execution, but the digest substitutes their normalized logical values captured
|
||||
before root-relative resolution, so relocating an equivalent configuration
|
||||
bundle does not change provenance. Because composition and resolution metadata
|
||||
are private, the digest excludes source layout, profile name, and ownership.
|
||||
Configuration stores environment variable names rather than resolving raw
|
||||
credentials, so raw secret values are neither loaded nor hashed.
|
||||
`recomputePipelineEffectiveDigest` is the single package-owned refresh point
|
||||
for later runtime expansion.
|
||||
|
||||
## Test Surfaces
|
||||
|
||||
`composition_test.go` protects the presence and merge algebra independently of
|
||||
the public schema. `pipeline_composition_test.go` exercises explicit imports,
|
||||
confinement, conflicts, strict decoding, metadata, and root-relative path
|
||||
behavior through `LoadPipeline`. `pipeline_profiles_test.go` covers selection,
|
||||
all-overlay validation, overlay behavior, provenance, option propagation, and
|
||||
effective-digest stability. Application configuration-loader tests protect the
|
||||
single-read boundary by changing the pipeline file after its initial load and
|
||||
confirming local session resolution retains the original pipeline. Other
|
||||
configuration tests continue to protect defaults and validation after assembly.
|
||||
`party_test.go` protects the versioned party schema, domain invariants, and
|
||||
deterministic players projection without involving campaign or runtime wiring.
|
||||
`party_resolution_test.go` protects campaign-owned party loading, canonical
|
||||
input restrictions, legacy overrides, source provenance, and virtual players
|
||||
input selection.
|
||||
|
||||
## Artifact Family Resolution
|
||||
|
||||
Pipeline loading retains `scriptorium.artifact_families` as a resolution-only
|
||||
declaration. Once campaign party resolution establishes a canonical roster,
|
||||
configuration expands families in sorted family-key and character-ID order
|
||||
into ordinary `ScriptoriumArtifactConfig` values. The expansion owns the narrow
|
||||
`{character_id}` output substitution, closed member-variable selectors, key and
|
||||
output collision checks, and the runtime-only family-origin catalog. It then
|
||||
removes family declarations from `ScriptoriumConfig`, runs ordinary Scriptorium
|
||||
validation, and refreshes the effective pipeline digest. Stages and adapters
|
||||
therefore receive only concrete artifact maps.
|
||||
|
||||
The catalog retains sorted family member keys plus family/character/source
|
||||
origins and the typed dependency/publish declarations for their later owners.
|
||||
`member_dependencies` add corresponding ordinary concrete dependencies, while
|
||||
the family-only `narratio.member_artifact.<family>` input form is rewritten to
|
||||
the matching ordinary configured-artifact source. The catalog records those
|
||||
resolved dependency and input identities with their declaring family and party
|
||||
member. No member-artifact source is registered as a runtime policy source.
|
||||
An enabled family publish declaration expands to ordinary configured-artifact
|
||||
publish rules before the existing publish and lock validators run. Runtime
|
||||
publication consequently receives no family wildcard or special matcher.
|
||||
71
docs/internal/fileops.md
Normal file
71
docs/internal/fileops.md
Normal file
@@ -0,0 +1,71 @@
|
||||
# Internal: File Operations
|
||||
|
||||
`internal/fileops` owns the narrow mechanics for durable replacement of one
|
||||
byte file. Callers keep ownership of serialization, validation, cancellation,
|
||||
and destination-directory policy.
|
||||
|
||||
## Destination Confinement
|
||||
|
||||
Before it creates, replaces, or installs a destination file, `fileops` opens
|
||||
each ancestor from the filesystem root and rejects symbolic links or components
|
||||
that change during traversal. The resulting parent-directory handle is retained
|
||||
for sibling temporary-file creation and rename, so a later pathname swap cannot
|
||||
redirect the replacement. Existing destination symlinks are replaced as leaf
|
||||
entries; their targets are never followed.
|
||||
|
||||
Remote object acquisition uses a writer supplied by the storage owner. The
|
||||
writer receives a `fileops`-owned, already-open sibling temporary file rather
|
||||
than a mutable destination path. Callers still own remote object selection,
|
||||
validation, conflict handling, and final mode.
|
||||
|
||||
Directory promotion keeps the verified destination parent open while it creates
|
||||
the temporary tree, copies regular source entries, and performs the platform
|
||||
no-replace rename. Platforms without a verified handle-relative atomic
|
||||
no-replace primitive reject promotion before writing a temporary tree.
|
||||
|
||||
## Cleanup Contract
|
||||
|
||||
`RemoveAllUnderRoot` accepts an explicit root and a proper descendant. It opens
|
||||
the root and each target ancestor without following symlinks, then removes the
|
||||
tree through those directory handles. It rejects root deletion and any symlink
|
||||
encountered in the target path or tree; repeated removal of a missing target is
|
||||
successful. Command and post-publish policy remains owned by `internal/app`.
|
||||
|
||||
## Confined Reads
|
||||
|
||||
`ReadRegularFileUnderRoot` is the no-follow, bounded read primitive for a
|
||||
caller-selected root and relative file path; `ReadRegularFile` is its
|
||||
path-based convenience wrapper. They verify every ancestor through directory
|
||||
handles and admit only a stable regular-file handle. Callers enforce their own
|
||||
byte limits and access policy. Credential mode policy and environment
|
||||
precedence remain owned by `internal/app`.
|
||||
|
||||
## Replacement Contract
|
||||
|
||||
`ReplaceFileAtomic` requires an existing destination directory. It creates a
|
||||
sibling temporary file, writes the complete byte sequence, applies the
|
||||
caller-supplied mode, syncs and closes the file, runs an optional pre-rename
|
||||
check, replaces the destination with a rename, then syncs the containing
|
||||
directory.
|
||||
|
||||
The pre-rename check is the last point at which a caller can cancel without
|
||||
installing a new destination. A failure before the rename leaves the old
|
||||
destination unchanged and removes the temporary file; any cleanup failure is
|
||||
returned alongside the primary failure. A failure after the rename may leave
|
||||
the new file visible, but it is not reported as crash-durable.
|
||||
|
||||
Replacement follows the operating system's same-filesystem rename semantics.
|
||||
If a platform cannot replace an existing destination, the operation returns an
|
||||
error and never removes the old file as an emulation step.
|
||||
|
||||
## Directory-Sync Support
|
||||
|
||||
Linux and macOS attempt to sync the destination directory. Windows opens the
|
||||
directory with backup semantics and flushes its buffers. If either operation
|
||||
is unavailable for the platform, directory handle, or filesystem,
|
||||
`ErrDirectorySyncUnsupported` is returned. Narratio does not treat that result
|
||||
as successful crash-durable replacement.
|
||||
|
||||
`WriteFileAtomic`, copy helpers, and downloaded temporary-file installation
|
||||
retain their compatibility behavior of creating the destination parent with
|
||||
the repository's workspace permissions before using this contract.
|
||||
@@ -16,6 +16,20 @@ Explain the session-progress and invocation-audit models implemented by
|
||||
- `inputs` records
|
||||
- durable `artifacts` records
|
||||
- per-stage `stages` map
|
||||
- an optional `post_publish_cleanup` obligation, which binds a committed run,
|
||||
remote commit identity, and each exact root-confined local target to its
|
||||
completion evidence
|
||||
|
||||
Session, campaign, and run identities in local and downloaded manifests must be
|
||||
portable opaque segments. Unsafe legacy identities are rejected with migration
|
||||
guidance rather than being normalized into a different workspace or remote
|
||||
namespace.
|
||||
|
||||
Prepare records independent `party` and `players` input checksums. In canonical
|
||||
party mode, the party record retains its campaign source identity while the
|
||||
players record uses `derived_from_party`; raw roster content is never embedded
|
||||
in manifest metadata. Both records remain the durable authority for consumers
|
||||
of their prepared input source IDs.
|
||||
|
||||
The model admits these stage states:
|
||||
|
||||
@@ -27,73 +41,275 @@ The model admits these stage states:
|
||||
- `stale`
|
||||
- `interrupted`
|
||||
|
||||
### Analyze-owned artifact state
|
||||
|
||||
The `analyze` stage record may carry `analyze_state_version: 1` and an
|
||||
`analyze_artifacts` map keyed by normalized configured artifact key. The
|
||||
version is the authority marker: version 1 with no entries is a valid evaluated
|
||||
empty set, while an absent version is legacy aggregate-only state and provides
|
||||
no current configured-artifact evidence.
|
||||
|
||||
Each analyze artifact record has one disposition:
|
||||
|
||||
- `current`: the configured artifact is available and carries a versioned
|
||||
fingerprint plus a complete output record and separate output size;
|
||||
- `stale`: the recorded semantic identity is no longer current;
|
||||
- `missing`: no validated current result exists;
|
||||
- `failed`: the attempted work failed and carries a bounded diagnostic; or
|
||||
- `unselected`: the artifact was intentionally outside the evaluated set.
|
||||
|
||||
Records bind their normalized key and dependencies, fingerprint contract when
|
||||
evaluated, canonical session-relative output identity when current, producing
|
||||
Narratio run, update time, and bounded non-secret Scriptorium provenance and
|
||||
diagnostic paths. A current output includes its configured source ID, contract,
|
||||
checksum, and positive byte size. Non-current records cannot carry an output,
|
||||
so an older file is not advertised through stale, missing, failed, or
|
||||
unselected state.
|
||||
|
||||
Family-produced records additionally retain optional `family` and
|
||||
`character_id` provenance supplied by configuration resolution. These fields
|
||||
do not replace the concrete configured key or infer family membership from a
|
||||
name, so older records without them remain valid.
|
||||
|
||||
The session-stage collection is the reconciled authority across invocations.
|
||||
The corresponding collection on an invocation's `analyze` stage record is an
|
||||
audit of only the artifacts evaluated or attempted by that run. These records
|
||||
remain analyze-owned data inside the fixed stage; they are not dynamic stages
|
||||
or generic subtasks.
|
||||
|
||||
The stage result contract has one analyze-specific projection boundary. On
|
||||
success, the runner validates and deep-copies the complete reconciled session
|
||||
collection and the invocation subset. Aggregate session outputs are rebuilt in
|
||||
configured-key order from current session records only; invocation outputs are
|
||||
limited to current records produced by that invocation's run ID. Ordinary
|
||||
stage outputs cannot accompany this projection, so there is one source of
|
||||
artifact authority.
|
||||
|
||||
Successful incremental execution replaces only evaluated artifact records and
|
||||
preserves valid unrelated current records. Rebuilt outputs are compared by
|
||||
bytes and contract: an unchanged identity permits an unselected dependent with
|
||||
the same recomputed fingerprint to remain current, while a changed identity
|
||||
removes output authority from every unselected transitive dependent by marking
|
||||
it stale. A partial analyze invocation can therefore succeed while unrelated
|
||||
configured records remain stale. Existing canonical files never create current
|
||||
records without validated execution and projection.
|
||||
|
||||
Aggregate analyze status is deliberately coarser than this collection. Resume
|
||||
validation may skip a succeeded aggregate record when the selected artifact
|
||||
closure is current even if unrelated records are stale. Conversely, a stale
|
||||
aggregate record may cross the ordinary runner boundary and perform zero
|
||||
Scriptorium calls when reconciliation proves every selected artifact current;
|
||||
the successful projection then restores the aggregate status.
|
||||
|
||||
Analyze may return a projection together with an error. That restricted result
|
||||
cannot carry ordinary outputs, skip state, aggregate logs, generated configs,
|
||||
or metadata. The runner persists only the validated per-artifact collections,
|
||||
then marks the aggregate analyze and run state failed and invalidates delivery
|
||||
dependents conservatively. Unrelated current records survive because the
|
||||
session projection is complete. A malformed projection is not applied, and a
|
||||
failed session projection save restores the prior per-artifact authority before
|
||||
terminal failure persistence.
|
||||
|
||||
The incremental executor constructs this restricted projection at each
|
||||
scheduled artifact boundary. The active record is failed without output,
|
||||
current transitive dependents are stale, unrelated current records survive, and
|
||||
only earlier validated and materialized completions remain current in the
|
||||
invocation subset. Session failure state is persisted before invocation failure
|
||||
state. If either terminal save fails, its persistence error is joined with the
|
||||
original adapter, validation, or filesystem cause; a failed projection save
|
||||
does not turn incidental canonical bytes into manifest authority.
|
||||
|
||||
## Run Manifest
|
||||
|
||||
`manifest.RunManifest` is created for each invocation and records:
|
||||
|
||||
- invocation identity and `force` flag
|
||||
- the selected profile (when any) and secret-free effective configuration digest
|
||||
- requested stages
|
||||
- per-stage action (`run` or `skip`)
|
||||
- per-stage status
|
||||
- overall run status (`running`, `succeeded`, `failed`)
|
||||
|
||||
## Remote Commit Manifest
|
||||
|
||||
`artifacts.RemoteCommitManifest` is a separate, versioned remote snapshot
|
||||
contract. It is not a serialized session manifest and contains no local
|
||||
post-publication assertion such as `current_pointer_written`. A remote commit
|
||||
identifies one campaign, session, and run and declares its immutable artifact
|
||||
set. Each artifact has a typed source, immutable destination key, SHA-256
|
||||
checksum, size, and storage generation.
|
||||
|
||||
`current/commit-pointer.json` is the sole mutable selector for the new
|
||||
contract. It identifies exactly one run-scoped `runs/{run_id}/commit.json` and
|
||||
binds that object by checksum, size, and generation. Readers strictly reject
|
||||
unknown fields, version mismatches, pointer/commit identity mismatches, and
|
||||
objects that do not match their declaration.
|
||||
|
||||
The reader retains a temporary, clearly isolated compatibility path for a
|
||||
coherent legacy `current/manifest.json` plus `current/run_id.txt` pair. That
|
||||
path is removable after migration and is never used to write new state.
|
||||
|
||||
## Persistence Semantics
|
||||
|
||||
`manifest.LocalStore`:
|
||||
|
||||
- validates loaded documents;
|
||||
- normalizes missing maps/stage records;
|
||||
- writes atomically via temp file + rename;
|
||||
- writes through a sibling temporary file, syncing the completed file and
|
||||
destination directory after atomic replacement;
|
||||
- updates `updated_at` on save.
|
||||
|
||||
If the operating system or filesystem cannot sync a directory, save returns an
|
||||
explicit error instead of claiming crash-durable replacement. A returned error
|
||||
after the rename can therefore leave the new manifest visible but not confirmed
|
||||
durable; callers must reload it before retrying.
|
||||
|
||||
## Execution Semantics
|
||||
|
||||
The application runner marks an executing stage running and then succeeded or
|
||||
failed in both manifests, persisting each transition. On success it records
|
||||
outputs, logs, generated configuration references, and metadata. Artifact
|
||||
outputs, logs, generated configuration references, metadata, and—when the
|
||||
stage implements the optional contract—a versioned semantic-configuration
|
||||
fingerprint. Artifact
|
||||
records may include optional contract and external provenance objects; old
|
||||
manifests remain compatible when those fields are absent. A successful forced
|
||||
rerun marks only succeeded downstream session-stage records stale.
|
||||
rerun marks only succeeded transitive dependent session-stage records stale.
|
||||
The application owns a fixed dependency relation distinct from execution order;
|
||||
dependents are returned in canonical order. Render and extract therefore never
|
||||
stale one another, while either can stale analyze, publish, and notify.
|
||||
|
||||
Starting an execution clears the current session-stage record's prior outputs,
|
||||
logs, generated configuration references, and metadata. Failed and skipped
|
||||
logs, generated configuration references, metadata, and semantic fingerprint.
|
||||
Failed and skipped
|
||||
transitions enforce the same clearing rule directly, while success repopulates
|
||||
only fields returned by the new result. Marking a record stale does not clear
|
||||
those details because resume validation and diagnosis may still require them
|
||||
before execution begins. Invocation run manifests remain immutable audit
|
||||
records of their own outcomes.
|
||||
|
||||
Aggregate lifecycle clearing deliberately preserves the analyze-owned
|
||||
per-artifact collection. This lets later reconciliation replace only evaluated
|
||||
entries without erasing unrelated current results. Other stages retain their
|
||||
existing aggregate-only lifecycle behavior and are forbidden from carrying the
|
||||
analyze-specific fields.
|
||||
|
||||
A stage may explicitly return a skipped disposition and stable reason. The
|
||||
runner persists that outcome in both manifests, clears older outputs for the
|
||||
session-stage record along with older logs, generated configuration references,
|
||||
and metadata, then applies any bounded details from the current skip and
|
||||
continues. This self-skip is distinct from deciding not to execute an
|
||||
already-succeeded stage and is reconsidered on later runs. Skipped results
|
||||
cannot contain outputs.
|
||||
cannot contain outputs. An intentional self-skip records the current semantic
|
||||
fingerprint because it is a completed, reusable stage result; failed or
|
||||
interrupted work never promotes one.
|
||||
|
||||
When an already-succeeded stage is skipped, the invocation run manifest records
|
||||
the `skip` action and reason. The session manifest deliberately retains its
|
||||
existing succeeded record because it remains the cross-invocation progress
|
||||
authority. Stages with a resume validator, currently extraction, may reject an
|
||||
otherwise eligible skip when the recorded durable result is obsolete; the
|
||||
runner marks it stale and executes it.
|
||||
authority. If a stage supplies semantic configuration evidence, reuse first
|
||||
requires the persisted positive schema version and lowercase SHA-256 digest to
|
||||
match the current resolved stage semantics. Missing legacy evidence, malformed
|
||||
evidence, or a mismatch makes the stage and its fixed transitive dependents
|
||||
stale. The invocation skip copies the matched fingerprint for provenance but
|
||||
does not rewrite session authority. The existing stage-specific resume
|
||||
validator runs only after this semantic check succeeds; both checks are
|
||||
required. Extraction and analyze have resume validators and may reject an
|
||||
otherwise eligible skip when their selected durable evidence is obsolete; the
|
||||
runner marks the aggregate record stale and executes it. Analyze's validator
|
||||
can still accept a partial selection when only unrelated artifact records are
|
||||
stale.
|
||||
|
||||
Implemented reuse coverage is deliberately split between aggregate semantic
|
||||
evidence and focused durable validators:
|
||||
|
||||
| Work | Reuse authority | Focused owners |
|
||||
| --- | --- | --- |
|
||||
| prepare | aggregate semantic fingerprint | [prepare](stage-prepare.md) |
|
||||
| transcribe | aggregate semantic fingerprint | [transcribe](stage-transcribe.md), [WhisperX](../integrations/whisperx.md) |
|
||||
| merge | aggregate semantic fingerprint | [merge](stage-merge.md), [Seriatim](../integrations/seriatim.md) |
|
||||
| polish | aggregate semantic fingerprint | [polish](stage-polish.md), [Audita](../integrations/audita.md) |
|
||||
| normalize | aggregate semantic fingerprint | [normalize](stage-normalize.md), [Seriatim](../integrations/seriatim.md) |
|
||||
| trim | aggregate semantic fingerprint | [trim](stage-trim.md), [Scriptorium](../integrations/scriptorium.md), [Seriatim](../integrations/seriatim.md) |
|
||||
| render | aggregate semantic fingerprint | [render](stage-render.md), [Seriatim](../integrations/seriatim.md) |
|
||||
| extract | aggregate semantic fingerprint plus reference/output validator | [extract](stage-extract.md), [Notarius](../integrations/notarius.md) |
|
||||
| analyze artifacts | per-artifact fingerprint, reconciliation, and output validator | [analyze](stage-analyze.md), [Scriptorium](../integrations/scriptorium.md) |
|
||||
| publish | aggregate semantic fingerprint plus immediate lock/commit checks | [publish](stage-publish.md), [storage adapter](adapters.md) |
|
||||
| notify | aggregate delivery-mode fingerprint | [pipeline overview](overview.md), [configuration](../config.md#notifications) |
|
||||
|
||||
These contracts record resolved choices Narratio can observe, not operational
|
||||
runner tuning. External model, module, prompt, profile, and configuration-file
|
||||
contents that a tool privately loads remain outside the contract when their
|
||||
configured identifier is unchanged; operators must force the affected work
|
||||
after such a private content change.
|
||||
|
||||
Session manifest is the authoritative stage-progress ledger across invocations.
|
||||
Run manifest is invocation-scoped audit state.
|
||||
|
||||
Both manifests retain the most recently resolved invocation's bounded
|
||||
configuration provenance. It identifies the selected profile name and source
|
||||
(`default` or `cli`) plus the effective configuration digest, but never a raw
|
||||
secret or profile content. This provenance is informational: it does not
|
||||
participate in stage resume or cache decisions. A profile change therefore
|
||||
invalidates only stages whose semantic configuration changed. When a private
|
||||
external-tool model, module, prompt, or profile changes behind an unchanged
|
||||
configured identifier, use `--force` for the affected work.
|
||||
|
||||
`session plan` computes the same current fingerprint and applies the same
|
||||
comparison and invalidation rules to a cloned manifest. It predicts the runner
|
||||
decision without persisting session or invocation state. The shared helper
|
||||
hashes deterministic JSON from stage-owned typed structs; stage providers must
|
||||
exclude secrets, complete effective-configuration dumps, and operational
|
||||
values that cannot affect canonical results. Concrete coverage is owned by the
|
||||
focused stage and integration documents linked above.
|
||||
|
||||
Before an explicitly bounded execution starts after `prepare`, the application
|
||||
reads the session manifest and accepts only `succeeded` or `skipped` for every
|
||||
excluded canonical prefix stage. The first other status or absent record fails
|
||||
the request before layout mutation, adapter initialization, session-manifest
|
||||
writes, or run-manifest creation. Excluded prefix records are not passed to
|
||||
resume validators. Records after the selected end are not prerequisites and
|
||||
may be made stale by selected work without being scheduled.
|
||||
|
||||
After a publish commits remotely, any configured local cleanup is first recorded
|
||||
as a session-manifest obligation before deletion begins. Each target becomes
|
||||
complete only after its confined deletion (or safe absence check) and a
|
||||
successful manifest save. An incomplete obligation is retried when publish
|
||||
executes again and retains the committed run and remote identity that authorized
|
||||
it; an invocation that does not execute publish does not perform cleanup.
|
||||
|
||||
Each invocation derives campaign, session, run, local-path, and remote-prefix
|
||||
metadata from the validated resolved configuration as one projection. A persisted
|
||||
session manifest must agree on campaign and session identity before execution;
|
||||
the current projection is refreshed for every invocation while stage progress,
|
||||
inputs, and durable artifacts remain session history.
|
||||
|
||||
For handled failures after an invocation record is created, the runner records
|
||||
the failure on the session ledger and persists it before persisting the failed
|
||||
run audit record. This preserves the resume authority while making a partial
|
||||
persistence disagreement visible. Abrupt process death remains an accepted case
|
||||
where a durable running record can require operator interpretation.
|
||||
|
||||
## Invariants
|
||||
|
||||
- stage resume/skip decisions are session-manifest driven.
|
||||
- semantic fingerprint comparison precedes stage-specific resume validation.
|
||||
- only successful and intentional-skipped results promote current semantic
|
||||
evidence; invocation reuse copies evidence without replacing session state.
|
||||
- running, failed, and self-skipped stages do not retain result payloads from
|
||||
an earlier success.
|
||||
- stale stages retain prior details until replacement execution starts.
|
||||
- force reruns stale downstream succeeded stages.
|
||||
- force reruns stale succeeded stages in the fixed dependency relation.
|
||||
- run manifest does not replace session manifest as progress authority.
|
||||
- remote commitment is established by a verified current pointer and remote
|
||||
commit relationship, never by a mutable session-manifest boolean.
|
||||
|
||||
## Implementation And Tests
|
||||
|
||||
- Models and transitions: `internal/manifest/manifest.go`,
|
||||
`internal/manifest/run_manifest.go`
|
||||
- Remote commit model and readers: `internal/artifacts/remote_commit.go`,
|
||||
`internal/artifacts/current_state_commit.go`,
|
||||
`internal/artifacts/current_state_legacy.go`
|
||||
- Persistence and validation: `internal/manifest/store.go`
|
||||
- Package tests: `internal/manifest/*_test.go`
|
||||
- Assembled execution behavior: `internal/app/runner_test.go`,
|
||||
|
||||
@@ -28,14 +28,14 @@ progress and artifact services resolve durable inputs and outputs.
|
||||
| --- | --- | --- |
|
||||
| Executable | `cmd/narratio` | Process entry, standard stream wiring, argument handoff, and exit status. |
|
||||
| Application orchestration | `internal/app` | Command dispatch, configuration selection, secret-file environment loading, production composition, session locking, planning, execution, restore, cleanup gates, and user-facing reporting. |
|
||||
| Configuration | `internal/config` | Strict YAML loading, discovery, defaults, normalization, session templating, and validation. |
|
||||
| Configuration | [`internal/config`](configuration.md) | Presence-aware root/import/profile composition, canonical party and family expansion, strict YAML loading, defaults, normalization, session templating, and validation. |
|
||||
| Pipeline stages | `internal/stage` | Canonical stage registry, shared stage contract, execution dependencies, and implemented stage behavior. |
|
||||
| External boundaries | `internal/adapters`, `internal/audio` | WhisperX HTTP, downstream subprocesses, notification, object storage, and S3 audio materialization behind Narratio contracts. |
|
||||
| Manifests | `internal/manifest` | Durable session progress, invocation audit state, stage transitions, validation, and atomic persistence. |
|
||||
| Artifacts and paths | `internal/artifacts`, `internal/pathsafe` | Artifact identities and resolution, local and remote path/key models, current-state discovery, and confined relative destinations. |
|
||||
| Previous-session cache | `internal/previouscache` | Deterministic planning and materialization requirements for configured previous-session inputs. |
|
||||
| Artifact policy | `internal/artifactpolicy` | Source and destination policy, configured artifact identity validation, and publish destination safety. |
|
||||
| Shared models and file operations | `internal/artifactmodel`, `internal/contracts`, `internal/fileops` | Transcript and artifact data contracts plus narrow atomic filesystem helpers. |
|
||||
| Shared models and file operations | `internal/artifactmodel`, `internal/contracts`, [`internal/fileops`](fileops.md) | Transcript and artifact data contracts plus durable single-file replacement helpers; unsupported directory syncing is reported explicitly. |
|
||||
| Logging | `internal/logging` | Application logger construction and shared structured logging behavior. |
|
||||
|
||||
The application boundary composes concrete implementations. Stages depend on
|
||||
@@ -43,6 +43,16 @@ Narratio-level contracts; external transport and SDK details remain in
|
||||
adapters. The normative rules for these relationships remain in
|
||||
[Architecture](../policy/architecture.md).
|
||||
|
||||
Pipeline execution and `session plan` share the same inclusive contiguous-range
|
||||
model. Planning clones session state and applies selected-stage transitions and
|
||||
resume validation in memory; it does not create invocation state or initialize
|
||||
stage-execution adapters. Command configuration loading can still retrieve a
|
||||
missing session file through configured remote storage. It retains the initially
|
||||
composed pipeline and selected campaign while resolving either a local or
|
||||
downloaded remote session, so one invocation cannot mix pipeline revisions.
|
||||
Analyze planning additionally exposes the artifact closure's targets,
|
||||
prerequisite rebuilds, execution order, and current reuse.
|
||||
|
||||
## Pipeline Stage Set
|
||||
|
||||
The implemented canonical order is:
|
||||
@@ -53,20 +63,30 @@ The implemented canonical order is:
|
||||
4. [`polish`](stage-polish.md)
|
||||
5. [`normalize`](stage-normalize.md)
|
||||
6. [`trim`](stage-trim.md)
|
||||
7. [`extract`](stage-extract.md)
|
||||
8. [`render`](stage-render.md)
|
||||
7. [`render`](stage-render.md)
|
||||
8. [`extract`](stage-extract.md)
|
||||
9. [`analyze`](stage-analyze.md)
|
||||
10. [`publish`](stage-publish.md)
|
||||
11. `notify` (placeholder)
|
||||
11. `notify` (no-op)
|
||||
|
||||
`notify` currently has optional notifier call behavior and no persisted pipeline
|
||||
outputs; its default collaborator is a no-op sender. The focused stage
|
||||
`notify` currently has no persisted pipeline outputs and uses the explicit
|
||||
`noop` notification mode. Its versioned semantic evidence records that delivery
|
||||
mode and excludes adapter credentials and response data. The focused stage
|
||||
documents own implementation mechanics. The
|
||||
[CLI](../cli.md) and [Operations](../operations.md) own user-visible invocation
|
||||
and execution semantics.
|
||||
|
||||
Execution order and invalidation are separate application contracts. The stage
|
||||
registry owns the flat execution sequence. The application orchestration owner
|
||||
uses a fixed, validated dependency relation to find transitive dependents in
|
||||
canonical order. In particular, `render` and `extract` are sibling consumers of
|
||||
trimmed transcript state: neither invalidates the other, while either can stale
|
||||
`analyze`, `publish`, and `notify`.
|
||||
|
||||
## Focused Documentation
|
||||
|
||||
- [Configuration Internals](configuration.md): pipeline composition, import
|
||||
confinement, field ownership, decoding, and root-relative path semantics.
|
||||
- [Adapter Internals](adapters.md): external adapter boundaries, composition,
|
||||
failure behavior, and test surfaces.
|
||||
- [Artifact Internals](artifacts.md): source identities, runtime catalog,
|
||||
@@ -84,8 +104,8 @@ and execution semantics.
|
||||
- [`polish`](stage-polish.md)
|
||||
- [`normalize`](stage-normalize.md)
|
||||
- [`trim`](stage-trim.md)
|
||||
- [`extract`](stage-extract.md)
|
||||
- [`render`](stage-render.md)
|
||||
- [`extract`](stage-extract.md)
|
||||
- [`analyze`](stage-analyze.md)
|
||||
- [`publish`](stage-publish.md)
|
||||
|
||||
|
||||
@@ -2,34 +2,127 @@
|
||||
|
||||
## Purpose
|
||||
|
||||
Execute selected configured Scriptorium artifacts in dependency order and materialize outputs.
|
||||
Reconcile configured Scriptorium artifacts, execute only required work in
|
||||
dependency order, and safely materialize validated outputs.
|
||||
|
||||
## Inputs
|
||||
|
||||
- configured artifacts from `pipeline.scriptorium.artifacts`
|
||||
- ordinary configured artifacts from `pipeline.scriptorium.artifacts`; canonical
|
||||
party artifact families have already expanded into this map during
|
||||
configuration resolution, including corresponding member dependencies and
|
||||
rewritten member-artifact input sources
|
||||
- optional selected artifact keys supplied through the stage environment
|
||||
- built-in/configured/previous-session source references in artifact inputs
|
||||
- built-in, configured, extraction, and previous-session source references in
|
||||
artifact inputs
|
||||
|
||||
Supported source families:
|
||||
- built-ins: `narratio.transcript.*`, `narratio.bounds.session`
|
||||
- prepared stable inputs: `narratio.input.players`, `narratio.input.party`,
|
||||
`narratio.input.glossary`
|
||||
`narratio.input.glossary`, `narratio.input.spell_catalog`
|
||||
- configured artifacts: `narratio.artifact.<key>`
|
||||
- extraction lanes: `narratio.extraction.<key>`
|
||||
- previous-session cache: `narratio.previous_session.artifact.<key>`
|
||||
|
||||
## Outputs
|
||||
|
||||
- one materialized output per executed configured artifact (`output_path`)
|
||||
- one current per-artifact manifest record per validated materialized output
|
||||
- stage metadata describing selected/generated/reused artifacts
|
||||
|
||||
## Key Behavior
|
||||
|
||||
- skips with metadata when Scriptorium config is missing or no executable artifacts remain.
|
||||
- builds runtime artifact catalog (built-ins + configured artifacts).
|
||||
- marks non-executable configured artifacts as reusable when output files already exist.
|
||||
- when `pipeline.scriptorium` is absent or no configured artifact is
|
||||
executable, completes successfully with no outputs and records explanatory
|
||||
metadata. This is not an explicit self-skip: both manifests record success,
|
||||
satisfy publish's prerequisite, and an ordinary later run reuses the result
|
||||
while the effective set remains empty. Enabling or selecting an artifact
|
||||
later makes missing versioned evidence non-resumable and schedules it without
|
||||
requiring force.
|
||||
- builds a runtime artifact catalog containing built-ins, configured artifacts,
|
||||
and configured extraction lanes. Extraction availability is hydrated only
|
||||
from compatible successful extraction evidence.
|
||||
- uses enabled configured artifacts by default. An explicit `--artifacts`
|
||||
selection is a one-invocation override that makes exactly the named
|
||||
configured artifacts explicit targets even when disabled. The work planner
|
||||
adds required configured prerequisites, reuses current ones, and schedules
|
||||
stale, missing, or otherwise non-current prerequisites before dependents.
|
||||
- makes a non-executable configured artifact reusable only when its current
|
||||
manifest record and durable output pass the configured-artifact evidence
|
||||
contract; an incidental or stale canonical file is unavailable.
|
||||
- validates selected artifact dependency order (cycle-safe topo ordering).
|
||||
- resolves required/optional inputs per artifact source definition.
|
||||
- resolves prepared stable input sources from `inputs/*.yml` materialized by `prepare`.
|
||||
- resolves required/optional inputs per artifact source definition into an
|
||||
ordered semantic identity. Each identity records the configured input name,
|
||||
canonical source ID, required policy, explicit presence, source contract,
|
||||
checksum, size, and a source-based logical identity. Workspace paths and
|
||||
producer run IDs are excluded.
|
||||
- orders input identities by configured input name independently of Go map
|
||||
iteration. Runtime adapter paths remain a separate execution-only map.
|
||||
- omits an unavailable optional input from the adapter request while retaining
|
||||
explicit absence in its semantic identity; an unavailable required input
|
||||
fails.
|
||||
- resolves prepared stable input sources through the shared manifest-authoritative
|
||||
identity resolver; it does not accept incidental files or fall back to
|
||||
campaign/session source paths.
|
||||
- reuses checksums and sizes from validated prepared, extraction, and current
|
||||
configured-artifact evidence. Other resolved inputs are hashed as confined
|
||||
regular files with streaming reads and the central resolved-artifact size
|
||||
limit.
|
||||
- owns a versioned SHA-256 fingerprint contract with one fixed-field canonical
|
||||
JSON payload and no map serialization. Configured artifacts are fingerprinted
|
||||
in deterministic dependency order.
|
||||
- fingerprints the normalized artifact key, prompt and profile identifiers,
|
||||
effective render-debug behavior, session-relative output identity, sorted
|
||||
dependency keys, ordered input declarations and semantic identities,
|
||||
validated current dependency-output identities, and sorted effective
|
||||
Scriptorium variables (including Narratio's sticky session variable).
|
||||
- provides read-only reconciliation that classifies each configured record as
|
||||
current, stale, missing, failed, legacy, or otherwise non-resumable, and
|
||||
separately identifies manifest records removed from current configuration.
|
||||
A record is current only when its fingerprint version and value match and its
|
||||
configured output still passes manifest-authoritative evidence validation.
|
||||
- owns a read-only typed work planner. Its explicit targets are enabled
|
||||
artifacts by default or the exact normalized `--artifacts` selection when
|
||||
supplied. It closes targets over configured prerequisites, orders the closure
|
||||
topologically, reuses current members, and schedules every non-current member
|
||||
before its dependents.
|
||||
- force applies only to explicit targets. A current prerequisite is reused
|
||||
unless it is itself an explicit forced target; disabled prerequisites may be
|
||||
rebuilt when required, while unrelated disabled artifacts are excluded.
|
||||
- the work plan carries explicit targets, prerequisite-only work, deterministic
|
||||
execution and reuse lists, invalidated and removed records, and a cloned
|
||||
projected record collection. Valid unrelated configured records survive the
|
||||
projection, removed records are omitted, and legacy files never become
|
||||
current without regeneration.
|
||||
- implements aggregate resume validation by running the same read-only catalog,
|
||||
fingerprint reconciliation, and work planner used by execution. A succeeded
|
||||
aggregate record is reusable exactly when the selected closure schedules no
|
||||
artifact work; stale unrelated records do not block a partial selection.
|
||||
- exposes the typed artifact decision to `session plan`. Planning applies it to
|
||||
a cloned manifest after modeling earlier selected stage transitions, so
|
||||
aggregate run/skip and artifact execute/reuse decisions match the ordinary
|
||||
runner without creating durable state or invoking Scriptorium.
|
||||
- executes only the work plan's scheduled entries. Manifest-validated current
|
||||
prerequisites remain available through the runtime catalog without invoking
|
||||
Scriptorium; newly produced prerequisites enter that catalog with the same
|
||||
contract, checksum, and size identity used for persisted current evidence.
|
||||
- keeps adapter output in the invocation's run-local analyze directory until
|
||||
it is a safe, non-empty, bounded regular file with a calculated checksum and
|
||||
complete output contract. Canonical replacement uses the shared atomic file
|
||||
operation boundary and verifies that the installed checksum matches the
|
||||
validated run-local bytes.
|
||||
- records each successful artifact's freshly computed fingerprint, canonical
|
||||
relative output path, contract, checksum, size, producer run ID, bounded
|
||||
Scriptorium provenance, logs, and generated configuration references in the
|
||||
analyze-owned projection.
|
||||
- preserves valid unrelated current records during partial execution. If a
|
||||
rebuilt output's bytes and contract are unchanged, unselected dependents may
|
||||
remain current. If that semantic identity changes, unselected transitive
|
||||
dependents become stale without being executed; dependents included in the
|
||||
invocation are evaluated in dependency order instead.
|
||||
- reports all evaluated targets and prerequisites in invocation state. The
|
||||
runner reconstructs aggregate session outputs from every current session
|
||||
record and invocation outputs from only records produced by the current run.
|
||||
Unrelated stale records do not make an otherwise successful partial
|
||||
invocation fail.
|
||||
- resolves previous-session sources from local `previous/` cache only.
|
||||
- runs optional render-debug, then artifact execution.
|
||||
- validates non-empty output files and materializes canonical outputs.
|
||||
@@ -44,17 +137,49 @@ Supported source families:
|
||||
guidance.
|
||||
- dependency cycles or unavailable required dependencies fail.
|
||||
- adapter validation failures fail stage.
|
||||
- a scheduled artifact failure returns the restricted analyze-state projection
|
||||
with the active artifact marked `failed`, a bounded error, and no output
|
||||
authority. Current transitive dependents become stale without execution.
|
||||
- earlier artifacts from the invocation remain current only after their
|
||||
run-local output passed validation and canonical materialization. They remain
|
||||
in invocation history; unattempted later artifacts do not appear there.
|
||||
- unrelated current records survive a partial failure. Old canonical bytes for
|
||||
the failed artifact and newly materialized bytes whose projection cannot be
|
||||
persisted are incidental, not current evidence.
|
||||
- the runner persists a valid partial projection before it marks aggregate
|
||||
analyze failed and invalidates publish and notify through the application
|
||||
dependency relation. Projection-persistence errors retain the last durable
|
||||
per-artifact authority and are joined with the original failure context.
|
||||
|
||||
## Invariants
|
||||
|
||||
- `analyze` performs no remote storage calls for previous-session source resolution.
|
||||
- input-identity resolution is read-only: it does not invoke adapters,
|
||||
materialize outputs, update status, or create run records.
|
||||
- fingerprints exclude timeouts, retries, timestamps, producer and Narratio run
|
||||
IDs, executable and config paths, workspace roots, diagnostic locations, and
|
||||
executable or private transitive configuration contents. A change that is
|
||||
visible only inside Scriptorium—such as a file privately loaded by its config
|
||||
path—requires an explicit forced regeneration.
|
||||
- output provenance and metadata are deterministic per execution.
|
||||
- a canonical file without current per-artifact manifest evidence is never
|
||||
promoted to current state.
|
||||
|
||||
## Related Contracts And Tests
|
||||
|
||||
- [Configuration](../config.md#scriptorium-artifact-entries) owns artifact
|
||||
fields and source-selection rules.
|
||||
fields and source-selection rules, including
|
||||
[artifact families](../config.md#scriptorium-artifact-families).
|
||||
- [CLI](../cli.md) owns user-visible artifact selection.
|
||||
- [Scriptorium](../integrations/scriptorium.md) owns the subprocess contract.
|
||||
- Implementation and tests: `internal/stage/analyze.go`,
|
||||
`internal/stage/analyze_test.go`
|
||||
`internal/stage/analyze_input_identity.go`, `internal/stage/analyze_test.go`,
|
||||
`internal/stage/analyze_input_identity_test.go`,
|
||||
`internal/stage/analyze_fingerprint.go`,
|
||||
`internal/stage/analyze_fingerprint_test.go`,
|
||||
`internal/stage/analyze_reconciliation.go`, and
|
||||
`internal/stage/analyze_reconciliation_test.go`,
|
||||
`internal/stage/analyze_plan.go`, `internal/stage/analyze_plan_test.go`, and
|
||||
`internal/stage/analyze_incremental_execution_test.go`, and
|
||||
`internal/stage/analyze_failure_test.go`,
|
||||
`internal/stage/analyze_resume.go`, and `internal/stage/analyze_resume_test.go`
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
## Responsibility
|
||||
|
||||
`extract` runs after `trim` and before `render`. It converts the canonical
|
||||
`extract` runs after `render` and before `analyze`. It converts the canonical
|
||||
`narratio.transcript.final_trimmed` JSON into configured Notarius lane artifacts.
|
||||
An omitted or disabled Notarius section makes the stage explicitly self-skip
|
||||
with reason `notarius_disabled`, no outputs, and no Notarius runner.
|
||||
@@ -17,33 +17,63 @@ procedures belong in [Operations](../operations.md).
|
||||
`internal/stage/extract.go`:
|
||||
|
||||
1. resolves the final trimmed transcript from the shared artifact catalog;
|
||||
2. resolves and fingerprints the Notarius invocation contract;
|
||||
3. creates a run-local staging directory and invokes the injected
|
||||
2. resolves every configured prepared reference through the shared
|
||||
manifest-authoritative identity resolver before creating run-local output;
|
||||
3. streams each verified reference into an invocation-local snapshot and
|
||||
rejects any source change observed while copying;
|
||||
4. fingerprints the byte- and provenance-bearing Notarius invocation evidence,
|
||||
including sorted reference identities;
|
||||
5. creates a run-local staging directory and invokes the injected
|
||||
`notarius.Runner`;
|
||||
4. validates the successful receipt, confined index, configured required lane
|
||||
descriptors, and regular payload files;
|
||||
5. atomically promotes the complete bundle to its immutable durable location;
|
||||
6. records one non-selectable `notarius_index` output and one selectable
|
||||
6. revalidates the reference snapshots, then validates the v2 successful
|
||||
receipt, confined index, management documents, configured required lane
|
||||
descriptors, validation summaries, and regular payload files;
|
||||
7. atomically promotes the complete bundle to its immutable durable location;
|
||||
8. records one non-selectable `notarius_index` output and one selectable
|
||||
`notarius_lane` output per configured lane; and
|
||||
7. registers each lane as `narratio.extraction.<output_key>` for downstream
|
||||
9. registers each lane as `narratio.extraction.<output_key>` for downstream
|
||||
Scriptorium and publish resolution.
|
||||
|
||||
Lane records retain checksum, contract, producer run ID, and Notarius system,
|
||||
run, pipeline, and lane provenance. Stage metadata retains the durable bundle
|
||||
root, receipt, diagnostic paths, rejection/warning summaries, producing
|
||||
Narratio run ID, and invocation fingerprint. Validation completes before
|
||||
Narratio run ID, the resolved trimmed-input identity, and invocation
|
||||
fingerprint. The input identity binds the exact transcript bytes, canonical
|
||||
source ID, producer stage/output/run identity, and resolution provenance.
|
||||
Reference metadata contains only selector, source ID, canonical session-relative
|
||||
path, checksum, and size; adapter requests receive selector and absolute
|
||||
invocation-local snapshot path, never payload contents. Snapshot bytes must
|
||||
match the prepared identity both before and after Notarius runs, so a concurrent
|
||||
prepared-file replacement cannot make recorded provenance describe different
|
||||
bytes from those supplied to Notarius.
|
||||
Validation completes before
|
||||
promotion, so a rejected result cannot expose a partial durable bundle.
|
||||
|
||||
Any executed extraction outcome that replaces a different effective outcome
|
||||
marks succeeded downstream stages stale. Repeating the same disabled self-skip
|
||||
with no outputs is stable and does not repeatedly invalidate downstream stages.
|
||||
marks succeeded analysis and delivery dependents stale. Render is an independent
|
||||
sibling and remains current. Repeating the same disabled self-skip with no
|
||||
outputs is stable and does not repeatedly invalidate dependent stages.
|
||||
|
||||
## Resume Validation
|
||||
|
||||
`internal/stage/extract_resume.go` permits a skip only when the existing stage
|
||||
record succeeded and still matches the current invocation fingerprint. The
|
||||
fingerprint covers the resolved executable and config paths, pipeline ID,
|
||||
timeout, working directory, and sorted configured output contracts.
|
||||
Before the focused validator runs, the application compares extract's versioned
|
||||
semantic fingerprint. It covers enablement, Notarius pipeline identity, sorted
|
||||
reference selector/source mappings, sorted declared output contracts, and each
|
||||
canonical `narratio.extraction.<key>` output identity. It excludes executable,
|
||||
timeout, working directory, config path, and private Notarius config contents.
|
||||
|
||||
`internal/stage/extract_resume.go` then permits a skip only when the existing
|
||||
stage record still matches the current byte- and provenance-bearing invocation
|
||||
evidence. That evidence covers the current direct trimmed-transcript identity,
|
||||
sorted prepared-reference identities, pipeline identity, and configured output
|
||||
contracts. The same reference helper and transcript identity are resolved again
|
||||
for artifact evidence, so changing current transcript bytes, reference bytes,
|
||||
or producer identity makes the prior extraction obsolete. Operational runner
|
||||
settings do not invalidate otherwise current durable evidence.
|
||||
|
||||
A valid prepared-reference change makes extraction non-resumable. Missing,
|
||||
unsafe, or checksum-inconsistent prepared evidence is a hard validation error
|
||||
with prepare-force guidance because an immediate extract rerun cannot succeed.
|
||||
|
||||
The validator then checks the producing run identity, canonical immutable
|
||||
bundle root, path confinement and absence of symlink components, receipt
|
||||
@@ -52,15 +82,18 @@ contracts and provenance, regular-file status, and stored checksums. Missing or
|
||||
obsolete results are non-resumable and run again; unsafe filesystem conditions
|
||||
return an error rather than silently accepting or replacing data.
|
||||
|
||||
The fingerprint cannot observe files imported by Notarius configuration,
|
||||
profile contents, prompt/module definitions, or other transitive inputs.
|
||||
Operators must force extraction after changing any such input.
|
||||
Neither contract can observe files imported by Notarius configuration, profile
|
||||
contents, prompt/module definitions, or other transitive inputs. Operators must
|
||||
force extraction after changing any such private input behind a stable
|
||||
identifier.
|
||||
|
||||
## Failure Behavior
|
||||
|
||||
Adapter startup, timeout, nonzero exit, receipt decoding, path confinement,
|
||||
index compatibility, required-lane rejection, payload inspection, checksum, or
|
||||
promotion errors fail the stage through ordinary manifest transition handling.
|
||||
index compatibility, inconsistent warning or diagnostic envelopes,
|
||||
required-lane rejection or incomplete validation, payload inspection,
|
||||
checksum, or promotion errors fail the stage through ordinary manifest
|
||||
transition handling.
|
||||
Stdout receipt and stderr diagnostics remain separate. Downstream stages are
|
||||
not given selectable extraction sources unless the complete configured result
|
||||
has passed validation and promotion.
|
||||
@@ -75,7 +108,8 @@ available for audit and recovery.
|
||||
|
||||
- Stage execution, selection, and resume validation: `internal/stage/extract.go`,
|
||||
`internal/stage/extract_resume.go`,
|
||||
`internal/stage/extract_test.go`
|
||||
`internal/stage/extract_test.go`,
|
||||
`internal/stage/semantic_contracts_delivery.go`
|
||||
- Subprocess boundary: `internal/adapters/notarius/subprocess.go`,
|
||||
`internal/adapters/notarius/subprocess_test.go`
|
||||
- Catalog hydration: `internal/artifacts/extraction_catalog.go`,
|
||||
|
||||
@@ -29,9 +29,23 @@ Normalize raw transcript inputs and merge into base transcript via Seriatim.
|
||||
- base transcript must validate before stage success.
|
||||
- report output is config-gated.
|
||||
|
||||
## Resume Evidence
|
||||
|
||||
Merge records a versioned semantic-configuration fingerprint for the Seriatim
|
||||
merge operation, output schema, coalesce gap, and every configured advanced
|
||||
merge transformation. A change reruns merge and stales only its fixed
|
||||
descendants; prepare and transcribe remain reusable. Binary path, timeout,
|
||||
report emission, logs, and diagnostic retention are operational exclusions.
|
||||
|
||||
Configuration or resources loaded privately inside Seriatim are outside
|
||||
Narratio's observable contract and require `--force` when changed. An existing
|
||||
successful merge record without evidence reruns once when selected.
|
||||
|
||||
## Related Contracts And Tests
|
||||
|
||||
- [Seriatim](../integrations/seriatim.md) owns subprocess and output semantics.
|
||||
- [Configuration](../config.md#pipeline) owns operator-selected Seriatim values.
|
||||
- Implementation and tests: `internal/stage/merge.go`,
|
||||
`internal/stage/merge_test.go`
|
||||
`internal/stage/merge_test.go`,
|
||||
`internal/stage/semantic_contracts_initial.go`, and
|
||||
`internal/stage/semantic_contracts_initial_test.go`
|
||||
|
||||
@@ -26,9 +26,17 @@ Normalize polished transcript into final transcript using Seriatim.
|
||||
- final transcript must validate as processed transcript JSON (`segments` array).
|
||||
- normalize defaults are applied when `pipeline.normalize` is unset.
|
||||
|
||||
## Resume Semantics
|
||||
|
||||
The versioned semantic fingerprint covers the Seriatim normalize operation,
|
||||
output schema and canonical output identity, plus the configured transcript
|
||||
transformations. Seriatim's executable and timeout and optional report
|
||||
generation are operational and do not invalidate the normalized transcript.
|
||||
|
||||
## Related Contracts And Tests
|
||||
|
||||
- [Seriatim](../integrations/seriatim.md) owns subprocess and output semantics.
|
||||
- [Configuration](../config.md#pipeline) owns normalize fields and defaults.
|
||||
- Implementation and tests: `internal/stage/normalize.go`,
|
||||
`internal/stage/normalize_test.go`
|
||||
`internal/stage/normalize_test.go`,
|
||||
`internal/stage/semantic_contracts_refinement.go`
|
||||
|
||||
@@ -17,7 +17,8 @@ Run Audita polishing on base transcript and produce polished transcript.
|
||||
## Key Behavior
|
||||
|
||||
- resolves base transcript from merge outputs/canonical fallback.
|
||||
- invokes Audita with configured model/module/runtime options.
|
||||
- invokes an Audita runner configured with static model/runtime options; the
|
||||
invocation supplies paths and modules.
|
||||
- validates processed transcript structure (`segments` array required).
|
||||
- validates optional report JSON.
|
||||
- materializes canonical outputs; records logs/generated config and adapter metadata.
|
||||
@@ -27,10 +28,26 @@ Run Audita polishing on base transcript and produce polished transcript.
|
||||
- polished transcript schema validation is mandatory.
|
||||
- report output is config-gated.
|
||||
|
||||
## Resume Semantics
|
||||
|
||||
The versioned semantic fingerprint covers the Audita service endpoint, model,
|
||||
validation model, module set, transcript description, output schema, selected
|
||||
external configuration path, and canonical polished-transcript identity. Module
|
||||
ordering is normalized because the configured modules form a set. Audita's
|
||||
executable, timeouts, concurrency, report and debug behavior, work retention,
|
||||
and credential environment name are operational and do not invalidate a
|
||||
successful result.
|
||||
|
||||
Narratio can fingerprint a selected model, module, or configuration identifier,
|
||||
but it cannot inspect content that Audita privately resolves behind that stable
|
||||
identifier. Force `polish` after changing such private content without changing
|
||||
its identifier.
|
||||
|
||||
## Related Contracts And Tests
|
||||
|
||||
- [Audita](../integrations/audita.md) owns subprocess, validation, and failure
|
||||
semantics.
|
||||
- [Configuration](../config.md#pipeline) owns operator-selected Audita values.
|
||||
- Implementation and tests: `internal/stage/polish.go`,
|
||||
`internal/stage/polish_test.go`
|
||||
`internal/stage/polish_test.go`,
|
||||
`internal/stage/semantic_contracts_refinement.go`
|
||||
|
||||
@@ -8,6 +8,7 @@ Materialize canonical current-session inputs before processing stages.
|
||||
|
||||
- resolved campaign, session, and pipeline configuration
|
||||
- stable input files (`speakers`, `autocorrect`, `glossary`, `players`, `party`)
|
||||
- optional spell-catalog overlay
|
||||
- one resolved local or S3 audio source
|
||||
- enabled configured artifact input requirements for previous-session sources
|
||||
|
||||
@@ -21,6 +22,7 @@ Materialize canonical current-session inputs before processing stages.
|
||||
- `inputs/glossary.yml`
|
||||
- `inputs/players.yml`
|
||||
- `inputs/party.yml`
|
||||
- optional `inputs/spell_catalog.json`
|
||||
- `audio/*.flac`
|
||||
- optional `previous/manifest.json`
|
||||
- optional `previous/artifacts/**`
|
||||
@@ -30,23 +32,54 @@ Materialize canonical current-session inputs before processing stages.
|
||||
|
||||
- validates required config/store state.
|
||||
- enforces local audio vs S3 audio mutual exclusivity.
|
||||
- rejects duplicate explicit local audio sources after resolution.
|
||||
- gives distinct local source paths with the same basename deterministic unique
|
||||
prepared filenames so neither source is overwritten.
|
||||
- materializes S3 audio through spool/cache-aware logic.
|
||||
- materializes a configured spell catalog with checksum and provenance, or
|
||||
safely removes an obsolete canonical spell catalog and its manifest record
|
||||
when the effective input is omitted.
|
||||
- in canonical party mode, copies the validated raw party bytes unchanged and
|
||||
deterministically generates the prepared players projection; legacy mode
|
||||
continues to copy its opaque party and explicit players sources.
|
||||
- scans enabled configured artifact inputs for `narratio.previous_session.artifact.*` requirements.
|
||||
- when previous requirements exist:
|
||||
- clears managed `previous/` state;
|
||||
- builds previous-cache remote plan;
|
||||
- clears managed `previous/` state on every invocation, then, when requirements exist:
|
||||
- resolves the pointer-selected previous source through the shared resolver;
|
||||
- downloads previous manifest/artifacts;
|
||||
- records previous inputs in `manifest.inputs`.
|
||||
|
||||
Required previous-session inputs fail when unavailable; optional missing inputs are skipped.
|
||||
Required previous-session inputs fail when unavailable; optional missing inputs
|
||||
are typed skipped results. Committed sources use their exact source-to-destination
|
||||
mapping, while the isolated legacy reader rejects ambiguous fallback matches.
|
||||
|
||||
## Invariants
|
||||
|
||||
- only `prepare` hydrates canonical `previous/` cache state.
|
||||
- managed previous artifacts are stored under `previous/artifacts/**` without
|
||||
duplicate `artifacts/artifacts/` nesting.
|
||||
- managed `previous/` state represents only the current requirement set.
|
||||
- `manifest.inputs` ordering is deterministic (`kind`, `path`).
|
||||
|
||||
## Resume Evidence
|
||||
|
||||
Prepare records a versioned semantic-configuration fingerprint for the
|
||||
resolved campaign/session selection, local-versus-S3 audio mode and canonical
|
||||
audio names, stable-input ownership/presence, previous-session identity, and
|
||||
the party mode plus canonical players projection version, and the effective
|
||||
previous-artifact requirement set. A change reruns prepare and
|
||||
stales its fixed descendants. Existing successful records without this
|
||||
evidence rerun once when selected.
|
||||
|
||||
Workspace, spool, and cache placement and absolute source relocation are not
|
||||
semantic when logical selection, canonical names, and bytes are equivalent.
|
||||
The fingerprint deliberately does not read or rehash large audio. Prepared
|
||||
input checksums remain the content provenance. Before reusing success, prepare
|
||||
validates every durable prepared copy and compares current stable-input bytes,
|
||||
canonical party and derived-player bytes, local audio membership/checksums, or
|
||||
S3 key/size/entity-tag identity with that provenance. Source relocation with
|
||||
equivalent names and bytes remains reusable; changed or unavailable evidence
|
||||
causes a normal prepare rerun.
|
||||
|
||||
## Related Contracts And Tests
|
||||
|
||||
- [Configuration](../config.md) owns audio selection, stable input fields, and
|
||||
@@ -56,5 +89,10 @@ Required previous-session inputs fail when unavailable; optional missing inputs
|
||||
- [Storage Internals](storage.md) and [Artifact Internals](artifacts.md) explain
|
||||
the internal collaborators.
|
||||
- Implementation and tests: `internal/stage/prepare.go`,
|
||||
`internal/stage/prepare_test.go`, `internal/audio/s3_audio_test.go`,
|
||||
`internal/stage/prepare_test.go`,
|
||||
`internal/stage/prepare_resume.go`,
|
||||
`internal/stage/prepare_resume_test.go`,
|
||||
`internal/stage/semantic_contracts_initial.go`,
|
||||
`internal/stage/semantic_contracts_initial_test.go`,
|
||||
`internal/audio/s3_audio_test.go`,
|
||||
`internal/previouscache/*_test.go`
|
||||
|
||||
@@ -9,34 +9,58 @@ Upload run/session outputs to object storage and atomically advance remote curre
|
||||
- successful preceding stages from the [canonical stage set](overview.md#pipeline-stage-set)
|
||||
- invocation-scoped run files
|
||||
- resolved publish output rules
|
||||
- effective publish locks (static + remote merged lock set)
|
||||
- effective publish locks (static + remote merged lock set), revalidated at the
|
||||
remote commit boundary
|
||||
- durable previous-session cache files when present
|
||||
|
||||
## Outputs
|
||||
|
||||
- uploaded invocation record and selected publish outputs;
|
||||
- uploaded durable previous-session cache files when present;
|
||||
- updated remote current manifest; and
|
||||
- remote current-run commit marker, written last.
|
||||
- immutable run-scoped commit manifest; and
|
||||
- current commit pointer, written last.
|
||||
|
||||
Exact remote placement and the operator workflow belong in
|
||||
[Operations](../operations.md#publish-workflow).
|
||||
|
||||
## Key Behavior
|
||||
|
||||
- stage can self-skip when publish disabled or run upload disabled.
|
||||
- when publishing or run upload is disabled, completes successfully with no
|
||||
outputs and records explanatory metadata. This is not an explicit self-skip:
|
||||
both manifests record success. Enablement and upload policy are fingerprinted,
|
||||
so changing either automatically makes the prior result non-resumable.
|
||||
- validates prerequisite stage success and object-store availability.
|
||||
- collects a deterministic run file list plus run `manifest.json`, excluding
|
||||
`audio/**` and the run-local `extract/notarius-output/**` staging bundle.
|
||||
- keeps run-local Notarius receipt and stderr diagnostics eligible for the run
|
||||
archive.
|
||||
- resolves publish output sources through runtime artifact catalog and manifest-aware resolution.
|
||||
- derives a deterministic run-archive allowlist from the validated run
|
||||
`manifest.json`: declared run-local outputs, logs, generated configs, and the
|
||||
manifest itself. Unlisted workspace files are not archive candidates.
|
||||
- opens each archive candidate beneath its archive root without following
|
||||
symlinked ancestors or leaf entries, verifies that it is a regular file and
|
||||
checks a declared checksum when present, then streams the opened descriptor.
|
||||
- derives the durable previous-cache archive from its validated manifest using
|
||||
the same confinement and regular-file checks.
|
||||
- resolves publish output sources through runtime artifact catalog and
|
||||
manifest-aware resolution. Configured Scriptorium outputs are publishable
|
||||
only from validated `current` per-artifact analyze evidence; an incidental
|
||||
canonical file, legacy aggregate output, stale/failed/unselected record, or
|
||||
mismatched path, size, or checksum remains unavailable. This does not change
|
||||
the explicit compatibility policies owned by built-in, extraction, or
|
||||
previous-session sources.
|
||||
- publishes extraction lanes only through explicit configured output rules;
|
||||
neither run-local nor durable Notarius bundles are scanned or uploaded wholesale.
|
||||
- selected artifact filter applies to configured artifact sources only.
|
||||
- locked outputs are skipped intentionally (including required ones).
|
||||
- optional missing outputs are skipped; required missing unlocked outputs fail.
|
||||
- writes remote current manifest before current run pointer.
|
||||
- creates one complete immutable source-to-destination mapping before upload;
|
||||
- uploads and verifies every declared immutable object and the commit manifest;
|
||||
- updates `current/commit-pointer.json` exactly once, last; and
|
||||
- does not write the legacy `current/manifest.json` or `current/run_id.txt` pair.
|
||||
- rechecks remote lock state immediately before the pointer update. A newly
|
||||
committed lock aborts selection, leaving any uploaded immutable attempt
|
||||
unselected.
|
||||
- reads the mutable remote lock document through a direct limit-plus-one read
|
||||
capped by `MaxRemoteLockStoreBytes` (1 MiB), retaining the generation returned
|
||||
with the opened body for conditional updates. Oversized lock documents fail
|
||||
before YAML decoding; published artifact payloads do not use this limit.
|
||||
|
||||
## Metadata Signals
|
||||
|
||||
@@ -47,16 +71,33 @@ Includes counts/lists for:
|
||||
- skipped optional outputs
|
||||
- skipped unselected outputs
|
||||
- locked outputs
|
||||
- current-state key paths
|
||||
- `current_pointer_written`
|
||||
- remote commit and current-pointer key paths
|
||||
- the run identifier selected by the commit
|
||||
|
||||
## Invariants
|
||||
|
||||
- `current/run_id.txt` is the remote commit marker and is written last.
|
||||
- run upload excludes `audio/**` and `extract/notarius-output/**`.
|
||||
- `extract/notarius.receipt.json` and `extract/notarius.stderr.log` remain
|
||||
eligible run-record diagnostics.
|
||||
- publish locks are not overridden by `--force`.
|
||||
- `current/commit-pointer.json` is the remote commit marker and is written last.
|
||||
- run files, selected outputs, previous-cache files, and the committed session
|
||||
manifest are all declared by an immutable commit under the run prefix.
|
||||
- run and previous uploads contain only manifest-declared regular files opened
|
||||
from verified descriptors; symlinks, special files, replacement races, and
|
||||
undeclared entries are rejected or ignored before uploads begin.
|
||||
- run-local diagnostics, including Notarius receipt and stderr files, are
|
||||
archived only when recorded by the run manifest.
|
||||
- publish locks are not overridden by `--force`; remote locks are revalidated
|
||||
immediately before current-state selection.
|
||||
- post-commit local cleanup is authorized by the committed publish metadata and
|
||||
is durably recorded by the application lifecycle before any local deletion.
|
||||
|
||||
## Resume Semantics
|
||||
|
||||
The versioned semantic fingerprint covers enabled behavior, run-upload policy,
|
||||
normalized source/destination/required output rules, static lock policy, and
|
||||
the remote backend, bucket, region, endpoint, and root-prefix identity. Rule
|
||||
and lock ordering is canonicalized. Credential environment names,
|
||||
path-addressing transport mode, local workspace placement, and run identifiers
|
||||
are excluded. Remote locks remain mutable state and are still revalidated at
|
||||
the commit boundary; semantic evidence does not replace that safety check.
|
||||
|
||||
The commit boundary and cleanup gate are normative architecture invariants; see
|
||||
[Architecture](../policy/architecture.md#publish-commit-boundary).
|
||||
@@ -70,5 +111,7 @@ The commit boundary and cleanup gate are normative architecture invariants; see
|
||||
- [Artifact Internals](artifacts.md) explains source resolution and current-state
|
||||
helpers.
|
||||
- Implementation and tests: `internal/stage/publish.go`,
|
||||
`internal/stage/publish_test.go`, `internal/app/operator_helpers_test.go`,
|
||||
`internal/stage/publish_test.go`,
|
||||
`internal/stage/semantic_contracts_delivery.go`,
|
||||
`internal/app/operator_helpers_test.go`, and
|
||||
`internal/app/post_publish_cleanup_test.go`
|
||||
|
||||
@@ -3,6 +3,10 @@
|
||||
## Purpose
|
||||
|
||||
Render Markdown transcript artifacts from normalized JSON transcripts via Seriatim.
|
||||
It runs after `trim` and before `extract` in the canonical sequence. Render and
|
||||
extract are independent sibling consumers: replacing render output does not
|
||||
invalidate extraction, but it does invalidate succeeded analysis and delivery
|
||||
records that may consume rendered transcripts.
|
||||
|
||||
## Inputs
|
||||
|
||||
@@ -20,7 +24,10 @@ Render Markdown transcript artifacts from normalized JSON transcripts via Seriat
|
||||
- resolves inputs manifest-first, then canonical fallback.
|
||||
- writes run-local outputs first, then materializes canonical session outputs.
|
||||
- records input provenance, output paths, adapter metadata, logs, and generated config refs.
|
||||
- skips with stage metadata when `pipeline.render.enabled=false`.
|
||||
- when `pipeline.render.enabled=false`, completes successfully with no outputs
|
||||
and records explanatory metadata. This is not an explicit self-skip: both
|
||||
manifests record success. Because enablement is fingerprinted, enabling
|
||||
render later automatically makes the prior result non-resumable.
|
||||
|
||||
## Failure Semantics
|
||||
|
||||
@@ -34,9 +41,20 @@ Render Markdown transcript artifacts from normalized JSON transcripts via Seriat
|
||||
- only `format: markdown` is supported.
|
||||
- render stage owns production of built-in Markdown transcript sources.
|
||||
|
||||
## Resume Semantics
|
||||
|
||||
The versioned semantic fingerprint covers enablement, final format, resolved
|
||||
title (including the session-title fallback), timestamp, segment-ID and
|
||||
metadata inclusion, both canonical input identities, and both Markdown output
|
||||
identities. Seriatim's executable, timeout, and report behavior are operational
|
||||
and do not invalidate rendered transcripts. A render-only change leaves the
|
||||
independent `extract` sibling reusable while invalidating their shared
|
||||
downstream consumers.
|
||||
|
||||
## Related Contracts And Tests
|
||||
|
||||
- [Seriatim](../integrations/seriatim.md) owns render subprocess behavior.
|
||||
- [Configuration](../config.md#pipeline) owns render fields and defaults.
|
||||
- Implementation and tests: `internal/stage/render.go`,
|
||||
`internal/stage/render_test.go`
|
||||
`internal/stage/render_test.go`,
|
||||
`internal/stage/semantic_contracts_refinement.go`
|
||||
|
||||
@@ -15,16 +15,34 @@ Generate raw per-speaker transcripts from prepared audio using WhisperX.
|
||||
## Key Behavior
|
||||
|
||||
- discovers prepared audio from manifest inputs or canonical audio directory.
|
||||
- derives speaker ID from `.flac` basename.
|
||||
- derives the transcript identity from the prepared `.flac` filename.
|
||||
- dispatches WhisperX requests through a bounded worker pool.
|
||||
- validates each output as JSON.
|
||||
- writes run-local outputs then materializes canonical transcript outputs.
|
||||
- writes run-local outputs then materializes canonical transcript outputs only
|
||||
after every planned request succeeds.
|
||||
|
||||
## Invariants
|
||||
|
||||
- speaker basenames must be unique.
|
||||
- prepared audio identities must be unique; prepare disambiguates distinct
|
||||
source paths that share a basename.
|
||||
- output path returned by adapter must match requested output path.
|
||||
- each successful output is validated before stage success.
|
||||
- an empty adapter result path means the requested path; adapters cannot select
|
||||
an alternate destination.
|
||||
- each successful output is validated before stage success, and cancellation or
|
||||
incomplete dispatch cannot be reported as a successful result.
|
||||
|
||||
## Resume Evidence
|
||||
|
||||
Transcribe records a versioned semantic-configuration fingerprint containing
|
||||
the Narratio-visible WhisperX service URL and recognition language. Changes to
|
||||
either rerun transcription and stale its fixed descendants while leaving
|
||||
prepare reusable. Retry count/delay, concurrency, timeout, credentials, and
|
||||
diagnostic locations are operational and do not change this evidence.
|
||||
|
||||
WhisperX models or private service configuration not exposed by Narratio's
|
||||
adapter contract cannot be fingerprinted; use `--force` after changing them.
|
||||
An existing successful transcribe record without evidence reruns once when
|
||||
selected.
|
||||
|
||||
## Related Contracts And Tests
|
||||
|
||||
@@ -33,4 +51,6 @@ Generate raw per-speaker transcripts from prepared audio using WhisperX.
|
||||
- [Configuration](../config.md#pipeline) owns concurrency and other
|
||||
operator-selected values.
|
||||
- Implementation and tests: `internal/stage/transcribe.go`,
|
||||
`internal/stage/transcribe_test.go`
|
||||
`internal/stage/transcribe_test.go`,
|
||||
`internal/stage/semantic_contracts_initial.go`, and
|
||||
`internal/stage/semantic_contracts_initial_test.go`
|
||||
|
||||
@@ -33,6 +33,19 @@ When `trim.enabled=false`:
|
||||
- bounds output exists only in enabled trim path.
|
||||
- render-debug output is diagnostic and not a declared stage output.
|
||||
|
||||
## Resume Semantics
|
||||
|
||||
The versioned semantic fingerprint covers enablement, the bounds prompt and
|
||||
profile identifiers, the Scriptorium configuration identity, transcript input
|
||||
name, sticky session variable, bounds and trimmed output identities, and the
|
||||
Seriatim trim operation. Diagnostic bounds rendering, diagnostic output paths,
|
||||
timeouts, executable paths, and optional reports are operational and do not
|
||||
invalidate the canonical trimmed transcript.
|
||||
|
||||
Narratio cannot inspect prompt, profile, or configuration content that
|
||||
Scriptorium or Seriatim privately resolves behind a stable identifier. Force
|
||||
`trim` after changing such private content without changing its identifier.
|
||||
|
||||
## Related Contracts And Tests
|
||||
|
||||
- [Scriptorium](../integrations/scriptorium.md) owns bounds generation and
|
||||
@@ -40,4 +53,5 @@ When `trim.enabled=false`:
|
||||
- [Seriatim](../integrations/seriatim.md) owns transcript trimming behavior.
|
||||
- [Configuration](../config.md#pipeline) owns trim fields and defaults.
|
||||
- Implementation and tests: `internal/stage/trim.go`,
|
||||
`internal/stage/trim_test.go`
|
||||
`internal/stage/trim_test.go`,
|
||||
`internal/stage/semantic_contracts_refinement.go`
|
||||
|
||||
@@ -12,14 +12,23 @@ operator-selected storage fields and credential mechanisms belong in
|
||||
`storage.ObjectStore` interface:
|
||||
|
||||
- `List(ctx, prefix)`
|
||||
- `Read(ctx, key)` returns an object body and the generation observed with it
|
||||
- `Download(ctx, key, localPath)`
|
||||
- `Upload(ctx, localPath, key, opts)`
|
||||
- `UploadConditional(ctx, source, key, opts, condition)`
|
||||
- `Exists(ctx, key)`
|
||||
|
||||
Key invariant:
|
||||
- callers pass full bucket-relative keys;
|
||||
- storage implementations do not infer campaign/session/run prefixes.
|
||||
|
||||
`ReadObjectBounded` is the shared mechanism for small control objects. It opens
|
||||
one object version, returns the metadata observed with that body, rejects an
|
||||
oversized known size before transfer, and still performs a context-aware
|
||||
limit-plus-one read. It closes the body on every exit. Callers own the policy
|
||||
limit and add the control-object category to errors; this helper is not used for
|
||||
large artifact payloads.
|
||||
|
||||
## Composition
|
||||
|
||||
`NewObjectStoreFromConfig` constructs the S3-backed implementation from
|
||||
@@ -31,13 +40,20 @@ not own discovery, defaults, or configuration validation.
|
||||
|
||||
- normalizes object keys.
|
||||
- `List` paginates and returns normalized `ObjectInfo`.
|
||||
- A truncated S3 listing must supply a new, non-empty continuation token;
|
||||
otherwise listing fails with bucket and prefix context instead of looping.
|
||||
- `Download` writes local files with parent directory creation.
|
||||
- `Upload` streams local file and returns remote metadata.
|
||||
- `Read` binds a returned body to its S3 ETag. `UploadConditional` maps an ETag
|
||||
match or absence precondition directly to the provider request and reports a
|
||||
failed precondition without performing a local check-then-write replacement.
|
||||
- `Exists` maps not-found responses to `false`.
|
||||
|
||||
## Invariants
|
||||
|
||||
- storage layer is stateless regarding manifest/stage progression.
|
||||
- bounded reads never retain more than the caller's limit plus one byte and do
|
||||
not replace owner-specific size policy.
|
||||
- publish ordering semantics are owned by stage/app code, not storage adapters.
|
||||
|
||||
## Implementation And Tests
|
||||
|
||||
@@ -13,8 +13,16 @@ previous-cache path construction. `SessionPathsFor` provides the session-scoped
|
||||
path model, and layout creation goes through `EnsureLayoutFor`. Callers should
|
||||
consume those helpers instead of rebuilding relative paths.
|
||||
|
||||
`internal/pathsafe` and application cleanup helpers enforce confinement for
|
||||
relative destinations and deletion targets.
|
||||
`internal/pathsafe` validates relative destinations. `internal/fileops` opens
|
||||
cleanup roots and their descendants through no-follow directory handles before
|
||||
removing them.
|
||||
|
||||
`internal/fileops` owns the ordinary workspace mode contract. On POSIX,
|
||||
`WorkspaceDirectoryMode` is setgid `02775` and `WorkspaceFileMode` is `0664`.
|
||||
`EnsureWorkspaceDirectory` reapplies the directory mode after creation so a
|
||||
restrictive umask cannot remove group access, while retaining existing ownership
|
||||
and group. Credential paths are outside this contract; the platform-specific
|
||||
operational requirements are in [Operations](../operations.md#workspace-permissions).
|
||||
|
||||
## Run-Local Stage Layout
|
||||
|
||||
@@ -36,21 +44,30 @@ existing destination. Exact physical paths belong in
|
||||
|
||||
## Locking
|
||||
|
||||
`artifacts.LocalStore` enforces the single-writer session lock via `.lock`
|
||||
(`ErrLockConflict` on contention).
|
||||
`artifacts.LocalStore` enforces the single-writer session lock via an
|
||||
operating-system lock held on `.lock` (`ErrLockConflict` on contention). The
|
||||
file retains owner metadata after release or process death; its existence is
|
||||
not evidence that a lock is active. Command and restore flows wait for this
|
||||
lock only while their context remains active, and report a release failure.
|
||||
|
||||
## Cleanup Semantics
|
||||
|
||||
Automatic post-publish cleanup:
|
||||
|
||||
- only runs when publish actually executed and succeeded;
|
||||
- requires `uploaded=true` and `current_pointer_written=true` metadata;
|
||||
- is created only after a successful publish commit with complete publish
|
||||
metadata, then is persisted before any deletion;
|
||||
- requires `uploaded=true`, a remote commit key, and a current commit-pointer
|
||||
key in publish metadata;
|
||||
- consumes the resolved cleanup policy described in
|
||||
[Configuration](../config.md);
|
||||
- refuses unsafe deletes (root delete, out-of-root delete, symlink paths).
|
||||
- refuses unsafe deletes (root delete, out-of-root delete, and symlinked
|
||||
ancestors or entries);
|
||||
- retries any recorded incomplete target on later invocations even when no
|
||||
publish work is selected. Missing targets are a successful, idempotent
|
||||
cleanup result only after the completion evidence is saved.
|
||||
|
||||
Manual cleanup uses the same scoped-target checks. Invocation syntax and exact
|
||||
deletion scope belong in [CLI](../cli.md#clean) and
|
||||
Manual cleanup uses the same root-confined deletion mechanism. Invocation
|
||||
syntax and exact deletion scope belong in [CLI](../cli.md#clean) and
|
||||
[Operations](../operations.md#cleanup).
|
||||
|
||||
## Invariants
|
||||
@@ -58,6 +75,8 @@ deletion scope belong in [CLI](../cli.md#clean) and
|
||||
- campaign-aware session root is mandatory.
|
||||
- manifest-driven stage state is durable across runs.
|
||||
- cleanup guardrails prevent destructive root/out-of-scope deletion.
|
||||
- ordinary workspace paths retain group-writable directory and file modes across
|
||||
nested creation, replacement, and Notarius promotion.
|
||||
|
||||
## Implementation And Tests
|
||||
|
||||
@@ -65,10 +84,11 @@ deletion scope belong in [CLI](../cli.md#clean) and
|
||||
`internal/artifacts/local.go`
|
||||
- Run-local materialization: `internal/stage/run_local.go`
|
||||
- Immutable bundle promotion: `internal/fileops/directory.go`
|
||||
- Cleanup confinement: `internal/app/cleanup_targets.go`,
|
||||
`internal/app/post_publish_cleanup.go`
|
||||
- Workspace modes: `internal/fileops/modes.go`
|
||||
- Cleanup confinement: `internal/fileops/cleanup.go`,
|
||||
`internal/app/cleanup_targets.go`, `internal/app/post_publish_cleanup.go`
|
||||
- Tests: `internal/artifacts/paths_model_test.go`,
|
||||
`internal/artifacts/local_test.go`, `internal/stage/run_local_test.go`,
|
||||
`internal/fileops/directory_test.go`,
|
||||
`internal/app/cleanup_targets_test.go`,
|
||||
`internal/fileops/directory_test.go`, `internal/fileops/modes_posix_test.go`,
|
||||
`internal/fileops/cleanup_test.go`, `internal/app/cleanup_targets_test.go`,
|
||||
`internal/app/post_publish_cleanup_test.go`
|
||||
|
||||
@@ -36,7 +36,12 @@ narratio session init 2026-04-04 --remote --force
|
||||
|
||||
If `campaign.yml` sets `session_template_file`, `session init` renders it. Template variables must resolve to concrete values.
|
||||
|
||||
Campaigns must provide stable input files for speakers, autocorrect, glossary, players, and party. Session files may override those paths for one session. The `prepare` stage materializes them under `inputs/`; configured Scriptorium artifacts can reference prepared `players`, `party`, and `glossary` files with `narratio.input.players`, `narratio.input.party`, and `narratio.input.glossary`.
|
||||
Campaigns must provide stable input files for speakers, autocorrect, glossary,
|
||||
players, and party, and may provide an optional spell-catalog overlay. Session
|
||||
files may override those paths for one session. The `prepare` stage
|
||||
materializes them under `inputs/`; configured consumers use the prepared files,
|
||||
never the original campaign or session source paths. Field definitions and
|
||||
source IDs are in [Configuration](./config.md#notarius-reference-bindings).
|
||||
|
||||
## Standard Session Workflow
|
||||
|
||||
@@ -65,6 +70,19 @@ narratio run 2026-04-04
|
||||
narratio session status 2026-04-04
|
||||
```
|
||||
|
||||
Run, plan, and status output identify the resolved pipeline profile (or `none`)
|
||||
and effective configuration digest. Status distinguishes the current resolved
|
||||
value from the last value persisted in the session manifest, which helps
|
||||
diagnose profile switches without changing resume authority.
|
||||
|
||||
Before switching an operational profile, compare its effective meaning with the
|
||||
current selection through `narratio config diff <left-profile> <right-profile>`.
|
||||
The command is read-only and succeeds whether it finds differences or not. Its
|
||||
sorted records describe defaulted, expanded concrete configuration—not source
|
||||
file layout—so it can be used to review model, artifact, and publish changes
|
||||
without creating a session or run. Select the same campaign explicitly when
|
||||
profiles could resolve different campaign paths; see the [CLI reference](cli.md#config-validate-config-show-config-sources-and-config-diff) for syntax and record format.
|
||||
|
||||
## Stage Execution and Continuation Behavior
|
||||
|
||||
Canonical stage order:
|
||||
@@ -75,22 +93,53 @@ Canonical stage order:
|
||||
4. `polish`
|
||||
5. `normalize`
|
||||
6. `trim`
|
||||
7. `extract`
|
||||
8. `render`
|
||||
7. `render`
|
||||
8. `extract`
|
||||
9. `analyze`
|
||||
10. `publish`
|
||||
11. `notify`
|
||||
|
||||
Execution rules:
|
||||
|
||||
- succeeded stages are skipped unless `--force` is set;
|
||||
- succeeded stages are skipped unless `--force` is set; stages with semantic
|
||||
configuration contracts additionally require matching versioned evidence,
|
||||
and missing legacy evidence causes a safe one-time rerun;
|
||||
- `run` continues interrupted or partially completed sessions by running non-succeeded stages;
|
||||
- forcing an upstream stage marks succeeded downstream stages as `stale` before
|
||||
the replacement runs; and
|
||||
- forcing a stage marks succeeded transitive dependents as `stale` before the
|
||||
replacement runs; render and extract are independent siblings; and
|
||||
- an executed failure, changed self-skip, or success that replaces a different
|
||||
effective upstream outcome also marks succeeded downstream stages stale. A
|
||||
effective outcome uses the same fixed dependency relation. A
|
||||
repeated self-skip with the same reason and no outputs is stable and does not
|
||||
perpetually rerun downstream work.
|
||||
perpetually rerun dependent work.
|
||||
|
||||
Every aggregate stage except analyze currently provides semantic-configuration
|
||||
evidence; analyze retains its more precise per-artifact fingerprints and
|
||||
validator.
|
||||
Prepare additionally validates current stable/local/S3 source identity and the
|
||||
checksums of its durable prepared copies before reuse. Changed bytes, audio
|
||||
membership, S3 object identity, or missing/tampered copies rerun prepare and
|
||||
its fixed descendants without requiring `--force`. Changing prepare selection
|
||||
semantics likewise reruns all fixed descendants; changing
|
||||
WhisperX language/service identity reuses prepare; changing a Seriatim merge
|
||||
transformation reuses prepare and transcribe; and changing an Audita model
|
||||
reuses prepare, transcribe, and merge while rebuilding transcript refinement.
|
||||
A trim change invalidates both render and extract through the fixed dependency
|
||||
relation, while a render-only change preserves the extract sibling.
|
||||
|
||||
Operational timeouts, retry/concurrency tuning, executable paths,
|
||||
workspace/cache/spool placement, reports, diagnostics, and secret values are
|
||||
excluded. Configuration, models, prompts, modules, or resources loaded
|
||||
privately inside external tools remain unobservable to Narratio. If their
|
||||
contents change behind the same configured identifier, explicitly force the
|
||||
affected stage.
|
||||
|
||||
An explicit self-skip is a durable `skipped` stage outcome that later runs
|
||||
reconsider. It differs from successful no-output execution: disabled `render`
|
||||
and `publish`, and absent or no-executable `analyze`, record `succeeded` with
|
||||
metadata and no outputs. Ordinary later runs reuse those successful results;
|
||||
force the affected stage after enabling or configuring it. Optional artifact
|
||||
inputs are omitted only from the consuming artifact invocation and do not make
|
||||
the stage self-skip.
|
||||
|
||||
Single-stage execution:
|
||||
|
||||
@@ -98,14 +147,86 @@ Single-stage execution:
|
||||
narratio run-stage normalize 2026-04-04 --force
|
||||
```
|
||||
|
||||
Contiguous bounded execution uses inclusive canonical endpoints:
|
||||
|
||||
```bash
|
||||
narratio session plan 2026-04-04 --from extract --through analyze --force
|
||||
narratio run 2026-04-04 --from extract --through analyze --force
|
||||
```
|
||||
|
||||
Omitting `--from` selects from `prepare`; omitting `--through` selects through
|
||||
`notify`. Force applies only within the selected range. Repeating `--from`,
|
||||
`--through`, or `--force` is rejected instead of resolving by argument order.
|
||||
The plan command uses the same selection contract and prints only the selected
|
||||
range. Planning is read-only: it clones the loaded manifest, models selected
|
||||
stage transitions and invalidation in memory, and invokes resume validation
|
||||
without writing the manifest, creating run directories, materializing files,
|
||||
or invoking pipeline adapters. Analyze detail separates explicit targets,
|
||||
prerequisite rebuilds, scheduled execution, and current reuse. This lets a
|
||||
coarsely stale aggregate analyze stage show zero artifact executions when its
|
||||
selected artifact evidence is still semantically current.
|
||||
|
||||
Before a bounded run or plan whose range starts after `prepare`, every excluded
|
||||
prefix stage must already have a session-manifest status of `succeeded` or
|
||||
`skipped`. Narratio reports the first absent, pending, running, failed, stale,
|
||||
or interrupted prerequisite without creating a run record or changing session
|
||||
state. Widen `--from` to include that stage, or recover it explicitly before
|
||||
retrying. Excluded prefix stages are not resume-validated or repaired as part
|
||||
of the bounded invocation; selected stages still reject missing, unsafe, or
|
||||
manifest-inconsistent inputs at their owning boundary.
|
||||
|
||||
Stages after `--through` are not prerequisites and are never scheduled by the
|
||||
bounded invocation. A selected forced stage can mark one of those succeeded
|
||||
dependents stale through the fixed invalidation relation, but the dependent
|
||||
does not execute until a later invocation selects it. Production composition
|
||||
likewise initializes only collaborators needed by the selected range and
|
||||
shared session lifecycle. In particular, render does not require Notarius or
|
||||
Scriptorium, extract does not require Scriptorium, and analyze does not require
|
||||
the transcription, Seriatim, Audita, or Notarius adapters.
|
||||
|
||||
For the common post-transcript development loop, use:
|
||||
|
||||
```bash
|
||||
narratio regenerate-artifacts 2026-04-04
|
||||
narratio regenerate-artifacts 2026-04-04 --artifacts session_recap,player_handout
|
||||
```
|
||||
|
||||
This command is a transparent expansion to a forced bounded `run` from
|
||||
`extract` through `analyze`. Extraction always rebuilds its complete configured
|
||||
bundle. Analysis rebuilds the selected targets and their required analysis
|
||||
prerequisites, or uses the normal default selection when no artifact names are
|
||||
given. The command does not run publish or notify; delivery remains a separate
|
||||
operator action.
|
||||
|
||||
Inspect current artifact evidence, then publish explicitly when the regenerated
|
||||
set is ready:
|
||||
|
||||
```bash
|
||||
narratio session artifacts 2026-04-04
|
||||
narratio publish 2026-04-04
|
||||
```
|
||||
|
||||
If planning or execution reports stale, missing, failed, legacy, or tampered
|
||||
analysis evidence, regenerate the affected target instead of copying an older
|
||||
canonical file into place or editing the manifest. See
|
||||
[Troubleshooting: Analysis artifact evidence is not current](./troubleshooting.md#analysis-artifact-evidence-is-not-current).
|
||||
|
||||
## Artifact Selection
|
||||
|
||||
`--artifacts` can be used on `run`, `run-stage`, `analyze`, and `publish`.
|
||||
For a configured artifact family, selecting its family key expands to every
|
||||
concrete character artifact. Select a concrete generated key to operate on one
|
||||
member only. Manifests and plan output retain the concrete key as the durable
|
||||
identity and include the family and character ID as optional provenance.
|
||||
|
||||
`--artifacts` can be used on `run`, `session plan`, `run-stage`, `analyze`, and
|
||||
`publish`. For a bounded run or plan, the selected range must contain `analyze`
|
||||
or `publish`.
|
||||
|
||||
Selection behavior:
|
||||
|
||||
- validates names against `pipeline.scriptorium.artifacts`;
|
||||
- filters analyze execution to selected configured artifacts;
|
||||
- selects explicit analyze targets and permits their required configured
|
||||
prerequisites to be reused or rebuilt first;
|
||||
- filters publish rules for `narratio.artifact.<name>` sources only;
|
||||
- does not suppress built-in transcript, bounds, or explicitly configured
|
||||
`narratio.extraction.<name>` publish sources; and
|
||||
@@ -127,29 +248,75 @@ The directory is immutable once promoted. Configured lanes become
|
||||
the bundle and `index.json` are retained for audit and resume validation but
|
||||
are not selectable or published implicitly.
|
||||
|
||||
Configured Notarius references resolve only from the current manifest-backed
|
||||
prepared inputs. Their canonical locations are `inputs/party.yml`,
|
||||
`inputs/players.yml`, `inputs/glossary.yml`, and, when configured,
|
||||
`inputs/spell_catalog.json`. Extraction supplies Notarius with verified copies
|
||||
under `runs/<run_id>/extract/references/` so a concurrent refresh of canonical
|
||||
prepared files cannot change the bytes consumed by an in-flight invocation.
|
||||
For a canonical party, preparation retains the validated authored party bytes
|
||||
at `inputs/party.yml` and generates `inputs/players.yml` from that roster.
|
||||
The manifest records their checksums separately, with the players input marked
|
||||
as derived from the party; refresh preparation after changing the roster rather
|
||||
than editing either prepared file.
|
||||
Inspect the effective stable-input inventory and
|
||||
prepared-file readiness with:
|
||||
|
||||
```bash
|
||||
narratio session status 2026-04-04
|
||||
narratio session validate 2026-04-04
|
||||
```
|
||||
|
||||
Reference metadata records selector, source ID, session-relative path,
|
||||
checksum, and byte size, but never payload contents. Changing a prepared
|
||||
reference changes extraction identity: ordinary continuation rejects the old
|
||||
result, reruns Notarius, and marks successful downstream stages stale. If the
|
||||
prepared file is missing or inconsistent with its manifest checksum, repair
|
||||
the source configuration and refresh prepared state first:
|
||||
|
||||
```bash
|
||||
narratio run-stage prepare 2026-04-04 --force
|
||||
```
|
||||
|
||||
Starting a replacement clears the previous extraction payload from the current
|
||||
session-stage record. If that replacement fails or self-skips, the current
|
||||
record does not fall back to the earlier outputs. The earlier run manifest and
|
||||
immutable bundle remain available for inspection, but downstream resolution
|
||||
requires a new current successful extraction record.
|
||||
|
||||
Atomic Notarius bundle promotion is supported on Linux, macOS, and Windows.
|
||||
On other operating systems, extraction fails before copying the bundle into a
|
||||
Atomic Notarius bundle promotion is supported on Linux and macOS. On Windows
|
||||
and other operating systems, extraction fails before copying the bundle into a
|
||||
temporary promotion tree because Narratio has no verified atomic no-replace
|
||||
directory primitive there. This is an extraction limitation, not a broader
|
||||
platform-support guarantee for every Narratio workflow.
|
||||
|
||||
## External Command Lifecycle
|
||||
|
||||
When an external command is cancelled or times out, Narratio terminates its
|
||||
owned descendants as well as the command itself. Cancellation first requests
|
||||
termination where the platform supports it, then force terminates after a
|
||||
bounded wait. A command is not considered finished until its leader has been
|
||||
reaped, and descendants that keep standard output or error open cannot keep
|
||||
the invocation blocked. Other operating systems fail closed rather than launch
|
||||
a command without tree ownership.
|
||||
|
||||
Subprocess stdout and stderr diagnostics are separately redacted and capped at
|
||||
8 MiB per invocation. Narratio does not retain configured credential values in
|
||||
these logs or their error tails; reaching a capture limit terminates the command
|
||||
tree and reports which stream exceeded the limit.
|
||||
|
||||
Run-local diagnostics are:
|
||||
|
||||
- `runs/{run_id}/extract/notarius.receipt.json`
|
||||
- `runs/{run_id}/extract/notarius.stderr.log`
|
||||
- `runs/{run_id}/extract/notarius-output/` before durable promotion
|
||||
|
||||
The run-record upload excludes the complete
|
||||
`extract/notarius-output/**` subtree. The receipt and stderr files remain
|
||||
eligible run-record diagnostics. The durable bundle is never scanned for
|
||||
implicit publication; only lanes named by explicit `pipeline.publish.outputs`
|
||||
rules are uploaded.
|
||||
The run-record upload is an allowlist derived from the validated run manifest,
|
||||
not a workspace scan. Each declared source is opened without following
|
||||
symlinked ancestors or the leaf, verified as a regular file, and streamed from
|
||||
that verified descriptor. Unlisted files and unsafe entries are never uploaded.
|
||||
The durable bundle is never scanned for implicit publication; only lanes named
|
||||
by explicit `pipeline.publish.outputs` rules are uploaded.
|
||||
|
||||
To intentionally replace the current extraction result, run:
|
||||
|
||||
@@ -157,15 +324,26 @@ To intentionally replace the current extraction result, run:
|
||||
narratio run-stage extract 2026-04-04 --force
|
||||
```
|
||||
|
||||
Narratio automatically reruns extraction when its recorded invocation contract
|
||||
or durable output validation changes. It cannot fingerprint configuration
|
||||
files, profiles, prompts, modules, or references loaded transitively by
|
||||
Notarius. Force extraction after changing any of those inputs, even when the
|
||||
top-level Narratio and Notarius config paths remain the same. A forced extract
|
||||
Narratio automatically reruns extraction when its recorded invocation contract,
|
||||
prepared Narratio reference identities, or durable output validation changes.
|
||||
The semantic portion covers Notarius enablement, pipeline identity, declared
|
||||
reference mapping, and output contracts. Executable, timeout, working directory,
|
||||
and private config-file paths are operational and do not invalidate a current
|
||||
result.
|
||||
It cannot fingerprint configuration files, profiles, prompts, modules, or
|
||||
other references loaded transitively by Notarius itself. Force extraction after
|
||||
changing any of those inputs, even when the top-level Narratio and Notarius
|
||||
config paths remain the same. A forced extract
|
||||
marks successful downstream stages stale. Ordinary extraction failures or
|
||||
outcome changes also stale affected downstream stages, while an identical
|
||||
repeated `notarius_disabled` self-skip does not repeatedly invalidate them.
|
||||
|
||||
Publish reuse additionally tracks enabled/run-upload behavior, normalized
|
||||
output rules, static locks, and remote backend/bucket/region/endpoint/root
|
||||
identity. Credential environment names, local workspace placement, and run IDs
|
||||
are excluded. Regardless of semantic reuse evidence, executing publish still
|
||||
revalidates mutable remote locks immediately before commit selection.
|
||||
|
||||
## Publish Workflow
|
||||
|
||||
Run publish only:
|
||||
@@ -184,13 +362,29 @@ Publish commit model:
|
||||
|
||||
- uploads eligible run files under `{session_prefix}/runs/{run_id}/`, excluding
|
||||
audio and the run-local Notarius staging bundle;
|
||||
- uploads configured published outputs, including only explicitly configured
|
||||
extraction lanes;
|
||||
- uploads `previous/**` cache files when present;
|
||||
- writes `current/manifest.json`;
|
||||
- writes `current/run_id.txt` last.
|
||||
- uploads configured published outputs and `previous/**` cache files into the
|
||||
same immutable run scope, including only explicitly configured extraction
|
||||
lanes;
|
||||
- writes `{session_prefix}/runs/{run_id}/commit.json` after all declared
|
||||
immutable objects are uploaded and verified; and
|
||||
- writes `{session_prefix}/current/commit-pointer.json` once, last.
|
||||
|
||||
`current/run_id.txt` is the remote current-state commit marker.
|
||||
`current/commit-pointer.json` is the remote current-state commit marker. It
|
||||
selects exactly one immutable commit, which declares the complete object set.
|
||||
|
||||
## Remote Commit Migration
|
||||
|
||||
The immutable remote commit contract uses
|
||||
`runs/{run_id}/commit.json` to declare a run's complete object set and a small
|
||||
`current/commit-pointer.json` to select it. The pointer binds the selected
|
||||
commit by version, checksum, size, and storage generation; committed artifacts
|
||||
are also checksum- and generation-bound. Readers accept this contract now and
|
||||
strictly reject mismatched or unknown data.
|
||||
|
||||
Legacy reads are limited to a coherent `current/manifest.json` and
|
||||
`current/run_id.txt` pair; a torn pair is rejected. New publication does not
|
||||
write that pair and remote commit state does not carry local
|
||||
`current_pointer_written` metadata.
|
||||
|
||||
## Publish Locks
|
||||
|
||||
@@ -204,7 +398,14 @@ Effective lock rules:
|
||||
- static and remote locks are merged;
|
||||
- static locks win on source collisions;
|
||||
- locked outputs are intentional skips;
|
||||
- lock add/remove commands mutate only remote lock state.
|
||||
- lock add/remove commands mutate only remote lock state through generation-bound
|
||||
conditional writes. A command retries a bounded number of concurrent
|
||||
conflicts while its invocation context remains active, so it never replaces a
|
||||
different lock-document generation; and
|
||||
- a publish re-reads remote locks immediately before it writes the current
|
||||
commit pointer. A lock committed before that recheck prevents selecting the
|
||||
new snapshot, even though its already-uploaded immutable objects may remain
|
||||
available for a later retry.
|
||||
|
||||
Examples:
|
||||
|
||||
@@ -230,20 +431,30 @@ Apply:
|
||||
narratio session restore 2026-04-04
|
||||
```
|
||||
|
||||
`--dry-run` does not write durable session files. It still reads the selected
|
||||
remote current state and may read object identity/content needed to classify the
|
||||
plan, so it is not a network-free operation.
|
||||
|
||||
Default restore scope:
|
||||
|
||||
- `manifest.json`
|
||||
- `transcripts/**`
|
||||
- `artifacts/**`
|
||||
- the committed session manifest and the committed transcript/artifact objects
|
||||
declared by the selected remote commit
|
||||
- `previous/**` when needed by configured previous-session artifact inputs
|
||||
|
||||
Optional:
|
||||
|
||||
- `--include-audio` to include `audio/**`
|
||||
- `--force` to overwrite local conflicts
|
||||
- `--force` to overwrite eligible conflicting regular files; it never replaces
|
||||
directories or other non-regular local targets
|
||||
|
||||
Restore writes an execution report at `reports/restore-latest.json`.
|
||||
|
||||
If restore fails after beginning installation, it leaves a durable
|
||||
`.restore-incomplete.json` marker in the session root. Pipeline runs will stop
|
||||
until you rerun the same restore command and it completes. Restore intentionally
|
||||
does not try to roll back files already installed; retrying the selected remote
|
||||
snapshot is the recovery procedure.
|
||||
|
||||
## Local State Layout
|
||||
|
||||
Session root:
|
||||
@@ -284,6 +495,38 @@ Cache layout (durable S3 audio cache):
|
||||
|
||||
- `{cache.root}/s3/{bucket}/...`
|
||||
|
||||
Each cached audio file has an adjacent managed identity record. It binds the
|
||||
file to its remote object version and verified digest; deleting or altering the
|
||||
record simply causes Narratio to download and verify the object again.
|
||||
|
||||
### Workspace Permissions
|
||||
|
||||
Ordinary Narratio workspace content is intentionally shareable with the
|
||||
workspace group. On POSIX systems, Narratio-created workspace, spool, and cache
|
||||
directories converge on setgid `02775`; ordinary files, including manifests,
|
||||
transcripts, generated configuration, logs, reports, and Notarius artifacts,
|
||||
converge on `0664`. Narratio explicitly applies these modes so a restrictive
|
||||
caller umask does not remove group write or setgid. It does not change file or
|
||||
directory ownership: the configured workspace's existing group is inherited.
|
||||
|
||||
Windows does not implement POSIX mode bits or setgid semantics. Configure the
|
||||
workspace, spool, and cache locations with an ACL that grants the collaborating
|
||||
group read/write access, and configure credential locations with an ACL limited
|
||||
to the intended credential owner. Do not use POSIX mode displays as evidence of
|
||||
Windows access control.
|
||||
|
||||
API keys are credentials, not ordinary workspace data. Store them outside the
|
||||
shared workspace or in a separately restricted credential location; ordinary
|
||||
workspace group access must never be treated as authorization to read keys.
|
||||
On POSIX, provision a credential directory as `0700` and credential files as
|
||||
`0600`; Narratio rejects group- or other-readable configured credential paths.
|
||||
On Windows, restrict the directory and files with ACLs to the credential owner.
|
||||
|
||||
External adapter results are individually bounded before Narratio validates or
|
||||
materializes them. These per-file limits do not reserve disk space: prevent hard
|
||||
disk exhaustion with filesystem, service, container, or volume quotas sized for
|
||||
the session workload.
|
||||
|
||||
## Cleanup
|
||||
|
||||
Session-scoped cleanup:
|
||||
@@ -309,9 +552,18 @@ Rules:
|
||||
|
||||
- `clean` deletes work/spool session state;
|
||||
- cache is preserved unless `--clear-cache` is set;
|
||||
- each deletion is confined beneath its configured workspace, spool, or cache
|
||||
root and refuses symlinked paths;
|
||||
- automatic post-publish cleanup is gated by successful publish commit plus:
|
||||
- `pipeline.spool.delete_audio_after_publish=true`
|
||||
- `pipeline.workspace.cleanup_after_publish=true`
|
||||
- Narratio first records the exact run-scoped cleanup obligation. If cleanup
|
||||
reports incomplete, the remote committed snapshot remains current; rerun
|
||||
publish to retry only the outstanding confined local cleanup.
|
||||
|
||||
Post-publish cleanup is evaluated only when `publish` actually executes in the
|
||||
current invocation. A bounded range that excludes publish does not replay a
|
||||
cleanup obligation as an unrelated side effect.
|
||||
|
||||
## Operational Caveats
|
||||
|
||||
|
||||
@@ -29,6 +29,10 @@ in the [integration documentation](../integrations/).
|
||||
The pipeline has one canonical ordered stage set. Configuration may enable,
|
||||
disable, or parameterize supported behavior, but it must not turn that sequence
|
||||
into an arbitrary DAG or hide orchestration in generic workflow abstractions.
|
||||
An invocation selects either the full sequence or one inclusive contiguous
|
||||
range of it. Execution remains flat and canonical even though invalidation is
|
||||
dependency-aware: the application owns a separate fixed relation used only to
|
||||
stale transitive dependents, including dependents outside a selected range.
|
||||
The implemented stage inventory belongs in the
|
||||
[Internal Overview](../internal/overview.md).
|
||||
|
||||
@@ -87,8 +91,10 @@ merely on incidental files existing on disk.
|
||||
|
||||
A failed or interrupted stage must not be presented as successful. Failure
|
||||
should preserve enough local state and diagnostics for inspection, recovery,
|
||||
and resume. Forcing an upstream stage invalidates succeeded downstream work
|
||||
according to the canonical stage order.
|
||||
and resume. Forcing a stage invalidates succeeded transitive dependents
|
||||
according to a fixed application-owned relation that is separate from canonical
|
||||
execution order. The relation is validated against the stage inventory and is
|
||||
not configurable.
|
||||
|
||||
A stage may explicitly self-skip with a stable reason and no outputs. That
|
||||
outcome is persisted, clears older outputs owned by the stage, and is
|
||||
@@ -119,6 +125,17 @@ install the validated session manifest after other restored durable files. The
|
||||
physical workflow and recovery procedures belong in
|
||||
[Operations](../operations.md).
|
||||
|
||||
Restore and runner transitions for one session use the same local lock. A
|
||||
durable incomplete-restore marker blocks runner reuse after a partial restore;
|
||||
safe retry, rather than rollback of arbitrary local effects, is the recovery
|
||||
mechanism. Restored manifest-local references must be confined to the selected
|
||||
local session root, never trusted as producer-machine absolute paths.
|
||||
|
||||
For the immutable remote-commit protocol, a restore or status operation binds
|
||||
to one pointer-selected commit and only its declared object identities. A force
|
||||
flag may replace an eligible regular managed file, but never turns a directory
|
||||
or other non-regular conflict into a successful restore.
|
||||
|
||||
## Configuration
|
||||
|
||||
Configuration is strict, explicit, centralized, and operator-oriented.
|
||||
@@ -128,6 +145,14 @@ Configuration is strict, explicit, centralized, and operator-oriented.
|
||||
- Empty configured values do not silently replace meaningful defaults.
|
||||
- Validation rejects invalid composition before stage execution where
|
||||
practical.
|
||||
- Root-owned imports and one selected profile resolve deterministically through
|
||||
the configuration owner; commands do not implement their own merge rules.
|
||||
- Canonical party rosters are campaign-owned. Their derived players projection
|
||||
and concrete character-family artifacts are resolved before runtime stages
|
||||
or adapters receive configuration.
|
||||
- Resume uses stage- or artifact-owned semantic evidence for observable
|
||||
result-affecting configuration; profile identity and an effective digest are
|
||||
provenance, never blanket cache keys.
|
||||
- Session templating remains narrow and deterministic rather than becoming a
|
||||
general configuration language.
|
||||
- Secret values are supplied indirectly and are not persisted in ordinary
|
||||
@@ -146,6 +171,10 @@ Canonical helpers own workspace, spool, cache, session, run, input, transcript,
|
||||
artifact, log, report, configuration, and publish-current paths. Callers must
|
||||
not reconstruct canonical paths through scattered string concatenation.
|
||||
|
||||
Reusable audio cache entries require a typed record that binds a confined,
|
||||
no-follow regular file and its digest to the selected remote object identity.
|
||||
Size alone and unqualified multipart ETags are not content-integrity evidence.
|
||||
|
||||
Artifact resolution is deterministic and manifest-aware. Producers materialize
|
||||
canonical outputs before reporting success, and consumers resolve declared
|
||||
artifact identities rather than infer files from unrelated directory contents.
|
||||
@@ -167,22 +196,38 @@ contracts belong under [Integrations](../integrations/).
|
||||
## Publish Commit Boundary
|
||||
|
||||
Publish has one explicit remote commit boundary. A remote run becomes current
|
||||
only after Narratio has successfully uploaded the run record, required published
|
||||
outputs, `current/manifest.json`, and finally `current/run_id.txt`.
|
||||
only after Narratio has successfully uploaded its immutable run-scoped objects,
|
||||
the immutable commit manifest, and finally the current commit pointer.
|
||||
|
||||
`current/run_id.txt` is the commit marker and must be written last. Failed,
|
||||
incomplete, skipped, or uncommitted publish attempts must not be presented as
|
||||
current remote state. Publish locks remain authoritative and are not bypassed by
|
||||
a forced run.
|
||||
`current/commit-pointer.json` is the sole mutable selector and must be written
|
||||
exactly once, last. Failed, incomplete, skipped, or uncommitted publish attempts
|
||||
must not be presented as current remote state. Publish locks remain authoritative
|
||||
and are not bypassed by a forced run. Mutable remote locks use provider-enforced
|
||||
generation preconditions and are revalidated immediately before pointer
|
||||
selection; loss of that check leaves the prior committed snapshot current.
|
||||
|
||||
Automatic local cleanup is permitted only after a successful publish commit,
|
||||
only when explicitly configured, and only through the path-safety guardrails.
|
||||
It is a durable local obligation bound to that committed run and its exact
|
||||
targets, not an inferred side effect of the current stage list. A cleanup
|
||||
failure makes the invocation incomplete while leaving the committed remote
|
||||
snapshot authoritative; later invocations resume the recorded obligation.
|
||||
|
||||
## Security, Privacy, And Diagnostics
|
||||
|
||||
Narratio handles private campaign material. Transcripts, prompts, generated
|
||||
artifacts, reports, logs, manifests, and diagnostic files are potentially
|
||||
sensitive.
|
||||
Narratio distinguishes ordinary workspace data from credentials. Campaign and
|
||||
session material—including manifests, transcripts, prompts, generated
|
||||
configuration, logs, reports, diagnostics, and Notarius artifacts—is
|
||||
intentionally shareable with the configured workspace group. API-key material
|
||||
is sensitive and is not covered by the ordinary workspace-sharing policy.
|
||||
|
||||
On POSIX systems, Narratio-created ordinary workspace directories converge on
|
||||
setgid `02775` and ordinary workspace files on `0664`, even when the caller's
|
||||
umask is restrictive. This preserves the existing workspace group for nested
|
||||
creation and atomic replacements without changing ownership. API-key storage
|
||||
uses a separate restrictive contract. On Windows, POSIX mode bits and setgid
|
||||
are not authoritative; operators must provide the equivalent shared-group and
|
||||
credential-restricted ACLs described in [Operations](../operations.md#workspace-permissions).
|
||||
|
||||
Raw secrets must not be stored in pipeline, campaign, or session YAML or written
|
||||
to manifests, logs, generated configuration, reports, publish metadata,
|
||||
|
||||
@@ -59,6 +59,8 @@ secret values.
|
||||
| --- | --- | --- | --- |
|
||||
| Product orientation and minimal end-to-end quickstart | `README.md` | What Narratio is, why it is useful, one shortest successful invocation, and links onward. | Complete command reference, configuration reference, operational procedures, implementation detail. |
|
||||
| Contributor entry point | `docs/development.md` | Task-oriented reading guide, minimal contributor orientation, baseline validation commands, and links to canonical docs. | Package inventory, architecture rules, subsystem behavior, detailed change recipes. |
|
||||
| Maintainer release procedure | `docs/release.md` | Version selection, candidate preparation and validation, guarded tag publication, release completion boundary, failure recovery, and optional asynchronous inspection. | Script implementation mechanics, current application contracts, and historical release summaries. |
|
||||
| Historical release summary | `docs/releases/<tag>.md` | Immutable summary, compatibility, upgrade, and changes for one released version. | Current maintainer procedure and current application contract details. |
|
||||
| Current application architecture | `docs/policy/architecture.md` | System shape, normative ownership, dependency direction, architectural boundaries, invariants, safety properties, and non-goals. | Concrete package inventory, implementation mechanics, contributor procedures, decision history, future work. |
|
||||
| Documentation organization | `docs/policy/documentation.md` | Documentation ownership, audience boundaries, maintenance rules, and ADR/document lifecycle. | Application architecture or product behavior. |
|
||||
| Testing policy | `docs/policy/testing.md` | Test philosophy, risk-based sufficiency, test boundaries, doubles, coverage guidance, regression-test policy, and criteria for adding, rewriting, or deleting tests. | Subsystem behavior, application contracts, subsystem-specific test inventories, and implementation plans. |
|
||||
|
||||
97
docs/release.md
Normal file
97
docs/release.md
Normal file
@@ -0,0 +1,97 @@
|
||||
# Releasing Narratio
|
||||
|
||||
This document is the maintainer procedure for creating a Narratio source and
|
||||
binary release. The synchronous release boundary is a successful push of one
|
||||
new tag to `origin`; Woodpecker and Gitea publication happen later and do not
|
||||
change that result.
|
||||
|
||||
## Choose a version and write its note
|
||||
|
||||
Narratio is past `v1.0.0`. Use an unused stable tag in the exact form
|
||||
`vMAJOR.MINOR.PATCH`:
|
||||
|
||||
- increment `MINOR` for backward-compatible features;
|
||||
- increment `PATCH` for backward-compatible fixes; and
|
||||
- reserve a new `MAJOR` for an intentional breaking documented contract.
|
||||
|
||||
Before preparing the candidate, create
|
||||
`docs/releases/vMAJOR.MINOR.PATCH.md` with this structure:
|
||||
|
||||
```markdown
|
||||
# Narratio vMAJOR.MINOR.PATCH
|
||||
|
||||
This release ...
|
||||
|
||||
## Summary
|
||||
|
||||
## Compatibility
|
||||
|
||||
## Upgrade
|
||||
|
||||
## Changes
|
||||
```
|
||||
|
||||
The compatibility section identifies relevant CLI, configuration, artifact,
|
||||
integration, or operating-contract changes. The upgrade section states the
|
||||
required operator action, or explicitly says that no special action is
|
||||
required. Release notes are immutable historical summaries; link to the
|
||||
current canonical documentation for detailed behavior.
|
||||
|
||||
Commit the note and all candidate changes, then use the ordinary development
|
||||
workflow to push that commit to `main`. Do not create a release tag before the
|
||||
candidate is committed and `origin/main` contains the exact same commit.
|
||||
|
||||
## Validate the candidate
|
||||
|
||||
Run the shared checker from any directory:
|
||||
|
||||
```sh
|
||||
scripts/check-release-candidate.sh vMAJOR.MINOR.PATCH
|
||||
```
|
||||
|
||||
It validates the version and matching note, module hygiene, formatting,
|
||||
whitespace, uncached tests, race tests, static checks, documentation, examples,
|
||||
and six official cross-build assets. It uses `GOWORK=off`, does not contact
|
||||
application services, CI, or Gitea, and does not create tags or modify tracked
|
||||
source. Fix any failure on `main`, commit it, push it normally, and rerun the
|
||||
checker.
|
||||
|
||||
The checker builds Linux, macOS, and Windows assets for `amd64` and `arm64`.
|
||||
Cross-builds prove compilation; they are not native macOS or Windows runtime
|
||||
evidence.
|
||||
|
||||
## Publish the tag
|
||||
|
||||
From a clean checkout on `main` whose `HEAD` equals `origin/main`, run:
|
||||
|
||||
```sh
|
||||
scripts/release.sh vMAJOR.MINOR.PATCH
|
||||
```
|
||||
|
||||
The command fetches and checks `origin/main`, re-runs candidate validation,
|
||||
then fetches and checks again before creating an explicitly unsigned lightweight
|
||||
tag for the originally recorded commit. It refuses dirty, divergent, changed,
|
||||
or already-tagged candidates. It pushes only:
|
||||
|
||||
```text
|
||||
refs/tags/vMAJOR.MINOR.PATCH:refs/tags/vMAJOR.MINOR.PATCH
|
||||
```
|
||||
|
||||
It never commits changes, pushes `main`, force-pushes, moves a tag, or pushes
|
||||
all tags. A successful push of that exact ref completes the release command;
|
||||
the command prints the tag and commit, then returns without waiting for CI,
|
||||
querying Gitea, downloading assets, or checking checksums.
|
||||
|
||||
If a failure occurs before the tag is created, correct the candidate on `main`
|
||||
and repeat validation. If the push fails after local tag creation, the local tag
|
||||
is intentionally retained for inspection and the command must not be retried
|
||||
blindly. Once the upstream tag has been pushed, it is immutable. Correct any
|
||||
defect or failed asynchronous publication with a new patch version, a new
|
||||
release note, and the complete procedure again.
|
||||
|
||||
## Optional asynchronous inspection
|
||||
|
||||
After a successful tag push, a human may later inspect the tag-triggered
|
||||
Woodpecker run and the corresponding Gitea release for binaries and checksums.
|
||||
This is optional follow-up only. Automated releasers must not wait for, poll,
|
||||
or treat CI/Gitea completion as a condition of the successful tag push.
|
||||
26
docs/releases/README.md
Normal file
26
docs/releases/README.md
Normal file
@@ -0,0 +1,26 @@
|
||||
# Release Notes
|
||||
|
||||
This directory contains immutable historical release notes for Narratio.
|
||||
Future notes are created with the matching stable version and use this minimum
|
||||
structure:
|
||||
|
||||
```markdown
|
||||
# Narratio vMAJOR.MINOR.PATCH
|
||||
|
||||
This release ...
|
||||
|
||||
## Summary
|
||||
|
||||
## Compatibility
|
||||
|
||||
## Upgrade
|
||||
|
||||
## Changes
|
||||
```
|
||||
|
||||
See the [release procedure](../release.md) for creating a candidate and tag.
|
||||
When asynchronous publication succeeds, the corresponding Gitea release is the
|
||||
canonical source for downloadable binaries and checksums.
|
||||
|
||||
- [v1.6.0](v1.6.0.md)
|
||||
- [v1.5.0](v1.5.0.md)
|
||||
47
docs/releases/v1.5.0.md
Normal file
47
docs/releases/v1.5.0.md
Normal file
@@ -0,0 +1,47 @@
|
||||
# Narratio v1.5.0
|
||||
|
||||
Narratio v1.5.0 makes repeated post-transcript artifact development faster and
|
||||
more explicit while retaining the fixed, stage-driven pipeline model.
|
||||
|
||||
## Highlights
|
||||
|
||||
- The canonical pipeline now completes deterministic rendering before
|
||||
extraction, cleanly separating transcript-generating stages from
|
||||
artifact-generating stages.
|
||||
- `narratio run` and `narratio session plan` accept inclusive `--from` and
|
||||
`--through` bounds. Excluded transcript stages are not executed or
|
||||
invalidated by a bounded artifact-regeneration run.
|
||||
- `narratio regenerate-artifacts SESSION` is an exact convenience alias for a
|
||||
forced run from `extract` through `analyze`, including focused
|
||||
`--artifacts` selections.
|
||||
- Configured Scriptorium artifacts now have independent,
|
||||
manifest-authoritative freshness. Narratio reuses validated current work,
|
||||
rebuilds stale prerequisites in dependency order, and persists successful,
|
||||
failed, and newly stale artifact state when an analysis invocation only
|
||||
partially succeeds.
|
||||
- Publish consumes only configured artifacts backed by current manifest
|
||||
evidence; incidental or tampered files are not promoted as current output.
|
||||
|
||||
## Reliability And Administration
|
||||
|
||||
- Bounded prerequisites are checked again under the session lock before any
|
||||
run mutation, closing a concurrent-run race.
|
||||
- Analysis fingerprints are stable across executable and configuration path
|
||||
changes and continue to cover only Narratio-observable semantic inputs.
|
||||
- Runner composition now carries one validated execution plan from command
|
||||
parsing through prerequisite validation, adapter composition, manifest
|
||||
recording, and stage execution.
|
||||
- `narratio version` reports the exact tag embedded in official release
|
||||
binaries; ordinary source builds report `dev`.
|
||||
|
||||
## Upgrade Notes
|
||||
|
||||
- Existing unbounded commands and direct `run-stage`, `analyze`, and `publish`
|
||||
workflows retain their meanings.
|
||||
- Manifests written before artifact-level analysis state remain readable.
|
||||
Legacy aggregate analysis success is not sufficient freshness evidence, so
|
||||
the first analysis evaluation after upgrading may regenerate configured
|
||||
artifacts once.
|
||||
- Narratio cannot observe executable contents or configuration, prompt,
|
||||
profile, module, and other files loaded privately by Scriptorium. Explicitly
|
||||
force affected artifacts after changing those private inputs.
|
||||
74
docs/releases/v1.6.0.md
Normal file
74
docs/releases/v1.6.0.md
Normal file
@@ -0,0 +1,74 @@
|
||||
# Narratio v1.6.0
|
||||
|
||||
Narratio v1.6.0 makes large pipeline configurations easier to organize,
|
||||
inspect, and vary while adding character-oriented artifact generation and a
|
||||
guarded, reproducible release procedure.
|
||||
|
||||
## Summary
|
||||
|
||||
Pipeline configuration can now be assembled from explicit additive imports and
|
||||
a selected production or testing profile. Campaigns can own a canonical,
|
||||
versioned party roster, and Scriptorium artifact families can expand one
|
||||
definition into concrete per-character artifacts, dependencies, variables, and
|
||||
publish rules.
|
||||
|
||||
New read-only configuration commands expose the fully resolved pipeline,
|
||||
source provenance, semantic digest, and profile differences before a session is
|
||||
run. Stage reuse now records configuration-sensitive semantic evidence so
|
||||
profile or configuration changes cannot silently reuse incompatible work.
|
||||
|
||||
## Compatibility
|
||||
|
||||
This is a backward-compatible feature release. Existing monolithic pipeline
|
||||
files, concrete Scriptorium artifacts, publish rules, and configurations
|
||||
without profiles remain supported. When profiles are declared and no explicit
|
||||
profile is selected, Narratio uses the configured production default.
|
||||
|
||||
An unversioned party file plus a separate players file remains available as an
|
||||
isolated legacy compatibility path, but it cannot drive artifact families. New
|
||||
campaigns and new party-oriented features should use the `narratio.party.v1`
|
||||
schema. Existing manifests remain readable; missing legacy semantic evidence
|
||||
is treated as stale rather than trusted.
|
||||
|
||||
Narratio continues to consume the documented Notarius D&D pipeline contract.
|
||||
The release workflow still cross-compiles Linux, macOS, and Windows binaries
|
||||
for `amd64` and `arm64`; cross-compilation is not native runtime evidence for
|
||||
macOS or Windows.
|
||||
|
||||
## Upgrade
|
||||
|
||||
No special action is required for existing monolithic configurations that do
|
||||
not adopt profiles or artifact families. On the first run after upgrading,
|
||||
stages recorded by older manifests may regenerate once because those records do
|
||||
not contain the new semantic configuration evidence.
|
||||
|
||||
To adopt the new configuration model, use the maintained
|
||||
`examples/production-testing` bundle as a migration reference: split stable
|
||||
settings into explicit imports, define a production default and optional
|
||||
testing profile, convert campaign party data to `narratio.party.v1`, remove the
|
||||
separate players file, and then introduce character artifact families. Review
|
||||
the result with `narratio config validate`, `config show`, `config sources`, and
|
||||
`config diff` before running a session.
|
||||
|
||||
Campaign, session, previous-session, and run identifiers must satisfy the
|
||||
documented portable identity grammar. Existing manifests or remote state with
|
||||
unsafe legacy identifiers must be migrated before use.
|
||||
|
||||
## Changes
|
||||
|
||||
- Added root-owned, non-recursive additive pipeline imports with strict,
|
||||
source-aware conflict detection.
|
||||
- Added named pipeline profiles with an explicit production default and
|
||||
deliberate command-line selection.
|
||||
- Added `config validate`, `config show`, `config sources`, and `config diff`
|
||||
for read-only inspection of resolved configuration and provenance.
|
||||
- Added the strict `narratio.party.v1` campaign roster, including stable
|
||||
character IDs, player and character names, optional aliases, and classes.
|
||||
- Added deterministic players derivation and canonical party delivery to
|
||||
downstream integrations.
|
||||
- Added character-oriented artifact families, corresponding member
|
||||
dependencies, family selection, and generated publish policies.
|
||||
- Added semantic configuration fingerprints and stage-specific resume checks
|
||||
across transcript and artifact stages.
|
||||
- Added shared release candidate, asset build, and guarded tag-publication
|
||||
scripts, with tag-triggered publication remaining asynchronous.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,332 +0,0 @@
|
||||
# Codebase Audit Plan
|
||||
|
||||
Status: proposed
|
||||
|
||||
## Purpose
|
||||
|
||||
This audit will evaluate Narratio for correctness, efficiency, maintainability,
|
||||
and test-suite value. It will identify defects and credible risks, duplicated or
|
||||
near-duplicated behavior, code that can be made smaller or more idiomatic, and
|
||||
complex code whose remaining invariants need focused explanation.
|
||||
|
||||
The audit is investigative. It should produce evidence-backed findings and a
|
||||
prioritized remediation backlog, not make opportunistic production changes as
|
||||
it proceeds. The [Audit Sequence](audit-sequence.md) assigns this scope to
|
||||
concrete execution stages.
|
||||
|
||||
## Authoritative Baseline
|
||||
|
||||
Review implemented behavior against its canonical owner rather than treating
|
||||
the current implementation or tests as the specification:
|
||||
|
||||
- [Architecture](../policy/architecture.md) for system boundaries, dependency
|
||||
direction, state and path ownership, safety properties, and pipeline
|
||||
invariants;
|
||||
- [Internal Overview](../internal/overview.md) and its focused internal
|
||||
documents for implemented ownership and mechanics;
|
||||
- [Testing Policy](../policy/testing.md) for risk-based sufficiency, durable
|
||||
boundaries, test-double guidance, and test lifecycle decisions;
|
||||
- the [CLI](../cli.md), [Configuration](../config.md),
|
||||
[Operations](../operations.md), and [integration contracts](../integrations/)
|
||||
for externally observable behavior; and
|
||||
- the [Documentation Policy](../policy/documentation.md) for canonical ownership
|
||||
and the distinction between current and proposed behavior.
|
||||
|
||||
Where code, tests, and documentation disagree, record the disagreement. Do not
|
||||
assume which one is wrong until the canonical contract and caller expectations
|
||||
have been traced.
|
||||
|
||||
## Audit Principles
|
||||
|
||||
1. Review correctness before cleanup. A shorter implementation is not an
|
||||
improvement if it weakens a state transition, safety check, or external
|
||||
contract.
|
||||
2. Trace behavior across boundaries. Narratio's most important properties often
|
||||
emerge from the interaction of application orchestration, stages, manifests,
|
||||
artifact resolution, filesystem operations, and adapters.
|
||||
3. Distinguish repeated syntax from repeated policy. Extract a helper only when
|
||||
the behavior has one stable owner and the shared abstraction makes that
|
||||
ownership clearer. Similar stage code may be intentionally explicit.
|
||||
4. Prefer narrow, idiomatic Go over generic frameworks. In particular, proposed
|
||||
refactors must preserve the explicit canonical stage sequence and must not
|
||||
turn Narratio into a workflow engine or a second configuration system for
|
||||
downstream tools.
|
||||
5. Optimize credible work. Flag repeated I/O, hashing, serialization, remote
|
||||
calls, subprocess work, allocation, or poor asymptotic behavior when the
|
||||
relevant path can matter. Require a benchmark or workload argument for
|
||||
performance changes whose benefit is not evident.
|
||||
6. Treat comments as explanations of intent. Recommend comments for invariants,
|
||||
ordering constraints, non-obvious failure policy, or security reasoning—not
|
||||
as narration of ordinary Go or a substitute for simplifying code.
|
||||
7. Judge tests as a suite. A test can be locally reasonable and still add no
|
||||
marginal protection, while a compact test can be inadequate for a
|
||||
consequential cross-component failure.
|
||||
|
||||
## Evidence And Finding Standard
|
||||
|
||||
Begin from a cleanly identified revision and record toolchain and platform
|
||||
assumptions. Use the code knowledge graph to find ownership, callers, callees,
|
||||
similarity candidates, high-complexity functions, and weakly protected
|
||||
boundaries. Confirm every candidate by reading the implementation, its focused
|
||||
tests, and the applicable contract. Text search and static analysis supplement
|
||||
the graph for literals, configuration, generated files, and patterns that are
|
||||
not modeled reliably.
|
||||
|
||||
Each finding should record:
|
||||
|
||||
- category: correctness defect, correctness risk, duplication, simplification,
|
||||
efficiency, architectural boundary, comment/clarity, or test-suite issue;
|
||||
- source locations and the affected contract or invariant;
|
||||
- concrete evidence and a realistic failure or maintenance scenario;
|
||||
- impact, likelihood, confidence, and estimated remediation scope separately;
|
||||
- the smallest plausible improvement and its intended owner;
|
||||
- tests that already protect the behavior, tests that should change or be
|
||||
added, and tests that may become redundant; and
|
||||
- dependencies on, or conflicts with, other findings.
|
||||
|
||||
Do not report a metric alone as a finding. Complexity, similarity, coverage,
|
||||
fan-in, file size, and test count are prioritization signals that require manual
|
||||
confirmation. Consolidate findings that share one root cause.
|
||||
|
||||
## Cross-Cutting Review Lenses
|
||||
|
||||
### Correctness And Pipeline Semantics
|
||||
|
||||
Construct an explicit lifecycle matrix for every stage outcome: first run,
|
||||
already-succeeded skip, self-skip, failure, interruption, forced replacement,
|
||||
non-resumable result, and successful rerun. Trace how each outcome changes the
|
||||
session manifest, invocation manifest, downstream stage state, artifacts,
|
||||
diagnostics, and cleanup eligibility.
|
||||
|
||||
Across the pipeline, verify:
|
||||
|
||||
- the registry exposes one deterministic canonical order;
|
||||
- each stage's declared inputs, outputs, configuration, adapters, and manifest
|
||||
effects agree with its implementation and focused documentation;
|
||||
- inputs are resolved through manifest and artifact contracts rather than
|
||||
incidental directory contents;
|
||||
- run-local outputs are fully validated before canonical materialization;
|
||||
- failure, cancellation, or process interruption cannot advertise partial work
|
||||
as successful;
|
||||
- force and changed outcomes invalidate exactly the intended succeeded
|
||||
downstream work;
|
||||
- repeated execution is idempotent where promised, and ordering is stable
|
||||
wherever maps, directory reads, remote listings, or dependency graphs are
|
||||
involved;
|
||||
- session, campaign, run, source, checksum, contract, and external provenance
|
||||
identities cannot be confused across runs; and
|
||||
- errors preserve useful causes and do not expose secrets or private content.
|
||||
|
||||
Use fault-oriented reasoning at durability boundaries: fail immediately before
|
||||
and after manifest saves, canonical renames, external process completion,
|
||||
uploads, current-manifest publication, the current-run commit marker, restore
|
||||
manifest installation, and cleanup. Determine which state is authoritative and
|
||||
whether the next invocation recovers safely.
|
||||
|
||||
### Duplication And Helper Ownership
|
||||
|
||||
Search for exact and semantic duplication in production and tests, including:
|
||||
|
||||
- repeated stage setup, input resolution, output validation, run-local
|
||||
materialization, metadata construction, and error adaptation;
|
||||
- repeated manifest create/load/save and session/run transition handling;
|
||||
- repeated adapter construction, timeout parsing, command execution, generated
|
||||
configuration, log handling, and output checks;
|
||||
- repeated source-ID, destination, remote-key, and path validation policy;
|
||||
- repeated sorting, deduplication, checksum, copy, and atomic-write mechanics;
|
||||
and
|
||||
- repeated test fixtures and assertions that encode the same policy at several
|
||||
layers.
|
||||
|
||||
For each candidate, decide whether it is coincidental similarity, a repeated
|
||||
mechanism, or duplicated policy. Recommend extraction only when the helper can
|
||||
have a clear package owner, a narrow contract, and callers that become easier
|
||||
to understand. Prefer an unexported local helper when sharing is package-local.
|
||||
Do not create a broad utility package, force unlike stage results into one data
|
||||
model, or move policy into storage/file-operation helpers.
|
||||
|
||||
Initial similarity and complexity signals should seed, but not predetermine,
|
||||
inspection of the single-stage command wrappers, session/run manifest
|
||||
persistence pairs, adapter constructors, Scriptorium operations, stage fakes,
|
||||
and common stage materialization paths.
|
||||
|
||||
### Simplification, Go Idioms, And Efficiency
|
||||
|
||||
Review long or branch-heavy functions for separable decisions, state
|
||||
transitions, or data transformations. Pay particular attention to orchestration,
|
||||
configuration validation, artifact dependency resolution, resume verification,
|
||||
restore/previous-cache planning, and analyze/publish selection logic. A useful
|
||||
refactor should reduce cognitive load while leaving the important ordering
|
||||
visible.
|
||||
|
||||
Check for:
|
||||
|
||||
- unnecessary nesting, defensive branches made unreachable by earlier
|
||||
validation, repeated normalization, and overly wide parameter lists;
|
||||
- interfaces defined for hypothetical extensibility rather than a demonstrated
|
||||
consumer boundary;
|
||||
- manual slice, map, string, error, and filesystem logic with a clearer standard
|
||||
library form;
|
||||
- incorrect or inconsistent `errors.Is`/`errors.As`, wrapping, context
|
||||
propagation, deferred cleanup, response-body closure, process waiting, and
|
||||
goroutine/channel ownership;
|
||||
- redundant filesystem scans, `stat`/checksum passes, whole-file buffering,
|
||||
copying, YAML/JSON round trips, sorting, remote listings, downloads, uploads,
|
||||
or adapter initialization;
|
||||
- linear searches nested in loops and repeated dependency or artifact lookup
|
||||
that should use an indexed map or a single planning pass;
|
||||
- unbounded concurrency, leaked work after cancellation, serialized independent
|
||||
work, and nondeterministic result collection; and
|
||||
- obsolete dependencies, portability assumptions, and platform-sensitive path
|
||||
or atomic-rename behavior.
|
||||
|
||||
Keep correctness and diagnosability ahead of micro-optimization. When a simpler
|
||||
algorithm changes performance characteristics, specify the representative
|
||||
input size and validation method.
|
||||
|
||||
### Comments And Local Explanation
|
||||
|
||||
Review high fan-in, high-complexity, security-sensitive, and commit-boundary
|
||||
code after likely simplifications have been identified. Add a comment
|
||||
recommendation when a maintainer needs to know why:
|
||||
|
||||
- state transitions or persistence operations occur in a specific order;
|
||||
- a stale record intentionally retains data while another transition clears it;
|
||||
- a path is checked more than once to resist traversal, symlink replacement, or
|
||||
time-of-check/time-of-use hazards;
|
||||
- an artifact is accepted only with particular manifest, checksum, contract, or
|
||||
provenance evidence;
|
||||
- a partial operation is intentionally not rolled back;
|
||||
- a remote pointer or local manifest must be installed last; or
|
||||
- concurrency, cancellation, compatibility, or downstream-tool behavior makes
|
||||
an apparently simpler approach unsafe.
|
||||
|
||||
Prefer a named helper, typed state, or smaller control flow when that removes the
|
||||
need for explanation. Check existing comments for stale claims as well as
|
||||
missing rationale.
|
||||
|
||||
### Test Suite Against The Canonical Policy
|
||||
|
||||
Build a risk-to-test matrix rather than auditing tests file by file in
|
||||
isolation. For each important behavior, identify its proper owner—parser,
|
||||
validator, domain package, adapter, orchestrator, CLI, integration, or end to
|
||||
end—and identify all tests that claim to protect it.
|
||||
|
||||
Evaluate:
|
||||
|
||||
- protection of data integrity, destructive operations, compatibility,
|
||||
security, concurrency, idempotency, recovery, and partial failure;
|
||||
- manifest transitions, force/invalidation, resume validation, atomic
|
||||
materialization, publish commit order, restore install order, and cleanup
|
||||
gates as assembled behaviors;
|
||||
- realistic HTTP, subprocess, filesystem, and object-store boundary behavior,
|
||||
including cancellation and malformed responses;
|
||||
- whether higher-level tests intentionally sample lower-level behavior or
|
||||
redundantly reproduce its full policy;
|
||||
- whether tests assert durable outcomes or private constants, exact error text,
|
||||
incidental paths, call choreography, or oversized snapshots;
|
||||
- whether real fast collaborators could replace elaborate doubles, and whether
|
||||
stateful fakes are realistic enough for the risk they protect;
|
||||
- fixture/helper duplication, oversized test cases, and setup that obscures the
|
||||
behavior under test without introducing a heavyweight test framework;
|
||||
- deterministic, offline, credential-free, order-independent execution and
|
||||
safe handling of environment and process-global state;
|
||||
- focused fuzz candidates in parsing, normalization, source IDs, remote/local
|
||||
path mapping, manifest decoding, and configuration boundaries; and
|
||||
- the presence and value of a small number of representative assembled
|
||||
workflows.
|
||||
|
||||
Use coverage only to locate unexpectedly weak consequential branches. Also
|
||||
inspect packages with extensive coverage for redundant tests and refactoring
|
||||
friction. For every proposed addition, deletion, or consolidation, state the
|
||||
realistic defect and marginal confidence involved.
|
||||
|
||||
The audit baseline should include the repository's canonical commands plus
|
||||
targeted diagnostic runs where supported:
|
||||
|
||||
```sh
|
||||
go test ./...
|
||||
go test -race ./...
|
||||
go vet ./...
|
||||
go build ./cmd/narratio
|
||||
```
|
||||
|
||||
Use focused repeated or shuffled runs to investigate state leakage and
|
||||
flakiness, and collect package/branch coverage for diagnosis. Review continuous
|
||||
integration to determine whether the appropriate offline validation is enforced;
|
||||
do not turn coverage percentage into a gate merely for this audit.
|
||||
|
||||
## Area-By-Area Inspection Map
|
||||
|
||||
| Area | Primary locations | What to inspect |
|
||||
| --- | --- | --- |
|
||||
| Process and application boundary | `cmd/narratio`, `internal/app` | Command dispatch, configuration selection, production composition, secret loading, lock lifetime, object-store initialization, context/error propagation, and separation of CLI reporting from orchestration policy. Review operator commands for consistent current-state authority and shared read-only mechanics. |
|
||||
| Stage registry and runner | `internal/stage/placeholders.go`, `internal/stage/stage.go`, `internal/app/planner.go`, `internal/app/runner.go`, `internal/app/run_stage.go` | Canonical order, action decisions, resume/force/self-skip/failure transitions, downstream invalidation, session/run manifest consistency, resource lifecycle, cleanup triggering, and opportunities to decompose the runner without hiding its state machine. |
|
||||
| Configuration | `internal/config` | Strict decoding, discovery and precedence, centralized defaults, normalization, templating, validation order, unknown fields, empty-value behavior, secret references, cross-field constraints, path confinement, deterministic errors, duplicated validator policy, and compatibility with maintained examples. |
|
||||
| Prepare and audio | `internal/stage/prepare.go`, `internal/audio`, `internal/previouscache` | Local/S3 exclusivity, cache and spool identity, partial downloads, checksum/reuse policy, previous-session required/optional planning, deterministic input records, clearing semantics, traversal safety, and avoiding repeated remote or filesystem work. |
|
||||
| Transcript stages | `internal/stage/transcribe.go`, `merge.go`, `polish.go`, `normalize.go`, `trim.go`, `render.go` | Contract parity across similar stages, bounded concurrency and cancellation, deterministic speaker/input ordering, run-local validation and canonical promotion, report/diagnostic classification, disabled behavior, and narrow opportunities for shared mechanics. |
|
||||
| Extraction | `internal/stage/extract.go`, `extract_resume.go`, `internal/adapters/notarius`, `internal/fileops/directory.go` | External receipt and lane validation, configuration fingerprint limits, immutable promotion, symlink/root replacement defenses, provenance and checksum checks, immediate and cross-invocation reuse, obsolete versus unsafe outcomes, failure residue, and whether dense verification logic can be clarified without weakening it. |
|
||||
| Analyze and artifact dependencies | `internal/stage/analyze.go`, `internal/artifacts`, `internal/artifactpolicy` | Source-family validation, runtime catalog state, enabled/selected/reused distinctions, topological ordering and cycle handling, required/optional inputs, local-only previous sources, deterministic metadata, repeated lookup/scanning, and ownership shared with config and publish. |
|
||||
| Publish and cleanup | `internal/stage/publish.go`, `internal/app/post_publish_cleanup.go`, `internal/app/cleanup_targets.go` | Prerequisite success, output selection, locks, required/optional behavior, exclusion rules, deterministic upload set, retry/idempotency implications, current-manifest then commit-marker ordering, metadata gates, and destructive path confinement. |
|
||||
| Manifest state | `internal/manifest` | Validation and backward compatibility, atomic persistence, timestamps, session/run identity, transition truth table, clearing versus retaining payload, create/load/save duplication, failure during dual-manifest updates, and whether state mutation has a single owner. |
|
||||
| Artifacts, paths, and policy | `internal/artifacts`, `internal/artifactpolicy`, `internal/pathsafe` | Canonical helper coverage, ad hoc reconstruction by callers, source-ID ownership, manifest-first resolution, extraction/current-state identity, destination normalization, stable ordering, typed missing-state errors, symlink/traversal defenses, and duplicate policy across config/stages/app. |
|
||||
| Restore | `internal/app/restore*.go`, `internal/previouscache`, `internal/audio` | Remote authority, confined mapping, deterministic plan actions, local conflict and force behavior, dry-run purity, temp-file installation, manifest-last ordering, partial failure/retry behavior, report accuracy, cache reuse, and shared current-state mechanics. |
|
||||
| File operations | `internal/fileops`, `internal/pathsafe`, local-store code in `internal/artifacts` | Atomic-write and promotion guarantees, permissions, close/sync/rename error handling, temp cleanup, same-filesystem assumptions, replacement policy, regular-file-only traversal, symlink and root-swap resistance, lock cleanup, and portability. |
|
||||
| External adapters and storage | `internal/adapters`, `internal/audio` | Transport isolation, shared subprocess mechanics versus adapter-specific policy, command/config duplication, quoting and working directories, timeouts/cancellation, stdout/stderr separation, HTTP body and retry behavior, S3 pagination/streaming/not-found mapping, credential independence, and external error adaptation. |
|
||||
| Shared models and diagnostics | `internal/artifactmodel`, `internal/contracts`, `internal/logging` | Serialization and validation invariants, unnecessary conversions, ownership of shared types, stable diagnostic structure, redaction, and whether small shared packages remain cohesive. |
|
||||
| Tests, examples, and automation | all `*_test.go`, `examples/`, `.woodpecker/` | Risk ownership, semantic duplication, fixture cost, policy-coupled assertions, realistic boundary tests, end-to-end sufficiency, default-suite isolation, example validation, diagnostic coverage, flakiness, runtime cost, and enforcement of canonical validation. |
|
||||
|
||||
## Narratio-Specific Cross-Boundary Scenarios
|
||||
|
||||
In addition to package-local review, trace these complete scenarios because a
|
||||
modular pipeline can look correct within every package while violating an
|
||||
end-to-end invariant:
|
||||
|
||||
1. A stage succeeds, its result becomes non-resumable, the rerun fails, and a
|
||||
later invocation decides what remains usable.
|
||||
2. An upstream forced or changed outcome interacts with already-succeeded,
|
||||
self-skipped, and disabled downstream stages.
|
||||
3. Extraction produces a valid immutable bundle, then configuration or
|
||||
transitive Notarius inputs change before analyze or publish.
|
||||
4. Previous-session state is published, restored or prepared into the local
|
||||
cache, and consumed by analyze without an unintended remote read.
|
||||
5. Publish fails at each upload boundary, especially between current manifest
|
||||
and current-run pointer, followed by status, restore, and retry.
|
||||
6. Restore encounters identical files, conflicting files, unsafe remote keys,
|
||||
cache hits, and a failure immediately before manifest installation.
|
||||
7. Automatic or manual cleanup is requested after skipped, failed, locked,
|
||||
partially uploaded, and fully committed publish outcomes.
|
||||
8. Cancellation reaches bounded transcription work, HTTP requests,
|
||||
subprocesses, object storage, and manifest reporting without leaks or false
|
||||
success.
|
||||
9. A configured artifact is disabled, unselected, reused, generated from
|
||||
another artifact, sourced from extraction, or sourced from a previous
|
||||
session, then filtered for publish.
|
||||
10. The same session is invoked concurrently, including lock contention and
|
||||
cleanup/release failures.
|
||||
|
||||
## Completion Criteria
|
||||
|
||||
The audit is complete when:
|
||||
|
||||
- every area in the inspection map has been reviewed against its canonical
|
||||
contracts and focused tests;
|
||||
- the stage lifecycle matrix and cross-boundary scenarios have explicit
|
||||
conclusions;
|
||||
- duplication candidates have been classified rather than merely counted;
|
||||
- simplification and performance recommendations explain their correctness
|
||||
constraints and expected benefit;
|
||||
- comment recommendations identify the non-obvious rationale to preserve;
|
||||
- the test suite has a risk-based sufficiency assessment, including gaps,
|
||||
redundancy, durability, execution properties, and automation;
|
||||
- findings are deduplicated, evidence-backed, and ranked by risk and dependency;
|
||||
and
|
||||
- unresolved questions and intentionally accepted risks are recorded rather
|
||||
than silently omitted.
|
||||
|
||||
## Execution
|
||||
|
||||
The [Audit Sequence](audit-sequence.md) is the canonical owner of execution
|
||||
order, stage boundaries, checkpoints, validation, and audit deliverables. This
|
||||
document remains the canonical owner of audit scope, review criteria, and the
|
||||
finding standard.
|
||||
@@ -1,734 +0,0 @@
|
||||
# Codebase Audit Sequence
|
||||
|
||||
Status: proposed
|
||||
|
||||
## Purpose And Relationship To The Audit Plan
|
||||
|
||||
This document turns the [Codebase Audit Plan](audit-plan.md) into a bounded,
|
||||
execution-ready sequence. The plan owns scope, review criteria, and the finding
|
||||
standard. This document owns ordering, dependencies, working records,
|
||||
validation, and exit gates.
|
||||
|
||||
The sequence is for investigation only. Do not mix production refactors or bug
|
||||
fixes into the audit. A confirmed urgent defect may justify stopping to request
|
||||
a separate remediation change, but its fix is not part of this sequence.
|
||||
|
||||
## Audit Run Records
|
||||
|
||||
Create `docs/roadmap/audit-findings.md` when the audit begins. It is the single
|
||||
working ledger and final audit report. Initialize it with:
|
||||
|
||||
- the audited revision, branch/worktree state, Go version, platform, and audit
|
||||
date;
|
||||
- baseline command results and timings;
|
||||
- an area coverage ledger;
|
||||
- the stage lifecycle matrix;
|
||||
- the cross-boundary scenario matrix from the audit plan;
|
||||
- a risk-to-test matrix;
|
||||
- candidate and confirmed finding registers; and
|
||||
- unresolved questions, accepted risks, and final conclusions.
|
||||
|
||||
Track each execution stage in the coverage ledger with one of `not_started`,
|
||||
`in_progress`, `complete`, or `blocked`. For a completed stage, record:
|
||||
|
||||
- contracts, packages, files, and important symbols reviewed;
|
||||
- graph traces, commands, tests, or other evidence used;
|
||||
- finding and candidate IDs produced;
|
||||
- explicit no-finding conclusions for reviewed high-risk behavior; and
|
||||
- follow-up questions assigned to later stages.
|
||||
|
||||
Use stable finding IDs with these prefixes:
|
||||
|
||||
| Prefix | Category |
|
||||
| --- | --- |
|
||||
| `COR` | Confirmed correctness defect |
|
||||
| `RSK` | Correctness or operational risk |
|
||||
| `ARC` | Ownership or architectural-boundary issue |
|
||||
| `DUP` | Duplicated mechanism or policy |
|
||||
| `SIM` | Simplification or idiomatic-Go opportunity |
|
||||
| `EFF` | Efficiency or resource-use issue |
|
||||
| `COM` | Missing, misleading, or stale explanatory comment |
|
||||
| `TST` | Test-suite gap, redundancy, brittleness, or execution issue |
|
||||
|
||||
Candidate IDs remain candidates until manual inspection confirms the behavior,
|
||||
contract, realistic scenario, and affected callers. Rejected candidates remain
|
||||
in a short classification log so later stages do not reopen them without new
|
||||
evidence.
|
||||
|
||||
## Execution Rules
|
||||
|
||||
1. Pin the audit to the revision recorded in Stage 0. If the worktree or HEAD
|
||||
changes, record the change and rerun every affected stage; do not silently
|
||||
combine evidence from different implementations.
|
||||
2. Use codebase graph search and call/data-flow traces before broad source
|
||||
search. Read the exact implementation, focused tests, and canonical contract
|
||||
before confirming a finding.
|
||||
3. Record test-policy observations during every behavior pass. Stage 12 owns the
|
||||
suite-wide conclusion but must not rediscover the suite from scratch.
|
||||
4. Record cross-area observations as candidates for the stage that owns the
|
||||
conclusion. Avoid producing duplicate findings from several review passes.
|
||||
5. Treat baseline failures as evidence, not automatic blockers. Continue when
|
||||
read-only inspection remains sound, and state the limitation. Stop only when
|
||||
the repository cannot be identified, required sources are unavailable, or a
|
||||
failure makes later evidence unreliable.
|
||||
6. Do not exercise a suspected destructive, credentialed, paid, or live-service
|
||||
path merely to prove a defect. Use source reasoning, existing safe fakes, or
|
||||
a narrowly controlled offline reproduction.
|
||||
7. Escalate a credible active data-loss, secret-exposure, or unsafe-cleanup
|
||||
defect immediately. Preserve the evidence and do not wait for final
|
||||
synthesis before reporting it.
|
||||
8. A stage is complete only when its exit gate is met. A package test passing is
|
||||
evidence, not proof that the review is complete.
|
||||
|
||||
## Sequence Overview
|
||||
|
||||
| Stage | Focus | Depends on | Primary result |
|
||||
| --- | --- | --- | --- |
|
||||
| 0 | Pin revision and establish baseline | None | Reproducible audit record |
|
||||
| 1 | Contract, boundary, and lifecycle map | 0 | Review matrices and ownership map |
|
||||
| 2 | Runner and manifest state machine | 1 | Lifecycle and dual-ledger conclusions |
|
||||
| 3 | Paths, artifacts, and filesystem safety | 1-2 | State/path authority and mutation conclusions |
|
||||
| 4 | Publish, remote commit, and cleanup | 2-3 | Commit-boundary and destructive-operation conclusions |
|
||||
| 5 | Restore and remote/previous state | 2-4 | Restore authority and recovery conclusions |
|
||||
| 6 | Configuration and application composition | 1-5 | Validation and wiring conclusions |
|
||||
| 7 | External adapters and shared support | 3, 6 | Boundary, cancellation, and resource conclusions |
|
||||
| 8 | Prepare and transcript-processing stages | 2-3, 6-7 | Ordinary stage-contract conclusions |
|
||||
| 9 | Extraction vertical slice | 2-3, 6-7 | Promotion, provenance, and resume conclusions |
|
||||
| 10 | Analyze and artifact dependency slice | 3, 6, 8-9 | Dependency and source-resolution conclusions |
|
||||
| 11 | Cross-codebase duplication, simplicity, efficiency, and comments | 2-10 | Classified maintainability candidates |
|
||||
| 12 | Test-suite policy audit | 2-11 | Risk-based suite sufficiency assessment |
|
||||
| 13 | Synthesis and audit closeout | 0-12 | Final deduplicated audit report |
|
||||
|
||||
Stages are intentionally ordered. Later stages may resolve candidates raised by
|
||||
earlier ones, but they must not invalidate an earlier stage silently. Return to
|
||||
the owning stage, update its coverage record, and note the new evidence.
|
||||
|
||||
## Stage 0: Pin Revision And Establish Baseline
|
||||
|
||||
### Entry
|
||||
|
||||
- Repository root and `docs/development.md` are available.
|
||||
- The audit plan and canonical policy documents can be read.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Record `git rev-parse HEAD`, branch/detached state, `git status --short`,
|
||||
`go version`, `go env GOOS GOARCH`, and the current date.
|
||||
2. Confirm that the code knowledge graph represents the recorded repository and
|
||||
revision; refresh the index if it is missing or stale.
|
||||
3. Capture the package/file/test inventory, entry points, architecture
|
||||
boundaries, high fan-in symbols, complexity signals, and similarity signals.
|
||||
4. Run the default offline baseline and record wall time and failures:
|
||||
|
||||
```sh
|
||||
go test -count=1 ./...
|
||||
go test -race -count=1 ./...
|
||||
go vet ./...
|
||||
```
|
||||
|
||||
5. Build into an external temporary directory so validation does not add a
|
||||
workspace binary:
|
||||
|
||||
```sh
|
||||
audit_build_dir="$(mktemp -d)"
|
||||
go build -o "$audit_build_dir/narratio" ./cmd/narratio
|
||||
go test -coverprofile="$audit_build_dir/coverage.out" ./...
|
||||
```
|
||||
|
||||
6. Inventory the repository's CI/release validation, maintained examples, fuzz
|
||||
tests, golden data, opt-in tests, and generated-test update mechanisms.
|
||||
|
||||
### Output
|
||||
|
||||
- Baseline and inventory sections in `audit-findings.md`.
|
||||
- Initial coverage ledger containing Stages 0-13.
|
||||
- Unconfirmed metric-driven candidates, clearly labeled as such.
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Revision and environment are reproducible.
|
||||
- Every baseline command has a recorded result.
|
||||
- Graph freshness is known.
|
||||
- Any limitation that affects later stages has an owner and disposition.
|
||||
|
||||
## Stage 1: Build The Contract, Boundary, And Lifecycle Map
|
||||
|
||||
### Entry
|
||||
|
||||
- Stage 0 is complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Read the architecture, internal overview, testing policy, focused internal
|
||||
documents, and the relevant CLI/configuration/operations/integration
|
||||
contracts using the development guide's routing rules.
|
||||
2. Map each package and important interface to its owned policy. Mark every
|
||||
cross-package dependency that appears to reverse or blur the intended
|
||||
direction for later confirmation.
|
||||
3. Build a stage-contract matrix with canonical order, declared inputs,
|
||||
outputs, configuration, adapters, skip behavior, resume validation,
|
||||
materialization boundary, manifest effects, and downstream invalidation.
|
||||
4. Build the lifecycle matrix required by the audit plan: first run,
|
||||
already-succeeded skip, self-skip, failure, interruption, forced replacement,
|
||||
non-resumable result, and successful rerun.
|
||||
5. Assign each of the ten cross-boundary scenarios in the audit plan to its
|
||||
primary execution stage and list supporting packages/tests.
|
||||
6. Seed the risk-to-test matrix with the intended test owner for each
|
||||
architectural invariant. Do not judge sufficiency yet.
|
||||
|
||||
### Output
|
||||
|
||||
- Package ownership, stage-contract, lifecycle, scenario, and preliminary
|
||||
risk-to-test matrices.
|
||||
- `ARC` and `RSK` candidates for apparent disagreements, without deciding from
|
||||
documentation alone which artifact is wrong.
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every area in the audit plan's inspection map has an assigned stage.
|
||||
- Every architectural invariant has an implementation owner and intended test
|
||||
owner.
|
||||
- Unknown or contradictory contracts are explicitly recorded.
|
||||
|
||||
## Stage 2: Audit The Runner And Manifest State Machine
|
||||
|
||||
### Entry
|
||||
|
||||
- Stage 1 matrices are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Trace the entry paths into full-run and single-stage execution through
|
||||
`internal/app/planner.go`, `runner.go`, `run_stage.go`, and related helpers.
|
||||
2. Inspect `internal/manifest` models, validation, session/run creation,
|
||||
loading, normalization, atomic saves, and all transition methods.
|
||||
3. Walk every lifecycle-matrix cell through both manifests. Verify clearing
|
||||
versus retention of outputs, diagnostics, generated configuration, metadata,
|
||||
errors, actions, timestamps, and downstream state.
|
||||
4. Reason about failures before and after each session-manifest and run-manifest
|
||||
save. Determine which disagreement states are possible and how a later
|
||||
invocation interprets them.
|
||||
5. Review force, changed-result, self-skip, failed-result, and non-resumable
|
||||
invalidation separately. Confirm behavior at the first and last canonical
|
||||
stage.
|
||||
6. Review session lock acquisition/release and concurrent invocation behavior,
|
||||
while leaving path implementation details to Stage 3.
|
||||
7. Classify the runner's complexity and repeated session/run persistence paths:
|
||||
state-machine clarity, justified explicitness, candidate local helpers, and
|
||||
comments that preserve ordering rationale.
|
||||
8. Review focused app/manifest tests against the matrix and add observations to
|
||||
the risk-to-test ledger.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/app ./internal/manifest
|
||||
go test -race -count=1 ./internal/app ./internal/manifest
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every lifecycle cell has a source-backed conclusion for both manifests.
|
||||
- Cross-boundary scenarios 1, 2, and the lock portion of 10 are resolved or
|
||||
carry explicit questions.
|
||||
- All runner/manifest candidates are confirmed, rejected, or assigned to a
|
||||
named later stage.
|
||||
|
||||
## Stage 3: Audit Paths, Artifacts, And Filesystem Safety
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 1-2 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Review `internal/artifacts`, `internal/artifactpolicy`, `internal/pathsafe`,
|
||||
`internal/fileops`, and local-store filesystem code.
|
||||
2. Inventory canonical path and key helpers, then search callers for ad hoc
|
||||
reconstruction, double normalization, mixed slash/filesystem semantics, or
|
||||
policy implemented outside its owner.
|
||||
3. Trace built-in, configured, extraction, previous-session, and current-state
|
||||
artifact resolution. Verify identity, checksum, contract, provenance,
|
||||
deterministic ordering, and typed missing-state behavior.
|
||||
4. Review atomic file writes, copies, directory promotion, temp cleanup,
|
||||
permission preservation, close/sync/rename errors, existing-destination
|
||||
behavior, same-filesystem assumptions, and platform sensitivity.
|
||||
5. Walk traversal, absolute path, broad root, symlink component, inspected-root
|
||||
replacement, non-regular file, and time-of-check/time-of-use scenarios.
|
||||
6. Confirm that low-level file/storage helpers receive explicit destinations
|
||||
and do not infer stage, campaign, session, run, or publish policy.
|
||||
7. Inspect lock-file implementation and cleanup errors to finish scenario 10.
|
||||
8. Record focused test ownership and gaps without duplicating Stage 2's state
|
||||
conclusions.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/artifacts ./internal/artifactpolicy ./internal/pathsafe ./internal/fileops
|
||||
go test -race -count=1 ./internal/artifacts ./internal/fileops
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every canonical path/key family has one identified owner.
|
||||
- Every material filesystem mutation has documented confinement and atomicity
|
||||
conclusions.
|
||||
- Scenario 10 is resolved.
|
||||
- Safety checks that appear repetitive are classified before any simplification
|
||||
recommendation is made.
|
||||
|
||||
## Stage 4: Audit Publish, Remote Commit, And Cleanup
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 2-3 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Trace publish from stage selection through object-store calls, manifest
|
||||
metadata, commit-marker publication, run completion, and post-publish
|
||||
cleanup.
|
||||
2. Verify prerequisite stage-state checks, selected/configured/extraction
|
||||
output resolution, required versus optional outputs, static and remote
|
||||
locks, run-file exclusions, previous-cache inclusion, and deterministic
|
||||
upload order.
|
||||
3. Enumerate failures before and after every upload. Prove that
|
||||
`current/run_id.txt` is written last and is the only remote-current commit
|
||||
point.
|
||||
4. Review retry/idempotency behavior, existing remote objects, partial uploads,
|
||||
pointer/manifest disagreement, and status/restore interpretation after each
|
||||
partial outcome.
|
||||
5. Trace automatic and manual cleanup gates. Confirm publish execution,
|
||||
`uploaded`, `current_pointer_written`, explicit policy, and confined targets
|
||||
are all required at the correct boundary.
|
||||
6. Confirm that `--force` cannot override publish locks or cleanup safety.
|
||||
7. Review duplication between publish planning, artifact destination policy,
|
||||
operator views, and cleanup metadata only after ownership is established.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/stage ./internal/app ./internal/artifacts ./internal/adapters/storage
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Cross-boundary scenarios 5 and 7 are resolved for every relevant failure
|
||||
boundary.
|
||||
- Remote-current authority and local-cleanup eligibility have explicit truth
|
||||
tables.
|
||||
- Publish findings distinguish stage policy from storage mechanics.
|
||||
|
||||
## Stage 5: Audit Restore And Remote/Previous State
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 2-4 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Trace restore discovery, planning, execution, reporting, audio
|
||||
materialization, and previous-cache planning through `internal/app`,
|
||||
`internal/artifacts`, `internal/previouscache`, `internal/audio`, and storage.
|
||||
2. Confirm remote pointer/manifest identity and campaign/session/run authority,
|
||||
including missing and inconsistent current state.
|
||||
3. Verify remote-to-local confinement, deterministic action ordering,
|
||||
`download`/`skip_same`/`conflict` decisions, force semantics, and dry-run
|
||||
purity.
|
||||
4. Walk failures during download, checksum or manifest validation, atomic
|
||||
install, report persistence, and the manifest-last boundary. Record the
|
||||
intentional lack of rollback and retry consequences.
|
||||
5. Review audio spool/cache identity, cache-hit verification, partial download
|
||||
behavior, and duplicate remote/filesystem work.
|
||||
6. Review previous-session requirement planning, required/optional behavior,
|
||||
identity checks, published-path fallback, and deterministic local mapping.
|
||||
7. Confirm which mechanics are shared with status/validate/operator commands
|
||||
and which caller-specific missing-state policies must remain separate.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/app ./internal/previouscache ./internal/audio ./internal/artifacts ./internal/adapters/storage
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Cross-boundary scenarios 4 and 6 are resolved through retry/recovery.
|
||||
- Restore authority, manifest-last installation, and partial-write behavior are
|
||||
explicit.
|
||||
- Previous-cache conclusions are ready for the prepare and analyze passes.
|
||||
|
||||
## Stage 6: Audit Configuration And Application Composition
|
||||
|
||||
### Entry
|
||||
|
||||
- Stage 1 is complete and Stages 2-5 have identified the policies that
|
||||
configuration and composition must supply.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Review `internal/config`, `cmd/narratio`, application command dispatch,
|
||||
configuration selection, secret-file environment loading, and production
|
||||
collaborator construction.
|
||||
2. Trace discovery, precedence, strict YAML decoding, defaults, empty values,
|
||||
normalization, session templating, and validation order across pipeline,
|
||||
campaign, and session configuration.
|
||||
3. Verify cross-field constraints for stage enablement, paths, timeouts,
|
||||
concurrency, artifacts, Notarius, Scriptorium, publish, storage, cleanup,
|
||||
audio, and previous-session behavior.
|
||||
4. Compare validation logic with maintained examples and the public
|
||||
configuration contract. Record contract drift rather than silently choosing
|
||||
code or docs.
|
||||
5. Check that filesystem secrets are loaded before the boundary that consumes
|
||||
them and are excluded from logs, manifests, reports, generated files, and
|
||||
errors.
|
||||
6. Review conditional construction of expensive/external collaborators and
|
||||
cleanup of anything with a lifecycle. Confirm test injection cannot create a
|
||||
behavior different from production composition.
|
||||
7. Classify repeated validators, path checks, timeout parsing, constructor
|
||||
wrappers, and single-stage command wrappers by policy owner.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/config ./internal/app ./cmd/narratio
|
||||
go vet ./...
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every operator-visible field used by audited behavior has a traced default,
|
||||
normalization, validation, and consumer.
|
||||
- Composition conclusions cover enabled and disabled stages without requiring
|
||||
live services or credentials.
|
||||
- Maintained examples have an explicit validity conclusion.
|
||||
|
||||
## Stage 7: Audit External Adapters And Shared Support
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 3 and 6 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Review `internal/adapters`, `internal/audio`, `internal/logging`,
|
||||
`internal/contracts`, and `internal/artifactmodel` at their public package
|
||||
boundaries.
|
||||
2. For each HTTP, subprocess, notification, and object-storage adapter, compare
|
||||
implementation with its integration contract and trace all production
|
||||
callers.
|
||||
3. Verify context cancellation, timeout ownership, process termination and
|
||||
waiting, goroutine/channel closure, HTTP response-body closure, retries,
|
||||
malformed responses, streaming, pagination, not-found mapping, and local
|
||||
file cleanup.
|
||||
4. Confirm command argument construction, working directory, environment,
|
||||
generated configuration, stdout/stderr separation, output validation, and
|
||||
external error adaptation stay inside the owning adapter.
|
||||
5. Compare subprocess implementations to the shared subprocess package. Classify
|
||||
repeated constructor/config/log/output mechanics separately from
|
||||
adapter-specific protocol policy.
|
||||
6. Review fakes for realistic state and concurrency behavior, but defer their
|
||||
suite-wide value judgment to Stage 12.
|
||||
7. Check shared models for avoidable conversions, stable serialization,
|
||||
validation ownership, and redaction-sensitive diagnostic fields.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/adapters/... ./internal/audio ./internal/logging ./internal/contracts ./internal/artifactmodel
|
||||
go test -race -count=1 ./internal/adapters/... ./internal/audio
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every external resource has an explicit acquisition, cancellation, and
|
||||
release conclusion.
|
||||
- Transport types and protocol policy have not leaked into stages.
|
||||
- Adapter duplication candidates identify the correct shared or specific
|
||||
owner.
|
||||
|
||||
## Stage 8: Audit Prepare And Transcript-Processing Stages
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 2-3 and 6-7 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Review `prepare`, `transcribe`, `merge`, `polish`, `normalize`, `trim`, and
|
||||
`render` as vertical slices from resolved configuration and manifest input
|
||||
through adapter call, run-local output, validation, canonical
|
||||
materialization, and recorded result.
|
||||
2. Verify each implementation against the Stage 1 contract matrix and focused
|
||||
internal document. Record any undeclared input, output, diagnostic, config,
|
||||
adapter, or skip/failure behavior.
|
||||
3. For prepare, confirm local/S3 exclusivity, stable input copying,
|
||||
previous-cache clearing/hydration, and deterministic manifest input records.
|
||||
4. For transcribe, confirm unique speaker identities, bounded runtime
|
||||
concurrency, cancellation, deterministic result ordering, adapter-returned
|
||||
path identity, and partial failure behavior.
|
||||
5. For transformation/render stages, confirm manifest-first resolution,
|
||||
run-local paths, schema/report validation, disabled/default behavior,
|
||||
canonical promotion, and diagnostic-versus-artifact classification.
|
||||
6. Compare similar stage implementations for shared mechanisms only after
|
||||
listing meaningful differences. Avoid a generic stage framework.
|
||||
7. Add stage-focused test ownership, gaps, and redundancy candidates to the
|
||||
risk-to-test matrix.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/stage ./internal/audio ./internal/previouscache ./internal/adapters/whisperx ./internal/adapters/seriatim ./internal/adapters/audita ./internal/adapters/scriptorium
|
||||
go test -race -count=1 ./internal/stage ./internal/audio
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every reviewed stage has a completed contract-matrix row.
|
||||
- Cross-boundary scenario 8 is resolved for transcription and subprocess-backed
|
||||
transformation stages.
|
||||
- Similarity candidates are classified as intentional explicitness, local
|
||||
helper candidates, or shared-owner findings.
|
||||
|
||||
## Stage 9: Audit The Extraction Vertical Slice
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 2-3 and 6-7 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Trace extraction from configuration validation and composition through
|
||||
transcript resolution, invocation fingerprint, Notarius execution, receipt
|
||||
and lane validation, directory promotion, manifest recording, catalog
|
||||
hydration, resume validation, analyze, and publish consumers.
|
||||
2. Verify run-local isolation, regular-file and confined-index requirements,
|
||||
required-lane policy, contract/provenance construction, checksum timing,
|
||||
immutable destination identity, and no-replacement promotion.
|
||||
3. Enumerate failures before and after subprocess completion, receipt parsing,
|
||||
payload inspection, promotion, and manifest persistence. Determine what
|
||||
remains diagnostic, durable, advertised, and reusable.
|
||||
4. Walk every resume validation branch. Distinguish obsolete/missing outcomes
|
||||
that trigger rerun from unsafe conditions that must stop execution.
|
||||
5. Evaluate the fingerprint's intentionally observable and unobservable inputs
|
||||
against documentation and force guidance.
|
||||
6. Review the dense validation code for named sub-decisions and comments while
|
||||
preserving the visible security proof and check ordering.
|
||||
7. Confirm focused tests cover immediate reuse, cross-invocation reuse,
|
||||
configuration change, payload tampering, provenance mismatch, symlinks/root
|
||||
replacement, failure residue, and downstream invalidation at the correct
|
||||
layers.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/stage ./internal/artifacts ./internal/fileops ./internal/adapters/notarius ./internal/app
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Cross-boundary scenario 3 is resolved, including transitive-input limits.
|
||||
- Promotion, advertisement, and resume each have a distinct authority and
|
||||
failure conclusion.
|
||||
- Every proposed simplification states which security or compatibility checks
|
||||
it preserves.
|
||||
|
||||
## Stage 10: Audit Analyze And Artifact Dependencies
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 3, 6, 8, and 9 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Trace all analyze source families from configuration validation through
|
||||
runtime catalog registration, availability, resolution, Scriptorium
|
||||
execution/reuse, materialization, metadata, and publish selection.
|
||||
2. Verify enabled, selected, executable, reused, generated, and unavailable
|
||||
states are distinct and deterministic.
|
||||
3. Review configured-artifact dependency validation and runtime topological
|
||||
ordering for cycles, missing dependencies, stable ordering, and consistency
|
||||
between configuration and execution.
|
||||
4. Confirm required/optional behavior and guidance for built-in transcripts,
|
||||
prepared stable inputs, configured artifacts, extraction sources, and
|
||||
previous-session sources.
|
||||
5. Prove previous-session resolution is local-only during analyze and that
|
||||
disabled artifacts are reused only under the documented conditions.
|
||||
6. Inspect repeated resolution branches, parameter width, nested lookup, and
|
||||
ordering work for a smaller representation or indexed plan without merging
|
||||
distinct source policies.
|
||||
7. Review tests for each state transition and source family at the narrowest
|
||||
stable owner, noting semantic duplication across config, artifacts, stage,
|
||||
publish, and assembled runner tests.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/stage ./internal/artifacts ./internal/artifactpolicy ./internal/config ./internal/adapters/scriptorium ./internal/app
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Cross-boundary scenario 9 is resolved for every source family and selection
|
||||
state.
|
||||
- Dependency ordering and source availability have explicit determinism and
|
||||
complexity conclusions.
|
||||
- Config, artifact-policy, catalog, stage, and publish ownership is unambiguous
|
||||
or represented by an `ARC` finding.
|
||||
|
||||
## Stage 11: Audit Duplication, Simplicity, Efficiency, And Comments
|
||||
|
||||
### Entry
|
||||
|
||||
- Behavior stages 2-10 are complete, so structural candidates can be judged
|
||||
against known contracts.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Rerun graph similarity, complexity, fan-in/fan-out, call-path, loop-depth,
|
||||
scan-in-loop, and change-coupling analyses on production code. Add targeted
|
||||
text/static searches for patterns the graph cannot represent.
|
||||
2. Revisit all `DUP`, `SIM`, `EFF`, and `COM` candidates collected earlier.
|
||||
Search for additional occurrences and trace all callers before assigning an
|
||||
owner.
|
||||
3. For duplication, classify coincidental syntax, shared mechanism, duplicated
|
||||
policy, or deliberately explicit security/state logic. Propose only the
|
||||
narrowest helper that improves ownership and comprehension.
|
||||
4. For complexity, sketch the smaller control flow or data model and verify it
|
||||
leaves state transitions, validation order, and commit boundaries visible.
|
||||
5. For efficiency, state the input scale or call frequency, current and proposed
|
||||
complexity/I/O behavior, expected benefit, and benchmark or measurement
|
||||
needed. Reject micro-optimizations without a credible workload.
|
||||
6. Review standard-library usage, errors, slices/maps, allocations, copying,
|
||||
sorting, serialization, filesystem passes, adapter initialization, remote
|
||||
calls, goroutines/channels, and interface breadth across the complete codebase.
|
||||
7. Review comments only after simplification decisions. Recommend why-comments
|
||||
for remaining invariants, compatibility limits, safety checks, partial
|
||||
failure, and ordering; flag comments that restate code or no longer match it.
|
||||
8. Check dependencies and platform assumptions for clear correctness,
|
||||
portability, complexity, or maintenance consequences.
|
||||
|
||||
### Validation
|
||||
|
||||
- Run focused package tests for any behavior used to disprove or confirm a
|
||||
candidate.
|
||||
- Run existing benchmarks where relevant. Propose a benchmark rather than
|
||||
inventing performance claims when representative measurement is absent.
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every structural candidate is confirmed, rejected with a reason, or merged
|
||||
into a stronger root-cause finding.
|
||||
- No helper recommendation creates a generic workflow abstraction or moves
|
||||
policy into a low-level utility.
|
||||
- Every efficiency finding has a credible workload and validation method.
|
||||
- Every comment finding states the non-obvious rationale that should be
|
||||
preserved.
|
||||
|
||||
## Stage 12: Audit The Test Suite Against Policy
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 2-11 have populated the risk-to-test matrix and test observations.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Complete the risk-to-test matrix. For every consequential invariant, list
|
||||
the current tests, proper owner, protected defect, missing failure modes, and
|
||||
overlap with other layers.
|
||||
2. Review tests by behavior cluster rather than filename: parsing/validation,
|
||||
domain/state, filesystem, adapters, orchestration, CLI, integration, and
|
||||
representative assembled workflows.
|
||||
3. Classify gaps for data integrity, destructive operations, compatibility,
|
||||
security, concurrency, idempotency, recovery, cancellation, and partial
|
||||
success. Confirm the gap is not credibly protected elsewhere.
|
||||
4. Classify redundancy and brittleness: private constants/defaults, exact error
|
||||
wording, incidental formatting/paths, mock choreography, oversized
|
||||
snapshots, helper-level duplication, and the same policy repeated across
|
||||
layers.
|
||||
5. Review doubles using the policy order: real deterministic collaborator,
|
||||
stateful fake, stub, then mock when interaction is contractual. Check that
|
||||
fakes model the failure and state semantics used by the tests.
|
||||
6. Inspect test helpers and large test functions for simplification and
|
||||
meaningful table-driven boundaries without creating a fixture framework
|
||||
whose maintenance cost exceeds its value.
|
||||
7. Review determinism and isolation: credentials, network access, paid APIs,
|
||||
environment, working directory, clocks, randomness, ports, temp paths,
|
||||
process-global state, ordering, cleanup, and parallel execution.
|
||||
8. Use coverage to investigate consequential weak branches, not as a score.
|
||||
Review heavily covered behavior for marginal-value duplication as well.
|
||||
9. Identify focused fuzz opportunities for parsers, YAML/JSON normalization,
|
||||
source IDs, confined paths, remote/local mapping, and manifest decoding.
|
||||
10. Compare local requirements with `.woodpecker/` and other automation. Record
|
||||
missing enforcement as a risk/cost decision, not an assumption that every
|
||||
diagnostic command belongs in CI.
|
||||
11. Investigate order dependence and flakiness with bounded runs, recording
|
||||
runtime and any reproducible seed:
|
||||
|
||||
```sh
|
||||
go test -shuffle=on -count=3 ./...
|
||||
go test -race -shuffle=on -count=1 ./...
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every important risk has a sufficiency conclusion and one intended test
|
||||
owner.
|
||||
- Every proposed test addition names the realistic defect and marginal value.
|
||||
- Every deletion/consolidation names the stronger remaining protection.
|
||||
- Default-suite determinism, offline behavior, runtime, flakiness, and CI
|
||||
enforcement have explicit conclusions.
|
||||
|
||||
## Stage 13: Synthesize And Close The Audit
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 0-12 meet their exit gates or have explicitly accepted limitations.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Reconcile candidates and findings across stages. Merge shared root causes and
|
||||
remove repeated symptoms while retaining all affected locations and
|
||||
contracts.
|
||||
2. Recheck every confirmed finding against current source, callers, tests, and
|
||||
canonical documentation. Downgrade or reject anything supported only by a
|
||||
metric or hypothetical preference.
|
||||
3. Rank impact, likelihood, confidence, and remediation scope separately. Order
|
||||
the recommended backlog by dependency: correctness/data safety first,
|
||||
architectural ownership next, then simplification/duplication, tests,
|
||||
efficiency, and comments where they remain necessary.
|
||||
4. Record positive conclusions for high-risk areas where the current design and
|
||||
tests are sufficient. The report should not imply that only defective areas
|
||||
were reviewed.
|
||||
5. Reconcile the area coverage ledger, lifecycle matrix, cross-boundary scenario
|
||||
matrix, and risk-to-test matrix with the audit plan's completion criteria.
|
||||
6. Record any accepted risks, ambiguous contracts, environmental limitations,
|
||||
and deferred investigations with an explicit rationale and owner.
|
||||
7. Check whether HEAD or the worktree changed since Stage 0. Rerun affected
|
||||
stages or clearly pin the report to the original revision.
|
||||
8. Validate the report and roadmap document links and run `git diff --check`.
|
||||
If implementation changed during the audit, rerun the full Stage 0 validation
|
||||
baseline against the final audited revision.
|
||||
|
||||
### Final Deliverable
|
||||
|
||||
`docs/roadmap/audit-findings.md` must contain:
|
||||
|
||||
- an executive assessment without unsupported quality scores;
|
||||
- the audited revision and validation baseline;
|
||||
- coverage and scenario completion summaries;
|
||||
- confirmed findings ordered by dependency and risk;
|
||||
- rejected candidate themes where their recurrence would otherwise waste work;
|
||||
- the test-suite sufficiency assessment;
|
||||
- positive conclusions and accepted risks; and
|
||||
- a recommended remediation order, without implementing the remediation.
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every completion criterion in the audit plan is satisfied or explicitly
|
||||
marked limited with rationale.
|
||||
- Every finding is evidence-backed, deduplicated, actionable, and assigned a
|
||||
stable ID.
|
||||
- No production change is included in the audit output.
|
||||
- The report is sufficient to prepare a separate remediation sequence without
|
||||
repeating discovery.
|
||||
@@ -71,6 +71,36 @@ Safe fix:
|
||||
|
||||
Relevant reference: [Configuration](./config.md).
|
||||
|
||||
## Unexpected imported or profile value
|
||||
|
||||
Symptom:
|
||||
|
||||
- an effective configuration value differs from the root file, or a duplicate
|
||||
ownership/configuration error is hard to locate.
|
||||
|
||||
Diagnostics:
|
||||
|
||||
```bash
|
||||
narratio config sources --config /path/pipeline.yml --profile testing
|
||||
```
|
||||
|
||||
Add `--campaign` or `--campaign-file` when the pipeline has party-driven
|
||||
artifact families. The output identifies each effective logical field's root,
|
||||
import, profile, default, campaign, party, or family source without printing
|
||||
the field value or credential contents.
|
||||
|
||||
Safe fix:
|
||||
|
||||
- move a duplicated base field so it has one owner;
|
||||
- correct the selected profile or its overlay; or
|
||||
- correct the campaign party/family declaration that owns generated values.
|
||||
|
||||
To review what would actually change before switching profiles, run `config
|
||||
diff` with the same pipeline and campaign selectors. It compares normalized
|
||||
effective values rather than YAML formatting or source-file layout.
|
||||
|
||||
Relevant reference: [Configuration inspection](./config.md#read-only-effective-pipeline-inspection).
|
||||
|
||||
## Audio mode conflict
|
||||
|
||||
Symptom:
|
||||
@@ -117,6 +147,40 @@ Safe fix:
|
||||
|
||||
Relevant reference: [CLI artifact selection](./cli.md).
|
||||
|
||||
## Bounded run prerequisite is unusable
|
||||
|
||||
Symptom:
|
||||
|
||||
- `run` or `session plan` reports that a prerequisite stage is absent or has a
|
||||
pending, running, failed, stale, or interrupted status before the selected
|
||||
start.
|
||||
|
||||
Likely cause:
|
||||
|
||||
- `--from` excludes upstream work that has not reached the terminal
|
||||
`succeeded` or `skipped` state in the session manifest.
|
||||
|
||||
Diagnostics:
|
||||
|
||||
```bash
|
||||
narratio session status 2026-04-04
|
||||
narratio session plan 2026-04-04 --from render --through analyze
|
||||
```
|
||||
|
||||
Safe fix:
|
||||
|
||||
- widen the bounded range to include the first reported stage, or recover that
|
||||
stage explicitly with `run-stage` before retrying. The failed check does not
|
||||
create a run record or modify the manifest. Narratio does not resume-validate
|
||||
excluded prefix stages, and stages after `--through` are not prerequisites.
|
||||
|
||||
If prerequisite statuses are terminal but a selected stage reports a missing,
|
||||
unsafe, or checksum-inconsistent artifact, repair the artifact at the stage
|
||||
that owns it; do not edit the manifest to bypass the selected stage's concrete
|
||||
input validation.
|
||||
|
||||
Relevant reference: [Operations: Stage Execution and Continuation Behavior](./operations.md#stage-execution-and-continuation-behavior).
|
||||
|
||||
## Notarius executable missing
|
||||
|
||||
Symptom:
|
||||
@@ -158,6 +222,66 @@ is expected audit state, not a signal to relink the old bundle manually.
|
||||
|
||||
Relevant reference: [Operations: Extraction Workflow](./operations.md#extraction-workflow).
|
||||
|
||||
## Prepared Notarius reference missing or inconsistent
|
||||
|
||||
Symptom:
|
||||
|
||||
- extraction or resume validation reports that a configured reference source is
|
||||
unavailable, unsafe, empty, or checksum-inconsistent and recommends
|
||||
`prepare --force`.
|
||||
|
||||
Likely causes:
|
||||
|
||||
- `prepare` has not run since the campaign/session stable input changed;
|
||||
- the configured source file is missing;
|
||||
- a prepared `inputs/` file or its manifest record was modified independently;
|
||||
- a spell-catalog binding exists without an effective `spell_catalog_file`.
|
||||
|
||||
Diagnostics:
|
||||
|
||||
```bash
|
||||
narratio session status 2026-04-04
|
||||
narratio session validate 2026-04-04
|
||||
```
|
||||
|
||||
Safe fix:
|
||||
|
||||
- correct the campaign/session input path, then refresh canonical prepared
|
||||
evidence before extraction:
|
||||
|
||||
```bash
|
||||
narratio run-stage prepare 2026-04-04 --force
|
||||
```
|
||||
|
||||
Do not point Notarius directly at the original source path or edit the manifest
|
||||
checksum. Relevant references: [Notarius reference configuration](./config.md#notarius-reference-bindings)
|
||||
and [Operations: Extraction Workflow](./operations.md#extraction-workflow).
|
||||
|
||||
## Notarius reference selector or generated-handoff collision
|
||||
|
||||
Symptom:
|
||||
|
||||
- Notarius exits nonzero with an undeclared reference-slot, incompatible media,
|
||||
or external/generated reference collision error.
|
||||
|
||||
Likely causes:
|
||||
|
||||
- a selector does not identify a slot declared by the selected Notarius target;
|
||||
- a prepared file does not satisfy that slot's Notarius media contract; or
|
||||
- a CLI binding attempts to replace a same-run generated D&D handoff.
|
||||
|
||||
Safe fix:
|
||||
|
||||
- compare external bindings with the selected Notarius pipeline's canonical
|
||||
consumer documentation;
|
||||
- keep only campaign-owned external slots on the CLI; and
|
||||
- leave registry, scene, combat, and occurrence handoffs to Notarius pipeline
|
||||
composition.
|
||||
|
||||
Narratio validates selector structure and prepared evidence, while Notarius
|
||||
owns slot declarations, media compatibility, and generated-handoff conflicts.
|
||||
Relevant reference: [Notarius integration](./integrations/notarius.md).
|
||||
|
||||
## Atomic Notarius promotion unsupported
|
||||
|
||||
Symptom:
|
||||
@@ -198,7 +322,7 @@ Safe fix:
|
||||
|
||||
- compare installed Notarius output with the canonical Notarius contracts,
|
||||
including receipt `index_file: index.json` and index management names
|
||||
`manifest.json`, `rejected.json`, and `warnings.json`; align
|
||||
`manifest.json`, `rejected.json`, `warnings.json`, and `diagnostics.json`; align
|
||||
`pipeline.notarius` constraints and rerun. Do not bypass confinement or schema
|
||||
checks.
|
||||
|
||||
@@ -230,6 +354,8 @@ Likely causes:
|
||||
|
||||
- the executable/config path, pipeline ID, timeout, working directory, or
|
||||
configured output contracts changed;
|
||||
- a configured prepared reference selector, source, path, checksum, or byte
|
||||
size changed;
|
||||
- the durable bundle, index, lane set, provenance, regular-file status, or
|
||||
checksum no longer validates.
|
||||
|
||||
@@ -252,12 +378,110 @@ Safe fix:
|
||||
narratio run-stage extract 2026-04-04 --force
|
||||
```
|
||||
|
||||
Narratio fingerprints its invocation contract, not the contents of transitive
|
||||
Notarius inputs. Always force extraction after changing them; downstream
|
||||
Narratio fingerprints its invocation contract and prepared Narratio reference
|
||||
identities, not the contents of other transitive Notarius inputs. Always force
|
||||
extraction after changing those external inputs; downstream
|
||||
successful stages are then marked stale normally.
|
||||
|
||||
Relevant reference: [Operations: Extraction Workflow](./operations.md#extraction-workflow).
|
||||
|
||||
## Analysis artifact evidence is not current
|
||||
|
||||
Symptom:
|
||||
|
||||
- ordinary continuation or `session plan` schedules one or more configured
|
||||
artifacts even though a canonical output file exists; or
|
||||
- publish reports a configured artifact source unavailable.
|
||||
|
||||
Likely causes:
|
||||
|
||||
- the per-artifact record is stale, missing, failed, unselected, malformed, or
|
||||
from the legacy aggregate-only manifest contract;
|
||||
- a configured prompt/profile, dependency, input identity, output path, or
|
||||
effective variable changed; or
|
||||
- the recorded output is missing, unsafe, empty, or has a size/checksum that no
|
||||
longer matches its manifest evidence.
|
||||
|
||||
Diagnostics:
|
||||
|
||||
```bash
|
||||
narratio session status 2026-04-04
|
||||
narratio session artifacts 2026-04-04
|
||||
narratio session plan 2026-04-04 --from analyze --through analyze
|
||||
```
|
||||
|
||||
Safe fix:
|
||||
|
||||
- investigate unexpected path or checksum changes as possible tampering;
|
||||
- otherwise let the selected analyze work rerun, or explicitly regenerate only
|
||||
the affected targets; and
|
||||
- never edit the fingerprint/checksum in the manifest or copy an old file into
|
||||
the canonical path as a substitute for current evidence.
|
||||
|
||||
```bash
|
||||
narratio analyze 2026-04-04 --artifacts session_recap
|
||||
```
|
||||
|
||||
Relevant references: [Operations: Artifact Selection](./operations.md#artifact-selection)
|
||||
and [Artifact Internals](./internal/artifacts.md#resolution-rules).
|
||||
|
||||
## Legacy aggregate analysis requires regeneration
|
||||
|
||||
Symptom:
|
||||
|
||||
- a manifest from an older Narratio version reports aggregate analyze success
|
||||
and the old files are present, but configured artifact sources remain
|
||||
unavailable.
|
||||
|
||||
Likely cause:
|
||||
|
||||
- the manifest has no supported per-artifact analyze state. Aggregate output
|
||||
lists do not establish current configured-artifact authority.
|
||||
|
||||
Safe fix:
|
||||
|
||||
- regenerate the required artifacts. A partial selection makes only its
|
||||
targets and prerequisites eligible for current state; unselected legacy
|
||||
files intentionally remain unavailable. Run full analysis later when every
|
||||
enabled configured artifact must become current.
|
||||
|
||||
```bash
|
||||
narratio analyze 2026-04-04 --artifacts session_recap
|
||||
narratio analyze 2026-04-04
|
||||
```
|
||||
|
||||
After current records exist, inspect them and publish explicitly. Do not delete
|
||||
the legacy files merely to influence selection; availability is manifest-owned.
|
||||
|
||||
Relevant references: [Operations: Stage Execution and Continuation Behavior](./operations.md#stage-execution-and-continuation-behavior)
|
||||
and [Manifest Internals](./internal/manifest.md#analyze-owned-artifact-state).
|
||||
|
||||
## Scriptorium private input changed without a rerun
|
||||
|
||||
Symptom:
|
||||
|
||||
- a prompt, profile, imported configuration file, executable, or other input
|
||||
loaded privately by Scriptorium changed, but Narratio still considers an
|
||||
artifact current.
|
||||
|
||||
Likely cause:
|
||||
|
||||
- analysis fingerprints cover Narratio-observable semantic identities, not
|
||||
executable contents or arbitrary files and transitive configuration that
|
||||
Scriptorium loads behind its configured paths and identifiers.
|
||||
|
||||
Safe fix:
|
||||
|
||||
- explicitly force the affected target after changing an unobserved private
|
||||
input. Force applies to explicit targets; current prerequisites remain
|
||||
reusable unless selected themselves.
|
||||
|
||||
```bash
|
||||
narratio analyze 2026-04-04 --artifacts session_recap
|
||||
```
|
||||
|
||||
Relevant reference: [Analyze Internals](./internal/stage-analyze.md#invariants).
|
||||
|
||||
## Previous-session artifact input missing
|
||||
|
||||
Symptom:
|
||||
@@ -299,7 +523,7 @@ Symptom:
|
||||
Likely causes:
|
||||
|
||||
- another process is running for the same session;
|
||||
- stale lock left by interrupted process.
|
||||
- a process still holds the operating-system lock while it is shutting down.
|
||||
|
||||
Diagnostics:
|
||||
|
||||
@@ -311,7 +535,8 @@ ps aux | grep narratio
|
||||
Safe fix:
|
||||
|
||||
- wait for active process completion;
|
||||
- remove stale lock only after confirming no live process owns it.
|
||||
- retry after an interrupted holder has exited; the kernel releases its lock
|
||||
even though the `.lock` metadata file remains for inspection.
|
||||
|
||||
Relevant reference: [Operations: Local State Layout](./operations.md#local-state-layout).
|
||||
|
||||
@@ -435,7 +660,7 @@ Diagnostics:
|
||||
|
||||
```bash
|
||||
ls -la /path/to/secrets_dir
|
||||
env | grep -E 'OBJECT_STORAGE|AWS|AUDITA|SCRIPTORIUM'
|
||||
env | sed 's/=.*//' | grep -E 'OBJECT_STORAGE|AWS|AUDITA|SCRIPTORIUM'
|
||||
```
|
||||
|
||||
Safe fix:
|
||||
|
||||
@@ -11,6 +11,18 @@ in the [configuration reference](../docs/config.md).
|
||||
WhisperX URL.
|
||||
- [Production-shaped pipeline](pipeline.production.yml): S3 storage, publish,
|
||||
external tools, and configured Scriptorium artifacts.
|
||||
- [Production/testing split bundle](production-testing/pipeline.yml): explicit
|
||||
`conf.d` imports, a production default, and selectable production/testing
|
||||
overlays. It also demonstrates canonical-party artifact families and a
|
||||
testing-only disabled artifact. Validate it with:
|
||||
|
||||
```sh
|
||||
narratio config validate --config examples/production-testing/pipeline.yml --campaign-file examples/campaigns/sample-campaign/campaign.yml
|
||||
narratio config diff production testing --config examples/production-testing/pipeline.yml --campaign-file examples/campaigns/sample-campaign/campaign.yml
|
||||
```
|
||||
|
||||
`config show` and `config sources` accept the same selectors and remain
|
||||
read-only.
|
||||
- [Full annotated pipeline](pipeline.full.annotated.yml): every implemented
|
||||
pipeline section with explanatory comments.
|
||||
- [Extraction subset pipeline](pipeline.extraction-subset.yml): a focused
|
||||
@@ -37,9 +49,11 @@ with the sample campaign and a compatible local- or S3-audio session.
|
||||
- The sample campaign references its local
|
||||
[speakers](campaigns/sample-campaign/speakers.yml),
|
||||
[autocorrect](campaigns/sample-campaign/autocorrect.yml),
|
||||
[glossary](campaigns/sample-campaign/glossary.yml),
|
||||
[players](campaigns/sample-campaign/players.yml), and
|
||||
[party](campaigns/sample-campaign/party.yml) fixtures.
|
||||
[glossary](campaigns/sample-campaign/glossary.yml), and canonical
|
||||
[party](campaigns/sample-campaign/party.yml) fixture, plus an optional
|
||||
[spell-catalog overlay](campaigns/sample-campaign/spell_catalog.json) that
|
||||
follows the Notarius v0.6 contract. Narratio derives the players projection
|
||||
from this party source; the campaign deliberately has no `players_file`.
|
||||
- [Sample speaker audio](audio/sample-speaker.flac) is a text placeholder that
|
||||
reserves the expected filename and directory shape. Replace it with a real
|
||||
FLAC file before running transcription.
|
||||
|
||||
@@ -4,5 +4,5 @@ inputs:
|
||||
speakers_file: ./speakers.yml
|
||||
autocorrect_file: ./autocorrect.yml
|
||||
glossary_file: ./glossary.yml
|
||||
players_file: ./players.yml
|
||||
party_file: ./party.yml
|
||||
spell_catalog_file: ./spell_catalog.json
|
||||
|
||||
@@ -1,2 +1,26 @@
|
||||
- name: Example Hero
|
||||
type: pc
|
||||
schema_version: narratio.party.v1
|
||||
|
||||
characters:
|
||||
arannis:
|
||||
player:
|
||||
name: Rowan Hale
|
||||
character:
|
||||
name: Arannis
|
||||
alias:
|
||||
- Ari
|
||||
- The Grey Owl
|
||||
classes:
|
||||
- name: wizard
|
||||
level: 8
|
||||
brenna:
|
||||
player:
|
||||
name: Rowan Hale
|
||||
character:
|
||||
name: Brenna
|
||||
alias:
|
||||
- Shield of Dawn
|
||||
classes:
|
||||
- name: paladin
|
||||
level: 6
|
||||
- name: warlock
|
||||
level: 2
|
||||
|
||||
@@ -1,2 +0,0 @@
|
||||
- name: Example Player
|
||||
role: player
|
||||
18
examples/campaigns/sample-campaign/spell_catalog.json
Normal file
18
examples/campaigns/sample-campaign/spell_catalog.json
Normal file
@@ -0,0 +1,18 @@
|
||||
{
|
||||
"schema_version": "notarius.dnd.spell-catalog-overlay.v1",
|
||||
"catalogs": [
|
||||
{
|
||||
"id": "narratio.sample-campaign",
|
||||
"ruleset": "dnd-5e-2014",
|
||||
"source": {
|
||||
"title": "Narratio sample campaign spell names"
|
||||
},
|
||||
"spells": [
|
||||
{
|
||||
"name": "Aegis of Emberfall",
|
||||
"aliases": ["Emberfall Aegis"]
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -14,6 +14,11 @@ notarius:
|
||||
config_path: /usr/local/etc/notarius/config.yml
|
||||
pipeline_id: dnd-session
|
||||
timeout: 3h
|
||||
references:
|
||||
glossary: narratio.input.glossary
|
||||
party: narratio.input.party
|
||||
players: narratio.input.players
|
||||
spell_catalog: narratio.input.spell_catalog
|
||||
outputs:
|
||||
npc_registry:
|
||||
lane_id: npc-registry
|
||||
@@ -52,4 +57,3 @@ scriptorium:
|
||||
scenes:
|
||||
source: narratio.extraction.scene_descriptions
|
||||
required: true
|
||||
|
||||
|
||||
@@ -12,7 +12,7 @@ workspace:
|
||||
# env_dir: ./secrets
|
||||
|
||||
storage:
|
||||
# Optional storage backend selector; use "s3" for publish + S3 audio workflows.
|
||||
# Defaults to "local". Use "s3" explicitly for publish + S3 audio workflows.
|
||||
backend: s3
|
||||
s3:
|
||||
# Required when using S3 audio or S3 publish uploads.
|
||||
@@ -136,6 +136,13 @@ notarius:
|
||||
pipeline_id: dnd-session
|
||||
timeout: 3h
|
||||
working_directory: /usr/local/etc/notarius
|
||||
# External campaign references use prepared Narratio source IDs. Omit an
|
||||
# optional binding when the selected Notarius pipeline does not need it.
|
||||
references:
|
||||
glossary: narratio.input.glossary
|
||||
party: narratio.input.party
|
||||
players: narratio.input.players
|
||||
spell_catalog: narratio.input.spell_catalog
|
||||
# Each key creates source narratio.extraction.<key>. These constraints match
|
||||
# the current Notarius D&D lane contracts; update them with Notarius.
|
||||
outputs:
|
||||
@@ -260,7 +267,5 @@ scriptorium:
|
||||
output_kind: player_handout
|
||||
|
||||
notification:
|
||||
# Optional notification settings.
|
||||
backend: ""
|
||||
recipient: ""
|
||||
timeout: 30s
|
||||
# No delivery provider is currently implemented.
|
||||
mode: noop
|
||||
|
||||
@@ -125,4 +125,4 @@ scriptorium:
|
||||
output_kind: player_handout
|
||||
|
||||
notification:
|
||||
timeout: 30s
|
||||
mode: noop
|
||||
|
||||
32
examples/production-testing/conf.d/artifacts.yml
Normal file
32
examples/production-testing/conf.d/artifacts.yml
Normal file
@@ -0,0 +1,32 @@
|
||||
scriptorium:
|
||||
binary: scriptorium
|
||||
config_path: ./scriptorium/config.yml
|
||||
artifact_families:
|
||||
character_meta:
|
||||
enabled: true
|
||||
for_each: party.characters
|
||||
prompt_id: dnd.character_meta
|
||||
output_path_pattern: artifacts/characters/{character_id}/meta.md
|
||||
member_vars:
|
||||
character_id: character_id
|
||||
character_name: character.name
|
||||
player_name: player.name
|
||||
class_summary: character.class_summary
|
||||
character_items:
|
||||
enabled: true
|
||||
for_each: party.characters
|
||||
prompt_id: dnd.character_items
|
||||
output_path_pattern: artifacts/characters/{character_id}/items.md
|
||||
member_dependencies: [character_meta]
|
||||
inputs:
|
||||
character_meta:
|
||||
source: narratio.member_artifact.character_meta
|
||||
required: true
|
||||
member_vars:
|
||||
character_id: character_id
|
||||
character_name: character.name
|
||||
aliases: character.alias_summary
|
||||
publish:
|
||||
enabled: true
|
||||
required: false
|
||||
dest_pattern: artifacts/characters/{character_id}/items.md
|
||||
12
examples/production-testing/conf.d/platform.yml
Normal file
12
examples/production-testing/conf.d/platform.yml
Normal file
@@ -0,0 +1,12 @@
|
||||
workspace:
|
||||
root: ./workspace
|
||||
|
||||
campaigns:
|
||||
root: ../../campaigns
|
||||
default_campaign_id: sample-campaign
|
||||
|
||||
cache:
|
||||
root: ./cache
|
||||
|
||||
spool:
|
||||
root: ./spool
|
||||
7
examples/production-testing/conf.d/publish.yml
Normal file
7
examples/production-testing/conf.d/publish.yml
Normal file
@@ -0,0 +1,7 @@
|
||||
publish:
|
||||
enabled: true
|
||||
upload_run: false
|
||||
outputs:
|
||||
- source: narratio.transcript.final_markdown
|
||||
dest: transcripts/final.md
|
||||
required: true
|
||||
2
examples/production-testing/conf.d/storage.yml
Normal file
2
examples/production-testing/conf.d/storage.yml
Normal file
@@ -0,0 +1,2 @@
|
||||
storage:
|
||||
backend: local
|
||||
16
examples/production-testing/conf.d/transcript.yml
Normal file
16
examples/production-testing/conf.d/transcript.yml
Normal file
@@ -0,0 +1,16 @@
|
||||
whisperx:
|
||||
transcribe_url: https://transcription.example.com/transcribe
|
||||
language: en
|
||||
|
||||
seriatim:
|
||||
binary: seriatim
|
||||
output_schema: seriatim-intermediate
|
||||
|
||||
audita:
|
||||
binary: audita
|
||||
modules: [glossary, grammar]
|
||||
output_schema: audita-v1
|
||||
|
||||
normalize:
|
||||
output_path: transcripts/final.json
|
||||
output_schema: seriatim-intermediate
|
||||
15
examples/production-testing/pipeline.yml
Normal file
15
examples/production-testing/pipeline.yml
Normal file
@@ -0,0 +1,15 @@
|
||||
# Copyable production/testing pipeline entry point. Every fragment is named
|
||||
# explicitly; Narratio never scans conf.d automatically.
|
||||
composition:
|
||||
imports:
|
||||
- conf.d/platform.yml
|
||||
- conf.d/storage.yml
|
||||
- conf.d/transcript.yml
|
||||
- conf.d/artifacts.yml
|
||||
- conf.d/publish.yml
|
||||
default_profile: production
|
||||
profiles:
|
||||
production:
|
||||
overlay: profiles/production.yml
|
||||
testing:
|
||||
overlay: profiles/testing.yml
|
||||
10
examples/production-testing/profiles/production.yml
Normal file
10
examples/production-testing/profiles/production.yml
Normal file
@@ -0,0 +1,10 @@
|
||||
audita:
|
||||
model: narratio-production-model-placeholder
|
||||
validation_model: narratio-production-validator-placeholder
|
||||
|
||||
scriptorium:
|
||||
artifact_families:
|
||||
character_meta:
|
||||
profile_id: production-placeholder
|
||||
character_items:
|
||||
profile_id: production-placeholder
|
||||
14
examples/production-testing/profiles/testing.yml
Normal file
14
examples/production-testing/profiles/testing.yml
Normal file
@@ -0,0 +1,14 @@
|
||||
audita:
|
||||
model: narratio-testing-model-placeholder
|
||||
validation_model: narratio-testing-validator-placeholder
|
||||
|
||||
scriptorium:
|
||||
artifacts:
|
||||
testing_notes:
|
||||
enabled: false
|
||||
output_path: artifacts/testing-notes.md
|
||||
artifact_families:
|
||||
character_meta:
|
||||
profile_id: testing-placeholder
|
||||
character_items:
|
||||
profile_id: testing-placeholder
|
||||
@@ -4,10 +4,10 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
// NoopRunner is a deterministic no-op audita adapter.
|
||||
@@ -84,17 +84,17 @@ func materializePlaceholders(req PolishRequest) error {
|
||||
"merged_transcript_path": req.MergedTranscriptPath,
|
||||
"output_path": req.OutputProcessedPath,
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("audita noop/fake stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("audita noop/fake stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("audita noop/fake stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("audita noop/fake stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
@@ -117,14 +117,14 @@ func writeJSONIfRequested(path string, payload any) error {
|
||||
if path == "" {
|
||||
return nil
|
||||
}
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
if err := fileops.EnsureWorkspaceDirectory(filepath.Dir(path)); err != nil {
|
||||
return fmt.Errorf("create parent directory %q: %w", filepath.Dir(path), err)
|
||||
}
|
||||
data, err := json.Marshal(payload)
|
||||
if err != nil {
|
||||
return fmt.Errorf("marshal placeholder json for %q: %w", path, err)
|
||||
}
|
||||
if err := subprocess.WriteFileAtomic(path, data, 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(path, data, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write placeholder json %q: %w", path, err)
|
||||
}
|
||||
return nil
|
||||
|
||||
@@ -6,8 +6,6 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
// TODO: implement a real Audita subprocess/service adapter.
|
||||
|
||||
// Runner is the adapter boundary for audita polish invocations.
|
||||
type Runner interface {
|
||||
Run(ctx context.Context, req PolishRequest) (PolishResult, error)
|
||||
@@ -22,16 +20,6 @@ type PolishRequest struct {
|
||||
ReportPath string
|
||||
WorkDir string
|
||||
Modules []string
|
||||
BaseURL string
|
||||
Model string
|
||||
TranscriptDescription string
|
||||
ConfigPath string
|
||||
OutputSchema string
|
||||
WorkDirRetention string
|
||||
TotalLLMConcurrency *int
|
||||
ProposalLLMConcurrency *int
|
||||
ValidationModel string
|
||||
ValidationLLMConcurrency *int
|
||||
StdoutLogPath string
|
||||
StderrLogPath string
|
||||
}
|
||||
|
||||
@@ -11,8 +11,15 @@ import (
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
// MaxProcessedOutputBytes bounds Audita's processed-transcript JSON result.
|
||||
const MaxProcessedOutputBytes int64 = 64 * 1024 * 1024
|
||||
|
||||
// MaxReportOutputBytes bounds Audita's optional report JSON result.
|
||||
const MaxReportOutputBytes int64 = 16 * 1024 * 1024
|
||||
|
||||
// SubprocessRunnerConfig defines deterministic settings for Audita CLI execution.
|
||||
type SubprocessRunnerConfig struct {
|
||||
Binary string
|
||||
@@ -211,6 +218,7 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
|
||||
Args: args,
|
||||
Timeout: r.timeout,
|
||||
EnvOverrides: env,
|
||||
DiagnosticOwner: "audita",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
@@ -370,13 +378,13 @@ func (r *SubprocessRunner) writeInvocationConfig(req PolishRequest, args []strin
|
||||
"credential_env_var": r.llmAPIKeyEnv,
|
||||
"credential_present": credentialPresent,
|
||||
}
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
func validateProcessedOutput(path string) error {
|
||||
data, err := os.ReadFile(path)
|
||||
data, err := readAuditaResult(path, MaxProcessedOutputBytes, "processed transcript")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read file: %w", err)
|
||||
return err
|
||||
}
|
||||
|
||||
var payload map[string]any
|
||||
@@ -405,9 +413,9 @@ func addSubprocessStreamHint(message string, runErr error) string {
|
||||
}
|
||||
|
||||
func validateJSONFile(path string) error {
|
||||
data, err := os.ReadFile(path)
|
||||
data, err := readAuditaResult(path, MaxReportOutputBytes, "report")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read file: %w", err)
|
||||
return err
|
||||
}
|
||||
var v any
|
||||
if err := json.Unmarshal(data, &v); err != nil {
|
||||
@@ -415,3 +423,11 @@ func validateJSONFile(path string) error {
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func readAuditaResult(path string, limit int64, category string) ([]byte, error) {
|
||||
data, err := fileops.ReadRegularFile(path, limit)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("audita %s result exceeds or cannot be read within %d-byte limit: %w", category, limit, err)
|
||||
}
|
||||
return data, nil
|
||||
}
|
||||
|
||||
@@ -189,7 +189,7 @@ func TestSubprocessRunnerUnconfiguredCredentialEnvOmitsCredential(t *testing.T)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSubprocessRunnerInheritsParentEnvironment(t *testing.T) {
|
||||
func TestSubprocessRunnerOmitsUnspecifiedParentEnvironment(t *testing.T) {
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("helper wrapper script uses /bin/sh")
|
||||
}
|
||||
@@ -214,8 +214,8 @@ func TestSubprocessRunnerInheritsParentEnvironment(t *testing.T) {
|
||||
}
|
||||
|
||||
rec := readAuditaHelperRecord(t, recordPath)
|
||||
if rec.Env["AUDITA_INHERITED_MARKER"] != "inherited-from-parent" {
|
||||
t.Fatalf("AUDITA_INHERITED_MARKER = %q, want inherited-from-parent", rec.Env["AUDITA_INHERITED_MARKER"])
|
||||
if rec.Env["AUDITA_INHERITED_MARKER"] != "" {
|
||||
t.Fatalf("AUDITA_INHERITED_MARKER = %q, want omitted from the child environment", rec.Env["AUDITA_INHERITED_MARKER"])
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -14,7 +14,9 @@ func (f *FakeRunner) Run(ctx context.Context, req RunRequest) (RunResult, error)
|
||||
if err := ctx.Err(); err != nil {
|
||||
return RunResult{}, err
|
||||
}
|
||||
f.Requests = append(f.Requests, req)
|
||||
copyRequest := req
|
||||
copyRequest.References = append([]ReferenceBinding(nil), req.References...)
|
||||
f.Requests = append(f.Requests, copyRequest)
|
||||
if f.Err != nil {
|
||||
return RunResult{}, f.Err
|
||||
}
|
||||
|
||||
@@ -6,13 +6,20 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
const ReceiptSchemaVersion = "notarius.run-result.v1"
|
||||
const ReceiptSchemaVersion = "notarius.run-result.v2"
|
||||
|
||||
// Runner is the adapter boundary for a complete Notarius pipeline invocation.
|
||||
type Runner interface {
|
||||
Run(ctx context.Context, req RunRequest) (RunResult, error)
|
||||
}
|
||||
|
||||
// ReferenceBinding maps one normalized Notarius selector to an absolute
|
||||
// external reference path.
|
||||
type ReferenceBinding struct {
|
||||
Selector string
|
||||
Path string
|
||||
}
|
||||
|
||||
// RunRequest contains the resolved inputs and diagnostic destinations for one invocation.
|
||||
type RunRequest struct {
|
||||
Binary string
|
||||
@@ -24,6 +31,7 @@ type RunRequest struct {
|
||||
ReceiptPath string
|
||||
LogPath string
|
||||
Timeout time.Duration
|
||||
References []ReferenceBinding
|
||||
}
|
||||
|
||||
// Receipt is the transport-neutral successful run receipt.
|
||||
@@ -35,11 +43,31 @@ type Receipt struct {
|
||||
IndexFile string
|
||||
NormalizedOutputCount int
|
||||
RejectedOutputCount int
|
||||
WarningCount int
|
||||
WarningGroupCount int
|
||||
WarningOccurrenceCount int
|
||||
DiagnosticGroupCount int
|
||||
DiagnosticOccurrenceCount int
|
||||
DiagnosticsTruncated bool
|
||||
ValidationStatus string
|
||||
ValidationSummaries []ValidationSummary
|
||||
DebugDirectory string
|
||||
}
|
||||
|
||||
// ValidationSummary retains the bounded outcome of one Notarius producer result.
|
||||
type ValidationSummary struct {
|
||||
Stage string
|
||||
StepID string
|
||||
LaneID string
|
||||
ModuleKey string
|
||||
ChunkID string
|
||||
Status string
|
||||
RejectingValidators []string
|
||||
ReasonCodes []string
|
||||
IncompleteValidators []string
|
||||
ProducerAttemptCount int
|
||||
TerminalAction string
|
||||
}
|
||||
|
||||
// LaneDescriptor identifies one normalized lane payload discovered through the index.
|
||||
type LaneDescriptor struct {
|
||||
LaneID string
|
||||
@@ -72,6 +100,8 @@ type Index struct {
|
||||
RejectedPath string
|
||||
WarningsFile string
|
||||
WarningsPath string
|
||||
DiagnosticsFile string
|
||||
DiagnosticsPath string
|
||||
Lanes []LaneDescriptor
|
||||
ChunkMap *PipelineDescriptor
|
||||
EvidenceContext *PipelineDescriptor
|
||||
@@ -90,8 +120,29 @@ type RejectionSummary struct {
|
||||
|
||||
// WarningSummary retains structured warning identity without free-form messages.
|
||||
type WarningSummary struct {
|
||||
Scope string
|
||||
Disposition string
|
||||
Category string
|
||||
ReasonCode string
|
||||
Origin DiagnosticOrigin
|
||||
OccurrenceCount int
|
||||
}
|
||||
|
||||
// DiagnosticOrigin identifies the framework-owned pipeline location of a finding.
|
||||
type DiagnosticOrigin struct {
|
||||
Stage string
|
||||
StepID string
|
||||
LaneID string
|
||||
ModuleKey string
|
||||
ValidatorKey string
|
||||
}
|
||||
|
||||
// DiagnosticSummary retains bounded advisory or observation group metadata.
|
||||
type DiagnosticSummary struct {
|
||||
Disposition string
|
||||
Category string
|
||||
ReasonCode string
|
||||
Origin DiagnosticOrigin
|
||||
OccurrenceCount int
|
||||
}
|
||||
|
||||
// RunResult describes a successfully decoded and validated Notarius bundle.
|
||||
@@ -105,4 +156,5 @@ type RunResult struct {
|
||||
Duration time.Duration
|
||||
Rejections []RejectionSummary
|
||||
Warnings []WarningSummary
|
||||
Diagnostics []DiagnosticSummary
|
||||
}
|
||||
|
||||
@@ -5,12 +5,13 @@ import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/notariusref"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/pathsafe"
|
||||
)
|
||||
|
||||
@@ -22,6 +23,12 @@ const (
|
||||
canonicalManifestFile = "manifest.json"
|
||||
canonicalRejectedFile = "rejected.json"
|
||||
canonicalWarningsFile = "warnings.json"
|
||||
canonicalDiagnosticsFile = "diagnostics.json"
|
||||
warningsSchemaVersion = "notarius.warnings.v2"
|
||||
diagnosticsSchemaVersion = "notarius.diagnostics.v1"
|
||||
maxWarningGroups = 128
|
||||
maxDiagnosticGroups = 256
|
||||
maxFindingSamples = 3
|
||||
)
|
||||
|
||||
type subprocessRun func(context.Context, subprocess.RunRequest) (subprocess.RunResult, error)
|
||||
@@ -41,7 +48,8 @@ func (r *SubprocessRunner) Run(ctx context.Context, req RunRequest) (RunResult,
|
||||
if r == nil || r.run == nil {
|
||||
return RunResult{}, fmt.Errorf("notarius subprocess runner is nil")
|
||||
}
|
||||
if err := validateRunRequest(req); err != nil {
|
||||
references, err := validateRunRequest(req)
|
||||
if err != nil {
|
||||
return RunResult{}, err
|
||||
}
|
||||
|
||||
@@ -50,13 +58,17 @@ func (r *SubprocessRunner) Run(ctx context.Context, req RunRequest) (RunResult,
|
||||
"--config", req.ConfigPath,
|
||||
"--input", req.InputPath,
|
||||
"--output-dir", req.OutputRoot,
|
||||
"--json",
|
||||
}
|
||||
for _, reference := range references {
|
||||
args = append(args, "--reference", reference.Selector+"="+reference.Path)
|
||||
}
|
||||
args = append(args, "--json")
|
||||
processResult, err := r.run(ctx, subprocess.RunRequest{
|
||||
Executable: req.Binary,
|
||||
Args: args,
|
||||
WorkingDir: req.WorkingDirectory,
|
||||
Timeout: req.Timeout,
|
||||
DiagnosticOwner: "notarius",
|
||||
StdoutLogPath: req.ReceiptPath,
|
||||
StderrLogPath: req.LogPath,
|
||||
})
|
||||
@@ -94,24 +106,42 @@ func (r *SubprocessRunner) Run(ctx context.Context, req RunRequest) (RunResult,
|
||||
if err != nil {
|
||||
return baseResult, err
|
||||
}
|
||||
diagnostics, diagnosticOccurrences, diagnosticsTruncated, err := loadDiagnostics(index.DiagnosticsPath)
|
||||
if err != nil {
|
||||
return baseResult, err
|
||||
}
|
||||
if receipt.NormalizedOutputCount != len(index.Lanes) || receipt.RejectedOutputCount != len(rejections) ||
|
||||
receipt.WarningGroupCount != len(warnings) || receipt.DiagnosticGroupCount != len(diagnostics) {
|
||||
return baseResult, fmt.Errorf("notarius receipt counts do not match published bundle")
|
||||
}
|
||||
warningOccurrences, err := sumWarningOccurrences(warnings)
|
||||
if err != nil {
|
||||
return baseResult, err
|
||||
}
|
||||
if receipt.WarningOccurrenceCount != warningOccurrences ||
|
||||
receipt.DiagnosticOccurrenceCount != diagnosticOccurrences ||
|
||||
receipt.DiagnosticsTruncated != diagnosticsTruncated {
|
||||
return baseResult, fmt.Errorf("notarius receipt occurrence counts do not match published bundle")
|
||||
}
|
||||
|
||||
baseResult.Receipt = receipt
|
||||
baseResult.Index = index
|
||||
baseResult.BundleRoot = bundleRoot
|
||||
baseResult.Rejections = rejections
|
||||
baseResult.Warnings = warnings
|
||||
baseResult.Diagnostics = diagnostics
|
||||
return baseResult, nil
|
||||
}
|
||||
|
||||
func validateRunRequest(req RunRequest) error {
|
||||
func validateRunRequest(req RunRequest) ([]ReferenceBinding, error) {
|
||||
if strings.TrimSpace(req.Binary) == "" {
|
||||
return fmt.Errorf("notarius binary is required")
|
||||
return nil, fmt.Errorf("notarius binary is required")
|
||||
}
|
||||
if strings.TrimSpace(req.PipelineID) == "" {
|
||||
return fmt.Errorf("notarius pipeline id is required")
|
||||
return nil, fmt.Errorf("notarius pipeline id is required")
|
||||
}
|
||||
if req.Timeout <= 0 {
|
||||
return fmt.Errorf("notarius timeout must be positive")
|
||||
return nil, fmt.Errorf("notarius timeout must be positive")
|
||||
}
|
||||
for label, path := range map[string]string{
|
||||
"config": req.ConfigPath,
|
||||
@@ -122,34 +152,53 @@ func validateRunRequest(req RunRequest) error {
|
||||
"log": req.LogPath,
|
||||
} {
|
||||
if strings.TrimSpace(path) == "" {
|
||||
return fmt.Errorf("notarius %s path is required", label)
|
||||
return nil, fmt.Errorf("notarius %s path is required", label)
|
||||
}
|
||||
if !filepath.IsAbs(path) {
|
||||
return fmt.Errorf("notarius %s path must be absolute", label)
|
||||
return nil, fmt.Errorf("notarius %s path must be absolute", label)
|
||||
}
|
||||
}
|
||||
if filepath.Clean(req.ReceiptPath) == filepath.Clean(req.LogPath) {
|
||||
return fmt.Errorf("notarius receipt and log paths must be different")
|
||||
return nil, fmt.Errorf("notarius receipt and log paths must be different")
|
||||
}
|
||||
references := make([]ReferenceBinding, 0, len(req.References))
|
||||
selectors := make(map[string]struct{}, len(req.References))
|
||||
for index, binding := range req.References {
|
||||
selector, err := notariusref.NormalizeSelector(binding.Selector)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("notarius reference %d selector: %w", index, err)
|
||||
}
|
||||
if _, duplicate := selectors[selector]; duplicate {
|
||||
return nil, fmt.Errorf("notarius reference selector %q is duplicated", selector)
|
||||
}
|
||||
selectors[selector] = struct{}{}
|
||||
if strings.TrimSpace(binding.Path) == "" {
|
||||
return nil, fmt.Errorf("notarius reference %q path is required", selector)
|
||||
}
|
||||
if !filepath.IsAbs(binding.Path) {
|
||||
return nil, fmt.Errorf("notarius reference %q path must be absolute", selector)
|
||||
}
|
||||
references = append(references, ReferenceBinding{Selector: selector, Path: binding.Path})
|
||||
}
|
||||
if err := requireRegularFile(req.ConfigPath); err != nil {
|
||||
return fmt.Errorf("validate notarius config path: %w", err)
|
||||
return nil, fmt.Errorf("validate notarius config path: %w", err)
|
||||
}
|
||||
if err := requireRegularFile(req.InputPath); err != nil {
|
||||
return fmt.Errorf("validate notarius input path: %w", err)
|
||||
return nil, fmt.Errorf("validate notarius input path: %w", err)
|
||||
}
|
||||
if err := requireDirectory(req.OutputRoot); err != nil {
|
||||
return fmt.Errorf("validate notarius output root: %w", err)
|
||||
return nil, fmt.Errorf("validate notarius output root: %w", err)
|
||||
}
|
||||
if err := requireDirectory(req.WorkingDirectory); err != nil {
|
||||
return fmt.Errorf("validate notarius working directory: %w", err)
|
||||
return nil, fmt.Errorf("validate notarius working directory: %w", err)
|
||||
}
|
||||
if err := validateLogDestination(req.ReceiptPath); err != nil {
|
||||
return fmt.Errorf("validate notarius receipt path: %w", err)
|
||||
return nil, fmt.Errorf("validate notarius receipt path: %w", err)
|
||||
}
|
||||
if err := validateLogDestination(req.LogPath); err != nil {
|
||||
return fmt.Errorf("validate notarius log path: %w", err)
|
||||
return nil, fmt.Errorf("validate notarius log path: %w", err)
|
||||
}
|
||||
return nil
|
||||
return references, nil
|
||||
}
|
||||
|
||||
type receiptDocument struct {
|
||||
@@ -160,11 +209,69 @@ type receiptDocument struct {
|
||||
IndexFile string `json:"index_file"`
|
||||
NormalizedOutputCount *int `json:"normalized_output_count"`
|
||||
RejectedOutputCount *int `json:"rejected_output_count"`
|
||||
WarningCount *int `json:"warning_count"`
|
||||
WarningGroupCount *int `json:"warning_group_count"`
|
||||
WarningOccurrenceCount *int `json:"warning_occurrence_count"`
|
||||
DiagnosticGroupCount *int `json:"diagnostic_group_count"`
|
||||
DiagnosticOccurrenceCount *int `json:"diagnostic_occurrence_count"`
|
||||
DiagnosticsTruncated *bool `json:"diagnostics_truncated"`
|
||||
ValidationStatus string `json:"validation_status"`
|
||||
ValidationSummaries []validationSummaryDocument `json:"validation_summaries"`
|
||||
DebugDirectory string `json:"debug_directory"`
|
||||
}
|
||||
|
||||
type validationSummaryDocument struct {
|
||||
Stage string `json:"stage"`
|
||||
StepID string `json:"step_id"`
|
||||
LaneID string `json:"lane_id"`
|
||||
ModuleKey string `json:"module_key"`
|
||||
ChunkID string `json:"chunk_id"`
|
||||
Status string `json:"status"`
|
||||
RejectingValidators []string `json:"rejecting_validators"`
|
||||
ReasonCodes []string `json:"reason_codes"`
|
||||
IncompleteValidators []string `json:"incomplete_validators"`
|
||||
ProducerAttemptCount *int `json:"producer_attempt_count"`
|
||||
TerminalAction string `json:"terminal_action"`
|
||||
}
|
||||
|
||||
func validValidationStatus(value string) bool {
|
||||
switch value {
|
||||
case "approved", "rejected", "incomplete":
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
func validateValidationSummaries(documents []validationSummaryDocument) ([]ValidationSummary, error) {
|
||||
summaries := make([]ValidationSummary, 0, len(documents))
|
||||
for _, document := range documents {
|
||||
if document.Status != "complete" && document.Status != "rejected" && document.Status != "incomplete" {
|
||||
return nil, fmt.Errorf("notarius validation summary status %q is invalid", document.Status)
|
||||
}
|
||||
if document.ProducerAttemptCount == nil || *document.ProducerAttemptCount <= 0 || !validTerminalAction(document.TerminalAction) {
|
||||
return nil, fmt.Errorf("notarius validation summary is missing required fields")
|
||||
}
|
||||
summaries = append(summaries, ValidationSummary{
|
||||
Stage: document.Stage, StepID: document.StepID, LaneID: document.LaneID,
|
||||
ModuleKey: document.ModuleKey, ChunkID: document.ChunkID, Status: document.Status,
|
||||
RejectingValidators: append([]string(nil), document.RejectingValidators...),
|
||||
ReasonCodes: append([]string(nil), document.ReasonCodes...),
|
||||
IncompleteValidators: append([]string(nil), document.IncompleteValidators...),
|
||||
ProducerAttemptCount: *document.ProducerAttemptCount, TerminalAction: document.TerminalAction,
|
||||
})
|
||||
}
|
||||
return summaries, nil
|
||||
}
|
||||
|
||||
func validTerminalAction(value string) bool {
|
||||
switch value {
|
||||
case "accepted", "reject_output", "warn_continue", "fail_run":
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
func loadReceipt(path, pipelineID string) (Receipt, error) {
|
||||
var document receiptDocument
|
||||
if err := decodeBoundedJSON(path, maxReceiptBytes, &document); err != nil {
|
||||
@@ -176,7 +283,9 @@ func loadReceipt(path, pipelineID string) (Receipt, error) {
|
||||
if strings.TrimSpace(document.RunID) == "" || strings.TrimSpace(document.PipelineID) == "" ||
|
||||
strings.TrimSpace(document.OutputDirectory) == "" || strings.TrimSpace(document.ValidationStatus) == "" ||
|
||||
document.NormalizedOutputCount == nil ||
|
||||
document.RejectedOutputCount == nil || document.WarningCount == nil {
|
||||
document.RejectedOutputCount == nil || document.WarningGroupCount == nil ||
|
||||
document.WarningOccurrenceCount == nil || document.DiagnosticGroupCount == nil ||
|
||||
document.DiagnosticOccurrenceCount == nil || document.DiagnosticsTruncated == nil {
|
||||
return Receipt{}, fmt.Errorf("notarius receipt is missing required fields")
|
||||
}
|
||||
if document.IndexFile != canonicalIndexFile {
|
||||
@@ -185,9 +294,18 @@ func loadReceipt(path, pipelineID string) (Receipt, error) {
|
||||
if document.PipelineID != pipelineID {
|
||||
return Receipt{}, fmt.Errorf("notarius receipt pipeline id %q does not match requested pipeline %q", document.PipelineID, pipelineID)
|
||||
}
|
||||
if *document.NormalizedOutputCount < 0 || *document.RejectedOutputCount < 0 || *document.WarningCount < 0 {
|
||||
if *document.NormalizedOutputCount < 0 || *document.RejectedOutputCount < 0 ||
|
||||
*document.WarningGroupCount < 0 || *document.WarningOccurrenceCount < 0 ||
|
||||
*document.DiagnosticGroupCount < 0 || *document.DiagnosticOccurrenceCount < 0 {
|
||||
return Receipt{}, fmt.Errorf("notarius receipt counts must be non-negative")
|
||||
}
|
||||
if !validValidationStatus(document.ValidationStatus) {
|
||||
return Receipt{}, fmt.Errorf("notarius receipt validation_status %q is invalid", document.ValidationStatus)
|
||||
}
|
||||
validationSummaries, err := validateValidationSummaries(document.ValidationSummaries)
|
||||
if err != nil {
|
||||
return Receipt{}, err
|
||||
}
|
||||
if !filepath.IsAbs(document.OutputDirectory) {
|
||||
return Receipt{}, fmt.Errorf("notarius receipt output directory must be absolute")
|
||||
}
|
||||
@@ -202,8 +320,13 @@ func loadReceipt(path, pipelineID string) (Receipt, error) {
|
||||
IndexFile: document.IndexFile,
|
||||
NormalizedOutputCount: *document.NormalizedOutputCount,
|
||||
RejectedOutputCount: *document.RejectedOutputCount,
|
||||
WarningCount: *document.WarningCount,
|
||||
WarningGroupCount: *document.WarningGroupCount,
|
||||
WarningOccurrenceCount: *document.WarningOccurrenceCount,
|
||||
DiagnosticGroupCount: *document.DiagnosticGroupCount,
|
||||
DiagnosticOccurrenceCount: *document.DiagnosticOccurrenceCount,
|
||||
DiagnosticsTruncated: *document.DiagnosticsTruncated,
|
||||
ValidationStatus: document.ValidationStatus,
|
||||
ValidationSummaries: validationSummaries,
|
||||
DebugDirectory: document.DebugDirectory,
|
||||
}, nil
|
||||
}
|
||||
@@ -213,6 +336,7 @@ type indexDocument struct {
|
||||
OutputFiles *[]laneDocument `json:"output_files"`
|
||||
RejectedFile string `json:"rejected_file"`
|
||||
WarningsFile string `json:"warnings_file"`
|
||||
DiagnosticsFile string `json:"diagnostics_file"`
|
||||
ChunkMap *pipelineDocument `json:"chunk_map"`
|
||||
EvidenceContext *pipelineDocument `json:"evidence_context"`
|
||||
}
|
||||
@@ -249,6 +373,7 @@ func loadIndex(bundleRoot, indexPath string) (Index, error) {
|
||||
{name: "manifest_file", got: document.ManifestFile, want: canonicalManifestFile},
|
||||
{name: "rejected_file", got: document.RejectedFile, want: canonicalRejectedFile},
|
||||
{name: "warnings_file", got: document.WarningsFile, want: canonicalWarningsFile},
|
||||
{name: "diagnostics_file", got: document.DiagnosticsFile, want: canonicalDiagnosticsFile},
|
||||
} {
|
||||
if field.got != field.want {
|
||||
return Index{}, fmt.Errorf("notarius index %s %q is incompatible; want %q", field.name, field.got, field.want)
|
||||
@@ -263,6 +388,7 @@ func loadIndex(bundleRoot, indexPath string) (Index, error) {
|
||||
ManifestFile: document.ManifestFile,
|
||||
RejectedFile: document.RejectedFile,
|
||||
WarningsFile: document.WarningsFile,
|
||||
DiagnosticsFile: document.DiagnosticsFile,
|
||||
}
|
||||
var err error
|
||||
if index.ManifestPath, err = resolveRegularFile(bundleRoot, index.ManifestFile); err != nil {
|
||||
@@ -274,6 +400,9 @@ func loadIndex(bundleRoot, indexPath string) (Index, error) {
|
||||
if index.WarningsPath, err = resolveRegularFile(bundleRoot, index.WarningsFile); err != nil {
|
||||
return Index{}, fmt.Errorf("resolve notarius warning file: %w", err)
|
||||
}
|
||||
if index.DiagnosticsPath, err = resolveRegularFile(bundleRoot, index.DiagnosticsFile); err != nil {
|
||||
return Index{}, fmt.Errorf("resolve notarius diagnostics file: %w", err)
|
||||
}
|
||||
|
||||
seenLanes := make(map[string]struct{}, len(*document.OutputFiles))
|
||||
for _, lane := range *document.OutputFiles {
|
||||
@@ -361,12 +490,34 @@ func loadRejections(path string) ([]RejectionSummary, error) {
|
||||
return summaries, nil
|
||||
}
|
||||
|
||||
type warningDocument struct {
|
||||
Warnings *[]struct {
|
||||
Scope string `json:"scope"`
|
||||
type findingGroupDocument struct {
|
||||
Disposition string `json:"disposition"`
|
||||
Category string `json:"category"`
|
||||
ReasonCode string `json:"reason_code"`
|
||||
Origin diagnosticOriginDocument `json:"origin"`
|
||||
OccurrenceCount *int `json:"occurrence_count"`
|
||||
Samples *[]struct {
|
||||
Scope string `json:"scope"`
|
||||
Message string `json:"message"`
|
||||
} `json:"warnings"`
|
||||
ChunkID string `json:"chunk_id"`
|
||||
ChunkIndex *int `json:"chunk_index"`
|
||||
} `json:"samples"`
|
||||
OmittedSampleCount *int `json:"omitted_sample_count"`
|
||||
}
|
||||
|
||||
type diagnosticOriginDocument struct {
|
||||
Stage string `json:"stage"`
|
||||
StepID string `json:"step_id"`
|
||||
LaneID string `json:"lane_id"`
|
||||
ModuleKey string `json:"module_key"`
|
||||
ValidatorKey string `json:"validator_key"`
|
||||
}
|
||||
|
||||
type warningDocument struct {
|
||||
SchemaVersion string `json:"schema_version"`
|
||||
GroupCount *int `json:"group_count"`
|
||||
OccurrenceCount *int `json:"occurrence_count"`
|
||||
Groups *[]findingGroupDocument `json:"groups"`
|
||||
}
|
||||
|
||||
func loadWarnings(path string) ([]WarningSummary, error) {
|
||||
@@ -374,46 +525,155 @@ func loadWarnings(path string) ([]WarningSummary, error) {
|
||||
if err := decodeBoundedJSON(path, maxSummaryBytes, &document); err != nil {
|
||||
return nil, fmt.Errorf("decode notarius warnings: %w", err)
|
||||
}
|
||||
if document.Warnings == nil {
|
||||
return nil, fmt.Errorf("notarius warning document is missing warnings array")
|
||||
if document.SchemaVersion != warningsSchemaVersion || document.GroupCount == nil ||
|
||||
document.OccurrenceCount == nil || document.Groups == nil {
|
||||
return nil, fmt.Errorf("notarius warning document is missing or incompatible required fields")
|
||||
}
|
||||
summaries := make([]WarningSummary, 0, len(*document.Warnings))
|
||||
for _, item := range *document.Warnings {
|
||||
if strings.TrimSpace(item.ReasonCode) == "" || strings.TrimSpace(item.Message) == "" {
|
||||
return nil, fmt.Errorf("notarius warning entries require reason_code and message")
|
||||
if *document.GroupCount < 0 || *document.GroupCount > maxWarningGroups || *document.OccurrenceCount < 0 ||
|
||||
*document.GroupCount != len(*document.Groups) {
|
||||
return nil, fmt.Errorf("notarius warning document counts are inconsistent")
|
||||
}
|
||||
summaries = append(summaries, WarningSummary{Scope: item.Scope, ReasonCode: item.ReasonCode})
|
||||
summaries := make([]WarningSummary, 0, len(*document.Groups))
|
||||
occurrences := 0
|
||||
for _, group := range *document.Groups {
|
||||
if err := validateFindingGroup(group); err != nil {
|
||||
return nil, fmt.Errorf("notarius warning group: %w", err)
|
||||
}
|
||||
if group.Disposition != "warning" {
|
||||
return nil, fmt.Errorf("notarius warning group disposition %q is invalid", group.Disposition)
|
||||
}
|
||||
if *group.OccurrenceCount > int(^uint(0)>>1)-occurrences {
|
||||
return nil, fmt.Errorf("notarius warning occurrence count overflows")
|
||||
}
|
||||
occurrences += *group.OccurrenceCount
|
||||
summaries = append(summaries, WarningSummary{
|
||||
Disposition: group.Disposition, Category: group.Category, ReasonCode: group.ReasonCode,
|
||||
Origin: diagnosticOrigin(group.Origin), OccurrenceCount: *group.OccurrenceCount,
|
||||
})
|
||||
}
|
||||
if occurrences != *document.OccurrenceCount {
|
||||
return nil, fmt.Errorf("notarius warning document occurrence count is inconsistent")
|
||||
}
|
||||
return summaries, nil
|
||||
}
|
||||
|
||||
type diagnosticDocument struct {
|
||||
SchemaVersion string `json:"schema_version"`
|
||||
GroupCount *int `json:"group_count"`
|
||||
OccurrenceCount *int `json:"occurrence_count"`
|
||||
Truncated *bool `json:"truncated"`
|
||||
UnrepresentedOccurrenceCount *int `json:"unrepresented_occurrence_count"`
|
||||
Groups *[]findingGroupDocument `json:"groups"`
|
||||
}
|
||||
|
||||
func loadDiagnostics(path string) ([]DiagnosticSummary, int, bool, error) {
|
||||
var document diagnosticDocument
|
||||
if err := decodeBoundedJSON(path, maxSummaryBytes, &document); err != nil {
|
||||
return nil, 0, false, fmt.Errorf("decode notarius diagnostics: %w", err)
|
||||
}
|
||||
if document.SchemaVersion != diagnosticsSchemaVersion || document.GroupCount == nil ||
|
||||
document.OccurrenceCount == nil || document.Truncated == nil ||
|
||||
document.UnrepresentedOccurrenceCount == nil || document.Groups == nil {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostics document is missing or incompatible required fields")
|
||||
}
|
||||
if *document.GroupCount < 0 || *document.GroupCount > maxDiagnosticGroups || *document.OccurrenceCount < 0 ||
|
||||
*document.UnrepresentedOccurrenceCount < 0 || *document.GroupCount != len(*document.Groups) {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostics document counts are inconsistent")
|
||||
}
|
||||
if !*document.Truncated && *document.UnrepresentedOccurrenceCount != 0 {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostics document has unrepresented occurrences without truncation")
|
||||
}
|
||||
summaries := make([]DiagnosticSummary, 0, len(*document.Groups))
|
||||
representedOccurrences := 0
|
||||
for _, group := range *document.Groups {
|
||||
if err := validateFindingGroup(group); err != nil {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostic group: %w", err)
|
||||
}
|
||||
if group.Disposition != "advisory" && group.Disposition != "observation" {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostic group disposition %q is invalid", group.Disposition)
|
||||
}
|
||||
if *group.OccurrenceCount > int(^uint(0)>>1)-representedOccurrences {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostic occurrence count overflows")
|
||||
}
|
||||
representedOccurrences += *group.OccurrenceCount
|
||||
summaries = append(summaries, DiagnosticSummary{
|
||||
Disposition: group.Disposition, Category: group.Category, ReasonCode: group.ReasonCode,
|
||||
Origin: diagnosticOrigin(group.Origin), OccurrenceCount: *group.OccurrenceCount,
|
||||
})
|
||||
}
|
||||
if *document.UnrepresentedOccurrenceCount > int(^uint(0)>>1)-representedOccurrences ||
|
||||
representedOccurrences+*document.UnrepresentedOccurrenceCount != *document.OccurrenceCount {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostics document occurrence count is inconsistent")
|
||||
}
|
||||
return summaries, *document.OccurrenceCount, *document.Truncated, nil
|
||||
}
|
||||
|
||||
func validateFindingGroup(group findingGroupDocument) error {
|
||||
if strings.TrimSpace(group.Disposition) == "" || strings.TrimSpace(group.Category) == "" ||
|
||||
strings.TrimSpace(group.ReasonCode) == "" || !validDiagnosticOriginStage(group.Origin.Stage) ||
|
||||
!validDiagnosticCategory(group.Disposition, group.Category) ||
|
||||
group.OccurrenceCount == nil || *group.OccurrenceCount <= 0 || group.Samples == nil ||
|
||||
group.OmittedSampleCount == nil || *group.OmittedSampleCount < 0 {
|
||||
return fmt.Errorf("missing required fields")
|
||||
}
|
||||
if len(*group.Samples) == 0 || len(*group.Samples) > maxFindingSamples ||
|
||||
*group.OmittedSampleCount != *group.OccurrenceCount-len(*group.Samples) {
|
||||
return fmt.Errorf("sample counts are inconsistent")
|
||||
}
|
||||
for _, sample := range *group.Samples {
|
||||
if strings.TrimSpace(sample.Scope) == "" || strings.TrimSpace(sample.Message) == "" ||
|
||||
(sample.ChunkIndex != nil && *sample.ChunkIndex < 0) {
|
||||
return fmt.Errorf("samples require scope and message")
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func validDiagnosticCategory(disposition, category string) bool {
|
||||
switch disposition {
|
||||
case "warning":
|
||||
return category == "configuration" || category == "degradation" ||
|
||||
category == "validation_incomplete" || category == "fallback"
|
||||
case "advisory":
|
||||
return category == "data_quality"
|
||||
case "observation":
|
||||
return category == "normalization"
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
func validDiagnosticOriginStage(stage string) bool {
|
||||
switch stage {
|
||||
case "references", "chunk", "extract", "merge", "normalize":
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
func diagnosticOrigin(document diagnosticOriginDocument) DiagnosticOrigin {
|
||||
return DiagnosticOrigin{
|
||||
Stage: document.Stage, StepID: document.StepID, LaneID: document.LaneID,
|
||||
ModuleKey: document.ModuleKey, ValidatorKey: document.ValidatorKey,
|
||||
}
|
||||
}
|
||||
|
||||
func sumWarningOccurrences(values []WarningSummary) (int, error) {
|
||||
total := 0
|
||||
for _, value := range values {
|
||||
if value.OccurrenceCount > int(^uint(0)>>1)-total {
|
||||
return 0, fmt.Errorf("notarius warning occurrence count overflows")
|
||||
}
|
||||
total += value.OccurrenceCount
|
||||
}
|
||||
return total, nil
|
||||
}
|
||||
|
||||
func decodeBoundedJSON(path string, limit int64, destination any) error {
|
||||
inspected, err := os.Lstat(path)
|
||||
data, err := fileops.ReadRegularFile(path, limit)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if inspected.Mode()&os.ModeSymlink != 0 || !inspected.Mode().IsRegular() {
|
||||
return fmt.Errorf("path %q must be a regular file without symlinks", path)
|
||||
}
|
||||
file, err := os.Open(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer func() { _ = file.Close() }()
|
||||
opened, err := file.Stat()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if !opened.Mode().IsRegular() || !os.SameFile(inspected, opened) {
|
||||
return fmt.Errorf("file %q changed before it could be read", path)
|
||||
}
|
||||
reader := io.LimitReader(file, limit+1)
|
||||
data, err := io.ReadAll(reader)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if int64(len(data)) > limit {
|
||||
return fmt.Errorf("file %q exceeds %d-byte limit", path, limit)
|
||||
return fmt.Errorf("notarius JSON result exceeds or cannot be read within %d-byte limit: %w", limit, err)
|
||||
}
|
||||
if err := json.Unmarshal(data, destination); err != nil {
|
||||
return err
|
||||
|
||||
@@ -52,6 +52,10 @@ func TestSubprocessRunnerBuildsExactInvocationAndDiscoversBundle(t *testing.T) {
|
||||
if result.Receipt.SchemaVersion != ReceiptSchemaVersion || result.Receipt.RunID != "notarius-run-1" {
|
||||
t.Fatalf("receipt = %#v", result.Receipt)
|
||||
}
|
||||
if len(result.Receipt.ValidationSummaries) != 1 || result.Receipt.ValidationSummaries[0].LaneID != "npc-registry" ||
|
||||
result.Receipt.ValidationSummaries[0].Status != "complete" {
|
||||
t.Fatalf("validation summaries = %#v", result.Receipt.ValidationSummaries)
|
||||
}
|
||||
if len(result.Index.Lanes) != 1 || result.Index.Lanes[0].LaneID != "npc-registry" {
|
||||
t.Fatalf("lanes = %#v", result.Index.Lanes)
|
||||
}
|
||||
@@ -64,12 +68,110 @@ func TestSubprocessRunnerBuildsExactInvocationAndDiscoversBundle(t *testing.T) {
|
||||
if len(result.Rejections) != 1 || result.Rejections[0].LaneID != "spells" || result.Rejections[0].ReasonCode != "invalid_spell" {
|
||||
t.Fatalf("rejections = %#v", result.Rejections)
|
||||
}
|
||||
if len(result.Warnings) != 1 || result.Warnings[0].Scope != "lane:npc-registry" || result.Warnings[0].ReasonCode != "normalized_name" {
|
||||
if len(result.Warnings) != 1 || result.Warnings[0].Category != "degradation" || result.Warnings[0].ReasonCode != "normalized_name" {
|
||||
t.Fatalf("warnings = %#v", result.Warnings)
|
||||
}
|
||||
if len(result.Diagnostics) != 1 || result.Diagnostics[0].Category != "data_quality" || result.Diagnostics[0].ReasonCode != "low_confidence" {
|
||||
t.Fatalf("diagnostics = %#v", result.Diagnostics)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSubprocessRunnerInheritsEnvironmentAndSeparatesStreams(t *testing.T) {
|
||||
func TestSubprocessRunnerBuildsOrderedReferenceArguments(t *testing.T) {
|
||||
req := validRunRequest(t)
|
||||
referenceRoot := t.TempDir()
|
||||
req.References = []ReferenceBinding{
|
||||
{Selector: " party ", Path: filepath.Join(referenceRoot, "party context=primary.json")},
|
||||
{Selector: " npc-registry . extract . glossary ", Path: filepath.Join(referenceRoot, "glossary.json")},
|
||||
}
|
||||
originalReferences := append([]ReferenceBinding(nil), req.References...)
|
||||
var captured sharedsubprocess.RunRequest
|
||||
runner := &SubprocessRunner{run: func(_ context.Context, processReq sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
|
||||
captured = processReq
|
||||
writeValidBundleAndReceipt(t, req, false)
|
||||
return sharedsubprocess.RunResult{ExitCode: 0}, nil
|
||||
}}
|
||||
|
||||
if _, err := runner.Run(context.Background(), req); err != nil {
|
||||
t.Fatalf("Run() error = %v", err)
|
||||
}
|
||||
wantArgs := []string{
|
||||
"run", req.PipelineID,
|
||||
"--config", req.ConfigPath,
|
||||
"--input", req.InputPath,
|
||||
"--output-dir", req.OutputRoot,
|
||||
"--reference", "party=" + req.References[0].Path,
|
||||
"--reference", "npc-registry.extract.glossary=" + req.References[1].Path,
|
||||
"--json",
|
||||
}
|
||||
if !reflect.DeepEqual(captured.Args, wantArgs) {
|
||||
t.Fatalf("subprocess args = %#v, want %#v", captured.Args, wantArgs)
|
||||
}
|
||||
if !reflect.DeepEqual(req.References, originalReferences) {
|
||||
t.Fatalf("Run() mutated caller references = %#v, want %#v", req.References, originalReferences)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSubprocessRunnerRejectsInvalidReferencesBeforeLaunch(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
references func(string) []ReferenceBinding
|
||||
wantErr string
|
||||
}{
|
||||
{
|
||||
name: "invalid selector",
|
||||
references: func(root string) []ReferenceBinding {
|
||||
return []ReferenceBinding{{Selector: "lane.prepare.party", Path: filepath.Join(root, "party.json")}}
|
||||
},
|
||||
wantErr: "selector",
|
||||
},
|
||||
{
|
||||
name: "duplicate normalized selector",
|
||||
references: func(root string) []ReferenceBinding {
|
||||
return []ReferenceBinding{
|
||||
{Selector: "lane.party", Path: filepath.Join(root, "party.json")},
|
||||
{Selector: " lane . party ", Path: filepath.Join(root, "party-2.json")},
|
||||
}
|
||||
},
|
||||
wantErr: "duplicated",
|
||||
},
|
||||
{
|
||||
name: "empty path",
|
||||
references: func(string) []ReferenceBinding {
|
||||
return []ReferenceBinding{{Selector: "party", Path: " "}}
|
||||
},
|
||||
wantErr: "path is required",
|
||||
},
|
||||
{
|
||||
name: "relative path",
|
||||
references: func(string) []ReferenceBinding {
|
||||
return []ReferenceBinding{{Selector: "party", Path: "references/party.json"}}
|
||||
},
|
||||
wantErr: "path must be absolute",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
req := validRunRequest(t)
|
||||
req.References = tt.references(t.TempDir())
|
||||
started := false
|
||||
runner := &SubprocessRunner{run: func(context.Context, sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
|
||||
started = true
|
||||
return sharedsubprocess.RunResult{}, nil
|
||||
}}
|
||||
|
||||
_, err := runner.Run(context.Background(), req)
|
||||
if err == nil || !strings.Contains(err.Error(), tt.wantErr) {
|
||||
t.Fatalf("Run() error = %v, want containing %q", err, tt.wantErr)
|
||||
}
|
||||
if started {
|
||||
t.Fatal("subprocess started after request validation failure")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestSubprocessRunnerUsesMinimalEnvironmentAndSeparatesStreams(t *testing.T) {
|
||||
req := validRunRequest(t)
|
||||
writeValidBundleAndReceipt(t, req, false)
|
||||
receiptFixture := req.ReceiptPath + ".fixture"
|
||||
@@ -103,7 +205,7 @@ cat "$NOTARIUS_RECEIPT_FIXTURE"
|
||||
t.Fatalf("Run() error = %v", err)
|
||||
}
|
||||
assertTextFile(t, filepath.Join(captureDir, "working-directory"), req.WorkingDirectory+"\n")
|
||||
assertTextFile(t, filepath.Join(captureDir, "environment"), "inherited-value")
|
||||
assertTextFile(t, filepath.Join(captureDir, "environment"), "")
|
||||
assertTextFile(t, req.LogPath, "diagnostic stream\n")
|
||||
receiptBytes, err := os.ReadFile(req.ReceiptPath)
|
||||
if err != nil {
|
||||
@@ -171,8 +273,10 @@ func TestLoadReceiptValidation(t *testing.T) {
|
||||
valid := map[string]any{
|
||||
"schema_version": ReceiptSchemaVersion, "run_id": "run-1", "pipeline_id": "pipeline-1",
|
||||
"output_directory": filepath.Join(root, "outputs", "run-1"), "index_file": "index.json",
|
||||
"normalized_output_count": 1, "rejected_output_count": 0, "warning_count": 0,
|
||||
"validation_status": "approved", "future_field": true,
|
||||
"normalized_output_count": 1, "rejected_output_count": 0,
|
||||
"warning_group_count": 0, "warning_occurrence_count": 0,
|
||||
"diagnostic_group_count": 0, "diagnostic_occurrence_count": 0,
|
||||
"diagnostics_truncated": false, "validation_status": "approved", "future_field": true,
|
||||
}
|
||||
tests := []struct {
|
||||
name string
|
||||
@@ -183,11 +287,12 @@ func TestLoadReceiptValidation(t *testing.T) {
|
||||
}{
|
||||
{name: "unknown fields tolerated", wantOK: true},
|
||||
{name: "malformed", raw: []byte("{")},
|
||||
{name: "unsupported version", mutate: func(v map[string]any) { v["schema_version"] = "notarius.run-result.v2" }},
|
||||
{name: "unsupported version", mutate: func(v map[string]any) { v["schema_version"] = "notarius.run-result.v1" }},
|
||||
{name: "missing field", mutate: func(v map[string]any) { delete(v, "run_id") }},
|
||||
{name: "pipeline mismatch", mutate: func(v map[string]any) { v["pipeline_id"] = "other" }},
|
||||
{name: "relative output", mutate: func(v map[string]any) { v["output_directory"] = "run-1" }},
|
||||
{name: "negative count", mutate: func(v map[string]any) { v["warning_count"] = -1 }},
|
||||
{name: "negative count", mutate: func(v map[string]any) { v["warning_group_count"] = -1 }},
|
||||
{name: "invalid validation status", mutate: func(v map[string]any) { v["validation_status"] = "valid" }},
|
||||
{
|
||||
name: "nested index", mutate: func(v map[string]any) { v["index_file"] = "nested/index.json" },
|
||||
wantError: `index_file "nested/index.json"`,
|
||||
@@ -369,22 +474,26 @@ func TestLoadDiagnosticSummariesValidateBoundsAndTolerateUnknownFields(t *testin
|
||||
root := t.TempDir()
|
||||
rejectedPath := filepath.Join(root, "rejected.json")
|
||||
warningsPath := filepath.Join(root, "warnings.json")
|
||||
diagnosticsPath := filepath.Join(root, "diagnostics.json")
|
||||
writeJSONFile(t, rejectedPath, map[string]any{"rejected": []any{map[string]any{
|
||||
"stage": "validate", "lane_id": "spells", "reason_code": "invalid", "message": "do not retain this", "future": true,
|
||||
}}, "future": true})
|
||||
writeJSONFile(t, warningsPath, map[string]any{"warnings": []any{map[string]any{
|
||||
"scope": "lane:spells", "reason_code": "bounded", "message": "do not retain this", "future": true,
|
||||
}}, "future": true})
|
||||
writeJSONFile(t, warningsPath, findingEnvelope(warningsSchemaVersion, []any{findingGroup("warning", "degradation", "bounded", "normalize", 2)}, 2, false, 0))
|
||||
writeJSONFile(t, diagnosticsPath, findingEnvelope(diagnosticsSchemaVersion, []any{findingGroup("advisory", "data_quality", "low_confidence", "normalize", 3)}, 4, true, 1))
|
||||
rejections, err := loadRejections(rejectedPath)
|
||||
if err != nil || len(rejections) != 1 || rejections[0].ReasonCode != "invalid" {
|
||||
t.Fatalf("loadRejections() = %#v, %v", rejections, err)
|
||||
}
|
||||
warnings, err := loadWarnings(warningsPath)
|
||||
if err != nil || len(warnings) != 1 || warnings[0].Scope != "lane:spells" {
|
||||
if err != nil || len(warnings) != 1 || warnings[0].Category != "degradation" || warnings[0].OccurrenceCount != 2 {
|
||||
t.Fatalf("loadWarnings() = %#v, %v", warnings, err)
|
||||
}
|
||||
diagnostics, occurrences, truncated, err := loadDiagnostics(diagnosticsPath)
|
||||
if err != nil || len(diagnostics) != 1 || occurrences != 4 || !truncated || diagnostics[0].Category != "data_quality" {
|
||||
t.Fatalf("loadDiagnostics() = %#v, %d, %t, %v", diagnostics, occurrences, truncated, err)
|
||||
}
|
||||
|
||||
for name, path := range map[string]string{"rejections": rejectedPath, "warnings": warningsPath} {
|
||||
for name, path := range map[string]string{"rejections": rejectedPath, "warnings": warningsPath, "diagnostics": diagnosticsPath} {
|
||||
t.Run("malformed "+name, func(t *testing.T) {
|
||||
if err := os.WriteFile(path, []byte("{"), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile() error = %v", err)
|
||||
@@ -392,8 +501,10 @@ func TestLoadDiagnosticSummariesValidateBoundsAndTolerateUnknownFields(t *testin
|
||||
var err error
|
||||
if name == "rejections" {
|
||||
_, err = loadRejections(path)
|
||||
} else {
|
||||
} else if name == "warnings" {
|
||||
_, err = loadWarnings(path)
|
||||
} else {
|
||||
_, _, _, err = loadDiagnostics(path)
|
||||
}
|
||||
if err == nil {
|
||||
t.Fatal("summary decoder error = nil")
|
||||
@@ -410,16 +521,23 @@ func TestLoadDiagnosticSummariesValidateBoundsAndTolerateUnknownFields(t *testin
|
||||
if _, err := loadRejections(oversized); err == nil || !strings.Contains(err.Error(), "exceeds") {
|
||||
t.Fatalf("loadRejections(oversized) error = %v", err)
|
||||
}
|
||||
if _, _, _, err := loadDiagnostics(oversized); err == nil || !strings.Contains(err.Error(), "exceeds") {
|
||||
t.Fatalf("loadDiagnostics(oversized) error = %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFakeRunnerCapturesRequestsAndHonorsContextAndError(t *testing.T) {
|
||||
req := RunRequest{PipelineID: "pipeline"}
|
||||
req := RunRequest{PipelineID: "pipeline", References: []ReferenceBinding{{Selector: "party", Path: "/references/party.json"}}}
|
||||
want := RunResult{BundleRoot: "/bundle"}
|
||||
fake := &FakeRunner{Result: want}
|
||||
got, err := fake.Run(context.Background(), req)
|
||||
if err != nil || !reflect.DeepEqual(got, want) || !reflect.DeepEqual(fake.Requests, []RunRequest{req}) {
|
||||
t.Fatalf("Run() = %#v, %v; requests = %#v", got, err, fake.Requests)
|
||||
}
|
||||
req.References[0].Path = "/references/changed.json"
|
||||
if fake.Requests[0].References[0].Path != "/references/party.json" {
|
||||
t.Fatalf("fake retained aliased request references: %#v", fake.Requests[0].References)
|
||||
}
|
||||
|
||||
wantErr := errors.New("configured failure")
|
||||
fake.Err = wantErr
|
||||
@@ -478,13 +596,12 @@ func writeValidBundleAndReceipt(t *testing.T, req RunRequest, includeUnknown boo
|
||||
}
|
||||
}
|
||||
rejection := map[string]any{"stage": "validate", "lane_id": "spells", "reason_code": "invalid_spell", "message": strings.Repeat("external detail", 20)}
|
||||
warning := map[string]any{"scope": "lane:npc-registry", "reason_code": "normalized_name", "message": strings.Repeat("external warning", 20)}
|
||||
if includeUnknown {
|
||||
rejection["future"] = true
|
||||
warning["future"] = true
|
||||
}
|
||||
writeJSONFile(t, filepath.Join(bundle, "rejected.json"), map[string]any{"rejected": []any{rejection}, "future": true})
|
||||
writeJSONFile(t, filepath.Join(bundle, "warnings.json"), map[string]any{"warnings": []any{warning}, "future": true})
|
||||
writeJSONFile(t, filepath.Join(bundle, "warnings.json"), findingEnvelope(warningsSchemaVersion, []any{findingGroup("warning", "degradation", "normalized_name", "normalize", 2)}, 2, false, 0))
|
||||
writeJSONFile(t, filepath.Join(bundle, "diagnostics.json"), findingEnvelope(diagnosticsSchemaVersion, []any{findingGroup("advisory", "data_quality", "low_confidence", "normalize", 3)}, 4, true, 1))
|
||||
index := validIndexValue([]any{map[string]any{
|
||||
"lane_id": "npc-registry", "file": "lanes/npc.json", "media_type": "application/json",
|
||||
"module_key": "dnd/npc-registry", "schema_id": "notarius.dnd.npc_registry",
|
||||
@@ -503,7 +620,13 @@ func writeValidBundleAndReceipt(t *testing.T, req RunRequest, includeUnknown boo
|
||||
receipt := map[string]any{
|
||||
"schema_version": ReceiptSchemaVersion, "run_id": "notarius-run-1", "pipeline_id": req.PipelineID,
|
||||
"output_directory": bundle, "index_file": "index.json", "normalized_output_count": 1,
|
||||
"rejected_output_count": 1, "warning_count": 1, "validation_status": "rejected",
|
||||
"rejected_output_count": 1, "warning_group_count": 1, "warning_occurrence_count": 2,
|
||||
"diagnostic_group_count": 1, "diagnostic_occurrence_count": 4,
|
||||
"diagnostics_truncated": true, "validation_status": "rejected",
|
||||
"validation_summaries": []any{map[string]any{
|
||||
"stage": "normalize", "lane_id": "npc-registry", "status": "complete",
|
||||
"producer_attempt_count": 1, "terminal_action": "accepted",
|
||||
}},
|
||||
}
|
||||
if includeUnknown {
|
||||
receipt["future"] = true
|
||||
@@ -517,7 +640,7 @@ func createBundleSkeleton(t *testing.T) string {
|
||||
if err := os.MkdirAll(filepath.Join(bundle, "lanes"), 0o755); err != nil {
|
||||
t.Fatalf("MkdirAll(bundle) error = %v", err)
|
||||
}
|
||||
for _, name := range []string{"manifest.json", "rejected.json", "warnings.json", "lanes/npc.json", "chunk-map.json"} {
|
||||
for _, name := range []string{"manifest.json", "rejected.json", "warnings.json", "diagnostics.json", "lanes/npc.json", "chunk-map.json"} {
|
||||
if err := os.WriteFile(filepath.Join(bundle, filepath.FromSlash(name)), []byte("{}"), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile(%q) error = %v", name, err)
|
||||
}
|
||||
@@ -528,10 +651,31 @@ func createBundleSkeleton(t *testing.T) string {
|
||||
func validIndexValue(lanes []any) map[string]any {
|
||||
return map[string]any{
|
||||
"manifest_file": "manifest.json", "output_files": lanes,
|
||||
"rejected_file": "rejected.json", "warnings_file": "warnings.json",
|
||||
"rejected_file": "rejected.json", "warnings_file": "warnings.json", "diagnostics_file": "diagnostics.json",
|
||||
}
|
||||
}
|
||||
|
||||
func findingGroup(disposition, category, reasonCode, origin string, occurrences int) map[string]any {
|
||||
return map[string]any{
|
||||
"disposition": disposition, "category": category, "reason_code": reasonCode,
|
||||
"origin": map[string]any{"stage": origin, "lane_id": "npc-registry"}, "occurrence_count": occurrences,
|
||||
"samples": []any{map[string]any{"scope": "lane:npc-registry", "message": "external detail"}},
|
||||
"omitted_sample_count": occurrences - 1,
|
||||
}
|
||||
}
|
||||
|
||||
func findingEnvelope(schema string, groups []any, occurrences int, truncated bool, unrepresented int) map[string]any {
|
||||
value := map[string]any{
|
||||
"schema_version": schema, "group_count": len(groups), "occurrence_count": occurrences,
|
||||
"groups": groups,
|
||||
}
|
||||
if schema == diagnosticsSchemaVersion {
|
||||
value["truncated"] = truncated
|
||||
value["unrepresented_occurrence_count"] = unrepresented
|
||||
}
|
||||
return value
|
||||
}
|
||||
|
||||
func writeJSONFile(t *testing.T, path string, value any) {
|
||||
t.Helper()
|
||||
data, err := json.Marshal(value)
|
||||
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"fmt"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
// NoopRunner is a deterministic no-op scriptorium adapter.
|
||||
@@ -142,7 +143,7 @@ func (f *FakeRunner) RenderArtifact(ctx context.Context, req RenderArtifactReque
|
||||
|
||||
func materializeRunPlaceholders(req RunArtifactRequest) error {
|
||||
if req.OutputPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("scriptorium noop/fake run artifact\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("scriptorium noop/fake run artifact\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write run output %q: %w", req.OutputPath, err)
|
||||
}
|
||||
}
|
||||
@@ -154,17 +155,17 @@ func materializeRunPlaceholders(req RunArtifactRequest) error {
|
||||
"prompt_id": req.PromptID,
|
||||
"output_path": req.OutputPath,
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake run stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake run stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake run stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake run stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
@@ -173,7 +174,7 @@ func materializeRunPlaceholders(req RunArtifactRequest) error {
|
||||
|
||||
func materializeRenderPlaceholders(req RenderArtifactRequest) error {
|
||||
if req.OutputPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("{\"schema\":\"scriptorium.render.v1\",\"placeholder\":true}\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("{\"schema\":\"scriptorium.render.v1\",\"placeholder\":true}\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write render output %q: %w", req.OutputPath, err)
|
||||
}
|
||||
}
|
||||
@@ -185,17 +186,17 @@ func materializeRenderPlaceholders(req RenderArtifactRequest) error {
|
||||
"prompt_id": req.PromptID,
|
||||
"output_path": req.OutputPath,
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake render stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake render stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake render stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake render stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,8 +9,12 @@ import (
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
// MaxOutputFileBytes bounds one Scriptorium artifact result.
|
||||
const MaxOutputFileBytes int64 = 64 * 1024 * 1024
|
||||
|
||||
// SubprocessRunner invokes Scriptorium through its public CLI.
|
||||
type SubprocessRunner struct{}
|
||||
|
||||
@@ -52,11 +56,15 @@ func (r *SubprocessRunner) RunArtifact(ctx context.Context, req RunArtifactReque
|
||||
}
|
||||
}
|
||||
|
||||
envOverrides, sensitiveNames := credentialEnvironment(req.APIKeyEnv)
|
||||
runRes, runErr := subprocess.Run(ctx, subprocess.RunRequest{
|
||||
Executable: req.Binary,
|
||||
Args: args,
|
||||
WorkingDir: req.WorkingDir,
|
||||
Timeout: req.Timeout,
|
||||
EnvOverrides: envOverrides,
|
||||
SensitiveEnvNames: sensitiveNames,
|
||||
DiagnosticOwner: "scriptorium",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
@@ -134,11 +142,15 @@ func (r *SubprocessRunner) RenderArtifact(ctx context.Context, req RenderArtifac
|
||||
}
|
||||
}
|
||||
|
||||
envOverrides, sensitiveNames := credentialEnvironment(req.APIKeyEnv)
|
||||
runRes, runErr := subprocess.Run(ctx, subprocess.RunRequest{
|
||||
Executable: req.Binary,
|
||||
Args: args,
|
||||
WorkingDir: req.WorkingDir,
|
||||
Timeout: req.Timeout,
|
||||
EnvOverrides: envOverrides,
|
||||
SensitiveEnvNames: sensitiveNames,
|
||||
DiagnosticOwner: "scriptorium",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
@@ -223,6 +235,15 @@ func validateCommonRunRequest(
|
||||
return true, nil
|
||||
}
|
||||
|
||||
func credentialEnvironment(apiKeyEnv string) (map[string]string, []string) {
|
||||
name := strings.TrimSpace(apiKeyEnv)
|
||||
if name == "" {
|
||||
return nil, nil
|
||||
}
|
||||
value, _ := os.LookupEnv(name)
|
||||
return map[string]string{name: value}, []string{name}
|
||||
}
|
||||
|
||||
func buildRunArgs(req RunArtifactRequest) []string {
|
||||
args := []string{"run", "--prompt", strings.TrimSpace(req.PromptID)}
|
||||
if cfgPath := strings.TrimSpace(req.ConfigPath); cfgPath != "" {
|
||||
@@ -321,18 +342,15 @@ func writeInvocationConfig(path string, payload invocationPayload) error {
|
||||
"render_format": payload.RenderFormat,
|
||||
"render_prompt_logged": payload.RenderPromptStore,
|
||||
}
|
||||
return subprocess.WriteYAMLAtomic(path, data, 0o644)
|
||||
return subprocess.WriteYAMLAtomic(path, data, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
func validateNonEmptyOutput(path string) error {
|
||||
info, err := os.Stat(path)
|
||||
data, err := fileops.ReadRegularFile(path, MaxOutputFileBytes)
|
||||
if err != nil {
|
||||
return fmt.Errorf("stat file: %w", err)
|
||||
return fmt.Errorf("scriptorium artifact output exceeds or cannot be read within %d-byte limit: %w", MaxOutputFileBytes, err)
|
||||
}
|
||||
if info.IsDir() {
|
||||
return fmt.Errorf("path is a directory")
|
||||
}
|
||||
if info.Size() <= 0 {
|
||||
if len(data) == 0 {
|
||||
return fmt.Errorf("file is empty")
|
||||
}
|
||||
return nil
|
||||
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"fmt"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
// NoopRunner is a deterministic no-op seriatim adapter.
|
||||
@@ -260,7 +261,7 @@ func (f *FakeRunner) Render(ctx context.Context, req RenderRequest) (RenderResul
|
||||
|
||||
func materializePlaceholders(req MergeRequest) error {
|
||||
if req.OutputMergedTranscriptPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputMergedTranscriptPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputMergedTranscriptPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write merged transcript %q: %w", req.OutputMergedTranscriptPath, err)
|
||||
}
|
||||
}
|
||||
@@ -271,22 +272,22 @@ func materializePlaceholders(req MergeRequest) error {
|
||||
"input_transcript_paths": req.InputTranscriptPaths,
|
||||
"output_path": req.OutputMergedTranscriptPath,
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.ReportPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write report %q: %w", req.ReportPath, err)
|
||||
}
|
||||
}
|
||||
@@ -295,7 +296,7 @@ func materializePlaceholders(req MergeRequest) error {
|
||||
|
||||
func materializeTrimPlaceholders(req TrimRequest) error {
|
||||
if req.OutputTrimmedPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputTrimmedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputTrimmedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write trimmed transcript %q: %w", req.OutputTrimmedPath, err)
|
||||
}
|
||||
}
|
||||
@@ -308,17 +309,17 @@ func materializeTrimPlaceholders(req TrimRequest) error {
|
||||
"output_path": req.OutputTrimmedPath,
|
||||
"keep_selector": req.KeepSelector,
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake trim stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake trim stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake trim stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake trim stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
@@ -327,7 +328,7 @@ func materializeTrimPlaceholders(req TrimRequest) error {
|
||||
|
||||
func materializeNormalizePlaceholders(req NormalizeRequest) error {
|
||||
if req.OutputNormalizedPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputNormalizedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputNormalizedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write normalized transcript %q: %w", req.OutputNormalizedPath, err)
|
||||
}
|
||||
}
|
||||
@@ -343,22 +344,22 @@ func materializeNormalizePlaceholders(req NormalizeRequest) error {
|
||||
if req.ReportPath != "" {
|
||||
payload["report_path"] = req.ReportPath
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake normalize stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake normalize stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake normalize stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake normalize stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.ReportPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write report %q: %w", req.ReportPath, err)
|
||||
}
|
||||
}
|
||||
@@ -367,7 +368,7 @@ func materializeNormalizePlaceholders(req NormalizeRequest) error {
|
||||
|
||||
func materializeRenderPlaceholders(req RenderRequest) error {
|
||||
if req.OutputRenderedPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputRenderedPath, []byte("# Transcript\n\nRendered markdown placeholder.\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputRenderedPath, []byte("# Transcript\n\nRendered markdown placeholder.\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write rendered transcript %q: %w", req.OutputRenderedPath, err)
|
||||
}
|
||||
}
|
||||
@@ -384,17 +385,17 @@ func materializeRenderPlaceholders(req RenderRequest) error {
|
||||
"include_segment_ids": req.IncludeSegmentIDs,
|
||||
"include_metadata": req.IncludeMetadata,
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake render stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake render stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake render stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake render stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,15 +4,18 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
"unicode/utf8"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
// MaxOutputFileBytes bounds each Seriatim JSON or rendered-text result.
|
||||
const MaxOutputFileBytes int64 = 64 * 1024 * 1024
|
||||
|
||||
// EnvConfig defines optional Seriatim environment tuning values.
|
||||
type EnvConfig struct {
|
||||
OverlapWordRunGap *float64
|
||||
@@ -133,6 +136,7 @@ func (r *SubprocessRunner) Run(ctx context.Context, req MergeRequest) (MergeResu
|
||||
Args: args,
|
||||
Timeout: r.timeout,
|
||||
EnvOverrides: env,
|
||||
DiagnosticOwner: "seriatim",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
@@ -235,6 +239,7 @@ func (r *SubprocessRunner) Trim(ctx context.Context, req TrimRequest) (TrimResul
|
||||
Executable: binary,
|
||||
Args: args,
|
||||
Timeout: timeout,
|
||||
DiagnosticOwner: "seriatim",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
@@ -323,6 +328,7 @@ func (r *SubprocessRunner) Normalize(ctx context.Context, req NormalizeRequest)
|
||||
Executable: binary,
|
||||
Args: args,
|
||||
Timeout: timeout,
|
||||
DiagnosticOwner: "seriatim",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
@@ -428,6 +434,7 @@ func (r *SubprocessRunner) Render(ctx context.Context, req RenderRequest) (Rende
|
||||
Executable: binary,
|
||||
Args: args,
|
||||
Timeout: timeout,
|
||||
DiagnosticOwner: "seriatim",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
@@ -546,7 +553,7 @@ func (r *SubprocessRunner) writeMergeInvocationConfig(req MergeRequest, args []s
|
||||
payload["coalesce_gap"] = *r.coalesceGap
|
||||
}
|
||||
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
func buildTrimArgs(req TrimRequest) []string {
|
||||
@@ -598,7 +605,7 @@ func writeTrimInvocationConfig(req TrimRequest, args []string, binary string, ti
|
||||
"output_path": req.OutputTrimmedPath,
|
||||
"keep_selector": req.KeepSelector,
|
||||
}
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
func writeNormalizeInvocationConfig(req NormalizeRequest, args []string, binary string, timeout time.Duration, outputSchema string) error {
|
||||
@@ -613,7 +620,7 @@ func writeNormalizeInvocationConfig(req NormalizeRequest, args []string, binary
|
||||
"output_schema": outputSchema,
|
||||
"report_path": req.ReportPath,
|
||||
}
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
func writeRenderInvocationConfig(req RenderRequest, args []string, binary string, timeout time.Duration, format string) error {
|
||||
@@ -631,13 +638,13 @@ func writeRenderInvocationConfig(req RenderRequest, args []string, binary string
|
||||
"include_segment_ids": req.IncludeSegmentIDs,
|
||||
"include_metadata": req.IncludeMetadata,
|
||||
}
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
func validateJSONFile(path string) error {
|
||||
data, err := os.ReadFile(path)
|
||||
data, err := readSeriatimResult(path, "JSON output")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read file: %w", err)
|
||||
return err
|
||||
}
|
||||
var v any
|
||||
if err := json.Unmarshal(data, &v); err != nil {
|
||||
@@ -647,9 +654,9 @@ func validateJSONFile(path string) error {
|
||||
}
|
||||
|
||||
func validateJSONFileWithSegments(path string) error {
|
||||
data, err := os.ReadFile(path)
|
||||
data, err := readSeriatimResult(path, "transcript JSON output")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read file: %w", err)
|
||||
return err
|
||||
}
|
||||
|
||||
var payload map[string]any
|
||||
@@ -668,9 +675,9 @@ func validateJSONFileWithSegments(path string) error {
|
||||
}
|
||||
|
||||
func validateNonEmptyTextFile(path string) error {
|
||||
data, err := os.ReadFile(path)
|
||||
data, err := readSeriatimResult(path, "rendered text output")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read file: %w", err)
|
||||
return err
|
||||
}
|
||||
if len(data) == 0 {
|
||||
return fmt.Errorf("file is empty")
|
||||
@@ -683,3 +690,11 @@ func validateNonEmptyTextFile(path string) error {
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func readSeriatimResult(path, category string) ([]byte, error) {
|
||||
data, err := fileops.ReadRegularFile(path, MaxOutputFileBytes)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("seriatim %s exceeds or cannot be read within %d-byte limit: %w", category, MaxOutputFileBytes, err)
|
||||
}
|
||||
return data, nil
|
||||
}
|
||||
|
||||
90
internal/adapters/storage/bounded_read.go
Normal file
90
internal/adapters/storage/bounded_read.go
Normal file
@@ -0,0 +1,90 @@
|
||||
package storage
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"math"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ReadLimitError reports that a remote object exceeded its caller-owned read
|
||||
// limit. The limit is enforced against both available object metadata and the
|
||||
// bytes returned by the opened object body.
|
||||
type ReadLimitError struct {
|
||||
Key string
|
||||
Limit int64
|
||||
Observed int64
|
||||
}
|
||||
|
||||
func (e *ReadLimitError) Error() string {
|
||||
return fmt.Sprintf("object %q exceeds %d-byte read limit (observed at least %d bytes)", e.Key, e.Limit, e.Observed)
|
||||
}
|
||||
|
||||
// ReadObjectBounded opens one object version and retains at most maxBytes of
|
||||
// its content. Object metadata may reject an oversized body early, but a
|
||||
// limit-plus-one read always enforces the boundary when transfer begins.
|
||||
func ReadObjectBounded(ctx context.Context, store ObjectStore, key string, maxBytes int64) (info ObjectInfo, data []byte, err error) {
|
||||
key = strings.TrimSpace(key)
|
||||
if store == nil {
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read bounded object: store is required")
|
||||
}
|
||||
if key == "" {
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read bounded object: key is required")
|
||||
}
|
||||
if maxBytes <= 0 || maxBytes == math.MaxInt64 {
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read bounded object %q: limit must be between 1 and %d bytes", key, int64(math.MaxInt64-1))
|
||||
}
|
||||
if err := ctx.Err(); err != nil {
|
||||
return ObjectInfo{}, nil, err
|
||||
}
|
||||
|
||||
info, body, err := store.Read(ctx, key)
|
||||
if err != nil {
|
||||
return ObjectInfo{}, nil, err
|
||||
}
|
||||
if body == nil {
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read bounded object %q: store returned no body", key)
|
||||
}
|
||||
defer func() {
|
||||
if closeErr := body.Close(); closeErr != nil {
|
||||
data = nil
|
||||
err = errors.Join(err, fmt.Errorf("close object %q: %w", key, closeErr))
|
||||
}
|
||||
}()
|
||||
|
||||
if info.Size > maxBytes {
|
||||
return info, nil, &ReadLimitError{Key: key, Limit: maxBytes, Observed: info.Size}
|
||||
}
|
||||
|
||||
data, err = io.ReadAll(io.LimitReader(contextReader{ctx: ctx, reader: body}, maxBytes+1))
|
||||
if err != nil {
|
||||
return info, nil, err
|
||||
}
|
||||
if err := ctx.Err(); err != nil {
|
||||
return info, nil, err
|
||||
}
|
||||
if int64(len(data)) > maxBytes {
|
||||
return info, nil, &ReadLimitError{Key: key, Limit: maxBytes, Observed: int64(len(data))}
|
||||
}
|
||||
return info, data, nil
|
||||
}
|
||||
|
||||
type contextReader struct {
|
||||
ctx context.Context
|
||||
reader io.Reader
|
||||
}
|
||||
|
||||
func (r contextReader) Read(p []byte) (int, error) {
|
||||
if err := r.ctx.Err(); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
n, err := r.reader.Read(p)
|
||||
if err == nil {
|
||||
if contextErr := r.ctx.Err(); contextErr != nil {
|
||||
return n, contextErr
|
||||
}
|
||||
}
|
||||
return n, err
|
||||
}
|
||||
135
internal/adapters/storage/bounded_read_test.go
Normal file
135
internal/adapters/storage/bounded_read_test.go
Normal file
@@ -0,0 +1,135 @@
|
||||
package storage
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestReadObjectBoundedAcceptsExactLimitWithAbsentSizeMetadata(t *testing.T) {
|
||||
body := &trackingReadCloser{reader: bytes.NewReader([]byte("12345678")), chunkSize: 2}
|
||||
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||
return ObjectInfo{Key: "control.json", ETag: "generation"}, body, nil
|
||||
}}
|
||||
|
||||
info, data, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
|
||||
if err != nil {
|
||||
t.Fatalf("ReadObjectBounded() error = %v", err)
|
||||
}
|
||||
if string(data) != "12345678" || info.ETag != "generation" {
|
||||
t.Fatalf("ReadObjectBounded() = (%#v, %q), want opened object metadata and bytes", info, data)
|
||||
}
|
||||
if !body.closed {
|
||||
t.Fatal("object body was not closed")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadObjectBoundedRejectsLimitPlusOneDespiteMissingOrInaccurateMetadata(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
metadataSize int64
|
||||
}{
|
||||
{name: "missing", metadataSize: 0},
|
||||
{name: "inaccurate", metadataSize: 2},
|
||||
}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
body := &trackingReadCloser{reader: bytes.NewReader([]byte("123456789")), chunkSize: 1}
|
||||
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||
return ObjectInfo{Key: "control.json", Size: test.metadataSize}, body, nil
|
||||
}}
|
||||
|
||||
_, data, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
|
||||
var limitErr *ReadLimitError
|
||||
if !errors.As(err, &limitErr) {
|
||||
t.Fatalf("ReadObjectBounded() error = %v, want ReadLimitError", err)
|
||||
}
|
||||
if data != nil || body.bytesRead != 9 || !body.closed {
|
||||
t.Fatalf("data=%q bytes read=%d closed=%t, want nil, 9, true", data, body.bytesRead, body.closed)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadObjectBoundedRejectsOversizedMetadataBeforeTransfer(t *testing.T) {
|
||||
body := &trackingReadCloser{reader: bytes.NewReader([]byte("small"))}
|
||||
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||
return ObjectInfo{Key: "control.json", Size: 9}, body, nil
|
||||
}}
|
||||
|
||||
_, _, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
|
||||
var limitErr *ReadLimitError
|
||||
if !errors.As(err, &limitErr) {
|
||||
t.Fatalf("ReadObjectBounded() error = %v, want ReadLimitError", err)
|
||||
}
|
||||
if body.bytesRead != 0 || !body.closed {
|
||||
t.Fatalf("bytes read=%d closed=%t, want zero-byte transfer and closed body", body.bytesRead, body.closed)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadObjectBoundedPropagatesCancellationAndClosesBody(t *testing.T) {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
body := &trackingReadCloser{reader: bytes.NewReader([]byte("12345678")), chunkSize: 1, afterRead: cancel}
|
||||
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||
return ObjectInfo{Key: "control.json"}, body, nil
|
||||
}}
|
||||
|
||||
_, data, err := ReadObjectBounded(ctx, store, "control.json", 8)
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("ReadObjectBounded() error = %v, want context cancellation", err)
|
||||
}
|
||||
if data != nil || body.bytesRead != 1 || !body.closed {
|
||||
t.Fatalf("data=%q bytes read=%d closed=%t, want nil, 1, true", data, body.bytesRead, body.closed)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadObjectBoundedReturnsCloseFailure(t *testing.T) {
|
||||
closeErr := errors.New("close failed")
|
||||
body := &trackingReadCloser{reader: bytes.NewReader([]byte("ok")), closeErr: closeErr}
|
||||
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||
return ObjectInfo{Key: "control.json", Size: 2}, body, nil
|
||||
}}
|
||||
|
||||
_, data, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
|
||||
if !errors.Is(err, closeErr) || data != nil || !body.closed {
|
||||
t.Fatalf("data=%q error=%v closed=%t, want close failure and no retained data", data, err, body.closed)
|
||||
}
|
||||
}
|
||||
|
||||
type boundedReadStore struct {
|
||||
ObjectStore
|
||||
read func(context.Context, string) (ObjectInfo, io.ReadCloser, error)
|
||||
}
|
||||
|
||||
func (s *boundedReadStore) Read(ctx context.Context, key string) (ObjectInfo, io.ReadCloser, error) {
|
||||
return s.read(ctx, key)
|
||||
}
|
||||
|
||||
type trackingReadCloser struct {
|
||||
reader io.Reader
|
||||
chunkSize int
|
||||
afterRead func()
|
||||
closeErr error
|
||||
bytesRead int
|
||||
closed bool
|
||||
}
|
||||
|
||||
func (r *trackingReadCloser) Read(p []byte) (int, error) {
|
||||
if r.chunkSize > 0 && len(p) > r.chunkSize {
|
||||
p = p[:r.chunkSize]
|
||||
}
|
||||
n, err := r.reader.Read(p)
|
||||
r.bytesRead += n
|
||||
if n > 0 && r.afterRead != nil {
|
||||
r.afterRead()
|
||||
r.afterRead = nil
|
||||
}
|
||||
return n, err
|
||||
}
|
||||
|
||||
func (r *trackingReadCloser) Close() error {
|
||||
r.closed = true
|
||||
return r.closeErr
|
||||
}
|
||||
23
internal/adapters/storage/download_writer.go
Normal file
23
internal/adapters/storage/download_writer.go
Normal file
@@ -0,0 +1,23 @@
|
||||
package storage
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
)
|
||||
|
||||
// WriterDownloader is implemented by storage backends that stream an object
|
||||
// into a caller-owned file handle.
|
||||
type WriterDownloader interface {
|
||||
DownloadTo(ctx context.Context, key string, destination io.Writer) error
|
||||
}
|
||||
|
||||
// DownloadTo streams one object into destination. Destination-confined callers
|
||||
// require this capability rather than granting a backend a mutable pathname.
|
||||
func DownloadTo(ctx context.Context, store ObjectStore, key string, destination io.Writer) error {
|
||||
writer, ok := store.(WriterDownloader)
|
||||
if !ok {
|
||||
return fmt.Errorf("object store does not support handle-confined downloads")
|
||||
}
|
||||
return writer.DownloadTo(ctx, key, destination)
|
||||
}
|
||||
@@ -14,16 +14,15 @@ func NewObjectStoreFromConfig(ctx context.Context, cfg *config.Config) (ObjectSt
|
||||
return nil, fmt.Errorf("pipeline config is required")
|
||||
}
|
||||
|
||||
if strings.EqualFold(strings.TrimSpace(cfg.Pipeline.Storage.Backend), "s3") {
|
||||
switch strings.ToLower(strings.TrimSpace(cfg.Pipeline.Storage.Backend)) {
|
||||
case config.StorageBackendS3:
|
||||
if cfg.Pipeline.Storage.S3 == nil {
|
||||
return nil, fmt.Errorf("pipeline.storage.s3 is required when pipeline.storage.backend is s3")
|
||||
}
|
||||
return NewS3BackendFromConfig(ctx, *cfg.Pipeline.Storage.S3)
|
||||
}
|
||||
|
||||
if cfg.Pipeline.Storage.S3 != nil && strings.TrimSpace(cfg.Pipeline.Storage.S3.Bucket) != "" {
|
||||
return NewS3BackendFromConfig(ctx, *cfg.Pipeline.Storage.S3)
|
||||
}
|
||||
|
||||
case "", config.StorageBackendLocal:
|
||||
return nil, fmt.Errorf("no remote object store backend is configured")
|
||||
default:
|
||||
return nil, fmt.Errorf("unsupported pipeline.storage.backend %q", cfg.Pipeline.Storage.Backend)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -54,3 +54,35 @@ func TestNewObjectStoreFromConfigNoRemoteBackendConfigured(t *testing.T) {
|
||||
t.Fatalf("NewObjectStoreFromConfig() error = %v, want no-backend error", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewObjectStoreFromConfigDoesNotInferS3FromProviderFields(t *testing.T) {
|
||||
called := false
|
||||
original := newS3Client
|
||||
t.Cleanup(func() { newS3Client = original })
|
||||
newS3Client = func(_ context.Context, _ s3ClientOptions) (s3API, error) {
|
||||
called = true
|
||||
return &fakeS3API{}, nil
|
||||
}
|
||||
|
||||
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
|
||||
Pipeline: &config.PipelineConfig{Storage: config.StorageConfig{
|
||||
Backend: config.StorageBackendLocal,
|
||||
S3: &config.StorageS3Config{Bucket: "my-archive"},
|
||||
}},
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), "no remote object store backend is configured") {
|
||||
t.Fatalf("NewObjectStoreFromConfig() error = %v, want no-backend error", err)
|
||||
}
|
||||
if called {
|
||||
t.Fatal("NewObjectStoreFromConfig() constructed S3 from incidental provider fields")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewObjectStoreFromConfigRejectsUnknownBackend(t *testing.T) {
|
||||
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
|
||||
Pipeline: &config.PipelineConfig{Storage: config.StorageConfig{Backend: "s33"}},
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), "unsupported pipeline.storage.backend") {
|
||||
t.Fatalf("NewObjectStoreFromConfig() error = %v, want unsupported-backend error", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,25 +1,35 @@
|
||||
package storage
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// FakeBackend provides a deterministic in-memory object store for tests.
|
||||
type FakeBackend struct {
|
||||
mu sync.RWMutex
|
||||
|
||||
Objects map[string]FakeObject
|
||||
Uploads []FakeUploadCall
|
||||
Downloads []FakeDownloadCall
|
||||
Reads []FakeReadCall
|
||||
|
||||
ListErr error
|
||||
DownloadErr error
|
||||
UploadErr error
|
||||
ExistsErr error
|
||||
UploadHook func(FakeUploadCall) error
|
||||
DownloadHook func(FakeDownloadCall) error
|
||||
}
|
||||
|
||||
// FakeUploadCall captures one upload invocation in call order.
|
||||
@@ -33,6 +43,12 @@ type FakeUploadCall struct {
|
||||
type FakeDownloadCall struct {
|
||||
Key string
|
||||
LocalPath string
|
||||
Bytes int64
|
||||
}
|
||||
|
||||
// FakeReadCall captures one opened object in call order.
|
||||
type FakeReadCall struct {
|
||||
Key string
|
||||
}
|
||||
|
||||
// FakeObject is a deterministic fake object-store record.
|
||||
@@ -46,6 +62,12 @@ type FakeObject struct {
|
||||
|
||||
// SeedObject inserts or replaces an object in the fake object store.
|
||||
func (f *FakeBackend) SeedObject(obj FakeObject) {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
f.seedObject(obj)
|
||||
}
|
||||
|
||||
func (f *FakeBackend) seedObject(obj FakeObject) {
|
||||
if f.Objects == nil {
|
||||
f.Objects = map[string]FakeObject{}
|
||||
}
|
||||
@@ -53,6 +75,9 @@ func (f *FakeBackend) SeedObject(obj FakeObject) {
|
||||
obj.Key = key
|
||||
obj.Data = append([]byte(nil), obj.Data...)
|
||||
obj.Metadata = copyMetadata(obj.Metadata)
|
||||
if obj.ETag == "" {
|
||||
obj.ETag = fakeObjectETag(obj.Data)
|
||||
}
|
||||
f.Objects[key] = obj
|
||||
}
|
||||
|
||||
@@ -65,6 +90,8 @@ func (f *FakeBackend) List(ctx context.Context, prefix string) ([]ObjectInfo, er
|
||||
return nil, f.ListErr
|
||||
}
|
||||
|
||||
f.mu.RLock()
|
||||
defer f.mu.RUnlock()
|
||||
normalizedPrefix := normalizeObjectKey(prefix)
|
||||
keys := make([]string, 0, len(f.Objects))
|
||||
for key := range f.Objects {
|
||||
@@ -87,34 +114,77 @@ func (f *FakeBackend) List(ctx context.Context, prefix string) ([]ObjectInfo, er
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Download writes one object to a local path.
|
||||
func (f *FakeBackend) Download(ctx context.Context, key, localPath string) error {
|
||||
// Read returns a stable object body and the generation observed with it.
|
||||
func (f *FakeBackend) Read(ctx context.Context, key string) (ObjectInfo, io.ReadCloser, error) {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return ObjectInfo{}, nil, err
|
||||
}
|
||||
if f.DownloadErr != nil {
|
||||
return ObjectInfo{}, nil, f.DownloadErr
|
||||
}
|
||||
normalizedKey := normalizeObjectKey(key)
|
||||
f.mu.RLock()
|
||||
obj, ok := f.Objects[normalizedKey]
|
||||
if ok {
|
||||
obj.Data = append([]byte(nil), obj.Data...)
|
||||
obj.Metadata = copyMetadata(obj.Metadata)
|
||||
}
|
||||
f.mu.RUnlock()
|
||||
if !ok {
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read object %q: %w", normalizedKey, os.ErrNotExist)
|
||||
}
|
||||
f.mu.Lock()
|
||||
f.Reads = append(f.Reads, FakeReadCall{Key: normalizedKey})
|
||||
f.mu.Unlock()
|
||||
return ObjectInfo{Key: obj.Key, Size: int64(len(obj.Data)), ETag: obj.ETag, LastModified: obj.LastModified}, io.NopCloser(bytes.NewReader(obj.Data)), nil
|
||||
}
|
||||
|
||||
// DownloadTo writes one object to a caller-owned destination writer.
|
||||
func (f *FakeBackend) DownloadTo(ctx context.Context, key string, destination io.Writer) error {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
if f.DownloadErr != nil {
|
||||
return f.DownloadErr
|
||||
}
|
||||
if destination == nil {
|
||||
return fmt.Errorf("download object: destination writer is required")
|
||||
}
|
||||
if f.DownloadHook != nil {
|
||||
if err := f.DownloadHook(FakeDownloadCall{Key: normalizeObjectKey(key)}); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
_, source, err := f.Read(ctx, key)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer source.Close()
|
||||
count, err := io.Copy(destination, source)
|
||||
if err != nil {
|
||||
return fmt.Errorf("download object %q: write destination: %w", key, err)
|
||||
}
|
||||
f.mu.Lock()
|
||||
f.Downloads = append(f.Downloads, FakeDownloadCall{Key: normalizeObjectKey(key), Bytes: count})
|
||||
f.mu.Unlock()
|
||||
return nil
|
||||
}
|
||||
|
||||
// Download writes one object to a local path.
|
||||
func (f *FakeBackend) Download(ctx context.Context, key, localPath string) error {
|
||||
if strings.TrimSpace(localPath) == "" {
|
||||
return fmt.Errorf("download object: local path is required")
|
||||
}
|
||||
|
||||
obj, ok := f.Objects[normalizeObjectKey(key)]
|
||||
if !ok {
|
||||
return fmt.Errorf("download object %q: %w", key, os.ErrNotExist)
|
||||
}
|
||||
f.Downloads = append(f.Downloads, FakeDownloadCall{
|
||||
Key: normalizeObjectKey(key),
|
||||
LocalPath: localPath,
|
||||
})
|
||||
|
||||
if err := os.MkdirAll(filepath.Dir(localPath), 0o755); err != nil {
|
||||
return fmt.Errorf("download object %q: create parent directory: %w", key, err)
|
||||
}
|
||||
if err := os.WriteFile(localPath, obj.Data, 0o644); err != nil {
|
||||
return fmt.Errorf("download object %q: write local file: %w", key, err)
|
||||
destination, err := os.Create(localPath)
|
||||
if err != nil {
|
||||
return fmt.Errorf("download object %q: create local file: %w", key, err)
|
||||
}
|
||||
return nil
|
||||
defer destination.Close()
|
||||
return f.DownloadTo(ctx, key, destination)
|
||||
}
|
||||
|
||||
// Upload reads a local file and stores it under key.
|
||||
@@ -132,33 +202,117 @@ func (f *FakeBackend) Upload(ctx context.Context, localPath, key string, opts Up
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
|
||||
}
|
||||
|
||||
data, err := os.ReadFile(localPath)
|
||||
file, err := os.Open(localPath)
|
||||
if err != nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", key, localPath, err)
|
||||
}
|
||||
defer file.Close()
|
||||
return f.uploadReader(ctx, file, key, opts, localPath)
|
||||
}
|
||||
|
||||
// UploadReader stores content provided by a caller-owned reader.
|
||||
func (f *FakeBackend) UploadReader(ctx context.Context, source io.Reader, key string, opts UploadOptions) (ObjectInfo, error) {
|
||||
return f.uploadReader(ctx, source, key, opts, "reader")
|
||||
}
|
||||
|
||||
func (f *FakeBackend) uploadReader(ctx context.Context, source io.Reader, key string, opts UploadOptions, localPath string) (ObjectInfo, error) {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return ObjectInfo{}, err
|
||||
}
|
||||
if f.UploadErr != nil {
|
||||
return ObjectInfo{}, f.UploadErr
|
||||
}
|
||||
if source == nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: source is required")
|
||||
}
|
||||
if strings.TrimSpace(key) == "" {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
|
||||
}
|
||||
|
||||
data, err := io.ReadAll(source)
|
||||
if err != nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", key, localPath, err)
|
||||
}
|
||||
|
||||
normalizedKey := normalizeObjectKey(key)
|
||||
f.Uploads = append(f.Uploads, FakeUploadCall{
|
||||
call := FakeUploadCall{
|
||||
LocalPath: localPath,
|
||||
Key: normalizedKey,
|
||||
Options: UploadOptions{
|
||||
Metadata: copyMetadata(opts.Metadata),
|
||||
ContentType: opts.ContentType,
|
||||
},
|
||||
})
|
||||
now := time.Now().UTC()
|
||||
obj := FakeObject{
|
||||
Key: normalizedKey,
|
||||
Data: data,
|
||||
Metadata: copyMetadata(opts.Metadata),
|
||||
LastModified: &now,
|
||||
}
|
||||
f.SeedObject(obj)
|
||||
return ObjectInfo{
|
||||
Key: normalizedKey,
|
||||
Size: int64(len(data)),
|
||||
LastModified: &now,
|
||||
}, nil
|
||||
f.mu.Lock()
|
||||
f.Uploads = append(f.Uploads, call)
|
||||
f.mu.Unlock()
|
||||
if f.UploadHook != nil {
|
||||
if err := f.UploadHook(call); err != nil {
|
||||
return ObjectInfo{}, err
|
||||
}
|
||||
}
|
||||
return f.storeUploadedObject(normalizedKey, data, opts), nil
|
||||
}
|
||||
|
||||
// UploadConditional atomically checks and replaces one mutable object.
|
||||
func (f *FakeBackend) UploadConditional(ctx context.Context, source io.Reader, key string, opts UploadOptions, condition WriteCondition) (ObjectInfo, error) {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return ObjectInfo{}, err
|
||||
}
|
||||
if err := validateWriteCondition(condition); err != nil {
|
||||
return ObjectInfo{}, err
|
||||
}
|
||||
if f.UploadErr != nil {
|
||||
return ObjectInfo{}, f.UploadErr
|
||||
}
|
||||
if source == nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: source is required")
|
||||
}
|
||||
normalizedKey := normalizeObjectKey(key)
|
||||
if normalizedKey == "" {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
|
||||
}
|
||||
data, err := io.ReadAll(source)
|
||||
if err != nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q: %w", normalizedKey, err)
|
||||
}
|
||||
call := FakeUploadCall{Key: normalizedKey, Options: UploadOptions{Metadata: copyMetadata(opts.Metadata), ContentType: opts.ContentType}}
|
||||
f.mu.Lock()
|
||||
f.Uploads = append(f.Uploads, call)
|
||||
f.mu.Unlock()
|
||||
if f.UploadHook != nil {
|
||||
if err := f.UploadHook(call); err != nil {
|
||||
return ObjectInfo{}, err
|
||||
}
|
||||
}
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
existing, found := f.Objects[normalizedKey]
|
||||
if condition.RequireAbsent && found {
|
||||
return ObjectInfo{}, ErrConditionNotMet
|
||||
}
|
||||
if expected := strings.TrimSpace(condition.MatchETag); expected != "" && (!found || existing.ETag != expected) {
|
||||
return ObjectInfo{}, ErrConditionNotMet
|
||||
}
|
||||
return f.storeUploadedObjectLocked(normalizedKey, data, opts), nil
|
||||
}
|
||||
|
||||
func (f *FakeBackend) storeUploadedObject(key string, data []byte, opts UploadOptions) ObjectInfo {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
return f.storeUploadedObjectLocked(key, data, opts)
|
||||
}
|
||||
|
||||
func (f *FakeBackend) storeUploadedObjectLocked(key string, data []byte, opts UploadOptions) ObjectInfo {
|
||||
now := time.Now().UTC()
|
||||
obj := FakeObject{Key: key, Data: append([]byte(nil), data...), Metadata: copyMetadata(opts.Metadata), LastModified: &now}
|
||||
f.seedObject(obj)
|
||||
return ObjectInfo{Key: key, Size: int64(len(data)), ETag: fakeObjectETag(data), LastModified: &now}
|
||||
}
|
||||
|
||||
func fakeObjectETag(data []byte) string {
|
||||
sum := sha256.Sum256(data)
|
||||
return hex.EncodeToString(sum[:])
|
||||
}
|
||||
|
||||
// Exists checks object presence.
|
||||
@@ -169,7 +323,9 @@ func (f *FakeBackend) Exists(ctx context.Context, key string) (bool, error) {
|
||||
if f.ExistsErr != nil {
|
||||
return false, f.ExistsErr
|
||||
}
|
||||
f.mu.RLock()
|
||||
_, ok := f.Objects[normalizeObjectKey(key)]
|
||||
f.mu.RUnlock()
|
||||
return ok, nil
|
||||
}
|
||||
|
||||
|
||||
@@ -71,6 +71,21 @@ func TestFakeBackendUploadAndExists(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestFakeBackendConditionalUploadRejectsStaleGeneration(t *testing.T) {
|
||||
fake := &FakeBackend{}
|
||||
fake.SeedObject(FakeObject{Key: "locks.yml", Data: []byte("old")})
|
||||
old := fake.Objects["locks.yml"].ETag
|
||||
if _, err := fake.UploadConditional(context.Background(), strings.NewReader("new"), "locks.yml", UploadOptions{}, WriteCondition{MatchETag: old}); err != nil {
|
||||
t.Fatalf("UploadConditional() error = %v", err)
|
||||
}
|
||||
if _, err := fake.UploadConditional(context.Background(), strings.NewReader("lost"), "locks.yml", UploadOptions{}, WriteCondition{MatchETag: old}); !errors.Is(err, ErrConditionNotMet) {
|
||||
t.Fatalf("UploadConditional() error = %v, want ErrConditionNotMet", err)
|
||||
}
|
||||
if got := string(fake.Objects["locks.yml"].Data); got != "new" {
|
||||
t.Fatalf("locks object = %q, want successful replacement preserved", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFakeBackendObjectErrors(t *testing.T) {
|
||||
fake := &FakeBackend{DownloadErr: errors.New("download fail"), UploadErr: errors.New("upload fail"), ListErr: errors.New("list fail"), ExistsErr: errors.New("exists fail")}
|
||||
|
||||
|
||||
@@ -2,9 +2,21 @@ package storage
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"time"
|
||||
)
|
||||
|
||||
// ErrConditionNotMet reports that an object changed or already existed before a
|
||||
// conditional write could be committed.
|
||||
var ErrConditionNotMet = errors.New("object write condition not met")
|
||||
|
||||
// ReaderUploader streams caller-owned, already-opened content to object storage.
|
||||
// Callers retain source-selection and filesystem-confinement policy.
|
||||
type ReaderUploader interface {
|
||||
UploadReader(ctx context.Context, source io.Reader, key string, opts UploadOptions) (ObjectInfo, error)
|
||||
}
|
||||
|
||||
// ObjectStore is a remote object storage boundary used by prepare, restore, and publish work.
|
||||
//
|
||||
// Key invariant:
|
||||
@@ -12,8 +24,10 @@ import (
|
||||
// infer Narratio session semantics and do not prepend root prefixes.
|
||||
type ObjectStore interface {
|
||||
List(ctx context.Context, prefix string) ([]ObjectInfo, error)
|
||||
Read(ctx context.Context, key string) (ObjectInfo, io.ReadCloser, error)
|
||||
Download(ctx context.Context, key, localPath string) error
|
||||
Upload(ctx context.Context, localPath, key string, opts UploadOptions) (ObjectInfo, error)
|
||||
UploadConditional(ctx context.Context, source io.Reader, key string, opts UploadOptions, condition WriteCondition) (ObjectInfo, error)
|
||||
Exists(ctx context.Context, key string) (bool, error)
|
||||
}
|
||||
|
||||
@@ -30,3 +44,10 @@ type UploadOptions struct {
|
||||
Metadata map[string]string
|
||||
ContentType string
|
||||
}
|
||||
|
||||
// WriteCondition protects a mutable object update against a stale snapshot.
|
||||
// Exactly one condition is required by UploadConditional.
|
||||
type WriteCondition struct {
|
||||
MatchETag string
|
||||
RequireAbsent bool
|
||||
}
|
||||
|
||||
@@ -117,6 +117,7 @@ func (b *S3Backend) List(ctx context.Context, prefix string) ([]ObjectInfo, erro
|
||||
normalizedPrefix := normalizeObjectKey(prefix)
|
||||
out := make([]ObjectInfo, 0)
|
||||
var token *string
|
||||
seenTokens := map[string]struct{}{}
|
||||
|
||||
for {
|
||||
resp, err := b.client.ListObjectsV2(ctx, &s3.ListObjectsV2Input{
|
||||
@@ -142,44 +143,78 @@ func (b *S3Backend) List(ctx context.Context, prefix string) ([]ObjectInfo, erro
|
||||
})
|
||||
}
|
||||
|
||||
if !valueOrFalseBool(resp.IsTruncated) || resp.NextContinuationToken == nil {
|
||||
if !valueOrFalseBool(resp.IsTruncated) {
|
||||
break
|
||||
}
|
||||
token = resp.NextContinuationToken
|
||||
next := strings.TrimSpace(valueOrEmpty(resp.NextContinuationToken))
|
||||
if next == "" {
|
||||
return nil, fmt.Errorf("s3 list objects bucket %q prefix %q: truncated response has an empty continuation token", b.bucket, normalizedPrefix)
|
||||
}
|
||||
if _, repeated := seenTokens[next]; repeated {
|
||||
return nil, fmt.Errorf("s3 list objects bucket %q prefix %q: truncated response repeated continuation token", b.bucket, normalizedPrefix)
|
||||
}
|
||||
seenTokens[next] = struct{}{}
|
||||
token = &next
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Read retrieves an object together with the generation observed for its body.
|
||||
func (b *S3Backend) Read(ctx context.Context, key string) (ObjectInfo, io.ReadCloser, error) {
|
||||
normalizedKey := normalizeObjectKey(key)
|
||||
resp, err := b.client.GetObject(ctx, &s3.GetObjectInput{Bucket: &b.bucket, Key: &normalizedKey})
|
||||
if err != nil {
|
||||
if isS3NotFound(err) {
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read object %q: %w", normalizedKey, os.ErrNotExist)
|
||||
}
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read object %q: %w", normalizedKey, err)
|
||||
}
|
||||
var lastModified *time.Time
|
||||
if resp.LastModified != nil {
|
||||
t := *resp.LastModified
|
||||
lastModified = &t
|
||||
}
|
||||
return ObjectInfo{
|
||||
Key: normalizedKey, Size: valueOrZeroInt64(resp.ContentLength),
|
||||
ETag: strings.Trim(valueOrEmpty(resp.ETag), "\""), LastModified: lastModified,
|
||||
}, resp.Body, nil
|
||||
}
|
||||
|
||||
// DownloadTo retrieves one object into the caller-owned destination writer.
|
||||
func (b *S3Backend) DownloadTo(ctx context.Context, key string, destination io.Writer) error {
|
||||
if destination == nil {
|
||||
return fmt.Errorf("download object: destination writer is required")
|
||||
}
|
||||
_, body, err := b.Read(ctx, key)
|
||||
if err != nil {
|
||||
return fmt.Errorf("download object %q: %w", normalizeObjectKey(key), err)
|
||||
}
|
||||
defer body.Close()
|
||||
|
||||
if _, err := io.Copy(destination, body); err != nil {
|
||||
return fmt.Errorf("download object %q: copy body: %w", normalizeObjectKey(key), err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Download retrieves one object to localPath, creating parent directories as needed.
|
||||
func (b *S3Backend) Download(ctx context.Context, key, localPath string) error {
|
||||
normalizedKey := normalizeObjectKey(key)
|
||||
if strings.TrimSpace(localPath) == "" {
|
||||
return fmt.Errorf("download object: local path is required")
|
||||
}
|
||||
|
||||
resp, err := b.client.GetObject(ctx, &s3.GetObjectInput{
|
||||
Bucket: &b.bucket,
|
||||
Key: &normalizedKey,
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("download object %q: %w", normalizedKey, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if err := os.MkdirAll(filepath.Dir(localPath), 0o755); err != nil {
|
||||
return fmt.Errorf("download object %q: create parent directory: %w", normalizedKey, err)
|
||||
return fmt.Errorf("download object %q: create parent directory: %w", key, err)
|
||||
}
|
||||
dst, err := os.Create(localPath)
|
||||
if err != nil {
|
||||
return fmt.Errorf("download object %q: create local file: %w", normalizedKey, err)
|
||||
return fmt.Errorf("download object %q: create local file: %w", key, err)
|
||||
}
|
||||
defer dst.Close()
|
||||
|
||||
if _, err := io.Copy(dst, resp.Body); err != nil {
|
||||
return fmt.Errorf("download object %q: copy body: %w", normalizedKey, err)
|
||||
if err := b.DownloadTo(ctx, key, dst); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := dst.Sync(); err != nil {
|
||||
return fmt.Errorf("download object %q: sync local file: %w", normalizedKey, err)
|
||||
return fmt.Errorf("download object %q: sync local file: %w", key, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -204,28 +239,65 @@ func (b *S3Backend) Upload(ctx context.Context, localPath, key string, opts Uplo
|
||||
if err != nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: stat local file: %w", normalizedKey, localPath, err)
|
||||
}
|
||||
return b.uploadReader(ctx, file, key, opts, stat.Size(), WriteCondition{})
|
||||
}
|
||||
|
||||
// UploadReader sends caller-owned content to key.
|
||||
func (b *S3Backend) UploadReader(ctx context.Context, source io.Reader, key string, opts UploadOptions) (ObjectInfo, error) {
|
||||
return b.uploadReader(ctx, source, key, opts, 0, WriteCondition{})
|
||||
}
|
||||
|
||||
// UploadConditional uploads a mutable object only when its observed generation
|
||||
// still matches, or when no object exists yet.
|
||||
func (b *S3Backend) UploadConditional(ctx context.Context, source io.Reader, key string, opts UploadOptions, condition WriteCondition) (ObjectInfo, error) {
|
||||
if err := validateWriteCondition(condition); err != nil {
|
||||
return ObjectInfo{}, err
|
||||
}
|
||||
return b.uploadReader(ctx, source, key, opts, 0, condition)
|
||||
}
|
||||
|
||||
func (b *S3Backend) uploadReader(ctx context.Context, source io.Reader, key string, opts UploadOptions, size int64, condition WriteCondition) (ObjectInfo, error) {
|
||||
normalizedKey := normalizeObjectKey(key)
|
||||
if source == nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: source is required")
|
||||
}
|
||||
if normalizedKey == "" {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
|
||||
}
|
||||
|
||||
input := &s3.PutObjectInput{
|
||||
Bucket: &b.bucket,
|
||||
Key: &normalizedKey,
|
||||
Body: file,
|
||||
Body: source,
|
||||
Metadata: copyMetadata(opts.Metadata),
|
||||
}
|
||||
if strings.TrimSpace(opts.ContentType) != "" {
|
||||
ct := strings.TrimSpace(opts.ContentType)
|
||||
input.ContentType = &ct
|
||||
}
|
||||
if condition.RequireAbsent {
|
||||
wildcard := "*"
|
||||
input.IfNoneMatch = &wildcard
|
||||
} else if expected := strings.TrimSpace(condition.MatchETag); expected != "" {
|
||||
input.IfMatch = &expected
|
||||
}
|
||||
|
||||
resp, err := b.client.PutObject(ctx, input)
|
||||
if err != nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", normalizedKey, localPath, err)
|
||||
if isS3ConditionalConflict(err) {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q: %w", normalizedKey, ErrConditionNotMet)
|
||||
}
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q: %w", normalizedKey, err)
|
||||
}
|
||||
|
||||
return ObjectInfo{
|
||||
info := ObjectInfo{
|
||||
Key: normalizedKey,
|
||||
Size: stat.Size(),
|
||||
ETag: strings.Trim(valueOrEmpty(resp.ETag), "\""),
|
||||
}, nil
|
||||
}
|
||||
if size > 0 {
|
||||
info.Size = size
|
||||
}
|
||||
return info, nil
|
||||
}
|
||||
|
||||
// Exists checks whether one object key exists.
|
||||
@@ -239,18 +311,43 @@ func (b *S3Backend) Exists(ctx context.Context, key string) (bool, error) {
|
||||
return true, nil
|
||||
}
|
||||
|
||||
if isS3NotFound(err) {
|
||||
return false, nil
|
||||
}
|
||||
return false, fmt.Errorf("head object %q: %w", normalizedKey, err)
|
||||
}
|
||||
|
||||
func validateWriteCondition(condition WriteCondition) error {
|
||||
if condition.RequireAbsent == (strings.TrimSpace(condition.MatchETag) != "") {
|
||||
return fmt.Errorf("conditional upload requires exactly one of MatchETag or RequireAbsent")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func isS3NotFound(err error) bool {
|
||||
var notFound *types.NotFound
|
||||
if errors.As(err, ¬Found) {
|
||||
return false, nil
|
||||
return true
|
||||
}
|
||||
var apiErr smithy.APIError
|
||||
if errors.As(err, &apiErr) {
|
||||
switch apiErr.ErrorCode() {
|
||||
case "NotFound", "NoSuchKey", "404":
|
||||
return false, nil
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false, fmt.Errorf("head object %q: %w", normalizedKey, err)
|
||||
return false
|
||||
}
|
||||
|
||||
func isS3ConditionalConflict(err error) bool {
|
||||
var apiErr smithy.APIError
|
||||
if errors.As(err, &apiErr) {
|
||||
switch apiErr.ErrorCode() {
|
||||
case "PreconditionFailed", "ConditionalRequestConflict", "412", "409":
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func valueOrEmpty(v *string) string {
|
||||
|
||||
@@ -2,6 +2,7 @@ package storage
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
@@ -18,10 +19,15 @@ import (
|
||||
|
||||
type fakeS3API struct {
|
||||
listOut *s3.ListObjectsV2Output
|
||||
listOutputs []*s3.ListObjectsV2Output
|
||||
listErr error
|
||||
listCalls int
|
||||
|
||||
getBody io.ReadCloser
|
||||
getErr error
|
||||
getSize *int64
|
||||
getETag *string
|
||||
getLastModified *time.Time
|
||||
|
||||
putOut *s3.PutObjectOutput
|
||||
putErr error
|
||||
@@ -29,6 +35,7 @@ type fakeS3API struct {
|
||||
headErr error
|
||||
|
||||
lastList *s3.ListObjectsV2Input
|
||||
lastLists []*s3.ListObjectsV2Input
|
||||
lastGet *s3.GetObjectInput
|
||||
lastPut *s3.PutObjectInput
|
||||
lastHead *s3.HeadObjectInput
|
||||
@@ -36,15 +43,56 @@ type fakeS3API struct {
|
||||
|
||||
func (f *fakeS3API) ListObjectsV2(_ context.Context, params *s3.ListObjectsV2Input, _ ...func(*s3.Options)) (*s3.ListObjectsV2Output, error) {
|
||||
f.lastList = params
|
||||
f.lastLists = append(f.lastLists, params)
|
||||
if f.listErr != nil {
|
||||
return nil, f.listErr
|
||||
}
|
||||
if f.listCalls < len(f.listOutputs) {
|
||||
out := f.listOutputs[f.listCalls]
|
||||
f.listCalls++
|
||||
return out, nil
|
||||
}
|
||||
if f.listOut == nil {
|
||||
return &s3.ListObjectsV2Output{}, nil
|
||||
}
|
||||
return f.listOut, nil
|
||||
}
|
||||
|
||||
func TestS3BackendListPaginatesAndRejectsNonProgressingTokens(t *testing.T) {
|
||||
t.Run("multiple pages", func(t *testing.T) {
|
||||
client := &fakeS3API{listOutputs: []*s3.ListObjectsV2Output{
|
||||
{Contents: []types.Object{{Key: strPtr("prefix/a"), Size: int64Ptr(1)}}, IsTruncated: boolPtr(true), NextContinuationToken: strPtr("next")},
|
||||
{Contents: []types.Object{{Key: strPtr("prefix/b"), Size: int64Ptr(2)}}, IsTruncated: boolPtr(false)},
|
||||
}}
|
||||
items, err := (&S3Backend{bucket: "bucket-1", client: client}).List(context.Background(), "prefix/")
|
||||
if err != nil {
|
||||
t.Fatalf("List() error = %v", err)
|
||||
}
|
||||
if len(items) != 2 || items[0].Key != "prefix/a" || items[1].Key != "prefix/b" {
|
||||
t.Fatalf("List() items = %#v", items)
|
||||
}
|
||||
if len(client.lastLists) != 2 || client.lastLists[1].ContinuationToken == nil || *client.lastLists[1].ContinuationToken != "next" {
|
||||
t.Fatalf("continuation calls = %#v", client.lastLists)
|
||||
}
|
||||
})
|
||||
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
outputs []*s3.ListObjectsV2Output
|
||||
want string
|
||||
}{
|
||||
{name: "empty", outputs: []*s3.ListObjectsV2Output{{IsTruncated: boolPtr(true)}}, want: "empty continuation token"},
|
||||
{name: "repeated", outputs: []*s3.ListObjectsV2Output{{IsTruncated: boolPtr(true), NextContinuationToken: strPtr("again")}, {IsTruncated: boolPtr(true), NextContinuationToken: strPtr("again")}}, want: "repeated continuation token"},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
_, err := (&S3Backend{bucket: "bucket-1", client: &fakeS3API{listOutputs: test.outputs}}).List(context.Background(), "prefix/")
|
||||
if err == nil || !strings.Contains(err.Error(), test.want) || !strings.Contains(err.Error(), "bucket-1") || !strings.Contains(err.Error(), "prefix/") {
|
||||
t.Fatalf("List() error = %v, want contextual %q", err, test.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func (f *fakeS3API) GetObject(_ context.Context, params *s3.GetObjectInput, _ ...func(*s3.Options)) (*s3.GetObjectOutput, error) {
|
||||
f.lastGet = params
|
||||
if f.getErr != nil {
|
||||
@@ -54,7 +102,7 @@ func (f *fakeS3API) GetObject(_ context.Context, params *s3.GetObjectInput, _ ..
|
||||
if body == nil {
|
||||
body = io.NopCloser(strings.NewReader(""))
|
||||
}
|
||||
return &s3.GetObjectOutput{Body: body}, nil
|
||||
return &s3.GetObjectOutput{Body: body, ContentLength: f.getSize, ETag: f.getETag, LastModified: f.getLastModified}, nil
|
||||
}
|
||||
|
||||
func (f *fakeS3API) PutObject(_ context.Context, params *s3.PutObjectInput, _ ...func(*s3.Options)) (*s3.PutObjectOutput, error) {
|
||||
@@ -126,6 +174,33 @@ func TestS3BackendDownloadCreatesParentDirectory(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestS3BackendReadReturnsOpenedObjectMetadata(t *testing.T) {
|
||||
lastModified := time.Date(2026, 8, 11, 1, 2, 3, 0, time.UTC)
|
||||
client := &fakeS3API{
|
||||
getBody: io.NopCloser(strings.NewReader("locks")),
|
||||
getSize: int64Ptr(5),
|
||||
getETag: strPtr(`"generation"`),
|
||||
getLastModified: &lastModified,
|
||||
}
|
||||
backend := &S3Backend{bucket: "bucket-1", client: client}
|
||||
|
||||
info, body, err := backend.Read(context.Background(), `sessions\locks.yml`)
|
||||
if err != nil {
|
||||
t.Fatalf("Read() error = %v", err)
|
||||
}
|
||||
data, readErr := io.ReadAll(body)
|
||||
closeErr := body.Close()
|
||||
if readErr != nil || closeErr != nil {
|
||||
t.Fatalf("read body error=%v close error=%v", readErr, closeErr)
|
||||
}
|
||||
if info.Key != "sessions/locks.yml" || info.Size != 5 || info.ETag != "generation" || info.LastModified == nil || !info.LastModified.Equal(lastModified) {
|
||||
t.Fatalf("Read() info = %#v, want opened object metadata", info)
|
||||
}
|
||||
if string(data) != "locks" || client.lastGet == nil || *client.lastGet.Key != "sessions/locks.yml" {
|
||||
t.Fatalf("Read() data=%q request=%#v", data, client.lastGet)
|
||||
}
|
||||
}
|
||||
|
||||
func TestS3BackendUploadAndExists(t *testing.T) {
|
||||
client := &fakeS3API{putOut: &s3.PutObjectOutput{ETag: strPtr(`"etag123"`)}}
|
||||
backend := &S3Backend{bucket: "bucket-1", client: client}
|
||||
@@ -160,6 +235,22 @@ func TestS3BackendUploadAndExists(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestS3BackendConditionalUploadUsesProviderPrecondition(t *testing.T) {
|
||||
client := &fakeS3API{putOut: &s3.PutObjectOutput{ETag: strPtr(`"etag123"`)}}
|
||||
backend := &S3Backend{bucket: "bucket-1", client: client}
|
||||
if _, err := backend.UploadConditional(context.Background(), strings.NewReader("payload"), "locks.yml", UploadOptions{}, WriteCondition{MatchETag: "before"}); err != nil {
|
||||
t.Fatalf("UploadConditional() error = %v", err)
|
||||
}
|
||||
if client.lastPut == nil || client.lastPut.IfMatch == nil || *client.lastPut.IfMatch != "before" || client.lastPut.IfNoneMatch != nil {
|
||||
t.Fatalf("PutObject conditional input = %#v", client.lastPut)
|
||||
}
|
||||
client.putErr = &smithy.GenericAPIError{Code: "PreconditionFailed", Message: "changed"}
|
||||
_, err := backend.UploadConditional(context.Background(), strings.NewReader("payload"), "locks.yml", UploadOptions{}, WriteCondition{RequireAbsent: true})
|
||||
if !errors.Is(err, ErrConditionNotMet) {
|
||||
t.Fatalf("UploadConditional() error = %v, want ErrConditionNotMet", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestS3BackendUploadMissingLocalFile(t *testing.T) {
|
||||
backend := &S3Backend{bucket: "bucket-1", client: &fakeS3API{}}
|
||||
_, err := backend.Upload(context.Background(), filepath.Join(t.TempDir(), "missing.txt"), "key.txt", UploadOptions{})
|
||||
@@ -249,5 +340,6 @@ func TestNewS3BackendFromConfigFallsBackWhenCredentialEnvMissing(t *testing.T) {
|
||||
|
||||
func strPtr(v string) *string { return &v }
|
||||
func int64Ptr(v int64) *int64 { return &v }
|
||||
func boolPtr(v bool) *bool { return &v }
|
||||
|
||||
var _ s3API = (*fakeS3API)(nil)
|
||||
|
||||
391
internal/adapters/subprocess/diagnostics.go
Normal file
391
internal/adapters/subprocess/diagnostics.go
Normal file
@@ -0,0 +1,391 @@
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
const (
|
||||
// MaxStdoutDiagnosticBytes bounds persisted stdout from one external command.
|
||||
MaxStdoutDiagnosticBytes int64 = 8 * 1024 * 1024
|
||||
// MaxStderrDiagnosticBytes bounds persisted stderr from one external command.
|
||||
MaxStderrDiagnosticBytes int64 = 8 * 1024 * 1024
|
||||
diagnosticTailBytes = 2048
|
||||
)
|
||||
|
||||
var inheritedEnvironmentNames = map[string]struct{}{
|
||||
"COMSPEC": {},
|
||||
"HOME": {},
|
||||
"PATH": {},
|
||||
"SYSTEMROOT": {},
|
||||
"TMP": {},
|
||||
"TMPDIR": {},
|
||||
"TEMP": {},
|
||||
"WINDIR": {},
|
||||
// These test-only helper destinations let the adapter package tests exercise
|
||||
// real command invocation without widening the production environment.
|
||||
"AUDITA_HELPER_RECORD_PATH": {},
|
||||
"AUDITA_HELPER_MODE": {},
|
||||
"GO_WANT_AUDITA_HELPER": {},
|
||||
"GO_WANT_SCRIPTORIUM_HELPER": {},
|
||||
"GO_WANT_SERIATIM_HELPER": {},
|
||||
"GO_WANT_SUBPROCESS_HELPER": {},
|
||||
"NOTARIUS_CAPTURE_DIR": {},
|
||||
"NOTARIUS_RECEIPT_FIXTURE": {},
|
||||
"SCRIPTORIUM_HELPER_RECORD_PATH": {},
|
||||
"SCRIPTORIUM_HELPER_MODE": {},
|
||||
"SERIATIM_HELPER_RECORD_PATH": {},
|
||||
"SERIATIM_HELPER_MODE": {},
|
||||
}
|
||||
|
||||
var sensitiveEnvironmentNames = map[string]struct{}{
|
||||
"ANTHROPIC_API_KEY": {},
|
||||
"API_KEY": {},
|
||||
"AUDITA_LLM_API_KEY": {},
|
||||
"AWS_ACCESS_KEY_ID": {},
|
||||
"AWS_SECRET_ACCESS_KEY": {},
|
||||
"AWS_SESSION_TOKEN": {},
|
||||
"OPENAI_API_KEY": {},
|
||||
"OPENROUTER_API_KEY": {},
|
||||
}
|
||||
|
||||
type captureLimitError struct {
|
||||
stream string
|
||||
owner string
|
||||
limit int64
|
||||
}
|
||||
|
||||
func (e *captureLimitError) Error() string {
|
||||
return fmt.Sprintf("%s diagnostic capture for %s exceeded %d bytes", e.stream, e.owner, e.limit)
|
||||
}
|
||||
|
||||
type logWriters struct {
|
||||
files []*os.File
|
||||
Stdout io.Writer
|
||||
Stderr io.Writer
|
||||
|
||||
limits chan *captureLimitError
|
||||
mu sync.Mutex
|
||||
limit *captureLimitError
|
||||
stdout *diagnosticWriter
|
||||
stderr *diagnosticWriter
|
||||
}
|
||||
|
||||
type diagnosticWriter struct {
|
||||
logs *logWriters
|
||||
stream string
|
||||
owner string
|
||||
target io.Writer
|
||||
limit int64
|
||||
received int64
|
||||
persisted int64
|
||||
redactor streamRedactor
|
||||
tail []byte
|
||||
}
|
||||
|
||||
func openLogWriters(stdoutPath, stderrPath, owner string, sensitiveValues []string) (*logWriters, error) {
|
||||
logs := &logWriters{limits: make(chan *captureLimitError, 1)}
|
||||
cleanStdout := cleanLogPath(stdoutPath)
|
||||
cleanStderr := cleanLogPath(stderrPath)
|
||||
|
||||
stdoutFile, err := openDiagnosticFile(cleanStdout)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open stdout log: %w", err)
|
||||
}
|
||||
stderrFile := stdoutFile
|
||||
if cleanStdout != cleanStderr {
|
||||
stderrFile, err = openDiagnosticFile(cleanStderr)
|
||||
if err != nil {
|
||||
_ = stdoutFile.Close()
|
||||
return nil, fmt.Errorf("open stderr log: %w", err)
|
||||
}
|
||||
}
|
||||
if cleanStdout == cleanStderr {
|
||||
logs.files = []*os.File{stdoutFile}
|
||||
} else {
|
||||
logs.files = []*os.File{stdoutFile, stderrFile}
|
||||
}
|
||||
|
||||
logs.stdout = newDiagnosticWriter(logs, "stdout", owner, stdoutFile, MaxStdoutDiagnosticBytes, sensitiveValues)
|
||||
logs.stderr = newDiagnosticWriter(logs, "stderr", owner, stderrFile, MaxStderrDiagnosticBytes, sensitiveValues)
|
||||
logs.Stdout = logs.stdout
|
||||
logs.Stderr = logs.stderr
|
||||
return logs, nil
|
||||
}
|
||||
|
||||
func newDiagnosticWriter(logs *logWriters, stream, owner string, target io.Writer, limit int64, sensitiveValues []string) *diagnosticWriter {
|
||||
return &diagnosticWriter{
|
||||
logs: logs,
|
||||
stream: stream,
|
||||
owner: owner,
|
||||
target: target,
|
||||
limit: limit,
|
||||
redactor: newStreamRedactor(sensitiveValues),
|
||||
}
|
||||
}
|
||||
|
||||
func (w *diagnosticWriter) Write(data []byte) (int, error) {
|
||||
if w.received >= w.limit {
|
||||
return len(data), w.reachLimit()
|
||||
}
|
||||
accepted := data
|
||||
if remaining := w.limit - w.received; int64(len(accepted)) > remaining {
|
||||
accepted = accepted[:remaining]
|
||||
}
|
||||
w.received += int64(len(accepted))
|
||||
if err := w.writeRedacted(w.redactor.Write(accepted)); err != nil {
|
||||
return len(data), err
|
||||
}
|
||||
if len(accepted) != len(data) {
|
||||
return len(data), w.reachLimit()
|
||||
}
|
||||
return len(data), nil
|
||||
}
|
||||
|
||||
func (w *diagnosticWriter) Flush() error {
|
||||
return w.writeRedacted(w.redactor.Flush())
|
||||
}
|
||||
|
||||
func (w *diagnosticWriter) writeRedacted(data []byte) error {
|
||||
if len(data) == 0 {
|
||||
return nil
|
||||
}
|
||||
w.logs.mu.Lock()
|
||||
remaining := w.limit - w.persisted
|
||||
if remaining <= 0 {
|
||||
w.logs.mu.Unlock()
|
||||
return w.reachLimit()
|
||||
}
|
||||
toWrite := data
|
||||
exceeded := int64(len(data)) > remaining
|
||||
if exceeded {
|
||||
toWrite = toWrite[:remaining]
|
||||
}
|
||||
written, err := w.target.Write(toWrite)
|
||||
w.persisted += int64(written)
|
||||
w.retainTail(toWrite[:written])
|
||||
w.logs.mu.Unlock()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if exceeded {
|
||||
return w.reachLimit()
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (w *diagnosticWriter) Tail() string {
|
||||
w.logs.mu.Lock()
|
||||
defer w.logs.mu.Unlock()
|
||||
return strings.TrimSpace(string(w.tail))
|
||||
}
|
||||
|
||||
func (w *diagnosticWriter) retainTail(data []byte) {
|
||||
if len(data) >= diagnosticTailBytes {
|
||||
if cap(w.tail) < diagnosticTailBytes {
|
||||
w.tail = make([]byte, diagnosticTailBytes)
|
||||
} else {
|
||||
w.tail = w.tail[:diagnosticTailBytes]
|
||||
}
|
||||
copy(w.tail, data[len(data)-diagnosticTailBytes:])
|
||||
return
|
||||
}
|
||||
if cap(w.tail) < diagnosticTailBytes {
|
||||
retained := make([]byte, len(w.tail), diagnosticTailBytes)
|
||||
copy(retained, w.tail)
|
||||
w.tail = retained
|
||||
}
|
||||
if overflow := len(w.tail) + len(data) - diagnosticTailBytes; overflow > 0 {
|
||||
copy(w.tail, w.tail[overflow:])
|
||||
w.tail = w.tail[:len(w.tail)-overflow]
|
||||
}
|
||||
w.tail = append(w.tail, data...)
|
||||
}
|
||||
|
||||
func (w *diagnosticWriter) reachLimit() error {
|
||||
limit := &captureLimitError{stream: w.stream, owner: w.owner, limit: w.limit}
|
||||
w.logs.mu.Lock()
|
||||
if w.logs.limit == nil {
|
||||
w.logs.limit = limit
|
||||
w.logs.limits <- limit
|
||||
}
|
||||
w.logs.mu.Unlock()
|
||||
return limit
|
||||
}
|
||||
|
||||
func (l *logWriters) Limits() <-chan *captureLimitError { return l.limits }
|
||||
|
||||
func (l *logWriters) Limit() *captureLimitError {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
return l.limit
|
||||
}
|
||||
|
||||
func (l *logWriters) Flush() error {
|
||||
return joinErrors(l.stdout.Flush(), l.stderr.Flush())
|
||||
}
|
||||
|
||||
func (l *logWriters) Close() {
|
||||
_ = l.Flush()
|
||||
for _, file := range l.files {
|
||||
_ = file.Close()
|
||||
}
|
||||
}
|
||||
|
||||
func cleanLogPath(path string) string {
|
||||
trimmed := strings.TrimSpace(path)
|
||||
if trimmed == "" {
|
||||
return ""
|
||||
}
|
||||
return filepath.Clean(trimmed)
|
||||
}
|
||||
|
||||
func openDiagnosticFile(path string) (*os.File, error) {
|
||||
if path == "" {
|
||||
return os.OpenFile(os.DevNull, os.O_WRONLY, 0)
|
||||
}
|
||||
if err := fileops.EnsureWorkspaceDirectory(filepath.Dir(path)); err != nil {
|
||||
return nil, fmt.Errorf("create log directory for %q: %w", path, err)
|
||||
}
|
||||
file, err := fileops.OpenFileConfined(path, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, fileops.WorkspaceFileMode)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open log file %q: %w", path, err)
|
||||
}
|
||||
if err := file.Chmod(fileops.WorkspaceFileMode); err != nil {
|
||||
_ = file.Close()
|
||||
return nil, fmt.Errorf("set log file permissions %q: %w", path, err)
|
||||
}
|
||||
return file, nil
|
||||
}
|
||||
|
||||
func (r RunRequest) diagnosticOwner() string {
|
||||
if owner := strings.TrimSpace(r.DiagnosticOwner); owner != "" {
|
||||
return owner
|
||||
}
|
||||
return "subprocess"
|
||||
}
|
||||
|
||||
func buildChildEnvironment(base []string, overrides map[string]string) []string {
|
||||
values := make(map[string]string, len(inheritedEnvironmentNames)+len(overrides))
|
||||
for _, item := range base {
|
||||
name, value, ok := strings.Cut(item, "=")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
normalized := strings.ToUpper(name)
|
||||
if _, allowed := inheritedEnvironmentNames[normalized]; allowed {
|
||||
values[name] = value
|
||||
}
|
||||
}
|
||||
for name, value := range overrides {
|
||||
values[name] = value
|
||||
}
|
||||
names := make([]string, 0, len(values))
|
||||
for name := range values {
|
||||
names = append(names, name)
|
||||
}
|
||||
sort.Strings(names)
|
||||
out := make([]string, 0, len(names))
|
||||
for _, name := range names {
|
||||
out = append(out, name+"="+values[name])
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func sensitiveEnvironmentValues(environment []string, additionalNames []string) []string {
|
||||
names := make(map[string]struct{}, len(sensitiveEnvironmentNames)+len(additionalNames))
|
||||
for name := range sensitiveEnvironmentNames {
|
||||
names[name] = struct{}{}
|
||||
}
|
||||
for _, name := range additionalNames {
|
||||
if trimmed := strings.ToUpper(strings.TrimSpace(name)); trimmed != "" {
|
||||
names[trimmed] = struct{}{}
|
||||
}
|
||||
}
|
||||
values := make([]string, 0, len(names))
|
||||
for _, item := range environment {
|
||||
name, value, ok := strings.Cut(item, "=")
|
||||
if !ok || strings.TrimSpace(value) == "" {
|
||||
continue
|
||||
}
|
||||
if _, sensitive := names[strings.ToUpper(name)]; sensitive {
|
||||
values = append(values, value)
|
||||
}
|
||||
}
|
||||
return values
|
||||
}
|
||||
|
||||
type streamRedactor struct {
|
||||
values []string
|
||||
buffer []byte
|
||||
maxLen int
|
||||
}
|
||||
|
||||
func newStreamRedactor(values []string) streamRedactor {
|
||||
unique := make(map[string]struct{}, len(values))
|
||||
for _, value := range values {
|
||||
if value != "" {
|
||||
unique[value] = struct{}{}
|
||||
}
|
||||
}
|
||||
sorted := make([]string, 0, len(unique))
|
||||
for value := range unique {
|
||||
sorted = append(sorted, value)
|
||||
}
|
||||
sort.Slice(sorted, func(i, j int) bool { return len(sorted[i]) > len(sorted[j]) })
|
||||
maxLen := 1
|
||||
for _, value := range sorted {
|
||||
if len(value) > maxLen {
|
||||
maxLen = len(value)
|
||||
}
|
||||
}
|
||||
return streamRedactor{values: sorted, maxLen: maxLen}
|
||||
}
|
||||
|
||||
func (r *streamRedactor) Write(data []byte) []byte {
|
||||
r.buffer = append(r.buffer, data...)
|
||||
safeCut := len(r.buffer) - r.maxLen + 1
|
||||
if safeCut <= 0 {
|
||||
return nil
|
||||
}
|
||||
emitCut := safeCut
|
||||
for _, value := range r.values {
|
||||
start := 0
|
||||
for {
|
||||
index := bytes.Index(r.buffer[start:], []byte(value))
|
||||
if index < 0 {
|
||||
break
|
||||
}
|
||||
index += start
|
||||
if index+len(value) > safeCut && index < emitCut {
|
||||
emitCut = index
|
||||
}
|
||||
start = index + 1
|
||||
}
|
||||
}
|
||||
output := redactBytes(r.buffer[:emitCut], r.values)
|
||||
r.buffer = append(r.buffer[:0], r.buffer[emitCut:]...)
|
||||
return output
|
||||
}
|
||||
|
||||
func (r *streamRedactor) Flush() []byte {
|
||||
output := redactBytes(r.buffer, r.values)
|
||||
r.buffer = nil
|
||||
return output
|
||||
}
|
||||
|
||||
func redactBytes(data []byte, values []string) []byte {
|
||||
out := append([]byte(nil), data...)
|
||||
for _, value := range values {
|
||||
out = bytes.ReplaceAll(out, []byte(value), []byte("<redacted>"))
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -1,3 +1,2 @@
|
||||
// Package subprocess provides reusable process execution and generated-config helpers.
|
||||
package subprocess
|
||||
|
||||
|
||||
66
internal/adapters/subprocess/process_tree.go
Normal file
66
internal/adapters/subprocess/process_tree.go
Normal file
@@ -0,0 +1,66 @@
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"time"
|
||||
)
|
||||
|
||||
const (
|
||||
gracefulTerminationWait = 2 * time.Second
|
||||
forcefulTerminationWait = 2 * time.Second
|
||||
)
|
||||
|
||||
// ownedProcessTree owns every process started by a command invocation.
|
||||
// Implementations must tolerate a leader that has already exited.
|
||||
type ownedProcessTree interface {
|
||||
Start(*exec.Cmd) error
|
||||
TerminateGracefully() error
|
||||
TerminateForcefully() error
|
||||
Dispose() error
|
||||
}
|
||||
|
||||
func waitForOwnedCommand(ctx context.Context, tree ownedProcessTree, waitCh <-chan error, captureLimits <-chan *captureLimitError) (waitErr, ctxErr error, captureLimit *captureLimitError, cleanupErr error) {
|
||||
select {
|
||||
case waitErr = <-waitCh:
|
||||
return waitErr, nil, nil, nil
|
||||
case <-ctx.Done():
|
||||
ctxErr = ctx.Err()
|
||||
case captureLimit = <-captureLimits:
|
||||
}
|
||||
|
||||
cleanupErr = tree.TerminateGracefully()
|
||||
gracefulTimer := time.NewTimer(gracefulTerminationWait)
|
||||
defer gracefulTimer.Stop()
|
||||
|
||||
select {
|
||||
case waitErr = <-waitCh:
|
||||
// The leader may exit before descendants finish graceful shutdown.
|
||||
cleanupErr = joinErrors(cleanupErr, tree.TerminateForcefully())
|
||||
return waitErr, ctxErr, captureLimit, cleanupErr
|
||||
case <-gracefulTimer.C:
|
||||
}
|
||||
|
||||
cleanupErr = joinErrors(cleanupErr, tree.TerminateForcefully())
|
||||
forcefulTimer := time.NewTimer(forcefulTerminationWait)
|
||||
defer forcefulTimer.Stop()
|
||||
|
||||
select {
|
||||
case waitErr = <-waitCh:
|
||||
return waitErr, ctxErr, captureLimit, cleanupErr
|
||||
case <-forcefulTimer.C:
|
||||
return nil, ctxErr, captureLimit, joinErrors(cleanupErr, fmt.Errorf("owned subprocess did not reap within %s after forceful termination", forcefulTerminationWait))
|
||||
}
|
||||
}
|
||||
|
||||
func joinErrors(errs ...error) error {
|
||||
filtered := make([]error, 0, len(errs))
|
||||
for _, err := range errs {
|
||||
if err != nil {
|
||||
filtered = append(filtered, err)
|
||||
}
|
||||
}
|
||||
return errors.Join(filtered...)
|
||||
}
|
||||
227
internal/adapters/subprocess/process_tree_supported_test.go
Normal file
227
internal/adapters/subprocess/process_tree_supported_test.go
Normal file
@@ -0,0 +1,227 @@
|
||||
//go:build linux || darwin || windows
|
||||
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestRunCancellationTerminatesProcessTree(t *testing.T) {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
resultCh := make(chan runOutcome, 1)
|
||||
req, sentinelPath := processTreeRequest(t)
|
||||
go func() {
|
||||
result, err := Run(ctx, req)
|
||||
resultCh <- runOutcome{result: result, err: err}
|
||||
}()
|
||||
|
||||
awaitHelperReady(t, req.EnvOverrides["SUBPROCESS_HELPER_READY_PATH"])
|
||||
cancel()
|
||||
|
||||
outcome := awaitRunOutcome(t, resultCh)
|
||||
if !outcome.result.Canceled {
|
||||
t.Fatalf("Canceled = %v, want true", outcome.result.Canceled)
|
||||
}
|
||||
if !errors.Is(outcome.err, context.Canceled) {
|
||||
t.Fatalf("error = %v, want context cancellation", outcome.err)
|
||||
}
|
||||
assertDescendantDidNotSurvive(t, sentinelPath)
|
||||
}
|
||||
|
||||
func TestRunTimeoutTerminatesProcessTree(t *testing.T) {
|
||||
req, sentinelPath := processTreeRequest(t)
|
||||
req.Timeout = 100 * time.Millisecond
|
||||
|
||||
result, err := Run(context.Background(), req)
|
||||
if !result.TimedOut {
|
||||
t.Fatalf("TimedOut = %v, want true", result.TimedOut)
|
||||
}
|
||||
if !errors.Is(err, context.DeadlineExceeded) {
|
||||
t.Fatalf("error = %v, want context deadline exceeded", err)
|
||||
}
|
||||
assertDescendantDidNotSurvive(t, sentinelPath)
|
||||
}
|
||||
|
||||
func TestRunCaptureLimitTerminatesProcessTree(t *testing.T) {
|
||||
req, sentinelPath := processTreeRequest(t)
|
||||
req.Args[len(req.Args)-1] = "tree-spam"
|
||||
|
||||
result, err := Run(context.Background(), req)
|
||||
if err == nil {
|
||||
t.Fatal("Run() error = nil, want capture-limit error")
|
||||
}
|
||||
if result.ExitCode == 0 {
|
||||
t.Fatalf("ExitCode = %d, want terminated process", result.ExitCode)
|
||||
}
|
||||
if !strings.Contains(err.Error(), "stdout diagnostic capture for subprocess exceeded") {
|
||||
t.Fatalf("error = %q, want stdout capture-limit context", err)
|
||||
}
|
||||
info, statErr := os.Stat(req.StdoutLogPath)
|
||||
if statErr != nil {
|
||||
t.Fatalf("stat stdout diagnostic: %v", statErr)
|
||||
}
|
||||
if info.Size() != MaxStdoutDiagnosticBytes {
|
||||
t.Fatalf("stdout diagnostic size = %d, want %d", info.Size(), MaxStdoutDiagnosticBytes)
|
||||
}
|
||||
assertDescendantDidNotSurvive(t, sentinelPath)
|
||||
}
|
||||
|
||||
func TestRunDisposesDescendantsAfterLeaderExit(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
mode string
|
||||
wantExitCode int
|
||||
wantWaitDelay bool
|
||||
ignoreTerm bool
|
||||
}{
|
||||
{name: "success retaining streams", mode: "leader-exit-retained", wantExitCode: 0, wantWaitDelay: true},
|
||||
{name: "success redirecting streams", mode: "leader-exit-redirected", wantExitCode: 0},
|
||||
{name: "failed leader", mode: "leader-fail-redirected", wantExitCode: 9, ignoreTerm: true},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
req, sentinelPath, releasePath := leaderExitRequest(t, tt.mode)
|
||||
if tt.ignoreTerm {
|
||||
req.EnvOverrides["SUBPROCESS_HELPER_IGNORE_TERM"] = "1"
|
||||
}
|
||||
outcomes := make(chan runOutcome, 1)
|
||||
go func() {
|
||||
result, err := Run(context.Background(), req)
|
||||
outcomes <- runOutcome{result: result, err: err}
|
||||
}()
|
||||
|
||||
var outcome runOutcome
|
||||
select {
|
||||
case outcome = <-outcomes:
|
||||
case <-time.After(6 * time.Second):
|
||||
t.Fatal("Run() did not complete bounded owned-tree disposal")
|
||||
}
|
||||
if outcome.result.ExitCode != tt.wantExitCode {
|
||||
t.Fatalf("ExitCode = %d, want %d", outcome.result.ExitCode, tt.wantExitCode)
|
||||
}
|
||||
if tt.wantWaitDelay {
|
||||
if !errors.Is(outcome.err, exec.ErrWaitDelay) {
|
||||
t.Fatalf("error = %v, want exec.ErrWaitDelay", outcome.err)
|
||||
}
|
||||
} else if tt.wantExitCode == 0 && outcome.err != nil {
|
||||
t.Fatalf("Run() error = %v, want nil", outcome.err)
|
||||
} else if tt.wantExitCode != 0 {
|
||||
var exitErr *exec.ExitError
|
||||
if !errors.As(outcome.err, &exitErr) || exitErr.ExitCode() != tt.wantExitCode {
|
||||
t.Fatalf("error = %v, want exit code %d", outcome.err, tt.wantExitCode)
|
||||
}
|
||||
}
|
||||
if _, err := os.Stat(req.EnvOverrides["SUBPROCESS_HELPER_READY_PATH"]); err != nil {
|
||||
t.Fatalf("descendant readiness file: %v", err)
|
||||
}
|
||||
if err := os.WriteFile(releasePath, []byte("release"), 0o600); err != nil {
|
||||
t.Fatalf("WriteFile(release) error = %v", err)
|
||||
}
|
||||
assertDescendantDidNotSurvive(t, sentinelPath)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
type runOutcome struct {
|
||||
result RunResult
|
||||
err error
|
||||
}
|
||||
|
||||
func processTreeRequest(t *testing.T) (RunRequest, string) {
|
||||
t.Helper()
|
||||
|
||||
executable, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
dir := t.TempDir()
|
||||
readyPath := filepath.Join(dir, "ready")
|
||||
sentinelPath := filepath.Join(dir, "descendant-survived")
|
||||
return RunRequest{
|
||||
Executable: executable,
|
||||
Args: []string{"-test.run=^TestSubprocessHelper$", "--", "tree"},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"SUBPROCESS_HELPER_READY_PATH": readyPath,
|
||||
"SUBPROCESS_HELPER_SENTINEL_PATH": sentinelPath,
|
||||
},
|
||||
StdoutLogPath: filepath.Join(dir, "stdout.log"),
|
||||
StderrLogPath: filepath.Join(dir, "stderr.log"),
|
||||
}, sentinelPath
|
||||
}
|
||||
|
||||
func leaderExitRequest(t *testing.T, mode string) (RunRequest, string, string) {
|
||||
t.Helper()
|
||||
|
||||
executable, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
dir := t.TempDir()
|
||||
readyPath := filepath.Join(dir, "ready")
|
||||
releasePath := filepath.Join(dir, "release")
|
||||
sentinelPath := filepath.Join(dir, "descendant-survived")
|
||||
return RunRequest{
|
||||
Executable: executable,
|
||||
Args: []string{"-test.run=^TestSubprocessHelper$", "--", mode},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"SUBPROCESS_HELPER_READY_PATH": readyPath,
|
||||
"SUBPROCESS_HELPER_RELEASE_PATH": releasePath,
|
||||
"SUBPROCESS_HELPER_SENTINEL_PATH": sentinelPath,
|
||||
},
|
||||
StdoutLogPath: filepath.Join(dir, "stdout.log"),
|
||||
StderrLogPath: filepath.Join(dir, "stderr.log"),
|
||||
}, sentinelPath, releasePath
|
||||
}
|
||||
|
||||
func awaitHelperReady(t *testing.T, readyPath string) {
|
||||
t.Helper()
|
||||
|
||||
deadline := time.Now().Add(2 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
if _, err := os.Stat(readyPath); err == nil {
|
||||
return
|
||||
} else if !errors.Is(err, os.ErrNotExist) {
|
||||
t.Fatalf("stat helper readiness: %v", err)
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
t.Fatal("helper did not start its descendant")
|
||||
}
|
||||
|
||||
func awaitRunOutcome(t *testing.T, outcomes <-chan runOutcome) runOutcome {
|
||||
t.Helper()
|
||||
|
||||
select {
|
||||
case outcome := <-outcomes:
|
||||
if outcome.err == nil {
|
||||
t.Fatal("Run() error = nil, want cancellation error")
|
||||
}
|
||||
return outcome
|
||||
case <-time.After(3 * time.Second):
|
||||
t.Fatal("Run() did not return after cancellation")
|
||||
return runOutcome{}
|
||||
}
|
||||
}
|
||||
|
||||
func assertDescendantDidNotSurvive(t *testing.T, sentinelPath string) {
|
||||
t.Helper()
|
||||
|
||||
time.Sleep(700 * time.Millisecond)
|
||||
if _, err := os.Stat(sentinelPath); err == nil {
|
||||
t.Fatal("descendant survived cancellation and wrote its sentinel")
|
||||
} else if !errors.Is(err, os.ErrNotExist) {
|
||||
t.Fatalf("stat descendant sentinel: %v", err)
|
||||
}
|
||||
}
|
||||
104
internal/adapters/subprocess/process_tree_unix.go
Normal file
104
internal/adapters/subprocess/process_tree_unix.go
Normal file
@@ -0,0 +1,104 @@
|
||||
//go:build linux || darwin
|
||||
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
const processGroupPollInterval = 10 * time.Millisecond
|
||||
|
||||
type unixProcessTree struct {
|
||||
processGroupID int
|
||||
}
|
||||
|
||||
func newOwnedProcessTree() (ownedProcessTree, error) {
|
||||
return &unixProcessTree{}, nil
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) Start(cmd *exec.Cmd) error {
|
||||
cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true}
|
||||
if err := cmd.Start(); err != nil {
|
||||
return err
|
||||
}
|
||||
tree.processGroupID = cmd.Process.Pid
|
||||
return nil
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) TerminateGracefully() error {
|
||||
return tree.signal(syscall.SIGTERM)
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) TerminateForcefully() error {
|
||||
return tree.signal(syscall.SIGKILL)
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) Dispose() error {
|
||||
hasMembers, err := tree.hasMembers()
|
||||
if err != nil || !hasMembers {
|
||||
return err
|
||||
}
|
||||
|
||||
cleanupErr := tree.TerminateGracefully()
|
||||
empty, waitErr := tree.waitUntilEmpty(gracefulTerminationWait)
|
||||
cleanupErr = joinErrors(cleanupErr, waitErr)
|
||||
if empty {
|
||||
return cleanupErr
|
||||
}
|
||||
|
||||
cleanupErr = joinErrors(cleanupErr, tree.TerminateForcefully())
|
||||
empty, waitErr = tree.waitUntilEmpty(forcefulTerminationWait)
|
||||
cleanupErr = joinErrors(cleanupErr, waitErr)
|
||||
if !empty {
|
||||
cleanupErr = joinErrors(cleanupErr, fmt.Errorf("owned subprocess group did not exit within %s after forceful termination", forcefulTerminationWait))
|
||||
}
|
||||
return cleanupErr
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) signal(signal syscall.Signal) error {
|
||||
if tree.processGroupID <= 0 {
|
||||
return nil
|
||||
}
|
||||
err := syscall.Kill(-tree.processGroupID, signal)
|
||||
if errors.Is(err, syscall.ESRCH) || errors.Is(err, os.ErrProcessDone) {
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) hasMembers() (bool, error) {
|
||||
if tree.processGroupID <= 0 {
|
||||
return false, nil
|
||||
}
|
||||
err := syscall.Kill(-tree.processGroupID, 0)
|
||||
if err == nil || errors.Is(err, syscall.EPERM) {
|
||||
return true, nil
|
||||
}
|
||||
if errors.Is(err, syscall.ESRCH) || errors.Is(err, os.ErrProcessDone) {
|
||||
return false, nil
|
||||
}
|
||||
return false, fmt.Errorf("inspect owned subprocess group: %w", err)
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) waitUntilEmpty(timeout time.Duration) (bool, error) {
|
||||
deadline := time.Now().Add(timeout)
|
||||
for {
|
||||
hasMembers, err := tree.hasMembers()
|
||||
if err != nil || !hasMembers {
|
||||
return !hasMembers, err
|
||||
}
|
||||
remaining := time.Until(deadline)
|
||||
if remaining <= 0 {
|
||||
return false, nil
|
||||
}
|
||||
if remaining > processGroupPollInterval {
|
||||
remaining = processGroupPollInterval
|
||||
}
|
||||
time.Sleep(remaining)
|
||||
}
|
||||
}
|
||||
12
internal/adapters/subprocess/process_tree_unsupported.go
Normal file
12
internal/adapters/subprocess/process_tree_unsupported.go
Normal file
@@ -0,0 +1,12 @@
|
||||
//go:build !linux && !darwin && !windows
|
||||
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"runtime"
|
||||
)
|
||||
|
||||
func newOwnedProcessTree() (ownedProcessTree, error) {
|
||||
return nil, fmt.Errorf("owned subprocess trees are unsupported on %s", runtime.GOOS)
|
||||
}
|
||||
122
internal/adapters/subprocess/process_tree_windows.go
Normal file
122
internal/adapters/subprocess/process_tree_windows.go
Normal file
@@ -0,0 +1,122 @@
|
||||
//go:build windows
|
||||
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
|
||||
"golang.org/x/sys/windows"
|
||||
)
|
||||
|
||||
type windowsProcessTree struct {
|
||||
job windows.Handle
|
||||
}
|
||||
|
||||
func newOwnedProcessTree() (ownedProcessTree, error) {
|
||||
return &windowsProcessTree{}, nil
|
||||
}
|
||||
|
||||
func (tree *windowsProcessTree) Start(cmd *exec.Cmd) error {
|
||||
job, err := windows.CreateJobObject(nil, nil)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create job object: %w", err)
|
||||
}
|
||||
|
||||
limits := windows.JOBOBJECT_EXTENDED_LIMIT_INFORMATION{}
|
||||
limits.BasicLimitInformation.LimitFlags = windows.JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE
|
||||
if _, err := windows.SetInformationJobObject(job, windows.JobObjectExtendedLimitInformation, uintptr(unsafe.Pointer(&limits)), uint32(unsafe.Sizeof(limits))); err != nil {
|
||||
_ = windows.CloseHandle(job)
|
||||
return fmt.Errorf("configure job object: %w", err)
|
||||
}
|
||||
|
||||
cmd.SysProcAttr = &syscall.SysProcAttr{CreationFlags: windows.CREATE_SUSPENDED}
|
||||
if err := cmd.Start(); err != nil {
|
||||
_ = windows.CloseHandle(job)
|
||||
return err
|
||||
}
|
||||
|
||||
process, err := windows.OpenProcess(windows.PROCESS_SET_QUOTA|windows.PROCESS_TERMINATE, false, uint32(cmd.Process.Pid))
|
||||
if err == nil {
|
||||
err = windows.AssignProcessToJobObject(job, process)
|
||||
_ = windows.CloseHandle(process)
|
||||
}
|
||||
if err == nil {
|
||||
err = resumeInitialThread(uint32(cmd.Process.Pid))
|
||||
}
|
||||
if err != nil {
|
||||
killErr := cmd.Process.Kill()
|
||||
waitErr := cmd.Wait()
|
||||
_ = windows.CloseHandle(job)
|
||||
return joinErrors(fmt.Errorf("assign process to job object: %w", err), killErr, waitErr)
|
||||
}
|
||||
|
||||
tree.job = job
|
||||
return nil
|
||||
}
|
||||
|
||||
func resumeInitialThread(processID uint32) error {
|
||||
snapshot, err := windows.CreateToolhelp32Snapshot(windows.TH32CS_SNAPTHREAD, 0)
|
||||
if err != nil {
|
||||
return fmt.Errorf("snapshot initial thread: %w", err)
|
||||
}
|
||||
defer func() { _ = windows.CloseHandle(snapshot) }()
|
||||
|
||||
entry := windows.ThreadEntry32{Size: uint32(unsafe.Sizeof(windows.ThreadEntry32{}))}
|
||||
if err := windows.Thread32First(snapshot, &entry); err != nil {
|
||||
return fmt.Errorf("find initial thread: %w", err)
|
||||
}
|
||||
for {
|
||||
if entry.OwnerProcessID != processID {
|
||||
// Keep enumerating until the suspended process's only initial thread
|
||||
// is found.
|
||||
} else {
|
||||
thread, openErr := windows.OpenThread(windows.THREAD_SUSPEND_RESUME, false, entry.ThreadID)
|
||||
if openErr != nil {
|
||||
return fmt.Errorf("open initial thread: %w", openErr)
|
||||
}
|
||||
defer func() { _ = windows.CloseHandle(thread) }()
|
||||
if _, resumeErr := windows.ResumeThread(thread); resumeErr != nil {
|
||||
return fmt.Errorf("resume initial thread: %w", resumeErr)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
if err := windows.Thread32Next(snapshot, &entry); err != nil {
|
||||
if errors.Is(err, windows.ERROR_NO_MORE_FILES) {
|
||||
break
|
||||
}
|
||||
return fmt.Errorf("find initial thread: %w", err)
|
||||
}
|
||||
}
|
||||
return fmt.Errorf("find initial thread: no thread found for process %d", processID)
|
||||
}
|
||||
|
||||
func (tree *windowsProcessTree) TerminateGracefully() error {
|
||||
// Windows jobs have no portable graceful signal. Terminating the owned job
|
||||
// is the safe fallback and prevents a descendant from escaping cleanup.
|
||||
return tree.terminate()
|
||||
}
|
||||
|
||||
func (tree *windowsProcessTree) TerminateForcefully() error {
|
||||
return tree.terminate()
|
||||
}
|
||||
|
||||
func (tree *windowsProcessTree) Dispose() error {
|
||||
if tree.job == 0 {
|
||||
return nil
|
||||
}
|
||||
err := windows.CloseHandle(tree.job)
|
||||
tree.job = 0
|
||||
return err
|
||||
}
|
||||
|
||||
func (tree *windowsProcessTree) terminate() error {
|
||||
if tree.job == 0 {
|
||||
return nil
|
||||
}
|
||||
return windows.TerminateJobObject(tree.job, 1)
|
||||
}
|
||||
@@ -2,16 +2,13 @@ package subprocess
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
"gopkg.in/yaml.v3"
|
||||
)
|
||||
|
||||
@@ -21,6 +18,8 @@ type RunRequest struct {
|
||||
Args []string
|
||||
WorkingDir string
|
||||
EnvOverrides map[string]string
|
||||
SensitiveEnvNames []string
|
||||
DiagnosticOwner string
|
||||
Timeout time.Duration
|
||||
StdoutLogPath string
|
||||
StderrLogPath string
|
||||
@@ -54,17 +53,26 @@ func Run(ctx context.Context, req RunRequest) (RunResult, error) {
|
||||
}
|
||||
defer cancel()
|
||||
|
||||
logs, err := openLogWriters(req.StdoutLogPath, req.StderrLogPath)
|
||||
childEnv := buildChildEnvironment(os.Environ(), req.EnvOverrides)
|
||||
logs, err := openLogWriters(req.StdoutLogPath, req.StderrLogPath, req.diagnosticOwner(), sensitiveEnvironmentValues(childEnv, req.SensitiveEnvNames))
|
||||
if err != nil {
|
||||
return RunResult{}, err
|
||||
}
|
||||
defer logs.Close()
|
||||
|
||||
cmd := exec.CommandContext(runCtx, req.Executable, req.Args...)
|
||||
tree, err := newOwnedProcessTree()
|
||||
if err != nil {
|
||||
return RunResult{}, fmt.Errorf("prepare owned subprocess tree: %w", err)
|
||||
}
|
||||
|
||||
cmd := exec.Command(req.Executable, req.Args...)
|
||||
cmd.Dir = req.WorkingDir
|
||||
cmd.Env = mergeEnv(os.Environ(), req.EnvOverrides)
|
||||
cmd.Env = childEnv
|
||||
cmd.Stdout = logs.Stdout
|
||||
cmd.Stderr = logs.Stderr
|
||||
// Streaming capture uses pipes. Bound their lifetime when a leader exits
|
||||
// while a descendant still holds a stream descriptor.
|
||||
cmd.WaitDelay = forcefulTerminationWait
|
||||
|
||||
started := time.Now().UTC()
|
||||
result := RunResult{
|
||||
@@ -74,45 +82,67 @@ func Run(ctx context.Context, req RunRequest) (RunResult, error) {
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
}
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
if err := runCtx.Err(); err != nil {
|
||||
result.CompletedAt = time.Now().UTC()
|
||||
result.Duration = result.CompletedAt.Sub(result.StartedAt)
|
||||
return result, fmt.Errorf("command was not started: %w", err)
|
||||
}
|
||||
|
||||
if err := tree.Start(cmd); err != nil {
|
||||
result.CompletedAt = time.Now().UTC()
|
||||
result.Duration = result.CompletedAt.Sub(result.StartedAt)
|
||||
return result, fmt.Errorf("start command %q with args %v: %w", req.Executable, req.Args, err)
|
||||
}
|
||||
|
||||
waitErr := cmd.Wait()
|
||||
waitCh := make(chan error, 1)
|
||||
go func() { waitCh <- cmd.Wait() }()
|
||||
|
||||
waitErr, ctxErr, captureLimit, cleanupErr := waitForOwnedCommand(runCtx, tree, waitCh, logs.Limits())
|
||||
cleanupErr = joinErrors(cleanupErr, tree.Dispose())
|
||||
cleanupErr = joinErrors(cleanupErr, logs.Flush())
|
||||
if captureLimit == nil {
|
||||
captureLimit = logs.Limit()
|
||||
}
|
||||
result.CompletedAt = time.Now().UTC()
|
||||
result.Duration = result.CompletedAt.Sub(result.StartedAt)
|
||||
if cmd.ProcessState != nil {
|
||||
result.ExitCode = cmd.ProcessState.ExitCode()
|
||||
}
|
||||
|
||||
ctxErr := runCtx.Err()
|
||||
if errors.Is(ctxErr, context.DeadlineExceeded) {
|
||||
if ctxErr == context.DeadlineExceeded {
|
||||
result.TimedOut = true
|
||||
}
|
||||
if errors.Is(ctxErr, context.Canceled) && !result.TimedOut {
|
||||
if ctxErr == context.Canceled && !result.TimedOut {
|
||||
result.Canceled = true
|
||||
}
|
||||
|
||||
if waitErr == nil {
|
||||
if waitErr == nil && ctxErr == nil && cleanupErr == nil {
|
||||
return result, nil
|
||||
}
|
||||
|
||||
stderrTail := readRedactedTail(req.StderrLogPath, req.EnvOverrides, 2048)
|
||||
stderrTail := logs.stderr.Tail()
|
||||
diagnostics := buildDiagnostics(req, result, stderrTail)
|
||||
|
||||
if captureLimit != nil {
|
||||
if cause := joinErrors(waitErr, cleanupErr); cause != nil {
|
||||
return result, fmt.Errorf("%w (%s): %w", captureLimit, diagnostics, cause)
|
||||
}
|
||||
return result, fmt.Errorf("%w (%s)", captureLimit, diagnostics)
|
||||
}
|
||||
if result.TimedOut {
|
||||
return result, fmt.Errorf("command timed out after %s (%s)", req.Timeout, diagnostics)
|
||||
return result, fmt.Errorf("command timed out after %s (%s): %w", req.Timeout, diagnostics, joinErrors(ctxErr, waitErr, cleanupErr))
|
||||
}
|
||||
if result.Canceled {
|
||||
return result, fmt.Errorf("command canceled (%s)", diagnostics)
|
||||
return result, fmt.Errorf("command canceled (%s): %w", diagnostics, joinErrors(ctxErr, waitErr, cleanupErr))
|
||||
}
|
||||
if exitErr, ok := waitErr.(*exec.ExitError); ok {
|
||||
return result, fmt.Errorf("command failed with exit code %d (%s): %w", exitErr.ExitCode(), diagnostics, waitErr)
|
||||
return result, fmt.Errorf("command failed with exit code %d (%s): %w", exitErr.ExitCode(), diagnostics, joinErrors(waitErr, cleanupErr))
|
||||
}
|
||||
if cleanupErr != nil {
|
||||
return result, fmt.Errorf("command cleanup failed (%s): %w", diagnostics, joinErrors(waitErr, cleanupErr))
|
||||
}
|
||||
|
||||
return result, fmt.Errorf("command failed to run (%s): %w", diagnostics, waitErr)
|
||||
return result, fmt.Errorf("command failed to run (%s): %w", diagnostics, joinErrors(waitErr, cleanupErr))
|
||||
}
|
||||
|
||||
// WriteYAMLAtomic marshals value as YAML and atomically writes it to path.
|
||||
@@ -127,141 +157,17 @@ func WriteYAMLAtomic(path string, value any, perm os.FileMode) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// WriteFileAtomic writes bytes via same-directory temp file + atomic rename.
|
||||
// WriteFileAtomic writes bytes through the shared durable replacement primitive.
|
||||
func WriteFileAtomic(path string, data []byte, perm os.FileMode) error {
|
||||
if strings.TrimSpace(path) == "" {
|
||||
return fmt.Errorf("write file: path is required")
|
||||
}
|
||||
|
||||
dir := filepath.Dir(path)
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
return fmt.Errorf("create parent directory %q: %w", dir, err)
|
||||
if err := fileops.WriteFileAtomic(path, data, perm); err != nil {
|
||||
return fmt.Errorf("write file %q: %w", path, err)
|
||||
}
|
||||
|
||||
base := filepath.Base(path)
|
||||
tmp, err := os.CreateTemp(dir, "."+base+".tmp-*")
|
||||
if err != nil {
|
||||
return fmt.Errorf("create temp file: %w", err)
|
||||
}
|
||||
tmpPath := tmp.Name()
|
||||
removeTmp := true
|
||||
defer func() {
|
||||
if removeTmp {
|
||||
_ = os.Remove(tmpPath)
|
||||
}
|
||||
}()
|
||||
|
||||
if _, err := tmp.Write(data); err != nil {
|
||||
_ = tmp.Close()
|
||||
return fmt.Errorf("write temp file: %w", err)
|
||||
}
|
||||
if err := tmp.Sync(); err != nil {
|
||||
_ = tmp.Close()
|
||||
return fmt.Errorf("sync temp file: %w", err)
|
||||
}
|
||||
if err := tmp.Close(); err != nil {
|
||||
return fmt.Errorf("close temp file: %w", err)
|
||||
}
|
||||
if err := os.Chmod(tmpPath, perm); err != nil {
|
||||
return fmt.Errorf("chmod temp file: %w", err)
|
||||
}
|
||||
if err := os.Rename(tmpPath, path); err != nil {
|
||||
return fmt.Errorf("rename temp file: %w", err)
|
||||
}
|
||||
removeTmp = false
|
||||
return nil
|
||||
}
|
||||
|
||||
type logWriters struct {
|
||||
files []*os.File
|
||||
Stdout io.Writer
|
||||
Stderr io.Writer
|
||||
}
|
||||
|
||||
func (l *logWriters) Close() {
|
||||
for _, f := range l.files {
|
||||
_ = f.Close()
|
||||
}
|
||||
}
|
||||
|
||||
func openLogWriters(stdoutPath, stderrPath string) (*logWriters, error) {
|
||||
cleanStdout := cleanLogPath(stdoutPath)
|
||||
cleanStderr := cleanLogPath(stderrPath)
|
||||
|
||||
// Keep stdout/stderr on the same file descriptor when both paths target
|
||||
// the same file to avoid descriptor aliasing surprises across runtimes.
|
||||
if cleanStdout != "" && cleanStdout == cleanStderr {
|
||||
f, err := openLogFile(cleanStdout)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open shared stdout/stderr log %q: %w", cleanStdout, err)
|
||||
}
|
||||
return &logWriters{
|
||||
files: []*os.File{f},
|
||||
Stdout: f,
|
||||
Stderr: f,
|
||||
}, nil
|
||||
}
|
||||
|
||||
stdoutFile, stdoutWriter, err := logWriter(cleanStdout)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open stdout log: %w", err)
|
||||
}
|
||||
stderrFile, stderrWriter, err := logWriter(cleanStderr)
|
||||
if err != nil {
|
||||
closeFile(stdoutFile)
|
||||
return nil, fmt.Errorf("open stderr log: %w", err)
|
||||
}
|
||||
|
||||
files := make([]*os.File, 0, 2)
|
||||
if stdoutFile != nil {
|
||||
files = append(files, stdoutFile)
|
||||
}
|
||||
if stderrFile != nil {
|
||||
files = append(files, stderrFile)
|
||||
}
|
||||
return &logWriters{
|
||||
files: files,
|
||||
Stdout: stdoutWriter,
|
||||
Stderr: stderrWriter,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func cleanLogPath(path string) string {
|
||||
trimmed := strings.TrimSpace(path)
|
||||
if trimmed == "" {
|
||||
return ""
|
||||
}
|
||||
return filepath.Clean(trimmed)
|
||||
}
|
||||
|
||||
func logWriter(path string) (*os.File, io.Writer, error) {
|
||||
if strings.TrimSpace(path) == "" {
|
||||
return nil, io.Discard, nil
|
||||
}
|
||||
f, err := openLogFile(path)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
return f, f, nil
|
||||
}
|
||||
|
||||
func openLogFile(path string) (*os.File, error) {
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
return nil, fmt.Errorf("create log directory for %q: %w", path, err)
|
||||
}
|
||||
f, err := os.Create(path)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open log file %q: %w", path, err)
|
||||
}
|
||||
return f, nil
|
||||
}
|
||||
|
||||
func closeFile(f *os.File) {
|
||||
if f != nil {
|
||||
_ = f.Close()
|
||||
}
|
||||
}
|
||||
|
||||
func buildDiagnostics(req RunRequest, result RunResult, stderrTail string) string {
|
||||
details := fmt.Sprintf(
|
||||
"executable=%q args=%v cwd=%q timeout=%s exit_code=%d timed_out=%t canceled=%t stdout_log=%q stderr_log=%q",
|
||||
@@ -298,87 +204,3 @@ func fdDiagnosticsHint(exitCode int, stderrTail string) string {
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func readRedactedTail(path string, envOverrides map[string]string, maxBytes int64) string {
|
||||
if strings.TrimSpace(path) == "" || maxBytes <= 0 {
|
||||
return ""
|
||||
}
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
info, err := f.Stat()
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
size := info.Size()
|
||||
start := int64(0)
|
||||
if size > maxBytes {
|
||||
start = size - maxBytes
|
||||
}
|
||||
if _, err := f.Seek(start, io.SeekStart); err != nil {
|
||||
return ""
|
||||
}
|
||||
data, err := io.ReadAll(f)
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
tail := strings.TrimSpace(string(data))
|
||||
if tail == "" {
|
||||
return ""
|
||||
}
|
||||
return redactSensitiveTail(tail, envOverrides)
|
||||
}
|
||||
|
||||
func redactSensitiveTail(tail string, envOverrides map[string]string) string {
|
||||
out := tail
|
||||
for k, v := range envOverrides {
|
||||
if strings.TrimSpace(v) == "" {
|
||||
continue
|
||||
}
|
||||
if looksSensitiveEnvKey(k) {
|
||||
out = strings.ReplaceAll(out, v, "<redacted>")
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func looksSensitiveEnvKey(key string) bool {
|
||||
k := strings.ToUpper(strings.TrimSpace(key))
|
||||
return strings.Contains(k, "KEY") ||
|
||||
strings.Contains(k, "TOKEN") ||
|
||||
strings.Contains(k, "SECRET") ||
|
||||
strings.Contains(k, "PASSWORD")
|
||||
}
|
||||
|
||||
func mergeEnv(base []string, overrides map[string]string) []string {
|
||||
if len(overrides) == 0 {
|
||||
return base
|
||||
}
|
||||
|
||||
kv := make(map[string]string, len(base)+len(overrides))
|
||||
for _, item := range base {
|
||||
k, v, ok := strings.Cut(item, "=")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
kv[k] = v
|
||||
}
|
||||
for k, v := range overrides {
|
||||
kv[k] = v
|
||||
}
|
||||
|
||||
keys := make([]string, 0, len(kv))
|
||||
for k := range kv {
|
||||
keys = append(keys, k)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
|
||||
out := make([]string, 0, len(keys))
|
||||
for _, k := range keys {
|
||||
out = append(out, k+"="+kv[k])
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
@@ -1,10 +1,17 @@
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"os/exec"
|
||||
"os/signal"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strconv"
|
||||
"strings"
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -120,6 +127,316 @@ func TestRunFailureRedactsSensitiveTail(t *testing.T) {
|
||||
if !strings.Contains(err.Error(), "<redacted>") {
|
||||
t.Fatalf("error = %q, want redacted stderr tail marker", err.Error())
|
||||
}
|
||||
for _, path := range []string{req.StdoutLogPath, req.StderrLogPath} {
|
||||
data, readErr := os.ReadFile(path)
|
||||
if readErr != nil {
|
||||
t.Fatalf("read diagnostic %q: %v", path, readErr)
|
||||
}
|
||||
if strings.Contains(string(data), secretValue) {
|
||||
t.Fatalf("diagnostic %q leaked secret: %q", path, data)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunRejectsSymlinkDiagnosticWithoutTruncatingTarget(t *testing.T) {
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("creating symlinks requires privileges that are not available on every Windows runner")
|
||||
}
|
||||
exe, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
|
||||
dir := t.TempDir()
|
||||
targetPath := filepath.Join(dir, "outside.log")
|
||||
const original = "must remain unchanged"
|
||||
if err := os.WriteFile(targetPath, []byte(original), 0o600); err != nil {
|
||||
t.Fatalf("WriteFile(target) error = %v", err)
|
||||
}
|
||||
stdoutPath := filepath.Join(dir, "stdout.log")
|
||||
if err := os.Symlink(targetPath, stdoutPath); err != nil {
|
||||
t.Fatalf("Symlink() error = %v", err)
|
||||
}
|
||||
|
||||
_, err = Run(context.Background(), RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=^TestSubprocessHelper$", "--", "success"},
|
||||
EnvOverrides: map[string]string{"GO_WANT_SUBPROCESS_HELPER": "1"},
|
||||
StdoutLogPath: stdoutPath,
|
||||
StderrLogPath: filepath.Join(dir, "stderr.log"),
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), "symbolic link") {
|
||||
t.Fatalf("Run() error = %v, want symbolic-link rejection", err)
|
||||
}
|
||||
data, readErr := os.ReadFile(targetPath)
|
||||
if readErr != nil {
|
||||
t.Fatalf("ReadFile(target) error = %v", readErr)
|
||||
}
|
||||
if string(data) != original {
|
||||
t.Fatalf("target content = %q, want %q", data, original)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunFailureUsesOpenedDiagnosticAfterPathReplacement(t *testing.T) {
|
||||
exe, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
|
||||
dir := t.TempDir()
|
||||
readyPath := filepath.Join(dir, "ready")
|
||||
releasePath := filepath.Join(dir, "release")
|
||||
stderrPath := filepath.Join(dir, "stderr.log")
|
||||
openedPath := filepath.Join(dir, "opened-stderr.log")
|
||||
const secretValue = "replacement-api-key-value"
|
||||
const commandContent = "trusted command failure"
|
||||
req := RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=^TestSubprocessHelper$", "--", "delayed-fail"},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"API_KEY": secretValue,
|
||||
"SUBPROCESS_HELPER_READY_PATH": readyPath,
|
||||
"SUBPROCESS_HELPER_RELEASE_PATH": releasePath,
|
||||
"SUBPROCESS_HELPER_STDERR": commandContent,
|
||||
},
|
||||
StdoutLogPath: filepath.Join(dir, "stdout.log"),
|
||||
StderrLogPath: stderrPath,
|
||||
}
|
||||
|
||||
resultCh := make(chan error, 1)
|
||||
go func() {
|
||||
_, runErr := Run(context.Background(), req)
|
||||
resultCh <- runErr
|
||||
}()
|
||||
waitForHelperFile(t, readyPath)
|
||||
if err := os.Rename(stderrPath, openedPath); err != nil {
|
||||
t.Fatalf("Rename(stderr log) error = %v", err)
|
||||
}
|
||||
if err := os.WriteFile(stderrPath, []byte(secretValue), 0o600); err != nil {
|
||||
t.Fatalf("WriteFile(replacement) error = %v", err)
|
||||
}
|
||||
if err := os.WriteFile(releasePath, []byte("continue"), 0o600); err != nil {
|
||||
t.Fatalf("WriteFile(release) error = %v", err)
|
||||
}
|
||||
|
||||
select {
|
||||
case runErr := <-resultCh:
|
||||
if runErr == nil {
|
||||
t.Fatal("Run() error = nil, want command failure")
|
||||
}
|
||||
if strings.Contains(runErr.Error(), secretValue) {
|
||||
t.Fatalf("error read replacement-path content: %q", runErr)
|
||||
}
|
||||
if !strings.Contains(runErr.Error(), commandContent) {
|
||||
t.Fatalf("error = %q, want retained command diagnostic", runErr)
|
||||
}
|
||||
case <-time.After(3 * time.Second):
|
||||
t.Fatal("Run() did not return after helper release")
|
||||
}
|
||||
|
||||
openedData, err := os.ReadFile(openedPath)
|
||||
if err != nil {
|
||||
t.Fatalf("ReadFile(opened diagnostic) error = %v", err)
|
||||
}
|
||||
if !strings.Contains(string(openedData), commandContent) {
|
||||
t.Fatalf("opened diagnostic = %q, want command content", openedData)
|
||||
}
|
||||
replacementData, err := os.ReadFile(stderrPath)
|
||||
if err != nil {
|
||||
t.Fatalf("ReadFile(replacement diagnostic) error = %v", err)
|
||||
}
|
||||
if string(replacementData) != secretValue {
|
||||
t.Fatalf("replacement diagnostic = %q, want %q", replacementData, secretValue)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunRedactsSplitCredentialInSeparateAndSharedDiagnostics(t *testing.T) {
|
||||
exe, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
const secretValue = "split-super-secret-value"
|
||||
|
||||
for _, shared := range []bool{false, true} {
|
||||
t.Run(map[bool]string{false: "separate", true: "shared"}[shared], func(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
stdoutPath := filepath.Join(dir, "stdout.log")
|
||||
stderrPath := filepath.Join(dir, "stderr.log")
|
||||
if shared {
|
||||
stderrPath = stdoutPath
|
||||
}
|
||||
req := RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=^TestSubprocessHelper$", "--", "splitsecret"},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"API_KEY": secretValue,
|
||||
},
|
||||
StdoutLogPath: stdoutPath,
|
||||
StderrLogPath: stderrPath,
|
||||
}
|
||||
|
||||
_, runErr := Run(context.Background(), req)
|
||||
if runErr == nil {
|
||||
t.Fatal("Run() error = nil, want command failure")
|
||||
}
|
||||
if strings.Contains(runErr.Error(), secretValue) || !strings.Contains(runErr.Error(), "<redacted>") {
|
||||
t.Fatalf("error = %q, want redacted credential", runErr)
|
||||
}
|
||||
paths := map[string]struct{}{stdoutPath: {}, stderrPath: {}}
|
||||
for path := range paths {
|
||||
data, readErr := os.ReadFile(path)
|
||||
if readErr != nil {
|
||||
t.Fatalf("ReadFile(%q) error = %v", path, readErr)
|
||||
}
|
||||
if strings.Contains(string(data), secretValue) || !strings.Contains(string(data), "<redacted>") {
|
||||
t.Fatalf("diagnostic %q = %q, want redacted credential", path, data)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunRedactsInheritedSensitiveEnvironment(t *testing.T) {
|
||||
exe, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
secretValue := "inherited-secret-value"
|
||||
t.Setenv("OPENROUTER_API_KEY", secretValue)
|
||||
dir := t.TempDir()
|
||||
req := RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=TestSubprocessHelper", "--", "echoenv"},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"SUBPROCESS_HELPER_ENV_KEY": "OPENROUTER_API_KEY",
|
||||
},
|
||||
StdoutLogPath: filepath.Join(dir, "stdout.log"),
|
||||
StderrLogPath: filepath.Join(dir, "stderr.log"),
|
||||
}
|
||||
_, err = Run(context.Background(), req)
|
||||
if err == nil {
|
||||
t.Fatal("Run() error = nil, want non-nil")
|
||||
}
|
||||
if strings.Contains(err.Error(), secretValue) {
|
||||
t.Fatalf("error leaked inherited secret: %q", err)
|
||||
}
|
||||
for _, path := range []string{req.StdoutLogPath, req.StderrLogPath} {
|
||||
data, readErr := os.ReadFile(path)
|
||||
if readErr != nil {
|
||||
t.Fatalf("read diagnostic %q: %v", path, readErr)
|
||||
}
|
||||
if strings.Contains(string(data), secretValue) {
|
||||
t.Fatalf("diagnostic %q leaked inherited secret: %q", path, data)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunRedactsSensitiveOutputAndErrorTail(t *testing.T) {
|
||||
exe, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
secretValue := "override-secret-value"
|
||||
dir := t.TempDir()
|
||||
req := RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=TestSubprocessHelper", "--", "echoenv"},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"SUBPROCESS_HELPER_ENV_KEY": "OPENROUTER_API_KEY",
|
||||
"OPENROUTER_API_KEY": secretValue,
|
||||
},
|
||||
StdoutLogPath: filepath.Join(dir, "stdout.log"),
|
||||
StderrLogPath: filepath.Join(dir, "stderr.log"),
|
||||
}
|
||||
_, err = Run(context.Background(), req)
|
||||
if err == nil {
|
||||
t.Fatal("Run() error = nil, want non-nil")
|
||||
}
|
||||
if strings.Contains(err.Error(), secretValue) || !strings.Contains(err.Error(), "<redacted>") {
|
||||
t.Fatalf("error = %q, want redacted secret", err)
|
||||
}
|
||||
for _, path := range []string{req.StdoutLogPath, req.StderrLogPath} {
|
||||
data, readErr := os.ReadFile(path)
|
||||
if readErr != nil {
|
||||
t.Fatalf("read diagnostic %q: %v", path, readErr)
|
||||
}
|
||||
if strings.Contains(string(data), secretValue) || !strings.Contains(string(data), "<redacted>") {
|
||||
t.Fatalf("diagnostic %q = %q, want redacted secret", path, data)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestStreamRedactorHandlesSplitAndOverlappingSecrets(t *testing.T) {
|
||||
redactor := newStreamRedactor([]string{"abc", "abcde", "cde", ""})
|
||||
var output bytes.Buffer
|
||||
output.Write(redactor.Write([]byte("start-ab")))
|
||||
output.Write(redactor.Write([]byte("cde-end")))
|
||||
output.Write(redactor.Flush())
|
||||
if got := output.String(); got != "start-<redacted>-end" {
|
||||
t.Fatalf("redacted output = %q, want one redacted marker", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDiagnosticWriterHonorsExactLimitAndCapPlusOne(t *testing.T) {
|
||||
exactLogs := &logWriters{limits: make(chan *captureLimitError, 1)}
|
||||
var exactOutput bytes.Buffer
|
||||
exact := newDiagnosticWriter(exactLogs, "stdout", "test", &exactOutput, 5, nil)
|
||||
if _, err := exact.Write([]byte("abcde")); err != nil {
|
||||
t.Fatalf("exact Write() error = %v", err)
|
||||
}
|
||||
if err := exact.Flush(); err != nil {
|
||||
t.Fatalf("exact Flush() error = %v", err)
|
||||
}
|
||||
if got := exactOutput.String(); got != "abcde" {
|
||||
t.Fatalf("exact output = %q, want abcde", got)
|
||||
}
|
||||
if exactLogs.Limit() != nil {
|
||||
t.Fatal("exact write recorded a capture limit")
|
||||
}
|
||||
|
||||
cappedLogs := &logWriters{limits: make(chan *captureLimitError, 1)}
|
||||
var cappedOutput bytes.Buffer
|
||||
capped := newDiagnosticWriter(cappedLogs, "stderr", "test", &cappedOutput, 5, nil)
|
||||
if _, err := capped.Write([]byte("abcdef")); err == nil {
|
||||
t.Fatal("cap-plus-one Write() error = nil, want capture limit")
|
||||
}
|
||||
if err := capped.Flush(); err != nil {
|
||||
t.Fatalf("cap-plus-one Flush() error = %v", err)
|
||||
}
|
||||
if got := cappedOutput.String(); got != "abcde" {
|
||||
t.Fatalf("capped output = %q, want abcde", got)
|
||||
}
|
||||
if limit := cappedLogs.Limit(); limit == nil || limit.stream != "stderr" || limit.limit != 5 {
|
||||
t.Fatalf("capture limit = %#v, want stderr limit 5", limit)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDiagnosticWriterRetainsBoundedRedactedTail(t *testing.T) {
|
||||
logs := &logWriters{limits: make(chan *captureLimitError, 1)}
|
||||
var output bytes.Buffer
|
||||
secret := "credential-value"
|
||||
writer := newDiagnosticWriter(logs, "stderr", "test", &output, 16*1024, []string{secret})
|
||||
prefix := strings.Repeat("x", diagnosticTailBytes+512)
|
||||
if _, err := writer.Write([]byte(prefix + secret[:7])); err != nil {
|
||||
t.Fatalf("first Write() error = %v", err)
|
||||
}
|
||||
if _, err := writer.Write([]byte(secret[7:] + "-failure")); err != nil {
|
||||
t.Fatalf("second Write() error = %v", err)
|
||||
}
|
||||
if err := writer.Flush(); err != nil {
|
||||
t.Fatalf("Flush() error = %v", err)
|
||||
}
|
||||
tail := writer.Tail()
|
||||
if len(tail) > diagnosticTailBytes {
|
||||
t.Fatalf("retained tail length = %d, want at most %d", len(tail), diagnosticTailBytes)
|
||||
}
|
||||
if strings.Contains(tail, secret) || !strings.Contains(tail, "<redacted>-failure") {
|
||||
t.Fatalf("retained tail = %q, want bounded redacted content", tail)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunFailureAddsBadDescriptorHint(t *testing.T) {
|
||||
@@ -184,15 +501,14 @@ func TestRunInheritsParentEnvironmentByDefault(t *testing.T) {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
|
||||
t.Setenv("GO_WANT_SUBPROCESS_HELPER", "1")
|
||||
t.Setenv("SUBPROCESS_HELPER_ENV_KEY", "SUBPROCESS_PARENT_VALUE")
|
||||
t.Setenv("SUBPROCESS_PARENT_VALUE", "inherited-value")
|
||||
t.Setenv("PATH", "inherited-value")
|
||||
|
||||
dir := t.TempDir()
|
||||
stdoutPath := filepath.Join(dir, "stdout.log")
|
||||
req := RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=TestSubprocessHelper", "--", "printenv"},
|
||||
EnvOverrides: map[string]string{"GO_WANT_SUBPROCESS_HELPER": "1", "SUBPROCESS_HELPER_ENV_KEY": "PATH"},
|
||||
StdoutLogPath: stdoutPath,
|
||||
}
|
||||
|
||||
@@ -214,16 +530,16 @@ func TestRunEnvOverridesWinOverInheritedValues(t *testing.T) {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
|
||||
t.Setenv("GO_WANT_SUBPROCESS_HELPER", "1")
|
||||
t.Setenv("SUBPROCESS_HELPER_ENV_KEY", "SUBPROCESS_PARENT_VALUE")
|
||||
t.Setenv("SUBPROCESS_PARENT_VALUE", "parent-value")
|
||||
|
||||
dir := t.TempDir()
|
||||
stdoutPath := filepath.Join(dir, "stdout.log")
|
||||
req := RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=TestSubprocessHelper", "--", "printenv"},
|
||||
EnvOverrides: map[string]string{"SUBPROCESS_PARENT_VALUE": "override-value"},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"SUBPROCESS_HELPER_ENV_KEY": "SUBPROCESS_PARENT_VALUE",
|
||||
"SUBPROCESS_PARENT_VALUE": "override-value",
|
||||
},
|
||||
StdoutLogPath: stdoutPath,
|
||||
}
|
||||
|
||||
@@ -361,6 +677,30 @@ func TestSubprocessHelper(t *testing.T) {
|
||||
case "failbadfd":
|
||||
_, _ = os.Stderr.WriteString("OSError: [Errno 9] Bad file descriptor\n")
|
||||
os.Exit(120)
|
||||
case "delayed-fail":
|
||||
if err := os.WriteFile(os.Getenv("SUBPROCESS_HELPER_READY_PATH"), []byte("ready"), 0o600); err != nil {
|
||||
os.Exit(4)
|
||||
}
|
||||
deadline := time.Now().Add(2 * time.Second)
|
||||
for {
|
||||
if _, err := os.Stat(os.Getenv("SUBPROCESS_HELPER_RELEASE_PATH")); err == nil {
|
||||
break
|
||||
} else if !errors.Is(err, os.ErrNotExist) || time.Now().After(deadline) {
|
||||
os.Exit(5)
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
_, _ = os.Stderr.WriteString(os.Getenv("SUBPROCESS_HELPER_STDERR"))
|
||||
os.Exit(6)
|
||||
case "splitsecret":
|
||||
secret := os.Getenv("API_KEY")
|
||||
split := len(secret) / 2
|
||||
for _, stream := range []*os.File{os.Stdout, os.Stderr} {
|
||||
_, _ = stream.WriteString(secret[:split])
|
||||
time.Sleep(20 * time.Millisecond)
|
||||
_, _ = stream.WriteString(secret[split:] + "\n")
|
||||
}
|
||||
os.Exit(7)
|
||||
case "sleep":
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
os.Exit(0)
|
||||
@@ -368,7 +708,115 @@ func TestSubprocessHelper(t *testing.T) {
|
||||
key := os.Getenv("SUBPROCESS_HELPER_ENV_KEY")
|
||||
_, _ = os.Stdout.WriteString(os.Getenv(key) + "\n")
|
||||
os.Exit(0)
|
||||
case "echoenv":
|
||||
key := os.Getenv("SUBPROCESS_HELPER_ENV_KEY")
|
||||
value := os.Getenv(key)
|
||||
_, _ = os.Stdout.WriteString(value)
|
||||
_, _ = os.Stderr.WriteString(value)
|
||||
os.Exit(5)
|
||||
case "spam":
|
||||
chunk := strings.Repeat("x", 64*1024)
|
||||
count, _ := strconv.Atoi(os.Getenv("SUBPROCESS_HELPER_CHUNKS"))
|
||||
for range count {
|
||||
_, _ = os.Stdout.WriteString(chunk)
|
||||
}
|
||||
os.Exit(0)
|
||||
case "tree-spam":
|
||||
descendant := exec.Command(os.Args[0], "-test.run=^TestSubprocessHelper$", "--", "descendant")
|
||||
descendant.Env = append(os.Environ(), "GO_WANT_SUBPROCESS_HELPER=1")
|
||||
descendant.Stdout = os.Stdout
|
||||
descendant.Stderr = os.Stderr
|
||||
if err := descendant.Start(); err != nil {
|
||||
os.Exit(3)
|
||||
}
|
||||
if err := os.WriteFile(os.Getenv("SUBPROCESS_HELPER_READY_PATH"), []byte("ready"), 0o600); err != nil {
|
||||
os.Exit(4)
|
||||
}
|
||||
chunk := strings.Repeat("x", 64*1024)
|
||||
for {
|
||||
_, _ = os.Stdout.WriteString(chunk)
|
||||
}
|
||||
case "tree":
|
||||
descendant := exec.Command(os.Args[0], "-test.run=^TestSubprocessHelper$", "--", "descendant")
|
||||
descendant.Env = append(os.Environ(), "GO_WANT_SUBPROCESS_HELPER=1")
|
||||
descendant.Stdout = os.Stdout
|
||||
descendant.Stderr = os.Stderr
|
||||
if err := descendant.Start(); err != nil {
|
||||
os.Exit(3)
|
||||
}
|
||||
if err := os.WriteFile(os.Getenv("SUBPROCESS_HELPER_READY_PATH"), []byte("ready"), 0o600); err != nil {
|
||||
os.Exit(4)
|
||||
}
|
||||
time.Sleep(10 * time.Second)
|
||||
os.Exit(0)
|
||||
case "leader-exit-retained", "leader-exit-redirected", "leader-fail-redirected":
|
||||
descendant := exec.Command(os.Args[0], "-test.run=^TestSubprocessHelper$", "--", "descendant-after-release")
|
||||
descendant.Env = append(os.Environ(), "GO_WANT_SUBPROCESS_HELPER=1")
|
||||
if mode == "leader-exit-retained" {
|
||||
descendant.Stdout = os.Stdout
|
||||
descendant.Stderr = os.Stderr
|
||||
}
|
||||
if err := descendant.Start(); err != nil {
|
||||
os.Exit(3)
|
||||
}
|
||||
if !helperFileAppeared(os.Getenv("SUBPROCESS_HELPER_READY_PATH"), 2*time.Second) {
|
||||
os.Exit(4)
|
||||
}
|
||||
if mode == "leader-fail-redirected" {
|
||||
os.Exit(9)
|
||||
}
|
||||
os.Exit(0)
|
||||
case "descendant-after-release":
|
||||
if os.Getenv("SUBPROCESS_HELPER_IGNORE_TERM") == "1" {
|
||||
signal.Ignore(syscall.SIGTERM)
|
||||
}
|
||||
if err := os.WriteFile(os.Getenv("SUBPROCESS_HELPER_READY_PATH"), []byte("ready"), 0o600); err != nil {
|
||||
os.Exit(4)
|
||||
}
|
||||
deadline := time.Now().Add(10 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
if _, err := os.Stat(os.Getenv("SUBPROCESS_HELPER_RELEASE_PATH")); err == nil {
|
||||
_ = os.WriteFile(os.Getenv("SUBPROCESS_HELPER_SENTINEL_PATH"), []byte("survived"), 0o600)
|
||||
os.Exit(0)
|
||||
} else if !errors.Is(err, os.ErrNotExist) {
|
||||
os.Exit(5)
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
os.Exit(0)
|
||||
case "descendant":
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
_ = os.WriteFile(os.Getenv("SUBPROCESS_HELPER_SENTINEL_PATH"), []byte("survived"), 0o600)
|
||||
time.Sleep(10 * time.Second)
|
||||
os.Exit(0)
|
||||
default:
|
||||
os.Exit(2)
|
||||
}
|
||||
}
|
||||
|
||||
func waitForHelperFile(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(2 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
if _, err := os.Stat(path); err == nil {
|
||||
return
|
||||
} else if !errors.Is(err, os.ErrNotExist) {
|
||||
t.Fatalf("Stat(%q) error = %v", path, err)
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
t.Fatalf("helper file %q was not created", path)
|
||||
}
|
||||
|
||||
func helperFileAppeared(path string, timeout time.Duration) bool {
|
||||
deadline := time.Now().Add(timeout)
|
||||
for time.Now().Before(deadline) {
|
||||
if _, err := os.Stat(path); err == nil {
|
||||
return true
|
||||
} else if !errors.Is(err, os.ErrNotExist) {
|
||||
return false
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -2,8 +2,10 @@ package whisperx
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sync"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
var minimalTranscriptJSON = []byte(`{"schema":"speaker_transcript.v1","segments":[]}`)
|
||||
@@ -31,7 +33,8 @@ func (n *NoopClient) Transcribe(ctx context.Context, req TranscribeRequest) (Tra
|
||||
|
||||
// FakeClient captures requests and returns deterministic responses for tests.
|
||||
type FakeClient struct {
|
||||
Requests []TranscribeRequest
|
||||
requestsMu sync.RWMutex
|
||||
requests []TranscribeRequest
|
||||
Err error
|
||||
Result TranscribeResult
|
||||
TranscribeFn func(ctx context.Context, req TranscribeRequest) (TranscribeResult, error)
|
||||
@@ -42,7 +45,9 @@ func (f *FakeClient) Transcribe(ctx context.Context, req TranscribeRequest) (Tra
|
||||
if err := ctx.Err(); err != nil {
|
||||
return TranscribeResult{}, err
|
||||
}
|
||||
f.Requests = append(f.Requests, req)
|
||||
f.requestsMu.Lock()
|
||||
f.requests = append(f.requests, req)
|
||||
f.requestsMu.Unlock()
|
||||
if f.TranscribeFn != nil {
|
||||
return f.TranscribeFn(ctx, req)
|
||||
}
|
||||
@@ -65,12 +70,19 @@ func (f *FakeClient) Transcribe(ctx context.Context, req TranscribeRequest) (Tra
|
||||
return res, nil
|
||||
}
|
||||
|
||||
// RequestsSnapshot returns a copy of captured requests safe for concurrent test assertions.
|
||||
func (f *FakeClient) RequestsSnapshot() []TranscribeRequest {
|
||||
f.requestsMu.RLock()
|
||||
defer f.requestsMu.RUnlock()
|
||||
return append([]TranscribeRequest(nil), f.requests...)
|
||||
}
|
||||
|
||||
func writeMinimalJSON(path string) error {
|
||||
if path == "" {
|
||||
return nil
|
||||
}
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
if err := fileops.EnsureWorkspaceDirectory(filepath.Dir(path)); err != nil {
|
||||
return err
|
||||
}
|
||||
return os.WriteFile(path, minimalTranscriptJSON, 0o644)
|
||||
return fileops.WriteFileAtomic(path, minimalTranscriptJSON, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@ package whisperx
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"sync"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -14,8 +15,9 @@ func TestFakeClientCapturesRequestAndReturnsPath(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("Transcribe() error = %v", err)
|
||||
}
|
||||
if len(fake.Requests) != 1 || fake.Requests[0].SpeakerID != "alice" {
|
||||
t.Fatalf("requests = %#v, want one alice request", fake.Requests)
|
||||
requests := fake.RequestsSnapshot()
|
||||
if len(requests) != 1 || requests[0].SpeakerID != "alice" {
|
||||
t.Fatalf("requests = %#v, want one alice request", requests)
|
||||
}
|
||||
if res.OutputRawTranscriptPath != req.OutputRawTranscriptPath {
|
||||
t.Fatalf("output path = %q, want %q", res.OutputRawTranscriptPath, req.OutputRawTranscriptPath)
|
||||
@@ -29,3 +31,22 @@ func TestFakeClientError(t *testing.T) {
|
||||
t.Fatal("expected error, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestFakeClientRequestsSnapshotSupportsConcurrentCalls(t *testing.T) {
|
||||
fake := &FakeClient{}
|
||||
const callers = 16
|
||||
var group sync.WaitGroup
|
||||
group.Add(callers)
|
||||
for i := 0; i < callers; i++ {
|
||||
go func() {
|
||||
defer group.Done()
|
||||
if _, err := fake.Transcribe(context.Background(), TranscribeRequest{}); err != nil {
|
||||
t.Errorf("Transcribe() error = %v", err)
|
||||
}
|
||||
}()
|
||||
}
|
||||
group.Wait()
|
||||
if got := len(fake.RequestsSnapshot()); got != callers {
|
||||
t.Fatalf("captured requests = %d, want %d", got, callers)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
package whisperx
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
@@ -14,10 +13,16 @@ import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
const defaultMaxResponseBytes int64 = 10 * 1024 * 1024
|
||||
const (
|
||||
defaultMaxWhisperXResponseBytes int64 = 10 * 1024 * 1024
|
||||
whisperXUploadBufferSize = 32 * 1024
|
||||
)
|
||||
|
||||
// HTTPClientConfig contains parsed, deterministic WhisperX HTTP client settings.
|
||||
type HTTPClientConfig struct {
|
||||
@@ -39,6 +44,7 @@ type HTTPClient struct {
|
||||
retryDelay time.Duration
|
||||
httpClient *http.Client
|
||||
maxResponseBytes int64
|
||||
openAudio func(string) (io.ReadCloser, error)
|
||||
}
|
||||
|
||||
// NewHTTPClientFromConfigValues builds a client from config values and parses durations once.
|
||||
@@ -72,11 +78,11 @@ func NewHTTPClient(cfg HTTPClientConfig) (*HTTPClient, error) {
|
||||
return nil, fmt.Errorf("whisperx transcribe_url is required")
|
||||
}
|
||||
u, err := url.Parse(cfg.TranscribeURL)
|
||||
if err != nil || u.Scheme == "" || u.Host == "" {
|
||||
if err != nil || !u.IsAbs() || u.Host == "" || !isHTTPURLScheme(u.Scheme) {
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("invalid whisperx transcribe_url %q: %w", cfg.TranscribeURL, err)
|
||||
}
|
||||
return nil, fmt.Errorf("invalid whisperx transcribe_url %q", cfg.TranscribeURL)
|
||||
return nil, fmt.Errorf("invalid whisperx transcribe_url %q: must be an absolute http or https URL", cfg.TranscribeURL)
|
||||
}
|
||||
if cfg.Timeout <= 0 {
|
||||
return nil, fmt.Errorf("whisperx timeout must be > 0")
|
||||
@@ -93,12 +99,14 @@ func NewHTTPClient(cfg HTTPClientConfig) (*HTTPClient, error) {
|
||||
|
||||
client := cfg.HTTPClient
|
||||
if client == nil {
|
||||
client = &http.Client{}
|
||||
transport := http.DefaultTransport.(*http.Transport).Clone()
|
||||
transport.ExpectContinueTimeout = 100 * time.Millisecond
|
||||
client = &http.Client{Transport: transport}
|
||||
}
|
||||
|
||||
maxBytes := cfg.MaxResponseBytes
|
||||
if maxBytes <= 0 {
|
||||
maxBytes = defaultMaxResponseBytes
|
||||
maxBytes = defaultMaxWhisperXResponseBytes
|
||||
}
|
||||
|
||||
return &HTTPClient{
|
||||
@@ -109,6 +117,7 @@ func NewHTTPClient(cfg HTTPClientConfig) (*HTTPClient, error) {
|
||||
retryDelay: cfg.RetryDelay,
|
||||
httpClient: client,
|
||||
maxResponseBytes: maxBytes,
|
||||
openAudio: func(path string) (io.ReadCloser, error) { return os.Open(path) },
|
||||
}, nil
|
||||
}
|
||||
|
||||
@@ -147,7 +156,7 @@ func (c *HTTPClient) Transcribe(ctx context.Context, req TranscribeRequest) (Tra
|
||||
result.Duration = time.Since(start)
|
||||
return result, fmt.Errorf("whisperx attempt %d returned invalid json: %w", attempt, err)
|
||||
}
|
||||
if err := writeFileAtomic(req.OutputRawTranscriptPath, body, 0o644); err != nil {
|
||||
if err := writeFileAtomic(req.OutputRawTranscriptPath, body, fileops.WorkspaceFileMode); err != nil {
|
||||
result.Duration = time.Since(start)
|
||||
return result, fmt.Errorf("whisperx write transcript output %q: %w", req.OutputRawTranscriptPath, err)
|
||||
}
|
||||
@@ -183,53 +192,45 @@ func (c *HTTPClient) Transcribe(ctx context.Context, req TranscribeRequest) (Tra
|
||||
}
|
||||
|
||||
func (c *HTTPClient) doTranscribeAttempt(ctx context.Context, audioPath string) (int, []byte, error) {
|
||||
bodyBuf := &bytes.Buffer{}
|
||||
writer := multipart.NewWriter(bodyBuf)
|
||||
upload := newMultipartUpload(ctx, audioPath, c.language, c.openAudio)
|
||||
defer upload.Close()
|
||||
|
||||
fileWriter, err := writer.CreateFormFile("file", filepath.Base(audioPath))
|
||||
if err != nil {
|
||||
return 0, nil, fmt.Errorf("create multipart file field: %w", err)
|
||||
}
|
||||
|
||||
audioFile, err := os.Open(audioPath)
|
||||
if err != nil {
|
||||
return 0, nil, fmt.Errorf("open audio file %q: %w", audioPath, err)
|
||||
}
|
||||
if _, err := io.Copy(fileWriter, audioFile); err != nil {
|
||||
_ = audioFile.Close()
|
||||
return 0, nil, fmt.Errorf("copy audio file %q: %w", audioPath, err)
|
||||
}
|
||||
if err := audioFile.Close(); err != nil {
|
||||
return 0, nil, fmt.Errorf("close audio file %q: %w", audioPath, err)
|
||||
}
|
||||
|
||||
if err := writer.WriteField("language", c.language); err != nil {
|
||||
return 0, nil, fmt.Errorf("write language form field: %w", err)
|
||||
}
|
||||
if err := writer.Close(); err != nil {
|
||||
return 0, nil, fmt.Errorf("close multipart writer: %w", err)
|
||||
}
|
||||
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, c.url.String(), bodyBuf)
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, c.url.String(), upload)
|
||||
if err != nil {
|
||||
return 0, nil, fmt.Errorf("build whisperx request: %w", err)
|
||||
}
|
||||
req.Header.Set("Content-Type", writer.FormDataContentType())
|
||||
req.Header.Set("Content-Type", upload.contentType)
|
||||
req.Header.Set("Expect", "100-continue")
|
||||
|
||||
resp, err := c.httpClient.Do(req)
|
||||
if err != nil {
|
||||
_ = upload.Close()
|
||||
if producerErr := upload.Wait(); producerErr != nil {
|
||||
return 0, nil, fmt.Errorf("stream whisperx request body: %w", producerErr)
|
||||
}
|
||||
return 0, nil, fmt.Errorf("perform whisperx request: %w", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
data, err := readBounded(resp.Body, c.maxResponseBytes)
|
||||
if err != nil {
|
||||
if resp.StatusCode < 200 || resp.StatusCode >= 300 {
|
||||
_ = upload.Close()
|
||||
if producerErr := upload.Wait(); producerErr != nil {
|
||||
return resp.StatusCode, nil, fmt.Errorf("stream whisperx request body: %w", producerErr)
|
||||
}
|
||||
if _, err := readWhisperXResponse(resp.Body, c.maxResponseBytes); err != nil {
|
||||
return resp.StatusCode, nil, fmt.Errorf("read whisperx response body: %w", err)
|
||||
}
|
||||
|
||||
if resp.StatusCode < 200 || resp.StatusCode >= 300 {
|
||||
return resp.StatusCode, nil, fmt.Errorf("whisperx returned status %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
if err := upload.Wait(); err != nil {
|
||||
return resp.StatusCode, nil, fmt.Errorf("stream whisperx request body: %w", err)
|
||||
}
|
||||
|
||||
data, err := readWhisperXResponse(resp.Body, c.maxResponseBytes)
|
||||
if err != nil {
|
||||
return resp.StatusCode, nil, fmt.Errorf("read whisperx response body: %w", err)
|
||||
}
|
||||
return resp.StatusCode, data, nil
|
||||
}
|
||||
|
||||
@@ -264,57 +265,188 @@ func (c *HTTPClient) shouldRetry(parent context.Context, err error, status int)
|
||||
return false
|
||||
}
|
||||
|
||||
func readBounded(r io.Reader, maxBytes int64) ([]byte, error) {
|
||||
func readWhisperXResponse(r io.Reader, maxBytes int64) ([]byte, error) {
|
||||
limited := io.LimitReader(r, maxBytes+1)
|
||||
data, err := io.ReadAll(limited)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if int64(len(data)) > maxBytes {
|
||||
return nil, fmt.Errorf("response exceeds max size %d bytes", maxBytes)
|
||||
return nil, fmt.Errorf("whisperx response exceeds configured limit of %d bytes", maxBytes)
|
||||
}
|
||||
return data, nil
|
||||
}
|
||||
|
||||
func isHTTPURLScheme(scheme string) bool {
|
||||
switch strings.ToLower(scheme) {
|
||||
case "http", "https":
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
type multipartUpload struct {
|
||||
reader *io.PipeReader
|
||||
writer *io.PipeWriter
|
||||
contentType string
|
||||
done chan struct{}
|
||||
|
||||
mu sync.Mutex
|
||||
audio io.Closer
|
||||
err error
|
||||
aborted bool
|
||||
}
|
||||
|
||||
func newMultipartUpload(ctx context.Context, audioPath, language string, openAudio func(string) (io.ReadCloser, error)) *multipartUpload {
|
||||
reader, writer := io.Pipe()
|
||||
multipartWriter := multipart.NewWriter(writer)
|
||||
upload := &multipartUpload{
|
||||
reader: reader,
|
||||
writer: writer,
|
||||
contentType: multipartWriter.FormDataContentType(),
|
||||
done: make(chan struct{}),
|
||||
}
|
||||
|
||||
go func() {
|
||||
err := upload.write(ctx, multipartWriter, audioPath, language, openAudio)
|
||||
if err != nil {
|
||||
_ = writer.CloseWithError(err)
|
||||
} else {
|
||||
_ = writer.Close()
|
||||
}
|
||||
upload.mu.Lock()
|
||||
upload.err = err
|
||||
upload.audio = nil
|
||||
upload.mu.Unlock()
|
||||
close(upload.done)
|
||||
}()
|
||||
|
||||
go func() {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
upload.abort()
|
||||
case <-upload.done:
|
||||
}
|
||||
}()
|
||||
|
||||
return upload
|
||||
}
|
||||
|
||||
func (u *multipartUpload) Read(p []byte) (int, error) {
|
||||
return u.reader.Read(p)
|
||||
}
|
||||
|
||||
func (u *multipartUpload) Close() error {
|
||||
u.abort()
|
||||
return nil
|
||||
}
|
||||
|
||||
func (u *multipartUpload) Wait() error {
|
||||
<-u.done
|
||||
u.mu.Lock()
|
||||
defer u.mu.Unlock()
|
||||
return u.err
|
||||
}
|
||||
|
||||
func (u *multipartUpload) write(ctx context.Context, writer *multipart.Writer, audioPath, language string, openAudio func(string) (io.ReadCloser, error)) error {
|
||||
fileWriter, err := writer.CreateFormFile("file", filepath.Base(audioPath))
|
||||
if err != nil {
|
||||
return u.producerError(ctx, fmt.Errorf("create multipart file field: %w", err))
|
||||
}
|
||||
|
||||
audioFile, err := openAudio(audioPath)
|
||||
if err != nil {
|
||||
return u.producerError(ctx, fmt.Errorf("open audio file %q: %w", audioPath, err))
|
||||
}
|
||||
u.setAudio(audioFile)
|
||||
|
||||
_, copyErr := io.CopyBuffer(fileWriter, &contextReader{ctx: ctx, reader: audioFile}, make([]byte, whisperXUploadBufferSize))
|
||||
closeErr := audioFile.Close()
|
||||
u.clearAudio(audioFile)
|
||||
if copyErr != nil {
|
||||
return u.producerError(ctx, fmt.Errorf("copy audio file %q: %w", audioPath, copyErr))
|
||||
}
|
||||
if closeErr != nil {
|
||||
return u.producerError(ctx, fmt.Errorf("close audio file %q: %w", audioPath, closeErr))
|
||||
}
|
||||
|
||||
if err := writer.WriteField("language", language); err != nil {
|
||||
return u.producerError(ctx, fmt.Errorf("write language form field: %w", err))
|
||||
}
|
||||
if err := writer.Close(); err != nil {
|
||||
return u.producerError(ctx, fmt.Errorf("close multipart writer: %w", err))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (u *multipartUpload) producerError(ctx context.Context, err error) error {
|
||||
if ctx.Err() != nil {
|
||||
return ctx.Err()
|
||||
}
|
||||
u.mu.Lock()
|
||||
aborted := u.aborted
|
||||
u.mu.Unlock()
|
||||
if aborted {
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
func (u *multipartUpload) setAudio(audio io.Closer) {
|
||||
u.mu.Lock()
|
||||
u.audio = audio
|
||||
aborted := u.aborted
|
||||
u.mu.Unlock()
|
||||
if aborted {
|
||||
_ = audio.Close()
|
||||
}
|
||||
}
|
||||
|
||||
func (u *multipartUpload) clearAudio(audio io.Closer) {
|
||||
u.mu.Lock()
|
||||
if u.audio == audio {
|
||||
u.audio = nil
|
||||
}
|
||||
u.mu.Unlock()
|
||||
}
|
||||
|
||||
func (u *multipartUpload) abort() {
|
||||
u.mu.Lock()
|
||||
if u.aborted {
|
||||
u.mu.Unlock()
|
||||
return
|
||||
}
|
||||
u.aborted = true
|
||||
audio := u.audio
|
||||
u.mu.Unlock()
|
||||
|
||||
_ = u.reader.Close()
|
||||
if audio != nil {
|
||||
_ = audio.Close()
|
||||
}
|
||||
}
|
||||
|
||||
type contextReader struct {
|
||||
ctx context.Context
|
||||
reader io.Reader
|
||||
}
|
||||
|
||||
func (r *contextReader) Read(p []byte) (int, error) {
|
||||
select {
|
||||
case <-r.ctx.Done():
|
||||
return 0, r.ctx.Err()
|
||||
default:
|
||||
return r.reader.Read(p)
|
||||
}
|
||||
}
|
||||
|
||||
func writeFileAtomic(path string, data []byte, perm os.FileMode) error {
|
||||
if strings.TrimSpace(path) == "" {
|
||||
return fmt.Errorf("path is required")
|
||||
}
|
||||
dir := filepath.Dir(path)
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
return fmt.Errorf("create parent dir %q: %w", dir, err)
|
||||
if err := fileops.WriteFileAtomic(path, data, perm); err != nil {
|
||||
return fmt.Errorf("write file %q: %w", path, err)
|
||||
}
|
||||
|
||||
base := filepath.Base(path)
|
||||
tmp, err := os.CreateTemp(dir, "."+base+".tmp-*")
|
||||
if err != nil {
|
||||
return fmt.Errorf("create temp file: %w", err)
|
||||
}
|
||||
tmpPath := tmp.Name()
|
||||
removeTmp := true
|
||||
defer func() {
|
||||
if removeTmp {
|
||||
_ = os.Remove(tmpPath)
|
||||
}
|
||||
}()
|
||||
|
||||
if _, err := tmp.Write(data); err != nil {
|
||||
_ = tmp.Close()
|
||||
return fmt.Errorf("write temp file: %w", err)
|
||||
}
|
||||
if err := tmp.Sync(); err != nil {
|
||||
_ = tmp.Close()
|
||||
return fmt.Errorf("sync temp file: %w", err)
|
||||
}
|
||||
if err := tmp.Close(); err != nil {
|
||||
return fmt.Errorf("close temp file: %w", err)
|
||||
}
|
||||
if err := os.Chmod(tmpPath, perm); err != nil {
|
||||
return fmt.Errorf("chmod temp file: %w", err)
|
||||
}
|
||||
if err := os.Rename(tmpPath, path); err != nil {
|
||||
return fmt.Errorf("rename temp file: %w", err)
|
||||
}
|
||||
removeTmp = false
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"io"
|
||||
"mime/multipart"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
@@ -19,6 +20,7 @@ func TestHTTPClientTranscribeSuccess(t *testing.T) {
|
||||
var gotLanguage string
|
||||
var gotFileField string
|
||||
var gotFileSize int
|
||||
var gotFileData string
|
||||
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method != http.MethodPost {
|
||||
@@ -40,6 +42,7 @@ func TestHTTPClientTranscribeSuccess(t *testing.T) {
|
||||
t.Fatalf("ReadAll(file) error = %v", err)
|
||||
}
|
||||
gotFileSize = len(data)
|
||||
gotFileData = string(data)
|
||||
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = w.Write([]byte(`{"schema":"speaker_transcript.v1","segments":[]}`))
|
||||
@@ -77,13 +80,27 @@ func TestHTTPClientTranscribeSuccess(t *testing.T) {
|
||||
if gotFileSize == 0 {
|
||||
t.Fatal("file size = 0, want >0")
|
||||
}
|
||||
if gotFileData != "audio-data" {
|
||||
t.Fatalf("file data = %q, want exact payload", gotFileData)
|
||||
}
|
||||
verifyJSONFile(t, outPath)
|
||||
}
|
||||
|
||||
func TestHTTPClientRetriesOnTransientAndSucceeds(t *testing.T) {
|
||||
var calls atomic.Int32
|
||||
var payloads []string
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
n := calls.Add(1)
|
||||
file, _, err := r.FormFile("file")
|
||||
if err != nil {
|
||||
t.Fatalf("FormFile(file) error = %v", err)
|
||||
}
|
||||
data, err := io.ReadAll(file)
|
||||
_ = file.Close()
|
||||
if err != nil {
|
||||
t.Fatalf("ReadAll(file) error = %v", err)
|
||||
}
|
||||
payloads = append(payloads, string(data))
|
||||
if n == 1 {
|
||||
http.Error(w, "temporary", http.StatusInternalServerError)
|
||||
return
|
||||
@@ -112,6 +129,9 @@ func TestHTTPClientRetriesOnTransientAndSucceeds(t *testing.T) {
|
||||
if calls.Load() != 2 {
|
||||
t.Fatalf("calls = %d, want 2", calls.Load())
|
||||
}
|
||||
if len(payloads) != 2 || payloads[0] != "audio-data" || payloads[1] != "audio-data" {
|
||||
t.Fatalf("retry payloads = %#v, want two exact audio payloads", payloads)
|
||||
}
|
||||
verifyJSONFile(t, outPath)
|
||||
}
|
||||
|
||||
@@ -247,7 +267,244 @@ func TestHTTPClientConstructorValidation(t *testing.T) {
|
||||
if err == nil {
|
||||
t.Fatal("expected bad retry_delay error")
|
||||
}
|
||||
for _, endpoint := range []string{"ftp://example.com/transcribe", "file:///tmp/transcribe", "//example.com/transcribe", "https:/missing-host"} {
|
||||
if _, err := NewHTTPClientFromConfigValues(endpoint, "en", "30m", "2s", 1); err == nil {
|
||||
t.Errorf("NewHTTPClientFromConfigValues(%q) error = nil, want endpoint validation error", endpoint)
|
||||
}
|
||||
}
|
||||
for _, endpoint := range []string{"http://example.com/transcribe", "https://example.com/transcribe"} {
|
||||
if _, err := NewHTTPClientFromConfigValues(endpoint, "en", "30m", "2s", 1); err != nil {
|
||||
t.Errorf("NewHTTPClientFromConfigValues(%q) error = %v", endpoint, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestHTTPClientStreamsUploadBeforeSourceCompletes(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
source := newGatedReadCloser([]byte("audio-data"), release)
|
||||
firstByteReceived := make(chan struct{})
|
||||
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
part := firstMultipartFilePart(t, r)
|
||||
buf := make([]byte, 1)
|
||||
if _, err := part.Read(buf); err != nil {
|
||||
t.Errorf("Read(file) error = %v", err)
|
||||
return
|
||||
}
|
||||
close(firstByteReceived)
|
||||
if _, err := io.Copy(io.Discard, part); err != nil {
|
||||
t.Errorf("discard remaining file data: %v", err)
|
||||
return
|
||||
}
|
||||
_, _ = w.Write([]byte(`{"ok":true}`))
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
client := newTestHTTPClient(t, srv.URL)
|
||||
client.openAudio = func(string) (io.ReadCloser, error) { return source, nil }
|
||||
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
_, err := client.Transcribe(context.Background(), TranscribeRequest{AudioPath: "audio.flac", OutputRawTranscriptPath: filepath.Join(t.TempDir(), "raw.json")})
|
||||
done <- err
|
||||
}()
|
||||
|
||||
select {
|
||||
case <-firstByteReceived:
|
||||
close(release)
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("server did not receive streamed audio before source completed")
|
||||
}
|
||||
if err := <-done; err != nil {
|
||||
t.Fatalf("Transcribe() error = %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHTTPClientSourceReadFailureReachesCaller(t *testing.T) {
|
||||
sourceErr := errors.New("source read failed")
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
_, _ = io.Copy(io.Discard, r.Body)
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
client := newTestHTTPClient(t, srv.URL)
|
||||
client.openAudio = func(string) (io.ReadCloser, error) {
|
||||
return &failingReadCloser{first: []byte("partial"), err: sourceErr}, nil
|
||||
}
|
||||
|
||||
_, err := client.Transcribe(context.Background(), TranscribeRequest{AudioPath: "audio.flac", OutputRawTranscriptPath: filepath.Join(t.TempDir(), "raw.json")})
|
||||
if !errors.Is(err, sourceErr) {
|
||||
t.Fatalf("Transcribe() error = %v, want source read failure", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHTTPClientEarlyServerResponseReturns(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
source := newGatedReadCloser([]byte("audio-data"), release)
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
http.Error(w, "bad request", http.StatusBadRequest)
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
client := newTestHTTPClient(t, srv.URL)
|
||||
client.openAudio = func(string) (io.ReadCloser, error) { return source, nil }
|
||||
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
_, err := client.Transcribe(context.Background(), TranscribeRequest{AudioPath: "audio.flac", OutputRawTranscriptPath: filepath.Join(t.TempDir(), "raw.json")})
|
||||
done <- err
|
||||
}()
|
||||
|
||||
select {
|
||||
case err := <-done:
|
||||
if err == nil {
|
||||
t.Fatal("Transcribe() error = nil, want HTTP status error")
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("Transcribe() did not finish after server closed the request early")
|
||||
}
|
||||
}
|
||||
|
||||
func TestHTTPClientCancellationReleasesBlockedProducer(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
source := newGatedReadCloser([]byte("audio-data"), release)
|
||||
firstByteReceived := make(chan struct{})
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
part := firstMultipartFilePart(t, r)
|
||||
buf := make([]byte, 1)
|
||||
if _, err := part.Read(buf); err != nil {
|
||||
t.Errorf("Read(file) error = %v", err)
|
||||
return
|
||||
}
|
||||
close(firstByteReceived)
|
||||
select {
|
||||
case <-r.Context().Done():
|
||||
case <-source.closed:
|
||||
}
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
client := newTestHTTPClient(t, srv.URL)
|
||||
client.openAudio = func(string) (io.ReadCloser, error) { return source, nil }
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
_, err := client.Transcribe(ctx, TranscribeRequest{AudioPath: "audio.flac", OutputRawTranscriptPath: filepath.Join(t.TempDir(), "raw.json")})
|
||||
done <- err
|
||||
}()
|
||||
|
||||
select {
|
||||
case <-firstByteReceived:
|
||||
cancel()
|
||||
case <-time.After(time.Second):
|
||||
cancel()
|
||||
t.Fatal("server did not receive initial streamed audio")
|
||||
}
|
||||
select {
|
||||
case err := <-done:
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("Transcribe() error = %v, want context cancellation", err)
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("Transcribe() did not finish after cancellation")
|
||||
}
|
||||
select {
|
||||
case <-source.closed:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("blocked audio source was not closed on cancellation")
|
||||
}
|
||||
}
|
||||
|
||||
func TestHTTPClientBoundsWhisperXResponse(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
_, _ = io.Copy(io.Discard, r.Body)
|
||||
_, _ = w.Write([]byte(`{"ok":true}`))
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
client, err := NewHTTPClient(HTTPClientConfig{TranscribeURL: srv.URL, Language: "en", Timeout: time.Second, MaxResponseBytes: 4})
|
||||
if err != nil {
|
||||
t.Fatalf("NewHTTPClient() error = %v", err)
|
||||
}
|
||||
audioPath := writeWhisperXTestFile(t, "audio.flac", "audio-data")
|
||||
_, err = client.Transcribe(context.Background(), TranscribeRequest{AudioPath: audioPath, OutputRawTranscriptPath: filepath.Join(t.TempDir(), "raw.json")})
|
||||
if err == nil || !strings.Contains(err.Error(), "whisperx response exceeds configured limit") {
|
||||
t.Fatalf("Transcribe() error = %v, want bounded WhisperX response error", err)
|
||||
}
|
||||
}
|
||||
|
||||
func newTestHTTPClient(t *testing.T, endpoint string) *HTTPClient {
|
||||
t.Helper()
|
||||
client, err := NewHTTPClientFromConfigValues(endpoint, "en", "2s", "1ms", 0)
|
||||
if err != nil {
|
||||
t.Fatalf("NewHTTPClientFromConfigValues() error = %v", err)
|
||||
}
|
||||
return client
|
||||
}
|
||||
|
||||
func firstMultipartFilePart(t *testing.T, r *http.Request) *multipart.Part {
|
||||
t.Helper()
|
||||
reader, err := r.MultipartReader()
|
||||
if err != nil {
|
||||
t.Fatalf("MultipartReader() error = %v", err)
|
||||
}
|
||||
part, err := reader.NextPart()
|
||||
if err != nil {
|
||||
t.Fatalf("NextPart() error = %v", err)
|
||||
}
|
||||
if part.FormName() != "file" {
|
||||
t.Fatalf("first form field = %q, want file", part.FormName())
|
||||
}
|
||||
return part
|
||||
}
|
||||
|
||||
type gatedReadCloser struct {
|
||||
first []byte
|
||||
release <-chan struct{}
|
||||
closed chan struct{}
|
||||
sent bool
|
||||
once atomic.Bool
|
||||
}
|
||||
|
||||
func newGatedReadCloser(first []byte, release <-chan struct{}) *gatedReadCloser {
|
||||
return &gatedReadCloser{first: first, release: release, closed: make(chan struct{})}
|
||||
}
|
||||
|
||||
func (r *gatedReadCloser) Read(p []byte) (int, error) {
|
||||
if !r.sent {
|
||||
r.sent = true
|
||||
return copy(p, r.first), nil
|
||||
}
|
||||
select {
|
||||
case <-r.release:
|
||||
return 0, io.EOF
|
||||
case <-r.closed:
|
||||
return 0, errors.New("audio source closed")
|
||||
}
|
||||
}
|
||||
|
||||
func (r *gatedReadCloser) Close() error {
|
||||
if r.once.CompareAndSwap(false, true) {
|
||||
close(r.closed)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
type failingReadCloser struct {
|
||||
first []byte
|
||||
err error
|
||||
sent bool
|
||||
}
|
||||
|
||||
func (r *failingReadCloser) Read(p []byte) (int, error) {
|
||||
if !r.sent {
|
||||
r.sent = true
|
||||
return copy(p, r.first), nil
|
||||
}
|
||||
return 0, r.err
|
||||
}
|
||||
|
||||
func (r *failingReadCloser) Close() error { return nil }
|
||||
|
||||
func writeWhisperXTestFile(t *testing.T, name, contents string) string {
|
||||
t.Helper()
|
||||
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
)
|
||||
|
||||
@@ -47,19 +48,99 @@ func (f *artifactSelectionFlag) Normalize() ([]string, error) {
|
||||
}
|
||||
|
||||
func validateSelectedArtifacts(cfg *config.Config, selected []string) error {
|
||||
if len(selected) == 0 {
|
||||
return nil
|
||||
_, err := resolveEffectiveArtifacts(cfg, selected)
|
||||
return err
|
||||
}
|
||||
|
||||
func resolveEffectiveArtifacts(cfg *config.Config, selected []string) (artifacts.EffectiveArtifactSet, error) {
|
||||
if cfg == nil || cfg.Pipeline == nil || cfg.Pipeline.Scriptorium == nil {
|
||||
return fmt.Errorf("--artifacts requires pipeline.scriptorium.artifacts to be configured")
|
||||
if len(selected) == 0 {
|
||||
return artifacts.ResolveEffectiveArtifactSet(nil, nil)
|
||||
}
|
||||
return artifacts.EffectiveArtifactSet{}, fmt.Errorf("--artifacts requires pipeline.scriptorium.artifacts to be configured")
|
||||
}
|
||||
configured := artifacts.ConfiguredArtifactDefinitions(cfg.Pipeline.Scriptorium.Artifacts)
|
||||
normalized, err := normalizeArtifactSelection(cfg, selected)
|
||||
if err != nil {
|
||||
return artifacts.EffectiveArtifactSet{}, err
|
||||
}
|
||||
if len(normalized) > 0 && len(configured) == 0 {
|
||||
return artifacts.EffectiveArtifactSet{}, fmt.Errorf("--artifacts requires at least one configured artifact in pipeline.scriptorium.artifacts")
|
||||
}
|
||||
effective, err := artifacts.ResolveEffectiveArtifactSet(configured, normalized)
|
||||
if err != nil {
|
||||
if strings.Contains(err.Error(), "is not configured") {
|
||||
return artifacts.EffectiveArtifactSet{}, fmt.Errorf("--artifacts includes unknown artifact %q", selectedArtifactName(err))
|
||||
}
|
||||
return artifacts.EffectiveArtifactSet{}, err
|
||||
}
|
||||
if err := validateEffectiveArtifactConfiguration(cfg.Pipeline.Scriptorium.Artifacts, effective); err != nil {
|
||||
return artifacts.EffectiveArtifactSet{}, err
|
||||
}
|
||||
return effective.WithOrigins(effectiveArtifactOrigins(cfg.Pipeline)), nil
|
||||
}
|
||||
|
||||
func normalizeArtifactSelection(cfg *config.Config, selected []string) ([]string, error) {
|
||||
if len(selected) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
configured := cfg.Pipeline.Scriptorium.Artifacts
|
||||
if len(configured) == 0 {
|
||||
return fmt.Errorf("--artifacts requires at least one configured artifact in pipeline.scriptorium.artifacts")
|
||||
families := config.ArtifactFamilies(cfg.Pipeline).Families
|
||||
set := make(map[string]struct{}, len(selected))
|
||||
for _, raw := range selected {
|
||||
key := strings.TrimSpace(raw)
|
||||
if key == "" {
|
||||
return nil, fmt.Errorf("artifact names must be non-empty")
|
||||
}
|
||||
for _, name := range selected {
|
||||
if _, ok := configured[name]; !ok {
|
||||
return fmt.Errorf("--artifacts includes unknown artifact %q", name)
|
||||
if family, ok := families[key]; ok {
|
||||
for _, member := range family.Members {
|
||||
set[member] = struct{}{}
|
||||
}
|
||||
continue
|
||||
}
|
||||
if _, ok := configured[key]; !ok {
|
||||
return nil, fmt.Errorf("--artifacts includes unknown artifact %q", key)
|
||||
}
|
||||
set[key] = struct{}{}
|
||||
}
|
||||
normalized := make([]string, 0, len(set))
|
||||
for key := range set {
|
||||
normalized = append(normalized, key)
|
||||
}
|
||||
sort.Strings(normalized)
|
||||
return normalized, nil
|
||||
}
|
||||
|
||||
func effectiveArtifactOrigins(pipeline *config.PipelineConfig) map[string]artifacts.EffectiveArtifactOrigin {
|
||||
catalog := config.ArtifactFamilies(pipeline)
|
||||
origins := make(map[string]artifacts.EffectiveArtifactOrigin, len(catalog.Members))
|
||||
for key, member := range catalog.Members {
|
||||
origins[key] = artifacts.EffectiveArtifactOrigin{Family: member.Family, CharacterID: member.CharacterID}
|
||||
}
|
||||
return origins
|
||||
}
|
||||
|
||||
func selectedArtifactName(err error) string {
|
||||
message := err.Error()
|
||||
start := strings.Index(message, "\"")
|
||||
if start < 0 {
|
||||
return ""
|
||||
}
|
||||
end := strings.Index(message[start+1:], "\"")
|
||||
if end < 0 {
|
||||
return ""
|
||||
}
|
||||
return message[start+1 : start+1+end]
|
||||
}
|
||||
|
||||
func validateEffectiveArtifactConfiguration(configured map[string]config.ScriptoriumArtifactConfig, effective artifacts.EffectiveArtifactSet) error {
|
||||
for _, name := range effective.Keys() {
|
||||
artifactCfg := configured[name]
|
||||
if strings.TrimSpace(artifactCfg.PromptID) == "" {
|
||||
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.prompt_id is required when selected", name)
|
||||
}
|
||||
if strings.TrimSpace(artifactCfg.OutputPath) == "" {
|
||||
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.output_path is required when selected", name)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
|
||||
@@ -11,7 +11,6 @@ import (
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/stage"
|
||||
)
|
||||
|
||||
func TestExecuteRunStageArtifactsUnsupportedStageFails(t *testing.T) {
|
||||
@@ -43,8 +42,8 @@ func TestExecuteRunStagePublishPropagatesSelectedArtifacts(t *testing.T) {
|
||||
t.Cleanup(func() {
|
||||
executeStagesFn = origExecuteStagesFn
|
||||
})
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
|
||||
for _, s := range stages {
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, plan BoundedPlan, opts RunOptions) (*RunSummary, error) {
|
||||
for _, s := range plan.Stages() {
|
||||
capturedStages = append(capturedStages, s.Name())
|
||||
}
|
||||
capturedArtifacts = append([]string(nil), opts.SelectedArtifacts...)
|
||||
@@ -115,8 +114,8 @@ func TestRunStageArtifactsDoesNotImplyForce(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("RunStage() error = %v", err)
|
||||
}
|
||||
if !strings.Contains(out.String(), "stage=analyze executed=0 skipped=1 force=false") {
|
||||
t.Fatalf("output = %q, want analyze skip without force", out.String())
|
||||
if !strings.Contains(out.String(), "stage=analyze executed=1 skipped=0 force=false") {
|
||||
t.Fatalf("output = %q, want legacy analyze evidence rebuilt without implying force", out.String())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -126,17 +125,28 @@ func TestRunArtifactsWithSucceededAnalyzeSkipsUnlessForced(t *testing.T) {
|
||||
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
|
||||
|
||||
store := &manifest.LocalStore{}
|
||||
seed := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
|
||||
if err := RunStage(
|
||||
context.Background(),
|
||||
[]string{"prepare", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath},
|
||||
&bytes.Buffer{},
|
||||
); err != nil {
|
||||
t.Fatalf("seed prepare stage: %v", err)
|
||||
}
|
||||
seed, err := store.Load(context.Background(), manifestPath)
|
||||
if err != nil {
|
||||
t.Fatalf("load prepared manifest: %v", err)
|
||||
}
|
||||
for _, stageName := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "render", "analyze", "publish", "notify"} {
|
||||
seed.MarkStageSucceeded(stageName, time.Date(2026, 5, 3, 10, 1, 0, 0, time.UTC), nil)
|
||||
}
|
||||
seed.MarkStageSkipped("extract", time.Date(2026, 5, 3, 10, 1, 0, 0, time.UTC), "notarius_disabled")
|
||||
seedCurrentSemanticEvidence(t, loadConfigForSemanticEvidence(t, pipelinePath, campaignPath, sessionPath), seed, "prepare", "transcribe", "merge", "polish", "normalize", "trim", "render")
|
||||
if err := store.Save(context.Background(), manifestPath, seed); err != nil {
|
||||
t.Fatalf("save manifest: %v", err)
|
||||
}
|
||||
|
||||
var out bytes.Buffer
|
||||
err := Run(
|
||||
err = Run(
|
||||
context.Background(),
|
||||
[]string{"2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "session_recap"},
|
||||
&out,
|
||||
@@ -144,8 +154,8 @@ func TestRunArtifactsWithSucceededAnalyzeSkipsUnlessForced(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("Run() error = %v", err)
|
||||
}
|
||||
if !strings.Contains(out.String(), "executed=1 skipped=11") {
|
||||
t.Fatalf("output = %q, want all stages skipped", out.String())
|
||||
if !strings.Contains(out.String(), "executed=4 skipped=8") {
|
||||
t.Fatalf("output = %q, want extract reconsidered and legacy analyze plus delivery rebuilt", out.String())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -159,8 +169,8 @@ func TestExecuteAnalyzeForceRunsAnalyze(t *testing.T) {
|
||||
t.Cleanup(func() {
|
||||
executeStagesFn = origExecuteStagesFn
|
||||
})
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
|
||||
for _, s := range stages {
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, plan BoundedPlan, opts RunOptions) (*RunSummary, error) {
|
||||
for _, s := range plan.Stages() {
|
||||
capturedStages = append(capturedStages, s.Name())
|
||||
}
|
||||
capturedForce = opts.Force
|
||||
@@ -200,7 +210,7 @@ func TestExecuteAnalyzePropagatesSelectedArtifacts(t *testing.T) {
|
||||
t.Cleanup(func() {
|
||||
executeStagesFn = origExecuteStagesFn
|
||||
})
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, _ []stage.Stage, opts RunOptions) (*RunSummary, error) {
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, _ BoundedPlan, opts RunOptions) (*RunSummary, error) {
|
||||
capturedArtifacts = append([]string(nil), opts.SelectedArtifacts...)
|
||||
return &RunSummary{ManifestPath: filepath.Join(workspaceRoot, "manifest.json"), Executed: []string{"analyze"}}, nil
|
||||
}
|
||||
@@ -293,8 +303,8 @@ func TestExecutePublishForceRunsPublish(t *testing.T) {
|
||||
t.Cleanup(func() {
|
||||
executeStagesFn = origExecuteStagesFn
|
||||
})
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
|
||||
for _, s := range stages {
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, plan BoundedPlan, opts RunOptions) (*RunSummary, error) {
|
||||
for _, s := range plan.Stages() {
|
||||
capturedStages = append(capturedStages, s.Name())
|
||||
}
|
||||
capturedForce = opts.Force
|
||||
|
||||
@@ -1,11 +1,71 @@
|
||||
package app
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
)
|
||||
|
||||
func TestResolveEffectiveArtifactsExpandsFamilySelections(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
write := func(name, body string) string {
|
||||
path := filepath.Join(dir, name)
|
||||
if err := os.WriteFile(path, []byte(body), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
pipeline := write("pipeline.yml", `workspace: {root: /tmp/narratio-work}
|
||||
whisperx: {transcribe_url: https://example.test/transcribe}
|
||||
notification: {mode: noop}
|
||||
scriptorium:
|
||||
artifact_families:
|
||||
character_meta:
|
||||
enabled: false
|
||||
for_each: party.characters
|
||||
prompt_id: dnd.character_meta
|
||||
output_path_pattern: artifacts/characters/{character_id}/meta.md
|
||||
`)
|
||||
campaign := write("campaign.yml", `campaign_id: campaign
|
||||
inputs: {speakers_file: speakers.yml, autocorrect_file: autocorrect.yml, glossary_file: glossary.yml, party_file: party.yml}
|
||||
`)
|
||||
session := write("session.yml", `session_id: session
|
||||
campaign: campaign
|
||||
inputs: {audio_dir: audio}
|
||||
`)
|
||||
write("party.yml", `schema_version: narratio.party.v1
|
||||
characters:
|
||||
zeta: {player: {name: Z}, character: {name: Zeta, classes: [{name: wizard}]}}
|
||||
alpha: {player: {name: A}, character: {name: Alpha, classes: [{name: ranger}]}}
|
||||
`)
|
||||
cfg, err := config.LoadWithSessionOptions(pipeline, campaign, session, config.SessionLoadOptions{})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
effective, err := resolveEffectiveArtifacts(cfg, []string{"character_meta", "character_meta_alpha"})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got, want := effective.Keys(), []string{"character_meta_alpha", "character_meta_zeta"}; !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("keys = %#v, want %#v", got, want)
|
||||
}
|
||||
if origin, ok := effective.Origin("character_meta_alpha"); !ok || origin.Family != "character_meta" || origin.CharacterID != "alpha" {
|
||||
t.Fatalf("origin = %#v, %t", origin, ok)
|
||||
}
|
||||
if _, err := resolveEffectiveArtifacts(cfg, []string{"unknown"}); err == nil || !strings.Contains(err.Error(), "unknown artifact") {
|
||||
t.Fatalf("unknown selection error = %v", err)
|
||||
}
|
||||
if defaultEffective, err := resolveEffectiveArtifacts(cfg, nil); err != nil {
|
||||
t.Fatal(err)
|
||||
} else if len(defaultEffective.Keys()) != 0 {
|
||||
t.Fatal("default selection should omit disabled family members")
|
||||
}
|
||||
}
|
||||
|
||||
func TestArtifactSelectionFlagNormalize(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
@@ -110,6 +170,20 @@ func TestValidateSelectedArtifacts(t *testing.T) {
|
||||
},
|
||||
selected: []string{"player_handout", "session_recap"},
|
||||
},
|
||||
{
|
||||
name: "selected disabled artifact must be executable",
|
||||
cfg: &config.Config{
|
||||
Pipeline: &config.PipelineConfig{
|
||||
Scriptorium: &config.ScriptoriumConfig{
|
||||
Artifacts: map[string]config.ScriptoriumArtifactConfig{
|
||||
"player_handout": {Enabled: false, OutputPath: "artifacts/player_handout.md"},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
selected: []string{"player_handout"},
|
||||
wantErr: "pipeline.scriptorium.artifacts.player_handout.prompt_id is required when selected",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
|
||||
37
internal/app/analyze_evidence_test_helpers_test.go
Normal file
37
internal/app/analyze_evidence_test_helpers_test.go
Normal file
@@ -0,0 +1,37 @@
|
||||
package app
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifactmodel"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||
)
|
||||
|
||||
func setAppAnalyzeEvidence(m *manifest.Manifest, key, relativePath string, body []byte) {
|
||||
now := time.Date(2026, 5, 19, 23, 0, 0, 0, time.UTC)
|
||||
record := m.Stages["analyze"]
|
||||
if record == nil {
|
||||
record = &manifest.StageRecord{Name: "analyze", Status: manifest.StatusSucceeded, CreatedAt: now, UpdatedAt: now}
|
||||
m.Stages["analyze"] = record
|
||||
}
|
||||
if record.AnalyzeArtifacts == nil {
|
||||
record.AnalyzeArtifacts = map[string]manifest.AnalyzeArtifactRecord{}
|
||||
}
|
||||
digest := sha256.Sum256(body)
|
||||
record.AnalyzeStateVersion = manifest.AnalyzeStateContractVersion
|
||||
record.AnalyzeArtifacts[key] = manifest.AnalyzeArtifactRecord{
|
||||
Key: key, Status: manifest.AnalyzeArtifactCurrent,
|
||||
FingerprintVersion: manifest.AnalyzeFingerprintContractVersion,
|
||||
Fingerprint: strings.Repeat("1", 64),
|
||||
Output: &manifest.ArtifactRecord{
|
||||
Kind: "scriptorium_artifact", SourceID: artifacts.ConfiguredArtifactSourceID(key), LocalPath: relativePath,
|
||||
Contract: &artifactmodel.ContractMetadata{MediaType: "text/markdown", SchemaID: "narratio." + key, SchemaVersion: "1"},
|
||||
ProducerRunID: "run-1", Checksum: hex.EncodeToString(digest[:]),
|
||||
},
|
||||
OutputSize: int64(len(body)), ProducerRunID: "run-1", UpdatedAt: now,
|
||||
}
|
||||
}
|
||||
134
internal/app/analyze_projection.go
Normal file
134
internal/app/analyze_projection.go
Normal file
@@ -0,0 +1,134 @@
|
||||
package app
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"reflect"
|
||||
"sort"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/stage"
|
||||
)
|
||||
|
||||
type validatedAnalyzeProjection struct {
|
||||
session map[string]manifest.AnalyzeArtifactRecord
|
||||
invocation map[string]manifest.AnalyzeArtifactRecord
|
||||
}
|
||||
|
||||
type analyzeStateSnapshot struct {
|
||||
version int
|
||||
records map[string]manifest.AnalyzeArtifactRecord
|
||||
}
|
||||
|
||||
func captureAnalyzeState(manifestValue *manifest.Manifest, stageName string) analyzeStateSnapshot {
|
||||
if manifestValue == nil || stageName != "analyze" || manifestValue.Stages["analyze"] == nil {
|
||||
return analyzeStateSnapshot{}
|
||||
}
|
||||
record := manifestValue.Stages["analyze"]
|
||||
return analyzeStateSnapshot{
|
||||
version: record.AnalyzeStateVersion,
|
||||
records: manifest.CloneAnalyzeArtifactCollection(record.AnalyzeArtifacts),
|
||||
}
|
||||
}
|
||||
|
||||
func restoreAnalyzeState(manifestValue *manifest.Manifest, snapshot analyzeStateSnapshot) {
|
||||
if manifestValue == nil || manifestValue.Stages["analyze"] == nil {
|
||||
return
|
||||
}
|
||||
record := manifestValue.Stages["analyze"]
|
||||
record.AnalyzeStateVersion = snapshot.version
|
||||
record.AnalyzeArtifacts = manifest.CloneAnalyzeArtifactCollection(snapshot.records)
|
||||
}
|
||||
|
||||
func validateSuccessfulAnalyzeProjection(stageName string, result *stage.StageResult) (*validatedAnalyzeProjection, error) {
|
||||
if result == nil || result.AnalyzeState == nil {
|
||||
return nil, nil
|
||||
}
|
||||
if stageName != "analyze" {
|
||||
return nil, fmt.Errorf("stage %q returned analyze-owned state projection", stageName)
|
||||
}
|
||||
if result.Disposition == stage.StageDispositionSkipped {
|
||||
return nil, fmt.Errorf("skipped analyze result cannot contain analyze-owned state projection")
|
||||
}
|
||||
if len(result.Outputs) != 0 {
|
||||
return nil, fmt.Errorf("analyze result with state projection cannot contain ordinary outputs")
|
||||
}
|
||||
return validateAndCloneAnalyzeProjection(result.AnalyzeState)
|
||||
}
|
||||
|
||||
func validateFailedAnalyzeProjection(stageName string, result *stage.StageResult) (*validatedAnalyzeProjection, error) {
|
||||
if result == nil || result.AnalyzeState == nil {
|
||||
return nil, nil
|
||||
}
|
||||
if stageName != "analyze" {
|
||||
return nil, fmt.Errorf("stage %q returned analyze-owned state projection with an error", stageName)
|
||||
}
|
||||
if result.Disposition != stage.StageDispositionSucceeded || result.SkipReason != "" || len(result.Outputs) != 0 || len(result.Logs) != 0 || len(result.GeneratedConfigs) != 0 || len(result.Metadata) != 0 {
|
||||
return nil, fmt.Errorf("analyze result with an error may contain only analyze-owned state projection")
|
||||
}
|
||||
return validateAndCloneAnalyzeProjection(result.AnalyzeState)
|
||||
}
|
||||
|
||||
func validateAndCloneAnalyzeProjection(projection *stage.AnalyzeStateProjection) (*validatedAnalyzeProjection, error) {
|
||||
session := manifest.CloneAnalyzeArtifactCollection(projection.Session)
|
||||
invocation := manifest.CloneAnalyzeArtifactCollection(projection.Invocation)
|
||||
if err := manifest.ValidateAnalyzeArtifactCollection(manifest.AnalyzeStateContractVersion, session); err != nil {
|
||||
return nil, fmt.Errorf("validate reconciled session analyze state: %w", err)
|
||||
}
|
||||
if err := manifest.ValidateAnalyzeArtifactCollection(manifest.AnalyzeStateContractVersion, invocation); err != nil {
|
||||
return nil, fmt.Errorf("validate invocation analyze state: %w", err)
|
||||
}
|
||||
for key, invocationRecord := range invocation {
|
||||
sessionRecord, ok := session[key]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("invocation analyze artifact %q is absent from reconciled session state", key)
|
||||
}
|
||||
if !reflect.DeepEqual(invocationRecord, sessionRecord) {
|
||||
return nil, fmt.Errorf("invocation analyze artifact %q contradicts reconciled session state", key)
|
||||
}
|
||||
}
|
||||
return &validatedAnalyzeProjection{session: session, invocation: invocation}, nil
|
||||
}
|
||||
|
||||
func applyAnalyzeProjection(
|
||||
sessionManifest *manifest.Manifest,
|
||||
runManifest *manifest.RunManifest,
|
||||
projection *validatedAnalyzeProjection,
|
||||
) {
|
||||
if projection == nil {
|
||||
return
|
||||
}
|
||||
if sessionManifest != nil && sessionManifest.Stages["analyze"] != nil {
|
||||
record := sessionManifest.Stages["analyze"]
|
||||
record.AnalyzeStateVersion = manifest.AnalyzeStateContractVersion
|
||||
record.AnalyzeArtifacts = manifest.CloneAnalyzeArtifactCollection(projection.session)
|
||||
}
|
||||
if runManifest != nil && runManifest.Stages["analyze"] != nil {
|
||||
record := runManifest.Stages["analyze"]
|
||||
record.AnalyzeStateVersion = manifest.AnalyzeStateContractVersion
|
||||
record.AnalyzeArtifacts = manifest.CloneAnalyzeArtifactCollection(projection.invocation)
|
||||
}
|
||||
}
|
||||
|
||||
func analyzeProjectionOutputs(records map[string]manifest.AnalyzeArtifactRecord, producerRunID string) []manifest.ArtifactRecord {
|
||||
keys := make([]string, 0, len(records))
|
||||
for key, record := range records {
|
||||
if record.Status != manifest.AnalyzeArtifactCurrent || record.Output == nil {
|
||||
continue
|
||||
}
|
||||
if producerRunID != "" && record.ProducerRunID != producerRunID {
|
||||
continue
|
||||
}
|
||||
keys = append(keys, key)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
outputs := make([]manifest.ArtifactRecord, 0, len(keys))
|
||||
for _, key := range keys {
|
||||
record := manifest.CloneAnalyzeArtifactCollection(map[string]manifest.AnalyzeArtifactRecord{key: records[key]})[key]
|
||||
output := *record.Output
|
||||
if output.ProducerRunID == "" {
|
||||
output.ProducerRunID = record.ProducerRunID
|
||||
}
|
||||
outputs = append(outputs, output)
|
||||
}
|
||||
return outputs
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user