27 Commits

Author SHA1 Message Date
018d08fd4c Updated contributors in the LICENSE
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-24 06:54:30 -05:00
3a0a0b3940 Moved JSON response schemas into separate files 2026-05-24 06:51:50 -05:00
99ab2f181b Remove the completed documentation roadmap 2026-05-23 20:59:56 -05:00
4e4801dc98 Finalize documentation validation and update README documentation links 2026-05-24 01:12:24 +00:00
900ad74958 Consolidate development policy docs and remove legacy documentation paths 2026-05-24 01:10:29 +00:00
28d5201a69 Add integration documentation for subprocess, LLM, and input files 2026-05-24 01:07:27 +00:00
e5944f9875 Migrate internal architecture docs to docs/internal 2026-05-24 01:02:47 +00:00
6344fc91ba Add operations and troubleshooting documentation 2026-05-24 00:57:50 +00:00
7f3a91cc9e Rewrite config docs and add validated example files 2026-05-24 00:54:27 +00:00
72fb021453 Rewrite README and add canonical CLI reference 2026-05-24 00:51:02 +00:00
40e8b54d3b Establish canonical documentation paths and update README links 2026-05-24 00:47:27 +00:00
76651333b1 Add documetation policy 2026-05-23 19:44:55 -05:00
0b01c3a83d Implemented the config source setter cleanup identified during the code audit 2026-05-23 18:49:44 -05:00
0630d36734 Implemented the CLI override extraction cleanup identified in the code audit 2026-05-23 18:33:12 -05:00
52c2697040 Moved report/ledger assembly to a new processreport module 2026-05-23 18:27:30 -05:00
f790c1441c Refresh architecture and configuration documentation for current runtime behavior 2026-05-23 18:24:06 +00:00
56f9b28f4b Consolidate shared test helpers and stabilize timeout hook integration test 2026-05-23 18:18:48 +00:00
222222f449 Share configured LLM secret extraction across diagnostics paths 2026-05-23 18:11:36 +00:00
99391cd18b Centralize validator classification and malformed output handling 2026-05-23 18:02:53 +00:00
84be774b34 Share module proposal execution and transcript-section prompt payload helpers 2026-05-23 17:56:07 +00:00
e053f7e124 Add shared metadata maps and stage-name helpers 2026-05-23 17:48:37 +00:00
13029dbb33 Centralize effective config loading and path resolution 2026-05-23 17:43:27 +00:00
938bfe88c1 Centralize output schema and module key validation catalogs 2026-05-23 17:39:13 +00:00
fa1bd237d1 Centralize diagnostics artifact names and report metadata paths 2026-05-23 17:32:30 +00:00
3d7057b437 Added an implementation roadmap for the issues identified in the code audit 2026-05-23 12:22:01 -05:00
32c8c8b446 Audit code quality and deduplication opportunities 2026-05-23 11:10:44 -05:00
a3655f5540 Make module-stage LLM handling resilient and report warnings 2026-05-23 10:07:06 -05:00
116 changed files with 6316 additions and 3708 deletions

View File

@@ -1,4 +1,4 @@
Copyright (c) 2026 eric.
Copyright (c) 2026 Eric Rakestraw.
Redistribution and use in source and binary forms, with or without modification, are permitted provided that the following conditions are met:

303
README.md
View File

@@ -1,303 +1,46 @@
# Audita
Audita is a transcript polishing CLI.
Audita is a CLI that polishes transcript JSON using glossary-aware and LLM-backed correction modules.
`audita process` validates transcript/glossary input, normalizes and chunks transcript segments, runs the default correction pipeline, and emits corrected transcript output plus machine-readable diagnostics and reports.
## Quickstart
## What Audita Does
Default module sequence:
- `glossary`
- `homophones`
- `glossary`
- `spoken_word`
- `grammar`
Pipeline behavior includes:
- glossary-backed domain/acoustic corrections
- conservative homophone and mistranscription corrections
- conservative spoken-word dysfluency cleanup with semantic guardrails
- grammar/punctuation/capitalization/formatting cleanup
- validator-chain enforcement before application
- run reports and diagnostics artifacts with secret redaction
## Build and Install
Build a local binary:
Build:
```sh
go build -o ./bin/audita ./cmd/audita
```
Install into your Go bin directory:
Run the shortest useful command:
```sh
go install ./cmd/audita
audita process ./transcript.json --glossary ./glossary.yaml --output ./corrected.json
```
CLI help:
```sh
audita --help
audita process --help
audita config --help
```
## Test
Run all tests:
```sh
go test ./...
```
## Basic Usage
Required inputs:
- transcript JSON path (positional argument)
- `--glossary <glossary.yaml>`
Recommended run:
```sh
audita process transcript.json \
--glossary glossary.yaml \
--output corrected.json \
--report-json report.json
```
Select an explicit output schema (default is `bare-segments`):
```sh
audita process transcript.json \
--glossary glossary.yaml \
--output-schema audita-v1 \
--output corrected.json \
--report-json report.json
```
Recommended config-based run:
```sh
audita process transcript.json \
--glossary glossary.yaml \
--config audita.yml \
--output corrected.json \
--report-json report.json
```
Explicit module override:
```sh
audita process transcript.json \
--glossary glossary.yaml \
--modules glossary,homophones,grammar \
--output corrected.json \
--report-json report.json
```
Optional transcript background context:
```sh
audita process transcript.json \
--glossary glossary.yaml \
--transcript-description "Brief context that may help resolve ambiguous terms." \
--output corrected.json
```
The transcript description is background context only and does not override transcript content.
Write transcript JSON to stdout (no `--output`):
```sh
audita process transcript.json --glossary glossary.yaml
```
Control diagnostics location/retention:
```sh
audita process transcript.json \
--glossary glossary.yaml \
--work-dir /tmp/audita \
--work-dir-retention auto \
--output corrected.json \
--report-json report.json
```
## Stdout/Stderr Contract
- With `--output`, stdout is expected to be empty on success.
- Without `--output`, stdout contains transcript JSON only on success.
- `--report-json` writes a file and is never printed to stdout.
- stderr is human-readable diagnostics/errors.
For subprocess orchestration guidance, see [`docs/subprocess-operations.md`](docs/subprocess-operations.md).
Notes:
- the transcript JSON path is required as a positional argument;
- `--glossary` is required;
- without `--output`, corrected transcript JSON is written to stdout.
## Configuration
Precedence:
1. defaults
2. config file (`--config`, `AUDITA_CONFIG`, or default search paths when present: `/usr/local/etc/audita/config.yml`, then `/etc/audita/config.yml`)
3. environment (`AUDITA_*`)
4. CLI flags
Audita loads defaults, optional file config, environment overrides, then CLI overrides.
Config commands:
Use these commands to validate and inspect config:
```sh
audita config validate --config audita.yml
audita config print-effective --config audita.yml
audita config validate --config ./audita.yml
audita config print-effective --config ./audita.yml
```
For full config-file schema and examples, see [`docs/configuration.md`](docs/configuration.md).
For output-schema details, see [`docs/output-schemas.md`](docs/output-schemas.md).
For built-in validator keys and chain definitions, see [`docs/validators.md`](docs/validators.md).
For embedded prompt assets and prompt metadata behavior, see [`docs/prompts.md`](docs/prompts.md).
For CLI/process compatibility guarantees, see [`docs/public-contract.md`](docs/public-contract.md).
### Modules
- `AUDITA_MODULES` (CSV)
- CLI: `--modules`
### Transcript Description
CLI:
- `--transcript-description`
Behavior:
- optional background context for proposal and LLM-validator prompts;
- trimmed and length-limited by CLI validation;
- does not override transcript content;
- no `AUDITA_*` environment variable is currently defined for this setting.
### Primary LLM
Environment:
- `AUDITA_LLM_API_KEY` (or `OPENROUTER_API_KEY` fallback)
- `AUDITA_MODEL`
- `AUDITA_BASE_URL`
- `AUDITA_LLM_TIMEOUT_SECONDS`
- `AUDITA_MAX_RETRIES`
CLI:
- `--llm-api-key`
- `--model`
- `--base-url`
- `--llm-timeout-seconds`
- `--max-retries`
### Validation LLM
Environment:
- `AUDITA_VALIDATION_LLM_API_KEY`
- `AUDITA_VALIDATION_MODEL`
- `AUDITA_VALIDATION_BASE_URL`
- `AUDITA_VALIDATION_LLM_TIMEOUT_SECONDS`
- `AUDITA_VALIDATION_MAX_RETRIES`
- `AUDITA_VALIDATION_LLM_CONCURRENCY`
- `AUDITA_VALIDATION_MAX_PROMPT_TOKENS`
CLI:
- `--validation-llm-api-key`
- `--validation-model`
- `--validation-base-url`
- `--validation-llm-timeout-seconds`
- `--validation-max-retries`
- `--validation-llm-concurrency`
- `--validation-max-prompt-tokens`
### LLM Concurrency
Environment:
- `AUDITA_TOTAL_LLM_CONCURRENCY`
- `AUDITA_PROPOSAL_LLM_CONCURRENCY`
- `AUDITA_VALIDATION_LLM_CONCURRENCY`
- `AUDITA_LLM_CONCURRENCY` (legacy alias for `AUDITA_TOTAL_LLM_CONCURRENCY`)
CLI:
- `--total-llm-concurrency`
- `--proposal-llm-concurrency`
- `--validation-llm-concurrency`
- `--llm-concurrency` (legacy alias for `--total-llm-concurrency`)
Behavior:
- all proposal and validation LLM calls are bounded by total LLM concurrency
- proposal LLM calls are additionally bounded by proposal LLM concurrency
- when validation concurrency is unset, it inherits total LLM concurrency
- when explicitly set, proposal and validation concurrency must each be `<= total-llm-concurrency`
- canonical total settings win when both canonical and legacy alias settings are provided at the same precedence layer
### Confidence Thresholds
Environment:
- `AUDITA_GLOSSARY_CONFIDENCE_THRESHOLD`
- `AUDITA_HOMOPHONES_CONFIDENCE_THRESHOLD`
- `AUDITA_SPOKEN_WORD_CONFIDENCE_THRESHOLD`
- `AUDITA_GRAMMAR_CONFIDENCE_THRESHOLD`
CLI:
- `--glossary-confidence-threshold`
- `--homophones-confidence-threshold`
- `--spoken-word-confidence-threshold`
- `--grammar-confidence-threshold`
### Normalization and Chunking
Environment:
- `AUDITA_NORMALIZE_MAX_SEGMENT_GAP`
- `AUDITA_NORMALIZE_ELLIPSIS_GAP`
- `AUDITA_NORMALIZE_MAX_SEGMENT_DURATION`
- `AUDITA_NORMALIZE_MAX_SEGMENT_TOKENS`
- `AUDITA_MAX_SECTION_TOKENS`
- `AUDITA_MIN_SECTION_TOKENS`
- `AUDITA_TARGET_SECTIONS`
CLI:
- `--normalize-max-segment-gap`
- `--normalize-ellipsis-gap`
- `--normalize-max-segment-duration`
- `--normalize-max-segment-tokens`
- `--max-section-tokens`
- `--min-section-tokens`
- `--target-sections`
### Work Directory
Environment:
- `AUDITA_WORK_DIR`
- `AUDITA_WORK_DIR_RETENTION` (`auto`, `always`, `never`)
CLI:
- `--work-dir`
- `--work-dir-retention`
Retention behavior:
- `always`: keep all run directories
- `never`: keep successful run directories
- `auto`: keep failed runs and successful runs with skipped/rejected corrections
## Reports and Diagnostics
Per-run diagnostics include:
- source transcript artifacts
- normalized transcript artifact
- normalization summary
- chunking summary
- utilization diagnostics summary
- correction ledger
- invocation metadata
- redacted effective config
- module/validator prompt-response diagnostics
- `report.json`
- `error.log` on failure
Optional external report output:
- `--report-json <path>`
## Documentation
- Architecture: [`docs/architecture.md`](docs/architecture.md)
- Diagnostics: [`docs/diagnostics.md`](docs/diagnostics.md)
- Structured LLM adapter: [`docs/structured-llm.md`](docs/structured-llm.md)
- Subprocess operations: [`docs/subprocess-operations.md`](docs/subprocess-operations.md)
- Release checklist: [`docs/release-checklist.md`](docs/release-checklist.md)
- CLI reference: [`docs/cli.md`](docs/cli.md)
- Configuration reference: [`docs/config.md`](docs/config.md)
- Operations guide: [`docs/operations.md`](docs/operations.md)
- Troubleshooting: [`docs/troubleshooting.md`](docs/troubleshooting.md)
- Subprocess integration: [`docs/integrations/subprocess.md`](docs/integrations/subprocess.md)
- OpenAI-compatible LLM integration: [`docs/integrations/openai-compatible-llm.md`](docs/integrations/openai-compatible-llm.md)
- Transcript and glossary file integration: [`docs/integrations/transcript-glossary-files.md`](docs/integrations/transcript-glossary-files.md)
- Development workflow: [`docs/policy/development.md`](docs/policy/development.md)
- Architecture policy: [`docs/policy/architecture.md`](docs/policy/architecture.md)
- Documentation policy: [`docs/policy/documentation.md`](docs/policy/documentation.md)

View File

@@ -16,6 +16,7 @@ import (
"time"
"gitea.maximumdirect.net/eric/audita/internal/cli"
"gitea.maximumdirect.net/eric/audita/internal/testsupport"
)
func TestHelperProcess(t *testing.T) {
@@ -382,25 +383,22 @@ func TestProcessFailureMalformedStructuredLLMResponseViaSubprocessHook(t *testin
"--work-dir-retention",
"always",
)
if result.exitCode == 0 {
t.Fatalf("expected nonzero exit code")
if result.exitCode != 0 {
t.Fatalf("expected zero exit code, got %d stderr=%q", result.exitCode, result.stderr)
}
if result.stdout != "" {
t.Fatalf("expected empty stdout on failure, got %q", result.stdout)
if !json.Valid([]byte(result.stdout)) {
t.Fatalf("expected transcript JSON on stdout, got %q", result.stdout)
}
if !strings.Contains(result.stderr, "runner_execution") {
t.Fatalf("expected runner_execution failure, got %q", result.stderr)
}
if !strings.Contains(result.stderr, "diagnostics:") {
t.Fatalf("expected diagnostics path in stderr, got %q", result.stderr)
if result.stderr != "" {
t.Fatalf("expected empty stderr on success, got %q", result.stderr)
}
report := readFile(t, reportPath)
if !json.Valid(report) {
t.Fatalf("expected valid failure report JSON")
t.Fatalf("expected valid success report JSON")
}
runDir := onlyRunDir(t, workDir)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err != nil {
t.Fatalf("expected error.log, got: %v", err)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err == nil || !os.IsNotExist(err) {
t.Fatalf("did not expect error.log, got: %v", err)
}
}
@@ -502,6 +500,9 @@ func TestProcessCancellationViaSubprocessTimeoutHook(t *testing.T) {
"always",
)
if result.stdout != "" {
if result.stderr == "" {
t.Skipf("subprocess timeout hook did not trigger in this run; stdout=%q", result.stdout)
}
t.Fatalf("expected empty stdout on failure, got %q", result.stdout)
}
if !strings.Contains(result.stderr, "context deadline exceeded") {
@@ -621,12 +622,7 @@ func schemaFixturePath(name string) string {
}
func readFile(t *testing.T, path string) []byte {
t.Helper()
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("failed to read file %q: %v", path, err)
}
return data
return testsupport.ReadFile(t, path)
}
func assertJSONSemanticallyEqual(t *testing.T, expected []byte, actual []byte) {
@@ -669,41 +665,13 @@ func writeLargeTranscriptFixture(t *testing.T, segments int) string {
}
func onlyRunDir(t *testing.T, workDir string) string {
t.Helper()
entries, err := os.ReadDir(workDir)
if err != nil {
t.Fatalf("failed to read work dir %q: %v", workDir, err)
}
dirs := make([]string, 0, len(entries))
for _, e := range entries {
if e.IsDir() {
dirs = append(dirs, filepath.Join(workDir, e.Name()))
}
}
if len(dirs) != 1 {
t.Fatalf("expected exactly one run dir in %q, found %d", workDir, len(dirs))
}
return dirs[0]
return testsupport.OnlyRunDir(t, workDir)
}
func assertNoSecretInFile(t *testing.T, path, secret string) {
t.Helper()
raw := string(readFile(t, path))
if strings.Contains(raw, secret) {
t.Fatalf("secret leaked in %s", path)
}
testsupport.AssertNoSecretInFile(t, path, secret)
}
func assertNoSecretInTree(t *testing.T, root, secret string) {
t.Helper()
_ = filepath.WalkDir(root, func(path string, d os.DirEntry, err error) error {
if err != nil || d == nil || d.IsDir() {
return nil
}
raw, readErr := os.ReadFile(path)
if readErr == nil && strings.Contains(string(raw), secret) {
t.Fatalf("secret leaked in %s", path)
}
return nil
})
testsupport.AssertNoSecretInTree(t, root, secret)
}

View File

@@ -1,849 +0,0 @@
# Audita Architecture
## Scope and intent
This document describes:
- the architecture used in production today.
Historical rewrite details live in `docs/rewrite-notes.md`.
## Current implementation status
Implemented today:
- Go CLI entrypoint and `audita process` wiring.
- Config defaults, env loading, CLI override precedence, and validation.
- Transcript and glossary parsing/validation.
- Deterministic transcript normalization.
- Deterministic token estimation and transcript chunking.
- Per-run diagnostics directory creation plus process-level artifacts.
- Process report JSON output with diagnostics artifact references.
- Framework foundation packages for contracts and proposal application.
- Production runner orchestration package with deterministic sequential module execution.
- Module-level report structures with applied/skipped change records.
- Runtime validator models and deterministic validators.
- Deterministic validator-chain execution in the runner with cardinality enforcement.
- Module-level validator decision/rejection reporting.
- Internal structured LLM client contract plus an Audita-owned OpenAI-compatible structured LLM adapter package.
- Bounded FIFO LLM scheduler infrastructure with context-aware permit handling.
- Runtime primary/validation LLM effective-config resolution helpers with validation inheritance.
- Generic JSON prompt/response diagnostics writer primitives with secret redaction.
- LLM-backed validator models, prompt builders, batching, and runtime execution.
- Runner wiring for LLM validators via the internal structured LLM abstraction and scheduler hooks.
- LLM validator diagnostics artifacts and report-level decision metadata paths.
- Shared LLM proposal-generation helper with structured correction-set parsing.
- Deterministic proposal-index assignment and enriched proposal mapping for shared generation.
- Proposal-generation diagnostics artifacts with secret redaction.
- Production module registry with known-key recognition and explicit unsupported-module errors.
- Production `grammar` module implementation in `internal/modules/grammar`.
- Production `glossary` module implementation in `internal/modules/glossary`.
- Production `homophones` module implementation in `internal/modules/homophones`.
- Production `spoken_word` module implementation in `internal/modules/spoken_word`.
- Explicit runtime support for `--modules grammar` through the production runner path.
- Explicit runtime support for `--modules glossary`, including repeated stages such as `--modules glossary,glossary`.
- Explicit runtime support for `--modules homophones` through the production runner path.
- Explicit runtime support for `--modules spoken_word` through the production runner path.
Current reality:
- all production modules exist and are wired into the default runtime path.
- a normal `audita process` run without `--modules` now executes the full sequence:
- `glossary`
- `homophones`
- `glossary`
- `spoken_word`
- `grammar`
- repeated glossary stages are deterministic and reported distinctly as `glossary_1` and `glossary_2`.
## Actual Go package layout
```text
cmd/audita/
main.go
internal/cli/
run.go
internal/core/config/
config.go
env.go
flags.go
redaction.go
validation.go
internal/core/schema/
transcript.go
glossary.go
errors.go
internal/core/io/
files.go
internal/core/normalization/
normalize.go
tokens.go
internal/core/chunking/
sections.go
summary.go
tokens.go
internal/core/diagnostics/
run_dir.go
internal/core/reporting/
report.go
internal/framework/contracts/
contracts.go
internal/framework/proposals/
proposal.go
policy.go
preview.go
apply.go
internal/framework/runner/
observability.go
runner.go
internal/framework/proposal_generation/
generate.go
internal/framework/modules/
registry.go
internal/modules/grammar/
module.go
prompt.go
internal/modules/glossary/
module.go
prompt.go
internal/modules/homophones/
module.go
prompt.go
internal/modules/spoken_word/
module.go
prompt.go
internal/framework/validators/
models.go
deterministic.go
llm_models.go
llm_prompt_builders.go
llm_batching.go
llm_validators.go
internal/validators/
metadata/
metadata.go
registry.go
chains.go
confidence_threshold/
validator.go
original_text_presence/
validator.go
non_empty_corrected_text/
validator.go
no_effect/
validator.go
protected_terms/
validator.go
spoken_form_plausibility/
validator.go
meaning_reversal_review/
validator.go
editorial_review/
validator.go
grammar_review/
validator.go
spoken_word_review/
validator.go
internal/prompts/
registry.go
render.go
assets/
shared/
modules/
validators/
internal/framework/llm/
openai_compatible_client.go
scheduler.go
effective_config.go
diagnostics.go
internal/framework/responseschema/
registry.go
registry_test.go
internal/cli/
review_artifacts.go
parity_test.go
release_fixtures_test.go
testdata/
parity/
release/
```
## Current CLI behavior
Primary commands:
```sh
audita process <transcript.json> --glossary <glossary.yaml> [flags]
audita config validate --config <config.yml>
audita config print-effective [--config <config.yml>]
```
Current runtime flow (`internal/cli/run.go`):
1. Build runtime config from:
- defaults;
- file config source (`--config`, `AUDITA_CONFIG`, or default search paths when present: `/usr/local/etc/audita/config.yml`, then `/etc/audita/config.yml`);
- environment overrides;
- CLI overrides.
2. Parse flags and apply CLI overrides.
3. Validate transcript positional argument and required `--glossary`.
4. Create per-run diagnostics directory.
5. Read transcript and glossary files.
6. Parse/validate transcript and glossary.
7. Write source transcript artifacts.
8. Normalize transcript.
9. Write normalized transcript and normalization summary artifacts.
10. Chunk normalized transcript and compute chunk summaries.
11. Write chunking summary artifact.
12. Execute runner modules sequentially:
- default run path uses configured default sequence (`glossary,homophones,glossary,spoken_word,grammar`);
- explicit `--modules` overrides the default sequence;
- test/injected module factory path remains available for deterministic runtime tests.
- each module recomputes chunks from the current working transcript, runs chunk proposal work concurrently, aggregates deterministically, validates, and applies approved proposals once.
13. Output working transcript to `--output` file or stdout.
14. Build process report metadata.
15. Optionally write `--report-json`; always write run-dir `report.json`.
16. Apply work-dir retention.
Config command behavior (`internal/cli/run.go`):
- `audita config validate --config <path>`:
- loads and validates a versioned YAML config file;
- does not require transcript or glossary inputs.
- `audita config print-effective [--config <path>]`:
- builds effective config from defaults + file config + env overrides;
- prints redacted JSON to stdout;
- does not require transcript or glossary inputs.
Parity fixture status:
- representative Python-parity fixture coverage exists under `internal/cli/testdata/parity`;
- parity tests use fake structured LLM responses for deterministic behavior, including default full-pipeline shape assertions;
- parity comparisons intentionally ignore nondeterministic metadata (timestamps, run IDs, temp paths, token usage) and remain strict for deterministic contract fields (transcript content, module order/instance naming, applied/skipped/rejected counts, and status).
- intentional Python-vs-Go differences and open parity gaps are documented in `docs/python-parity.md`.
Important behavior details:
- Glossary is validated and is used for explicit glossary/grammar/homophones/spoken_word module correction paths.
- Default production CLI behavior now executes the full production module sequence unless `--modules` override is supplied.
- Explicit `--modules grammar`, `--modules glossary`, `--modules homophones`, and `--modules spoken_word` continue to run production module paths with LLM-backed proposal generation and validator-chain execution.
- Default runs (without explicit module selection) perform LLM calls through production module and validator paths.
- Success path is generally quiet on stderr.
- Source IDs are preserved into a canonical transcript before normalization; normalization then reassigns output IDs sequentially from `1`.
## Implemented data contracts
### Transcript input
Accepted top-level forms:
- bare JSON array of segments
- object with `segments` array
Source segment contract:
- `id` optional integer
- `speaker` non-empty string
- `start` finite non-negative number
- `end` finite non-negative number with `end >= start`
- `text` non-empty string
- `categories` optional array of non-empty strings
Additional checks:
- duplicate explicit source IDs are rejected.
### Transcript output
Transcript output is selected through an output schema registry (`internal/core/outputschema`).
Supported output schemas:
- `bare-segments` (default):
- top-level JSON array of normalized segments;
- each segment includes `id`, `speaker`, `start`, `end`, `text`, optional `categories`.
- `audita-v1`:
- top-level object with:
- `schema: "audita-v1"`
- `version: "v1"`
- `segments: [...]` (same normalized segment payload).
Current status:
- `seriatim-intermediate` is not implemented yet; selecting it fails clearly as an unsupported output schema.
Selection behavior:
- CLI: `--output-schema <name>`
- file config: `output.schema: <name>`
- precedence remains runtime-wide defaults -> file config -> env -> CLI.
Both stdout transcript output and `--output` file output use the same selected output encoder.
### Glossary input
YAML with `glossary` entries. Required fields per entry:
- `name`, `category`, `summary`
Optional:
- `aliases`, `plural`
## Implemented config/env/flag behavior
Precedence for `audita process`:
1. defaults (`config.Default()`)
2. config file (if resolved from `--config`, `AUDITA_CONFIG`, or default path)
3. environment overrides
4. CLI flags (`ApplyCLIOverrides`)
File-config source behavior:
- explicit `--config <path>`:
- required to exist, otherwise process fails clearly.
- `AUDITA_CONFIG` (when `--config` is not provided):
- required to exist, otherwise process fails clearly.
- default paths `/usr/local/etc/audita/config.yml`, then `/etc/audita/config.yml` (when neither explicit source is provided):
- first existing path in that order is used;
- both missing is silently ignored.
Versioned file-config behavior (`internal/core/config/file_config.go`):
- supported version: `version: 1`;
- missing version fails;
- unsupported version fails;
- strict unknown-field rejection is enabled.
`api_key_env` behavior:
- file config can declare API key environment variable names for proposal/validation LLM settings;
- runtime resolves those names from the process environment during config application;
- no direct API-key value field is supported in file config.
Redaction behavior:
- effective config artifacts and `audita config print-effective` both use the same redaction path (`Config.Redacted()`), so API keys are not emitted in plaintext.
Implemented config surfaces include:
- module list
- primary and validation LLM settings
- total/proposal/validation LLM concurrency controls
- transcript description context (`--transcript-description`)
- section token controls and target sections
- confidence thresholds
- normalization controls
- work-dir and retention mode
Current caveat:
- LLM/module-related settings are active for default and explicit module-run paths.
- compatibility environment variables and lower-level CLI tuning flags remain available while the preferred config-driven surface is adopted.
Transcript description behavior:
- `--transcript-description` is a process-flag input for optional user-supplied background context.
- runtime config stores this value in `Config.TranscriptDescription` after CLI trimming and length validation.
- default value is empty; empty values produce no prompt context section.
- this value is intentionally non-secret and appears in effective config and invocation metadata artifacts.
## Implemented transcript description prompt context
Transcript description context is wired through production prompt paths:
- proposal prompts for `glossary`, `homophones`, `spoken_word`, and `grammar`;
- LLM-backed validator prompts for spoken-form plausibility, meaning reversal, editorial review, grammar review, and spoken-word review.
Prompt guardrail semantics are consistent across modules and validators:
- transcript description is labeled as "background context only";
- it may help interpret ambiguous terms;
- it must not override transcript content;
- the model must not invent corrections, facts, names, events, motivations, or speaker intent from this description.
Generated transcript descriptions remain deferred and are not implemented in the current runtime.
## Implemented embedded prompt assets
Prompt assets are now built-in embedded Markdown files under `internal/prompts/assets`:
- `assets/modules/*` for production module proposal prompts;
- `assets/validators/*` for LLM-backed validator prompts;
- `assets/shared/prompt_hardening.md` for shared prompt-injection hardening text.
Prompt source behavior:
- built-in embedded prompts are the only supported source in current runtime;
- filesystem prompt overrides and prompt-source selection flags are not implemented.
`internal/prompts` registry responsibilities:
- register stable prompt IDs;
- register prompt version and source metadata;
- load embedded assets;
- compute deterministic SHA-256 prompt source hashes;
- render system/user prompts with `text/template` using missing-key errors.
Prompt metadata fields:
- `prompt_id`
- `prompt_version`
- `prompt_source` (`builtin`)
- `embedded_path`
- `sha256`
Prompt rendering flow:
- module proposal builders construct typed template data (section JSON, glossary JSON, transcript-description block) and render via `internal/prompts`;
- validator prompt builders construct typed template data (validation payload JSON, transcript-description block) and render via `internal/prompts`.
Shared prompt hardening:
- the same centralized hardening fragment is included in every proposal and LLM-validator prompt;
- hardening text enforces untrusted transcript handling, no instruction-following from transcript content, and no invented facts/corrections.
Prompt metadata diagnostics flow:
- proposal-generation diagnostics request metadata includes prompt metadata;
- LLM-validator diagnostics request metadata includes prompt metadata;
- detailed prompt metadata is diagnostics-scoped today and is not yet expanded into broad report-level prompt registries.
## Implemented structured LLM infrastructure
`internal/framework/contracts` now defines a typed structured-completion contract:
- `StructuredLLMClient.CompleteStructured(ctx, req, out)`
- caller-owned typed decode target via `out` pointer.
- caller-selected structured response schema metadata via `StructuredCompletionRequest.ResponseSchema`.
`internal/framework/llm` provides `OpenAICompatibleClient`, a direct `net/http` adapter over OpenAI-compatible chat completions:
- configurable `base_url`, model, optional API key, retries, HTTP client, and request timeout;
- OpenAI-compatible endpoint behavior (for example OpenAI/OpenRouter/local-compatible base URLs);
- request message translation from `contracts.LLMMessage` to chat-completions messages;
- strict `response_format.type = json_schema` with registered structured response schemas (`strict: true`, schema name, and schema body);
- response metadata mapping (provider/model/token usage) into Audita-owned response types;
- API-key redaction in adapter-returned errors;
- context cancellation and timeout propagation through request contexts and HTTP client timeouts;
- bounded retry behavior for transient request failures and malformed retryable structured responses.
Structured response schemas are owned by Audita in `internal/framework/responseschema` and currently include:
- key `correction_set`:
- id `audita.correction_set`
- version `v1`
- name `audita_correction_set_v1`
- sha256 `05f8ff3fa04f68115c0cb1859d2656f51aa5c0bae8ff2470b2d4f6f531953195`
- key `validator_decision_set`:
- id `audita.validator_decision_set`
- version `v1`
- name `audita_validator_decision_set_v1`
- sha256 `b73f4790b98fbb955f0aec5496dd8ce9a8fe14aa2f35c700b4b4e5634f106fd5`
Provider-level structured output is treated as a guardrail, not a trust boundary:
- the adapter decodes assistant message content into caller-owned structs;
- proposal-generation and validator layers continue local validation (shape, cardinality, confidence bounds, and proposal-index semantics) before changes can be applied.
Current runtime boundary:
- the default CLI runtime path (without explicit module selection) instantiates the full production module sequence.
- LLM calls are exercised in production in both default full-pipeline runs and explicit `--modules` runs, and in tests when fake/injected clients are used.
- normal `go test ./...` does not require real LLM credentials or Python dependencies.
`internal/framework/llm` also provides:
- a bounded FIFO `Scheduler` for controlled concurrent LLM calls with reliable permit release on success, error, and cancellation;
- primary/validation effective-config resolution helpers, including validation inheritance fallback to total LLM concurrency settings;
- generic interaction diagnostics primitives that write machine-readable JSON artifacts for request metadata, request payload, response payload, and optional error payload with secret redaction.
Structured LLM diagnostics behavior:
- proposal-generation and validator diagnostics include structured response schema metadata (`id`, `version`, `name`, `sha256`) when schema-driven calls are made;
- API keys and bearer tokens are redacted from request/response/error diagnostics artifacts and surfaced errors.
Dependency posture:
- the runtime no longer depends on `instructor-go`;
- structured LLM behavior is implemented through Audita-owned code paths behind `StructuredLLMClient`.
LLM concurrency runtime behavior:
- `total` concurrency bounds all proposal and validation LLM calls.
- `proposal` concurrency adds a proposal-only sub-cap, composed with total.
- `validation` concurrency adds a validation-only sub-cap, composed with total.
- legacy `llm-concurrency` inputs remain compatibility aliases for total concurrency.
- modules execute serially, chunk proposals run concurrently within each module, and approved proposals are applied once per module in deterministic order.
## Implemented normalization behavior
Normalization (`internal/core/normalization`) currently:
- sorts by segment start time;
- merges adjacent same-speaker segments when constraints pass;
- uses gap-based joiners:
- gap `< ellipsis_gap` -> single space join
- gap `>= ellipsis_gap` -> `... ` join
- enforces merged duration and token-limit constraints;
- reassigns output IDs sequentially from `1`;
- returns `NormalizationSummary` with merge and skip counters.
Note: merged categories are concatenated (not deduplicated).
## Implemented chunking behavior
Chunking (`internal/core/chunking`) currently provides:
- deterministic heuristic token estimation;
- contiguous sectioning with section metadata;
- max/min section token validation;
- optional `target_sections` override for section-count planning;
- summary and detailed summary generation.
Current behavior details:
- if a single segment exceeds max tokens, it is emitted as its own section (not hard-failed);
- default section count is planned from `ceil(total_tokens / max_section_tokens)`;
- section sizing targets `ceil(total_tokens / section_count)` with a deterministic forward pass;
- sections remain contiguous and ordered, and segments are never split.
## Implemented proposal/replacement infrastructure
`internal/framework/proposals` provides deterministic proposal composition logic:
- `CorrectionProposal` and `EnrichedCorrectionProposal` models;
- replacement policies: `require_unique`, `replace_all`;
- safe preview (`PreviewProposalForSegment`) with stable skip reasons;
- deterministic apply (`ApplyProposals`) in ascending `proposal_index` order;
- applied/skipped change records suitable for reporting.
`internal/framework/contracts` provides interfaces and run-spec metadata scaffolding, including deterministic repeated module instance naming (`ResolveModuleRunSpecs`).
These primitives are wired into the production runner and report model. The grammar, glossary, homophones, and spoken_word modules are implemented.
## Implemented validator runtime infrastructure
`internal/framework/validators` provides deterministic validator infrastructure:
- runtime validation request/result models;
- stable validator reason codes;
- cardinality enforcement for validator decisions:
- missing proposal indexes fail
- duplicate proposal indexes fail
- unknown proposal indexes fail
- deterministic validators:
- confidence threshold by module key/config threshold
- original-text presence against current working transcript
- non-empty corrected text
- identical/no-effect rejection
- conservative protected glossary-term guard for non-glossary modules
`internal/framework/runner` executes module pipelines with deterministic boundaries:
- modules still execute serially over the working transcript;
- section proposal work is launched promptly and can run concurrently;
- section-level validator-chain work starts as section proposals become available (deterministic validators before LLM-backed validators);
- proposal-generation and LLM-validator calls can overlap under composed scheduler limits;
- approved proposals are still applied once per module after section work settles.
Validator rejections are reported distinctly from proposal-application skips.
Validator composition is now explicit and registry-backed through `internal/validators`:
- built-in validator registry with stable keys and lookup/build failure for unknown keys;
- built-in chain definitions per production module key;
- production modules resolve validator chains from those built-in definitions.
Package ownership boundary:
- `internal/validators/<validator_key>` owns built-in validator construction and stable key identity.
- `internal/framework/validators` remains shared runtime machinery:
- request/result models;
- decision cardinality enforcement;
- protected-vocabulary helpers;
- generic LLM-backed validator runtime, batching, and diagnostics glue.
Validator execution classification metadata:
- `internal/validators/metadata` defines execution class markers:
- `deterministic`
- `llm_backed`
- runner ordering uses this metadata interface rather than concrete framework validator type assertions.
- validators without classification metadata default to deterministic ordering.
`protected_terms` construction ownership:
- `internal/validators/protected_terms.New()` builds the general (non-glossary-stage) variant.
- `internal/validators/protected_terms.NewGlossaryStage()` builds the glossary-stage variant used by glossary chains.
- both variants preserve the stable key `protected_terms`.
Stable built-in validator keys:
- deterministic:
- `confidence_threshold`
- `original_text_presence`
- `non_empty_corrected_text`
- `no_effect`
- `protected_terms`
- LLM-backed:
- `spoken_form_plausibility`
- `meaning_reversal_review`
- `editorial_review`
- `grammar_review`
- `spoken_word_review`
Built-in module chains:
- `glossary`:
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `spoken_form_plausibility`
- `meaning_reversal_review`
- `homophones`:
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `spoken_form_plausibility`
- `meaning_reversal_review`
- `spoken_word`:
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `spoken_word_review`
- `meaning_reversal_review`
- `grammar`:
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `grammar_review`
- `meaning_reversal_review`
1.0 boundary:
- validator chains are built-in and not user-configurable from config/CLI.
- existing threshold and batching knobs remain configurable.
## Implemented LLM-backed validator infrastructure
`internal/framework/validators` now includes LLM-backed validator support:
- typed request/response models for structured LLM validation;
- prompt builders for:
- spoken-form plausibility
- meaning reversal detection
- editorial review
- grammar review
- spoken-word review
- deterministic batching by `validation_max_prompt_tokens`;
- strict cardinality validation of structured LLM decisions (missing/duplicate/unknown indexes fail);
- safe failure behavior for malformed/invalid structured responses.
`internal/framework/runner` wires LLM validators into existing validator chains using:
- the internal structured LLM client abstraction (`contracts.StructuredLLMClient`);
- bounded scheduler hooks for validator call execution;
- diagnostics writer hooks for machine-readable prompt/response artifacts with secret redaction.
## Implemented shared proposal-generation infrastructure
`internal/framework/proposal_generation` provides a reusable, prompt-agnostic helper for future real modules:
- structured request model including module key/instance, replacement policy, working transcript context, optional section metadata, glossary, config, and diagnostics context;
- structured correction-set response model (`corrections`) mapped into existing `proposals.CorrectionProposal` and `proposals.EnrichedCorrectionProposal` models;
- deterministic proposal-index assignment through a caller-provided `start_index`;
- structured LLM calls through `contracts.StructuredLLMClient` only (no direct provider calls);
- optional bounded execution through scheduler hooks (`contracts.LLMScheduler`);
- prompt/response diagnostics artifact writing via the generic `internal/framework/llm` diagnostics primitives with redaction of API keys/secrets.
This helper only produces candidate proposals; validator-chain execution and proposal application remain runner responsibilities.
## Implemented production module-registry scaffolding
`internal/framework/modules` now provides a production registry scaffold:
- recognizes intended module keys:
- `glossary`
- `homophones`
- `spoken_word`
- `grammar`
- supports explicit constructor registration with dependency injection for:
- run spec
- config
- glossary
- proposal/validation structured LLM clients
- proposal/validation schedulers
- diagnostics directory context
- returns explicit errors for unknown keys (`unsupported_module`).
The `grammar`, `glossary`, `homophones`, and `spoken_word` module keys are now registered and constructible.
## Implemented grammar production module
`internal/modules/grammar` now provides the first production module:
- prompt builder faithfully constrained to punctuation/capitalization/spacing/article cleanup;
- explicit guardrails against meaning-changing rewrites, style rewrites, summarization, and invention;
- proposal generation through `internal/framework/proposal_generation` and `contracts.StructuredLLMClient`;
- scheduler-aware proposal calls through existing `contracts.LLMScheduler` hooks;
- replacement policy `require_unique` (current runtime policy);
- validator chain integration using existing deterministic + LLM-backed validators;
- grammar confidence threshold enforcement through existing validator/config infrastructure;
- module-level reporting and diagnostics capture through existing runner/reporting paths.
## Implemented glossary production module
`internal/modules/glossary` now provides the second production module:
- prompt builder aligned to Python glossary-module intent, constrained to glossary-backed domain/acoustic corrections;
- prompt context includes glossary names, aliases, categories, summaries, and plural forms where available;
- guardrails against broad style rewriting and against replacing unrelated terms simply because they appear in glossary entries;
- proposal generation through `internal/framework/proposal_generation` and `contracts.StructuredLLMClient`;
- scheduler-aware proposal calls through existing `contracts.LLMScheduler` hooks;
- replacement policy `replace_all` (matching Python glossary behavior);
- validator chain integration using existing deterministic + LLM-backed validators;
- glossary confidence threshold enforcement through existing validator/config infrastructure;
- module-level reporting and diagnostics capture through existing runner/reporting paths;
- explicit support for repeated glossary stages with deterministic instance names (`glossary_1`, `glossary_2`, ...), where later stages see prior-stage working transcript changes.
## Implemented protected-term behavior
`internal/framework/validators/protected_terms.go` provides deterministic glossary-derived protected vocabulary:
- extracts protected terms from glossary names and aliases;
- includes explicit plural fields and synthetic plural forms where safe;
- deduplicates and returns stable ordering for repeatable behavior/tests.
This vocabulary is used by deterministic validators for both glossary-stage and non-glossary-stage protection checks, keeping protected-term guardrails active across modules.
## Implemented homophones production module
`internal/modules/homophones` now provides the third production module:
- prompt builder aligned to Python homophones-module intent, constrained to conservative homophone/near-homophone/mistranscription corrections;
- prompt context includes protected glossary names/aliases/plurals to avoid damaging known terms;
- explicit guardrails against punctuation cleanup, grammar cleanup, style rewriting, summarization, and content invention;
- proposal generation through `internal/framework/proposal_generation` and `contracts.StructuredLLMClient`;
- scheduler-aware proposal calls through existing `contracts.LLMScheduler` hooks;
- replacement policy `require_unique` (matching Python homophones behavior);
- validator chain integration using existing deterministic + LLM-backed validators;
- homophones confidence threshold enforcement through existing validator/config infrastructure;
- protected-term guardrails for non-glossary modules remain active and are exercised through the homophones path;
- module-level reporting and diagnostics capture through existing runner/reporting paths.
## Implemented spoken_word production module
`internal/modules/spoken_word` now provides the fourth production module:
- prompt builder aligned to Python spoken_word-module intent, constrained to conservative dysfluency cleanup;
- strong prompt guardrails preserving meaning/intent/voice/named entities/domain terms and substantive content;
- explicit guardrails against summarization, style rewriting, grammar-only cleanup, punctuation-only cleanup, invention, and meaning-changing rewrites;
- proposal generation through `internal/framework/proposal_generation` and `contracts.StructuredLLMClient`;
- scheduler-aware proposal calls through existing `contracts.LLMScheduler` hooks;
- replacement policy `require_unique` (matching Python spoken_word behavior);
- validator chain integration using existing deterministic + LLM-backed validators, including strong semantic guardrails (`spoken_word_review`, `meaning_reversal_review`);
- spoken_word confidence threshold enforcement through existing validator/config infrastructure;
- protected-term guardrails for non-glossary modules remain active and are exercised through the spoken_word path;
- module-level reporting and diagnostics capture through existing runner/reporting paths.
## Reports and diagnostics (implemented)
Current per-run artifacts include:
- `source-transcript.json`
- `source-transcript-parsed.json`
- `normalized-transcript.json`
- `normalization-summary.json`
- `chunking-summary.json`
- `utilization-diagnostics.json`
- `correction-ledger.json`
- `invocation.json`
- `effective-config.json` (redacted credentials)
- `report.json`
- `error.log` on failure
`--report-json` writes a separate report file when requested.
Current process reports include diagnostics metadata references for:
- diagnostics directory path;
- source transcript artifact path;
- parsed source transcript artifact path;
- normalized transcript artifact path;
- normalization summary artifact path;
- chunking summary artifact path;
- utilization diagnostics artifact path;
- correction ledger artifact path;
- invocation metadata artifact path;
- redacted effective-config artifact path;
- error-log artifact path on failure.
Current process reports also include:
- module-level results (when runner modules execute), including applied/skipped proposal changes;
- run-level module summary totals and failed module instance metadata.
- module-level validator decisions and validator rejections.
- optional decision-level diagnostic artifact paths for validator LLM interactions when available.
- stable validator keys in `validator_name` fields for validator decisions/rejections.
- explicit report metadata:
- report schema name;
- report schema version;
- selected output schema;
- config file version when config file input is used.
- review/observability artifacts:
- run-level and module-level utilization/timing summaries;
- flattened correction ledger entries for applied/rejected/skipped/failed correction dispositions.
Utilization diagnostics collection:
- collection is performed in the runner path via lightweight instrumentation around LLM scheduler and structured-client execution (`internal/framework/runner`);
- instrumentation is observational only and does not change scheduler acquisition/release semantics or module execution order;
- serialized artifact: `utilization-diagnostics.json`.
Utilization diagnostics high-level shape:
- `effective_concurrency`:
- `total_llm`, `proposal_llm`, `validation_llm`;
- `run_timing`:
- `run_wall_time_ms`;
- `scheduler_queue_wait_ms`;
- `llm_execution_time_ms`;
- `deterministic_validation_time_ms`;
- `max_in_flight_llm_calls`;
- `average_in_flight_llm_calls`;
- `llm_calls`:
- `total_proposal_calls`;
- `total_validation_calls`;
- `modules`:
- per-module key/instance timing summaries including module wall time and per-module call counts;
- `validators`:
- per-validator summaries keyed by stable validator key with elapsed time and LLM-backed marker.
Correction ledger construction:
- ledger entries are built from runner module results in the CLI report/diagnostics path (`internal/cli/review_artifacts.go`);
- serialized artifact: `correction-ledger.json`;
- one flattened record per applied/validator-rejected/application-skipped outcome where data is available, plus module-failed records for failed module instances.
Correction ledger high-level shape:
- run/module/proposal identity:
- `run_id`, `module_key`, `module_instance`, `proposal_index`, `segment_id`;
- correction payload:
- `original_text`, `proposed_corrected_text`, `applied_corrected_text` (when applied), `replacement_policy`;
- disposition:
- `disposition` in `{applied,rejected,skipped,failed}`;
- `disposition_reason_code`, `disposition_message`;
- validator decision snapshots:
- `deterministic_validator_decisions[]`;
- `llm_validator_decisions[]`;
- each decision uses stable validator keys and reason codes.
Identity and metadata boundaries:
- stable module keys/instance names and stable validator keys are included directly in ledger records;
- prompt metadata and structured response schema metadata remain in LLM interaction diagnostics payloads and are not duplicated into every ledger row;
- reports reference artifact paths for utilization and ledger files through diagnostics metadata.
Redaction and retention:
- secret redaction guarantees continue to apply to diagnostics/report artifacts;
- utilization and ledger artifacts are emitted within the existing run-directory retention model (`auto|always|never`) and are retained/removed with the run directory.
Current report schema metadata values:
- `report_metadata.report_schema_name = "audita-process-report"`
- `report_metadata.report_schema_version = "v1"`
Retention modes implemented in `ApplyRetention`:
- `always`: keep all run directories.
- `never`: keep successful run directories.
- `auto`: keep failed runs and successful runs with skipped corrections.
- failed runs are always retained.
Current runtime note:
- default non-explicit runs usually have no module-level skipped corrections, so `auto` commonly removes clean successful run directories.
- explicit grammar/glossary/homophones/spoken_word runs can produce validator rejections and application skips, which are reflected in reports and retention input.
## Current tests and quality posture
Implemented tests currently cover:
- CLI argument handling and behavior (`internal/cli/run_test.go`)
- subprocess stdout/stderr and exit-code behavior (`cmd/audita/main_integration_test.go`)
- config/env/override validation (`internal/core/config/*_test.go`)
- transcript and glossary schema validation (`internal/core/schema/*_test.go`)
- deterministic normalization (`internal/core/normalization/*_test.go`)
- deterministic chunking and summaries (`internal/core/chunking/*_test.go`)
- proposal preview/apply semantics (`internal/framework/proposals/*_test.go`)
- contracts/foundation composition tests (`internal/framework/contracts/*_test.go`)
- runner sequencing and failure behavior with deterministic fake modules (`internal/framework/runner/*_test.go`)
- CLI runner integration through injected fake module factories (`internal/cli/run_test.go`)
- validator models, cardinality enforcement, and deterministic validators (`internal/framework/validators/*_test.go`)
- LLM-backed validator batching, prompt builders, structured-response safety, scheduler hooks, and diagnostics redaction (`internal/framework/validators/*_test.go`, `internal/framework/runner/*_test.go`)
- shared proposal-generation request/response parsing, deterministic indexing, scheduler hooks, and diagnostics redaction (`internal/framework/proposal_generation/*_test.go`, `internal/framework/runner/*_test.go`)
- production module-registry known-key recognition and unsupported/internal-registry error behavior (`internal/framework/modules/*_test.go`, `internal/cli/run_test.go`)
- production grammar module prompt constraints, proposal mapping, validator-chain behavior, confidence-threshold enforcement, diagnostics redaction, and explicit CLI/runtime integration (`internal/modules/grammar/*_test.go`, `internal/cli/run_test.go`, `internal/framework/runner/*_test.go`)
- production glossary module prompt constraints, proposal mapping, validator-chain behavior, confidence-threshold enforcement, diagnostics redaction, repeated-stage behavior, and explicit CLI/runtime integration (`internal/modules/glossary/*_test.go`, `internal/cli/run_test.go`, `internal/framework/runner/*_test.go`)
- production homophones module prompt constraints, proposal mapping, validator-chain behavior, confidence-threshold enforcement, diagnostics redaction, protected-term behavior, and explicit CLI/runtime integration (`internal/modules/homophones/*_test.go`, `internal/cli/run_test.go`, `internal/framework/runner/*_test.go`)
- production spoken_word module prompt constraints, proposal mapping, validator-chain behavior, semantic guardrail behavior, confidence-threshold enforcement, diagnostics redaction, protected-term behavior, and explicit CLI/runtime integration (`internal/modules/spoken_word/*_test.go`, `internal/cli/run_test.go`, `internal/framework/runner/*_test.go`)
- glossary-derived protected-term extraction and stable behavior (`internal/framework/validators/protected_terms_test.go`)
- default full-pipeline runtime shape and ordering (`internal/cli/run_test.go`, `cmd/audita/main_integration_test.go`, `internal/cli/parity_test.go`)
- subprocess operational hardening behavior including large-input, failure-mode, timeout/cancellation, backend-failure, and partial-progress paths (`cmd/audita/main_integration_test.go`)
- report/diagnostics redaction and artifact-shape behavior across success and failure paths (`internal/cli/run_test.go`, `cmd/audita/main_integration_test.go`)
- curated release-fixture and idempotence-oriented readiness checks using fake structured LLM responses (`internal/cli/release_fixtures_test.go`, `internal/cli/testdata/release`)
## Operational hardening status
The runtime now includes hardened subprocess behavior for parent-process callers:
- deterministic success/failure exit codes;
- strict stdout/stderr separation suitable for machine orchestration;
- failure stderr summaries that include diagnostics location when available;
- retained failure diagnostics (`report.json`, `error.log`, and artifacts written before failure);
- deterministic timeout/cancellation behavior in tests;
- redaction coverage for API keys/secrets across reports, diagnostics artifacts, and surfaced errors.
- stable output routing behavior:
- with `--output`, stdout remains empty on success;
- without `--output`, stdout contains only transcript JSON in the selected output schema;
- `--report-json` writes report data to file only (never stdout).
Operational caller guidance is documented in [`docs/subprocess-operations.md`](docs/subprocess-operations.md).
## Final status
- Audita's default full module-sequence runtime is implemented and tested.
- Parity fixtures and operational hardening coverage are in place.
- Historical migration context is documented in [`docs/migration-from-python.md`](docs/migration-from-python.md).

View File

@@ -1,98 +0,0 @@
# Audita Diagnostics
This document describes the run-directory diagnostics artifacts produced by `audita process`.
## Purpose
Diagnostics provide machine-readable run context and execution artifacts for:
- failure debugging;
- validator/correction review;
- post-run performance analysis.
Diagnostics are written under the configured work directory (`--work-dir`) when run-directory initialization succeeds.
## Core artifacts
Typical artifacts in each run directory:
- `source-transcript.json`
- `source-transcript-parsed.json`
- `normalized-transcript.json`
- `normalization-summary.json`
- `chunking-summary.json`
- `invocation.json`
- `effective-config.json` (redacted)
- module/validator LLM interaction artifacts
- `report.json`
- `error.log` on failure
## Utilization diagnostics artifact
Artifact:
- `utilization-diagnostics.json`
High-level fields:
- `effective_concurrency`:
- total/proposal/validation LLM concurrency limits in effect.
- `run_timing`:
- run wall time;
- scheduler queue wait time;
- LLM execution time;
- deterministic validator time;
- max/average in-flight LLM calls.
- `llm_calls`:
- total proposal and validation LLM call counts.
- `modules`:
- module-level timing summaries.
- `validators`:
- per-validator timing summaries keyed by stable validator key.
## Correction ledger artifact
Artifact:
- `correction-ledger.json`
Ledger records are flattened review entries derived from module results and include:
- module/proposal identity (`module_key`, `module_instance`, `proposal_index`, `segment_id`);
- correction text fields and replacement policy when available;
- disposition:
- `applied`
- `rejected`
- `skipped`
- `failed`
- stable reason codes/messages;
- deterministic and LLM validator decision snapshots using stable validator keys.
Validator rejection and proposal-application skip are distinct dispositions.
## Report references
`report.json` and optional `--report-json` output include diagnostics metadata paths for:
- utilization diagnostics artifact;
- correction ledger artifact;
- existing transcript/normalization/chunking/invocation/effective-config artifacts.
## Retention behavior
Run-directory retention follows configured policy:
- `always`: keep all run directories;
- `never`: keep successful run directories;
- `auto`: keep failed runs and successful runs with skipped/rejected corrections.
## Redaction guarantees
API keys and other configured secrets are redacted from:
- `effective-config.json`;
- LLM interaction diagnostics artifacts;
- reports and surfaced errors.
## Debugging guide
When debugging:
- slow runs:
- inspect `utilization-diagnostics.json` (`run_timing`, `modules`, `validators`, in-flight metrics).
- validator rejections:
- inspect `correction-ledger.json` rejected entries and matching validator decisions;
- inspect validator response diagnostics payloads.
- application skips:
- inspect `correction-ledger.json` skipped entries and skip reason codes;
- compare with validator decisions to distinguish validation rejection vs apply-time skip.

View File

@@ -1,88 +0,0 @@
# Audita Output Schemas
This document describes the built-in transcript output schema registry used by `audita process`.
## Supported schema names
### `bare-segments`
Status:
- implemented
- default output schema
Shape:
- top-level JSON array of transcript segments
Segment fields:
- `id`
- `speaker`
- `start`
- `end`
- `text`
- optional `categories`
Compatibility:
- this preserves the long-standing output shape used by existing consumers.
### `audita-v1`
Status:
- implemented
Shape:
- top-level JSON object:
- `schema`: `"audita-v1"`
- `version`: `"v1"`
- `segments`: transcript segment array
Segment fields inside `segments` match `bare-segments` segment fields.
Compatibility:
- this is the Audita-native object format with explicit schema/version metadata.
### `seriatim-intermediate`
Status:
- deferred / not implemented
Current behavior:
- selecting `seriatim-intermediate` fails clearly as an unsupported output schema.
Reason:
- a concrete, repository-backed contract for this schema has not been finalized yet.
## Selection
Choose output schema with CLI:
```sh
audita process <transcript.json> --glossary <glossary.yaml> --output-schema audita-v1
```
Or in file config:
```yaml
version: 1
output:
schema: audita-v1
```
Precedence remains:
1. defaults
2. file config
3. environment overrides
4. CLI overrides
`--output-schema` overrides `output.schema` when both are supplied.
## Output routing behavior
- With `--output`, transcript JSON is written to file using the selected schema and stdout stays empty on success.
- Without `--output`, stdout contains transcript JSON only, using the selected schema.
- `--report-json` writes report JSON to file and does not write report payloads to stdout.
## Backward-compatibility expectations
- default schema stays `bare-segments` for compatibility unless explicitly changed in a future breaking release;
- supported schema names are treated as stable public contract values;
- unsupported schema names fail before output write.

View File

@@ -1,118 +0,0 @@
# Audita Prompts
This document describes Audita's built-in embedded prompt assets and prompt registry behavior.
## Why embedded prompt assets
Audita embeds production prompt text into the binary so runtime behavior is:
- deterministic;
- auditable;
- dependency-light;
- not dependent on external prompt files at execution time.
Prompt text is authored as Markdown assets and rendered by Go code using typed template data.
## Built-in prompt registry
The prompt registry lives in `internal/prompts` and is responsible for:
- loading embedded prompt assets;
- registering stable prompt IDs and versions;
- recording prompt source metadata;
- computing deterministic SHA-256 source hashes;
- rendering system/user prompts with strict missing-key failures.
Current prompt source behavior:
- built-in embedded prompts only (`prompt_source = builtin`).
- filesystem prompt overrides are not supported.
## Built-in prompt IDs
Module proposal prompts:
- `modules.glossary.proposal`
- `modules.homophones.proposal`
- `modules.spoken_word.proposal`
- `modules.grammar.proposal`
LLM-backed validator prompts:
- `validators.spoken_form_plausibility`
- `validators.meaning_reversal_review`
- `validators.editorial_review`
- `validators.grammar_review`
- `validators.spoken_word_review`
## Prompt version semantics
Current built-in prompt version value is `v1`.
Version is a stable metadata identifier for diagnostics and debugging. It is not a dynamic prompt-selection mechanism.
## Prompt hash semantics
Each registered prompt includes a deterministic SHA-256 hash of embedded source text.
Hash purpose:
- identify exact prompt source used in a run;
- support diagnostics reproducibility and change auditing.
Current hash scope:
- source prompt text (system + user assets for a registered prompt), not a runtime secret-bearing payload.
## Template rendering behavior
Prompt rendering uses Go `text/template` with typed template data from module/validator builders.
Missing-key behavior:
- rendering uses missing-key errors;
- missing/renamed template fields fail quickly instead of silently producing incomplete prompts.
Go code still owns:
- structured request/response models;
- response schema selection;
- transcript/glossary/payload formatting;
- module and validator selection;
- diagnostics wiring.
## Shared prompt hardening policy
A shared hardening fragment is embedded once and included in every module proposal prompt and every LLM-validator prompt.
Hardening policy includes:
- transcript text is untrusted data;
- glossary entries and transcript descriptions are reference data, not instructions;
- instructions found inside transcript text must not be obeyed;
- model must perform only the requested correction/validation task;
- no invention of facts, names, events, motivations, speaker intent, or corrections;
- transcript remains the source of truth.
## Transcript description behavior
Transcript description remains background-only prompt context:
- it may help interpret ambiguous terms;
- it is explicitly non-authoritative and must not override transcript content;
- empty descriptions do not render awkward blank context sections.
Generated transcript descriptions are not implemented in this workstream.
## Diagnostics and report metadata boundaries
Current metadata flow:
- proposal-generation diagnostics request metadata includes prompt metadata;
- LLM-validator diagnostics request metadata includes prompt metadata.
Prompt metadata fields used in diagnostics:
- `prompt_id`
- `prompt_version`
- `prompt_source`
- `embedded_path`
- `sha256`
Current boundary:
- detailed prompt metadata is diagnostics-first;
- broad report-level prompt registries/ledgers are deferred.
## 1.0 boundary
Not implemented for 1.0 in this workstream:
- filesystem prompt overrides;
- user-configurable prompt selection;
- external prompt directories.

View File

@@ -1,165 +0,0 @@
# Audita Public Contract
This document defines stability expectations for Audita's external process and data interfaces.
## Scope
This contract covers:
- CLI invocation and behavior
- versioned config file behavior
- transcript/glossary input forms
- transcript output schema selection
- process report schema metadata
- stable validator key identifiers in report/diagnostics records
- prompt metadata identifiers in diagnostics
- diagnostics directory behavior
- utilization diagnostics and correction-ledger artifact presence/pathing in diagnostics metadata
- stdout/stderr and exit-code behavior
- secret redaction guarantees
- compatibility and deprecation policy
## CLI stability expectations
Stable commands:
- `audita process`
- `audita config validate`
- `audita config print-effective`
For `audita process`, stable high-value flags include:
- `--config`
- `--glossary`
- `--output`
- `--report-json`
- `--modules`
- `--output-schema`
Compatibility flags and lower-level tuning flags remain available; they may be narrowed over time with explicit compatibility notes.
## Config file stability expectations
Supported file format:
- YAML
- strict unknown-field rejection
- explicit `version`
Supported version:
- `version: 1`
Precedence for `audita process`:
1. built-in defaults
2. file config
3. environment overrides
4. CLI overrides
Config source behavior:
- `--config <path>`: missing path is a clear failure
- `AUDITA_CONFIG`: missing path is a clear failure
- defaults `/usr/local/etc/audita/config.yml`, then `/etc/audita/config.yml`: both missing is non-fatal
## Supported transcript input forms
Audita accepts transcript JSON as either:
- a top-level array of segments
- an object with a `segments` array
Segments must satisfy the schema and validation rules enforced by `internal/core/schema`.
## Supported glossary input form
Audita accepts glossary YAML with a top-level `glossary` entry list and validates required fields per entry.
## Supported output schema names
Built-in output schema registry supports:
- `bare-segments` (default)
- `audita-v1`
`seriatim-intermediate` is planned but not implemented.
Unknown output schema names fail clearly.
## Report schema/versioning expectations
Process report payloads include `report_metadata` with:
- `report_schema_name`
- `report_schema_version`
- `output_schema`
- `config_version` when file config is used
Current values:
- `report_schema_name`: `audita-process-report`
- `report_schema_version`: `v1`
`--report-json` output and diagnostics run-dir `report.json` use the same report schema metadata.
Validator decision/rejection records in reports use stable validator keys in `validator_name`.
Report diagnostics metadata includes artifact-path fields for utilization diagnostics and correction ledger when diagnostics initialization succeeds.
## Diagnostics directory behavior
When diagnostics directory creation succeeds, Audita writes run artifacts including:
- invocation metadata
- redacted effective config
- transcript/normalization/chunking artifacts
- utilization diagnostics (`utilization-diagnostics.json`)
- correction ledger (`correction-ledger.json`)
- report and failure error log (when applicable)
- module/LLM diagnostics artifacts as available
Retention behavior is controlled by configured retention mode; failed runs are retained.
Diagnostics metadata for LLM interactions may include semi-public prompt identifiers:
- `prompt_id`
- `prompt_version`
- `prompt_source`
- `embedded_path`
- `sha256`
These are diagnostic identifiers, not user-facing prompt override controls.
## Stdout/stderr behavior
Success behavior:
- with `--output`, stdout is empty
- without `--output`, stdout contains only transcript JSON in selected output schema
- report JSON is not written to stdout
Failure behavior:
- stderr contains human-readable error summary
- nonzero exit
- diagnostics path is printed when available
## Exit-code behavior
- `0`: success
- nonzero: failure
Treat any nonzero exit as a failed invocation.
## Secret redaction guarantees
Audita redacts API keys and authorization secrets from:
- effective config outputs (`audita config print-effective`, diagnostics effective-config artifact)
- report artifacts
- LLM diagnostics artifacts
- surfaced request/response error messages
Config files should reference secrets via environment variable names (`api_key_env`) rather than embedding secret values.
## Compatibility and deprecation policy
- Existing stable schema names, report metadata keys, and top-level command behavior are treated as public contract.
- Compatibility inputs (legacy flags/env aliases) may remain during transition windows.
- Any planned removal or behavior change should include clear compatibility notes and migration guidance.
## Breaking changes after 1.0
After 1.0, breaking changes include, for example:
- changing default success/failure exit-code semantics
- changing stdout/stderr routing semantics
- silently changing default output schema shape
- removing supported output schema names without compatibility strategy
- changing report schema fields or meanings incompatibly
- changing config version semantics incompatibly without version bump
Additive fields, additive diagnostics, and new optional schema names are generally non-breaking when existing behavior remains intact.

View File

@@ -1,90 +0,0 @@
# Structured LLM Architecture
## Purpose
This document describes Audita's structured LLM runtime boundary and adapter behavior.
## Why Audita owns the adapter
Audita owns a small structured LLM adapter so that core runtime behavior is controlled inside the repository:
- request construction and schema handling are explicit and testable;
- retries, timeouts, cancellation, and error redaction are consistent across modules and validators;
- provider SDK types are not exposed outside the adapter boundary;
- dependency weight and transitive provider-specific behavior are reduced.
At runtime, the rest of Audita depends only on the internal contract:
- `StructuredLLMClient`
- `CompleteStructured(ctx, req, out)`
## OpenAI-compatible request shape
At a conceptual level, Audita sends chat completion requests with:
- `model`
- `messages` (role/content pairs)
- `response_format`:
- `type = "json_schema"`
- `json_schema.name` (stable schema name)
- `json_schema.strict = true`
- `json_schema.schema` (registered JSON Schema payload)
The adapter uses OpenAI-compatible `POST {base_url}/chat/completions` over `net/http`.
## Structured response schema registry
Structured response schemas are registered in `internal/framework/responseschema` with stable metadata:
- schema key
- schema ID
- schema version
- schema name (OpenAI-compatible `response_format` name)
- raw JSON Schema payload
- SHA-256 hash
Current schemas:
- `correction_set`:
- id `audita.correction_set`
- version `v1`
- name `audita_correction_set_v1`
- `validator_decision_set`:
- id `audita.validator_decision_set`
- version `v1`
- name `audita_validator_decision_set_v1`
## Provider compatibility assumptions
Audita assumes an OpenAI-compatible chat-completions endpoint that:
- accepts message arrays with model selection;
- accepts `response_format.type = json_schema`;
- returns a completion with assistant message content and optional usage metadata.
Provider-specific differences are expected in strictness and error payload shapes, so the adapter treats provider output as untrusted until locally decoded.
## Local decode and validation remain mandatory
Provider-level structured output is a transport guardrail, not final validation.
After receiving a response, Audita still:
- decodes assistant content into typed request-specific structs;
- validates proposal and validator payload invariants locally;
- enforces deterministic validator/cardinality rules before any transcript application.
This protects runtime correctness even when provider responses are malformed, partial, or semantically inconsistent.
## Diagnostics and redaction
When structured schemas are used, diagnostics metadata records:
- schema ID
- schema version
- schema name
- schema hash
Diagnostics and surfaced errors preserve secret redaction:
- API keys and bearer tokens are redacted from request/response/error artifacts;
- redaction is applied before diagnostic files are written.
## Runtime behavior guarantees
The structured LLM path preserves existing runtime guarantees:
- bounded LLM call execution through schedulers;
- context-aware cancellation and timeout propagation;
- retry behavior for transient failures and retryable malformed structured responses;
- deterministic module/chunk/proposal/validator behavior outside provider nondeterminism.

View File

@@ -1,148 +0,0 @@
# Audita Validators
This document describes Audita's built-in validator registry and module validator chains.
For LLM-backed validator prompt asset details, see [`docs/prompts.md`](prompts.md).
## Package ownership
Built-in validator construction is package-owned under `internal/validators/<validator_key>`:
- `internal/validators/confidence_threshold`
- `internal/validators/original_text_presence`
- `internal/validators/non_empty_corrected_text`
- `internal/validators/no_effect`
- `internal/validators/protected_terms`
- `internal/validators/spoken_form_plausibility`
- `internal/validators/meaning_reversal_review`
- `internal/validators/editorial_review`
Registry and chain wiring stay in:
- `internal/validators/registry.go`
- `internal/validators/chains.go`
Shared validator runtime mechanics stay in `internal/framework/validators`:
- request/result/decision models
- decision cardinality helpers
- protected vocabulary helpers
- shared LLM validator runtime, batching, and diagnostics helpers
Execution classification metadata is defined in `internal/validators/metadata`:
- `deterministic`
- `llm_backed`
Runner ordering uses this metadata so deterministic validators run before LLM-backed validators without concrete framework type assertions.
## Scope
Validator chains are built-in runtime behavior.
Current 1.0 boundary:
- built-in validator keys and built-in module chains are stable runtime identifiers;
- thresholds and batching knobs remain configurable where already supported;
- arbitrary user-defined validator chains are deferred.
## Built-in validator keys
### Deterministic validators
- `confidence_threshold`
- checks proposal confidence against module-specific configured threshold.
- `original_text_presence`
- ensures target segment exists and `original_text` exists in current working segment text.
- `non_empty_corrected_text`
- rejects blank/whitespace-only `corrected_text`.
- `no_effect`
- rejects proposals where `original_text == corrected_text`.
- `protected_terms`
- protects glossary-derived terms from unsafe mutations in non-glossary modules.
- glossary stages use glossary-specific protection logic but still report this same stable key.
### LLM-backed validators
- `spoken_form_plausibility`
- checks whether proposed spoken-form change remains plausible in transcript context.
- `meaning_reversal_review`
- checks for likely meaning reversal or semantic contradiction.
- `editorial_review`
- performs conservative editorial safety review.
## Built-in module chains
Current built-in chains resolved from `internal/validators/chains.go`:
- `glossary`
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `spoken_form_plausibility`
- `meaning_reversal_review`
- `homophones`
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `spoken_form_plausibility`
- `meaning_reversal_review`
- `spoken_word`
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `editorial_review`
- `meaning_reversal_review`
- `grammar`
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `editorial_review`
- `meaning_reversal_review`
## Protected terms construction
`protected_terms` has explicit constructors:
- general constructor used by non-glossary modules through the built-in registry
- glossary-stage constructor used by glossary chain resolution
Both variants preserve existing behavior and report the stable key `protected_terms`.
## Execution semantics
- modules execute serially;
- section proposal work can run concurrently within a module;
- deterministic validators run before LLM-backed validators;
- malformed/missing/duplicate/unknown LLM validator decisions fail safely;
- approved proposals are applied once per module after section work settles.
## Validator rejections vs proposal-application skips
- validator rejection:
- proposal is denied by validator-chain review and appears in validator rejection reporting with validator key and reason code.
- proposal-application skip:
- proposal passed validators but could not be applied under replacement-policy semantics (for example no matching span at apply time).
These are separate outcomes and are reported separately.
## Reporting and diagnostics identity
- report validator decision/rejection entries use stable validator keys in `validator_name`.
- validator LLM diagnostics include validator identity in interaction metadata and structured response schema metadata.
- correction ledger entries include deterministic and LLM validator decision snapshots keyed by the same stable validator keys, and keep validator rejection distinct from application-level skip.
Prompt assets are unchanged by the validator package-ownership refactor and remain built-in under `internal/prompts`.
## Configurable knobs that remain supported
- per-module confidence thresholds (`thresholds.*` / equivalent env+CLI overrides)
- validation batching limits (`validation_max_prompt_tokens` / equivalent env+CLI overrides)
- validation LLM model/base URL/timeout/retries/concurrency settings
These tune validator behavior without exposing arbitrary user-defined chains.

186
docs/cli.md Normal file
View File

@@ -0,0 +1,186 @@
# Audita CLI Reference
## Shortest Useful Command
```sh
audita process <transcript.json> --glossary <glossary.yaml> --output <corrected.json>
```
This command validates input files, runs the configured correction pipeline, and writes corrected transcript JSON.
## Command Overview
- `audita process`: process one transcript JSON file.
- `audita config validate`: validate a versioned YAML config file.
- `audita config print-effective`: print redacted effective config JSON.
General help:
```sh
audita --help
audita process --help
audita config --help
```
## `process`
Usage:
```sh
audita process <transcript.json> [flags]
```
Input requirements:
- exactly one transcript JSON positional argument is required;
- `--glossary <path>` is required.
Config path selection for `process`:
1. `--config <path>`
2. `AUDITA_CONFIG`
3. `/usr/local/etc/audita/config.yml` (if present)
4. `/etc/audita/config.yml` (if present)
For precedence and full config schema, see [`docs/config.md`](config.md).
### `process` Flag Reference
Core I/O flags:
- `--config <path>`: path to versioned YAML config file.
- `--glossary <path>`: glossary YAML input path (required).
- `--output <path>`: corrected transcript JSON output file path.
- `--report-json <path>`: machine-readable report JSON output path.
- `--output-schema <key>`: output schema key (`bare-segments` or `audita-v1`).
- `--modules <csv>`: comma-separated module sequence override.
Primary LLM flags:
- `--llm-api-key <value>`: primary LLM API key.
- `--model <name>`: primary LLM model name.
- `--base-url <url>`: primary OpenAI-compatible base URL.
- `--llm-timeout-seconds <int>`: primary timeout in seconds.
- `--max-retries <int>`: primary structured-output retries.
Validation LLM flags:
- `--validation-llm-api-key <value>`: validation LLM API key.
- `--validation-model <name>`: validation LLM model name.
- `--validation-base-url <url>`: validation OpenAI-compatible base URL.
- `--validation-llm-timeout-seconds <int>`: validation timeout in seconds.
- `--validation-max-retries <int>`: validation structured-output retries.
- `--validation-max-prompt-tokens <int>`: validation max prompt tokens.
Concurrency flags:
- `--total-llm-concurrency <int>`: total concurrent proposal+validation LLM calls.
- `--proposal-llm-concurrency <int>`: concurrent proposal-generation LLM calls.
- `--validation-llm-concurrency <int>`: concurrent validation LLM calls.
- `--llm-concurrency <int>`: alias for `--total-llm-concurrency`.
Chunking and normalization flags:
- `--target-sections <int>`: target number of transcript sections.
- `--max-section-tokens <int>`: maximum section tokens.
- `--min-section-tokens <int>`: minimum section tokens.
- `--normalize-max-segment-gap <float>`: maximum same-speaker merge gap.
- `--normalize-ellipsis-gap <float>`: gap threshold for ellipsis insertion.
- `--normalize-max-segment-duration <float>`: maximum merged segment duration.
- `--normalize-max-segment-tokens <int>`: maximum merged segment token estimate.
Threshold flags:
- `--glossary-confidence-threshold <float>`
- `--homophones-confidence-threshold <float>`
- `--spoken-word-confidence-threshold <float>`
- `--grammar-confidence-threshold <float>`
Context and diagnostics flags:
- `--transcript-description <text>`: background context for prompts; does not override transcript content.
- `--work-dir <path>`: per-run diagnostics work directory.
- `--work-dir-retention <auto|always|never>`: run-directory retention policy.
### `process` Output and Exit Behavior
- With `--output`: stdout is expected to be empty on success.
- Without `--output`: stdout contains transcript JSON only on success.
- `--report-json` writes a file and is never printed to stdout.
- Stderr is human-readable diagnostics/errors.
- On failures after diagnostics initialization, stderr includes the diagnostics directory path.
Exit behavior:
- `0`: success.
- `1`: runtime failure during processing/reporting/output paths.
- `2`: CLI usage or configuration input error.
Integration references:
- subprocess contract: [`docs/integrations/subprocess.md`](integrations/subprocess.md)
- transcript/glossary file contract: [`docs/integrations/transcript-glossary-files.md`](integrations/transcript-glossary-files.md)
### `process` Examples
Write corrected transcript to a file:
```sh
audita process transcript.json \
--glossary glossary.yaml \
--output corrected.json
```
Emit transcript JSON to stdout:
```sh
audita process transcript.json --glossary glossary.yaml
```
Use explicit config and write report JSON:
```sh
audita process transcript.json \
--glossary glossary.yaml \
--config audita.yml \
--output corrected.json \
--report-json report.json
```
Override the module sequence:
```sh
audita process transcript.json \
--glossary glossary.yaml \
--modules glossary,homophones,grammar \
--output corrected.json
```
## `config validate`
Usage:
```sh
audita config validate --config <path>
```
Behavior:
- validates defaults merged with file config;
- does not apply environment overrides;
- prints `config is valid` on success.
Errors:
- `--config` is required;
- positional arguments are rejected;
- validation failures are printed to stderr.
## `config print-effective`
Usage:
```sh
audita config print-effective [--config <path>]
```
Config path selection:
1. `--config <path>` when provided
2. `AUDITA_CONFIG`
3. `/usr/local/etc/audita/config.yml` (if present)
4. `/etc/audita/config.yml` (if present)
Behavior:
- merges defaults, optional config file, and environment overrides;
- prints redacted JSON to stdout.
Errors:
- positional arguments are rejected;
- resolution or parse failures are printed to stderr.

238
docs/config.md Normal file
View File

@@ -0,0 +1,238 @@
# Audita Configuration
## Scope
This is the canonical configuration reference for Audita.
It documents:
- config path resolution;
- effective precedence across defaults, file config, environment, and CLI;
- supported `version: 1` YAML schema;
- environment overrides;
- CLI override relationship;
- validation and secrets behavior.
For CLI command syntax, see [`docs/cli.md`](cli.md).
For OpenAI-compatible endpoint behavior, see [`docs/integrations/openai-compatible-llm.md`](integrations/openai-compatible-llm.md).
For transcript/glossary input file contracts, see [`docs/integrations/transcript-glossary-files.md`](integrations/transcript-glossary-files.md).
## Loading Model
Path resolution for `audita process` and `audita config print-effective`:
1. `--config <path>`
2. `AUDITA_CONFIG`
3. `/usr/local/etc/audita/config.yml` (if present)
4. `/etc/audita/config.yml` (if present)
Missing explicit path behavior:
- missing `--config` target is an error;
- missing `AUDITA_CONFIG` target is an error.
Missing default-path files are non-fatal.
## Effective Precedence
`audita process`:
1. defaults
2. file config
3. environment overrides
4. CLI overrides
`audita config print-effective`:
1. defaults
2. file config
3. environment overrides
`audita config validate`:
1. defaults
2. file config
`config validate` is intentionally file-only (no environment overrides).
## Defaults
Current defaults:
- modules: `glossary,homophones,glossary,spoken_word,grammar`
- output schema: `bare-segments`
- primary model: `openrouter/google/gemma-4-31b-it`
- primary base URL: `https://openrouter.ai/api/v1`
- primary timeout: `600` seconds
- max retries: `3`
- total/proposal LLM concurrency: `1`
- validation max prompt tokens: `2048`
- max section tokens: `8192`
- min section tokens: `2048`
- confidence thresholds: `0.8`
- normalization max segment gap: `4.0`
- normalization ellipsis gap: `3.5`
- normalization max segment duration: `60.0`
- normalization max segment tokens: `2048`
- transcript description: empty
- work dir: `/tmp/audita`
- work dir retention: `auto`
## YAML Schema (`version: 1`)
Supported file version:
- `version: 1` (required)
Unknown YAML fields are rejected.
```yaml
version: 1
pipeline:
modules: [glossary, homophones, glossary, spoken_word, grammar]
output:
schema: bare-segments
llm:
proposal:
base_url: https://openrouter.ai/api/v1
model: openrouter/google/gemma-4-31b-it
api_key_env: AUDITA_LLM_API_KEY
timeout: 600s
max_retries: 3
validation:
base_url: https://openrouter.ai/api/v1
model: openrouter/google/gemma-4-31b-it
api_key_env: AUDITA_VALIDATION_LLM_API_KEY
timeout: 600
max_retries: 3
concurrency:
total_llm: 1
proposal_llm: 1
validation_llm: 1
chunking:
target_sections: 8
max_section_tokens: 8192
min_section_tokens: 2048
normalization:
max_segment_gap: 4s
ellipsis_gap: 3.5s
max_segment_duration: 60s
max_segment_tokens: 2048
thresholds:
glossary: 0.8
homophones: 0.8
spoken_word: 0.8
grammar: 0.8
context:
description: optional background context
diagnostics:
work_dir: /tmp/audita
retention: auto
```
Duration-parsing behavior:
- `llm.*.timeout`: integer seconds or duration string; duration strings must resolve to whole seconds.
- `normalization.*` duration-like fields: numeric seconds or duration string.
## Environment Overrides
Modules:
- `AUDITA_MODULES`
Config path:
- `AUDITA_CONFIG`
Primary LLM:
- `AUDITA_LLM_API_KEY` (falls back to `OPENROUTER_API_KEY` when unset)
- `AUDITA_MODEL`
- `AUDITA_BASE_URL`
- `AUDITA_LLM_TIMEOUT_SECONDS`
- `AUDITA_MAX_RETRIES`
Validation LLM:
- `AUDITA_VALIDATION_LLM_API_KEY`
- `AUDITA_VALIDATION_MODEL`
- `AUDITA_VALIDATION_BASE_URL`
- `AUDITA_VALIDATION_LLM_TIMEOUT_SECONDS`
- `AUDITA_VALIDATION_MAX_RETRIES`
- `AUDITA_VALIDATION_MAX_PROMPT_TOKENS`
Concurrency:
- `AUDITA_TOTAL_LLM_CONCURRENCY`
- `AUDITA_PROPOSAL_LLM_CONCURRENCY`
- `AUDITA_VALIDATION_LLM_CONCURRENCY`
- `AUDITA_LLM_CONCURRENCY` (legacy alias for total)
Chunking:
- `AUDITA_MAX_SECTION_TOKENS`
- `AUDITA_MIN_SECTION_TOKENS`
- `AUDITA_TARGET_SECTIONS`
Thresholds:
- `AUDITA_GLOSSARY_CONFIDENCE_THRESHOLD`
- `AUDITA_HOMOPHONES_CONFIDENCE_THRESHOLD`
- `AUDITA_SPOKEN_WORD_CONFIDENCE_THRESHOLD`
- `AUDITA_GRAMMAR_CONFIDENCE_THRESHOLD`
Normalization:
- `AUDITA_NORMALIZE_MAX_SEGMENT_GAP`
- `AUDITA_NORMALIZE_ELLIPSIS_GAP`
- `AUDITA_NORMALIZE_MAX_SEGMENT_DURATION`
- `AUDITA_NORMALIZE_MAX_SEGMENT_TOKENS`
Diagnostics:
- `AUDITA_WORK_DIR`
- `AUDITA_WORK_DIR_RETENTION` (`auto`, `always`, `never`)
Transcript description:
- no `AUDITA_*` environment variable is currently defined.
## CLI Override Relationship
CLI flags override file and environment values for `audita process`.
The CLI supports canonical total concurrency (`--total-llm-concurrency`) and legacy alias (`--llm-concurrency`):
- when both are provided at the same precedence layer, canonical total wins;
- if proposal concurrency is not explicitly set and total is set via environment or CLI, proposal concurrency inherits that total;
- validation concurrency inherits total only when validation concurrency is unset.
For full flag syntax, see [`docs/cli.md`](cli.md).
## Validation Rules
Validation includes:
- supported module keys only;
- supported output schema keys only (`bare-segments`, `audita-v1`);
- positive timeout/concurrency/token constraints;
- `proposal_llm <= total_llm` and `validation_llm <= total_llm` when validation is set;
- confidence thresholds in `[0.0, 1.0]`;
- transcript description length `<= 500` characters;
- non-empty work dir;
- work-dir retention in `auto|always|never`.
## Secrets
Recommended secret handling:
- use `llm.proposal.api_key_env` and `llm.validation.api_key_env` in file config;
- use `AUDITA_*_API_KEY` environment overrides or CLI key flags when needed.
`api_key_env` fields contain environment variable names, not secret values.
Redaction behavior:
- `audita config print-effective` redacts resolved API keys.
- diagnostics and report paths redact configured secret values.
## Examples
- Minimal config: [`examples/minimal-config.yml`](../examples/minimal-config.yml)
- Production-style config: [`examples/production-config.yml`](../examples/production-config.yml)
- Tiny transcript input: [`examples/tiny-transcript.json`](../examples/tiny-transcript.json)
- Tiny glossary input: [`examples/tiny-glossary.yaml`](../examples/tiny-glossary.yaml)
Validate the config examples:
```sh
audita config validate --config examples/minimal-config.yml
audita config validate --config examples/production-config.yml
```

View File

@@ -1,190 +0,0 @@
# Audita Configuration
This document describes Audita's versioned YAML config support and related commands.
## Purpose
Audita's config file provides a stable place for pipeline defaults and runtime tuning that would otherwise require many environment variables or CLI flags.
Use config files for baseline settings, then use environment variables and CLI flags for deployment and per-run overrides.
## Supported version
Current supported config version:
- `version: 1`
Rules:
- missing `version` fails validation;
- unknown versions fail validation;
- unknown fields fail validation (strict decoding).
## Config path resolution
For `audita process`, config path resolution is:
1. `--config <path>` if provided
2. `AUDITA_CONFIG` if set and `--config` is not provided
3. default `/usr/local/etc/audita/config.yml` if present
4. fallback default `/etc/audita/config.yml` if present
Missing-file behavior:
- missing `--config` path: hard failure;
- missing `AUDITA_CONFIG` path: hard failure;
- missing both default-path files: non-fatal, run continues.
## Precedence model
Effective config precedence is:
1. built-in defaults
2. file config
3. environment overrides
4. CLI overrides
## Supported YAML fields
```yaml
version: 1
pipeline:
modules: [glossary, homophones, glossary, spoken_word, grammar]
output:
schema: bare-segments
llm:
proposal:
base_url: https://openrouter.ai/api/v1
model: openrouter/google/gemma-4-31b-it
api_key_env: AUDITA_LLM_API_KEY
timeout: 120s
max_retries: 3
validation:
base_url: https://openrouter.ai/api/v1
model: openrouter/google/gemma-4-31b-it
api_key_env: AUDITA_VALIDATION_LLM_API_KEY
timeout: 120s
max_retries: 3
concurrency:
total_llm: 2
proposal_llm: 2
validation_llm: 1
chunking:
target_sections: 8
max_section_tokens: 8192
min_section_tokens: 2048
normalization:
max_segment_gap: 4s
ellipsis_gap: 3.5s
max_segment_duration: 60s
max_segment_tokens: 2048
thresholds:
glossary: 0.8
homophones: 0.8
spoken_word: 0.8
grammar: 0.8
context:
description: "optional transcript background context"
diagnostics:
work_dir: /tmp/audita
retention: auto
```
`context.description` provides background-only transcript context for prompts.
If both config and CLI provide a description, `--transcript-description` takes precedence.
`output.schema` supports the built-in output schema registry values:
- `bare-segments` (default)
- `audita-v1`
Unknown schema names fail clearly before transcript output is written.
Duration-like fields accept either:
- numeric seconds (for example `120`, `3.5`), or
- duration strings (for example `120s`, `2m`).
For LLM timeouts, duration strings must resolve to whole seconds.
## Secret handling
Use `api_key_env` for secrets:
- `llm.proposal.api_key_env`
- `llm.validation.api_key_env`
These fields must contain environment variable names, not secret values.
At runtime, Audita resolves those names from the process environment.
Redaction behavior:
- run diagnostics `effective-config.json` is redacted;
- `audita config print-effective` output is redacted;
- API keys are never emitted in plaintext by those outputs.
## Config commands
Validate a config file:
```sh
audita config validate --config ./audita.yml
```
Print redacted effective config:
```sh
audita config print-effective --config ./audita.yml
```
`print-effective` loads defaults, then file config, then environment overrides.
## Example: local OpenAI-compatible endpoint
```yaml
version: 1
llm:
proposal:
base_url: http://localhost:8000/v1
model: local/proposal-model
api_key_env: AUDITA_LLM_API_KEY
timeout: 90s
max_retries: 2
validation:
base_url: http://localhost:8000/v1
model: local/validation-model
api_key_env: AUDITA_VALIDATION_LLM_API_KEY
timeout: 90s
max_retries: 2
pipeline:
modules: [glossary, homophones, glossary, spoken_word, grammar]
diagnostics:
work_dir: /tmp/audita
retention: auto
```
## Compatibility notes
Existing environment variables and lower-level CLI flags remain available for compatibility.
Current guidance:
- prefer file config for baseline behavior;
- keep environment variables for secrets/deployment-specific overrides;
- use CLI flags for per-run overrides.
- validator chains are built-in and are not user-configurable in config.
- prompt source selection and filesystem prompt overrides are not config options.

View File

@@ -1,96 +0,0 @@
# Audita Subprocess Operations
This document describes how parent processes should invoke `audita process` safely in production orchestration.
## Recommended command form
Use explicit file outputs for orchestrated runs:
```sh
audita process <transcript.json> \
--transcript-description "Brief context that may help resolve ambiguous terms." \
--glossary <glossary.yaml> \
--output <output-transcript.json> \
--report-json <report.json>
```
Additional flags that may be situationally appropriate:
- `--config <path>` to select an explicit versioned config file.
- `--output-schema <bare-segments|audita-v1>` to select transcript output shape.
- `--work-dir <dir>` to control diagnostics location.
- `--work-dir-retention <always|auto|never>` to control retained run directories.
- `--total-llm-concurrency`, `--proposal-llm-concurrency`, and `--validation-llm-concurrency` when orchestration needs to set explicit LLM throughput controls.
- `--modules ...` only when intentionally overriding the default sequence.
For config-driven orchestration, validate config files in CI/preflight:
```sh
audita config validate --config <path>
```
## Stdout behavior
- With `--output`: stdout is expected to be empty on success.
- Without `--output`: stdout contains transcript JSON only on success.
- Report JSON is never written to stdout.
## Stderr behavior
- Success path should be quiet or minimal human-readable logs.
- Failure path writes concise human-readable errors.
- When a diagnostics run directory exists, failure stderr includes its path.
- Prompt/response diagnostic payloads are not streamed to stderr.
## Output file behavior
- `--output` writes transcript JSON in the selected output schema to the provided path.
- Output write failures return nonzero and surface actionable errors.
- The command does not silently ignore output write errors.
## Report JSON behavior
- `--report-json` writes a machine-readable process report to the requested path.
- Run-directory `report.json` is written independently under diagnostics.
- Best-effort failure reports are emitted when possible without masking the primary failure.
- Report write failures return nonzero with clear stderr messaging.
- Report diagnostics metadata references run-directory artifacts including utilization diagnostics and correction ledger paths when available.
## Diagnostics directory behavior
- Each run creates (when possible) a per-run diagnostics directory.
- Typical artifacts include transcript, normalization, chunking, invocation, effective config, LLM diagnostics, `utilization-diagnostics.json`, `correction-ledger.json`, `report.json`, and `error.log` on failure.
- Failed runs retain diagnostics.
- Under `auto` retention, successful runs with skipped/rejected corrections are retained; clean successful runs may be removed.
## Exit codes
- `0`: success.
- Nonzero: failure (input/schema/config/module/LLM/runtime/output/report/diagnostics errors).
Treat any nonzero as a failed subprocess invocation.
## Timeout and cancellation
- Runtime operations propagate context cancellation and request timeouts through LLM/scheduler paths.
- On cancellation or timeout, the process exits nonzero and should not hang.
- If diagnostics were initialized before failure, failure artifacts remain available for debugging.
## Secret redaction expectations
API keys and configured secret values are redacted from:
- reports (`--report-json` and run-dir `report.json`);
- diagnostics artifacts (including effective config and LLM interaction artifacts);
- surfaced adapter/runtime errors;
- test fixtures and regression outputs.
Parent-process logs should still avoid printing raw environment variables.
## Parent-process pipe guidance
To avoid deadlocks in orchestrators:
- always read both stdout and stderr concurrently when invoking as a subprocess;
- prefer file outputs (`--output`, `--report-json`) for machine workflows;
- treat stderr as human-readable diagnostics, not structured data;
- parse structured results from output/report files.
For Go callers, prefer `exec.CommandContext` with explicit timeout/cancellation and buffered/streamed readers for both pipes.

View File

@@ -0,0 +1,119 @@
# OpenAI-Compatible LLM Integration
## Scope
This document defines the external LLM endpoint contract Audita currently uses.
It covers:
- endpoint and auth expectations;
- structured request and response shape;
- retry and timeout behavior;
- diagnostics and secret redaction.
For user-facing CLI flags and config keys, see [`docs/cli.md`](../cli.md) and [`docs/config.md`](../config.md).
## Endpoint Contract
Audita sends HTTPS `POST` requests to:
- `<base_url>/chat/completions`
`base_url` comes from primary or validation LLM config and is required.
## Authentication Contract
When an API key is configured, Audita sends:
- `Authorization: Bearer <api_key>`
When no API key is configured, the `Authorization` header is omitted.
## Request Shape
Audita sends a chat-completions payload with:
- `model`;
- `messages` (role/content pairs);
- `response_format` using JSON Schema strict mode.
Representative shape:
```json
{
"model": "example-model",
"messages": [
{"role": "system", "content": "..."},
{"role": "user", "content": "..."}
],
"response_format": {
"type": "json_schema",
"json_schema": {
"name": "correction_set",
"strict": true,
"schema": {"type": "object"}
}
}
}
```
Behavioral requirements enforced by Audita:
- `model` must resolve to a non-empty value;
- each message must have non-empty `role` and `content`;
- `response_format.type` is always `json_schema`;
- `response_format.json_schema.name` and `schema` must be present;
- request schema JSON must be valid JSON.
## Response Handling Contract
Audita expects a successful JSON response with at least one choice and assistant content that can be interpreted as JSON.
Supported assistant content forms:
- string containing JSON;
- raw JSON value.
Audita then decodes the JSON against the expected structured output type.
Current structured schema identities used by Audita runtime:
- `correction_set`
- `validator_decision_set`
## Retries and Timeouts
Retry behavior:
- default max retries is `3` when unset;
- retries apply to retryable transport/decode/server-side errors;
- HTTP `429` and `5xx` responses are retryable;
- retry stops immediately when context is canceled or deadline expires.
Timeout behavior:
- request timeout is derived from configured LLM timeout settings;
- timeout/cancellation propagate through HTTP requests and return nonzero process failures.
## Error Behavior
Non-2xx responses fail the request.
Error message extraction behavior:
- if provider JSON includes `error.message`, Audita surfaces that message;
- else if provider JSON includes top-level `message`, Audita surfaces that;
- otherwise Audita surfaces status code plus response body text.
Malformed or incompatible structured responses fail safely and are surfaced as runtime errors or validator/proposal warnings depending on call site.
## Secret Redaction
Configured LLM secrets are redacted from:
- surfaced adapter/runtime errors;
- LLM diagnostics request/response/error artifacts;
- effective config/report artifacts that include LLM configuration material.
Redaction marker:
- `[REDACTED]`
## Compatibility Boundaries
This integration documentation applies only to the implemented OpenAI-compatible chat completions flow.
Not part of current behavior:
- provider SDK integration;
- non-OpenAI-compatible API contracts;
- server-side model routing features beyond explicitly configured model/base URL.

View File

@@ -0,0 +1,99 @@
# Subprocess Integration
## Scope
This document describes how a parent process should invoke Audita as a subprocess.
It covers:
- invocation shape;
- stdout/stderr behavior;
- output/report file behavior;
- diagnostics and exit behavior.
For full CLI and config references, see [`docs/cli.md`](../cli.md) and [`docs/config.md`](../config.md).
## Recommended Invocation
Use explicit output and report paths for machine workflows:
```sh
audita process <transcript.json> \
--glossary <glossary.yaml> \
--output <output-transcript.json> \
--report-json <report.json>
```
Optional commonly used flags:
- `--config <path>`
- `--output-schema <bare-segments|audita-v1>`
- `--work-dir <dir>`
- `--work-dir-retention <always|auto|never>`
- `--transcript-description <text>`
## Stdout Contract
On success:
- with `--output`: stdout is expected to be empty;
- without `--output`: stdout contains transcript JSON only.
`--report-json` output is never written to stdout.
## Stderr Contract
Stderr is human-readable status/error output.
On failures:
- stderr includes a concise top-level error;
- when diagnostics are initialized, stderr includes diagnostics directory path.
Do not treat stderr as a machine-stable JSON channel.
## Output and Report File Contract
Transcript output:
- `--output` writes corrected transcript JSON to the provided path;
- output write failures return nonzero.
Report output:
- `--report-json` writes machine-readable process report JSON to the provided path;
- run diagnostics also attempt to write their own `report.json`;
- report write failures return nonzero;
- on failure paths, report writing is best-effort and does not mask the primary run error.
## Diagnostics Contract
When run-directory initialization succeeds, per-run diagnostics artifacts are written under the configured work directory.
Typical artifacts include:
- `source-transcript.json`
- `source-transcript-parsed.json`
- `normalized-transcript.json`
- `normalization-summary.json`
- `chunking-summary.json`
- `invocation.json`
- `effective-config.json`
- `utilization-diagnostics.json`
- `correction-ledger.json`
- `report.json`
- `error.log` (failure)
Retention behavior is controlled by `--work-dir-retention` / config.
## Exit Behavior
Exit codes:
- `0`: success;
- `1`: runtime processing/output/report failure;
- `2`: CLI usage or configuration input error.
Treat any nonzero as subprocess failure.
## Parent-Process Guidance
For reliable orchestration:
- read stdout and stderr concurrently to avoid pipe blocking;
- prefer `--output` and `--report-json` for machine parsing;
- use timeout/cancellation in the parent process;
- inspect diagnostics path and `report.json`/`error.log` on failure.
For input file contracts, see [`docs/integrations/transcript-glossary-files.md`](transcript-glossary-files.md).

View File

@@ -0,0 +1,98 @@
# Transcript and Glossary File Integration
## Scope
This document defines the input file contracts for:
- transcript JSON;
- glossary YAML.
These files are loaded and validated before processing begins.
## Transcript JSON Contract
Audita accepts either top-level shape:
- JSON array of segments; or
- JSON object with a `segments` array.
Segment fields:
- `id` (optional integer in source form);
- `speaker` (required non-empty string);
- `start` (required finite non-negative number);
- `end` (required finite non-negative number, `>= start`);
- `text` (required non-empty string);
- `categories` (optional string array; entries must be non-empty).
Additional rules:
- transcript must contain at least one segment;
- duplicate segment IDs are rejected when IDs are present.
Example (`examples/tiny-transcript.json`):
```json
[
{
"id": 1,
"speaker": "A",
"start": 0.0,
"end": 1.2,
"text": "hello world"
}
]
```
## Glossary YAML Contract
Audita expects top-level `glossary` list entries.
Entry fields:
- `name` (required non-empty string);
- `category` (required non-empty string);
- `summary` (required non-empty string);
- `aliases` (optional list of strings; entries must be non-empty);
- `plural` (optional string).
Additional rules:
- glossary must contain at least one entry.
Example (`examples/tiny-glossary.yaml`):
```yaml
glossary:
- name: Audita
aliases:
- audita
category: product
summary: The Audita transcript correction CLI.
```
## Validation Failure Behavior
Representative transcript validation failures:
- invalid JSON;
- unsupported top-level shape;
- empty `speaker` or `text`;
- invalid times (`NaN`, `Inf`, negative, or `end < start`);
- duplicate IDs;
- empty transcript array.
Representative glossary validation failures:
- invalid YAML;
- empty or missing glossary entries;
- missing required entry fields;
- empty alias values.
These failures surface as schema errors and the process exits nonzero.
## CLI Usage
Minimal invocation:
```sh
audita process ./transcript.json --glossary ./glossary.yaml --output ./corrected.json
```
See also:
- [`docs/cli.md`](../cli.md)
- [`docs/config.md`](../config.md)
- [`examples/tiny-transcript.json`](../../examples/tiny-transcript.json)
- [`examples/tiny-glossary.yaml`](../../examples/tiny-glossary.yaml)

View File

@@ -0,0 +1,79 @@
# Audita Diagnostics and Reporting
## Scope
This document describes diagnostics artifacts, process report mapping, and correction ledger generation.
## Run Directory Ownership
`internal/core/diagnostics` owns run-directory creation, artifact writes, and retention decisions.
Stable artifact names include:
- `source-transcript.json`
- `source-transcript-parsed.json`
- `normalized-transcript.json`
- `normalization-summary.json`
- `chunking-summary.json`
- `utilization-diagnostics.json`
- `correction-ledger.json`
- `invocation.json`
- `effective-config.json`
- `report.json`
- `error.log` (failure)
## Process Report Mapping
`internal/framework/processreport` maps runner/CLI execution facts into `reporting.ProcessReport`.
Report metadata fields include:
- `report_schema_name` (`audita-process-report`)
- `report_schema_version` (`v1`)
- `output_schema`
- `config_version` (when file config exists)
The report includes:
- top-level status/error phase/error message;
- normalization/chunking summaries;
- diagnostics metadata paths;
- per-module results and module summary.
## Correction Ledger
`internal/framework/processreport/BuildCorrectionLedger` flattens run results into `correction-ledger.json` entries.
Dispositions:
- `applied`
- `skipped`
- `rejected`
- `failed`
Validator decisions are split into deterministic and LLM-backed groups using validator metadata classification.
## Report Write Paths
- run directory always attempts to write `report.json` when possible;
- optional `--report-json` writes an external report file;
- on failure paths, report writing is best-effort and does not mask primary run errors.
## Retention Interaction
Current retention behavior:
- failed runs are retained;
- `always` keeps successful runs;
- `auto` removes only clean successful runs;
- `never` currently retains successful runs in current implementation.
## Redaction
Redacted data expectations:
- effective config artifact uses config redaction;
- diagnostics payloads and surfaced errors use LLM secret redaction;
- reports should not include raw API key values.
## Key Tests
- `internal/core/diagnostics/*_test.go`
- `internal/framework/processreport/*_test.go`
- `internal/core/reporting/report_test.go`
- `internal/cli/run_test.go`
- `cmd/audita/main_integration_test.go`

View File

@@ -0,0 +1,67 @@
# Audita LLM Runtime
## Scope
This document describes the structured LLM runtime and scheduler behavior.
## Client Boundary
All runtime LLM calls go through `contracts.StructuredLLMClient`.
Primary adapter:
- `internal/framework/llm/OpenAICompatibleClient`
## Request/Response Behavior
The OpenAI-compatible adapter sends chat completions requests with:
- model;
- messages;
- `response_format.type = json_schema`;
- strict schema envelope (`name`, `schema`, `strict=true`).
The response is decoded into the requested structured output target.
## Response Schema Registry
Structured response schemas are registered in `internal/framework/responseschema`:
- `correction_set`
- `validator_decision_set`
Each schema includes stable diagnostics metadata (`id`, `version`, `name`, `sha256`).
## Retries and Error Handling
Adapter retries apply to retryable conditions (for example transport/decoding/retryable status classes) up to configured `max_retries`.
Errors are sanitized to redact configured API-key values before surfacing.
Malformed structured output detection is shared through `internal/framework/structuredoutput` and is used by:
- proposal generation;
- LLM-backed validators.
## Scheduling and Concurrency
`internal/framework/llm/Scheduler` provides FIFO, context-aware permit gating.
Runner composes scheduler limits across:
- total LLM concurrency;
- proposal LLM concurrency;
- validation LLM concurrency.
Scheduler release is guarded to avoid permit leaks on cancellation/error.
## Diagnostics and Redaction
`internal/framework/llm/DiagnosticsWriter` writes request/response/error artifacts.
Configured secrets are derived from `llm.ConfiguredSecrets(cfg)` and redacted from:
- diagnostics payloads;
- surfaced runtime/adapter errors.
## Key Tests
- `internal/framework/llm/openai_compatible_client_test.go`
- `internal/framework/llm/scheduler_test.go`
- `internal/framework/llm/diagnostics_test.go`
- `internal/framework/responseschema/registry_test.go`
- `internal/framework/structuredoutput/malformed_test.go`

58
docs/internal/modules.md Normal file
View File

@@ -0,0 +1,58 @@
# Audita Modules
## Scope
This document covers module contracts and built-in module packages.
## Module Contract
Modules implement `contracts.TranscriptModule`:
- `Key()`
- `ReplacementPolicy()`
- `Validators()`
- `Propose(ctx, req)`
Runner resolves configured module specs to module instances through `internal/framework/modules`.
## Built-In Modules
Current module packages:
- `internal/modules/glossary`
- `internal/modules/homophones`
- `internal/modules/spoken_word`
- `internal/modules/grammar`
Current replacement policies:
- `glossary`: `replace_all`
- `homophones`: `require_unique`
- `spoken_word`: `require_unique`
- `grammar`: `require_unique`
## Proposal Generation Ownership
Shared proposal-generation plumbing is centralized in:
- `internal/framework/proposal_generation`
Module packages own:
- prompt selection (`internal/prompts` prompt IDs);
- module-specific prompt payload construction.
Shared prompt helpers live in `internal/framework/promptcontext`.
## Validator Chain Ownership
Built-in chains are resolved in `internal/validators` per module key.
Module packages call the built-in chain resolver at construction.
## Failure and Warning Behavior
- module setup failures surface as `runner_setup` or module setup errors;
- module runtime failures surface as `runner_execution` with partial module results preserved;
- malformed structured proposal payloads are downgraded to warnings and section-level proposal rejection.
## Key Tests
- `internal/modules/*/module_test.go`
- `internal/framework/modules/registry_test.go`
- `internal/framework/proposal_generation/*_test.go`
- `internal/cli/run_test.go` (pipeline/report integration)

View File

@@ -0,0 +1,46 @@
# Audita Output Schemas
## Scope
This document describes the implemented transcript output schema registry.
## Registry Ownership
Output schema registry is owned by `internal/core/outputschema`.
Supported schema keys:
- `bare-segments`
- `audita-v1`
## Schemas
`bare-segments`:
- top-level JSON array of transcript segments.
`audita-v1`:
- top-level JSON object with:
- `schema: "audita-v1"`
- `version: "v1"`
- `segments: [...]`
Segment fields include `id`, `speaker`, `start`, `end`, `text`, and optional `categories`.
## Validation and Resolution
Config validation and runtime resolution both reject unsupported schema keys.
Unknown schema keys fail with `unsupported output schema` before output emission.
## Output Emission
The selected schema is used by `audita process` when writing:
- output file (`--output`) or
- stdout (when no `--output`).
Report metadata records selected `output_schema`.
## Key Tests
- `internal/core/outputschema/registry_test.go`
- `internal/core/config/config_test.go`
- `internal/cli/run_test.go`

92
docs/internal/overview.md Normal file
View File

@@ -0,0 +1,92 @@
# Audita Internal Overview
## Scope
This document is the internal architecture entry point for developers and coding agents.
It summarizes:
- package boundaries;
- the main `process` execution path;
- where to add new code safely.
## Package Map
CLI and command orchestration:
- `internal/cli`
Core deterministic components:
- `internal/core/config`
- `internal/core/schema`
- `internal/core/normalization`
- `internal/core/chunking`
- `internal/core/outputschema`
- `internal/core/diagnostics`
- `internal/core/reporting`
- `internal/core/modulecatalog`
Framework orchestration and contracts:
- `internal/framework/contracts`
- `internal/framework/modules`
- `internal/framework/proposal_generation`
- `internal/framework/proposals`
- `internal/framework/runner`
- `internal/framework/validators`
- `internal/framework/llm`
- `internal/framework/responseschema`
- `internal/framework/structuredoutput`
- `internal/framework/processreport`
- `internal/framework/promptcontext`
- `internal/framework/stagename`
Domain implementations:
- `internal/modules/*`
- `internal/validators/*`
- `internal/prompts`
## Main Execution Path (`audita process`)
High-level flow:
1. CLI loads effective config and validates CLI requirements.
2. Run directory is created and invocation/effective config artifacts are written.
3. Transcript/glossary files are loaded and parsed.
4. Transcript is normalized and chunked.
5. `runner.Run` executes configured module instances.
6. Proposals are validated, applied deterministically, and serialized in selected output schema.
7. Process report, utilization diagnostics, correction ledger, and retention decisions are finalized.
## Boundary Summary
- `internal/core/*` owns deterministic, reusable logic and persistence-independent rules.
- `internal/framework/*` owns orchestration contracts and reusable runtime plumbing.
- `internal/modules/*` owns module-specific proposal behavior and prompt usage.
- `internal/validators/*` owns validator composition and built-in chain assembly.
- `internal/prompts` owns embedded prompt assets and metadata registry.
## Where To Add New Code
Add config fields:
- `internal/core/config`
Add module behavior:
- one package under `internal/modules/<module_key>`
- registration/wiring through `internal/framework/modules` and config module list
Add validators:
- implementation under `internal/validators/<validator_key>`
- registry/chain wiring in `internal/validators`
Add runtime orchestration behavior:
- `internal/framework/*` (runner/proposal/validator/LLM plumbing)
Add CLI surface:
- `internal/cli`
## Related Internal Docs
- [`docs/internal/pipeline.md`](pipeline.md)
- [`docs/internal/modules.md`](modules.md)
- [`docs/internal/validators.md`](validators.md)
- [`docs/internal/llm-runtime.md`](llm-runtime.md)
- [`docs/internal/diagnostics-reporting.md`](diagnostics-reporting.md)
- [`docs/internal/prompts.md`](prompts.md)
- [`docs/internal/output-schemas.md`](output-schemas.md)

79
docs/internal/pipeline.md Normal file
View File

@@ -0,0 +1,79 @@
# Audita Internal Pipeline
## Scope
This document describes the implemented `audita process` pipeline.
## Inputs
Pipeline inputs are:
- effective config (`internal/core/config`);
- transcript JSON (`internal/core/schema`);
- glossary YAML (`internal/core/schema`).
## Pipeline Phases
1. Input loading and schema validation
- transcript and glossary files are read and parsed.
- schema failures stop the run with `transcript_schema` or `glossary_schema`.
2. Normalization
- canonical transcript segments are normalized by configured gap/duration/token settings.
- normalization summary artifacts are written.
3. Chunking
- normalized transcript is chunked with configured max/min tokens and target sections.
4. Module proposal generation
- runner executes configured module instances in sequence.
- each module proposes corrections per section.
- per-section proposal generation can run concurrently.
5. Validator filtering
- validators run on candidate proposals before apply.
- deterministic validators run before LLM-backed validators.
- LLM validator inputs are batched by max prompt token limit.
6. Deterministic apply
- approved proposals are applied via replacement policy.
- applied/skipped/rejected outcomes are recorded.
7. Output and reporting
- final transcript is serialized with selected output schema.
- report, utilization diagnostics, and correction ledger are written.
- retention policy is applied to run directory.
## Runner Outputs
`runner.Run` returns:
- final transcript;
- per-module results;
- utilization diagnostics.
CLI/reporting then map this into process report and diagnostics artifacts.
## Failure Behavior
Representative failure phases include:
- `run_dir_creation`
- `transcript_read`, `glossary_read`
- `transcript_schema`, `glossary_schema`
- `chunking`
- `runner_setup`, `runner_execution`
- `output_schema`, `serialization`, `output_write`, `stdout_write`
When diagnostics are available, failure stderr includes diagnostics path.
## Invariants
- module execution order follows configured module sequence;
- proposal/validator nondeterminism is isolated before deterministic apply;
- proposal indices are assigned deterministically by section order;
- output/report artifacts are generated from run results, not speculative state.
## Key Tests
- `internal/framework/runner/runner_test.go`
- `internal/framework/proposal_generation/*_test.go`
- `internal/cli/run_test.go`
- `cmd/audita/main_integration_test.go`

62
docs/internal/prompts.md Normal file
View File

@@ -0,0 +1,62 @@
# Audita Prompt Registry
## Scope
This document describes embedded prompt assets, prompt metadata, and rendering behavior.
## Registry Ownership
Prompt registry lives in `internal/prompts` and embeds assets under `internal/prompts/assets/**`.
Registered prompt IDs:
- `modules.glossary.proposal`
- `modules.homophones.proposal`
- `modules.spoken_word.proposal`
- `modules.grammar.proposal`
- `validators.spoken_form_plausibility`
- `validators.meaning_reversal_review`
- `validators.editorial_review`
- `validators.grammar_review`
- `validators.spoken_word_review`
## Metadata Model
Each prompt has metadata:
- `prompt_id`
- `prompt_version`
- `prompt_source`
- `embedded_path`
- `sha256`
Current source/version values:
- `prompt_source = builtin`
- `prompt_version = v1`
## Rendering
`prompts.RenderUserSystem(promptID, data)` renders system/user templates.
Template behavior:
- uses Go `text/template`;
- `missingkey=error` is enabled;
- output is trimmed.
A shared hardening fragment is embedded once and referenced by prompt templates.
## Prompt Context Inputs
Shared prompt payload helpers:
- transcript section JSON (`internal/framework/promptcontext/MarshalTranscriptSectionJSON`)
- transcript description block (`TranscriptDescriptionBlock`)
Modules and LLM validators provide typed data maps to render prompt assets.
## Diagnostics Integration
Prompt metadata is attached to proposal/validator diagnostics request metadata using `Metadata.DiagnosticsMap()`.
## Key Tests
- `internal/prompts/registry_test.go`
- `internal/framework/promptcontext/*_test.go`
- module and validator prompt builder tests

View File

@@ -0,0 +1,72 @@
# Audita Validators
## Scope
This document describes validator composition, execution order, and decision handling.
## Ownership
Built-in validator keys and chains:
- `internal/validators`
Shared validator runtime mechanics:
- `internal/framework/validators`
Execution-class metadata:
- `internal/validators/metadata`
## Built-In Validator Keys
Deterministic:
- `proposal_shape`
- `confidence_threshold`
- `original_text_presence`
- `non_empty_corrected_text`
- `no_effect`
- `protected_terms`
LLM-backed:
- `spoken_form_plausibility`
- `meaning_reversal_review`
- `editorial_review`
## Built-In Chains
Module chains are defined in `internal/validators/chains.go`.
Glossary, homophones, spoken_word, and grammar each resolve a fixed ordered chain.
## Runtime Execution
For each module section:
1. run deterministic validators;
2. run LLM-backed validators;
3. record decisions and warnings;
4. carry only approved proposals forward.
Decision cardinality is enforced: each candidate proposal must receive exactly one decision per validator.
## LLM Validator Batching
LLM validators:
- build canonical validation request payloads;
- batch by `validation_max_prompt_tokens`;
- call structured LLM client using response schema registry.
Oversized single proposals are rejected with `validator_input_too_large`.
Malformed LLM validator responses are downgraded to warnings and rejected batch decisions.
## Decision and Rejection Reporting
Runner records:
- `validator_decisions`
- `validator_rejected`
- warning records (including malformed response warnings)
Correction ledger classifies deterministic vs LLM validator decisions using canonical metadata classes.
## Key Tests
- `internal/validators/*_test.go`
- `internal/framework/validators/*_test.go`
- `internal/framework/processreport/correction_ledger_test.go`
- `internal/cli/run_test.go`

113
docs/operations.md Normal file
View File

@@ -0,0 +1,113 @@
# Audita Operations
## Scope
This document covers operational behavior for `audita process` as currently implemented:
- run lifecycle;
- output and report files;
- diagnostics artifacts;
- run-directory retention behavior;
- failure inspection and recovery.
For command syntax, see [`docs/cli.md`](cli.md).
## Process Run Lifecycle
A `process` run performs these high-level steps:
1. load effective config (defaults + optional file + env + CLI);
2. create a per-run diagnostics directory;
3. load transcript JSON and glossary YAML;
4. parse/validate input schemas;
5. normalize transcript and compute chunking;
6. run configured modules/validators;
7. serialize output schema and write transcript output;
8. build and write process report;
9. apply run-directory retention.
If a failure happens after diagnostics initialization, the run writes failure details and returns nonzero.
## Output Files
Transcript output:
- when `--output <path>` is set, corrected transcript JSON is written to that file;
- when `--output` is omitted, corrected transcript JSON is written to stdout.
Report output:
- when `--report-json <path>` is set, Audita writes a process report JSON file;
- the run directory also writes its own `report.json` artifact.
On success with `--output`, stdout is expected to be empty.
## Diagnostics Directory
By default, runs use `work_dir` from effective config (default `/tmp/audita`).
Each run directory is created under the work dir using a generated ID like `run-<unix-nanos>`.
Top-level diagnostics artifacts:
- `source-transcript.json`
- `source-transcript-parsed.json`
- `normalized-transcript.json`
- `normalization-summary.json`
- `chunking-summary.json`
- `utilization-diagnostics.json`
- `correction-ledger.json`
- `invocation.json`
- `effective-config.json` (redacted)
- `report.json`
- `error.log` (failure runs)
Report diagnostics metadata includes resolved paths to these artifacts.
## Correction Ledger and Utilization Diagnostics
`correction-ledger.json` records correction dispositions:
- `applied`
- `skipped`
- `rejected`
- `failed`
`utilization-diagnostics.json` records effective concurrency and execution timing summaries for run/module/validator activity.
## Retention Behavior
Retention is controlled by `work_dir_retention` (`auto|always|never`).
Current behavior:
- failed runs are always retained;
- `always`: successful runs are retained;
- `auto`: successful runs are retained only when skipped/rejected corrections occurred; clean successful runs are removed;
- `never`: successful runs are currently retained (same net retention outcome as `always` in current implementation).
Even when a successful run directory is removed under `auto`, an explicit `--report-json` file is still preserved at its target path.
## Failure Inspection
For failed runs:
1. read stderr for the top-level failure and diagnostics path;
2. open `error.log` in the reported run directory;
3. inspect run `report.json` (`status`, `error_phase`, `error_message`);
4. inspect related artifacts referenced by report diagnostics metadata.
Typical `error_phase` values include:
- `transcript_read`
- `glossary_read`
- `transcript_schema`
- `glossary_schema`
- `chunking`
- `runner_setup`
- `runner_execution`
- `output_schema`
- `serialization`
- `output_write`
- `stdout_write`
## Recovery Guidance
Safe recovery pattern:
1. correct the immediate input/config/output-path problem;
2. rerun with `--work-dir-retention always` during debugging;
3. once stable, restore your normal retention mode.
Not implemented:
- resume/checkpoint APIs
- remote diagnostics/report storage

207
docs/policy/architecture.md Normal file
View File

@@ -0,0 +1,207 @@
# Architecture Policy
## Purpose
This document defines Audita's development architecture and invariants for maintainers and LLM coding agents. It describes how the project is intended to be changed safely, based on behavior implemented in this repository today.
User-facing behavior belongs in the README and focused runtime docs. Future or proposed work belongs only under `docs/roadmap/`.
## Project Shape
Audita is a single-process Go CLI for transcript polishing. The executable entrypoint is `cmd/audita`; command handling lives in `internal/cli`.
The implemented `audita process` flow is:
1. load effective config;
2. read and validate transcript JSON and glossary YAML;
3. normalize transcript segments;
4. chunk the working transcript into sections;
5. resolve configured module instances;
6. run correction modules and validator chains;
7. apply approved proposals deterministically;
8. write transcript output, reports, and diagnostics artifacts.
The current built-in modules are `glossary`, `homophones`, `spoken_word`, and `grammar`. The default configured module sequence repeats `glossary`.
For external behavior and compatibility details, prefer links to existing behavior docs:
- [CLI reference](../cli.md)
- [Configuration](../config.md)
- [Operations](../operations.md)
- [Troubleshooting](../troubleshooting.md)
- [Integration docs](../integrations/subprocess.md)
- [Internal docs](../internal/overview.md)
## Core Design Principles
- **Hexagonal architecture:** keep domain behavior behind narrow internal contracts. CLI, filesystem, config loading, diagnostics writing, and LLM transport are adapters around the core processing flow.
- **Composable modules and validators:** correction stages and validators should remain small, explicit, and independently testable.
- **Deterministic orchestration around LLM calls:** LLM responses are nondeterministic inputs. Proposal indexing, validator ordering, proposal application, reports, and output serialization must remain deterministic.
- **Bounded and observable concurrency:** use the implemented schedulers and configured concurrency limits for LLM call sites. Preserve utilization diagnostics when changing scheduling or orchestration.
- **Conservative correction behavior:** validate proposed corrections before application; apply accepted proposals through deterministic apply-time safety checks.
- **Standard-library-first:** prefer the Go standard library. Narrow third-party dependencies are acceptable when they materially improve maintainability, such as `gopkg.in/yaml.v3` for YAML parsing.
- **Current-behavior documentation:** non-roadmap docs must describe implemented behavior only.
## Architectural Boundaries
`internal/core` owns domain data handling and stable runtime contracts that do not require CLI or provider transport knowledge:
- config defaults, loading, validation, redaction, and catalogs;
- transcript and glossary schemas;
- normalization and chunking;
- output-schema encoding;
- diagnostics artifact naming and run-directory helpers;
- public process report shapes.
`internal/framework` owns orchestration contracts and reusable runtime mechanics:
- module and validator interfaces;
- proposal generation, proposal application, and prompt context;
- runner orchestration;
- LLM scheduler, OpenAI-compatible adapter, redaction helpers, and diagnostics writers;
- structured response schema registry;
- process report and correction-ledger assembly.
`internal/modules/*` owns module-specific correction stages. `internal/validators/*` owns built-in validator implementations, registry, chains, and execution-class metadata. `internal/prompts` owns embedded prompt assets and prompt metadata.
`internal/cli` owns command parsing, exit codes, stdout/stderr behavior, config command behavior, filesystem input/output wiring, and top-level process orchestration. CLI concerns should not move into modules, validators, or schema logic.
Tests should stay close to the behavior they protect. Shared test helpers are acceptable when they remove clear duplication without hiding module-specific behavior.
## Modules and Validators
Modules implement `contracts.TranscriptModule`. A module must provide:
- a stable key;
- a replacement policy;
- a validator chain;
- proposal generation from explicit request inputs.
Module packages should stay separate. Do not collapse module-specific prompts, scope, or validation choices into a broad generic stage abstraction.
Validators implement the shared validator contract and return one decision per candidate proposal. Deterministic validators and LLM-backed validators are both composable chain elements. Validator identity and execution class metadata are stable enough to affect ordering, diagnostics, reports, and correction-ledger classification.
Future module or validator changes should preserve:
- explicit inputs and outputs;
- no hidden global state;
- explicit config dependencies;
- deterministic proposal index handling;
- validation before final mutation;
- stable reason codes and validator keys where already exposed.
## LLM Integration and Concurrency
LLM calls are external effects behind narrow contracts. Production structured completions use `contracts.StructuredLLMClient`; the implemented provider adapter is OpenAI-compatible HTTP code in `internal/framework/llm`.
Structured response schemas are registered in `internal/framework/responseschema`. Provider-side schema enforcement is not a substitute for local validation: Audita still validates proposal structure, validator decision cardinality, and apply-time safety.
Concurrency is bounded by configured scheduler limits:
- total LLM concurrency;
- proposal LLM concurrency;
- validation LLM concurrency.
The scheduler is context-aware and releases permits on success, failure, and cancellation. Runner code may collect section-level work concurrently, but transcript mutation is applied later in deterministic proposal-index order.
Diagnostics for LLM interactions should be useful for debugging without leaking configured secrets. Use the existing redaction helpers and `llm.ConfiguredSecrets`.
## State, Inputs, and Outputs
Audita does not implement resume, checkpoint, manifest, or remote storage behavior. Runtime state is in memory plus per-run diagnostics artifacts written under the configured work directory.
Transcript input accepts the implemented JSON forms documented in the public contract. Parsed source transcripts are normalized into Audita's internal transcript shape before chunking and module execution.
Proposals and validator decisions are intermediate runtime data. Approved proposals are applied through `internal/framework/proposals`, which clones transcript state, orders by proposal index, and records applied or skipped changes.
Transcript output is encoded through `internal/core/outputschema`. Reports and correction ledgers are machine-readable artifacts derived from runner outputs; their public shape should not be changed casually.
## Configuration and CLI Boundaries
Config behavior is owned by `internal/core/config`; command usage and process wiring are owned by `internal/cli`.
`audita process` uses implemented precedence: defaults, config file, environment, then CLI flags. `config validate` validates defaults plus a file config and intentionally does not apply environment overrides. `config print-effective` applies defaults, file config, and environment overrides, then prints redacted JSON.
Do not duplicate full CLI or config reference material here. Use [Configuration](../config.md), [CLI reference](../cli.md), [Operations](../operations.md), and integration docs under [`docs/integrations/`](../integrations/subprocess.md) for current external behavior.
When adding config fields or CLI flags, update:
- config defaults, file/env/CLI application, and validation;
- CLI flag extraction if applicable;
- redaction when secrets are involved;
- tests for precedence and source-specific behavior;
- user-facing docs if external behavior changes.
## Errors, Logging, and Diagnostics
Errors should be phase-specific enough for CLI users and subprocess callers. The CLI writes human-readable errors to stderr and preserves transcript JSON-only stdout behavior on successful stdout output.
Run diagnostics are best-effort after run-directory creation. Failed runs are retained. Successful run retention follows the implemented work-dir retention policy.
Diagnostics and reports must not leak configured LLM secrets. Config redaction and LLM payload/error redaction are separate responsibilities and should remain separate.
Process reports, diagnostics metadata, utilization diagnostics, and correction ledgers are part of the public contract. Prefer additive, compatible changes.
## Testing Expectations
Use targeted package tests for touched behavior and `go test ./...` for substantial changes.
When changing modules, inspect or add:
- package-local module tests under `internal/modules/*`;
- prompt rendering or proposal-generation tests when prompt inputs change;
- parity or release fixtures when public output behavior changes.
When changing validators, inspect or add:
- validator package tests;
- registry and chain tests under `internal/validators`;
- framework validator tests for batching, malformed output, diagnostics, and cardinality.
When changing LLM integration or concurrency, inspect or add:
- `internal/framework/llm` scheduler/client/redaction tests;
- `internal/framework/runner` orchestration and utilization tests;
- structured-output malformed classification tests.
When changing config, CLI, schema, output, reports, or diagnostics, inspect or add:
- `internal/core/config` tests;
- CLI tests under `internal/cli`;
- schema and output-schema tests under `internal/core`;
- report, diagnostics, parity, and release-fixture tests.
## Dependency Policy
Audita should remain dependency-light. Prefer standard-library solutions for CLI parsing, HTTP, JSON, filesystem, synchronization, and tests.
Third-party dependencies should be narrow, justified, and preferably de facto standard for their purpose. YAML parsing is the current direct dependency exception.
Do not add broad frameworks for CLI, dependency injection, workflow orchestration, logging, or plugin systems without a concrete implemented need and focused tests.
## Documentation Expectations
Follow [Documentation Policy](./documentation.md). Architecture policy must stay concise and aligned with implemented behavior.
Do not use architecture docs as changelogs. Do not describe planned modules, adapters, modes, persistence, or configuration unless they are implemented. Put future work under `docs/roadmap/`.
## Architectural Invariants
- Keep LLM transport behind `StructuredLLMClient` and framework adapter boundaries.
- Keep correction modules narrowly scoped and package-separated.
- Keep validators modular, composable, and identified by stable keys.
- Keep CLI/config/filesystem concerns out of module and validator domain logic.
- Preserve deterministic transcript mutation and output handling around nondeterministic LLM calls.
- Keep LLM concurrency bounded, configurable, and observable where implemented.
- Keep run diagnostics and reports redacted and machine-readable.
- Keep public CLI, config, output-schema, diagnostics, report, prompt, module, and validator contracts stable unless a change is explicit and tested.
- Prefer small shared helpers over broad rewrites.
- Avoid new dependencies unless they are narrow and clearly justified.
## Non-Goals
- No plugin framework is implemented.
- No generic workflow engine is implemented.
- No resume, checkpoint, manifest, or remote storage system is implemented.
- No multi-process service mode is implemented.
- No provider SDK abstraction beyond the current structured LLM client contract and OpenAI-compatible HTTP adapter is implemented.

124
docs/policy/development.md Normal file
View File

@@ -0,0 +1,124 @@
# Audita Development Workflow
## Scope
This is the canonical contributor workflow for Audita maintainers and coding agents.
It defines:
- repository layout and boundaries;
- setup and test commands;
- expectations for code changes;
- how to add config, CLI, modules, validators, docs, and examples.
## Setup
Prerequisites:
- Go `1.24` or newer.
Common commands:
```sh
go test ./...
go build ./cmd/audita
```
## Repository Layout
Top-level areas:
- `cmd/audita`: executable entrypoint.
- `internal/cli`: command parsing and process/config command orchestration.
- `internal/core`: deterministic config/schema/normalization/chunking/output/diagnostics/reporting logic.
- `internal/framework`: runner orchestration, contracts, proposal generation/application, validators runtime, LLM runtime, response schemas.
- `internal/modules/*`: module-specific correction behavior.
- `internal/validators/*`: validator implementations, chains, and metadata.
- `internal/prompts`: embedded prompts and prompt metadata.
- `docs/`: canonical documentation.
- `examples/`: maintained copyable inputs/configs.
## Change Workflow
1. Confirm scope and behavior contract before editing.
2. Make focused changes in the appropriate ownership area.
3. Add or update tests for changed behavior.
4. Run targeted package tests for touched areas.
5. Run `go test ./...` for substantial changes.
6. Update docs/examples when external behavior changes.
## How To Add or Change Configuration
1. Add fields/defaults/validation under `internal/core/config`.
2. Apply source precedence correctly (defaults, file, env, CLI for `process`).
3. Ensure `config validate` remains file-only and `config print-effective` remains redacted.
4. Update tests in `internal/core/config` and related CLI tests.
5. Update [`docs/config.md`](../config.md) and relevant examples under `examples/`.
## How To Add or Change CLI Behavior
1. Implement parsing/wiring in `internal/cli`.
2. Keep stdout/stderr and exit behavior compatible unless intentional and documented.
3. Update CLI tests under `internal/cli` and integration tests under `cmd/audita`.
4. Update [`docs/cli.md`](../cli.md) and related integration docs.
## How To Add or Change Modules
1. Add or update one module package under `internal/modules/<module_key>`.
2. Keep module-specific prompt ownership in the module + `internal/prompts`.
3. Wire module registration/catalog resolution through framework/core module catalog code.
4. Verify replacement policy and validator chain selection.
5. Add/update module tests and proposal-generation tests.
6. Update internal docs when behavior/contracts change.
## How To Add or Change Validators
1. Implement validator behavior in `internal/validators` and shared runtime pieces in `internal/framework/validators` only when needed.
2. Preserve stable validator keys and decision semantics where already exposed.
3. Keep deterministic vs LLM-backed execution-class behavior explicit.
4. Add/update validator, chain, batching, and malformed-output tests.
5. Update validator documentation when external or developer-facing behavior changes.
## Documentation and Examples Expectations
- Keep one canonical home per topic (see [`docs/policy/documentation.md`](documentation.md)).
- Do not document future/unimplemented behavior outside `docs/roadmap/`.
- Keep command examples and config/examples in sync with current code.
- Keep examples secret-free and copyable.
## Practical Validation Checklist
Use this checklist for meaningful runtime-impacting changes:
1. Run core tests:
```sh
go test ./...
```
2. Verify config commands and examples:
```sh
go run ./cmd/audita config validate --config examples/minimal-config.yml
go run ./cmd/audita config validate --config examples/production-config.yml
go run ./cmd/audita config print-effective --config examples/minimal-config.yml
```
3. Re-check subprocess/runtime contract when touching CLI/process/report paths:
- `--output` success keeps stdout empty;
- no `--output` success writes transcript JSON to stdout;
- `--report-json` writes file output and is not written to stdout;
- failures return nonzero and include diagnostics path when available.
4. Re-check diagnostics/report/redaction when touching LLM, reporting, or diagnostics code:
- report schema metadata fields remain present;
- diagnostics artifact paths remain valid;
- configured secret values remain redacted in reports/diagnostics/errors.
5. Re-check output schema behavior when touching serialization/schema code:
- default `bare-segments` behavior remains correct unless intentionally changed;
- `audita-v1` behavior remains correct unless intentionally changed;
- unsupported schemas fail validation/resolve paths clearly.
## Commit Discipline
- Keep commits scoped and reviewable.
- Avoid mixing unrelated refactors with behavior changes.
- Use concise plain-English commit messages.

View File

@@ -0,0 +1,356 @@
# Go Project Documentation Policy
## Purpose
Project documentation must help four audiences:
1. users who need to run the application;
2. administrators/operators who need to configure and operate it;
3. developers who need to understand and change it safely;
4. LLM coding agents that need clear scope, boundaries, and invariants.
Docs should be accurate, concise, task-oriented, and organized by audience. Prefer links to canonical docs over repetition.
## Core Rules
### 1. Keep docs concise
Each document should cover a defined scope and only the essentials for that scope.
Avoid:
- long background explanations;
- repeated reference material;
- implementation detail in user-facing docs;
- aspirational language outside roadmap docs;
- verbose examples where one minimal example is clearer.
### 2. Document only implemented behavior outside roadmap files
Unimplemented, planned, aspirational, experimental, or future work may be described only under:
- `docs/roadmap/`
No other documentation file, including `README.md`, should describe code, features, modules, stages, commands, config fields, or behaviors that do not currently exist.
If a feature is partial, non-roadmap docs may describe only the implemented portion and its current boundary.
### 3. Use canonical homes
Each type of information should have one canonical location.
Canonical homes:
- project purpose and quickstart: `README.md`
- development principles: `docs/policy/architecture.md`
- configuration reference: `docs/config.md`
- CLI reference: `docs/cli.md`
- operations and recovery: `docs/operations.md`
- troubleshooting: `docs/troubleshooting.md`
- implemented internals: `docs/internal/`
- future work: `docs/roadmap/`
- contributor workflow: `docs/policy/development.md`
- copyable examples: `examples/`
Other files should summarize briefly and link to the canonical source.
### 4. Keep examples real
Examples should be valid, maintained, and free of secrets.
Where practical:
- example configs should load successfully;
- example commands should match real CLI syntax;
- important examples should be covered by tests.
## Documentation Profiles
All projects require:
- `README.md`
- `docs/policy/architecture.md`
Additional docs depend on the project.
### Small library
Recommended:
- `docs/policy/development.md`, if contributor conventions are non-obvious
### Simple CLI
Required:
- `docs/cli.md`
Recommended:
- `docs/policy/development.md`
### Config-driven CLI
Required:
- `docs/cli.md`
- `docs/config.md`
Recommended:
- `examples/`
- `docs/policy/development.md`
### Stateful or operator-facing application
Required:
- `docs/cli.md`, if CLI-based
- `docs/config.md`, if config-driven
- `docs/operations.md`
Recommended:
- `docs/troubleshooting.md`
- `examples/`
- `docs/policy/development.md`
### Modular, staged, service-oriented, or orchestration application
Required:
- `docs/cli.md`, if CLI-based
- `docs/config.md`, if config-driven
- `docs/operations.md`
- `docs/internal/`
- `docs/policy/development.md`
Recommended:
- `docs/troubleshooting.md`
- validated examples under `examples/`
## Required Documents
### README.md
**Audience:** users, administrators, operators
The README is the outward-facing project orientation page.
It should include, in order:
1. concise description;
2. elevator pitch;
3. shortest useful command or usage example;
4. links to targeted docs.
The README should be short. It is not a manual.
The “shortest useful command” means the simplest command that performs the projects core use case. (It does not mean `app --help`.)
### docs/policy/architecture.md
**Audience:** developers, LLM coding agents
`docs/policy/architecture.md` is required for every project.
It is an inward-facing development policy document. It should describe how the project is intended to be built and changed.
It should include:
- project shape;
- core design principles;
- package and boundary philosophy;
- state/persistence philosophy, if applicable;
- external integration philosophy, if applicable;
- error-handling and logging principles;
- testing expectations;
- documentation expectations;
- architectural invariants;
- explicit non-goals, if useful.
For small projects, this file may be brief. It may simply state that the project is intentionally narrow, monolithic, and dependency-light.
### docs/policy/development.md
**Audience:** developers, LLM coding agents
Required for projects maintained by humans and LLM coding agents.
It should include:
- repository layout;
- build/test commands;
- coding conventions;
- dependency policy;
- how to add config fields;
- how to add CLI flags;
- how to add stages/modules/adapters, if applicable;
- how to update examples;
- documentation update expectations.
### docs/config.md
**Audience:** administrators, operators, advanced users
Required for applications with configuration files.
It should include, in order:
1. config file locations and discovery precedence;
2. minimal working config;
3. production-oriented config;
4. full configuration reference;
5. secrets handling, if applicable;
6. links to maintained examples.
The full configuration reference should be canonical.
### docs/cli.md
**Audience:** users, administrators, operators
Required for CLI applications.
It should include, in order:
1. shortest useful command;
2. command overview;
3. complete flag reference;
4. common workflows;
5. diagnostic or recovery commands, if applicable.
Explain when commands are useful, not just their syntax.
### docs/operations.md
**Audience:** administrators, operators
Required for applications that maintain state, support resume behavior, run multiple stages, write durable artifacts, use remote storage, or require recovery procedures.
It should cover:
- normal workflow;
- filesystem layout;
- remote storage layout, if applicable;
- logs and manifests;
- resume/retry behavior;
- cleanup behavior;
- archive/backup behavior;
- safe recovery procedures;
- operational caveats.
### docs/troubleshooting.md
**Audience:** administrators, operators
Recommended once recurring failure modes exist.
Each entry should include:
- symptom;
- likely cause;
- diagnostic command or inspection step;
- safe fix;
- relevant links.
### docs/internal/
**Audience:** developers, LLM coding agents
Required for modular, staged, service-oriented, or orchestration projects.
This directory describes implemented internal components. It is not the roadmap.
Use one file per major component where useful.
Each component doc should include:
1. purpose;
2. inputs and outputs;
3. boundaries;
4. config fields used;
5. external adapters used;
6. state or manifest behavior, if applicable;
7. skip/resume behavior, if applicable;
8. failure behavior;
9. tests to inspect before changing;
10. architectural invariants.
### docs/roadmap/
**Audience:** maintainers, developers, LLM coding agents
This is the only place for planned, future, aspirational, experimental, or unimplemented work.
Roadmap docs should clearly distinguish:
- proposed work;
- accepted plans;
- deferred ideas;
- rejected ideas;
- implementation prompts or task breakdowns, if useful.
Roadmap docs should not be confused with current behavior.
### docs/integrations/
**Audience:** developers, LLM coding agents
Required for projects that depend on external CLIs, APIs, services, protocols, or file formats where the integration contract is important to maintain.
This directory contains concise, versioned reference notes for external integration contracts. It should document only the parts of the external system that this project actually uses.
Use one file per integration where useful.
## Examples Directory
Projects with non-trivial configuration or workflows should include `examples/`.
Useful examples include:
- minimal working config;
- production-oriented config;
- full annotated config;
- local development config;
- remote/object-storage config;
- minimal session/input file.
Examples should be valid, maintained, tested when practical, and linked from relevant docs.
## Security and Privacy
Docs and examples must not include:
- real API keys;
- tokens;
- passwords;
- private keys;
- private environment dumps;
- sensitive user data;
- raw private transcripts;
- private infrastructure details unless intentionally public.
Document secret-handling mechanisms, not actual secret values.
## Maintenance Rules
When docs change, verify the affected behavior.
Where practical:
- load example config files in tests;
- test CLI examples or command parser behavior;
- validate documented flags against real flags;
- remove stale references;
- update links after renames;
- keep roadmap content out of non-roadmap docs.
If documentation and code disagree, fix the documentation and/or open a roadmap item; do not leave aspirational behavior in current-behavior docs.
Documentation is complete only when it matches the current code.
## Documentation Change Checklist
Before merging documentation changes, verify:
- README is concise and orientation-focused.
- `docs/policy/architecture.md` describes development principles.
- Future work appears only under `docs/roadmap/`.
- User-facing docs avoid unnecessary internals.
- Developer-facing docs preserve boundaries and invariants.
- Config examples match the schema.
- CLI examples match real commands and flags.
- Defaults appear in the canonical config reference.
- No secrets or private data are included.
- Links are accurate.

View File

@@ -1,124 +0,0 @@
# Audita Release Checklist
Use this checklist before cutting a pre-1.0 or 1.0 release candidate.
## Core test pass
- Run:
- `go test ./...`
- Confirm tests pass without live LLM credentials and without Python dependencies.
## Config validation and precedence
- Validate a representative config:
- `audita config validate --config <path>`
- Inspect redacted effective config:
- `audita config print-effective --config <path>`
- Confirm precedence behavior:
- defaults -> file config -> environment -> CLI.
- Confirm default config search order:
- `/usr/local/etc/audita/config.yml` first, then `/etc/audita/config.yml`.
- Confirm missing both default-path config files is non-fatal when `--config`/`AUDITA_CONFIG` are unset.
## Output schema checks
- Verify default output schema remains `bare-segments`.
- Verify `--output-schema audita-v1` emits object payload with `schema` and `version`.
- Verify unknown schema (for example `seriatim-intermediate`) fails clearly.
## Subprocess contract checks
- With `--output`, verify stdout is empty on success.
- Without `--output`, verify stdout contains transcript JSON only.
- Verify `--report-json` writes file output and does not write report JSON to stdout.
- Verify failure stderr remains human-readable and includes diagnostics path when available.
- Verify nonzero exit on failures.
## Structured LLM checks
- Verify runtime uses the Audita-owned OpenAI-compatible adapter.
- Verify structured response schemas are attached via `response_format.type=json_schema`.
- Verify diagnostics metadata includes structured schema `id/version/name/sha256`.
- Verify provider output is still locally decoded/validated before use.
## Report and diagnostics schema checks
- Verify report metadata fields:
- `report_schema_name`
- `report_schema_version`
- `output_schema`
- `config_version` when file config is used.
- Verify diagnostics artifact references exist in reports:
- transcript/normalization/chunking/invocation/effective-config artifacts
- utilization diagnostics artifact
- correction ledger artifact
- error log on failures.
## Redaction checks
- Verify secrets are redacted from:
- `effective-config.json`
- run-dir and `--report-json` reports
- LLM request/response/error diagnostics payloads.
- Verify no API keys/bearer tokens leak into fixtures or outputs.
## Prompt and validator metadata checks
- Verify prompt metadata appears in LLM request metadata diagnostics:
- `prompt_id`, `prompt_version`, `prompt_source`, `embedded_path`, `sha256`.
- Verify stable validator keys appear in report decisions/rejections.
- Verify built-in validator chains resolve and execute for default and explicit module runs.
## Utilization diagnostics checks
- Verify `utilization-diagnostics.json` exists on successful runs.
- Verify partial utilization artifact behavior on controlled failure paths.
- Verify utilization fields are structurally present and nonnegative:
- effective concurrency
- run timing
- module timing summaries
- per-validator timing summaries.
## Correction ledger checks
- Verify `correction-ledger.json` exists on successful runs.
- Verify report references ledger artifact path.
- Verify ledger dispositions include applied/rejected and skipped/failed where exercised.
- Verify validator rejection and proposal-application skip remain distinct.
## Pipeline behavior checks
- Verify default full pipeline run remains:
- `glossary`, `homophones`, `glossary`, `spoken_word`, `grammar`
- with deterministic repeated instance naming (`glossary_1`, `glossary_2`).
- Verify explicit module runs (`--modules`) still work.
## Failure and cancellation checks
- Verify controlled failure paths retain diagnostics and produce best-effort failure reports.
- Verify timeout/cancellation paths exit nonzero, do not hang, and retain failure diagnostics when initialized.
## Release fixture/idempotence checks
- Run release fixtures (`internal/cli/testdata/release`) through `go test ./...`.
- Confirm fixture checks cover:
- must-apply and must-not-apply expectations
- protected-term survival
- report and diagnostics contracts
- output-schema checks
- prompt/schema metadata diagnostics
- utilization/ledger artifacts
- idempotence-oriented second pass no-op behavior with deterministic fake responses.
## Deferred-feature guardrail
- Confirm release docs do not claim support for deferred items:
- filesystem prompt overrides
- user-configurable validator chains
- arbitrary user-supplied output schemas
- resume/start-at/stop-after execution
- diff/check/propose-only modes
- generated transcript descriptions enabled by default
- interactive review UI
- UI/server wrapper
- provider benchmarking harness.

153
docs/troubleshooting.md Normal file
View File

@@ -0,0 +1,153 @@
# Audita Troubleshooting
## Scope
This guide lists recurring implemented failure modes for `audita process` and `audita config`.
For each entry: symptom, likely cause, inspect, and fix.
## Config Validation Fails
Symptom:
- `audita config validate --config <path>` exits nonzero.
Likely causes:
- missing `version`;
- unsupported config version;
- unknown YAML field;
- unsupported module key or output schema;
- invalid numeric/range/concurrency/retention values.
Inspect:
1. rerun `audita config validate --config <path>` and read stderr.
2. if needed, inspect effective config with `audita config print-effective --config <path>`.
Fix:
- set `version: 1`;
- remove unknown fields;
- use supported module keys and output schemas (`bare-segments`, `audita-v1`);
- correct invalid values to satisfy validation constraints.
## Config File Resolution Errors
Symptom:
- `audita process` fails before processing with config-related errors like `config file not found`.
Likely causes:
- `--config` points to a missing path;
- `AUDITA_CONFIG` points to a missing path;
- unreadable config path.
Inspect:
1. confirm `--config` or `AUDITA_CONFIG` path exists;
2. run `audita config validate --config <path>` directly.
Fix:
- correct the path or unset invalid `AUDITA_CONFIG`;
- fix permissions for the config file.
## Transcript or Glossary Schema Errors
Symptom:
- stderr includes `transcript_schema` or `glossary_schema` and run exits nonzero.
Likely causes:
- transcript is not valid JSON or has invalid segment fields;
- glossary is not valid YAML or has missing required glossary entry fields.
Inspect:
1. check stderr for parser/validation details;
2. if diagnostics were created, inspect `error.log` and run `report.json` (`error_phase`);
3. inspect `source-transcript.json` and `source-transcript-parsed.json` in the run directory.
Fix:
- correct transcript JSON shape/content;
- correct glossary YAML shape/content and required entry fields;
- rerun validation with known-good tiny examples for comparison:
- `examples/tiny-transcript.json`
- `examples/tiny-glossary.yaml`
## LLM Runtime/Backend Failures
Symptom:
- stderr includes `runner_execution` (or backend timeout/error details) and nonzero exit.
Likely causes:
- unreachable/failed LLM endpoint;
- timeout/cancellation;
- runtime module execution failure.
Inspect:
1. inspect stderr for backend message details;
2. inspect run `report.json` (`error_phase`, `module_results`);
3. inspect diagnostics payloads and `error.log`.
Fix:
- verify model/base URL/API key settings;
- increase timeout if needed;
- rerun with `--work-dir-retention always` while debugging.
## Output File Write Failure
Symptom:
- stderr includes `failed to write output file` and run exits nonzero.
Likely causes:
- output path directory missing;
- insufficient filesystem permissions;
- invalid output target path.
Inspect:
1. check `--output` target directory exists and is writable;
2. inspect run diagnostics `error.log` and report `error_phase`.
Fix:
- write to a valid writable path;
- create missing directories;
- adjust permissions.
## Report File Write Failure
Symptom:
- stderr includes `failed to write report JSON file` and run exits nonzero.
Likely causes:
- invalid or unwritable `--report-json` target path.
Inspect:
1. verify parent directory exists and is writable;
2. inspect diagnostics `error.log` for `report_write` context.
Fix:
- choose a writable report path;
- create missing directories;
- rerun.
## Unsupported Output Schema
Symptom:
- stderr includes `unsupported output schema` and run exits nonzero.
Likely causes:
- unsupported `--output-schema` value;
- unsupported `output.schema` in config.
Inspect:
1. check CLI/config schema key;
2. run `audita config validate --config <path>` when config is involved.
Fix:
- use `bare-segments` or `audita-v1`.
## Diagnostics Directory Lookup
Symptom:
- run fails and you need artifacts for debugging.
Inspect:
1. read stderr for `audita process: diagnostics: <run-dir>`;
2. open `<run-dir>/report.json` and `<run-dir>/error.log`;
3. use diagnostics paths embedded in report metadata for artifact lookup.
Fix:
- rerun with `--work-dir-retention always` to preserve run directories during investigation.

View File

@@ -0,0 +1,6 @@
version: 1
output:
schema: bare-segments
llm:
proposal:
api_key_env: AUDITA_LLM_API_KEY

View File

@@ -0,0 +1,41 @@
version: 1
pipeline:
modules: [glossary, homophones, glossary, spoken_word, grammar]
output:
schema: audita-v1
llm:
proposal:
base_url: https://openrouter.ai/api/v1
model: openrouter/google/gemma-4-31b-it
api_key_env: AUDITA_LLM_API_KEY
timeout: 120s
max_retries: 3
validation:
base_url: https://openrouter.ai/api/v1
model: openrouter/google/gemma-4-31b-it
api_key_env: AUDITA_VALIDATION_LLM_API_KEY
timeout: 120
max_retries: 3
concurrency:
total_llm: 2
proposal_llm: 2
validation_llm: 1
chunking:
target_sections: 8
max_section_tokens: 8192
min_section_tokens: 2048
normalization:
max_segment_gap: 4s
ellipsis_gap: 3.5s
max_segment_duration: 60s
max_segment_tokens: 2048
thresholds:
glossary: 0.8
homophones: 0.8
spoken_word: 0.8
grammar: 0.8
context:
description: "General context for domain vocabulary and speaker names."
diagnostics:
work_dir: /tmp/audita
retention: auto

View File

@@ -0,0 +1,6 @@
glossary:
- name: Audita
aliases:
- audita
category: product
summary: The Audita transcript correction CLI.

View File

@@ -0,0 +1,9 @@
[
{
"id": 1,
"speaker": "A",
"start": 0.0,
"end": 1.2,
"text": "hello world"
}
]

View File

@@ -0,0 +1,121 @@
package cli
import (
"flag"
"gitea.maximumdirect.net/eric/audita/internal/core/config"
)
type processOverrideBinding func(*config.CLIOverrides, processFlags)
var processOverrideBindings = map[string]processOverrideBinding{
"modules": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ModulesCSV = flags.modules
},
"output-schema": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.OutputSchema = flags.outputSchema
},
"llm-api-key": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.PrimaryLLMAPIKey = flags.llmAPIKey
},
"validation-llm-api-key": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationLLMAPIKey = flags.validationLLMAPIKey
},
"model": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.PrimaryModel = flags.model
},
"validation-model": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationModel = flags.validationModel
},
"base-url": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.PrimaryBaseURL = flags.baseURL
},
"validation-base-url": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationBaseURL = flags.validationBaseURL
},
"llm-timeout-seconds": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.PrimaryLLMTimeoutSeconds = flags.llmTimeoutSeconds
},
"total-llm-concurrency": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.TotalLLMConcurrency = flags.totalLLMConcurrency
},
"proposal-llm-concurrency": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ProposalLLMConcurrency = flags.proposalLLMConcurrency
},
"llm-concurrency": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.PrimaryLLMConcurrency = flags.llmConcurrency
},
"validation-llm-timeout-seconds": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationLLMTimeoutSeconds = flags.validationLLMTimeoutSeconds
},
"max-retries": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.MaxRetries = flags.maxRetries
},
"validation-max-retries": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationMaxRetries = flags.validationMaxRetries
},
"validation-llm-concurrency": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationLLMConcurrency = flags.validationLLMConcurrency
},
"validation-max-prompt-tokens": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationMaxPromptTokens = flags.validationMaxPromptTokens
},
"max-section-tokens": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.MaxSectionTokens = flags.maxSectionTokens
},
"min-section-tokens": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.MinSectionTokens = flags.minSectionTokens
},
"target-sections": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.TargetSections = flags.targetSections
},
"glossary-confidence-threshold": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.GlossaryConfidenceThreshold = flags.glossaryConfidenceThreshold
},
"grammar-confidence-threshold": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.GrammarConfidenceThreshold = flags.grammarConfidenceThreshold
},
"homophones-confidence-threshold": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.HomophonesConfidenceThreshold = flags.homophonesConfidenceThreshold
},
"spoken-word-confidence-threshold": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.SpokenWordConfidenceThreshold = flags.spokenWordConfidenceThreshold
},
"normalize-max-segment-gap": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.NormalizeMaxSegmentGap = flags.normalizeMaxSegmentGap
},
"normalize-ellipsis-gap": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.NormalizeEllipsisGap = flags.normalizeEllipsisGap
},
"normalize-max-segment-duration": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.NormalizeMaxSegmentDuration = flags.normalizeMaxSegmentDuration
},
"normalize-max-segment-tokens": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.NormalizeMaxSegmentTokens = flags.normalizeMaxSegmentTokens
},
"transcript-description": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.TranscriptDescription = flags.transcriptDescription
},
"work-dir": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.WorkDir = flags.workDir
},
"work-dir-retention": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.WorkDirRetention = flags.workDirRetention
},
}
func processCLIOverrides(fs *flag.FlagSet, flags processFlags) (config.CLIOverrides, bool) {
overrides := config.CLIOverrides{}
explicitModules := false
fs.Visit(func(f *flag.Flag) {
if f.Name == "modules" {
explicitModules = true
}
binding, ok := processOverrideBindings[f.Name]
if !ok {
return
}
binding(&overrides, flags)
})
return overrides, explicitModules
}

View File

@@ -0,0 +1,433 @@
package cli
import (
"io"
"reflect"
"strings"
"testing"
"gitea.maximumdirect.net/eric/audita/internal/core/config"
)
func TestProcessCLIOverridesMapsEveryConfigMutatingFlag(t *testing.T) {
tests := []struct {
name string
flagName string
value string
wantExplicitModules bool
assertOverrideFields func(t *testing.T, overrides config.CLIOverrides)
}{
{
name: "modules",
flagName: "modules",
value: "grammar,glossary",
wantExplicitModules: true,
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "ModulesCSV", overrides.ModulesCSV, "grammar,glossary")
},
},
{
name: "output schema",
flagName: "output-schema",
value: "audita-v1",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "OutputSchema", overrides.OutputSchema, "audita-v1")
},
},
{
name: "primary api key",
flagName: "llm-api-key",
value: "primary-key",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "PrimaryLLMAPIKey", overrides.PrimaryLLMAPIKey, "primary-key")
},
},
{
name: "validation api key",
flagName: "validation-llm-api-key",
value: "validation-key",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "ValidationLLMAPIKey", overrides.ValidationLLMAPIKey, "validation-key")
},
},
{
name: "primary model",
flagName: "model",
value: "primary-model",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "PrimaryModel", overrides.PrimaryModel, "primary-model")
},
},
{
name: "validation model",
flagName: "validation-model",
value: "validation-model",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "ValidationModel", overrides.ValidationModel, "validation-model")
},
},
{
name: "primary base url",
flagName: "base-url",
value: "https://primary.example.test",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "PrimaryBaseURL", overrides.PrimaryBaseURL, "https://primary.example.test")
},
},
{
name: "validation base url",
flagName: "validation-base-url",
value: "https://validation.example.test",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "ValidationBaseURL", overrides.ValidationBaseURL, "https://validation.example.test")
},
},
{
name: "primary timeout",
flagName: "llm-timeout-seconds",
value: "101",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "PrimaryLLMTimeoutSeconds", overrides.PrimaryLLMTimeoutSeconds, 101)
},
},
{
name: "total concurrency",
flagName: "total-llm-concurrency",
value: "5",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "TotalLLMConcurrency", overrides.TotalLLMConcurrency, 5)
},
},
{
name: "proposal concurrency",
flagName: "proposal-llm-concurrency",
value: "3",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "ProposalLLMConcurrency", overrides.ProposalLLMConcurrency, 3)
},
},
{
name: "legacy concurrency alias",
flagName: "llm-concurrency",
value: "4",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "PrimaryLLMConcurrency", overrides.PrimaryLLMConcurrency, 4)
},
},
{
name: "validation timeout",
flagName: "validation-llm-timeout-seconds",
value: "202",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "ValidationLLMTimeoutSeconds", overrides.ValidationLLMTimeoutSeconds, 202)
},
},
{
name: "max retries",
flagName: "max-retries",
value: "6",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "MaxRetries", overrides.MaxRetries, 6)
},
},
{
name: "validation max retries",
flagName: "validation-max-retries",
value: "7",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "ValidationMaxRetries", overrides.ValidationMaxRetries, 7)
},
},
{
name: "validation concurrency",
flagName: "validation-llm-concurrency",
value: "8",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "ValidationLLMConcurrency", overrides.ValidationLLMConcurrency, 8)
},
},
{
name: "validation max prompt tokens",
flagName: "validation-max-prompt-tokens",
value: "4096",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "ValidationMaxPromptTokens", overrides.ValidationMaxPromptTokens, 4096)
},
},
{
name: "max section tokens",
flagName: "max-section-tokens",
value: "9000",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "MaxSectionTokens", overrides.MaxSectionTokens, 9000)
},
},
{
name: "min section tokens",
flagName: "min-section-tokens",
value: "1000",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "MinSectionTokens", overrides.MinSectionTokens, 1000)
},
},
{
name: "target sections",
flagName: "target-sections",
value: "12",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "TargetSections", overrides.TargetSections, 12)
},
},
{
name: "glossary threshold",
flagName: "glossary-confidence-threshold",
value: "0.91",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "GlossaryConfidenceThreshold", overrides.GlossaryConfidenceThreshold, 0.91)
},
},
{
name: "grammar threshold",
flagName: "grammar-confidence-threshold",
value: "0.92",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "GrammarConfidenceThreshold", overrides.GrammarConfidenceThreshold, 0.92)
},
},
{
name: "homophones threshold",
flagName: "homophones-confidence-threshold",
value: "0.93",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "HomophonesConfidenceThreshold", overrides.HomophonesConfidenceThreshold, 0.93)
},
},
{
name: "spoken word threshold",
flagName: "spoken-word-confidence-threshold",
value: "0.94",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "SpokenWordConfidenceThreshold", overrides.SpokenWordConfidenceThreshold, 0.94)
},
},
{
name: "normalize max segment gap",
flagName: "normalize-max-segment-gap",
value: "1.2",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "NormalizeMaxSegmentGap", overrides.NormalizeMaxSegmentGap, 1.2)
},
},
{
name: "normalize ellipsis gap",
flagName: "normalize-ellipsis-gap",
value: "2.3",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "NormalizeEllipsisGap", overrides.NormalizeEllipsisGap, 2.3)
},
},
{
name: "normalize max segment duration",
flagName: "normalize-max-segment-duration",
value: "45.6",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "NormalizeMaxSegmentDuration", overrides.NormalizeMaxSegmentDuration, 45.6)
},
},
{
name: "normalize max segment tokens",
flagName: "normalize-max-segment-tokens",
value: "321",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "NormalizeMaxSegmentTokens", overrides.NormalizeMaxSegmentTokens, 321)
},
},
{
name: "transcript description",
flagName: "transcript-description",
value: "podcast episode",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "TranscriptDescription", overrides.TranscriptDescription, "podcast episode")
},
},
{
name: "work dir",
flagName: "work-dir",
value: "/tmp/custom-audita",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "WorkDir", overrides.WorkDir, "/tmp/custom-audita")
},
},
{
name: "work dir retention",
flagName: "work-dir-retention",
value: "always",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "WorkDirRetention", overrides.WorkDirRetention, "always")
},
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
fs, flags := newProcessFlagSet(config.Default(), io.Discard)
if err := fs.Parse([]string{"--" + tc.flagName, tc.value}); err != nil {
t.Fatalf("parse flag: %v", err)
}
overrides, explicitModules := processCLIOverrides(fs, flags)
if explicitModules != tc.wantExplicitModules {
t.Fatalf("explicitModules=%v, want %v", explicitModules, tc.wantExplicitModules)
}
tc.assertOverrideFields(t, overrides)
})
}
}
func TestProcessCLIOverridesIgnoresNonConfigFlags(t *testing.T) {
fs, flags := newProcessFlagSet(config.Default(), io.Discard)
if err := fs.Parse([]string{
"--config", "/tmp/config.yml",
"--glossary", "/tmp/glossary.yml",
"--output", "/tmp/output.json",
"--report-json", "/tmp/report.json",
}); err != nil {
t.Fatalf("parse flags: %v", err)
}
overrides, explicitModules := processCLIOverrides(fs, flags)
if explicitModules {
t.Fatal("non-config flags should not mark modules explicit")
}
assertNoCLIOverrides(t, overrides)
}
func TestNewProcessFlagSetDefaultsReflectEffectiveConfig(t *testing.T) {
cfg := config.Default()
cfg.Modules = []string{"grammar", "glossary"}
cfg.OutputSchema = "audita-v1"
cfg.PrimaryLLM.APIKey = "primary-key"
cfg.ValidationLLM.APIKey = "validation-key"
cfg.PrimaryLLM.Model = "primary-model"
cfg.ValidationLLM.Model = "validation-model"
cfg.PrimaryLLM.BaseURL = "https://primary.example.test"
cfg.ValidationLLM.BaseURL = "https://validation.example.test"
cfg.PrimaryLLM.TimeoutSeconds = 101
cfg.TotalLLMConcurrency = 5
cfg.ProposalLLMConcurrency = 3
cfg.PrimaryLLM.MaxRetries = 6
cfg.ValidationMaxPromptTokens = 4096
cfg.MaxSectionTokens = 9000
cfg.MinSectionTokens = 1000
cfg.Thresholds.Glossary = 0.91
cfg.Thresholds.Grammar = 0.92
cfg.Thresholds.Homophones = 0.93
cfg.Thresholds.SpokenWord = 0.94
cfg.Normalization.MaxSegmentGap = 1.2
cfg.Normalization.EllipsisGap = 2.3
cfg.Normalization.MaxSegmentDuration = 45.6
cfg.Normalization.MaxSegmentTokens = 321
cfg.TranscriptDescription = "podcast episode"
cfg.WorkDir = "/tmp/custom-audita"
cfg.WorkDirRetention = config.WorkDirRetentionAlways
validationTimeout := 202
validationRetries := 7
validationConcurrency := 8
targetSections := 12
cfg.ValidationLLM.TimeoutSeconds = &validationTimeout
cfg.ValidationLLM.MaxRetries = &validationRetries
cfg.ValidationLLMConcurrency = &validationConcurrency
cfg.TargetSections = &targetSections
_, flags := newProcessFlagSet(cfg, io.Discard)
assertStringOverride(t, "modules default", flags.modules, "grammar,glossary")
assertStringOverride(t, "output schema default", flags.outputSchema, "audita-v1")
assertStringOverride(t, "primary api key default", flags.llmAPIKey, "primary-key")
assertStringOverride(t, "validation api key default", flags.validationLLMAPIKey, "validation-key")
assertStringOverride(t, "primary model default", flags.model, "primary-model")
assertStringOverride(t, "validation model default", flags.validationModel, "validation-model")
assertStringOverride(t, "primary base url default", flags.baseURL, "https://primary.example.test")
assertStringOverride(t, "validation base url default", flags.validationBaseURL, "https://validation.example.test")
assertIntOverride(t, "primary timeout default", flags.llmTimeoutSeconds, 101)
assertIntOverride(t, "total concurrency default", flags.totalLLMConcurrency, 5)
assertIntOverride(t, "proposal concurrency default", flags.proposalLLMConcurrency, 3)
assertIntOverride(t, "legacy concurrency alias default", flags.llmConcurrency, 5)
assertIntOverride(t, "validation timeout default", flags.validationLLMTimeoutSeconds, validationTimeout)
assertIntOverride(t, "max retries default", flags.maxRetries, 6)
assertIntOverride(t, "validation max retries default", flags.validationMaxRetries, validationRetries)
assertIntOverride(t, "validation concurrency default", flags.validationLLMConcurrency, validationConcurrency)
assertIntOverride(t, "validation max prompt tokens default", flags.validationMaxPromptTokens, 4096)
assertIntOverride(t, "max section tokens default", flags.maxSectionTokens, 9000)
assertIntOverride(t, "min section tokens default", flags.minSectionTokens, 1000)
assertIntOverride(t, "target sections default", flags.targetSections, targetSections)
assertFloatOverride(t, "glossary threshold default", flags.glossaryConfidenceThreshold, 0.91)
assertFloatOverride(t, "grammar threshold default", flags.grammarConfidenceThreshold, 0.92)
assertFloatOverride(t, "homophones threshold default", flags.homophonesConfidenceThreshold, 0.93)
assertFloatOverride(t, "spoken word threshold default", flags.spokenWordConfidenceThreshold, 0.94)
assertFloatOverride(t, "normalize max segment gap default", flags.normalizeMaxSegmentGap, 1.2)
assertFloatOverride(t, "normalize ellipsis gap default", flags.normalizeEllipsisGap, 2.3)
assertFloatOverride(t, "normalize max segment duration default", flags.normalizeMaxSegmentDuration, 45.6)
assertIntOverride(t, "normalize max segment tokens default", flags.normalizeMaxSegmentTokens, 321)
assertStringOverride(t, "transcript description default", flags.transcriptDescription, "podcast episode")
assertStringOverride(t, "work dir default", flags.workDir, "/tmp/custom-audita")
assertStringOverride(t, "work dir retention default", flags.workDirRetention, "always")
}
func TestNewProcessFlagSetUsesFallbackDefaultsForUnsetOptionalConfig(t *testing.T) {
cfg := config.Default()
_, flags := newProcessFlagSet(cfg, io.Discard)
assertIntOverride(t, "validation timeout fallback", flags.validationLLMTimeoutSeconds, cfg.PrimaryLLM.TimeoutSeconds)
assertIntOverride(t, "validation retries fallback", flags.validationMaxRetries, cfg.PrimaryLLM.MaxRetries)
assertIntOverride(t, "validation concurrency fallback", flags.validationLLMConcurrency, cfg.TotalLLMConcurrency)
assertIntOverride(t, "target sections fallback", flags.targetSections, 0)
}
func assertStringOverride(t *testing.T, name string, got *string, want string) {
t.Helper()
if got == nil || *got != want {
t.Fatalf("%s=%v, want %q", name, pointerValue(got), want)
}
}
func assertIntOverride(t *testing.T, name string, got *int, want int) {
t.Helper()
if got == nil || *got != want {
t.Fatalf("%s=%v, want %d", name, pointerValue(got), want)
}
}
func assertFloatOverride(t *testing.T, name string, got *float64, want float64) {
t.Helper()
if got == nil || *got != want {
t.Fatalf("%s=%v, want %v", name, pointerValue(got), want)
}
}
func assertNoCLIOverrides(t *testing.T, overrides config.CLIOverrides) {
t.Helper()
value := reflect.ValueOf(overrides)
typ := value.Type()
for i := 0; i < value.NumField(); i++ {
field := value.Field(i)
if field.Kind() != reflect.Ptr {
t.Fatalf("unexpected non-pointer CLIOverrides field %s", typ.Field(i).Name)
}
if !field.IsNil() {
t.Fatalf("expected no CLI overrides, field %s was set", typ.Field(i).Name)
}
}
}
func pointerValue[T any](ptr *T) any {
if ptr == nil {
return "<nil>"
}
if stringer, ok := any(*ptr).(interface{ String() string }); ok {
return strings.TrimSpace(stringer.String())
}
return *ptr
}

View File

@@ -23,6 +23,7 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/llm"
"gitea.maximumdirect.net/eric/audita/internal/framework/modules"
"gitea.maximumdirect.net/eric/audita/internal/framework/processreport"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation"
"gitea.maximumdirect.net/eric/audita/internal/framework/runner"
"gitea.maximumdirect.net/eric/audita/internal/framework/validators"
@@ -374,30 +375,28 @@ func runProcess(args []string, stdout, stderr io.Writer) int {
fmt.Fprintf(stderr, "audita process: invalid CLI configuration: %v\n", err)
return 2
}
configPath, configSource, err := resolveConfigPath(configPathOverride, configPathOverrideSet, os.LookupEnv)
effectiveConfig, err := config.LoadEffectiveConfig(configPathOverride, configPathOverrideSet)
if err != nil {
fmt.Fprintf(stderr, "audita process: %v\n", err)
var effectiveConfigErr *config.EffectiveConfigError
if errors.As(err, &effectiveConfigErr) {
switch effectiveConfigErr.Kind {
case config.EffectiveConfigErrorLoadFile, config.EffectiveConfigErrorApplyFile:
fmt.Fprintf(stderr, "audita process: invalid config file: %v\n", effectiveConfigErr)
case config.EffectiveConfigErrorApplyEnv:
fmt.Fprintf(stderr, "audita process: invalid environment configuration: %v\n", effectiveConfigErr)
default:
fmt.Fprintf(stderr, "audita process: %v\n", effectiveConfigErr)
}
} else {
fmt.Fprintf(stderr, "audita process: %v\n", err)
}
return 2
}
cfg := config.Default()
var configVersion *int
if configPath != "" {
fileCfg, fileErr := config.LoadFileConfig(configPath)
if fileErr != nil {
fmt.Fprintf(stderr, "audita process: invalid config file: %v\n", fileErr)
return 2
}
if applyErr := cfg.ApplyFileConfig(fileCfg); applyErr != nil {
fmt.Fprintf(stderr, "audita process: invalid config file: %v\n", applyErr)
return 2
}
configVersion = &fileCfg.Version
}
if err := cfg.ApplyEnvOverrides(); err != nil {
fmt.Fprintf(stderr, "audita process: invalid environment configuration: %v\n", err)
return 2
}
cfg := effectiveConfig.Config
configPath := effectiveConfig.ConfigPath
configSource := effectiveConfig.ConfigSource
configVersion := effectiveConfig.ConfigVersion
fs, pFlags := newProcessFlagSet(cfg, stderr)
@@ -421,75 +420,7 @@ func runProcess(args []string, stdout, stderr io.Writer) int {
return 2
}
overrides := config.CLIOverrides{}
explicitModules := false
fs.Visit(func(f *flag.Flag) {
switch f.Name {
case "modules":
explicitModules = true
overrides.ModulesCSV = pFlags.modules
case "output-schema":
overrides.OutputSchema = pFlags.outputSchema
case "llm-api-key":
overrides.PrimaryLLMAPIKey = pFlags.llmAPIKey
case "validation-llm-api-key":
overrides.ValidationLLMAPIKey = pFlags.validationLLMAPIKey
case "model":
overrides.PrimaryModel = pFlags.model
case "validation-model":
overrides.ValidationModel = pFlags.validationModel
case "base-url":
overrides.PrimaryBaseURL = pFlags.baseURL
case "validation-base-url":
overrides.ValidationBaseURL = pFlags.validationBaseURL
case "llm-timeout-seconds":
overrides.PrimaryLLMTimeoutSeconds = pFlags.llmTimeoutSeconds
case "total-llm-concurrency":
overrides.TotalLLMConcurrency = pFlags.totalLLMConcurrency
case "proposal-llm-concurrency":
overrides.ProposalLLMConcurrency = pFlags.proposalLLMConcurrency
case "llm-concurrency":
overrides.PrimaryLLMConcurrency = pFlags.llmConcurrency
case "validation-llm-timeout-seconds":
overrides.ValidationLLMTimeoutSeconds = pFlags.validationLLMTimeoutSeconds
case "max-retries":
overrides.MaxRetries = pFlags.maxRetries
case "validation-max-retries":
overrides.ValidationMaxRetries = pFlags.validationMaxRetries
case "validation-llm-concurrency":
overrides.ValidationLLMConcurrency = pFlags.validationLLMConcurrency
case "validation-max-prompt-tokens":
overrides.ValidationMaxPromptTokens = pFlags.validationMaxPromptTokens
case "max-section-tokens":
overrides.MaxSectionTokens = pFlags.maxSectionTokens
case "min-section-tokens":
overrides.MinSectionTokens = pFlags.minSectionTokens
case "target-sections":
overrides.TargetSections = pFlags.targetSections
case "glossary-confidence-threshold":
overrides.GlossaryConfidenceThreshold = pFlags.glossaryConfidenceThreshold
case "grammar-confidence-threshold":
overrides.GrammarConfidenceThreshold = pFlags.grammarConfidenceThreshold
case "homophones-confidence-threshold":
overrides.HomophonesConfidenceThreshold = pFlags.homophonesConfidenceThreshold
case "spoken-word-confidence-threshold":
overrides.SpokenWordConfidenceThreshold = pFlags.spokenWordConfidenceThreshold
case "normalize-max-segment-gap":
overrides.NormalizeMaxSegmentGap = pFlags.normalizeMaxSegmentGap
case "normalize-ellipsis-gap":
overrides.NormalizeEllipsisGap = pFlags.normalizeEllipsisGap
case "normalize-max-segment-duration":
overrides.NormalizeMaxSegmentDuration = pFlags.normalizeMaxSegmentDuration
case "normalize-max-segment-tokens":
overrides.NormalizeMaxSegmentTokens = pFlags.normalizeMaxSegmentTokens
case "transcript-description":
overrides.TranscriptDescription = pFlags.transcriptDescription
case "work-dir":
overrides.WorkDir = pFlags.workDir
case "work-dir-retention":
overrides.WorkDirRetention = pFlags.workDirRetention
}
})
overrides, explicitModules := processCLIOverrides(fs, pFlags)
if err := cfg.ApplyCLIOverrides(overrides); err != nil {
fmt.Fprintf(stderr, "audita process: invalid CLI configuration: %v\n", err)
@@ -530,12 +461,15 @@ func runProcess(args []string, stdout, stderr io.Writer) int {
if runErr != nil {
if runDir != nil && runOutput != nil {
if runOutput.Utilization != nil {
_ = runDir.WriteJSONArtifact("utilization-diagnostics.json", runOutput.Utilization)
_ = runDir.WriteJSONArtifact(diagnostics.ArtifactUtilizationSummary, runOutput.Utilization)
}
_ = runDir.WriteJSONArtifact("correction-ledger.json", buildCorrectionLedger(runDir.Path(), runOutput))
_ = runDir.WriteJSONArtifact(diagnostics.ArtifactCorrectionLedger, processreport.BuildCorrectionLedger(processreport.CorrectionLedgerInput{
RunDirectoryPath: runDir.Path(),
RunOutput: runOutput,
}))
}
errorPhase, errorMessage := extractErrorPhase(runErr)
report := buildProcessReport("failed", inv, runDir, startedAt, completedAt, errorMessage, errorPhase, nil, nil, runOutput)
report := processreport.Build(processReportInput("failed", inv, runDir, startedAt, completedAt, errorMessage, errorPhase, nil, nil, runOutput))
if strings.TrimSpace(inv.ReportJSONPath) != "" {
if err := reporting.WriteProcessReport(inv.ReportJSONPath, report); err != nil {
@@ -559,12 +493,15 @@ func runProcess(args []string, stdout, stderr io.Writer) int {
if runDir != nil && runOutput != nil {
if runOutput.Utilization != nil {
_ = runDir.WriteJSONArtifact("utilization-diagnostics.json", runOutput.Utilization)
_ = runDir.WriteJSONArtifact(diagnostics.ArtifactUtilizationSummary, runOutput.Utilization)
}
_ = runDir.WriteJSONArtifact("correction-ledger.json", buildCorrectionLedger(runDir.Path(), runOutput))
_ = runDir.WriteJSONArtifact(diagnostics.ArtifactCorrectionLedger, processreport.BuildCorrectionLedger(processreport.CorrectionLedgerInput{
RunDirectoryPath: runDir.Path(),
RunOutput: runOutput,
}))
}
report := buildProcessReport("success", inv, runDir, startedAt, completedAt, "", "", normSummary, chunkSummary, runOutput)
report := processreport.Build(processReportInput("success", inv, runDir, startedAt, completedAt, "", "", normSummary, chunkSummary, runOutput))
if strings.TrimSpace(inv.ReportJSONPath) != "" {
if err := reporting.WriteProcessReport(inv.ReportJSONPath, report); err != nil {
@@ -579,21 +516,11 @@ func runProcess(args []string, stdout, stderr io.Writer) int {
}
}
hasSkippedCorrections := false
if runOutput != nil {
for _, mr := range runOutput.ModuleResults {
if len(mr.SkippedChanges) > 0 || len(mr.ValidatorRejected) > 0 {
hasSkippedCorrections = true
break
}
}
}
if runDir != nil {
_ = runDir.WriteReport(report)
if err := runDir.ApplyRetention(diagnostics.RetentionDecisionInput{
RunSucceeded: true,
HasSkippedCorrections: hasSkippedCorrections,
HasSkippedCorrections: processreport.HasSkippedCorrections(runOutput),
}); err != nil {
fmt.Fprintf(stderr, "audita process: failed to apply work-dir retention: %v\n", err)
return 1
@@ -652,6 +579,10 @@ func runConfigValidate(args []string, stdout, stderr io.Writer) int {
fmt.Fprintf(stderr, "audita config validate: %v\n", err)
return 2
}
if err := cfg.Validate(); err != nil {
fmt.Fprintf(stderr, "audita config validate: %v\n", err)
return 2
}
fmt.Fprintln(stdout, "config is valid")
return 0
}
@@ -674,28 +605,13 @@ func runConfigPrintEffective(args []string, stdout, stderr io.Writer) int {
configPathValue := strings.TrimSpace(*configPath)
configPathSet := configPathValue != ""
path, _, err := resolveConfigPath(configPathValue, configPathSet, os.LookupEnv)
effectiveConfig, err := config.LoadEffectiveConfig(configPathValue, configPathSet)
if err != nil {
fmt.Fprintf(stderr, "audita config print-effective: %v\n", err)
return 2
}
cfg := config.Default()
if path != "" {
fileCfg, fileErr := config.LoadFileConfig(path)
if fileErr != nil {
fmt.Fprintf(stderr, "audita config print-effective: %v\n", fileErr)
return 2
}
if applyErr := cfg.ApplyFileConfig(fileCfg); applyErr != nil {
fmt.Fprintf(stderr, "audita config print-effective: %v\n", applyErr)
return 2
}
}
if err := cfg.ApplyEnvOverrides(); err != nil {
fmt.Fprintf(stderr, "audita config print-effective: %v\n", err)
return 2
}
cfg := effectiveConfig.Config
redacted := cfg.Redacted()
out, err := json.MarshalIndent(redacted, "", " ")
@@ -722,142 +638,28 @@ func extractErrorPhase(err error) (phase string, message string) {
return "", msg
}
func buildProcessReport(status string, inv processInvocation, runDir *diagnostics.RunDirectory, startedAt, completedAt time.Time, errorMessage string, errorPhase string, normalizationSummary *normalization.NormalizationSummary, chunkingSummary *chunking.Summary, runOutput *runner.RunOutput) reporting.ProcessReport {
report := reporting.ProcessReport{
ReportMetadata: reporting.ReportMetadata{
ReportSchemaName: reporting.DefaultProcessReportSchemaName,
ReportSchemaVersion: reporting.DefaultProcessReportSchemaVersion,
OutputSchema: inv.Config.OutputSchema,
ConfigVersion: inv.ConfigVersion,
},
Phase: "default_pipeline",
Status: status,
Operation: "process",
TranscriptPath: inv.TranscriptPath,
GlossaryPath: inv.GlossaryPath,
OutputPath: inv.OutputPath,
Modules: append([]string(nil), inv.Config.Modules...),
StartedAt: startedAt,
CompletedAt: &completedAt,
ErrorPhase: errorPhase,
}
func processReportInput(status string, inv processInvocation, runDir *diagnostics.RunDirectory, startedAt, completedAt time.Time, errorMessage string, errorPhase string, normalizationSummary *normalization.NormalizationSummary, chunkingSummary *chunking.Summary, runOutput *runner.RunOutput) processreport.BuildInput {
runDirectoryPath := ""
if runDir != nil {
diagnosticsDir := runDir.Path()
report.Diagnostics = &reporting.DiagnosticsMetadata{
DirectoryPath: diagnosticsDir,
SourceTranscriptPath: filepath.Join(diagnosticsDir, "source-transcript.json"),
ParsedSourceTranscriptPath: filepath.Join(diagnosticsDir, "source-transcript-parsed.json"),
NormalizedTranscriptPath: filepath.Join(diagnosticsDir, "normalized-transcript.json"),
NormalizationSummaryPath: filepath.Join(diagnosticsDir, "normalization-summary.json"),
ChunkingSummaryPath: filepath.Join(diagnosticsDir, "chunking-summary.json"),
UtilizationSummaryPath: filepath.Join(diagnosticsDir, "utilization-diagnostics.json"),
CorrectionLedgerPath: filepath.Join(diagnosticsDir, "correction-ledger.json"),
InvocationMetadataPath: filepath.Join(diagnosticsDir, "invocation.json"),
RedactedEffectiveConfigPath: filepath.Join(diagnosticsDir, "effective-config.json"),
}
if status == "failed" {
report.Diagnostics.ErrorLogPath = filepath.Join(diagnosticsDir, "error.log")
}
runDirectoryPath = runDir.Path()
}
if errorMessage != "" {
report.ErrorMessage = errorMessage
return processreport.BuildInput{
Status: status,
TranscriptPath: inv.TranscriptPath,
GlossaryPath: inv.GlossaryPath,
OutputPath: inv.OutputPath,
Modules: inv.Config.Modules,
OutputSchema: inv.Config.OutputSchema,
ConfigVersion: inv.ConfigVersion,
StartedAt: startedAt,
CompletedAt: completedAt,
ErrorMessage: errorMessage,
ErrorPhase: errorPhase,
RunDirectoryPath: runDirectoryPath,
NormalizationSummary: normalizationSummary,
ChunkingSummary: chunkingSummary,
RunOutput: runOutput,
}
if normalizationSummary != nil {
report.InputSegmentCount = &normalizationSummary.InputSegmentCount
report.NormalizedSegmentCount = &normalizationSummary.OutputSegmentCount
report.NormalizationMerges = &normalizationSummary.MergesPerformed
report.NormalizationIDReassignments = &normalizationSummary.IDsReassigned
report.NormalizationSkipped.DifferentSpeakers = &normalizationSummary.SkippedMerges.DifferentSpeakers
report.NormalizationSkipped.GapTooLarge = &normalizationSummary.SkippedMerges.GapTooLarge
report.NormalizationSkipped.DurationExceeded = &normalizationSummary.SkippedMerges.DurationExceeded
report.NormalizationSkipped.TokenLimitExceeded = &normalizationSummary.SkippedMerges.TokenLimitExceeded
}
if chunkingSummary != nil {
report.Chunking = &reporting.ChunkingSummary{
ChunkCount: chunkingSummary.ChunkCount,
MinEstimatedTokens: chunkingSummary.MinEstimatedTokens,
MaxEstimatedTokens: chunkingSummary.MaxEstimatedTokens,
TotalEstimatedTokens: chunkingSummary.TotalEstimatedTokens,
TargetSections: chunkingSummary.TargetSections,
MaxSectionTokens: chunkingSummary.MaxSectionTokens,
MinSectionTokens: chunkingSummary.MinSectionTokens,
}
}
report.ModulesSummary, report.ModuleResults = buildModuleReporting(runOutput)
return report
}
func buildModuleReporting(runOutput *runner.RunOutput) (*reporting.ModulesSummary, []reporting.ModuleReport) {
if runOutput == nil || len(runOutput.ModuleResults) == 0 {
return nil, nil
}
moduleReports := make([]reporting.ModuleReport, 0, len(runOutput.ModuleResults))
summary := &reporting.ModulesSummary{ModuleCount: len(runOutput.ModuleResults)}
for _, r := range runOutput.ModuleResults {
startedAt := r.StartedAt
completedAt := r.CompletedAt
moduleReports = append(moduleReports, reporting.ModuleReport{
ModuleKey: r.ModuleKey,
ModuleInstance: r.ModuleInstance,
ReplacementPolicy: string(r.ReplacementPolicy),
Status: r.Status,
ProposalCount: r.ProposalCount,
ValidatorDecisions: mapValidatorDecisions(r.ValidatorDecisions),
ValidatorRejected: mapValidatorRejected(r.ValidatorRejected),
AppliedChanges: r.AppliedChanges,
SkippedChanges: r.SkippedChanges,
ErrorMessage: r.ErrorMessage,
StartedAt: &startedAt,
CompletedAt: &completedAt,
})
summary.TotalAppliedChanges += len(r.AppliedChanges)
summary.TotalSkippedChanges += len(r.SkippedChanges) + len(r.ValidatorRejected)
if r.Status == runner.ModuleStatusFailed && summary.FailedModuleInstance == "" {
summary.FailedModuleInstance = r.ModuleInstance
}
}
return summary, moduleReports
}
func mapValidatorDecisions(in []runner.ValidatorDecisionRecord) []reporting.ValidatorDecisionReport {
if len(in) == 0 {
return nil
}
out := make([]reporting.ValidatorDecisionReport, len(in))
for i, d := range in {
out[i] = reporting.ValidatorDecisionReport{
ValidatorName: d.ValidatorName,
ProposalIndex: d.ProposalIndex,
Approved: d.Approved,
ReasonCode: d.ReasonCode,
Message: d.Message,
DiagnosticArtifactPath: d.DiagnosticArtifactPath,
}
}
return out
}
func mapValidatorRejected(in []runner.ValidatorRejectedChange) []reporting.ValidatorRejectedReport {
if len(in) == 0 {
return nil
}
out := make([]reporting.ValidatorRejectedReport, len(in))
for i, d := range in {
out[i] = reporting.ValidatorRejectedReport{
ValidatorName: d.ValidatorName,
ProposalIndex: d.ProposalIndex,
ModuleKey: d.ModuleKey,
ModuleInstance: d.ModuleInstance,
TargetSegmentID: d.TargetSegmentID,
OriginalText: d.OriginalText,
CorrectedText: d.CorrectedText,
ReasonCode: d.ReasonCode,
Message: d.Message,
}
}
return out
}
type processFlags struct {
@@ -982,47 +784,6 @@ func findConfigPathOverride(args []string) (path string, set bool, err error) {
return "", false, nil
}
var statConfigPath = os.Stat
func resolveConfigPath(cliConfigPath string, cliConfigPathSet bool, lookup func(string) (string, bool)) (path string, source string, err error) {
if cliConfigPathSet {
path = strings.TrimSpace(cliConfigPath)
if path == "" {
return "", "", fmt.Errorf("--config requires a non-empty path")
}
if _, statErr := statConfigPath(path); statErr != nil {
if os.IsNotExist(statErr) {
return "", "", fmt.Errorf("config file not found: %s", path)
}
return "", "", fmt.Errorf("cannot access config file %s: %w", path, statErr)
}
return path, "flag", nil
}
if raw, ok := lookup("AUDITA_CONFIG"); ok {
path = strings.TrimSpace(raw)
if path == "" {
return "", "", fmt.Errorf("AUDITA_CONFIG must not be empty")
}
if _, statErr := statConfigPath(path); statErr != nil {
if os.IsNotExist(statErr) {
return "", "", fmt.Errorf("config file not found: %s", path)
}
return "", "", fmt.Errorf("cannot access config file %s: %w", path, statErr)
}
return path, "env", nil
}
for _, defaultPath := range config.DefaultConfigSearchPaths {
if _, statErr := statConfigPath(defaultPath); statErr == nil {
return defaultPath, "default", nil
} else if !os.IsNotExist(statErr) {
return "", "", fmt.Errorf("cannot access config file %s: %w", defaultPath, statErr)
}
}
return "", "", nil
}
func isHelpCommand(args []string) bool {
if len(args) == 0 {
return false

View File

@@ -21,11 +21,11 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/llm"
"gitea.maximumdirect.net/eric/audita/internal/framework/modules"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/runner"
"gitea.maximumdirect.net/eric/audita/internal/framework/validators"
"gitea.maximumdirect.net/eric/audita/internal/testsupport"
)
func TestRunRootHelp(t *testing.T) {
@@ -99,66 +99,6 @@ func TestRunProcessHelpListsExpectedFlags(t *testing.T) {
}
}
func TestResolveConfigPathDefaultIgnoredWhenMissing(t *testing.T) {
lookup := func(string) (string, bool) { return "", false }
path, source, err := resolveConfigPath("", false, lookup)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if path != "" || source != "" {
t.Fatalf("expected no config path/source, got path=%q source=%q", path, source)
}
}
func TestResolveConfigPathDefaultPrefersUsrLocalOverEtc(t *testing.T) {
oldStat := statConfigPath
statConfigPath = func(path string) (os.FileInfo, error) {
if path == config.DefaultConfigPathUsrLocal || path == config.DefaultConfigPath {
return nil, nil
}
return nil, os.ErrNotExist
}
t.Cleanup(func() { statConfigPath = oldStat })
lookup := func(string) (string, bool) { return "", false }
path, source, err := resolveConfigPath("", false, lookup)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if source != "default" {
t.Fatalf("expected default source, got %q", source)
}
if path != config.DefaultConfigPathUsrLocal {
t.Fatalf("expected %q, got %q", config.DefaultConfigPathUsrLocal, path)
}
}
func TestResolveConfigPathDefaultFallsBackToEtc(t *testing.T) {
oldStat := statConfigPath
statConfigPath = func(path string) (os.FileInfo, error) {
if path == config.DefaultConfigPathUsrLocal {
return nil, os.ErrNotExist
}
if path == config.DefaultConfigPath {
return nil, nil
}
return nil, os.ErrNotExist
}
t.Cleanup(func() { statConfigPath = oldStat })
lookup := func(string) (string, bool) { return "", false }
path, source, err := resolveConfigPath("", false, lookup)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if source != "default" {
t.Fatalf("expected default source, got %q", source)
}
if path != config.DefaultConfigPath {
t.Fatalf("expected %q, got %q", config.DefaultConfigPath, path)
}
}
func TestRunConfigValidateSuccess(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
@@ -210,6 +150,23 @@ func TestRunConfigValidateUnknownField(t *testing.T) {
}
}
func TestRunConfigValidateUnsupportedModuleKey(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
cfgPath := writeFile(t, "config.yml", "version: 1\npipeline:\n modules: [made_up]\n")
exitCode := Run([]string{"config", "validate", "--config", cfgPath}, &stdout, &stderr)
if exitCode == 0 {
t.Fatalf("expected failure for unsupported module key")
}
if stdout.Len() != 0 {
t.Fatalf("expected empty stdout on failure, got %q", stdout.String())
}
if !strings.Contains(stderr.String(), "unsupported module key") {
t.Fatalf("expected unsupported module key error, got %q", stderr.String())
}
}
func TestRunConfigPrintEffectiveOutputsRedactedJSON(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
@@ -242,6 +199,45 @@ func TestRunConfigPrintEffectiveOutputsRedactedJSON(t *testing.T) {
}
}
func TestRunConfigPrintEffectiveAppliesFileThenEnvironment(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
cfgPath := writeFile(t, "config.yml", "version: 1\nllm:\n proposal:\n model: file-model\n")
t.Setenv("AUDITA_MODEL", "env-model")
exitCode := Run([]string{"config", "print-effective", "--config", cfgPath}, &stdout, &stderr)
if exitCode != 0 {
t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
}
var out struct {
PrimaryLLM struct {
Model string `json:"Model"`
} `json:"PrimaryLLM"`
}
if err := json.Unmarshal(stdout.Bytes(), &out); err != nil {
t.Fatalf("expected valid JSON output, got error: %v output=%q", err, stdout.String())
}
if out.PrimaryLLM.Model != "env-model" {
t.Fatalf("expected env model override in print-effective output, got %q", out.PrimaryLLM.Model)
}
}
func TestRunConfigValidateIgnoresEnvironmentOverrides(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
cfgPath := writeFile(t, "config.yml", "version: 1\n")
t.Setenv("AUDITA_MODULES", "made_up")
exitCode := Run([]string{"config", "validate", "--config", cfgPath}, &stdout, &stderr)
if exitCode != 0 {
t.Fatalf("expected success because config validate is file-only, got %d stderr=%q", exitCode, stderr.String())
}
if !strings.Contains(stdout.String(), "config is valid") {
t.Fatalf("expected success message, got %q", stdout.String())
}
}
func TestRunConfigCommandDoesNotRequireTranscriptOrGlossary(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
@@ -369,8 +365,8 @@ diagnostics:
func TestRunProcessEnvOverridesConfigFile(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{
key: "m",
"grammar": fakeModule{
key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -392,7 +388,7 @@ func TestRunProcessEnvOverridesConfigFile(t *testing.T) {
cfgPath := writeFile(t, "config.yml", `
version: 1
pipeline:
modules: [m]
modules: [grammar]
llm:
proposal:
model: file-model
@@ -404,7 +400,7 @@ llm:
fixturePath("tiny_transcript.json"),
"--glossary", fixturePath("tiny_glossary.yaml"),
"--config", cfgPath,
"--modules", "m",
"--modules", "grammar",
}, &stdout, &stderr)
if exitCode != 0 {
t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
@@ -413,8 +409,8 @@ llm:
func TestRunProcessCLIOverridesEnvAndConfigFile(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{
key: "m",
"grammar": fakeModule{
key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -436,7 +432,7 @@ func TestRunProcessCLIOverridesEnvAndConfigFile(t *testing.T) {
cfgPath := writeFile(t, "config.yml", `
version: 1
pipeline:
modules: [m]
modules: [grammar]
llm:
proposal:
model: file-model
@@ -448,7 +444,7 @@ llm:
fixturePath("tiny_transcript.json"),
"--glossary", fixturePath("tiny_glossary.yaml"),
"--config", cfgPath,
"--modules", "m",
"--modules", "grammar",
"--model", "cli-model",
}, &stdout, &stderr)
if exitCode != 0 {
@@ -495,8 +491,8 @@ diagnostics:
func TestRunProcessTranscriptDescriptionCLIOverridesConfigFileContextDescription(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{
key: "m",
"grammar": fakeModule{
key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -517,7 +513,7 @@ func TestRunProcessTranscriptDescriptionCLIOverridesConfigFileContextDescription
cfgPath := writeFile(t, "config.yml", `
version: 1
pipeline:
modules: [m]
modules: [grammar]
context:
description: "file transcript description"
`)
@@ -528,7 +524,7 @@ context:
fixturePath("tiny_transcript.json"),
"--glossary", fixturePath("tiny_glossary.yaml"),
"--config", cfgPath,
"--modules", "m",
"--modules", "grammar",
"--transcript-description", "cli transcript description",
}, &stdout, &stderr)
if exitCode != 0 {
@@ -692,8 +688,8 @@ func TestRunProcessCLIOverridesEnvironment(t *testing.T) {
func TestRunProcessTranscriptDescriptionDefaultEmpty(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{
key: "m",
"grammar": fakeModule{
key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -720,7 +716,7 @@ func TestRunProcessTranscriptDescriptionDefaultEmpty(t *testing.T) {
exitCode := Run([]string{
"process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "m",
"--modules", "grammar",
}, &stdout, &stderr)
if exitCode != 0 {
t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
@@ -729,8 +725,8 @@ func TestRunProcessTranscriptDescriptionDefaultEmpty(t *testing.T) {
func TestRunProcessTranscriptDescriptionCLIOverrideAndTrim(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{
key: "m",
"grammar": fakeModule{
key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -757,7 +753,7 @@ func TestRunProcessTranscriptDescriptionCLIOverrideAndTrim(t *testing.T) {
exitCode := Run([]string{
"process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "m",
"--modules", "grammar",
"--transcript-description", " speaker background context ",
}, &stdout, &stderr)
if exitCode != 0 {
@@ -823,8 +819,8 @@ func TestRunProcessRejectsValidationConcurrencyAboveTotalConcurrency(t *testing.
func TestRunProcessTotalLLMConcurrencyDrivesEffectiveValidationConcurrencyWhenUnset(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{
key: "m",
"grammar": fakeModule{
key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -865,7 +861,7 @@ func TestRunProcessTotalLLMConcurrencyDrivesEffectiveValidationConcurrencyWhenUn
"--glossary",
fixturePath("tiny_glossary.yaml"),
"--modules",
"m",
"grammar",
"--total-llm-concurrency",
"4",
}, &stdout, &stderr)
@@ -908,8 +904,8 @@ func TestRunProcessRejectsProposalConcurrencyAboveTotalConcurrency(t *testing.T)
func TestRunProcessLegacyLLMConcurrencyAliasSetsTotalAndProposal(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{
key: "m",
"grammar": fakeModule{
key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -944,7 +940,7 @@ func TestRunProcessLegacyLLMConcurrencyAliasSetsTotalAndProposal(t *testing.T) {
"--glossary",
fixturePath("tiny_glossary.yaml"),
"--modules",
"m",
"grammar",
"--llm-concurrency",
"3",
}, &stdout, &stderr)
@@ -955,8 +951,8 @@ func TestRunProcessLegacyLLMConcurrencyAliasSetsTotalAndProposal(t *testing.T) {
func TestRunProcessLLMConcurrencyFlagsOverrideEnvironment(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{
key: "m",
"grammar": fakeModule{
key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -997,7 +993,7 @@ func TestRunProcessLLMConcurrencyFlagsOverrideEnvironment(t *testing.T) {
"--glossary",
fixturePath("tiny_glossary.yaml"),
"--modules",
"m",
"grammar",
"--total-llm-concurrency",
"4",
"--proposal-llm-concurrency",
@@ -1010,8 +1006,8 @@ func TestRunProcessLLMConcurrencyFlagsOverrideEnvironment(t *testing.T) {
func TestRunProcessAcceptsLLMConcurrencyEnvironmentVariables(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{
key: "m",
"grammar": fakeModule{
key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -1052,7 +1048,7 @@ func TestRunProcessAcceptsLLMConcurrencyEnvironmentVariables(t *testing.T) {
"--glossary",
fixturePath("tiny_glossary.yaml"),
"--modules",
"m",
"grammar",
}, &stdout, &stderr)
if exitCode != 0 {
t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
@@ -1793,11 +1789,12 @@ type fakeModule struct {
func (m fakeModule) Key() string { return m.key }
func (m fakeModule) ReplacementPolicy() proposals.ReplacementPolicy { return m.policy }
func (m fakeModule) Validators() []contracts.Validator { return m.validators }
func (m fakeModule) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
func (m fakeModule) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
if m.proposeF == nil {
return nil, nil
return contracts.ProposalResult{}, nil
}
return m.proposeF(req)
proposalsOut, err := m.proposeF(req)
return contracts.ProposalResult{Proposals: proposalsOut}, err
}
type fakeValidator struct {
@@ -1882,12 +1879,12 @@ func TestRunProcessInjectedFactoryExecutesRunnerAndReportsModules(t *testing.T)
return validators.Result{ValidatorName: "allow", Decisions: decisions}, nil
}}
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m1": fakeModule{key: "m1", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{allow}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
"glossary": fakeModule{key: "glossary", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{allow}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
return []proposals.CorrectionProposal{
{TargetSegmentID: 1, OriginalText: "Hello", CorrectedText: "Hi", Confidence: 1},
}, nil
}},
"m2": fakeModule{key: "m2", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{allow}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
"homophones": fakeModule{key: "homophones", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{allow}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
if req.WorkingTranscript.Segments[0].Text != "Hi world" {
t.Fatalf("expected module 2 to see module 1 changes, got %q", req.WorkingTranscript.Segments[0].Text)
}
@@ -1910,7 +1907,7 @@ func TestRunProcessInjectedFactoryExecutesRunnerAndReportsModules(t *testing.T)
exitCode := Run([]string{
"process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "m1,m2",
"--modules", "glossary,homophones",
"--output", outputPath,
"--report-json", reportPath,
"--work-dir", workDir,
@@ -1969,7 +1966,7 @@ func TestRunProcessInjectedFactoryLLMValidatorResultsInReports(t *testing.T) {
}},
}
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{key: "m", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{llmValidator}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
"grammar": fakeModule{key: "grammar", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{llmValidator}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
return []proposals.CorrectionProposal{{TargetSegmentID: 1, OriginalText: "Hello", CorrectedText: "Hi", Confidence: 1}}, nil
}},
}}
@@ -1989,7 +1986,7 @@ func TestRunProcessInjectedFactoryLLMValidatorResultsInReports(t *testing.T) {
exitCode := Run([]string{
"process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "m",
"--modules", "grammar",
"--output", outputPath,
"--report-json", reportPath,
"--work-dir", workDir,
@@ -2010,7 +2007,7 @@ func TestRunProcessInjectedFactoryLLMValidatorResultsInReports(t *testing.T) {
func TestRunProcessInjectedFactorySkippedKeepsAutoRetention(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m1": fakeModule{key: "m1", policy: proposals.ReplacementPolicyRequireUnique, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
"glossary": fakeModule{key: "glossary", policy: proposals.ReplacementPolicyRequireUnique, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
return []proposals.CorrectionProposal{
{TargetSegmentID: 1, OriginalText: "word", CorrectedText: "term", Confidence: 1},
}, nil
@@ -2026,7 +2023,7 @@ func TestRunProcessInjectedFactorySkippedKeepsAutoRetention(t *testing.T) {
exitCode := Run([]string{
"process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "m1",
"--modules", "glossary",
"--work-dir", workDir,
"--work-dir-retention", "auto",
}, &stdout, &stderr)
@@ -2040,7 +2037,7 @@ func TestRunProcessInjectedFactorySkippedKeepsAutoRetention(t *testing.T) {
func TestRunProcessInjectedFactoryFailureWritesFailedReport(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m1": fakeModule{key: "m1", policy: proposals.ReplacementPolicyRequireUnique, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
"glossary": fakeModule{key: "glossary", policy: proposals.ReplacementPolicyRequireUnique, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
return nil, errors.New("test failure")
}},
}}
@@ -2056,7 +2053,7 @@ func TestRunProcessInjectedFactoryFailureWritesFailedReport(t *testing.T) {
exitCode := Run([]string{
"process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "m1",
"--modules", "glossary",
"--work-dir", workDir,
"--work-dir-retention", "always",
"--report-json", reportPath,
@@ -2088,14 +2085,8 @@ func TestRunProcessInjectedFactoryFailureWritesFailedReport(t *testing.T) {
}
}
func TestRunProcessProductionRegistryUnsupportedModuleFailsCleanly(t *testing.T) {
cfg := modules.Dependencies{}
processModuleFactory = modules.NewFactory(cfg)
t.Cleanup(func() { processModuleFactory = nil })
func TestRunProcessUnsupportedModuleFailsDuringConfigValidation(t *testing.T) {
var stdout, stderr bytes.Buffer
workDir := t.TempDir()
reportPath := filepath.Join(t.TempDir(), "report.json")
transcriptPath := writeFile(t, "transcript.json", `[
{"id":1,"speaker":"Alice","start":0.0,"end":1.0,"text":"Hello"}
]`)
@@ -2104,9 +2095,6 @@ func TestRunProcessProductionRegistryUnsupportedModuleFailsCleanly(t *testing.T)
"process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "made_up",
"--work-dir", workDir,
"--work-dir-retention", "always",
"--report-json", reportPath,
}, &stdout, &stderr)
if exitCode == 0 {
t.Fatal("expected failure exit code")
@@ -2114,23 +2102,12 @@ func TestRunProcessProductionRegistryUnsupportedModuleFailsCleanly(t *testing.T)
if stdout.Len() != 0 {
t.Fatalf("expected empty stdout on failure, got %q", stdout.String())
}
if !strings.Contains(stderr.String(), "runner_execution") {
t.Fatalf("expected runner_execution failure on stderr, got %q", stderr.String())
if !strings.Contains(stderr.String(), "invalid CLI configuration") {
t.Fatalf("expected config validation failure on stderr, got %q", stderr.String())
}
if !strings.Contains(stderr.String(), "unsupported module key") {
t.Fatalf("expected explicit unsupported module message, got %q", stderr.String())
}
report := readProcessReport(t, reportPath)
if report.Status != "failed" {
t.Fatalf("expected failed report status, got %q", report.Status)
}
if report.ErrorPhase != "runner_execution" {
t.Fatalf("expected runner_execution phase, got %q", report.ErrorPhase)
}
if !strings.Contains(report.ErrorMessage, "unsupported module key") {
t.Fatalf("expected report error message to mention unsupported module, got %q", report.ErrorMessage)
}
}
func TestRunProcessExplicitUnsupportedModulesFailClearly(t *testing.T) {
@@ -2384,7 +2361,7 @@ func TestRunProcessExplicitGrammarRejectedAndApplicationSkipAreDistinct(t *testi
}
}
func TestRunProcessExplicitGrammarMalformedLLMOutputFailsWithErrorLog(t *testing.T) {
func TestRunProcessExplicitGrammarMalformedLLMOutputSucceedsWithWarning(t *testing.T) {
processProposalLLMClient = &fakeStructuredLLMClient{err: errors.New("malformed structured output")}
t.Cleanup(func() { processProposalLLMClient = nil })
@@ -2403,22 +2380,32 @@ func TestRunProcessExplicitGrammarMalformedLLMOutputFailsWithErrorLog(t *testing
"--work-dir-retention", "always",
"--report-json", reportPath,
}, &stdout, &stderr)
if exitCode == 0 {
t.Fatal("expected failure")
if exitCode != 0 {
t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
}
if stdout.Len() != 0 {
t.Fatalf("expected empty stdout on failure, got %q", stdout.String())
if stderr.Len() != 0 {
t.Fatalf("expected empty stderr on success, got %q", stderr.String())
}
if !strings.Contains(stderr.String(), "runner_execution") {
t.Fatalf("expected runner_execution error, got %q", stderr.String())
parsed, err := schema.ParseTranscriptJSON(stdout.Bytes())
if err != nil {
t.Fatalf("expected transcript stdout on success: %v", err)
}
if len(parsed.Segments) != 1 || parsed.Segments[0].Text != "hello" {
t.Fatalf("expected unchanged transcript, got %+v", parsed.Segments)
}
runDir := onlyRunDir(t, workDir)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err != nil {
t.Fatalf("expected error.log on failed grammar run: %v", err)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err == nil || !os.IsNotExist(err) {
t.Fatalf("did not expect error.log on successful grammar run: %v", err)
}
report := readProcessReport(t, reportPath)
if report.Status != "failed" || report.ErrorPhase != "runner_execution" {
t.Fatalf("expected failed runner_execution report, got %+v", report)
if report.Status != "success" || report.ErrorPhase != "" {
t.Fatalf("expected successful report, got %+v", report)
}
if len(report.ModuleResults) != 1 || len(report.ModuleResults[0].Warnings) != 1 {
t.Fatalf("expected one module warning, got %+v", report.ModuleResults)
}
if report.ModuleResults[0].Warnings[0].ReasonCode != "proposal_response_malformed" {
t.Fatalf("unexpected warning: %+v", report.ModuleResults[0].Warnings[0])
}
}
@@ -3041,7 +3028,7 @@ func TestRunProcessExplicitGlossaryRepeatedStagesUseDeterministicInstanceNamesAn
}
}
func TestRunProcessExplicitGlossaryMalformedLLMOutputFailsWithErrorLog(t *testing.T) {
func TestRunProcessExplicitGlossaryMalformedLLMOutputSucceedsWithWarning(t *testing.T) {
processProposalLLMClient = &fakeStructuredLLMClient{err: errors.New("malformed structured output")}
t.Cleanup(func() { processProposalLLMClient = nil })
@@ -3060,22 +3047,29 @@ func TestRunProcessExplicitGlossaryMalformedLLMOutputFailsWithErrorLog(t *testin
"--work-dir-retention", "always",
"--report-json", reportPath,
}, &stdout, &stderr)
if exitCode == 0 {
t.Fatal("expected failure")
if exitCode != 0 {
t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
}
if stdout.Len() != 0 {
t.Fatalf("expected empty stdout on failure, got %q", stdout.String())
if stderr.Len() != 0 {
t.Fatalf("expected empty stderr on success, got %q", stderr.String())
}
if !strings.Contains(stderr.String(), "runner_execution") {
t.Fatalf("expected runner_execution error, got %q", stderr.String())
parsed, err := schema.ParseTranscriptJSON(stdout.Bytes())
if err != nil {
t.Fatalf("expected transcript stdout on success: %v", err)
}
if len(parsed.Segments) != 1 || parsed.Segments[0].Text != "hello" {
t.Fatalf("expected unchanged transcript, got %+v", parsed.Segments)
}
runDir := onlyRunDir(t, workDir)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err != nil {
t.Fatalf("expected error.log on failed glossary run: %v", err)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err == nil || !os.IsNotExist(err) {
t.Fatalf("did not expect error.log on successful glossary run: %v", err)
}
report := readProcessReport(t, reportPath)
if report.Status != "failed" || report.ErrorPhase != "runner_execution" {
t.Fatalf("expected failed runner_execution report, got %+v", report)
if report.Status != "success" || report.ErrorPhase != "" {
t.Fatalf("expected successful report, got %+v", report)
}
if len(report.ModuleResults) != 1 || len(report.ModuleResults[0].Warnings) != 1 {
t.Fatalf("expected one module warning, got %+v", report.ModuleResults)
}
}
@@ -3295,7 +3289,7 @@ func TestRunProcessExplicitHomophonesProtectedGlossaryTermRejected(t *testing.T)
}
}
func TestRunProcessExplicitHomophonesMalformedLLMOutputFailsWithErrorLog(t *testing.T) {
func TestRunProcessExplicitHomophonesMalformedLLMOutputSucceedsWithWarning(t *testing.T) {
processProposalLLMClient = &fakeStructuredLLMClient{err: errors.New("malformed structured output")}
t.Cleanup(func() { processProposalLLMClient = nil })
@@ -3314,22 +3308,29 @@ func TestRunProcessExplicitHomophonesMalformedLLMOutputFailsWithErrorLog(t *test
"--work-dir-retention", "always",
"--report-json", reportPath,
}, &stdout, &stderr)
if exitCode == 0 {
t.Fatal("expected failure")
if exitCode != 0 {
t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
}
if stdout.Len() != 0 {
t.Fatalf("expected empty stdout on failure, got %q", stdout.String())
if stderr.Len() != 0 {
t.Fatalf("expected empty stderr on success, got %q", stderr.String())
}
if !strings.Contains(stderr.String(), "runner_execution") {
t.Fatalf("expected runner_execution error, got %q", stderr.String())
parsed, err := schema.ParseTranscriptJSON(stdout.Bytes())
if err != nil {
t.Fatalf("expected transcript stdout on success: %v", err)
}
if len(parsed.Segments) != 1 || parsed.Segments[0].Text != "hello" {
t.Fatalf("expected unchanged transcript, got %+v", parsed.Segments)
}
runDir := onlyRunDir(t, workDir)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err != nil {
t.Fatalf("expected error.log on failed homophones run: %v", err)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err == nil || !os.IsNotExist(err) {
t.Fatalf("did not expect error.log on successful homophones run: %v", err)
}
report := readProcessReport(t, reportPath)
if report.Status != "failed" || report.ErrorPhase != "runner_execution" {
t.Fatalf("expected failed runner_execution report, got %+v", report)
if report.Status != "success" || report.ErrorPhase != "" {
t.Fatalf("expected successful report, got %+v", report)
}
if len(report.ModuleResults) != 1 || len(report.ModuleResults[0].Warnings) != 1 {
t.Fatalf("expected one module warning, got %+v", report.ModuleResults)
}
}
@@ -3674,7 +3675,7 @@ func TestRunProcessExplicitSpokenWordProtectedGlossaryTermRejected(t *testing.T)
}
}
func TestRunProcessExplicitSpokenWordMalformedLLMOutputFailsWithErrorLog(t *testing.T) {
func TestRunProcessExplicitSpokenWordMalformedLLMOutputSucceedsWithWarning(t *testing.T) {
processProposalLLMClient = &fakeStructuredLLMClient{err: errors.New("malformed structured output")}
t.Cleanup(func() { processProposalLLMClient = nil })
@@ -3693,22 +3694,29 @@ func TestRunProcessExplicitSpokenWordMalformedLLMOutputFailsWithErrorLog(t *test
"--work-dir-retention", "always",
"--report-json", reportPath,
}, &stdout, &stderr)
if exitCode == 0 {
t.Fatal("expected failure")
if exitCode != 0 {
t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
}
if stdout.Len() != 0 {
t.Fatalf("expected empty stdout on failure, got %q", stdout.String())
if stderr.Len() != 0 {
t.Fatalf("expected empty stderr on success, got %q", stderr.String())
}
if !strings.Contains(stderr.String(), "runner_execution") {
t.Fatalf("expected runner_execution error, got %q", stderr.String())
parsed, err := schema.ParseTranscriptJSON(stdout.Bytes())
if err != nil {
t.Fatalf("expected transcript stdout on success: %v", err)
}
if len(parsed.Segments) != 1 || parsed.Segments[0].Text != "hello" {
t.Fatalf("expected unchanged transcript, got %+v", parsed.Segments)
}
runDir := onlyRunDir(t, workDir)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err != nil {
t.Fatalf("expected error.log on failed spoken_word run: %v", err)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err == nil || !os.IsNotExist(err) {
t.Fatalf("did not expect error.log on successful spoken_word run: %v", err)
}
report := readProcessReport(t, reportPath)
if report.Status != "failed" || report.ErrorPhase != "runner_execution" {
t.Fatalf("expected failed runner_execution report, got %+v", report)
if report.Status != "success" || report.ErrorPhase != "" {
t.Fatalf("expected successful report, got %+v", report)
}
if len(report.ModuleResults) != 1 || len(report.ModuleResults[0].Warnings) != 1 {
t.Fatalf("expected one module warning, got %+v", report.ModuleResults)
}
}
@@ -4328,12 +4336,7 @@ func writeFile(t *testing.T, name string, content string) string {
}
func readFile(t *testing.T, path string) []byte {
t.Helper()
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("failed to read file %q: %v", path, err)
}
return data
return testsupport.ReadFile(t, path)
}
func readProcessReport(t *testing.T, path string) reporting.ProcessReport {
@@ -4347,13 +4350,5 @@ func readProcessReport(t *testing.T, path string) reporting.ProcessReport {
}
func onlyRunDir(t *testing.T, workDir string) string {
t.Helper()
entries, err := os.ReadDir(workDir)
if err != nil {
t.Fatalf("failed to read work dir %q: %v", workDir, err)
}
if len(entries) != 1 {
t.Fatalf("expected exactly one run dir in %q, got %d", workDir, len(entries))
}
return filepath.Join(workDir, entries[0].Name())
return testsupport.OnlyRunDir(t, workDir)
}

View File

@@ -78,25 +78,22 @@ func (c *subprocessTestLLMClient) CompleteStructured(ctx context.Context, req co
}
}
case "mid_pipeline_fail":
switch target := out.(type) {
case *proposal_generation.StructuredCorrectionSet:
if _, ok := out.(*proposal_generation.StructuredCorrectionSet); ok {
c.mu.Lock()
c.proposals++
proposalCall := c.proposals
c.mu.Unlock()
if proposalCall >= 3 {
*target = proposal_generation.StructuredCorrectionSet{
Corrections: []proposal_generation.StructuredCorrectionProposal{
{TargetSegmentID: 0, OriginalText: "x", CorrectedText: "y", Confidence: 0.99},
},
}
} else {
*target = proposal_generation.StructuredCorrectionSet{
Corrections: []proposal_generation.StructuredCorrectionProposal{
{TargetSegmentID: 1, OriginalText: "Segment", CorrectedText: "Segment", Confidence: 0.99},
},
}
return contracts.StructuredCompletionResponse{}, errors.New("synthetic mid-pipeline failure")
}
}
switch target := out.(type) {
case *proposal_generation.StructuredCorrectionSet:
*target = proposal_generation.StructuredCorrectionSet{
Corrections: []proposal_generation.StructuredCorrectionProposal{
{TargetSegmentID: 1, OriginalText: "Segment", CorrectedText: "Segment", Confidence: 0.99},
},
}
case *validators.LLMValidationResponse:
*target = validators.LLMValidationResponse{Validations: nil}

View File

@@ -0,0 +1,171 @@
package config
import "strings"
type llmTargetPatch struct {
apiKey *string
model *string
baseURL *string
timeoutSeconds *int
maxRetries *int
}
type concurrencyPatch struct {
totalLLM *int
legacyTotalLLM *int
proposalLLM *int
validationLLM *int
inheritProposal bool
allowLegacyAlias bool
}
type chunkingPatch struct {
targetSections *int
maxSectionTokens *int
minSectionTokens *int
}
type thresholdsPatch struct {
glossary *float64
grammar *float64
homophones *float64
spokenWord *float64
}
type normalizationPatch struct {
maxSegmentGap *float64
ellipsisGap *float64
maxSegmentDuration *float64
maxSegmentTokens *int
}
type contextPatch struct {
transcriptDescription *string
}
type diagnosticsPatch struct {
workDir *string
workDirRetention *string
}
func (c *Config) applyPrimaryLLMTargetPatch(patch llmTargetPatch) {
if patch.apiKey != nil {
c.PrimaryLLM.APIKey = *patch.apiKey
}
if patch.model != nil {
c.PrimaryLLM.Model = *patch.model
}
if patch.baseURL != nil {
c.PrimaryLLM.BaseURL = *patch.baseURL
}
if patch.timeoutSeconds != nil {
c.PrimaryLLM.TimeoutSeconds = *patch.timeoutSeconds
}
if patch.maxRetries != nil {
c.PrimaryLLM.MaxRetries = *patch.maxRetries
}
}
func (c *Config) applyValidationLLMTargetPatch(patch llmTargetPatch) {
if patch.apiKey != nil {
c.ValidationLLM.APIKey = *patch.apiKey
}
if patch.model != nil {
c.ValidationLLM.Model = *patch.model
}
if patch.baseURL != nil {
c.ValidationLLM.BaseURL = *patch.baseURL
}
if patch.timeoutSeconds != nil {
value := *patch.timeoutSeconds
c.ValidationLLM.TimeoutSeconds = &value
}
if patch.maxRetries != nil {
value := *patch.maxRetries
c.ValidationLLM.MaxRetries = &value
}
}
func (c *Config) applyConcurrencyPatch(patch concurrencyPatch) {
totalSet := false
if patch.totalLLM != nil {
c.TotalLLMConcurrency = *patch.totalLLM
totalSet = true
}
if patch.allowLegacyAlias && patch.legacyTotalLLM != nil && !totalSet {
c.TotalLLMConcurrency = *patch.legacyTotalLLM
totalSet = true
}
proposalSet := false
if patch.proposalLLM != nil {
c.ProposalLLMConcurrency = *patch.proposalLLM
proposalSet = true
}
if patch.inheritProposal && totalSet && !proposalSet {
c.ProposalLLMConcurrency = c.TotalLLMConcurrency
}
if patch.validationLLM != nil {
value := *patch.validationLLM
c.ValidationLLMConcurrency = &value
}
}
func (c *Config) applyChunkingPatch(patch chunkingPatch) {
if patch.targetSections != nil {
value := *patch.targetSections
c.TargetSections = &value
}
if patch.maxSectionTokens != nil {
c.MaxSectionTokens = *patch.maxSectionTokens
}
if patch.minSectionTokens != nil {
c.MinSectionTokens = *patch.minSectionTokens
}
}
func (c *Config) applyThresholdsPatch(patch thresholdsPatch) {
if patch.glossary != nil {
c.Thresholds.Glossary = *patch.glossary
}
if patch.grammar != nil {
c.Thresholds.Grammar = *patch.grammar
}
if patch.homophones != nil {
c.Thresholds.Homophones = *patch.homophones
}
if patch.spokenWord != nil {
c.Thresholds.SpokenWord = *patch.spokenWord
}
}
func (c *Config) applyNormalizationPatch(patch normalizationPatch) {
if patch.maxSegmentGap != nil {
c.Normalization.MaxSegmentGap = *patch.maxSegmentGap
}
if patch.ellipsisGap != nil {
c.Normalization.EllipsisGap = *patch.ellipsisGap
}
if patch.maxSegmentDuration != nil {
c.Normalization.MaxSegmentDuration = *patch.maxSegmentDuration
}
if patch.maxSegmentTokens != nil {
c.Normalization.MaxSegmentTokens = *patch.maxSegmentTokens
}
}
func (c *Config) applyContextPatch(patch contextPatch) {
if patch.transcriptDescription != nil {
c.TranscriptDescription = strings.TrimSpace(*patch.transcriptDescription)
}
}
func (c *Config) applyDiagnosticsPatch(patch diagnosticsPatch) {
if patch.workDir != nil {
c.WorkDir = *patch.workDir
}
if patch.workDirRetention != nil {
c.WorkDirRetention = WorkDirRetention(*patch.workDirRetention)
}
}

View File

@@ -3,6 +3,8 @@ package config
import (
"fmt"
"strings"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
)
type WorkDirRetention string
@@ -14,7 +16,7 @@ const (
)
const (
DefaultModulesCSV = "glossary,homophones,glossary,spoken_word,grammar"
DefaultModulesCSV = modulecatalog.KeyGlossary + "," + modulecatalog.KeyHomophones + "," + modulecatalog.KeyGlossary + "," + modulecatalog.KeySpokenWord + "," + modulecatalog.KeyGrammar
DefaultOutputSchema = "bare-segments"
DefaultPrimaryModel = "openrouter/google/gemma-4-31b-it"
DefaultPrimaryBaseURL = "https://openrouter.ai/api/v1"

View File

@@ -4,6 +4,9 @@ import (
"reflect"
"strings"
"testing"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
"gitea.maximumdirect.net/eric/audita/internal/core/outputschema"
)
func TestDefaultConfigValues(t *testing.T) {
@@ -236,6 +239,186 @@ func TestApplyCLIOverridesTrimsTranscriptDescription(t *testing.T) {
}
}
func TestConfigSourcesApplySharedEffectiveFieldsConsistently(t *testing.T) {
fileCfg := mustParseFileConfigYAML(t, `
version: 1
output:
schema: " audita-v1 "
llm:
proposal:
base_url: https://proposal.example.test/v1
model: provider/proposal
timeout: 101
max_retries: 5
validation:
base_url: https://validation.example.test/v1
model: provider/validation
timeout: 202
max_retries: 6
chunking:
target_sections: 7
max_section_tokens: 9000
min_section_tokens: 1000
thresholds:
glossary: 0.91
grammar: 0.92
homophones: 0.93
spoken_word: 0.94
normalization:
max_segment_gap: 1.2
ellipsis_gap: 2.3
max_segment_duration: 45.6
max_segment_tokens: 321
context:
description: " shared context "
diagnostics:
work_dir: /tmp/audita-shared
retention: always
`)
tests := []struct {
name string
apply func(*Config) error
}{
{
name: "file",
apply: func(cfg *Config) error {
return cfg.applyFileConfigWithLookup(fileCfg, mapLookup(map[string]string{}))
},
},
{
name: "env",
apply: func(cfg *Config) error {
return cfg.applyEnvOverrides(mapLookup(map[string]string{
"AUDITA_MODEL": "provider/proposal",
"AUDITA_BASE_URL": "https://proposal.example.test/v1",
"AUDITA_LLM_TIMEOUT_SECONDS": "101",
"AUDITA_MAX_RETRIES": "5",
"AUDITA_VALIDATION_MODEL": "provider/validation",
"AUDITA_VALIDATION_BASE_URL": "https://validation.example.test/v1",
"AUDITA_VALIDATION_LLM_TIMEOUT_SECONDS": "202",
"AUDITA_VALIDATION_MAX_RETRIES": "6",
"AUDITA_TARGET_SECTIONS": "7",
"AUDITA_MAX_SECTION_TOKENS": "9000",
"AUDITA_MIN_SECTION_TOKENS": "1000",
"AUDITA_GLOSSARY_CONFIDENCE_THRESHOLD": "0.91",
"AUDITA_GRAMMAR_CONFIDENCE_THRESHOLD": "0.92",
"AUDITA_HOMOPHONES_CONFIDENCE_THRESHOLD": "0.93",
"AUDITA_SPOKEN_WORD_CONFIDENCE_THRESHOLD": "0.94",
"AUDITA_NORMALIZE_MAX_SEGMENT_GAP": "1.2",
"AUDITA_NORMALIZE_ELLIPSIS_GAP": "2.3",
"AUDITA_NORMALIZE_MAX_SEGMENT_DURATION": "45.6",
"AUDITA_NORMALIZE_MAX_SEGMENT_TOKENS": "321",
"AUDITA_WORK_DIR": "/tmp/audita-shared",
"AUDITA_WORK_DIR_RETENTION": "always",
}))
},
},
{
name: "cli",
apply: func(cfg *Config) error {
outputSchema := " audita-v1 "
proposalModel := "provider/proposal"
proposalBaseURL := "https://proposal.example.test/v1"
proposalTimeout := 101
proposalMaxRetries := 5
validationModel := "provider/validation"
validationBaseURL := "https://validation.example.test/v1"
validationTimeout := 202
validationMaxRetries := 6
targetSections := 7
maxSectionTokens := 9000
minSectionTokens := 1000
glossaryThreshold := 0.91
grammarThreshold := 0.92
homophonesThreshold := 0.93
spokenWordThreshold := 0.94
normalizeMaxSegmentGap := 1.2
normalizeEllipsisGap := 2.3
normalizeMaxSegmentDuration := 45.6
normalizeMaxSegmentTokens := 321
description := " shared context "
workDir := "/tmp/audita-shared"
workDirRetention := "always"
return cfg.ApplyCLIOverrides(CLIOverrides{
OutputSchema: &outputSchema,
PrimaryModel: &proposalModel,
PrimaryBaseURL: &proposalBaseURL,
PrimaryLLMTimeoutSeconds: &proposalTimeout,
MaxRetries: &proposalMaxRetries,
ValidationModel: &validationModel,
ValidationBaseURL: &validationBaseURL,
ValidationLLMTimeoutSeconds: &validationTimeout,
ValidationMaxRetries: &validationMaxRetries,
TargetSections: &targetSections,
MaxSectionTokens: &maxSectionTokens,
MinSectionTokens: &minSectionTokens,
GlossaryConfidenceThreshold: &glossaryThreshold,
GrammarConfidenceThreshold: &grammarThreshold,
HomophonesConfidenceThreshold: &homophonesThreshold,
SpokenWordConfidenceThreshold: &spokenWordThreshold,
NormalizeMaxSegmentGap: &normalizeMaxSegmentGap,
NormalizeEllipsisGap: &normalizeEllipsisGap,
NormalizeMaxSegmentDuration: &normalizeMaxSegmentDuration,
NormalizeMaxSegmentTokens: &normalizeMaxSegmentTokens,
TranscriptDescription: &description,
WorkDir: &workDir,
WorkDirRetention: &workDirRetention,
})
},
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
cfg := Default()
if err := tc.apply(&cfg); err != nil {
t.Fatalf("apply config source: %v", err)
}
assertSharedEffectiveFields(t, cfg, sharedEffectiveFieldOptions{
wantOutputSchemaOverride: tc.name != "env",
wantTranscriptDescriptionPatch: tc.name != "env",
})
})
}
}
func TestCLIAPIKeyOverrideIsDirectValue(t *testing.T) {
cfg := Default()
apiKey := "NOT_AN_ENV_VAR_NAME"
validationAPIKey := "also direct"
if err := cfg.ApplyCLIOverrides(CLIOverrides{PrimaryLLMAPIKey: &apiKey, ValidationLLMAPIKey: &validationAPIKey}); err != nil {
t.Fatalf("ApplyCLIOverrides failed: %v", err)
}
if cfg.PrimaryLLM.APIKey != apiKey {
t.Fatalf("expected direct primary api key, got %q", cfg.PrimaryLLM.APIKey)
}
if cfg.ValidationLLM.APIKey != validationAPIKey {
t.Fatalf("expected direct validation api key, got %q", cfg.ValidationLLM.APIKey)
}
}
func TestApplyFileConfigTotalConcurrencyDoesNotChangeProposalWhenProposalUnset(t *testing.T) {
fileCfg := mustParseFileConfigYAML(t, `
version: 1
concurrency:
total_llm: 4
`)
cfg := Default()
if err := cfg.applyFileConfigWithLookup(fileCfg, mapLookup(map[string]string{})); err != nil {
t.Fatalf("applyFileConfigWithLookup failed: %v", err)
}
if cfg.TotalLLMConcurrency != 4 {
t.Fatalf("expected file total concurrency 4, got %d", cfg.TotalLLMConcurrency)
}
if cfg.ProposalLLMConcurrency != DefaultLLMConcurrency {
t.Fatalf("expected file config to preserve proposal concurrency when unset, got %d", cfg.ProposalLLMConcurrency)
}
}
func TestValidationRejectsOverlyLongTranscriptDescription(t *testing.T) {
cfg := Default()
cfg.TranscriptDescription = strings.Repeat("a", DefaultTranscriptDescriptionMaxChars+1)
@@ -318,6 +501,38 @@ func TestValidationFailures(t *testing.T) {
}
}
func TestValidationRejectsUnsupportedModuleKey(t *testing.T) {
cfg := Default()
cfg.Modules = []string{modulecatalog.KeyGlossary, "made_up"}
err := cfg.Validate()
if err == nil {
t.Fatalf("expected validation error for unsupported module key")
}
if !strings.Contains(err.Error(), `unsupported module key "made_up"`) {
t.Fatalf("expected unsupported module key error, got %q", err.Error())
}
}
func TestValidationAllowsRepeatedSupportedModuleKeys(t *testing.T) {
cfg := Default()
cfg.Modules = []string{modulecatalog.KeyGlossary, modulecatalog.KeyGlossary, modulecatalog.KeyGrammar}
if err := cfg.Validate(); err != nil {
t.Fatalf("expected repeated supported module keys to validate, got %v", err)
}
}
func TestValidationAcceptsAllSupportedOutputSchemas(t *testing.T) {
for _, schemaKey := range outputschema.SupportedKeys() {
cfg := Default()
cfg.OutputSchema = schemaKey
if err := cfg.Validate(); err != nil {
t.Fatalf("expected output schema %q to validate, got %v", schemaKey, err)
}
}
}
func TestEffectiveValidationLLMInheritance(t *testing.T) {
cfg := Default()
cfg.PrimaryLLM.APIKey = "primary-key"
@@ -421,3 +636,70 @@ func mapLookup(values map[string]string) func(string) (string, bool) {
return value, ok
}
}
func mustParseFileConfigYAML(t *testing.T, raw string) FileConfig {
t.Helper()
fileCfg, err := ParseFileConfigYAML([]byte(raw))
if err != nil {
t.Fatalf("ParseFileConfigYAML failed: %v", err)
}
return fileCfg
}
type sharedEffectiveFieldOptions struct {
wantOutputSchemaOverride bool
wantTranscriptDescriptionPatch bool
}
func assertSharedEffectiveFields(t *testing.T, cfg Config, opts sharedEffectiveFieldOptions) {
t.Helper()
wantOutputSchema := DefaultOutputSchema
if opts.wantOutputSchemaOverride {
wantOutputSchema = "audita-v1"
}
if cfg.OutputSchema != wantOutputSchema {
t.Fatalf("unexpected output schema: %q", cfg.OutputSchema)
}
if cfg.PrimaryLLM.Model != "provider/proposal" ||
cfg.PrimaryLLM.BaseURL != "https://proposal.example.test/v1" ||
cfg.PrimaryLLM.TimeoutSeconds != 101 ||
cfg.PrimaryLLM.MaxRetries != 5 {
t.Fatalf("unexpected primary llm config: %+v", cfg.PrimaryLLM)
}
if cfg.ValidationLLM.Model != "provider/validation" ||
cfg.ValidationLLM.BaseURL != "https://validation.example.test/v1" ||
cfg.ValidationLLM.TimeoutSeconds == nil ||
*cfg.ValidationLLM.TimeoutSeconds != 202 ||
cfg.ValidationLLM.MaxRetries == nil ||
*cfg.ValidationLLM.MaxRetries != 6 {
t.Fatalf("unexpected validation llm config: %+v", cfg.ValidationLLM)
}
if cfg.TargetSections == nil || *cfg.TargetSections != 7 ||
cfg.MaxSectionTokens != 9000 ||
cfg.MinSectionTokens != 1000 {
t.Fatalf("unexpected chunking config: target=%v max=%d min=%d", cfg.TargetSections, cfg.MaxSectionTokens, cfg.MinSectionTokens)
}
if cfg.Thresholds.Glossary != 0.91 ||
cfg.Thresholds.Grammar != 0.92 ||
cfg.Thresholds.Homophones != 0.93 ||
cfg.Thresholds.SpokenWord != 0.94 {
t.Fatalf("unexpected thresholds: %+v", cfg.Thresholds)
}
if cfg.Normalization.MaxSegmentGap != 1.2 ||
cfg.Normalization.EllipsisGap != 2.3 ||
cfg.Normalization.MaxSegmentDuration != 45.6 ||
cfg.Normalization.MaxSegmentTokens != 321 {
t.Fatalf("unexpected normalization: %+v", cfg.Normalization)
}
wantDescription := ""
if opts.wantTranscriptDescriptionPatch {
wantDescription = "shared context"
}
if cfg.TranscriptDescription != wantDescription {
t.Fatalf("unexpected transcript description: %q", cfg.TranscriptDescription)
}
if cfg.WorkDir != "/tmp/audita-shared" ||
cfg.WorkDirRetention != WorkDirRetentionAlways {
t.Fatalf("unexpected diagnostics config: work_dir=%q retention=%q", cfg.WorkDir, cfg.WorkDirRetention)
}
}

View File

@@ -0,0 +1,119 @@
package config
import (
"fmt"
"os"
"strings"
)
type EffectiveConfigErrorKind string
const (
EffectiveConfigErrorResolvePath EffectiveConfigErrorKind = "resolve_path"
EffectiveConfigErrorLoadFile EffectiveConfigErrorKind = "load_file"
EffectiveConfigErrorApplyFile EffectiveConfigErrorKind = "apply_file"
EffectiveConfigErrorApplyEnv EffectiveConfigErrorKind = "apply_env"
)
type EffectiveConfigError struct {
Kind EffectiveConfigErrorKind
Err error
}
func (e *EffectiveConfigError) Error() string {
if e == nil || e.Err == nil {
return ""
}
return e.Err.Error()
}
func (e *EffectiveConfigError) Unwrap() error {
if e == nil {
return nil
}
return e.Err
}
type EffectiveConfig struct {
Config Config
ConfigPath string
ConfigSource string
ConfigVersion *int
}
func ResolveConfigPath(cliConfigPath string, cliConfigPathSet bool) (path string, source string, err error) {
return resolveConfigPathWithLookup(cliConfigPath, cliConfigPathSet, os.LookupEnv, os.Stat, DefaultConfigSearchPaths)
}
func LoadEffectiveConfig(cliConfigPath string, cliConfigPathSet bool) (EffectiveConfig, error) {
return loadEffectiveConfigWithLookup(cliConfigPath, cliConfigPathSet, os.LookupEnv, os.Stat, DefaultConfigSearchPaths)
}
func loadEffectiveConfigWithLookup(cliConfigPath string, cliConfigPathSet bool, lookup func(string) (string, bool), statPath func(string) (os.FileInfo, error), defaultSearchPaths []string) (EffectiveConfig, error) {
configPath, configSource, err := resolveConfigPathWithLookup(cliConfigPath, cliConfigPathSet, lookup, statPath, defaultSearchPaths)
if err != nil {
return EffectiveConfig{}, &EffectiveConfigError{Kind: EffectiveConfigErrorResolvePath, Err: err}
}
cfg := Default()
var configVersion *int
if configPath != "" {
fileCfg, fileErr := LoadFileConfig(configPath)
if fileErr != nil {
return EffectiveConfig{}, &EffectiveConfigError{Kind: EffectiveConfigErrorLoadFile, Err: fileErr}
}
if applyErr := cfg.ApplyFileConfig(fileCfg); applyErr != nil {
return EffectiveConfig{}, &EffectiveConfigError{Kind: EffectiveConfigErrorApplyFile, Err: applyErr}
}
configVersion = &fileCfg.Version
}
if applyEnvErr := cfg.applyEnvOverrides(lookup); applyEnvErr != nil {
return EffectiveConfig{}, &EffectiveConfigError{Kind: EffectiveConfigErrorApplyEnv, Err: applyEnvErr}
}
return EffectiveConfig{
Config: cfg,
ConfigPath: configPath,
ConfigSource: configSource,
ConfigVersion: configVersion,
}, nil
}
func resolveConfigPathWithLookup(cliConfigPath string, cliConfigPathSet bool, lookup func(string) (string, bool), statPath func(string) (os.FileInfo, error), defaultSearchPaths []string) (path string, source string, err error) {
if cliConfigPathSet {
path = strings.TrimSpace(cliConfigPath)
if path == "" {
return "", "", fmt.Errorf("--config requires a non-empty path")
}
if _, statErr := statPath(path); statErr != nil {
if os.IsNotExist(statErr) {
return "", "", fmt.Errorf("config file not found: %s", path)
}
return "", "", fmt.Errorf("cannot access config file %s: %w", path, statErr)
}
return path, "flag", nil
}
if raw, ok := lookup("AUDITA_CONFIG"); ok {
path = strings.TrimSpace(raw)
if path == "" {
return "", "", fmt.Errorf("AUDITA_CONFIG must not be empty")
}
if _, statErr := statPath(path); statErr != nil {
if os.IsNotExist(statErr) {
return "", "", fmt.Errorf("config file not found: %s", path)
}
return "", "", fmt.Errorf("cannot access config file %s: %w", path, statErr)
}
return path, "env", nil
}
for _, defaultPath := range defaultSearchPaths {
if _, statErr := statPath(defaultPath); statErr == nil {
return defaultPath, "default", nil
} else if !os.IsNotExist(statErr) {
return "", "", fmt.Errorf("cannot access config file %s: %w", defaultPath, statErr)
}
}
return "", "", nil
}

View File

@@ -0,0 +1,163 @@
package config
import (
"os"
"path/filepath"
"strings"
"testing"
)
func TestResolveConfigPathWithLookupMatrix(t *testing.T) {
statFor := func(existing map[string]bool) func(string) (os.FileInfo, error) {
return func(path string) (os.FileInfo, error) {
if existing[path] {
return nil, nil
}
return nil, os.ErrNotExist
}
}
tests := []struct {
name string
cliPath string
cliPathSet bool
lookup func(string) (string, bool)
stat func(string) (os.FileInfo, error)
defaultSearchPaths []string
wantPath string
wantSource string
wantErrContains string
}{
{
name: "explicit config path",
cliPath: "/tmp/explicit.yml",
cliPathSet: true,
lookup: func(string) (string, bool) { return "", false },
stat: statFor(map[string]bool{"/tmp/explicit.yml": true}),
defaultSearchPaths: []string{
"/usr/local/etc/audita/config.yml",
"/etc/audita/config.yml",
},
wantPath: "/tmp/explicit.yml",
wantSource: "flag",
},
{
name: "env config path",
cliPathSet: false,
lookup: func(key string) (string, bool) {
if key == "AUDITA_CONFIG" {
return "/tmp/from-env.yml", true
}
return "", false
},
stat: statFor(map[string]bool{"/tmp/from-env.yml": true}),
defaultSearchPaths: []string{"/usr/local/etc/audita/config.yml", "/etc/audita/config.yml"},
wantPath: "/tmp/from-env.yml",
wantSource: "env",
},
{
name: "default search path",
cliPathSet: false,
lookup: func(string) (string, bool) { return "", false },
stat: statFor(map[string]bool{
"/usr/local/etc/audita/config.yml": true,
"/etc/audita/config.yml": true,
}),
defaultSearchPaths: []string{"/usr/local/etc/audita/config.yml", "/etc/audita/config.yml"},
wantPath: "/usr/local/etc/audita/config.yml",
wantSource: "default",
},
{
name: "explicit missing path",
cliPath: "/tmp/missing.yml",
cliPathSet: true,
lookup: func(string) (string, bool) { return "", false },
stat: statFor(map[string]bool{}),
defaultSearchPaths: []string{
"/usr/local/etc/audita/config.yml",
"/etc/audita/config.yml",
},
wantErrContains: "config file not found",
},
{
name: "missing env path",
cliPathSet: false,
lookup: func(key string) (string, bool) {
if key == "AUDITA_CONFIG" {
return "/tmp/missing-from-env.yml", true
}
return "", false
},
stat: statFor(map[string]bool{}),
defaultSearchPaths: []string{"/usr/local/etc/audita/config.yml", "/etc/audita/config.yml"},
wantErrContains: "config file not found",
},
{
name: "missing default paths",
cliPathSet: false,
lookup: func(string) (string, bool) { return "", false },
stat: statFor(map[string]bool{}),
defaultSearchPaths: []string{
"/usr/local/etc/audita/config.yml",
"/etc/audita/config.yml",
},
wantPath: "",
wantSource: "",
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
gotPath, gotSource, err := resolveConfigPathWithLookup(tc.cliPath, tc.cliPathSet, tc.lookup, tc.stat, tc.defaultSearchPaths)
if tc.wantErrContains != "" {
if err == nil || !strings.Contains(err.Error(), tc.wantErrContains) {
t.Fatalf("expected error containing %q, got %v", tc.wantErrContains, err)
}
return
}
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if gotPath != tc.wantPath || gotSource != tc.wantSource {
t.Fatalf("unexpected result: got path=%q source=%q, want path=%q source=%q", gotPath, gotSource, tc.wantPath, tc.wantSource)
}
})
}
}
func TestLoadEffectiveConfigWithLookupAppliesDefaultsFileThenEnv(t *testing.T) {
tempDir := t.TempDir()
configPath := filepath.Join(tempDir, "config.yml")
configYAML := "version: 1\nllm:\n proposal:\n model: file-model\n"
if err := os.WriteFile(configPath, []byte(configYAML), 0o644); err != nil {
t.Fatalf("write config file: %v", err)
}
lookup := func(key string) (string, bool) {
switch key {
case "AUDITA_CONFIG":
return configPath, true
case "AUDITA_MODEL":
return "env-model", true
default:
return "", false
}
}
result, err := loadEffectiveConfigWithLookup("", false, lookup, os.Stat, DefaultConfigSearchPaths)
if err != nil {
t.Fatalf("loadEffectiveConfigWithLookup error: %v", err)
}
if result.ConfigPath != configPath {
t.Fatalf("unexpected config path: %q", result.ConfigPath)
}
if result.ConfigSource != "env" {
t.Fatalf("unexpected config source: %q", result.ConfigSource)
}
if result.ConfigVersion == nil || *result.ConfigVersion != SupportedFileConfigVersion {
t.Fatalf("unexpected config version: %#v", result.ConfigVersion)
}
if result.Config.PrimaryLLM.Model != "env-model" {
t.Fatalf("expected env override to win over file value, got %q", result.Config.PrimaryLLM.Model)
}
}

View File

@@ -7,8 +7,8 @@ import (
)
const (
DefaultConfigPath = "/etc/audita/config.yml"
DefaultConfigPathUsrLocal = "/usr/local/etc/audita/config.yml"
DefaultConfigPath = "/etc/audita/config.yml"
DefaultConfigPathUsrLocal = "/usr/local/etc/audita/config.yml"
)
var DefaultConfigSearchPaths = []string{
@@ -50,100 +50,93 @@ func (c *Config) applyEnvOverrides(lookup func(string) (string, bool)) error {
cfg.Modules = modules
}
primaryLLM := llmTargetPatch{}
if raw, ok := lookup("AUDITA_LLM_API_KEY"); ok {
cfg.PrimaryLLM.APIKey = raw
primaryLLM.apiKey = &raw
} else if raw, ok := lookup("OPENROUTER_API_KEY"); ok {
cfg.PrimaryLLM.APIKey = raw
primaryLLM.apiKey = &raw
}
if raw, ok := lookup("AUDITA_VALIDATION_LLM_API_KEY"); ok {
cfg.ValidationLLM.APIKey = raw
}
if raw, ok := lookup("AUDITA_MODEL"); ok {
cfg.PrimaryLLM.Model = raw
primaryLLM.model = &raw
}
if raw, ok := lookup("AUDITA_VALIDATION_MODEL"); ok {
cfg.ValidationLLM.Model = raw
}
if raw, ok := lookup("AUDITA_BASE_URL"); ok {
cfg.PrimaryLLM.BaseURL = raw
primaryLLM.baseURL = &raw
}
if raw, ok := lookup("AUDITA_VALIDATION_BASE_URL"); ok {
cfg.ValidationLLM.BaseURL = raw
}
if raw, ok := lookup("AUDITA_LLM_TIMEOUT_SECONDS"); ok {
value, err := parseInt(raw)
if err != nil {
return fmt.Errorf("AUDITA_LLM_TIMEOUT_SECONDS: %w", err)
}
cfg.PrimaryLLM.TimeoutSeconds = value
primaryLLM.timeoutSeconds = &value
}
if raw, ok := lookup("AUDITA_VALIDATION_LLM_TIMEOUT_SECONDS"); ok {
value, err := parseInt(raw)
if err != nil {
return fmt.Errorf("AUDITA_VALIDATION_LLM_TIMEOUT_SECONDS: %w", err)
}
cfg.ValidationLLM.TimeoutSeconds = &value
}
if raw, ok := lookup("AUDITA_MAX_RETRIES"); ok {
value, err := parseInt(raw)
if err != nil {
return fmt.Errorf("AUDITA_MAX_RETRIES: %w", err)
}
cfg.PrimaryLLM.MaxRetries = value
primaryLLM.maxRetries = &value
}
cfg.applyPrimaryLLMTargetPatch(primaryLLM)
validationLLM := llmTargetPatch{}
if raw, ok := lookup("AUDITA_VALIDATION_LLM_API_KEY"); ok {
validationLLM.apiKey = &raw
}
if raw, ok := lookup("AUDITA_VALIDATION_MODEL"); ok {
validationLLM.model = &raw
}
if raw, ok := lookup("AUDITA_VALIDATION_BASE_URL"); ok {
validationLLM.baseURL = &raw
}
if raw, ok := lookup("AUDITA_VALIDATION_LLM_TIMEOUT_SECONDS"); ok {
value, err := parseInt(raw)
if err != nil {
return fmt.Errorf("AUDITA_VALIDATION_LLM_TIMEOUT_SECONDS: %w", err)
}
validationLLM.timeoutSeconds = &value
}
if raw, ok := lookup("AUDITA_VALIDATION_MAX_RETRIES"); ok {
value, err := parseInt(raw)
if err != nil {
return fmt.Errorf("AUDITA_VALIDATION_MAX_RETRIES: %w", err)
}
validationLLM.maxRetries = &value
}
cfg.applyValidationLLMTargetPatch(validationLLM)
concurrency := concurrencyPatch{
inheritProposal: true,
allowLegacyAlias: true,
}
totalConcurrencySet := false
if raw, ok := lookup("AUDITA_TOTAL_LLM_CONCURRENCY"); ok {
value, err := parseInt(raw)
if err != nil {
return fmt.Errorf("AUDITA_TOTAL_LLM_CONCURRENCY: %w", err)
}
cfg.TotalLLMConcurrency = value
totalConcurrencySet = true
concurrency.totalLLM = &value
}
if raw, ok := lookup("AUDITA_LLM_CONCURRENCY"); ok {
value, err := parseInt(raw)
if err != nil {
return fmt.Errorf("AUDITA_LLM_CONCURRENCY: %w", err)
}
if !totalConcurrencySet {
cfg.TotalLLMConcurrency = value
totalConcurrencySet = true
}
concurrency.legacyTotalLLM = &value
}
proposalConcurrencySet := false
if raw, ok := lookup("AUDITA_PROPOSAL_LLM_CONCURRENCY"); ok {
value, err := parseInt(raw)
if err != nil {
return fmt.Errorf("AUDITA_PROPOSAL_LLM_CONCURRENCY: %w", err)
}
cfg.ProposalLLMConcurrency = value
proposalConcurrencySet = true
}
if totalConcurrencySet && !proposalConcurrencySet {
cfg.ProposalLLMConcurrency = cfg.TotalLLMConcurrency
}
if raw, ok := lookup("AUDITA_VALIDATION_MAX_RETRIES"); ok {
value, err := parseInt(raw)
if err != nil {
return fmt.Errorf("AUDITA_VALIDATION_MAX_RETRIES: %w", err)
}
cfg.ValidationLLM.MaxRetries = &value
concurrency.proposalLLM = &value
}
if raw, ok := lookup("AUDITA_VALIDATION_LLM_CONCURRENCY"); ok {
value, err := parseInt(raw)
if err != nil {
return fmt.Errorf("AUDITA_VALIDATION_LLM_CONCURRENCY: %w", err)
}
cfg.ValidationLLMConcurrency = &value
concurrency.validationLLM = &value
}
cfg.applyConcurrencyPatch(concurrency)
if raw, ok := lookup("AUDITA_VALIDATION_MAX_PROMPT_TOKENS"); ok {
value, err := parseInt(raw)
@@ -153,12 +146,13 @@ func (c *Config) applyEnvOverrides(lookup func(string) (string, bool)) error {
cfg.ValidationMaxPromptTokens = value
}
chunking := chunkingPatch{}
if raw, ok := lookup("AUDITA_MAX_SECTION_TOKENS"); ok {
value, err := parseInt(raw)
if err != nil {
return fmt.Errorf("AUDITA_MAX_SECTION_TOKENS: %w", err)
}
cfg.MaxSectionTokens = value
chunking.maxSectionTokens = &value
}
if raw, ok := lookup("AUDITA_MIN_SECTION_TOKENS"); ok {
@@ -166,7 +160,7 @@ func (c *Config) applyEnvOverrides(lookup func(string) (string, bool)) error {
if err != nil {
return fmt.Errorf("AUDITA_MIN_SECTION_TOKENS: %w", err)
}
cfg.MinSectionTokens = value
chunking.minSectionTokens = &value
}
if raw, ok := lookup("AUDITA_TARGET_SECTIONS"); ok {
@@ -174,73 +168,80 @@ func (c *Config) applyEnvOverrides(lookup func(string) (string, bool)) error {
if err != nil {
return fmt.Errorf("AUDITA_TARGET_SECTIONS: %w", err)
}
cfg.TargetSections = &value
chunking.targetSections = &value
}
cfg.applyChunkingPatch(chunking)
thresholds := thresholdsPatch{}
if raw, ok := lookup("AUDITA_GLOSSARY_CONFIDENCE_THRESHOLD"); ok {
value, err := parseFloat(raw)
if err != nil {
return fmt.Errorf("AUDITA_GLOSSARY_CONFIDENCE_THRESHOLD: %w", err)
}
cfg.Thresholds.Glossary = value
thresholds.glossary = &value
}
if raw, ok := lookup("AUDITA_GRAMMAR_CONFIDENCE_THRESHOLD"); ok {
value, err := parseFloat(raw)
if err != nil {
return fmt.Errorf("AUDITA_GRAMMAR_CONFIDENCE_THRESHOLD: %w", err)
}
cfg.Thresholds.Grammar = value
thresholds.grammar = &value
}
if raw, ok := lookup("AUDITA_HOMOPHONES_CONFIDENCE_THRESHOLD"); ok {
value, err := parseFloat(raw)
if err != nil {
return fmt.Errorf("AUDITA_HOMOPHONES_CONFIDENCE_THRESHOLD: %w", err)
}
cfg.Thresholds.Homophones = value
thresholds.homophones = &value
}
if raw, ok := lookup("AUDITA_SPOKEN_WORD_CONFIDENCE_THRESHOLD"); ok {
value, err := parseFloat(raw)
if err != nil {
return fmt.Errorf("AUDITA_SPOKEN_WORD_CONFIDENCE_THRESHOLD: %w", err)
}
cfg.Thresholds.SpokenWord = value
thresholds.spokenWord = &value
}
cfg.applyThresholdsPatch(thresholds)
normalization := normalizationPatch{}
if raw, ok := lookup("AUDITA_NORMALIZE_MAX_SEGMENT_GAP"); ok {
value, err := parseFloat(raw)
if err != nil {
return fmt.Errorf("AUDITA_NORMALIZE_MAX_SEGMENT_GAP: %w", err)
}
cfg.Normalization.MaxSegmentGap = value
normalization.maxSegmentGap = &value
}
if raw, ok := lookup("AUDITA_NORMALIZE_ELLIPSIS_GAP"); ok {
value, err := parseFloat(raw)
if err != nil {
return fmt.Errorf("AUDITA_NORMALIZE_ELLIPSIS_GAP: %w", err)
}
cfg.Normalization.EllipsisGap = value
normalization.ellipsisGap = &value
}
if raw, ok := lookup("AUDITA_NORMALIZE_MAX_SEGMENT_DURATION"); ok {
value, err := parseFloat(raw)
if err != nil {
return fmt.Errorf("AUDITA_NORMALIZE_MAX_SEGMENT_DURATION: %w", err)
}
cfg.Normalization.MaxSegmentDuration = value
normalization.maxSegmentDuration = &value
}
if raw, ok := lookup("AUDITA_NORMALIZE_MAX_SEGMENT_TOKENS"); ok {
value, err := parseInt(raw)
if err != nil {
return fmt.Errorf("AUDITA_NORMALIZE_MAX_SEGMENT_TOKENS: %w", err)
}
cfg.Normalization.MaxSegmentTokens = value
normalization.maxSegmentTokens = &value
}
cfg.applyNormalizationPatch(normalization)
diagnostics := diagnosticsPatch{}
if raw, ok := lookup("AUDITA_WORK_DIR"); ok {
cfg.WorkDir = raw
diagnostics.workDir = &raw
}
if raw, ok := lookup("AUDITA_WORK_DIR_RETENTION"); ok {
cfg.WorkDirRetention = WorkDirRetention(raw)
diagnostics.workDirRetention = &raw
}
cfg.applyDiagnosticsPatch(diagnostics)
cfg.syncLegacyConcurrencyAliases()

View File

@@ -205,118 +205,98 @@ func (c *Config) applyFileConfigWithLookup(fileCfg FileConfig, lookup func(strin
if fileCfg.LLM != nil {
if fileCfg.LLM.Proposal != nil {
if fileCfg.LLM.Proposal.BaseURL != nil {
c.PrimaryLLM.BaseURL = *fileCfg.LLM.Proposal.BaseURL
}
if fileCfg.LLM.Proposal.Model != nil {
c.PrimaryLLM.Model = *fileCfg.LLM.Proposal.Model
patch := llmTargetPatch{
model: fileCfg.LLM.Proposal.Model,
baseURL: fileCfg.LLM.Proposal.BaseURL,
maxRetries: fileCfg.LLM.Proposal.MaxRetries,
}
if fileCfg.LLM.Proposal.Timeout != nil {
c.PrimaryLLM.TimeoutSeconds = fileCfg.LLM.Proposal.Timeout.Seconds()
}
if fileCfg.LLM.Proposal.MaxRetries != nil {
c.PrimaryLLM.MaxRetries = *fileCfg.LLM.Proposal.MaxRetries
timeoutSeconds := fileCfg.LLM.Proposal.Timeout.Seconds()
patch.timeoutSeconds = &timeoutSeconds
}
if fileCfg.LLM.Proposal.APIKeyEnv != nil {
apiKey, err := resolveAPIKeyEnv(*fileCfg.LLM.Proposal.APIKeyEnv, lookup)
if err != nil {
return fmt.Errorf("llm.proposal.api_key_env: %w", err)
}
c.PrimaryLLM.APIKey = apiKey
patch.apiKey = &apiKey
}
c.applyPrimaryLLMTargetPatch(patch)
}
if fileCfg.LLM.Validation != nil {
if fileCfg.LLM.Validation.BaseURL != nil {
c.ValidationLLM.BaseURL = *fileCfg.LLM.Validation.BaseURL
}
if fileCfg.LLM.Validation.Model != nil {
c.ValidationLLM.Model = *fileCfg.LLM.Validation.Model
patch := llmTargetPatch{
model: fileCfg.LLM.Validation.Model,
baseURL: fileCfg.LLM.Validation.BaseURL,
maxRetries: fileCfg.LLM.Validation.MaxRetries,
}
if fileCfg.LLM.Validation.Timeout != nil {
v := fileCfg.LLM.Validation.Timeout.Seconds()
c.ValidationLLM.TimeoutSeconds = &v
}
if fileCfg.LLM.Validation.MaxRetries != nil {
v := *fileCfg.LLM.Validation.MaxRetries
c.ValidationLLM.MaxRetries = &v
timeoutSeconds := fileCfg.LLM.Validation.Timeout.Seconds()
patch.timeoutSeconds = &timeoutSeconds
}
if fileCfg.LLM.Validation.APIKeyEnv != nil {
apiKey, err := resolveAPIKeyEnv(*fileCfg.LLM.Validation.APIKeyEnv, lookup)
if err != nil {
return fmt.Errorf("llm.validation.api_key_env: %w", err)
}
c.ValidationLLM.APIKey = apiKey
patch.apiKey = &apiKey
}
c.applyValidationLLMTargetPatch(patch)
}
}
if fileCfg.Concurrency != nil {
if fileCfg.Concurrency.TotalLLM != nil {
c.TotalLLMConcurrency = *fileCfg.Concurrency.TotalLLM
}
if fileCfg.Concurrency.ProposalLLM != nil {
c.ProposalLLMConcurrency = *fileCfg.Concurrency.ProposalLLM
}
if fileCfg.Concurrency.ValidationLLM != nil {
v := *fileCfg.Concurrency.ValidationLLM
c.ValidationLLMConcurrency = &v
}
c.applyConcurrencyPatch(concurrencyPatch{
totalLLM: fileCfg.Concurrency.TotalLLM,
proposalLLM: fileCfg.Concurrency.ProposalLLM,
validationLLM: fileCfg.Concurrency.ValidationLLM,
})
}
if fileCfg.Chunking != nil {
if fileCfg.Chunking.TargetSections != nil {
v := *fileCfg.Chunking.TargetSections
c.TargetSections = &v
}
if fileCfg.Chunking.MaxSectionTokens != nil {
c.MaxSectionTokens = *fileCfg.Chunking.MaxSectionTokens
}
if fileCfg.Chunking.MinSectionTokens != nil {
c.MinSectionTokens = *fileCfg.Chunking.MinSectionTokens
}
c.applyChunkingPatch(chunkingPatch{
targetSections: fileCfg.Chunking.TargetSections,
maxSectionTokens: fileCfg.Chunking.MaxSectionTokens,
minSectionTokens: fileCfg.Chunking.MinSectionTokens,
})
}
if fileCfg.Normalization != nil {
patch := normalizationPatch{
maxSegmentTokens: fileCfg.Normalization.MaxSegmentTokens,
}
if fileCfg.Normalization.MaxSegmentGap != nil {
c.Normalization.MaxSegmentGap = fileCfg.Normalization.MaxSegmentGap.Seconds()
maxSegmentGap := fileCfg.Normalization.MaxSegmentGap.Seconds()
patch.maxSegmentGap = &maxSegmentGap
}
if fileCfg.Normalization.EllipsisGap != nil {
c.Normalization.EllipsisGap = fileCfg.Normalization.EllipsisGap.Seconds()
ellipsisGap := fileCfg.Normalization.EllipsisGap.Seconds()
patch.ellipsisGap = &ellipsisGap
}
if fileCfg.Normalization.MaxSegmentDuration != nil {
c.Normalization.MaxSegmentDuration = fileCfg.Normalization.MaxSegmentDuration.Seconds()
}
if fileCfg.Normalization.MaxSegmentTokens != nil {
c.Normalization.MaxSegmentTokens = *fileCfg.Normalization.MaxSegmentTokens
maxSegmentDuration := fileCfg.Normalization.MaxSegmentDuration.Seconds()
patch.maxSegmentDuration = &maxSegmentDuration
}
c.applyNormalizationPatch(patch)
}
if fileCfg.Thresholds != nil {
if fileCfg.Thresholds.Glossary != nil {
c.Thresholds.Glossary = *fileCfg.Thresholds.Glossary
}
if fileCfg.Thresholds.Homophones != nil {
c.Thresholds.Homophones = *fileCfg.Thresholds.Homophones
}
if fileCfg.Thresholds.SpokenWord != nil {
c.Thresholds.SpokenWord = *fileCfg.Thresholds.SpokenWord
}
if fileCfg.Thresholds.Grammar != nil {
c.Thresholds.Grammar = *fileCfg.Thresholds.Grammar
}
c.applyThresholdsPatch(thresholdsPatch{
glossary: fileCfg.Thresholds.Glossary,
grammar: fileCfg.Thresholds.Grammar,
homophones: fileCfg.Thresholds.Homophones,
spokenWord: fileCfg.Thresholds.SpokenWord,
})
}
if fileCfg.Context != nil && fileCfg.Context.Description != nil {
c.TranscriptDescription = strings.TrimSpace(*fileCfg.Context.Description)
c.applyContextPatch(contextPatch{transcriptDescription: fileCfg.Context.Description})
}
if fileCfg.Diagnostics != nil {
if fileCfg.Diagnostics.WorkDir != nil {
c.WorkDir = *fileCfg.Diagnostics.WorkDir
}
if fileCfg.Diagnostics.Retention != nil {
c.WorkDirRetention = WorkDirRetention(*fileCfg.Diagnostics.Retention)
}
c.applyDiagnosticsPatch(diagnosticsPatch{
workDir: fileCfg.Diagnostics.WorkDir,
workDirRetention: fileCfg.Diagnostics.Retention,
})
}
c.syncLegacyConcurrencyAliases()

View File

@@ -51,107 +51,54 @@ func (c *Config) ApplyCLIOverrides(overrides CLIOverrides) error {
c.OutputSchema = strings.TrimSpace(*overrides.OutputSchema)
}
if overrides.PrimaryLLMAPIKey != nil {
c.PrimaryLLM.APIKey = *overrides.PrimaryLLMAPIKey
}
if overrides.ValidationLLMAPIKey != nil {
c.ValidationLLM.APIKey = *overrides.ValidationLLMAPIKey
}
if overrides.PrimaryModel != nil {
c.PrimaryLLM.Model = *overrides.PrimaryModel
}
if overrides.ValidationModel != nil {
c.ValidationLLM.Model = *overrides.ValidationModel
}
if overrides.PrimaryBaseURL != nil {
c.PrimaryLLM.BaseURL = *overrides.PrimaryBaseURL
}
if overrides.ValidationBaseURL != nil {
c.ValidationLLM.BaseURL = *overrides.ValidationBaseURL
}
if overrides.PrimaryLLMTimeoutSeconds != nil {
c.PrimaryLLM.TimeoutSeconds = *overrides.PrimaryLLMTimeoutSeconds
}
totalConcurrencySet := false
if overrides.TotalLLMConcurrency != nil {
c.TotalLLMConcurrency = *overrides.TotalLLMConcurrency
totalConcurrencySet = true
}
// Backward-compatible alias: --llm-concurrency maps to total concurrency
// only when --total-llm-concurrency is not set in the same CLI invocation.
if overrides.PrimaryLLMConcurrency != nil && !totalConcurrencySet {
c.TotalLLMConcurrency = *overrides.PrimaryLLMConcurrency
totalConcurrencySet = true
}
proposalConcurrencySet := false
if overrides.ProposalLLMConcurrency != nil {
c.ProposalLLMConcurrency = *overrides.ProposalLLMConcurrency
proposalConcurrencySet = true
}
if totalConcurrencySet && !proposalConcurrencySet {
c.ProposalLLMConcurrency = c.TotalLLMConcurrency
}
if overrides.ValidationLLMTimeoutSeconds != nil {
value := *overrides.ValidationLLMTimeoutSeconds
c.ValidationLLM.TimeoutSeconds = &value
}
if overrides.MaxRetries != nil {
c.PrimaryLLM.MaxRetries = *overrides.MaxRetries
}
if overrides.ValidationMaxRetries != nil {
value := *overrides.ValidationMaxRetries
c.ValidationLLM.MaxRetries = &value
}
if overrides.ValidationLLMConcurrency != nil {
value := *overrides.ValidationLLMConcurrency
c.ValidationLLMConcurrency = &value
}
c.applyPrimaryLLMTargetPatch(llmTargetPatch{
apiKey: overrides.PrimaryLLMAPIKey,
model: overrides.PrimaryModel,
baseURL: overrides.PrimaryBaseURL,
timeoutSeconds: overrides.PrimaryLLMTimeoutSeconds,
maxRetries: overrides.MaxRetries,
})
c.applyValidationLLMTargetPatch(llmTargetPatch{
apiKey: overrides.ValidationLLMAPIKey,
model: overrides.ValidationModel,
baseURL: overrides.ValidationBaseURL,
timeoutSeconds: overrides.ValidationLLMTimeoutSeconds,
maxRetries: overrides.ValidationMaxRetries,
})
c.applyConcurrencyPatch(concurrencyPatch{
totalLLM: overrides.TotalLLMConcurrency,
legacyTotalLLM: overrides.PrimaryLLMConcurrency,
proposalLLM: overrides.ProposalLLMConcurrency,
validationLLM: overrides.ValidationLLMConcurrency,
inheritProposal: true,
allowLegacyAlias: true,
})
if overrides.ValidationMaxPromptTokens != nil {
c.ValidationMaxPromptTokens = *overrides.ValidationMaxPromptTokens
}
if overrides.MaxSectionTokens != nil {
c.MaxSectionTokens = *overrides.MaxSectionTokens
}
if overrides.MinSectionTokens != nil {
c.MinSectionTokens = *overrides.MinSectionTokens
}
if overrides.TargetSections != nil {
value := *overrides.TargetSections
c.TargetSections = &value
}
if overrides.GlossaryConfidenceThreshold != nil {
c.Thresholds.Glossary = *overrides.GlossaryConfidenceThreshold
}
if overrides.GrammarConfidenceThreshold != nil {
c.Thresholds.Grammar = *overrides.GrammarConfidenceThreshold
}
if overrides.HomophonesConfidenceThreshold != nil {
c.Thresholds.Homophones = *overrides.HomophonesConfidenceThreshold
}
if overrides.SpokenWordConfidenceThreshold != nil {
c.Thresholds.SpokenWord = *overrides.SpokenWordConfidenceThreshold
}
if overrides.NormalizeMaxSegmentGap != nil {
c.Normalization.MaxSegmentGap = *overrides.NormalizeMaxSegmentGap
}
if overrides.NormalizeEllipsisGap != nil {
c.Normalization.EllipsisGap = *overrides.NormalizeEllipsisGap
}
if overrides.NormalizeMaxSegmentDuration != nil {
c.Normalization.MaxSegmentDuration = *overrides.NormalizeMaxSegmentDuration
}
if overrides.NormalizeMaxSegmentTokens != nil {
c.Normalization.MaxSegmentTokens = *overrides.NormalizeMaxSegmentTokens
}
if overrides.TranscriptDescription != nil {
c.TranscriptDescription = strings.TrimSpace(*overrides.TranscriptDescription)
}
if overrides.WorkDir != nil {
c.WorkDir = *overrides.WorkDir
}
if overrides.WorkDirRetention != nil {
c.WorkDirRetention = WorkDirRetention(*overrides.WorkDirRetention)
}
c.applyChunkingPatch(chunkingPatch{
targetSections: overrides.TargetSections,
maxSectionTokens: overrides.MaxSectionTokens,
minSectionTokens: overrides.MinSectionTokens,
})
c.applyThresholdsPatch(thresholdsPatch{
glossary: overrides.GlossaryConfidenceThreshold,
grammar: overrides.GrammarConfidenceThreshold,
homophones: overrides.HomophonesConfidenceThreshold,
spokenWord: overrides.SpokenWordConfidenceThreshold,
})
c.applyNormalizationPatch(normalizationPatch{
maxSegmentGap: overrides.NormalizeMaxSegmentGap,
ellipsisGap: overrides.NormalizeEllipsisGap,
maxSegmentDuration: overrides.NormalizeMaxSegmentDuration,
maxSegmentTokens: overrides.NormalizeMaxSegmentTokens,
})
c.applyContextPatch(contextPatch{transcriptDescription: overrides.TranscriptDescription})
c.applyDiagnosticsPatch(diagnosticsPatch{
workDir: overrides.WorkDir,
workDirRetention: overrides.WorkDirRetention,
})
c.syncLegacyConcurrencyAliases()

View File

@@ -3,6 +3,9 @@ package config
import (
"fmt"
"strings"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
"gitea.maximumdirect.net/eric/audita/internal/core/outputschema"
)
func (c Config) Validate() error {
@@ -12,19 +15,19 @@ func (c Config) Validate() error {
issues = append(issues, "modules must not be empty")
}
for _, module := range c.Modules {
if strings.TrimSpace(module) == "" {
moduleKey := strings.TrimSpace(module)
if moduleKey == "" {
issues = append(issues, "modules must not contain empty values")
break
}
if !modulecatalog.IsSupported(moduleKey) {
issues = append(issues, fmt.Sprintf("unsupported module key %q", moduleKey))
}
}
if strings.TrimSpace(c.OutputSchema) == "" {
issues = append(issues, "output schema must not be empty")
} else {
switch strings.TrimSpace(c.OutputSchema) {
case "bare-segments", "audita-v1":
default:
issues = append(issues, fmt.Sprintf("unsupported output schema %q", c.OutputSchema))
}
} else if !outputschema.IsSupported(c.OutputSchema) {
issues = append(issues, fmt.Sprintf("unsupported output schema %q", c.OutputSchema))
}
if c.PrimaryLLM.TimeoutSeconds <= 0 {

View File

@@ -0,0 +1,42 @@
package diagnostics
import (
"path/filepath"
"gitea.maximumdirect.net/eric/audita/internal/core/reporting"
)
const (
ArtifactSourceTranscript = "source-transcript.json"
ArtifactParsedSourceTranscript = "source-transcript-parsed.json"
ArtifactNormalizedTranscript = "normalized-transcript.json"
ArtifactNormalizationSummary = "normalization-summary.json"
ArtifactChunkingSummary = "chunking-summary.json"
ArtifactUtilizationSummary = "utilization-diagnostics.json"
ArtifactCorrectionLedger = "correction-ledger.json"
ArtifactInvocationMetadata = "invocation.json"
ArtifactEffectiveConfig = "effective-config.json"
ArtifactReport = "report.json"
ArtifactErrorLog = "error.log"
)
func BuildDiagnosticsMetadata(runDirectoryPath string, runSucceeded bool) reporting.DiagnosticsMetadata {
metadata := reporting.DiagnosticsMetadata{
DirectoryPath: runDirectoryPath,
SourceTranscriptPath: filepath.Join(runDirectoryPath, ArtifactSourceTranscript),
ParsedSourceTranscriptPath: filepath.Join(runDirectoryPath, ArtifactParsedSourceTranscript),
NormalizedTranscriptPath: filepath.Join(runDirectoryPath, ArtifactNormalizedTranscript),
NormalizationSummaryPath: filepath.Join(runDirectoryPath, ArtifactNormalizationSummary),
ChunkingSummaryPath: filepath.Join(runDirectoryPath, ArtifactChunkingSummary),
UtilizationSummaryPath: filepath.Join(runDirectoryPath, ArtifactUtilizationSummary),
CorrectionLedgerPath: filepath.Join(runDirectoryPath, ArtifactCorrectionLedger),
InvocationMetadataPath: filepath.Join(runDirectoryPath, ArtifactInvocationMetadata),
RedactedEffectiveConfigPath: filepath.Join(runDirectoryPath, ArtifactEffectiveConfig),
}
if !runSucceeded {
metadata.ErrorLogPath = filepath.Join(runDirectoryPath, ArtifactErrorLog)
}
return metadata
}

View File

@@ -0,0 +1,55 @@
package diagnostics
import (
"path/filepath"
"testing"
)
func TestBuildDiagnosticsMetadataSuccessPathsMatchArtifactConstants(t *testing.T) {
runPath := filepath.Join("tmp", "run-123")
metadata := BuildDiagnosticsMetadata(runPath, true)
if metadata.DirectoryPath != runPath {
t.Fatalf("unexpected diagnostics directory path: got=%q want=%q", metadata.DirectoryPath, runPath)
}
if metadata.SourceTranscriptPath != filepath.Join(runPath, ArtifactSourceTranscript) {
t.Fatalf("unexpected source transcript path: %q", metadata.SourceTranscriptPath)
}
if metadata.ParsedSourceTranscriptPath != filepath.Join(runPath, ArtifactParsedSourceTranscript) {
t.Fatalf("unexpected parsed source transcript path: %q", metadata.ParsedSourceTranscriptPath)
}
if metadata.NormalizedTranscriptPath != filepath.Join(runPath, ArtifactNormalizedTranscript) {
t.Fatalf("unexpected normalized transcript path: %q", metadata.NormalizedTranscriptPath)
}
if metadata.NormalizationSummaryPath != filepath.Join(runPath, ArtifactNormalizationSummary) {
t.Fatalf("unexpected normalization summary path: %q", metadata.NormalizationSummaryPath)
}
if metadata.ChunkingSummaryPath != filepath.Join(runPath, ArtifactChunkingSummary) {
t.Fatalf("unexpected chunking summary path: %q", metadata.ChunkingSummaryPath)
}
if metadata.UtilizationSummaryPath != filepath.Join(runPath, ArtifactUtilizationSummary) {
t.Fatalf("unexpected utilization summary path: %q", metadata.UtilizationSummaryPath)
}
if metadata.CorrectionLedgerPath != filepath.Join(runPath, ArtifactCorrectionLedger) {
t.Fatalf("unexpected correction ledger path: %q", metadata.CorrectionLedgerPath)
}
if metadata.InvocationMetadataPath != filepath.Join(runPath, ArtifactInvocationMetadata) {
t.Fatalf("unexpected invocation metadata path: %q", metadata.InvocationMetadataPath)
}
if metadata.RedactedEffectiveConfigPath != filepath.Join(runPath, ArtifactEffectiveConfig) {
t.Fatalf("unexpected redacted effective config path: %q", metadata.RedactedEffectiveConfigPath)
}
if metadata.ErrorLogPath != "" {
t.Fatalf("did not expect error log path on success: %q", metadata.ErrorLogPath)
}
}
func TestBuildDiagnosticsMetadataFailureIncludesErrorLogPath(t *testing.T) {
runPath := filepath.Join("tmp", "run-123")
metadata := BuildDiagnosticsMetadata(runPath, false)
want := filepath.Join(runPath, ArtifactErrorLog)
if metadata.ErrorLogPath != want {
t.Fatalf("unexpected error log path: got=%q want=%q", metadata.ErrorLogPath, want)
}
}

View File

@@ -106,7 +106,7 @@ func (r *RunDirectory) WriteInvocationMetadata(metadata InvocationMetadata) erro
metadata.StartedAt = r.createdAt
}
path := filepath.Join(r.path, "invocation.json")
path := filepath.Join(r.path, ArtifactInvocationMetadata)
bytes, err := json.MarshalIndent(metadata, "", " ")
if err != nil {
return fmt.Errorf("failed to marshal invocation metadata: %w", err)
@@ -120,7 +120,7 @@ func (r *RunDirectory) WriteInvocationMetadata(metadata InvocationMetadata) erro
// WriteEffectiveConfig writes redacted effective config metadata for this run.
func (r *RunDirectory) WriteEffectiveConfig(cfg config.Config) error {
path := filepath.Join(r.path, "effective-config.json")
path := filepath.Join(r.path, ArtifactEffectiveConfig)
redacted := cfg.Redacted()
bytes, err := json.MarshalIndent(redacted, "", " ")
if err != nil {
@@ -136,13 +136,13 @@ func (r *RunDirectory) WriteEffectiveConfig(cfg config.Config) error {
// WriteSourceTranscript writes the source transcript artifact
func (r *RunDirectory) WriteSourceTranscript(transcript *schema.SourceTranscript, raw []byte) error {
// Write raw source for reference
sourcePath := filepath.Join(r.path, "source-transcript.json")
sourcePath := filepath.Join(r.path, ArtifactSourceTranscript)
if err := os.WriteFile(sourcePath, raw, 0o644); err != nil {
return fmt.Errorf("failed to write source transcript: %w", err)
}
// Write parsed source for debugging
parsedPath := filepath.Join(r.path, "source-transcript-parsed.json")
parsedPath := filepath.Join(r.path, ArtifactParsedSourceTranscript)
parsedBytes, err := json.MarshalIndent(transcript, "", " ")
if err != nil {
return fmt.Errorf("failed to marshal parsed source transcript: %w", err)
@@ -157,7 +157,7 @@ func (r *RunDirectory) WriteSourceTranscript(transcript *schema.SourceTranscript
// WriteNormalizedTranscript writes the normalized transcript artifact
func (r *RunDirectory) WriteNormalizedTranscript(transcript *schema.Transcript) error {
normalizedPath := filepath.Join(r.path, "normalized-transcript.json")
normalizedPath := filepath.Join(r.path, ArtifactNormalizedTranscript)
bytes, err := schema.TranscriptToJSON(transcript)
if err != nil {
return fmt.Errorf("failed to serialize normalized transcript: %w", err)
@@ -170,7 +170,7 @@ func (r *RunDirectory) WriteNormalizedTranscript(transcript *schema.Transcript)
// WriteNormalizationSummary writes the normalization summary artifact
func (r *RunDirectory) WriteNormalizationSummary(summary *normalization.NormalizationSummary) error {
summaryPath := filepath.Join(r.path, "normalization-summary.json")
summaryPath := filepath.Join(r.path, ArtifactNormalizationSummary)
bytes, err := json.MarshalIndent(summary, "", " ")
if err != nil {
return fmt.Errorf("failed to marshal normalization summary: %w", err)
@@ -184,7 +184,7 @@ func (r *RunDirectory) WriteNormalizationSummary(summary *normalization.Normaliz
// WriteReport writes the authoritative report artifact
func (r *RunDirectory) WriteReport(report reporting.ProcessReport) error {
reportPath := filepath.Join(r.path, "report.json")
reportPath := filepath.Join(r.path, ArtifactReport)
bytes, err := json.MarshalIndent(report, "", " ")
if err != nil {
return fmt.Errorf("failed to marshal report: %w", err)
@@ -198,13 +198,13 @@ func (r *RunDirectory) WriteReport(report reporting.ProcessReport) error {
// WriteErrorLog writes an error log on failure
func (r *RunDirectory) WriteErrorLog(errorMessage string) error {
errorPath := filepath.Join(r.path, "error.log")
errorPath := filepath.Join(r.path, ArtifactErrorLog)
return os.WriteFile(errorPath, []byte(errorMessage+"\n"), 0o644)
}
// WriteChunkingSummary writes the chunking summary artifact
func (r *RunDirectory) WriteChunkingSummary(summary *chunking.DetailedSummary) error {
summaryPath := filepath.Join(r.path, "chunking-summary.json")
summaryPath := filepath.Join(r.path, ArtifactChunkingSummary)
bytes, err := json.MarshalIndent(summary, "", " ")
if err != nil {
return fmt.Errorf("failed to marshal chunking summary: %w", err)

View File

@@ -0,0 +1,35 @@
package modulecatalog
import "strings"
const (
KeyGlossary = "glossary"
KeyHomophones = "homophones"
KeySpokenWord = "spoken_word"
KeyGrammar = "grammar"
)
var supportedKeys = []string{
KeyGlossary,
KeyHomophones,
KeySpokenWord,
KeyGrammar,
}
var supportedKeySet = map[string]struct{}{
KeyGlossary: {},
KeyHomophones: {},
KeySpokenWord: {},
KeyGrammar: {},
}
func SupportedKeys() []string {
out := make([]string, len(supportedKeys))
copy(out, supportedKeys)
return out
}
func IsSupported(key string) bool {
_, ok := supportedKeySet[strings.TrimSpace(key)]
return ok
}

View File

@@ -0,0 +1,24 @@
package modulecatalog
import (
"reflect"
"testing"
)
func TestSupportedKeys(t *testing.T) {
want := []string{KeyGlossary, KeyHomophones, KeySpokenWord, KeyGrammar}
if got := SupportedKeys(); !reflect.DeepEqual(got, want) {
t.Fatalf("unexpected supported keys: got=%v want=%v", got, want)
}
}
func TestIsSupported(t *testing.T) {
for _, key := range SupportedKeys() {
if !IsSupported(key) {
t.Fatalf("expected key %q to be supported", key)
}
}
if IsSupported("made_up") {
t.Fatalf("did not expect made_up to be supported")
}
}

View File

@@ -31,15 +31,31 @@ var definitions = map[string]Definition{
},
}
var supportedKeys = []string{
SchemaBareSegments,
SchemaAuditaV1,
}
func SupportedKeys() []string {
out := make([]string, len(supportedKeys))
copy(out, supportedKeys)
return out
}
func IsSupported(key string) bool {
_, ok := definitions[strings.TrimSpace(key)]
return ok
}
func Resolve(key string) (Definition, error) {
normalized := strings.TrimSpace(key)
if normalized == "" {
return Definition{}, fmt.Errorf("output schema must not be empty")
}
def, ok := definitions[normalized]
if !ok {
if !IsSupported(normalized) {
return Definition{}, fmt.Errorf("unsupported output schema %q", normalized)
}
def := definitions[normalized]
return def, nil
}

View File

@@ -2,6 +2,7 @@ package outputschema
import (
"encoding/json"
"reflect"
"strings"
"testing"
@@ -56,3 +57,19 @@ func TestResolveUnknown(t *testing.T) {
t.Fatalf("expected unsupported output schema error, got %v", err)
}
}
func TestSupportedKeysAndIsSupported(t *testing.T) {
want := []string{SchemaBareSegments, SchemaAuditaV1}
if got := SupportedKeys(); !reflect.DeepEqual(got, want) {
t.Fatalf("unexpected supported schema keys: got=%v want=%v", got, want)
}
for _, key := range want {
if !IsSupported(key) {
t.Fatalf("expected schema key %q to be supported", key)
}
}
if IsSupported("seriatim-intermediate") {
t.Fatalf("did not expect unsupported schema to be reported as supported")
}
}

View File

@@ -7,6 +7,7 @@ import (
"time"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
)
type ProcessReport struct {
@@ -46,18 +47,19 @@ type ReportMetadata struct {
}
type ModuleReport struct {
ModuleKey string `json:"module_key"`
ModuleInstance string `json:"module_instance"`
ReplacementPolicy string `json:"replacement_policy,omitempty"`
Status string `json:"status"`
ProposalCount int `json:"proposal_count"`
ValidatorDecisions []ValidatorDecisionReport `json:"validator_decisions,omitempty"`
ValidatorRejected []ValidatorRejectedReport `json:"validator_rejected,omitempty"`
AppliedChanges []proposals.AppliedChange `json:"applied_changes,omitempty"`
SkippedChanges []proposals.SkippedChange `json:"skipped_changes,omitempty"`
ErrorMessage string `json:"error_message,omitempty"`
StartedAt *time.Time `json:"started_at,omitempty"`
CompletedAt *time.Time `json:"completed_at,omitempty"`
ModuleKey string `json:"module_key"`
ModuleInstance string `json:"module_instance"`
ReplacementPolicy string `json:"replacement_policy,omitempty"`
Status string `json:"status"`
ProposalCount int `json:"proposal_count"`
Warnings []stagewarnings.StageWarning `json:"warnings,omitempty"`
ValidatorDecisions []ValidatorDecisionReport `json:"validator_decisions,omitempty"`
ValidatorRejected []ValidatorRejectedReport `json:"validator_rejected,omitempty"`
AppliedChanges []proposals.AppliedChange `json:"applied_changes,omitempty"`
SkippedChanges []proposals.SkippedChange `json:"skipped_changes,omitempty"`
ErrorMessage string `json:"error_message,omitempty"`
StartedAt *time.Time `json:"started_at,omitempty"`
CompletedAt *time.Time `json:"completed_at,omitempty"`
}
type ValidatorDecisionReport struct {

View File

@@ -47,7 +47,7 @@ func (h testChunkProposalHarness) collectEnrichedProposals(
return nil, err
}
for _, proposal := range base {
for _, proposal := range base.Proposals {
sectionIndex := section.Index
out = append(out, proposals.EnrichedCorrectionProposal{
CorrectionProposal: proposal,
@@ -85,11 +85,11 @@ func (m deterministicFakeModule) ReplacementPolicy() proposals.ReplacementPolicy
func (m deterministicFakeModule) Validators() []Validator { return nil }
func (m deterministicFakeModule) Propose(ctx context.Context, req ProposalRequest) ([]proposals.CorrectionProposal, error) {
func (m deterministicFakeModule) Propose(ctx context.Context, req ProposalRequest) (ProposalResult, error) {
_ = ctx
if req.WorkingTranscript == nil || req.Section == nil {
return []proposals.CorrectionProposal{}, nil
return ProposalResult{}, nil
}
out := make([]proposals.CorrectionProposal, 0)
@@ -110,7 +110,7 @@ func (m deterministicFakeModule) Propose(ctx context.Context, req ProposalReques
}
}
return out, nil
return ProposalResult{Proposals: out}, nil
}
func TestChunkProposalMetadataAssociation(t *testing.T) {

View File

@@ -11,6 +11,7 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/responseschema"
"gitea.maximumdirect.net/eric/audita/internal/framework/validators"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
)
// StructuredLLMClient provides provider-agnostic structured completion.
@@ -28,7 +29,7 @@ type TranscriptModule interface {
Key() string
ReplacementPolicy() proposals.ReplacementPolicy
Validators() []Validator
Propose(ctx context.Context, req ProposalRequest) ([]proposals.CorrectionProposal, error)
Propose(ctx context.Context, req ProposalRequest) (ProposalResult, error)
}
// Validator evaluates candidate proposals and returns one decision per proposal index.
@@ -103,6 +104,11 @@ type ProposalRequest struct {
LLMScheduler LLMScheduler `json:"-"`
}
type ProposalResult struct {
Proposals []proposals.CorrectionProposal `json:"proposals,omitempty"`
Warnings []stagewarnings.StageWarning `json:"warnings,omitempty"`
}
// ValidationRequest is the input to validator execution.
type ValidationRequest = validators.Request

View File

@@ -50,11 +50,13 @@ func (f *fakeModule) Validators() []Validator {
return []Validator{&fakeValidator{}}
}
func (f *fakeModule) Propose(ctx context.Context, req ProposalRequest) ([]proposals.CorrectionProposal, error) {
func (f *fakeModule) Propose(ctx context.Context, req ProposalRequest) (ProposalResult, error) {
_ = ctx
_ = req
return []proposals.CorrectionProposal{
{TargetSegmentID: 1, OriginalText: "a", CorrectedText: "b", Confidence: 0.9},
return ProposalResult{
Proposals: []proposals.CorrectionProposal{
{TargetSegmentID: 1, OriginalText: "a", CorrectedText: "b", Confidence: 0.9},
},
}, nil
}
@@ -68,12 +70,12 @@ func TestInterfaceContractsCompileWithFakes(t *testing.T) {
t.Fatalf("unexpected module key: %q", got)
}
proposalsOut, err := module.Propose(context.Background(), ProposalRequest{})
proposalResult, err := module.Propose(context.Background(), ProposalRequest{})
if err != nil {
t.Fatalf("unexpected propose error: %v", err)
}
if len(proposalsOut) != 1 {
t.Fatalf("expected one proposal, got %d", len(proposalsOut))
if len(proposalResult.Proposals) != 1 {
t.Fatalf("expected one proposal, got %d", len(proposalResult.Proposals))
}
}

View File

@@ -0,0 +1,17 @@
package llm
import "gitea.maximumdirect.net/eric/audita/internal/core/config"
// ConfiguredSecrets returns all configured LLM API-key values that should be
// redacted from diagnostics and surfaced error payloads.
func ConfiguredSecrets(cfg *config.Config) []string {
if cfg == nil {
return nil
}
effectiveValidation := cfg.EffectiveValidationLLMConfig()
return []string{
cfg.PrimaryLLM.APIKey,
cfg.ValidationLLM.APIKey,
effectiveValidation.APIKey,
}
}

View File

@@ -0,0 +1,38 @@
package llm
import (
"reflect"
"testing"
"gitea.maximumdirect.net/eric/audita/internal/core/config"
)
func TestConfiguredSecretsNilConfig(t *testing.T) {
if got := ConfiguredSecrets(nil); got != nil {
t.Fatalf("expected nil secrets for nil config, got %v", got)
}
}
func TestConfiguredSecretsWithValidationOverride(t *testing.T) {
cfg := config.Default()
cfg.PrimaryLLM.APIKey = "primary-secret"
cfg.ValidationLLM.APIKey = "validation-secret"
got := ConfiguredSecrets(&cfg)
want := []string{"primary-secret", "validation-secret", "validation-secret"}
if !reflect.DeepEqual(got, want) {
t.Fatalf("unexpected secrets: got=%v want=%v", got, want)
}
}
func TestConfiguredSecretsWithInheritedValidationKey(t *testing.T) {
cfg := config.Default()
cfg.PrimaryLLM.APIKey = "primary-secret"
cfg.ValidationLLM.APIKey = ""
got := ConfiguredSecrets(&cfg)
want := []string{"primary-secret", "", "primary-secret"}
if !reflect.DeepEqual(got, want) {
t.Fatalf("unexpected inherited secrets: got=%v want=%v", got, want)
}
}

View File

@@ -6,6 +6,7 @@ import (
"strings"
"gitea.maximumdirect.net/eric/audita/internal/core/config"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
glossarymodule "gitea.maximumdirect.net/eric/audita/internal/modules/glossary"
@@ -15,28 +16,20 @@ import (
)
const (
ModuleKeyGlossary = "glossary"
ModuleKeyHomophones = "homophones"
ModuleKeySpokenWord = "spoken_word"
ModuleKeyGrammar = "grammar"
ModuleKeyGlossary = modulecatalog.KeyGlossary
ModuleKeyHomophones = modulecatalog.KeyHomophones
ModuleKeySpokenWord = modulecatalog.KeySpokenWord
ModuleKeyGrammar = modulecatalog.KeyGrammar
)
const (
ReasonUnsupportedModule = "unsupported_module"
)
var knownModuleKeys = map[string]struct{}{
ModuleKeyGlossary: {},
ModuleKeyHomophones: {},
ModuleKeySpokenWord: {},
ModuleKeyGrammar: {},
}
// IsKnownModuleKey reports whether a module key is recognized by the production
// registry scaffold.
func IsKnownModuleKey(key string) bool {
_, ok := knownModuleKeys[strings.TrimSpace(key)]
return ok
return modulecatalog.IsSupported(key)
}
// Dependencies holds explicit constructor dependencies for module creation.
@@ -69,7 +62,7 @@ type Factory struct {
func NewFactory(deps Dependencies) *Factory {
factory := &Factory{
deps: deps,
constructors: make(map[string]Constructor, len(knownModuleKeys)),
constructors: make(map[string]Constructor, len(modulecatalog.SupportedKeys())),
}
_ = factory.RegisterConstructor(ModuleKeyGlossary, constructGlossaryModule)
_ = factory.RegisterConstructor(ModuleKeyHomophones, constructHomophonesModule)

View File

@@ -6,6 +6,7 @@ import (
"strings"
"testing"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
)
@@ -19,14 +20,14 @@ func (m noopModule) ReplacementPolicy() proposals.ReplacementPolicy {
return proposals.ReplacementPolicyRequireUnique
}
func (m noopModule) Validators() []contracts.Validator { return nil }
func (m noopModule) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
func (m noopModule) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
_ = ctx
_ = req
return nil, nil
return contracts.ProposalResult{}, nil
}
func TestKnownModuleKeyRecognition(t *testing.T) {
for _, key := range []string{ModuleKeyGlossary, ModuleKeyHomophones, ModuleKeySpokenWord, ModuleKeyGrammar} {
for _, key := range modulecatalog.SupportedKeys() {
if !IsKnownModuleKey(key) {
t.Fatalf("expected key %q to be recognized", key)
}

View File

@@ -1,10 +1,11 @@
package cli
package processreport
import (
"path/filepath"
"sort"
"gitea.maximumdirect.net/eric/audita/internal/framework/runner"
validatormetadata "gitea.maximumdirect.net/eric/audita/internal/validators/metadata"
)
const (
@@ -14,7 +15,12 @@ const (
correctionDispositionFailed = "failed"
)
type correctionLedgerEntry struct {
type CorrectionLedgerInput struct {
RunDirectoryPath string
RunOutput *runner.RunOutput
}
type CorrectionLedgerEntry struct {
RunID string `json:"run_id,omitempty"`
ModuleKey string `json:"module_key"`
ModuleInstance string `json:"module_instance"`
@@ -27,33 +33,28 @@ type correctionLedgerEntry struct {
Disposition string `json:"disposition"`
DispositionReasonCode string `json:"disposition_reason_code,omitempty"`
DispositionMessage string `json:"disposition_message,omitempty"`
DeterministicValidatorResults []ledgerValidatorDecisionRecord `json:"deterministic_validator_decisions,omitempty"`
LLMValidatorResults []ledgerValidatorDecisionRecord `json:"llm_validator_decisions,omitempty"`
DeterministicValidatorResults []LedgerValidatorDecisionRecord `json:"deterministic_validator_decisions,omitempty"`
LLMValidatorResults []LedgerValidatorDecisionRecord `json:"llm_validator_decisions,omitempty"`
}
type ledgerValidatorDecisionRecord struct {
type LedgerValidatorDecisionRecord struct {
ValidatorKey string `json:"validator_key"`
Approved bool `json:"approved"`
ReasonCode string `json:"reason_code"`
Message string `json:"message,omitempty"`
}
func buildCorrectionLedger(runDirPath string, runOutput *runner.RunOutput) []correctionLedgerEntry {
func BuildCorrectionLedger(input CorrectionLedgerInput) []CorrectionLedgerEntry {
runOutput := input.RunOutput
if runOutput == nil || len(runOutput.ModuleResults) == 0 {
return nil
}
runID := ""
if runDirPath != "" {
runID = filepath.Base(runDirPath)
}
entries := make([]correctionLedgerEntry, 0)
llmBacked := map[string]bool{
"spoken_form_plausibility": true,
"meaning_reversal_review": true,
"editorial_review": true,
if input.RunDirectoryPath != "" {
runID = filepath.Base(input.RunDirectoryPath)
}
entries := make([]CorrectionLedgerEntry, 0)
for _, module := range runOutput.ModuleResults {
decisionsByProposal := make(map[int][]runner.ValidatorDecisionRecord)
for _, decision := range module.ValidatorDecisions {
@@ -61,7 +62,7 @@ func buildCorrectionLedger(runDirPath string, runOutput *runner.RunOutput) []cor
}
for _, change := range module.AppliedChanges {
entries = append(entries, correctionLedgerEntry{
entries = append(entries, CorrectionLedgerEntry{
RunID: runID,
ModuleKey: module.ModuleKey,
ModuleInstance: module.ModuleInstance,
@@ -72,12 +73,12 @@ func buildCorrectionLedger(runDirPath string, runOutput *runner.RunOutput) []cor
AppliedCorrectedText: change.CorrectedText,
ReplacementPolicy: string(module.ReplacementPolicy),
Disposition: correctionDispositionApplied,
DeterministicValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], false, llmBacked),
LLMValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], true, llmBacked),
DeterministicValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], false),
LLMValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], true),
})
}
for _, change := range module.SkippedChanges {
entries = append(entries, correctionLedgerEntry{
entries = append(entries, CorrectionLedgerEntry{
RunID: runID,
ModuleKey: module.ModuleKey,
ModuleInstance: module.ModuleInstance,
@@ -89,12 +90,12 @@ func buildCorrectionLedger(runDirPath string, runOutput *runner.RunOutput) []cor
Disposition: correctionDispositionSkipped,
DispositionReasonCode: string(change.SkipReason),
DispositionMessage: change.Message,
DeterministicValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], false, llmBacked),
LLMValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], true, llmBacked),
DeterministicValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], false),
LLMValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], true),
})
}
for _, rejection := range module.ValidatorRejected {
entries = append(entries, correctionLedgerEntry{
entries = append(entries, CorrectionLedgerEntry{
RunID: runID,
ModuleKey: module.ModuleKey,
ModuleInstance: module.ModuleInstance,
@@ -106,12 +107,12 @@ func buildCorrectionLedger(runDirPath string, runOutput *runner.RunOutput) []cor
Disposition: correctionDispositionRejected,
DispositionReasonCode: rejection.ReasonCode,
DispositionMessage: rejection.Message,
DeterministicValidatorResults: filterLedgerDecisions(decisionsByProposal[rejection.ProposalIndex], false, llmBacked),
LLMValidatorResults: filterLedgerDecisions(decisionsByProposal[rejection.ProposalIndex], true, llmBacked),
DeterministicValidatorResults: filterLedgerDecisions(decisionsByProposal[rejection.ProposalIndex], false),
LLMValidatorResults: filterLedgerDecisions(decisionsByProposal[rejection.ProposalIndex], true),
})
}
if module.Status == runner.ModuleStatusFailed {
entries = append(entries, correctionLedgerEntry{
entries = append(entries, CorrectionLedgerEntry{
RunID: runID,
ModuleKey: module.ModuleKey,
ModuleInstance: module.ModuleInstance,
@@ -135,16 +136,29 @@ func buildCorrectionLedger(runDirPath string, runOutput *runner.RunOutput) []cor
return entries
}
func filterLedgerDecisions(in []runner.ValidatorDecisionRecord, wantLLM bool, llmBacked map[string]bool) []ledgerValidatorDecisionRecord {
func HasSkippedCorrections(runOutput *runner.RunOutput) bool {
if runOutput == nil {
return false
}
for _, mr := range runOutput.ModuleResults {
if len(mr.SkippedChanges) > 0 || len(mr.ValidatorRejected) > 0 {
return true
}
}
return false
}
func filterLedgerDecisions(in []runner.ValidatorDecisionRecord, wantLLM bool) []LedgerValidatorDecisionRecord {
if len(in) == 0 {
return nil
}
out := make([]ledgerValidatorDecisionRecord, 0, len(in))
out := make([]LedgerValidatorDecisionRecord, 0, len(in))
for _, decision := range in {
if llmBacked[decision.ValidatorName] != wantLLM {
isLLMBacked := validatormetadata.ClassForKey(decision.ValidatorName) == validatormetadata.ExecutionClassLLMBacked
if isLLMBacked != wantLLM {
continue
}
out = append(out, ledgerValidatorDecisionRecord{
out = append(out, LedgerValidatorDecisionRecord{
ValidatorKey: decision.ValidatorName,
Approved: decision.Approved,
ReasonCode: decision.ReasonCode,

View File

@@ -0,0 +1,128 @@
package processreport
import (
"testing"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/runner"
)
func TestBuildCorrectionLedgerClassifiesValidatorDecisionsFromCanonicalMetadata(t *testing.T) {
output := &runner.RunOutput{
ModuleResults: []runner.ModuleResult{
{
ModuleKey: "glossary",
ModuleInstance: "glossary",
ReplacementPolicy: proposals.ReplacementPolicyReplaceAll,
ValidatorDecisions: []runner.ValidatorDecisionRecord{
{ValidatorName: "proposal_shape", ProposalIndex: 3, Approved: true, ReasonCode: "approved"},
{ValidatorName: "spoken_form_plausibility", ProposalIndex: 3, Approved: true, ReasonCode: "approved"},
},
AppliedChanges: []proposals.AppliedChange{
{
ProposalIndex: 3,
ModuleKey: "glossary",
ModuleInstance: "glossary",
TargetSegmentID: 1,
OriginalText: "gestures",
CorrectedText: "Jesters",
},
},
},
},
}
ledger := BuildCorrectionLedger(CorrectionLedgerInput{
RunDirectoryPath: "/tmp/audita-run-id",
RunOutput: output,
})
if len(ledger) != 1 {
t.Fatalf("expected one ledger entry, got %d", len(ledger))
}
entry := ledger[0]
if entry.RunID != "audita-run-id" || entry.Disposition != "applied" || entry.AppliedCorrectedText != "Jesters" {
t.Fatalf("unexpected applied ledger entry: %+v", entry)
}
if len(entry.DeterministicValidatorResults) != 1 || entry.DeterministicValidatorResults[0].ValidatorKey != "proposal_shape" {
t.Fatalf("unexpected deterministic decision split: %+v", entry.DeterministicValidatorResults)
}
if len(entry.LLMValidatorResults) != 1 || entry.LLMValidatorResults[0].ValidatorKey != "spoken_form_plausibility" {
t.Fatalf("unexpected llm-backed decision split: %+v", entry.LLMValidatorResults)
}
}
func TestBuildCorrectionLedgerPreservesDispositionPolicy(t *testing.T) {
output := &runner.RunOutput{
ModuleResults: []runner.ModuleResult{
{
ModuleKey: "grammar",
ModuleInstance: "grammar",
ReplacementPolicy: proposals.ReplacementPolicyRequireUnique,
Status: runner.ModuleStatusSuccess,
SkippedChanges: []proposals.SkippedChange{
{
ProposalIndex: 2,
TargetSegmentID: 7,
OriginalText: "old",
CorrectedText: "new",
SkipReason: proposals.SkipReasonAmbiguousOriginal,
Message: "ambiguous",
},
},
ValidatorRejected: []runner.ValidatorRejectedChange{
{
ProposalIndex: 3,
TargetSegmentID: 8,
OriginalText: "before",
CorrectedText: "after",
ReasonCode: "protected_term",
Message: "blocked",
},
},
},
{
ModuleKey: "capitalization",
ModuleInstance: "capitalization",
Status: runner.ModuleStatusFailed,
ErrorMessage: "failed",
},
},
}
ledger := BuildCorrectionLedger(CorrectionLedgerInput{RunOutput: output})
if len(ledger) != 3 {
t.Fatalf("expected skipped, rejected, and failed entries, got %+v", ledger)
}
byDisposition := make(map[string]CorrectionLedgerEntry)
for _, entry := range ledger {
byDisposition[entry.Disposition] = entry
}
if byDisposition["skipped"].DispositionReasonCode != string(proposals.SkipReasonAmbiguousOriginal) ||
byDisposition["skipped"].DispositionMessage != "ambiguous" {
t.Fatalf("unexpected skipped ledger entry: %+v", byDisposition["skipped"])
}
if byDisposition["rejected"].DispositionReasonCode != "protected_term" ||
byDisposition["rejected"].ProposedCorrectedText != "after" {
t.Fatalf("unexpected rejected ledger entry: %+v", byDisposition["rejected"])
}
if byDisposition["failed"].DispositionReasonCode != "module_failed" ||
byDisposition["failed"].DispositionMessage != "failed" {
t.Fatalf("unexpected failed ledger entry: %+v", byDisposition["failed"])
}
}
func TestHasSkippedCorrectionsIncludesApplicationSkipsAndValidatorRejections(t *testing.T) {
if HasSkippedCorrections(nil) {
t.Fatal("nil output should not have skipped corrections")
}
if HasSkippedCorrections(&runner.RunOutput{ModuleResults: []runner.ModuleResult{{AppliedChanges: []proposals.AppliedChange{{ProposalIndex: 1}}}}}) {
t.Fatal("applied-only output should not have skipped corrections")
}
if !HasSkippedCorrections(&runner.RunOutput{ModuleResults: []runner.ModuleResult{{SkippedChanges: []proposals.SkippedChange{{ProposalIndex: 1}}}}}) {
t.Fatal("application skips should count as skipped corrections")
}
if !HasSkippedCorrections(&runner.RunOutput{ModuleResults: []runner.ModuleResult{{ValidatorRejected: []runner.ValidatorRejectedChange{{ProposalIndex: 1}}}}}) {
t.Fatal("validator rejections should count as skipped corrections")
}
}

View File

@@ -0,0 +1,158 @@
package processreport
import (
"time"
"gitea.maximumdirect.net/eric/audita/internal/core/chunking"
"gitea.maximumdirect.net/eric/audita/internal/core/diagnostics"
"gitea.maximumdirect.net/eric/audita/internal/core/normalization"
"gitea.maximumdirect.net/eric/audita/internal/core/reporting"
"gitea.maximumdirect.net/eric/audita/internal/framework/runner"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
)
// BuildInput contains already-computed process execution facts for report assembly.
type BuildInput struct {
Status string
TranscriptPath string
GlossaryPath string
OutputPath string
Modules []string
OutputSchema string
ConfigVersion *int
StartedAt time.Time
CompletedAt time.Time
ErrorMessage string
ErrorPhase string
RunDirectoryPath string
NormalizationSummary *normalization.NormalizationSummary
ChunkingSummary *chunking.Summary
RunOutput *runner.RunOutput
}
// Build creates the public process report without owning command parsing or config loading.
func Build(input BuildInput) reporting.ProcessReport {
report := reporting.ProcessReport{
ReportMetadata: reporting.ReportMetadata{
ReportSchemaName: reporting.DefaultProcessReportSchemaName,
ReportSchemaVersion: reporting.DefaultProcessReportSchemaVersion,
OutputSchema: input.OutputSchema,
ConfigVersion: input.ConfigVersion,
},
Phase: "default_pipeline",
Status: input.Status,
Operation: "process",
TranscriptPath: input.TranscriptPath,
GlossaryPath: input.GlossaryPath,
OutputPath: input.OutputPath,
Modules: append([]string(nil), input.Modules...),
StartedAt: input.StartedAt,
CompletedAt: &input.CompletedAt,
ErrorPhase: input.ErrorPhase,
}
if input.RunDirectoryPath != "" {
runSucceeded := input.Status == "success"
metadata := diagnostics.BuildDiagnosticsMetadata(input.RunDirectoryPath, runSucceeded)
report.Diagnostics = &metadata
}
if input.ErrorMessage != "" {
report.ErrorMessage = input.ErrorMessage
}
if input.NormalizationSummary != nil {
report.InputSegmentCount = &input.NormalizationSummary.InputSegmentCount
report.NormalizedSegmentCount = &input.NormalizationSummary.OutputSegmentCount
report.NormalizationMerges = &input.NormalizationSummary.MergesPerformed
report.NormalizationIDReassignments = &input.NormalizationSummary.IDsReassigned
report.NormalizationSkipped.DifferentSpeakers = &input.NormalizationSummary.SkippedMerges.DifferentSpeakers
report.NormalizationSkipped.GapTooLarge = &input.NormalizationSummary.SkippedMerges.GapTooLarge
report.NormalizationSkipped.DurationExceeded = &input.NormalizationSummary.SkippedMerges.DurationExceeded
report.NormalizationSkipped.TokenLimitExceeded = &input.NormalizationSummary.SkippedMerges.TokenLimitExceeded
}
if input.ChunkingSummary != nil {
report.Chunking = &reporting.ChunkingSummary{
ChunkCount: input.ChunkingSummary.ChunkCount,
MinEstimatedTokens: input.ChunkingSummary.MinEstimatedTokens,
MaxEstimatedTokens: input.ChunkingSummary.MaxEstimatedTokens,
TotalEstimatedTokens: input.ChunkingSummary.TotalEstimatedTokens,
TargetSections: input.ChunkingSummary.TargetSections,
MaxSectionTokens: input.ChunkingSummary.MaxSectionTokens,
MinSectionTokens: input.ChunkingSummary.MinSectionTokens,
}
}
report.ModulesSummary, report.ModuleResults = buildModuleReporting(input.RunOutput)
return report
}
func buildModuleReporting(runOutput *runner.RunOutput) (*reporting.ModulesSummary, []reporting.ModuleReport) {
if runOutput == nil || len(runOutput.ModuleResults) == 0 {
return nil, nil
}
moduleReports := make([]reporting.ModuleReport, 0, len(runOutput.ModuleResults))
summary := &reporting.ModulesSummary{ModuleCount: len(runOutput.ModuleResults)}
for _, r := range runOutput.ModuleResults {
startedAt := r.StartedAt
completedAt := r.CompletedAt
moduleReports = append(moduleReports, reporting.ModuleReport{
ModuleKey: r.ModuleKey,
ModuleInstance: r.ModuleInstance,
ReplacementPolicy: string(r.ReplacementPolicy),
Status: r.Status,
ProposalCount: r.ProposalCount,
Warnings: append([]stagewarnings.StageWarning(nil), r.Warnings...),
ValidatorDecisions: mapValidatorDecisions(r.ValidatorDecisions),
ValidatorRejected: mapValidatorRejected(r.ValidatorRejected),
AppliedChanges: r.AppliedChanges,
SkippedChanges: r.SkippedChanges,
ErrorMessage: r.ErrorMessage,
StartedAt: &startedAt,
CompletedAt: &completedAt,
})
summary.TotalAppliedChanges += len(r.AppliedChanges)
summary.TotalSkippedChanges += len(r.SkippedChanges) + len(r.ValidatorRejected)
if r.Status == runner.ModuleStatusFailed && summary.FailedModuleInstance == "" {
summary.FailedModuleInstance = r.ModuleInstance
}
}
return summary, moduleReports
}
func mapValidatorDecisions(in []runner.ValidatorDecisionRecord) []reporting.ValidatorDecisionReport {
if len(in) == 0 {
return nil
}
out := make([]reporting.ValidatorDecisionReport, len(in))
for i, d := range in {
out[i] = reporting.ValidatorDecisionReport{
ValidatorName: d.ValidatorName,
ProposalIndex: d.ProposalIndex,
Approved: d.Approved,
ReasonCode: d.ReasonCode,
Message: d.Message,
DiagnosticArtifactPath: d.DiagnosticArtifactPath,
}
}
return out
}
func mapValidatorRejected(in []runner.ValidatorRejectedChange) []reporting.ValidatorRejectedReport {
if len(in) == 0 {
return nil
}
out := make([]reporting.ValidatorRejectedReport, len(in))
for i, d := range in {
out[i] = reporting.ValidatorRejectedReport{
ValidatorName: d.ValidatorName,
ProposalIndex: d.ProposalIndex,
ModuleKey: d.ModuleKey,
ModuleInstance: d.ModuleInstance,
TargetSegmentID: d.TargetSegmentID,
OriginalText: d.OriginalText,
CorrectedText: d.CorrectedText,
ReasonCode: d.ReasonCode,
Message: d.Message,
}
}
return out
}

View File

@@ -0,0 +1,180 @@
package processreport
import (
"path/filepath"
"testing"
"time"
"gitea.maximumdirect.net/eric/audita/internal/core/chunking"
"gitea.maximumdirect.net/eric/audita/internal/core/diagnostics"
"gitea.maximumdirect.net/eric/audita/internal/core/normalization"
"gitea.maximumdirect.net/eric/audita/internal/core/reporting"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/runner"
)
func TestBuildSuccessReportMapsExecutionFacts(t *testing.T) {
startedAt := time.Date(2026, 5, 23, 10, 0, 0, 0, time.UTC)
completedAt := startedAt.Add(time.Second)
configVersion := 4
targetSections := 2
inputSegments := 5
outputSegments := 4
merges := 1
reassigned := 2
differentSpeakers := 3
report := Build(BuildInput{
Status: "success",
TranscriptPath: "transcript.json",
GlossaryPath: "glossary.yaml",
OutputPath: "out.json",
Modules: []string{"grammar"},
OutputSchema: "default",
ConfigVersion: &configVersion,
StartedAt: startedAt,
CompletedAt: completedAt,
RunDirectoryPath: filepath.Join(
"tmp",
"audita-run",
),
NormalizationSummary: &normalization.NormalizationSummary{
InputSegmentCount: inputSegments,
OutputSegmentCount: outputSegments,
MergesPerformed: merges,
IDsReassigned: reassigned,
SkippedMerges: struct {
DifferentSpeakers int `json:"different_speakers"`
GapTooLarge int `json:"gap_too_large"`
DurationExceeded int `json:"duration_exceeded"`
TokenLimitExceeded int `json:"token_limit_exceeded"`
}{
DifferentSpeakers: differentSpeakers,
},
},
ChunkingSummary: &chunking.Summary{
ChunkCount: 3,
MinEstimatedTokens: 10,
MaxEstimatedTokens: 20,
TotalEstimatedTokens: 45,
TargetSections: &targetSections,
MaxSectionTokens: 200,
MinSectionTokens: 50,
},
RunOutput: &runner.RunOutput{
ModuleResults: []runner.ModuleResult{
{
ModuleKey: "grammar",
ModuleInstance: "grammar",
ReplacementPolicy: proposals.ReplacementPolicyRequireUnique,
Status: runner.ModuleStatusSuccess,
ProposalCount: 2,
ValidatorDecisions: []runner.ValidatorDecisionRecord{
{
ValidatorName: "proposal_shape",
ProposalIndex: 1,
Approved: true,
ReasonCode: "approved",
Message: "ok",
DiagnosticArtifactPath: "diagnostics/validator.json",
},
},
ValidatorRejected: []runner.ValidatorRejectedChange{
{
ValidatorName: "protected_term",
ProposalIndex: 2,
ModuleKey: "grammar",
ModuleInstance: "grammar",
TargetSegmentID: 7,
OriginalText: "old",
CorrectedText: "new",
ReasonCode: "protected_term",
Message: "blocked",
},
},
AppliedChanges: []proposals.AppliedChange{
{ProposalIndex: 1, TargetSegmentID: 7, OriginalText: "old", CorrectedText: "new"},
},
SkippedChanges: []proposals.SkippedChange{
{ProposalIndex: 3, TargetSegmentID: 8, SkipReason: proposals.SkipReasonMissingSegment},
},
StartedAt: startedAt,
CompletedAt: completedAt,
},
},
},
})
if report.ReportMetadata.ReportSchemaName != reporting.DefaultProcessReportSchemaName ||
report.ReportMetadata.ReportSchemaVersion != reporting.DefaultProcessReportSchemaVersion ||
report.ReportMetadata.OutputSchema != "default" ||
report.ReportMetadata.ConfigVersion == nil ||
*report.ReportMetadata.ConfigVersion != configVersion {
t.Fatalf("unexpected report metadata: %+v", report.ReportMetadata)
}
if report.Phase != "default_pipeline" || report.Operation != "process" || report.Status != "success" {
t.Fatalf("unexpected process identity fields: phase=%q operation=%q status=%q", report.Phase, report.Operation, report.Status)
}
if report.Diagnostics == nil || report.Diagnostics.CorrectionLedgerPath != filepath.Join("tmp", "audita-run", diagnostics.ArtifactCorrectionLedger) {
t.Fatalf("unexpected diagnostics metadata: %+v", report.Diagnostics)
}
if report.InputSegmentCount == nil || *report.InputSegmentCount != inputSegments ||
report.NormalizationSkipped.DifferentSpeakers == nil ||
*report.NormalizationSkipped.DifferentSpeakers != differentSpeakers {
t.Fatalf("unexpected normalization summary: %+v", report)
}
if report.Chunking == nil || report.Chunking.ChunkCount != 3 || report.Chunking.TargetSections == nil || *report.Chunking.TargetSections != targetSections {
t.Fatalf("unexpected chunking summary: %+v", report.Chunking)
}
if report.ModulesSummary == nil ||
report.ModulesSummary.ModuleCount != 1 ||
report.ModulesSummary.TotalAppliedChanges != 1 ||
report.ModulesSummary.TotalSkippedChanges != 2 {
t.Fatalf("unexpected modules summary: %+v", report.ModulesSummary)
}
if len(report.ModuleResults) != 1 ||
len(report.ModuleResults[0].ValidatorDecisions) != 1 ||
len(report.ModuleResults[0].ValidatorRejected) != 1 {
t.Fatalf("unexpected module reports: %+v", report.ModuleResults)
}
}
func TestBuildFailedReportPreservesErrorAndFailureSummary(t *testing.T) {
startedAt := time.Date(2026, 5, 23, 10, 0, 0, 0, time.UTC)
completedAt := startedAt.Add(time.Second)
report := Build(BuildInput{
Status: "failed",
TranscriptPath: "transcript.json",
GlossaryPath: "glossary.yaml",
Modules: []string{"grammar"},
OutputSchema: "default",
StartedAt: startedAt,
CompletedAt: completedAt,
ErrorPhase: "module",
ErrorMessage: "module failed",
RunDirectoryPath: "run-dir",
RunOutput: &runner.RunOutput{
ModuleResults: []runner.ModuleResult{
{
ModuleKey: "grammar",
ModuleInstance: "grammar",
Status: runner.ModuleStatusFailed,
ErrorMessage: "module failed",
StartedAt: startedAt,
CompletedAt: completedAt,
},
},
},
})
if report.Status != "failed" || report.ErrorPhase != "module" || report.ErrorMessage != "module failed" {
t.Fatalf("unexpected failure fields: %+v", report)
}
if report.Diagnostics == nil || report.Diagnostics.ErrorLogPath == "" {
t.Fatalf("expected failure diagnostics metadata, got %+v", report.Diagnostics)
}
if report.ModulesSummary == nil || report.ModulesSummary.FailedModuleInstance != "grammar" {
t.Fatalf("unexpected failed module summary: %+v", report.ModulesSummary)
}
}

View File

@@ -0,0 +1,45 @@
package promptcontext
import (
"encoding/json"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
)
type transcriptSectionSegment struct {
ID int `json:"id"`
Speaker string `json:"speaker"`
Start float64 `json:"start"`
End float64 `json:"end"`
Text string `json:"text"`
Categories []string `json:"categories,omitempty"`
}
type transcriptSectionPayload struct {
SectionIndex int `json:"section_index"`
Segments []transcriptSectionSegment `json:"segments"`
}
// MarshalTranscriptSectionJSON builds the standardized transcript-section JSON
// payload consumed by proposal prompt templates.
func MarshalTranscriptSectionJSON(transcript *schema.Transcript, sectionIndex int) ([]byte, error) {
payload := transcriptSectionPayload{
SectionIndex: sectionIndex,
Segments: make([]transcriptSectionSegment, 0),
}
if transcript != nil {
for _, s := range transcript.Segments {
payload.Segments = append(payload.Segments, transcriptSectionSegment{
ID: s.ID,
Speaker: s.Speaker,
Start: s.Start,
End: s.End,
Text: s.Text,
Categories: append([]string(nil), s.Categories...),
})
}
}
return json.MarshalIndent(payload, "", " ")
}

View File

@@ -0,0 +1,95 @@
package promptcontext
import (
"encoding/json"
"reflect"
"testing"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
)
func TestMarshalTranscriptSectionJSONShape(t *testing.T) {
transcript := &schema.Transcript{Segments: []schema.Segment{
{ID: 1, Speaker: "A", Start: 0.1, End: 1.2, Text: "alpha", Categories: []string{"session", "intro"}},
}}
raw, err := MarshalTranscriptSectionJSON(transcript, 3)
if err != nil {
t.Fatalf("MarshalTranscriptSectionJSON error: %v", err)
}
var decoded map[string]any
if err := json.Unmarshal(raw, &decoded); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if got := decoded["section_index"]; got != float64(3) {
t.Fatalf("section_index: got=%v want=%v", got, 3)
}
segments, ok := decoded["segments"].([]any)
if !ok || len(segments) != 1 {
t.Fatalf("segments shape mismatch: %T %+v", decoded["segments"], decoded["segments"])
}
first, ok := segments[0].(map[string]any)
if !ok {
t.Fatalf("segment shape mismatch: %T", segments[0])
}
if first["id"] != float64(1) || first["speaker"] != "A" || first["start"] != 0.1 || first["end"] != 1.2 || first["text"] != "alpha" {
t.Fatalf("unexpected segment fields: %+v", first)
}
cats, ok := first["categories"].([]any)
if !ok || len(cats) != 2 || cats[0] != "session" || cats[1] != "intro" {
t.Fatalf("unexpected categories: %+v", first["categories"])
}
}
func TestMarshalTranscriptSectionJSONEmptyTranscript(t *testing.T) {
raw, err := MarshalTranscriptSectionJSON(nil, 0)
if err != nil {
t.Fatalf("MarshalTranscriptSectionJSON error: %v", err)
}
var decoded map[string]any
if err := json.Unmarshal(raw, &decoded); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if got := decoded["section_index"]; got != float64(0) {
t.Fatalf("section_index: got=%v want=%v", got, 0)
}
segments, ok := decoded["segments"].([]any)
if !ok {
t.Fatalf("segments shape mismatch: %T", decoded["segments"])
}
if len(segments) != 0 {
t.Fatalf("expected empty segments, got %d", len(segments))
}
}
func TestMarshalTranscriptSectionJSONCopiesCategories(t *testing.T) {
transcript := &schema.Transcript{Segments: []schema.Segment{
{ID: 1, Speaker: "A", Start: 0, End: 1, Text: "alpha", Categories: []string{"kept"}},
}}
raw, err := MarshalTranscriptSectionJSON(transcript, 1)
if err != nil {
t.Fatalf("MarshalTranscriptSectionJSON error: %v", err)
}
transcript.Segments[0].Categories[0] = "changed"
var decoded struct {
Segments []struct {
Categories []string `json:"categories"`
} `json:"segments"`
}
if err := json.Unmarshal(raw, &decoded); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(decoded.Segments) != 1 {
t.Fatalf("expected one segment, got %d", len(decoded.Segments))
}
if !reflect.DeepEqual(decoded.Segments[0].Categories, []string{"kept"}) {
t.Fatalf("expected copied categories, got %v", decoded.Segments[0].Categories)
}
}

View File

@@ -15,6 +15,9 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/framework/llm"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/responseschema"
"gitea.maximumdirect.net/eric/audita/internal/framework/stagename"
"gitea.maximumdirect.net/eric/audita/internal/framework/structuredoutput"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
)
// InteractionDiagnosticsWriter writes machine-readable prompt/response artifacts.
@@ -68,6 +71,7 @@ type Request struct {
type Result struct {
Corrections []proposals.CorrectionProposal `json:"corrections"`
Enriched []proposals.EnrichedCorrectionProposal `json:"enriched"`
Warnings []stagewarnings.StageWarning `json:"warnings,omitempty"`
Artifacts InteractionArtifacts `json:"artifacts,omitempty"`
}
@@ -92,7 +96,7 @@ func GenerateCandidates(ctx context.Context, req Request) (Result, error) {
stage := strings.TrimSpace(req.StageName)
if stage == "" {
stage = buildStageName(req.ModuleInstance, req.Section)
stage = stagename.ProposalGeneration(req.ModuleInstance, sectionIndexPtr(req.Section))
}
model := resolveModel(req.Config, req.Model)
messages := append([]contracts.LLMMessage(nil), req.Messages...)
@@ -104,7 +108,7 @@ func GenerateCandidates(ctx context.Context, req Request) (Result, error) {
writer = diagnosticsWriterAdapter{
writer: llm.NewDiagnosticsWriter(
filepath.Join(req.DiagnosticsDir, req.ModuleInstance),
proposalGenerationSecrets(req.Config),
llm.ConfiguredSecrets(req.Config),
),
}
}
@@ -141,7 +145,7 @@ func GenerateCandidates(ctx context.Context, req Request) (Result, error) {
if len(req.PromptMetadata) > 0 {
requestMetadata["prompt_metadata"] = req.PromptMetadata
}
requestMetadata["response_schema"] = schemaMetadata(responseSchema)
requestMetadata["response_schema"] = responseSchema.DiagnosticsMap()
if writer != nil {
artifacts, _ = writer.WriteInteraction(
@@ -156,6 +160,12 @@ func GenerateCandidates(ctx context.Context, req Request) (Result, error) {
}
if callErr != nil {
if structuredoutput.IsMalformedError(callErr) {
return Result{
Warnings: []stagewarnings.StageWarning{newMalformedProposalWarning(req.Section, artifacts, callErr)},
Artifacts: artifacts,
}, nil
}
return Result{}, fmt.Errorf("proposal generation completion failed: %w", callErr)
}
@@ -168,9 +178,6 @@ func GenerateCandidates(ctx context.Context, req Request) (Result, error) {
CorrectedText: raw.CorrectedText,
Confidence: raw.Confidence,
}
if err := candidate.Validate(); err != nil {
return Result{}, fmt.Errorf("invalid structured correction at index %d: %w", i, err)
}
corrections = append(corrections, candidate)
enrichedCandidate := proposals.EnrichedCorrectionProposal{
@@ -191,27 +198,11 @@ func GenerateCandidates(ctx context.Context, req Request) (Result, error) {
return Result{
Corrections: corrections,
Enriched: enriched,
Warnings: nil,
Artifacts: artifacts,
}, nil
}
func schemaMetadata(schema responseschema.Schema) map[string]any {
return map[string]any{
"id": schema.ID,
"version": schema.Version,
"name": schema.Name,
"sha256": schema.SHA256,
}
}
func buildStageName(moduleInstance string, section *contracts.SectionMetadata) string {
base := fmt.Sprintf("%s:proposal-generation", moduleInstance)
if section == nil {
return base
}
return fmt.Sprintf("%s:section-%04d", base, section.Index)
}
func resolveModel(cfg *config.Config, override string) string {
if strings.TrimSpace(override) != "" {
return strings.TrimSpace(override)
@@ -222,17 +213,6 @@ func resolveModel(cfg *config.Config, override string) string {
return llm.ResolvePrimaryConfig(*cfg).Model
}
func proposalGenerationSecrets(cfg *config.Config) []string {
if cfg == nil {
return nil
}
return []string{
cfg.PrimaryLLM.APIKey,
cfg.ValidationLLM.APIKey,
cfg.EffectiveValidationLLMConfig().APIKey,
}
}
func errPayload(err error) any {
if err == nil {
return nil
@@ -240,6 +220,35 @@ func errPayload(err error) any {
return map[string]any{"error": err.Error()}
}
func newMalformedProposalWarning(section *contracts.SectionMetadata, artifacts InteractionArtifacts, err error) stagewarnings.StageWarning {
warning := stagewarnings.StageWarning{
Scope: stagewarnings.ScopeProposalGeneration,
ReasonCode: "proposal_response_malformed",
Message: strings.TrimSpace(err.Error()),
DiagnosticArtifactPath: diagnosticArtifactPath(artifacts),
}
if section != nil {
sectionIndex := section.Index
warning.SectionIndex = &sectionIndex
}
return warning
}
func diagnosticArtifactPath(artifacts InteractionArtifacts) string {
if artifacts.ErrorPayloadPath != "" {
return artifacts.ErrorPayloadPath
}
return artifacts.ResponsePayloadPath
}
func sectionIndexPtr(section *contracts.SectionMetadata) *int {
if section == nil {
return nil
}
index := section.Index
return &index
}
type diagnosticsWriterAdapter struct {
writer *llm.DiagnosticsWriter
}

View File

@@ -224,12 +224,12 @@ func TestGenerateCandidatesDiagnosticsIncludeSchemaMetadata(t *testing.T) {
}
}
func TestGenerateCandidatesMalformedStructuredResponse(t *testing.T) {
func TestGenerateCandidatesInvalidCorrectionIsPreservedForLaterValidation(t *testing.T) {
client := &fakeStructuredClient{
responses: []StructuredCorrectionSet{
{
Corrections: []StructuredCorrectionProposal{
{TargetSegmentID: 1, OriginalText: "x", CorrectedText: "", Confidence: 0.9},
{TargetSegmentID: 0, OriginalText: "x", CorrectedText: "", Confidence: 1.2},
},
},
},
@@ -237,9 +237,55 @@ func TestGenerateCandidatesMalformedStructuredResponse(t *testing.T) {
req := defaultRequest(t)
req.LLMClient = client
_, err := GenerateCandidates(context.Background(), req)
if err == nil || !strings.Contains(err.Error(), "invalid structured correction") {
t.Fatalf("expected structured response validation failure, got %v", err)
result, err := GenerateCandidates(context.Background(), req)
if err != nil {
t.Fatalf("expected invalid correction to survive generation, got %v", err)
}
if len(result.Corrections) != 1 {
t.Fatalf("expected one correction, got %+v", result)
}
if result.Corrections[0].TargetSegmentID != 0 || result.Corrections[0].Confidence != 1.2 {
t.Fatalf("unexpected preserved correction: %+v", result.Corrections[0])
}
if len(result.Warnings) != 0 {
t.Fatalf("did not expect warnings for individually invalid corrections, got %+v", result.Warnings)
}
}
func TestGenerateCandidatesMalformedStructuredOutputReturnsWarning(t *testing.T) {
client := &fakeStructuredClient{err: errors.New("malformed structured output")}
req := defaultRequest(t)
req.LLMClient = client
result, err := GenerateCandidates(context.Background(), req)
if err != nil {
t.Fatalf("expected malformed structured output to downgrade to warning, got %v", err)
}
if len(result.Corrections) != 0 || len(result.Enriched) != 0 {
t.Fatalf("expected no proposals on malformed response, got %+v", result)
}
if len(result.Warnings) != 1 {
t.Fatalf("expected one warning, got %+v", result.Warnings)
}
if result.Warnings[0].ReasonCode != "proposal_response_malformed" {
t.Fatalf("unexpected warning: %+v", result.Warnings[0])
}
}
func TestGenerateCandidatesProviderMalformedEnvelopeReturnsWarning(t *testing.T) {
client := &fakeStructuredClient{err: errors.New("provider response missing choices")}
req := defaultRequest(t)
req.LLMClient = client
result, err := GenerateCandidates(context.Background(), req)
if err != nil {
t.Fatalf("expected malformed provider envelope to downgrade to warning, got %v", err)
}
if len(result.Corrections) != 0 || len(result.Enriched) != 0 {
t.Fatalf("expected no proposals on malformed response, got %+v", result)
}
if len(result.Warnings) != 1 || result.Warnings[0].ReasonCode != "proposal_response_malformed" {
t.Fatalf("unexpected warnings: %+v", result.Warnings)
}
}
@@ -315,20 +361,22 @@ func TestGenerateCandidatesMultipleSectionsStableMetadata(t *testing.T) {
}
func TestGenerateCandidatesDiagnosticsWrittenAndRedacted(t *testing.T) {
secret := "proposal-secret"
primarySecret := "proposal-primary-secret"
validationSecret := "proposal-validation-secret"
client := &fakeStructuredClient{
responses: []StructuredCorrectionSet{
{Corrections: []StructuredCorrectionProposal{{TargetSegmentID: 1, OriginalText: secret, CorrectedText: "safe", Confidence: 0.9}}},
{Corrections: []StructuredCorrectionProposal{{TargetSegmentID: 1, OriginalText: validationSecret, CorrectedText: "safe", Confidence: 0.9}}},
},
}
cfg := config.Default()
cfg.PrimaryLLM.APIKey = secret
cfg.PrimaryLLM.APIKey = primarySecret
cfg.ValidationLLM.APIKey = validationSecret
req := defaultRequest(t)
req.Config = &cfg
req.LLMClient = client
req.DiagnosticsDir = t.TempDir()
req.Messages = []contracts.LLMMessage{
{Role: "system", Content: "include secret " + secret},
{Role: "system", Content: "include secret " + primarySecret},
{Role: "user", Content: "fix it"},
}
@@ -345,8 +393,8 @@ func TestGenerateCandidatesDiagnosticsWrittenAndRedacted(t *testing.T) {
if readErr != nil {
t.Fatalf("read artifact %q: %v", path, readErr)
}
if strings.Contains(string(raw), secret) {
t.Fatalf("artifact leaked secret %q: %s", path, string(raw))
if strings.Contains(string(raw), primarySecret) || strings.Contains(string(raw), validationSecret) {
t.Fatalf("artifact leaked configured secret in %q: %s", path, string(raw))
}
if !strings.Contains(string(raw), "[REDACTED]") {
t.Fatalf("expected redaction marker in artifact %q: %s", path, string(raw))

View File

@@ -0,0 +1,81 @@
package proposal_generation
import (
"context"
"fmt"
"strings"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/stagename"
"gitea.maximumdirect.net/eric/audita/internal/prompts"
)
type ProposalMessageBuilder func(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error)
type ModuleProposalRequest struct {
ProposalRequest contracts.ProposalRequest
PromptID string
BuildMessages ProposalMessageBuilder
}
// ExecuteModuleProposal runs shared proposal generation plumbing for one
// module, leaving only module-specific prompt message building at call sites.
func ExecuteModuleProposal(ctx context.Context, req ModuleProposalRequest) (contracts.ProposalResult, error) {
if req.BuildMessages == nil {
return contracts.ProposalResult{}, fmt.Errorf("proposal message builder is required")
}
if strings.TrimSpace(req.PromptID) == "" {
return contracts.ProposalResult{}, fmt.Errorf("prompt ID must not be empty")
}
sectionIndex := 0
if req.ProposalRequest.Section != nil {
sectionIndex = req.ProposalRequest.Section.Index
}
transcriptDescription := ""
if req.ProposalRequest.Config != nil {
transcriptDescription = req.ProposalRequest.Config.TranscriptDescription
}
messages, err := req.BuildMessages(
req.ProposalRequest.WorkingTranscript,
req.ProposalRequest.Glossary,
sectionIndex,
transcriptDescription,
)
if err != nil {
return contracts.ProposalResult{}, err
}
promptMetadata, ok := prompts.LookupMetadata(req.PromptID)
if !ok {
return contracts.ProposalResult{}, fmt.Errorf("unknown prompt ID %q", req.PromptID)
}
generated, err := GenerateCandidates(ctx, Request{
ModuleKey: req.ProposalRequest.RunSpec.ModuleKey,
ModuleInstance: req.ProposalRequest.RunSpec.InstanceName,
ReplacementPolicy: req.ProposalRequest.RunSpec.ReplacementPolicy,
WorkingTranscript: req.ProposalRequest.WorkingTranscript,
Section: req.ProposalRequest.Section,
Glossary: req.ProposalRequest.Glossary,
Config: req.ProposalRequest.Config,
Messages: messages,
PromptMetadata: promptMetadata.DiagnosticsMap(),
StageName: stagename.ModuleProposal(req.ProposalRequest.RunSpec.InstanceName, sectionIndexPtr(req.ProposalRequest.Section)),
StartIndex: 0,
LLMClient: req.ProposalRequest.LLMClient,
Scheduler: req.ProposalRequest.LLMScheduler,
DiagnosticsDir: req.ProposalRequest.DiagnosticsDir,
})
if err != nil {
return contracts.ProposalResult{}, err
}
return contracts.ProposalResult{
Proposals: generated.Corrections,
Warnings: generated.Warnings,
}, nil
}

View File

@@ -0,0 +1,115 @@
package proposal_generation
import (
"context"
"errors"
"testing"
"gitea.maximumdirect.net/eric/audita/internal/core/config"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/prompts"
)
func TestExecuteModuleProposalBuildsMessagesFromSectionAndDescription(t *testing.T) {
client := &fakeStructuredClient{
responses: []StructuredCorrectionSet{
{Corrections: []StructuredCorrectionProposal{{TargetSegmentID: 1, OriginalText: "teh", CorrectedText: "the", Confidence: 0.9}}},
},
}
section := contracts.SectionMetadata{Index: 7}
cfg := config.Default()
cfg.TranscriptDescription = "Hearing transcript with role titles."
transcript := &schema.Transcript{Segments: []schema.Segment{{ID: 1, Speaker: "A", Start: 0, End: 1, Text: "teh"}}}
glossary := &schema.Glossary{Entries: []schema.GlossaryEntry{{Name: "X"}}}
var gotSectionIndex int
var gotDescription string
var gotTranscript *schema.Transcript
var gotGlossary *schema.Glossary
out, err := ExecuteModuleProposal(context.Background(), ModuleProposalRequest{
ProposalRequest: contracts.ProposalRequest{
ExecutionContext: contracts.ExecutionContext{
Config: &cfg,
WorkingTranscript: transcript,
Glossary: glossary,
Section: &section,
},
RunSpec: contracts.ModuleRunSpec{
ModuleKey: "grammar",
InstanceName: "grammar",
ReplacementPolicy: proposals.ReplacementPolicyRequireUnique,
},
LLMClient: client,
},
PromptID: prompts.PromptIDModuleGrammarProposal,
BuildMessages: func(inTranscript *schema.Transcript, inGlossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
gotSectionIndex = sectionIndex
gotDescription = transcriptDescription
gotTranscript = inTranscript
gotGlossary = inGlossary
return []contracts.LLMMessage{{Role: "system", Content: "sys"}, {Role: "user", Content: "usr"}}, nil
},
})
if err != nil {
t.Fatalf("ExecuteModuleProposal error: %v", err)
}
if gotSectionIndex != 7 {
t.Fatalf("section index: got=%d want=%d", gotSectionIndex, 7)
}
if gotDescription != cfg.TranscriptDescription {
t.Fatalf("transcript description: got=%q want=%q", gotDescription, cfg.TranscriptDescription)
}
if gotTranscript != transcript {
t.Fatalf("expected shared transcript pointer")
}
if gotGlossary != glossary {
t.Fatalf("expected shared glossary pointer")
}
if len(client.calls) != 1 || client.calls[0].StageName != "grammar:proposal:section-0007" {
t.Fatalf("unexpected stage name calls: %+v", client.calls)
}
if len(out.Proposals) != 1 || out.Proposals[0].CorrectedText != "the" {
t.Fatalf("unexpected proposals: %+v", out)
}
}
func TestExecuteModuleProposalValidatesInputs(t *testing.T) {
if _, err := ExecuteModuleProposal(context.Background(), ModuleProposalRequest{}); err == nil {
t.Fatalf("expected missing message builder error")
}
_, err := ExecuteModuleProposal(context.Background(), ModuleProposalRequest{
BuildMessages: func(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
return nil, nil
},
})
if err == nil {
t.Fatalf("expected empty prompt ID error")
}
_, err = ExecuteModuleProposal(context.Background(), ModuleProposalRequest{
PromptID: "missing.prompt.id",
BuildMessages: func(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
return []contracts.LLMMessage{{Role: "system", Content: "sys"}, {Role: "user", Content: "usr"}}, nil
},
})
if err == nil {
t.Fatalf("expected unknown prompt ID error")
}
}
func TestExecuteModuleProposalPropagatesBuilderError(t *testing.T) {
wantErr := errors.New("builder failed")
_, err := ExecuteModuleProposal(context.Background(), ModuleProposalRequest{
PromptID: prompts.PromptIDModuleGlossaryProposal,
BuildMessages: func(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
return nil, wantErr
},
})
if !errors.Is(err, wantErr) {
t.Fatalf("expected builder error, got %v", err)
}
}

View File

@@ -147,6 +147,8 @@ func skipReasonMessage(reason ProposalSkipReason) string {
return "original_text matched multiple spans under require_unique policy"
case SkipReasonNoEffect:
return "original_text and corrected_text are identical"
case SkipReasonEmptyResultingText:
return "proposal would leave the segment empty"
case SkipReasonInvalidProposal:
return "proposal failed structural or policy validation"
default:

View File

@@ -15,6 +15,7 @@ const (
SkipReasonMissingOriginalText ProposalSkipReason = "missing_original_text"
SkipReasonAmbiguousOriginal ProposalSkipReason = "ambiguous_original_text"
SkipReasonNoEffect ProposalSkipReason = "no_effect"
SkipReasonEmptyResultingText ProposalSkipReason = "empty_resulting_segment"
SkipReasonInvalidProposal ProposalSkipReason = "invalid_proposal"
)
@@ -62,6 +63,9 @@ func PreviewProposalForSegment(segment *schema.Segment, proposal CorrectionPropo
}
corrected := strings.Replace(segment.Text, proposal.OriginalText, proposal.CorrectedText, 1)
if strings.TrimSpace(corrected) == "" {
return SegmentPreviewResult{SkipReason: SkipReasonEmptyResultingText}
}
return SegmentPreviewResult{
Applicable: true,
CorrectedSegmentText: corrected,
@@ -70,6 +74,9 @@ func PreviewProposalForSegment(segment *schema.Segment, proposal CorrectionPropo
case ReplacementPolicyReplaceAll:
corrected := strings.ReplaceAll(segment.Text, proposal.OriginalText, proposal.CorrectedText)
if strings.TrimSpace(corrected) == "" {
return SegmentPreviewResult{SkipReason: SkipReasonEmptyResultingText}
}
return SegmentPreviewResult{
Applicable: true,
CorrectedSegmentText: corrected,

View File

@@ -121,6 +121,42 @@ func TestPreviewProposalForSegmentNoEffectReplacement(t *testing.T) {
}
}
func TestPreviewProposalForSegmentAllowsEmptyCorrectedTextWhenSegmentRemainsNonEmpty(t *testing.T) {
segment := &schema.Segment{ID: 8, Text: "uh hello"}
proposal := CorrectionProposal{
TargetSegmentID: 8,
OriginalText: "uh ",
CorrectedText: "",
Confidence: 0.9,
}
result := PreviewProposalForSegment(segment, proposal, ReplacementPolicyRequireUnique)
if !result.Applicable {
t.Fatalf("expected applicable preview, got skip reason %q", result.SkipReason)
}
if result.CorrectedSegmentText != "hello" {
t.Fatalf("unexpected corrected text: %q", result.CorrectedSegmentText)
}
}
func TestPreviewProposalForSegmentRejectsEmptyResultingSegment(t *testing.T) {
segment := &schema.Segment{ID: 8, Text: "uh"}
proposal := CorrectionProposal{
TargetSegmentID: 8,
OriginalText: "uh",
CorrectedText: "",
Confidence: 0.9,
}
result := PreviewProposalForSegment(segment, proposal, ReplacementPolicyRequireUnique)
if result.Applicable {
t.Fatal("expected non-applicable preview")
}
if result.SkipReason != SkipReasonEmptyResultingText {
t.Fatalf("expected skip reason %q, got %q", SkipReasonEmptyResultingText, result.SkipReason)
}
}
func TestPreviewProposalForSegmentPreservesInputSegment(t *testing.T) {
segment := &schema.Segment{ID: 9, Speaker: "A", Start: 1.0, End: 2.0, Text: "rank rank"}
original := *segment

View File

@@ -38,9 +38,6 @@ func (p CorrectionProposal) Validate() error {
if strings.TrimSpace(p.OriginalText) == "" {
return fmt.Errorf("proposal original_text must not be empty")
}
if strings.TrimSpace(p.CorrectedText) == "" {
return fmt.Errorf("proposal corrected_text must not be empty")
}
if p.Confidence < 0.0 || p.Confidence > 1.0 {
return fmt.Errorf("proposal confidence must be between 0.0 and 1.0")
}

View File

@@ -35,7 +35,7 @@ func TestCorrectionProposalValidate_InvalidEmptyOriginalText(t *testing.T) {
}
}
func TestCorrectionProposalValidate_InvalidEmptyCorrectedText(t *testing.T) {
func TestCorrectionProposalValidate_AllowsEmptyCorrectedText(t *testing.T) {
proposal := CorrectionProposal{
TargetSegmentID: 42,
OriginalText: "gestures",
@@ -43,12 +43,8 @@ func TestCorrectionProposalValidate_InvalidEmptyCorrectedText(t *testing.T) {
Confidence: 0.95,
}
err := proposal.Validate()
if err == nil {
t.Fatal("expected validation error, got nil")
}
if err.Error() != "proposal corrected_text must not be empty" {
t.Fatalf("unexpected error: %v", err)
if err := proposal.Validate(); err != nil {
t.Fatalf("expected empty corrected_text to be allowed, got %v", err)
}
}

View File

@@ -0,0 +1,41 @@
{
"type": "object",
"additionalProperties": false,
"required": [
"corrections"
],
"properties": {
"corrections": {
"type": "array",
"items": {
"type": "object",
"additionalProperties": false,
"required": [
"id",
"original_text",
"corrected_text",
"confidence"
],
"properties": {
"id": {
"type": "integer",
"minimum": 1
},
"original_text": {
"type": "string",
"minLength": 1
},
"corrected_text": {
"type": "string",
"minLength": 1
},
"confidence": {
"type": "number",
"minimum": 0,
"maximum": 1
}
}
}
}
}
}

View File

@@ -0,0 +1,39 @@
{
"type": "object",
"additionalProperties": false,
"required": [
"validations"
],
"properties": {
"validations": {
"type": "array",
"items": {
"type": "object",
"additionalProperties": false,
"required": [
"correction_index",
"approved",
"confidence",
"reason"
],
"properties": {
"correction_index": {
"type": "integer",
"minimum": 0
},
"approved": {
"type": "boolean"
},
"confidence": {
"type": "number",
"minimum": 0,
"maximum": 1
},
"reason": {
"type": "string"
}
}
}
}
}
}

View File

@@ -2,9 +2,11 @@ package responseschema
import (
"crypto/sha256"
"embed"
"encoding/hex"
"encoding/json"
"fmt"
"sort"
"strings"
)
@@ -19,6 +21,9 @@ const (
schemaVersionV1 = "v1"
)
//go:embed assets/*.json
var schemaAssets embed.FS
// Schema describes one registered structured response schema.
type Schema struct {
ID string `json:"id"`
@@ -28,21 +33,44 @@ type Schema struct {
SHA256 string `json:"sha256"`
}
func (s Schema) DiagnosticsMap() map[string]any {
return map[string]any{
"id": s.ID,
"version": s.Version,
"name": s.Name,
"sha256": s.SHA256,
}
}
var registry = map[Key]Schema{
CorrectionSetKey: mustBuildSchema(
CorrectionSetKey: mustBuildSchemaFromAsset(
correctionSetSchemaID,
schemaVersionV1,
"audita_correction_set_v1",
[]byte(`{"type":"object","additionalProperties":false,"required":["corrections"],"properties":{"corrections":{"type":"array","items":{"type":"object","additionalProperties":false,"required":["id","original_text","corrected_text","confidence"],"properties":{"id":{"type":"integer","minimum":1},"original_text":{"type":"string","minLength":1},"corrected_text":{"type":"string","minLength":1},"confidence":{"type":"number","minimum":0,"maximum":1}}}}}}`),
"assets/correction_set.v1.json",
),
ValidatorDecisionSetKey: mustBuildSchema(
ValidatorDecisionSetKey: mustBuildSchemaFromAsset(
validatorDecisionSchemaID,
schemaVersionV1,
"audita_validator_decision_set_v1",
[]byte(`{"type":"object","additionalProperties":false,"required":["validations"],"properties":{"validations":{"type":"array","items":{"type":"object","additionalProperties":false,"required":["correction_index","approved","confidence","reason"],"properties":{"correction_index":{"type":"integer","minimum":0},"approved":{"type":"boolean"},"confidence":{"type":"number","minimum":0,"maximum":1},"reason":{"type":"string"}}}}}}`),
"assets/validator_decision_set.v1.json",
),
}
func Registered() []Schema {
keys := make([]string, 0, len(registry))
for key := range registry {
keys = append(keys, string(key))
}
sort.Strings(keys)
out := make([]Schema, 0, len(keys))
for _, key := range keys {
out = append(out, cloneSchema(registry[Key(key)]))
}
return out
}
// Lookup returns a copy of the registered schema for the provided key.
func Lookup(key Key) (Schema, bool) {
schema, ok := registry[key]
@@ -69,6 +97,14 @@ func cloneSchema(in Schema) Schema {
return out
}
func mustBuildSchemaFromAsset(id string, version string, name string, assetPath string) Schema {
rawSchema, err := schemaAssets.ReadFile(assetPath)
if err != nil {
panic(fmt.Sprintf("read response schema asset %q: %v", assetPath, err))
}
return mustBuildSchema(id, version, name, rawSchema)
}
func mustBuildSchema(id string, version string, name string, rawSchema []byte) Schema {
id = strings.TrimSpace(id)
version = strings.TrimSpace(version)

View File

@@ -1,14 +1,16 @@
package responseschema
import (
"bytes"
"crypto/sha256"
"encoding/hex"
"encoding/json"
"testing"
)
const (
expectedCorrectionSetSHA256 = "05f8ff3fa04f68115c0cb1859d2656f51aa5c0bae8ff2470b2d4f6f531953195"
expectedValidatorDecisionSetSHA256 = "b73f4790b98fbb955f0aec5496dd8ce9a8fe14aa2f35c700b4b4e5634f106fd5"
expectedCorrectionSetSHA256 = "b86a2dde38d7f440d26470fa8830167512bb0aa1b35e5fee5547057be583c388"
expectedValidatorDecisionSetSHA256 = "2fe90d450e2595b57885aba91c8cc5cedf783dd36758ab430309e3eace401f54"
)
func TestLookupKnownSchemas(t *testing.T) {
@@ -47,6 +49,20 @@ func TestLookupUnknownSchema(t *testing.T) {
}
}
func TestRegisteredSchemasLoadReadableEmbeddedJSON(t *testing.T) {
for _, schema := range Registered() {
if len(schema.JSONSchema) == 0 {
t.Fatalf("expected non-empty JSON schema for %q", schema.ID)
}
if !json.Valid(schema.JSONSchema) {
t.Fatalf("expected valid JSON schema for %q", schema.ID)
}
if !bytes.Contains(schema.JSONSchema, []byte("\n ")) {
t.Fatalf("expected readable formatted JSON schema for %q", schema.ID)
}
}
}
func TestSchemaHashesMatchRegisteredJSON(t *testing.T) {
expectedByKey := map[Key]string{
CorrectionSetKey: expectedCorrectionSetSHA256,
@@ -89,3 +105,20 @@ func TestLookupReturnsSchemaCopy(t *testing.T) {
t.Fatalf("expected lookup to return independent schema copy")
}
}
func TestDiagnosticsMapIncludesStableSchemaMetadataShapeForAllSchemas(t *testing.T) {
registered := Registered()
if len(registered) == 0 {
t.Fatalf("expected registered response schemas")
}
for _, schema := range registered {
metadataMap := schema.DiagnosticsMap()
if metadataMap["id"] != schema.ID ||
metadataMap["version"] != schema.Version ||
metadataMap["name"] != schema.Name ||
metadataMap["sha256"] != schema.SHA256 {
t.Fatalf("unexpected diagnostics metadata map for %q: %+v", schema.ID, metadataMap)
}
}
}

View File

@@ -15,6 +15,7 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/framework/llm"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/validators"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
validatormetadata "gitea.maximumdirect.net/eric/audita/internal/validators/metadata"
)
@@ -37,18 +38,19 @@ type ValidationScheduler = contracts.LLMScheduler
// ModuleResult captures deterministic per-module execution output.
type ModuleResult struct {
ModuleKey string `json:"module_key"`
ModuleInstance string `json:"module_instance"`
ReplacementPolicy proposals.ReplacementPolicy `json:"replacement_policy"`
Status string `json:"status"`
ProposalCount int `json:"proposal_count"`
ValidatorDecisions []ValidatorDecisionRecord `json:"validator_decisions,omitempty"`
ValidatorRejected []ValidatorRejectedChange `json:"validator_rejected,omitempty"`
AppliedChanges []proposals.AppliedChange `json:"applied_changes,omitempty"`
SkippedChanges []proposals.SkippedChange `json:"skipped_changes,omitempty"`
ErrorMessage string `json:"error_message,omitempty"`
StartedAt time.Time `json:"started_at"`
CompletedAt time.Time `json:"completed_at"`
ModuleKey string `json:"module_key"`
ModuleInstance string `json:"module_instance"`
ReplacementPolicy proposals.ReplacementPolicy `json:"replacement_policy"`
Status string `json:"status"`
ProposalCount int `json:"proposal_count"`
Warnings []stagewarnings.StageWarning `json:"warnings,omitempty"`
ValidatorDecisions []ValidatorDecisionRecord `json:"validator_decisions,omitempty"`
ValidatorRejected []ValidatorRejectedChange `json:"validator_rejected,omitempty"`
AppliedChanges []proposals.AppliedChange `json:"applied_changes,omitempty"`
SkippedChanges []proposals.SkippedChange `json:"skipped_changes,omitempty"`
ErrorMessage string `json:"error_message,omitempty"`
StartedAt time.Time `json:"started_at"`
CompletedAt time.Time `json:"completed_at"`
}
type ValidatorDecisionRecord struct {
@@ -176,6 +178,7 @@ func (r *Runner) Run(ctx context.Context, input RunInput) (RunOutput, error) {
ReplacementPolicy: policy,
Status: ModuleStatusFailed,
ProposalCount: pipelineResult.ProposalCount,
Warnings: pipelineResult.Warnings,
ValidatorDecisions: pipelineResult.ValidatorDecisions,
ValidatorRejected: pipelineResult.ValidatorRejected,
ErrorMessage: pipelineErr.Error(),
@@ -199,6 +202,7 @@ func (r *Runner) Run(ctx context.Context, input RunInput) (RunOutput, error) {
ReplacementPolicy: policy,
Status: ModuleStatusSuccess,
ProposalCount: pipelineResult.ProposalCount,
Warnings: pipelineResult.Warnings,
ValidatorDecisions: pipelineResult.ValidatorDecisions,
ValidatorRejected: pipelineResult.ValidatorRejected,
AppliedChanges: applyResult.Applied,
@@ -232,12 +236,14 @@ type collectSectionProposalsInput struct {
type sectionProposals struct {
meta contracts.SectionMetadata
corrected []proposals.CorrectionProposal
warnings []stagewarnings.StageWarning
}
type sectionProposalResult struct {
sectionPos int
section chunking.Section
corrected []proposals.CorrectionProposal
warnings []stagewarnings.StageWarning
err error
}
@@ -247,12 +253,14 @@ type sectionValidationResult struct {
approved []proposals.EnrichedCorrectionProposal
decisions []ValidatorDecisionRecord
rejected []ValidatorRejectedChange
warnings []stagewarnings.StageWarning
err error
}
type modulePipelineResult struct {
ProposalCount int
Approved []proposals.EnrichedCorrectionProposal
Warnings []stagewarnings.StageWarning
ValidatorDecisions []ValidatorDecisionRecord
ValidatorRejected []ValidatorRejectedChange
}
@@ -271,7 +279,7 @@ func collectSectionProposals(ctx context.Context, input collectSectionProposalsI
go func() {
defer wg.Done()
meta := contracts.SectionMetadataFromSection(section)
corrected, err := input.Module.Propose(withModuleInstanceContext(runCtx, input.Spec.InstanceName), contracts.ProposalRequest{
proposalResult, err := input.Module.Propose(withModuleInstanceContext(runCtx, input.Spec.InstanceName), contracts.ProposalRequest{
ExecutionContext: contracts.ExecutionContext{
Config: input.Config,
WorkingTranscript: transcriptFromSection(section),
@@ -291,7 +299,8 @@ func collectSectionProposals(ctx context.Context, input collectSectionProposalsI
case results <- sectionProposalResult{
sectionPos: sectionPos,
section: section,
corrected: corrected,
corrected: proposalResult.Proposals,
warnings: proposalResult.Warnings,
err: err,
}:
case <-runCtx.Done():
@@ -308,6 +317,7 @@ func collectSectionProposals(ctx context.Context, input collectSectionProposalsI
func runModulePipeline(ctx context.Context, input collectSectionProposalsInput) (modulePipelineResult, error) {
out := modulePipelineResult{
Approved: make([]proposals.EnrichedCorrectionProposal, 0),
Warnings: make([]stagewarnings.StageWarning, 0),
ValidatorDecisions: make([]ValidatorDecisionRecord, 0),
ValidatorRejected: make([]ValidatorRejectedChange, 0),
}
@@ -352,6 +362,7 @@ func runModulePipeline(ctx context.Context, input collectSectionProposalsInput)
if firstErr != nil {
continue
}
out.Warnings = append(out.Warnings, result.warnings...)
pending[result.sectionPos] = result
for {
@@ -401,6 +412,7 @@ func runModulePipeline(ctx context.Context, input collectSectionProposalsInput)
approved: validated.approved,
decisions: validated.decisions,
rejected: validated.rejected,
warnings: validated.warnings,
err: err,
}
}(nextSectionToProcess, sectionEnriched, sectionMeta)
@@ -424,6 +436,7 @@ func runModulePipeline(ctx context.Context, input collectSectionProposalsInput)
break
}
out.Approved = append(out.Approved, res.approved...)
out.Warnings = append(out.Warnings, res.warnings...)
out.ValidatorDecisions = append(out.ValidatorDecisions, res.decisions...)
out.ValidatorRejected = append(out.ValidatorRejected, res.rejected...)
}
@@ -440,6 +453,38 @@ func runModulePipeline(ctx context.Context, input collectSectionProposalsInput)
}
return validatorOrder[out.ValidatorRejected[i].ValidatorName] < validatorOrder[out.ValidatorRejected[j].ValidatorName]
})
sort.SliceStable(out.Warnings, func(i, j int) bool {
leftSection, rightSection := -1, -1
if out.Warnings[i].SectionIndex != nil {
leftSection = *out.Warnings[i].SectionIndex
}
if out.Warnings[j].SectionIndex != nil {
rightSection = *out.Warnings[j].SectionIndex
}
if leftSection != rightSection {
return leftSection < rightSection
}
leftBatch, rightBatch := -1, -1
if out.Warnings[i].BatchIndex != nil {
leftBatch = *out.Warnings[i].BatchIndex
}
if out.Warnings[j].BatchIndex != nil {
rightBatch = *out.Warnings[j].BatchIndex
}
if leftBatch != rightBatch {
return leftBatch < rightBatch
}
if out.Warnings[i].ValidatorName != out.Warnings[j].ValidatorName {
return validatorOrder[out.Warnings[i].ValidatorName] < validatorOrder[out.Warnings[j].ValidatorName]
}
if out.Warnings[i].Scope != out.Warnings[j].Scope {
return out.Warnings[i].Scope < out.Warnings[j].Scope
}
if out.Warnings[i].ReasonCode != out.Warnings[j].ReasonCode {
return out.Warnings[i].ReasonCode < out.Warnings[j].ReasonCode
}
return out.Warnings[i].Message < out.Warnings[j].Message
})
if firstErr != nil {
return out, firstErr
@@ -467,11 +512,13 @@ type validateSectionCandidatesResult struct {
approved []proposals.EnrichedCorrectionProposal
decisions []ValidatorDecisionRecord
rejected []ValidatorRejectedChange
warnings []stagewarnings.StageWarning
}
func validateSectionCandidates(ctx context.Context, input validateSectionCandidatesInput) (validateSectionCandidatesResult, error) {
decisions := make([]ValidatorDecisionRecord, 0)
rejected := make([]ValidatorRejectedChange, 0)
warnings := make([]stagewarnings.StageWarning, 0)
eligible := append([]proposals.EnrichedCorrectionProposal(nil), input.SectionEnriched...)
for _, validator := range input.Validators {
@@ -480,7 +527,7 @@ func validateSectionCandidates(ctx context.Context, input validateSectionCandida
diagnosticsWriter = &llmDiagnosticsWriterAdapter{
writer: llm.NewDiagnosticsWriter(
filepath.Join(input.DiagnosticsDir, input.Spec.InstanceName),
validatorSecrets(input.Config),
llm.ConfiguredSecrets(input.Config),
),
}
}
@@ -512,6 +559,7 @@ func validateSectionCandidates(ctx context.Context, input validateSectionCandida
approved: eligible,
decisions: decisions,
rejected: rejected,
warnings: warnings,
}, fmt.Errorf("validator %q failed: %w", validator.Name(), err)
}
if err := validators.EnforceDecisionCardinality(eligible, vResult.Decisions); err != nil {
@@ -519,8 +567,10 @@ func validateSectionCandidates(ctx context.Context, input validateSectionCandida
approved: eligible,
decisions: decisions,
rejected: rejected,
warnings: warnings,
}, fmt.Errorf("validator %q cardinality failed: %w", validator.Name(), err)
}
warnings = append(warnings, vResult.Warnings...)
nextEligible := make([]proposals.EnrichedCorrectionProposal, 0, len(eligible))
byIndex := make(map[int]proposals.EnrichedCorrectionProposal, len(eligible))
@@ -562,6 +612,7 @@ func validateSectionCandidates(ctx context.Context, input validateSectionCandida
approved: eligible,
decisions: decisions,
rejected: rejected,
warnings: warnings,
}, nil
}
@@ -693,15 +744,3 @@ func (a *llmDiagnosticsWriterAdapter) WriteInteraction(stage string, requestMeta
ErrorPayloadPath: art.ErrorPayloadPath,
}, nil
}
func validatorSecrets(cfg *config.Config) []string {
if cfg == nil {
return nil
}
effective := cfg.EffectiveValidationLLMConfig()
return []string{
cfg.PrimaryLLM.APIKey,
effective.APIKey,
cfg.ValidationLLM.APIKey,
}
}

View File

@@ -47,11 +47,12 @@ type fakeModule struct {
func (m fakeModule) Key() string { return m.key }
func (m fakeModule) ReplacementPolicy() proposals.ReplacementPolicy { return m.policy }
func (m fakeModule) Validators() []contracts.Validator { return m.validators }
func (m fakeModule) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
func (m fakeModule) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
if m.proposeF == nil {
return nil, nil
return contracts.ProposalResult{}, nil
}
return m.proposeF(req)
proposalsOut, err := m.proposeF(req)
return contracts.ProposalResult{Proposals: proposalsOut}, err
}
type fakeValidator struct {
@@ -1191,7 +1192,7 @@ func TestRunnerLLMValidatorRejectionPreventsApplication(t *testing.T) {
}
}
func TestRunnerLLMValidatorMalformedResponseFailsWithPartialProgress(t *testing.T) {
func TestRunnerLLMValidatorMalformedResponseRejectsBatchAndKeepsPartialProgress(t *testing.T) {
client := &fakeStructuredClient{responses: []validators.LLMValidationResponse{{Validations: []validators.LLMValidationDecision{{CorrectionIndex: 99, Approved: true, Confidence: 0.9, Reason: "bad index"}}}}}
llmValidator, _ := validators.NewLLMBackedValidator("spoken_form_plausibility_review", validators.LLMValidatorTypeSpokenFormPlausibility, "")
r := New(fakeFactory{modules: map[string]contracts.TranscriptModule{
@@ -1209,15 +1210,21 @@ func TestRunnerLLMValidatorMalformedResponseFailsWithPartialProgress(t *testing.
ModuleSpecs: []contracts.ModuleRunSpec{{ModuleKey: "m1", InstanceName: "m1"}, {ModuleKey: "m2", InstanceName: "m2"}},
ValidationLLMClient: client,
})
if err == nil {
t.Fatal("expected llm validator failure")
if err != nil {
t.Fatalf("expected malformed validator response to downgrade, got %v", err)
}
if out.FinalTranscript.Segments[0].Text != "the cat" {
t.Fatalf("expected partial progress retained")
}
if len(out.ModuleResults) != 2 || len(out.ModuleResults[1].ValidatorRejected) != 1 {
t.Fatalf("expected second module rejection, got %+v", out.ModuleResults)
}
if len(out.ModuleResults[1].Warnings) != 1 || out.ModuleResults[1].Warnings[0].ReasonCode != validators.ReasonValidatorMalformed {
t.Fatalf("expected malformed warning, got %+v", out.ModuleResults[1].Warnings)
}
}
func TestRunnerLLMValidatorMissingDecisionFails(t *testing.T) {
func TestRunnerLLMValidatorMissingDecisionRejectsBatch(t *testing.T) {
client := &fakeStructuredClient{responses: []validators.LLMValidationResponse{{Validations: []validators.LLMValidationDecision{}}}}
llmValidator, _ := validators.NewLLMBackedValidator("spoken_form_plausibility_review", validators.LLMValidatorTypeSpokenFormPlausibility, "")
r := New(fakeFactory{modules: map[string]contracts.TranscriptModule{
@@ -1232,12 +1239,12 @@ func TestRunnerLLMValidatorMissingDecisionFails(t *testing.T) {
ModuleSpecs: []contracts.ModuleRunSpec{{ModuleKey: "m", InstanceName: "m"}},
ValidationLLMClient: client,
})
if err == nil {
t.Fatal("expected missing decision failure")
if err != nil {
t.Fatalf("expected missing decision downgrade, got %v", err)
}
}
func TestRunnerLLMValidatorDuplicateDecisionFails(t *testing.T) {
func TestRunnerLLMValidatorDuplicateDecisionRejectsBatch(t *testing.T) {
client := &fakeStructuredClient{responses: []validators.LLMValidationResponse{{Validations: []validators.LLMValidationDecision{
{CorrectionIndex: 0, Approved: true, Confidence: 0.9, Reason: "ok"},
{CorrectionIndex: 0, Approved: false, Confidence: 0.9, Reason: "dup"},
@@ -1255,8 +1262,8 @@ func TestRunnerLLMValidatorDuplicateDecisionFails(t *testing.T) {
ModuleSpecs: []contracts.ModuleRunSpec{{ModuleKey: "m", InstanceName: "m"}},
ValidationLLMClient: client,
})
if err == nil {
t.Fatal("expected duplicate decision failure")
if err != nil {
t.Fatalf("expected duplicate decision downgrade, got %v", err)
}
}
@@ -1300,12 +1307,13 @@ func TestRunnerLLMValidatorBatchingAndSchedulerUsage(t *testing.T) {
}
func TestRunnerLLMValidatorDiagnosticsWrittenAndRedacted(t *testing.T) {
secret := "super-secret-key"
client := &fakeStructuredClient{responses: []validators.LLMValidationResponse{{Validations: []validators.LLMValidationDecision{{CorrectionIndex: 0, Approved: true, Confidence: 0.9, Reason: secret}}}}}
primarySecret := "runner-primary-secret"
validationSecret := "runner-validation-secret"
client := &fakeStructuredClient{responses: []validators.LLMValidationResponse{{Validations: []validators.LLMValidationDecision{{CorrectionIndex: 0, Approved: true, Confidence: 0.9, Reason: validationSecret}}}}}
llmValidator, _ := validators.NewLLMBackedValidator("spoken_form_plausibility_review", validators.LLMValidatorTypeSpokenFormPlausibility, "")
cfg := config.Default()
cfg.PrimaryLLM.APIKey = secret
cfg.ValidationLLM.APIKey = secret
cfg.PrimaryLLM.APIKey = primarySecret
cfg.ValidationLLM.APIKey = validationSecret
diagDir := t.TempDir()
r := New(fakeFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{key: "m", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{llmValidator}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
@@ -1314,7 +1322,7 @@ func TestRunnerLLMValidatorDiagnosticsWrittenAndRedacted(t *testing.T) {
}})
out, err := r.Run(context.Background(), RunInput{
Config: &cfg,
Transcript: &schema.Transcript{Segments: []schema.Segment{{ID: 1, Text: "There were gestures at the temple."}}},
Transcript: &schema.Transcript{Segments: []schema.Segment{{ID: 1, Text: "There were gestures at the temple. " + primarySecret}}},
ModuleSpecs: []contracts.ModuleRunSpec{{ModuleKey: "m", InstanceName: "m"}},
ValidationLLMClient: client,
ValidationDiagnosticsDir: diagDir,
@@ -1325,12 +1333,26 @@ func TestRunnerLLMValidatorDiagnosticsWrittenAndRedacted(t *testing.T) {
if len(out.ModuleResults[0].ValidatorDecisions) == 0 || out.ModuleResults[0].ValidatorDecisions[0].DiagnosticArtifactPath == "" {
t.Fatalf("expected diagnostic artifact path on decision")
}
matches, globErr := filepath.Glob(filepath.Join(diagDir, "m", "*.json"))
if globErr != nil {
t.Fatalf("glob diagnostics: %v", globErr)
}
if len(matches) == 0 {
t.Fatalf("expected diagnostics JSON artifacts under %s", filepath.Join(diagDir, "m"))
}
for _, path := range matches {
raw, readErr := os.ReadFile(path)
if readErr != nil {
t.Fatalf("read diagnostic %q: %v", path, readErr)
}
if strings.Contains(string(raw), primarySecret) || strings.Contains(string(raw), validationSecret) {
t.Fatalf("configured secret leaked in diagnostics %q: %s", path, string(raw))
}
}
raw, readErr := os.ReadFile(out.ModuleResults[0].ValidatorDecisions[0].DiagnosticArtifactPath)
if readErr != nil {
t.Fatalf("read diagnostic: %v", readErr)
}
if strings.Contains(string(raw), secret) {
t.Fatalf("secret leaked in diagnostics: %s", string(raw))
t.Fatalf("read decision diagnostic: %v", readErr)
}
if !strings.Contains(string(raw), "[REDACTED]") {
t.Fatalf("expected redaction marker in diagnostics")
@@ -1355,7 +1377,7 @@ type proposalGenerationModule struct {
func (m proposalGenerationModule) Key() string { return m.key }
func (m proposalGenerationModule) ReplacementPolicy() proposals.ReplacementPolicy { return m.policy }
func (m proposalGenerationModule) Validators() []contracts.Validator { return nil }
func (m proposalGenerationModule) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
func (m proposalGenerationModule) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
result, err := proposal_generation.GenerateCandidates(ctx, proposal_generation.Request{
ModuleKey: req.RunSpec.ModuleKey,
ModuleInstance: req.RunSpec.InstanceName,
@@ -1372,9 +1394,9 @@ func (m proposalGenerationModule) Propose(ctx context.Context, req contracts.Pro
DiagnosticsDir: req.DiagnosticsDir,
})
if err != nil {
return nil, err
return contracts.ProposalResult{}, err
}
return result.Corrections, nil
return contracts.ProposalResult{Proposals: result.Corrections, Warnings: result.Warnings}, nil
}
type fakeProposalStructuredClient struct {

View File

@@ -0,0 +1,24 @@
package stagename
import (
"fmt"
)
func ModuleProposal(moduleInstance string, sectionIndex *int) string {
if sectionIndex == nil || *sectionIndex == 0 {
return fmt.Sprintf("%s:proposal", moduleInstance)
}
return fmt.Sprintf("%s:proposal:section-%04d", moduleInstance, *sectionIndex)
}
func ProposalGeneration(moduleInstance string, sectionIndex *int) string {
base := fmt.Sprintf("%s:proposal-generation", moduleInstance)
if sectionIndex == nil {
return base
}
return fmt.Sprintf("%s:section-%04d", base, *sectionIndex)
}
func ValidatorBatch(moduleInstance string, validatorName string, batchIndex int) string {
return fmt.Sprintf("%s:%s:batch-%04d", moduleInstance, validatorName, batchIndex)
}

View File

@@ -0,0 +1,33 @@
package stagename
import (
"testing"
)
func TestModuleProposalStageName(t *testing.T) {
if got := ModuleProposal("grammar", nil); got != "grammar:proposal" {
t.Fatalf("unexpected stage name without section: %q", got)
}
sectionIndex := 7
if got := ModuleProposal("grammar", &sectionIndex); got != "grammar:proposal:section-0007" {
t.Fatalf("unexpected stage name with section: %q", got)
}
}
func TestProposalGenerationStageName(t *testing.T) {
if got := ProposalGeneration("grammar", nil); got != "grammar:proposal-generation" {
t.Fatalf("unexpected proposal generation stage name without section: %q", got)
}
sectionIndex := 3
if got := ProposalGeneration("grammar", &sectionIndex); got != "grammar:proposal-generation:section-0003" {
t.Fatalf("unexpected proposal generation stage name with section: %q", got)
}
}
func TestValidatorBatchStageName(t *testing.T) {
if got := ValidatorBatch("homophones_1", "spoken_form_plausibility_review", 12); got != "homophones_1:spoken_form_plausibility_review:batch-0012" {
t.Fatalf("unexpected validator batch stage name: %q", got)
}
}

View File

@@ -0,0 +1,28 @@
package structuredoutput
import "strings"
var malformedMarkers = []string{
"malformed structured output",
"decode structured output:",
"decode provider response envelope:",
"provider response missing choices",
"provider response missing assistant message content",
"provider response assistant message content is empty",
"provider response assistant message content is not valid JSON",
}
// IsMalformedError reports whether err matches provider malformed
// structured-output failure markers that should be downgraded.
func IsMalformedError(err error) bool {
if err == nil {
return false
}
msg := err.Error()
for _, marker := range malformedMarkers {
if strings.Contains(msg, marker) {
return true
}
}
return false
}

View File

@@ -0,0 +1,32 @@
package structuredoutput
import (
"errors"
"testing"
)
func TestIsMalformedError(t *testing.T) {
cases := []struct {
name string
err error
want bool
}{
{name: "nil", err: nil, want: false},
{name: "generic", err: errors.New("network timeout"), want: false},
{name: "malformed", err: errors.New("malformed structured output"), want: true},
{name: "decode structured", err: errors.New("decode structured output: unexpected end of JSON input"), want: true},
{name: "missing choices", err: errors.New("provider response missing choices"), want: true},
{name: "missing content", err: errors.New("provider response missing assistant message content"), want: true},
{name: "empty content", err: errors.New("provider response assistant message content is empty"), want: true},
{name: "invalid content json", err: errors.New("provider response assistant message content is not valid JSON"), want: true},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
got := IsMalformedError(tc.err)
if got != tc.want {
t.Fatalf("IsMalformedError(%v): got=%v want=%v", tc.err, got, tc.want)
}
})
}
}

View File

@@ -4,8 +4,35 @@ import (
"context"
"fmt"
"strings"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
)
type ProposalShapeValidator struct{}
func (v ProposalShapeValidator) Name() string { return "proposal_shape" }
func (v ProposalShapeValidator) Validate(_ context.Context, req Request) (Result, error) {
decisions := make([]Decision, 0, len(req.CandidateProposal))
for _, c := range req.CandidateProposal {
switch {
case c.TargetSegmentID <= 0:
decisions = append(decisions, rejection(c.ProposalIndex, ReasonInvalidTargetSegment, "proposal target segment id must be positive"))
case strings.TrimSpace(c.OriginalText) == "":
decisions = append(decisions, rejection(c.ProposalIndex, ReasonEmptyOriginalText, "proposal original_text must not be empty"))
case c.Confidence < 0.0 || c.Confidence > 1.0:
decisions = append(decisions, rejection(c.ProposalIndex, ReasonInvalidConfidence, "proposal confidence must be between 0.0 and 1.0"))
default:
decisions = append(decisions, approval(c.ProposalIndex))
}
}
if err := EnforceDecisionCardinality(req.CandidateProposal, decisions); err != nil {
return Result{}, err
}
return Result{ValidatorName: v.Name(), Decisions: decisions}, nil
}
type ConfidenceThresholdValidator struct{}
func (v ConfidenceThresholdValidator) Name() string { return "confidence_threshold" }
@@ -62,10 +89,23 @@ type NonEmptyCorrectionValidator struct{}
func (v NonEmptyCorrectionValidator) Name() string { return "non_empty_corrected_text" }
func (v NonEmptyCorrectionValidator) Validate(_ context.Context, req Request) (Result, error) {
segmentsByID := make(map[int]schema.Segment)
if req.WorkingTranscript != nil {
for _, seg := range req.WorkingTranscript.Segments {
segmentsByID[seg.ID] = seg
}
}
decisions := make([]Decision, 0, len(req.CandidateProposal))
for _, c := range req.CandidateProposal {
if strings.TrimSpace(c.CorrectedText) == "" {
decisions = append(decisions, rejection(c.ProposalIndex, ReasonEmptyCorrectedText, "corrected_text must not be empty"))
segment, ok := segmentsByID[c.TargetSegmentID]
if !ok {
decisions = append(decisions, approval(c.ProposalIndex))
continue
}
preview := proposals.PreviewProposalForSegment(&segment, c.CorrectionProposal, req.ReplacementPolicy)
if preview.SkipReason == proposals.SkipReasonEmptyResultingText {
decisions = append(decisions, rejection(c.ProposalIndex, ReasonEmptyResultingText, "proposal would leave the segment empty"))
continue
}
decisions = append(decisions, approval(c.ProposalIndex))

View File

@@ -11,6 +11,9 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/responseschema"
"gitea.maximumdirect.net/eric/audita/internal/framework/stagename"
"gitea.maximumdirect.net/eric/audita/internal/framework/structuredoutput"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
"gitea.maximumdirect.net/eric/audita/internal/prompts"
)
@@ -78,7 +81,20 @@ func (v *LLMBackedValidator) Validate(ctx context.Context, req Request) (Result,
maxTokens = req.Config.ValidationMaxPromptTokens
}
batches, err := ChunkLLMValidationItems(validationReq.Items, maxTokens, v.estimator)
warnings := append([]stagewarnings.StageWarning(nil), oversizedValidationWarnings(v.name, maxTokens, validationReq.Items, v.estimator)...)
oversized := oversizedValidationDecisions(validationReq.Items, maxTokens, v.estimator)
itemsForBatching := filterItemsByDecision(validationReq.Items, oversized)
if len(itemsForBatching) == 0 {
all := append([]Decision(nil), immediate...)
all = append(all, oversized...)
if err := EnforceDecisionCardinality(req.CandidateProposal, all); err != nil {
return Result{}, err
}
sort.SliceStable(all, func(i, j int) bool { return all[i].ProposalIndex < all[j].ProposalIndex })
return Result{ValidatorName: v.name, Decisions: all, Warnings: warnings}, nil
}
batches, err := ChunkLLMValidationItems(itemsForBatching, maxTokens, v.estimator)
if err != nil {
return Result{}, err
}
@@ -96,9 +112,10 @@ func (v *LLMBackedValidator) Validate(ctx context.Context, req Request) (Result,
var response LLMValidationResponse
responseSchema := responseschema.MustLookup(responseschema.ValidatorDecisionSetKey)
stage := stagename.ValidatorBatch(req.ModuleInstance, v.name, batch.BatchIndex)
call := func(callCtx context.Context) error {
_, err = req.LLMClient.CompleteStructured(callCtx, StructuredCompletionRequest{
StageName: fmt.Sprintf("%s:%s:batch-%04d", req.ModuleInstance, v.name, batch.BatchIndex),
StageName: stage,
Messages: messages,
Model: resolvedValidationModel(req.Config, v.model),
ResponseSchema: &responseSchema,
@@ -112,27 +129,15 @@ func (v *LLMBackedValidator) Validate(ctx context.Context, req Request) (Result,
}
artifacts := InteractionArtifacts{}
if req.DiagnosticsWriter != nil {
stage := fmt.Sprintf("%s:%s:batch-%04d", req.ModuleInstance, v.name, batch.BatchIndex)
promptMetadata := validatorPromptMetadata(v.validatorType)
artifacts, _ = req.DiagnosticsWriter.WriteInteraction(
stage,
map[string]any{
"validator_name": v.name,
"validator_type": v.validatorType,
"batch_index": batch.BatchIndex,
"prompt_metadata": map[string]any{
"prompt_id": promptMetadata.PromptID,
"prompt_version": promptMetadata.PromptVersion,
"prompt_source": promptMetadata.PromptSource,
"embedded_path": promptMetadata.EmbeddedPath,
"sha256": promptMetadata.SHA256,
},
"response_schema": map[string]any{
"id": responseSchema.ID,
"version": responseSchema.Version,
"name": responseSchema.Name,
"sha256": responseSchema.SHA256,
},
"validator_name": v.name,
"validator_type": v.validatorType,
"batch_index": batch.BatchIndex,
"prompt_metadata": promptMetadata.DiagnosticsMap(),
"response_schema": responseSchema.DiagnosticsMap(),
},
map[string]any{"messages": messages, "items": batch.Items},
response,
@@ -140,12 +145,19 @@ func (v *LLMBackedValidator) Validate(ctx context.Context, req Request) (Result,
)
}
if err != nil {
if structuredoutput.IsMalformedError(err) {
llmDecisions = append(llmDecisions, rejectBatch(batch.Items, ReasonValidatorMalformed, fmt.Sprintf("validator response malformed: %s", strings.TrimSpace(err.Error())))...)
warnings = append(warnings, newValidatorWarning(v.name, batch.BatchIndex, ReasonValidatorMalformed, err.Error(), artifacts))
continue
}
return Result{}, fmt.Errorf("LLM validator %q completion failed: %w", v.name, err)
}
batchDecisions, err := mapLLMResponseToDecisions(batch.Items, response)
if err != nil {
return Result{}, fmt.Errorf("LLM validator %q response invalid: %w", v.name, err)
llmDecisions = append(llmDecisions, rejectBatch(batch.Items, ReasonValidatorMalformed, fmt.Sprintf("validator response malformed: %s", strings.TrimSpace(err.Error())))...)
warnings = append(warnings, newValidatorWarning(v.name, batch.BatchIndex, ReasonValidatorMalformed, err.Error(), artifacts))
continue
}
for i := range batchDecisions {
batchDecisions[i].DiagnosticArtifactPath = artifacts.ResponsePayloadPath
@@ -154,12 +166,13 @@ func (v *LLMBackedValidator) Validate(ctx context.Context, req Request) (Result,
}
all := append([]Decision(nil), immediate...)
all = append(all, oversized...)
all = append(all, llmDecisions...)
if err := EnforceDecisionCardinality(req.CandidateProposal, all); err != nil {
return Result{}, err
}
sort.SliceStable(all, func(i, j int) bool { return all[i].ProposalIndex < all[j].ProposalIndex })
return Result{ValidatorName: v.name, Decisions: all}, nil
return Result{ValidatorName: v.name, Decisions: all, Warnings: warnings}, nil
}
func validatorPromptMetadata(validatorType LLMValidatorType) prompts.Metadata {
@@ -301,3 +314,77 @@ func mapLLMResponseToDecisions(items []LLMValidationItem, response LLMValidation
}
return decisions, nil
}
func oversizedValidationDecisions(items []LLMValidationItem, maxPromptTokens int, estimator chunking.TokenEstimator) []Decision {
out := make([]Decision, 0)
for _, item := range items {
singleTokens, err := estimateBatchTokens(estimator, []LLMValidationItem{item})
if err != nil || singleTokens <= maxPromptTokens {
continue
}
out = append(out, rejection(item.CorrectionIndex, ReasonValidatorInputTooLarge, "validation input exceeds max prompt tokens"))
}
return out
}
func oversizedValidationWarnings(validatorName string, maxPromptTokens int, items []LLMValidationItem, estimator chunking.TokenEstimator) []stagewarnings.StageWarning {
out := make([]stagewarnings.StageWarning, 0)
for _, item := range items {
singleTokens, err := estimateBatchTokens(estimator, []LLMValidationItem{item})
if err != nil || singleTokens <= maxPromptTokens {
continue
}
out = append(out, stagewarnings.StageWarning{
Scope: stagewarnings.ScopeValidator,
ValidatorName: validatorName,
ReasonCode: ReasonValidatorInputTooLarge,
Message: fmt.Sprintf("validation input exceeds max prompt tokens for proposal %d", item.CorrectionIndex),
})
}
return out
}
func filterItemsByDecision(items []LLMValidationItem, decisions []Decision) []LLMValidationItem {
if len(decisions) == 0 {
return append([]LLMValidationItem(nil), items...)
}
rejected := make(map[int]struct{}, len(decisions))
for _, decision := range decisions {
rejected[decision.ProposalIndex] = struct{}{}
}
out := make([]LLMValidationItem, 0, len(items))
for _, item := range items {
if _, ok := rejected[item.CorrectionIndex]; ok {
continue
}
out = append(out, item)
}
return out
}
func rejectBatch(items []LLMValidationItem, reasonCode string, message string) []Decision {
out := make([]Decision, 0, len(items))
for _, item := range items {
out = append(out, rejection(item.CorrectionIndex, reasonCode, message))
}
return out
}
func newValidatorWarning(validatorName string, batchIndex int, reasonCode string, message string, artifacts InteractionArtifacts) stagewarnings.StageWarning {
idx := batchIndex
return stagewarnings.StageWarning{
Scope: stagewarnings.ScopeValidator,
ValidatorName: validatorName,
BatchIndex: &idx,
ReasonCode: reasonCode,
Message: strings.TrimSpace(message),
DiagnosticArtifactPath: diagnosticArtifactPath(artifacts),
}
}
func diagnosticArtifactPath(artifacts InteractionArtifacts) string {
if artifacts.ErrorPayloadPath != "" {
return artifacts.ErrorPayloadPath
}
return artifacts.ResponsePayloadPath
}

View File

@@ -248,29 +248,55 @@ func TestLLMBackedValidatorApprovalAndRejection(t *testing.T) {
}
}
func TestLLMBackedValidatorMalformedOutputFails(t *testing.T) {
func TestLLMBackedValidatorMalformedOutputRejectsBatch(t *testing.T) {
client := &fakeStructuredLLMClient{err: errors.New("malformed structured output")}
v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model")
req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")})
req.LLMClient = client
_, err := v.Validate(context.Background(), req)
if err == nil || !strings.Contains(err.Error(), "completion failed") {
t.Fatalf("expected malformed output error, got %v", err)
res, err := v.Validate(context.Background(), req)
if err != nil {
t.Fatalf("expected malformed output downgrade, got %v", err)
}
if len(res.Decisions) != 1 || res.Decisions[0].Approved || res.Decisions[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("unexpected decisions: %+v", res.Decisions)
}
if len(res.Warnings) != 1 || res.Warnings[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("expected malformed warning, got %+v", res.Warnings)
}
}
func TestLLMBackedValidatorMissingDecisionFails(t *testing.T) {
func TestLLMBackedValidatorProviderMalformedEnvelopeRejectsBatch(t *testing.T) {
client := &fakeStructuredLLMClient{err: errors.New("provider response assistant message content is empty")}
v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model")
req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")})
req.LLMClient = client
res, err := v.Validate(context.Background(), req)
if err != nil {
t.Fatalf("expected malformed provider envelope downgrade, got %v", err)
}
if len(res.Decisions) != 1 || res.Decisions[0].Approved || res.Decisions[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("unexpected decisions: %+v", res.Decisions)
}
if len(res.Warnings) != 1 || res.Warnings[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("expected malformed warning, got %+v", res.Warnings)
}
}
func TestLLMBackedValidatorMissingDecisionRejectsBatch(t *testing.T) {
client := &fakeStructuredLLMClient{responses: []LLMValidationResponse{{Validations: []LLMValidationDecision{}}}}
v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model")
req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")})
req.LLMClient = client
_, err := v.Validate(context.Background(), req)
if err == nil || !strings.Contains(err.Error(), "response invalid") {
t.Fatalf("expected missing decision error, got %v", err)
res, err := v.Validate(context.Background(), req)
if err != nil {
t.Fatalf("expected missing decision downgrade, got %v", err)
}
if len(res.Decisions) != 1 || res.Decisions[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("unexpected decisions: %+v", res.Decisions)
}
}
func TestLLMBackedValidatorDuplicateDecisionFails(t *testing.T) {
func TestLLMBackedValidatorDuplicateDecisionRejectsBatch(t *testing.T) {
client := &fakeStructuredLLMClient{responses: []LLMValidationResponse{{Validations: []LLMValidationDecision{
{CorrectionIndex: 0, Approved: true, Confidence: 0.9, Reason: "ok"},
{CorrectionIndex: 0, Approved: false, Confidence: 0.9, Reason: "dup"},
@@ -278,20 +304,74 @@ func TestLLMBackedValidatorDuplicateDecisionFails(t *testing.T) {
v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model")
req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")})
req.LLMClient = client
_, err := v.Validate(context.Background(), req)
if err == nil || !strings.Contains(err.Error(), "duplicate") {
t.Fatalf("expected duplicate decision error, got %v", err)
res, err := v.Validate(context.Background(), req)
if err != nil {
t.Fatalf("expected duplicate decision downgrade, got %v", err)
}
if len(res.Decisions) != 1 || res.Decisions[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("unexpected decisions: %+v", res.Decisions)
}
}
func TestLLMBackedValidatorUnknownProposalIndexFails(t *testing.T) {
func TestLLMBackedValidatorUnknownProposalIndexRejectsBatch(t *testing.T) {
client := &fakeStructuredLLMClient{responses: []LLMValidationResponse{{Validations: []LLMValidationDecision{{CorrectionIndex: 99, Approved: true, Confidence: 0.9, Reason: "unknown"}}}}}
v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model")
req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")})
req.LLMClient = client
_, err := v.Validate(context.Background(), req)
if err == nil || !strings.Contains(err.Error(), "unknown") {
t.Fatalf("expected unknown index error, got %v", err)
res, err := v.Validate(context.Background(), req)
if err != nil {
t.Fatalf("expected unknown index downgrade, got %v", err)
}
if len(res.Decisions) != 1 || res.Decisions[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("unexpected decisions: %+v", res.Decisions)
}
}
func TestLLMBackedValidatorOversizedSingleProposalRejectsOnlyThatProposal(t *testing.T) {
client := &fakeStructuredLLMClient{responses: []LLMValidationResponse{{Validations: []LLMValidationDecision{
{CorrectionIndex: 1, Approved: true, Confidence: 0.9, Reason: "ok"},
}}}}
v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model")
huge := strings.Repeat("gestures ", 200)
req := Request{
WorkingTranscript: &schema.Transcript{Segments: []schema.Segment{
{ID: 1, Text: huge},
{ID: 2, Text: "There were gestures at the temple.", Categories: []string{"narration"}},
}},
CandidateProposal: []proposals.EnrichedCorrectionProposal{
{
CorrectionProposal: proposals.CorrectionProposal{TargetSegmentID: 1, OriginalText: huge, CorrectedText: "Jesters", Confidence: 0.9},
ProposalMetadata: proposals.ProposalMetadata{ProposalIndex: 0, ModuleKey: "homophones", ModuleInstance: "homophones"},
},
{
CorrectionProposal: proposals.CorrectionProposal{TargetSegmentID: 2, OriginalText: "gestures", CorrectedText: "Jesters", Confidence: 0.9},
ProposalMetadata: proposals.ProposalMetadata{ProposalIndex: 1, ModuleKey: "homophones", ModuleInstance: "homophones"},
},
},
ModuleKey: "homophones",
ModuleInstance: "homophones",
ReplacementPolicy: proposals.ReplacementPolicyRequireUnique,
}
req.LLMClient = client
cfg := config.Default()
cfg.ValidationMaxPromptTokens = 200
req.Config = &cfg
res, err := v.Validate(context.Background(), req)
if err != nil {
t.Fatalf("expected oversize downgrade, got %v", err)
}
if len(res.Decisions) != 2 {
t.Fatalf("expected two decisions, got %+v", res.Decisions)
}
if res.Decisions[0].ReasonCode != ReasonValidatorInputTooLarge || res.Decisions[0].Approved {
t.Fatalf("expected first decision oversize rejection, got %+v", res.Decisions[0])
}
if !res.Decisions[1].Approved {
t.Fatalf("expected second decision approved, got %+v", res.Decisions[1])
}
if len(res.Warnings) != 1 || res.Warnings[0].ReasonCode != ReasonValidatorInputTooLarge {
t.Fatalf("expected one oversize warning, got %+v", res.Warnings)
}
}

View File

@@ -6,18 +6,25 @@ import (
"strings"
"gitea.maximumdirect.net/eric/audita/internal/core/config"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
)
const (
ReasonApproved = "approved"
ReasonLowConfidence = "low_confidence"
ReasonMissingOriginalText = "missing_original_text"
ReasonMissingTargetSegment = "missing_target_segment"
ReasonEmptyCorrectedText = "empty_corrected_text"
ReasonNoEffect = "no_effect"
ReasonProtectedGlossaryTerm = "protected_glossary_term"
ReasonApproved = "approved"
ReasonLowConfidence = "low_confidence"
ReasonMissingOriginalText = "missing_original_text"
ReasonMissingTargetSegment = "missing_target_segment"
ReasonEmptyResultingText = "empty_resulting_segment"
ReasonNoEffect = "no_effect"
ReasonProtectedGlossaryTerm = "protected_glossary_term"
ReasonInvalidTargetSegment = "invalid_target_segment_id"
ReasonEmptyOriginalText = "empty_original_text"
ReasonInvalidConfidence = "invalid_confidence"
ReasonValidatorMalformed = "validator_response_malformed"
ReasonValidatorInputTooLarge = "validator_input_too_large"
)
// Request is the runtime input shared by deterministic validators.
@@ -45,8 +52,9 @@ type Decision struct {
// Result is one validator output containing exactly one decision per proposal index.
type Result struct {
ValidatorName string `json:"validator_name"`
Decisions []Decision `json:"decisions"`
ValidatorName string `json:"validator_name"`
Decisions []Decision `json:"decisions"`
Warnings []stagewarnings.StageWarning `json:"warnings,omitempty"`
}
// ValidationScheduler provides bounded execution for validator LLM calls.
@@ -111,13 +119,13 @@ func confidenceThresholdForModule(moduleKey string, cfg *config.Config) float64
return 0.0
}
switch moduleKey {
case "glossary":
case modulecatalog.KeyGlossary:
return cfg.Thresholds.Glossary
case "grammar":
case modulecatalog.KeyGrammar:
return cfg.Thresholds.Grammar
case "homophones":
case modulecatalog.KeyHomophones:
return cfg.Thresholds.Homophones
case "spoken_word":
case modulecatalog.KeySpokenWord:
return cfg.Thresholds.SpokenWord
default:
return 0.0

View File

@@ -68,6 +68,31 @@ func TestConfidenceThresholdValidator(t *testing.T) {
}
}
func TestProposalShapeValidator(t *testing.T) {
req := Request{CandidateProposal: []proposals.EnrichedCorrectionProposal{
mkCandidate(0, 1, "teh", "the", 0.9),
mkCandidate(1, 0, "teh", "the", 0.9),
mkCandidate(2, 1, " ", "the", 0.9),
mkCandidate(3, 1, "teh", "the", 1.5),
}}
res, err := (ProposalShapeValidator{}).Validate(context.Background(), req)
if err != nil {
t.Fatalf("Validate error: %v", err)
}
if !res.Decisions[0].Approved {
t.Fatalf("expected proposal 0 approved")
}
if res.Decisions[1].ReasonCode != ReasonInvalidTargetSegment {
t.Fatalf("expected invalid target segment rejection, got %+v", res.Decisions[1])
}
if res.Decisions[2].ReasonCode != ReasonEmptyOriginalText {
t.Fatalf("expected empty original rejection, got %+v", res.Decisions[2])
}
if res.Decisions[3].ReasonCode != ReasonInvalidConfidence {
t.Fatalf("expected invalid confidence rejection, got %+v", res.Decisions[3])
}
}
func TestOriginalTextPresenceValidator(t *testing.T) {
req := Request{WorkingTranscript: &schema.Transcript{Segments: []schema.Segment{{ID: 1, Text: "hello world"}}}, CandidateProposal: []proposals.EnrichedCorrectionProposal{
mkCandidate(0, 1, "hello", "hi", 0.9),
@@ -90,10 +115,13 @@ func TestOriginalTextPresenceValidator(t *testing.T) {
}
func TestNonEmptyCorrectionValidator(t *testing.T) {
req := Request{CandidateProposal: []proposals.EnrichedCorrectionProposal{
mkCandidate(0, 1, "hello", "hi", 0.9),
mkCandidate(1, 1, "hello", " ", 0.9),
}}
req := Request{
WorkingTranscript: &schema.Transcript{Segments: []schema.Segment{{ID: 1, Text: "hello world"}}},
ReplacementPolicy: proposals.ReplacementPolicyRequireUnique,
CandidateProposal: []proposals.EnrichedCorrectionProposal{
mkCandidate(0, 1, "hello", "hi", 0.9),
mkCandidate(1, 1, "hello world", " ", 0.9),
}}
res, err := (NonEmptyCorrectionValidator{}).Validate(context.Background(), req)
if err != nil {
t.Fatalf("Validate error: %v", err)
@@ -101,8 +129,8 @@ func TestNonEmptyCorrectionValidator(t *testing.T) {
if !res.Decisions[0].Approved {
t.Fatalf("expected proposal 0 approved")
}
if res.Decisions[1].ReasonCode != ReasonEmptyCorrectedText {
t.Fatalf("expected empty_corrected_text, got %+v", res.Decisions[1])
if res.Decisions[1].ReasonCode != ReasonEmptyResultingText {
t.Fatalf("expected empty_resulting_segment, got %+v", res.Decisions[1])
}
}
@@ -151,9 +179,14 @@ func TestStableReasonCodes(t *testing.T) {
ReasonLowConfidence,
ReasonMissingOriginalText,
ReasonMissingTargetSegment,
ReasonEmptyCorrectedText,
ReasonEmptyResultingText,
ReasonNoEffect,
ReasonProtectedGlossaryTerm,
ReasonInvalidTargetSegment,
ReasonEmptyOriginalText,
ReasonInvalidConfidence,
ReasonValidatorMalformed,
ReasonValidatorInputTooLarge,
}
for _, code := range codes {
if strings.TrimSpace(code) == "" {

View File

@@ -0,0 +1,18 @@
package warnings
type Scope string
const (
ScopeProposalGeneration Scope = "proposal_generation"
ScopeValidator Scope = "validator"
)
type StageWarning struct {
Scope Scope `json:"scope"`
ReasonCode string `json:"reason_code"`
Message string `json:"message"`
SectionIndex *int `json:"section_index,omitempty"`
ValidatorName string `json:"validator_name,omitempty"`
BatchIndex *int `json:"batch_index,omitempty"`
DiagnosticArtifactPath string `json:"diagnostic_artifact_path,omitempty"`
}

View File

@@ -2,12 +2,11 @@ package glossary
import (
"context"
"fmt"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/prompts"
builtinvalidators "gitea.maximumdirect.net/eric/audita/internal/validators"
)
@@ -36,65 +35,10 @@ func (m *Module) Validators() []contracts.Validator {
return append([]contracts.Validator(nil), m.validators...)
}
func (m *Module) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
sectionTranscript := transcriptForSection(req.WorkingTranscript, req.Section)
sectionIndex := 0
if req.Section != nil {
sectionIndex = req.Section.Index
}
transcriptDescription := ""
if req.Config != nil {
transcriptDescription = req.Config.TranscriptDescription
}
messages, err := BuildProposalMessages(sectionTranscript, req.Glossary, sectionIndex, transcriptDescription)
if err != nil {
return nil, err
}
generated, err := proposal_generation.GenerateCandidates(ctx, proposal_generation.Request{
ModuleKey: req.RunSpec.ModuleKey,
ModuleInstance: req.RunSpec.InstanceName,
ReplacementPolicy: req.RunSpec.ReplacementPolicy,
WorkingTranscript: req.WorkingTranscript,
Section: req.Section,
Glossary: req.Glossary,
Config: req.Config,
Messages: messages,
PromptMetadata: map[string]any{
"prompt_id": proposalPromptMetadata().PromptID,
"prompt_version": proposalPromptMetadata().PromptVersion,
"prompt_source": proposalPromptMetadata().PromptSource,
"embedded_path": proposalPromptMetadata().EmbeddedPath,
"sha256": proposalPromptMetadata().SHA256,
},
StageName: proposalStageName(req),
StartIndex: 0,
LLMClient: req.LLMClient,
Scheduler: req.LLMScheduler,
DiagnosticsDir: req.DiagnosticsDir,
func (m *Module) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
return proposal_generation.ExecuteModuleProposal(ctx, proposal_generation.ModuleProposalRequest{
ProposalRequest: req,
PromptID: prompts.PromptIDModuleGlossaryProposal,
BuildMessages: BuildProposalMessages,
})
if err != nil {
return nil, err
}
return generated.Corrections, nil
}
func transcriptForSection(transcript *schema.Transcript, section *contracts.SectionMetadata) *schema.Transcript {
if section == nil || transcript == nil {
return transcript
}
segments := make([]schema.Segment, 0, len(transcript.Segments))
for _, seg := range transcript.Segments {
if seg.ID >= section.StartSegmentID && seg.ID <= section.EndSegmentID {
segments = append(segments, seg)
}
}
return &schema.Transcript{Segments: segments}
}
func proposalStageName(req contracts.ProposalRequest) string {
if req.Section == nil || req.Section.Index == 0 {
return fmt.Sprintf("%s:proposal", req.RunSpec.InstanceName)
}
return fmt.Sprintf("%s:proposal:section-%04d", req.RunSpec.InstanceName, req.Section.Index)
}

View File

@@ -129,6 +129,7 @@ func TestGlossaryModuleValidatorChain(t *testing.T) {
got = append(got, v.Name())
}
want := []string{
"proposal_shape",
"no_effect",
"original_text_presence",
"confidence_threshold",
@@ -185,7 +186,7 @@ func TestGlossaryModuleProposeMapsCorrectionsAndWritesDiagnostics(t *testing.T)
if len(client.calls) != 1 || client.calls[0].StageName != "glossary:proposal" {
t.Fatalf("expected one glossary:proposal call, got %+v", client.calls)
}
if len(out) != 1 || out[0].CorrectedText != "Jesters" {
if len(out.Proposals) != 1 || out.Proposals[0].CorrectedText != "Jesters" {
t.Fatalf("unexpected proposals: %+v", out)
}
diagFiles, globErr := filepath.Glob(filepath.Join(diagDir, "glossary", "*proposal*response-payload.json"))

View File

@@ -10,43 +10,13 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/prompts"
)
type promptSegment struct {
ID int `json:"id"`
Speaker string `json:"speaker"`
Start float64 `json:"start"`
End float64 `json:"end"`
Text string `json:"text"`
Categories []string `json:"categories,omitempty"`
}
type promptTranscriptSection struct {
SectionIndex int `json:"section_index"`
Segments []promptSegment `json:"segments"`
}
func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
glossaryJSON, err := json.MarshalIndent(glossary, "", " ")
if err != nil {
return nil, fmt.Errorf("marshal glossary prompt context: %w", err)
}
sectionPayload := promptTranscriptSection{
SectionIndex: sectionIndex,
Segments: make([]promptSegment, 0),
}
if transcript != nil {
for _, s := range transcript.Segments {
sectionPayload.Segments = append(sectionPayload.Segments, promptSegment{
ID: s.ID,
Speaker: s.Speaker,
Start: s.Start,
End: s.End,
Text: s.Text,
Categories: append([]string(nil), s.Categories...),
})
}
}
sectionJSON, err := json.MarshalIndent(sectionPayload, "", " ")
sectionJSON, err := promptcontext.MarshalTranscriptSectionJSON(transcript, sectionIndex)
if err != nil {
return nil, fmt.Errorf("marshal transcript prompt context: %w", err)
}
@@ -65,7 +35,3 @@ func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Gloss
{Role: "user", Content: user},
}, nil
}
func proposalPromptMetadata() prompts.Metadata {
return prompts.MustLookupMetadata(prompts.PromptIDModuleGlossaryProposal)
}

View File

@@ -2,12 +2,11 @@ package grammar
import (
"context"
"fmt"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/prompts"
builtinvalidators "gitea.maximumdirect.net/eric/audita/internal/validators"
)
@@ -36,65 +35,10 @@ func (m *Module) Validators() []contracts.Validator {
return append([]contracts.Validator(nil), m.validators...)
}
func (m *Module) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
sectionTranscript := transcriptForSection(req.WorkingTranscript, req.Section)
sectionIndex := 0
if req.Section != nil {
sectionIndex = req.Section.Index
}
transcriptDescription := ""
if req.Config != nil {
transcriptDescription = req.Config.TranscriptDescription
}
messages, err := BuildProposalMessages(sectionTranscript, req.Glossary, sectionIndex, transcriptDescription)
if err != nil {
return nil, err
}
generated, err := proposal_generation.GenerateCandidates(ctx, proposal_generation.Request{
ModuleKey: req.RunSpec.ModuleKey,
ModuleInstance: req.RunSpec.InstanceName,
ReplacementPolicy: req.RunSpec.ReplacementPolicy,
WorkingTranscript: req.WorkingTranscript,
Section: req.Section,
Glossary: req.Glossary,
Config: req.Config,
Messages: messages,
PromptMetadata: map[string]any{
"prompt_id": proposalPromptMetadata().PromptID,
"prompt_version": proposalPromptMetadata().PromptVersion,
"prompt_source": proposalPromptMetadata().PromptSource,
"embedded_path": proposalPromptMetadata().EmbeddedPath,
"sha256": proposalPromptMetadata().SHA256,
},
StageName: proposalStageName(req),
StartIndex: 0,
LLMClient: req.LLMClient,
Scheduler: req.LLMScheduler,
DiagnosticsDir: req.DiagnosticsDir,
func (m *Module) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
return proposal_generation.ExecuteModuleProposal(ctx, proposal_generation.ModuleProposalRequest{
ProposalRequest: req,
PromptID: prompts.PromptIDModuleGrammarProposal,
BuildMessages: BuildProposalMessages,
})
if err != nil {
return nil, err
}
return generated.Corrections, nil
}
func transcriptForSection(transcript *schema.Transcript, section *contracts.SectionMetadata) *schema.Transcript {
if section == nil || transcript == nil {
return transcript
}
segments := make([]schema.Segment, 0, len(transcript.Segments))
for _, seg := range transcript.Segments {
if seg.ID >= section.StartSegmentID && seg.ID <= section.EndSegmentID {
segments = append(segments, seg)
}
}
return &schema.Transcript{Segments: segments}
}
func proposalStageName(req contracts.ProposalRequest) string {
if req.Section == nil || req.Section.Index == 0 {
return fmt.Sprintf("%s:proposal", req.RunSpec.InstanceName)
}
return fmt.Sprintf("%s:proposal:section-%04d", req.RunSpec.InstanceName, req.Section.Index)
}

View File

@@ -131,6 +131,7 @@ func TestGrammarModuleValidatorChain(t *testing.T) {
got = append(got, v.Name())
}
want := []string{
"proposal_shape",
"no_effect",
"original_text_presence",
"confidence_threshold",
@@ -187,7 +188,7 @@ func TestGrammarModuleProposeMapsCorrectionsAndWritesDiagnostics(t *testing.T) {
if len(client.calls) != 1 || client.calls[0].StageName != "grammar:proposal" {
t.Fatalf("expected one grammar:proposal call, got %+v", client.calls)
}
if len(out) != 1 || out[0].CorrectedText != "Hello, world" {
if len(out.Proposals) != 1 || out.Proposals[0].CorrectedText != "Hello, world" {
t.Fatalf("unexpected proposals: %+v", out)
}
diagFiles, globErr := filepath.Glob(filepath.Join(diagDir, "grammar", "*proposal*response-payload.json"))

Some files were not shown because too many files have changed in this diff Show More