Compare commits
31 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| b1eb37d80d | |||
| 0fc92f3643 | |||
| 0dfd06c349 | |||
| da3720693d | |||
| 6dfc1ea527 | |||
| 761d70bbc6 | |||
| 451cc19418 | |||
| c37ea70dcb | |||
| a90859114a | |||
| 9202ccddb9 | |||
| f40d4add91 | |||
| 16bb12face | |||
| f18e2428dc | |||
| 3b64e784a1 | |||
| 3744d229a2 | |||
| 9bbe1fb7f1 | |||
| b7a66f6cc4 | |||
| c8efdb53d3 | |||
| ab4b252b08 | |||
| e9028e08a4 | |||
| 332884f887 | |||
| e5173c78fe | |||
| 546be2ab92 | |||
| 7743b397a6 | |||
| d23a95471c | |||
| f8ab117bfc | |||
| 88018c9e76 | |||
| b3e7dc3136 | |||
| 385c62a5b4 | |||
| 7d9bf33d18 | |||
| b7cc5fb980 |
2
LICENSE
2
LICENSE
@@ -1,4 +1,4 @@
|
||||
Copyright (c) 2026 eric.
|
||||
Copyright (c) 2026 Eric Rakestraw.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without modification, are permitted provided that the following conditions are met:
|
||||
|
||||
|
||||
540
README.md
540
README.md
@@ -1,513 +1,49 @@
|
||||
# seriatim
|
||||
|
||||
`seriatim` merges per-speaker WhisperX-style JSON transcripts into a single JSON transcript that preserves speaker identity and chronological order. It also trims existing seriatim output artifacts by segment ID and normalizes external transcript-like JSON into standard seriatim output schemas.
|
||||
`seriatim` is a Go CLI for transcript artifact processing.
|
||||
|
||||
The current implementation supports the `merge`, `trim`, and `normalize` commands. `merge` reads one or more input JSON files, optionally maps each input file to a canonical speaker using `speakers.yml`, sorts all segments by timestamp, detects and resolves overlaps when word-level timing is available, assigns consecutive numeric `id` values, and writes a merged JSON artifact. `trim` reads an existing seriatim output artifact and projects it to a retained segment subset. `normalize` reads transcript-like JSON input, validates required segment fields, sorts deterministically, assigns fresh IDs, and emits a selected seriatim output schema.
|
||||
It merges per-speaker WhisperX-style JSON into deterministic seriatim JSON,
|
||||
trims existing seriatim artifacts by segment ID, normalizes transcript-like JSON
|
||||
into supported output schemas, and renders existing seriatim artifacts as
|
||||
human-readable Markdown.
|
||||
|
||||
## Usage
|
||||
## Quickstart
|
||||
|
||||
Run from source:
|
||||
Shortest useful merge command:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim merge \
|
||||
--input-file samples/raw/2026-04-19-Eric_Rakestraw.json \
|
||||
--input-file samples/raw/2026-04-19-Mike_Brown.json \
|
||||
--input-file speaker-a.json \
|
||||
--input-file speaker-b.json \
|
||||
--output-file merged.json
|
||||
```
|
||||
|
||||
Optional report output:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim merge \
|
||||
--input-file eric.json \
|
||||
--input-file mike.json \
|
||||
--output-file merged.json \
|
||||
--report-file report.json
|
||||
```
|
||||
|
||||
Trim an existing seriatim artifact:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim trim \
|
||||
--input-file merged.json \
|
||||
--output-file trimmed.json \
|
||||
--keep "1-10, 15, 20-25"
|
||||
```
|
||||
|
||||
Normalize external transcript-style JSON:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim normalize \
|
||||
--input-file transcript.json \
|
||||
--output-file normalized.json
|
||||
```
|
||||
|
||||
Normalize an Audita-style bare segment array to full schema with report output:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim normalize \
|
||||
--input-file audita-segments.json \
|
||||
--output-file normalized-full.json \
|
||||
--output-schema seriatim-full \
|
||||
--report-file normalize-report.json
|
||||
```
|
||||
|
||||
## CLI
|
||||
|
||||
```text
|
||||
seriatim merge [flags]
|
||||
seriatim trim [flags]
|
||||
seriatim normalize [flags]
|
||||
```
|
||||
|
||||
Global flags:
|
||||
|
||||
| Flag | Description |
|
||||
| --- | --- |
|
||||
| `--help` | Show command help. |
|
||||
| `--version` | Show application version. Local builds default to `dev`; release builds inject the release version. |
|
||||
|
||||
`merge` flags:
|
||||
|
||||
| Flag | Required | Default | Description |
|
||||
| --- | --- | --- | --- |
|
||||
| `--input-file` | Yes | none | Input transcript JSON file. Repeat once per speaker/input file. |
|
||||
| `--output-file` | Yes | none | Merged transcript JSON output path. |
|
||||
| `--report-file` | No | none | Optional report JSON output path. |
|
||||
| `--speakers` | No | none | Speaker map YAML file. When omitted, input file basenames are used as speaker labels. |
|
||||
| `--autocorrect` | No | none | Autocorrect rules YAML file. When omitted, the default `autocorrect` module leaves text unchanged. |
|
||||
| `--input-reader` | No | `json-files` | Input reader module. |
|
||||
| `--output-modules` | No | `json` | Comma-separated output modules. |
|
||||
| `--output-schema` | No | `seriatim-intermediate` | JSON output contract. Allowed values are `seriatim-minimal`, `seriatim-intermediate`, and `seriatim-full`. If omitted, the runtime default is used; consumers that depend on a specific shape should set this explicitly. |
|
||||
| `--preprocessing-modules` | No | `validate-raw,normalize-speakers,trim-text` | Comma-separated preprocessing modules, evaluated in order. |
|
||||
| `--postprocessing-modules` | No | `detect-overlaps,resolve-overlaps,backchannel,filler,resolve-danglers,coalesce,detect-overlaps,autocorrect,assign-ids,validate-output` | Comma-separated postprocessing modules, evaluated in order. |
|
||||
| `--coalesce-gap` | No | `3.0` | Maximum same-speaker gap in seconds for `coalesce`; also used as the `resolve-overlaps` context window. Must be a non-negative float. |
|
||||
|
||||
`trim` flags:
|
||||
|
||||
| Flag | Required | Default | Description |
|
||||
| --- | --- | --- | --- |
|
||||
| `--input-file` | Yes | none | Input seriatim output artifact JSON file. |
|
||||
| `--output-file` | Yes | none | Trimmed transcript JSON output path. |
|
||||
| `--keep` | Exactly one of `--keep` or `--remove` is required | none | Segment ID selector to retain. |
|
||||
| `--remove` | Exactly one of `--keep` or `--remove` is required | none | Segment ID selector to drop. |
|
||||
| `--output-schema` | No | preserve input artifact schema | Optional output schema override: `seriatim-minimal`, `seriatim-intermediate`, or `seriatim-full`. |
|
||||
| `--report-file` | No | none | Optional report JSON output path. |
|
||||
| `--allow-empty` | No | `false` | Allow trimming to zero retained segments. |
|
||||
|
||||
`trim` selection rules:
|
||||
|
||||
- `--keep` and `--remove` are mutually exclusive.
|
||||
- Exactly one of `--keep` or `--remove` is required.
|
||||
- Selection is by segment ID only.
|
||||
- Invalid selected segment IDs fail the command by default.
|
||||
|
||||
`trim` selector syntax:
|
||||
|
||||
- Segment IDs are positive 1-based integers.
|
||||
- Inclusive ranges are supported: `1-10`.
|
||||
- Comma-separated selectors are supported: `1-10,15,20-25`.
|
||||
- Whitespace around numbers, commas, and hyphens is allowed: `1 - 10, 15, 20 - 25`.
|
||||
- Duplicate and overlapping ranges are accepted and normalized as a union.
|
||||
- Descending ranges (for example `10-1`) are rejected.
|
||||
|
||||
`trim` behavior:
|
||||
|
||||
- `trim` consumes existing seriatim JSON output artifacts only.
|
||||
- `trim` does not accept raw WhisperX transcript JSON as input.
|
||||
- Retained output segment IDs are renumbered sequentially from `1` to `N`.
|
||||
- Transcript order is preserved from input transcript order; selector order does not reorder output.
|
||||
- When output schema is `seriatim-full`, overlap groups are recomputed from retained segments.
|
||||
- `--output-schema seriatim-full` is supported when trim has full-schema artifact data to emit; trim does not synthesize missing full-schema provenance from minimal/intermediate input artifacts.
|
||||
- `trim` does not run merge postprocessors such as `resolve-overlaps`, `coalesce`, or `autocorrect`.
|
||||
|
||||
`trim` report output:
|
||||
|
||||
- When `--report-file` is provided, the report includes standard trim/validation/output events.
|
||||
- The report includes a `trim-audit` event containing trim operation metadata, including selected IDs, retained/removed counts, removed IDs, and old-to-new segment ID mapping.
|
||||
- Old-to-new ID mapping is emitted as a deterministic ordered array of `{old_id, new_id}` pairs.
|
||||
|
||||
`normalize` flags:
|
||||
|
||||
| Flag | Required | Default | Description |
|
||||
| --- | --- | --- | --- |
|
||||
| `--input-file` | Yes | none | Input transcript JSON file. |
|
||||
| `--output-file` | Yes | none | Normalized transcript JSON output path. |
|
||||
| `--output-schema` | No | `seriatim-intermediate` (resolved via `SERIATIM_OUTPUT_SCHEMA` when set) | Output JSON schema: `seriatim-minimal`, `seriatim-intermediate`, or `seriatim-full`. |
|
||||
| `--output-modules` | No | `json` | Comma-separated output modules. Current normalize support is `json` only. |
|
||||
| `--report-file` | No | none | Optional report JSON output path. |
|
||||
|
||||
`normalize` input shapes:
|
||||
|
||||
- Top-level object with a `segments` array.
|
||||
- Bare top-level array of segment objects (for example, Audita-style output).
|
||||
|
||||
`normalize` behavior:
|
||||
|
||||
- Repairs missing timing fields deterministically:
|
||||
if one of `start`/`end` is present, sets both to that value;
|
||||
if both are missing, uses midpoint of previous `end` and next `start`,
|
||||
with edge fallback to available neighbor and `0.0` for single-segment inputs.
|
||||
- If `end < start`, swaps them.
|
||||
- Fills missing/empty `speaker` with `Unknown_Speaker`.
|
||||
- Drops segments with missing, empty, or whitespace-only `text`.
|
||||
- Validates repaired timing with `start >= 0`.
|
||||
- Accepts existing input `id` values as provenance only.
|
||||
- Reassigns output segment IDs sequentially from `1` to `N`.
|
||||
- Sorts deterministically by `(start, end, original_input_index, speaker)`.
|
||||
- Uses original input order only as a tie-breaker.
|
||||
- Does not run merge postprocessors such as overlap detection, overlap resolution, coalescing, or autocorrect.
|
||||
- Useful for converting external transcript outputs into standard seriatim artifacts.
|
||||
|
||||
`normalize` report output:
|
||||
|
||||
- When `--report-file` is provided, normalize emits deterministic report events with input shape detection, segment counts, schema/module selections, sorting/ID diagnostics, and output write/validation summaries.
|
||||
- A machine-readable `normalize-audit` event is included for downstream tooling.
|
||||
|
||||
Environment variables:
|
||||
|
||||
| Environment Variable | Default | Description |
|
||||
| --- | --- | --- |
|
||||
| `SERIATIM_OUTPUT_SCHEMA` | `seriatim-intermediate` | Output schema used when `--output-schema` is not explicitly provided. Allowed values are `seriatim-minimal`, `seriatim-intermediate`, and `seriatim-full`. The CLI flag takes precedence. |
|
||||
| `SERIATIM_OVERLAP_WORD_RUN_GAP` | `1.0` | Maximum gap in seconds between adjacent timed words when `resolve-overlaps` builds word-run replacement segments. Must be a positive float. |
|
||||
| `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW` | `1.0` | Near-start window in seconds for ordering replacement word runs shortest-first. Must be a positive float. |
|
||||
| `SERIATIM_BACKCHANNEL_MAX_DURATION` | `2.0` | Maximum duration in seconds for `backchannel` classification. Must be a positive float. |
|
||||
| `SERIATIM_FILLER_MAX_DURATION` | `1.25` | Maximum duration in seconds for `filler` classification. Must be a positive float. |
|
||||
|
||||
## Input JSON Format
|
||||
|
||||
Each input file must be valid JSON with a top-level `segments` array. The current parser accepts the WhisperX segment subset needed for merging:
|
||||
|
||||
```json
|
||||
{
|
||||
"segments": [
|
||||
{
|
||||
"start": 1.25,
|
||||
"end": 3.5,
|
||||
"text": "Hello there.",
|
||||
"words": [
|
||||
{"word": "Hello", "start": 1.25, "end": 1.55, "score": 0.98},
|
||||
{"word": "there.", "start": 1.7, "end": 2.0}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
Required segment fields:
|
||||
|
||||
- `start`: number, must be `>= 0`.
|
||||
- `end`: number, must be `>= start`.
|
||||
- `text`: string.
|
||||
|
||||
Optional word fields:
|
||||
|
||||
- `words`: array of word timing objects.
|
||||
- `words[].word`: string.
|
||||
- `words[].start`: optional number, must be `>= 0` when present.
|
||||
- `words[].end`: optional number, must be `>= start` when present with `start`.
|
||||
- `words[].score`: optional number.
|
||||
- `words[].speaker`: optional raw speaker label string.
|
||||
|
||||
Word-level timing is preserved internally for overlap resolution. If a word is missing `start` or `end`, seriatim keeps the word text, emits a warning in the optional report, and does not use that word as a timing anchor. Word timing is not emitted in the final JSON artifact.
|
||||
|
||||
## Speaker Map Format
|
||||
|
||||
`speakers.yml` maps input files to canonical speaker names using ordered substring rules:
|
||||
|
||||
This file is optional. If `--speakers` is omitted, `seriatim` uses each input file basename as the segment speaker label.
|
||||
|
||||
```yaml
|
||||
match:
|
||||
- speaker: "Eric Rakestraw"
|
||||
match:
|
||||
- "Eric_Rakestraw"
|
||||
- "Eric"
|
||||
|
||||
- speaker: "Mike Brown"
|
||||
match:
|
||||
- "Mike_Brown"
|
||||
- "mb"
|
||||
```
|
||||
|
||||
For each `--input-file`, `seriatim` takes the file basename and evaluates the rules in order. The first rule with a matching substring wins, and no later rules are evaluated.
|
||||
|
||||
For example, this input:
|
||||
|
||||
```text
|
||||
samples/raw/2026-04-19-Eric_Rakestraw.json
|
||||
```
|
||||
|
||||
matches this rule because the basename contains `Eric_Rakestraw`:
|
||||
|
||||
```yaml
|
||||
- speaker: "Eric Rakestraw"
|
||||
match:
|
||||
- "Eric_Rakestraw"
|
||||
```
|
||||
|
||||
Important details:
|
||||
|
||||
- Matching is against the input file basename, not the full path.
|
||||
- Matching is case-insensitive.
|
||||
- Rules are evaluated from first to last.
|
||||
- Each rule must have a non-empty `speaker`.
|
||||
- Each rule must have at least one non-empty `match` string.
|
||||
- Duplicate speaker names are invalid.
|
||||
- Every input file must match at least one rule or the command fails.
|
||||
|
||||
Deprecated old format:
|
||||
|
||||
```yaml
|
||||
inputs:
|
||||
eric.json:
|
||||
speaker: "Eric Rakestraw"
|
||||
```
|
||||
|
||||
The old `inputs:` direct mapping format is no longer supported.
|
||||
|
||||
## Output JSON Format
|
||||
|
||||
`--output-modules json` controls the writer. `--output-schema` controls the JSON contract that writer serializes.
|
||||
|
||||
The named schemas are stable public contracts. If a consumer depends on a specific shape, it should request that schema explicitly at runtime. The runtime default selection may change in a future release.
|
||||
|
||||
The `seriatim-intermediate` schema is the current default selection when neither `--output-schema` nor `SERIATIM_OUTPUT_SCHEMA` is set. It stays close to the minimal schema, but adds optional `categories` on each segment:
|
||||
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"application": "seriatim",
|
||||
"version": "dev",
|
||||
"output_schema": "seriatim-intermediate"
|
||||
},
|
||||
"segments": [
|
||||
{
|
||||
"id": 1,
|
||||
"start": 1.25,
|
||||
"end": 3.5,
|
||||
"speaker": "Eric Rakestraw",
|
||||
"text": "Hello there.",
|
||||
"categories": ["backchannel"]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
The `seriatim-full` schema uses the full seriatim envelope:
|
||||
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"application": "seriatim",
|
||||
"version": "dev",
|
||||
"input_reader": "json-files",
|
||||
"input_files": ["eric.json", "mike.json"],
|
||||
"preprocessing_modules": ["validate-raw", "normalize-speakers", "trim-text"],
|
||||
"postprocessing_modules": ["detect-overlaps", "resolve-overlaps", "backchannel", "filler", "resolve-danglers", "coalesce", "detect-overlaps", "autocorrect", "assign-ids", "validate-output"],
|
||||
"output_modules": ["json"]
|
||||
},
|
||||
"segments": [
|
||||
{
|
||||
"id": 1,
|
||||
"source": "eric.json",
|
||||
"source_segment_index": 0,
|
||||
"speaker": "Eric Rakestraw",
|
||||
"start": 1.25,
|
||||
"end": 3.5,
|
||||
"text": "Hello there.",
|
||||
"overlap_group_id": 1
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"source": "eric.json",
|
||||
"source_ref": "word-run:1:1:1",
|
||||
"derived_from": ["eric.json#0"],
|
||||
"speaker": "Eric Rakestraw",
|
||||
"start": 2.0,
|
||||
"end": 2.5,
|
||||
"text": "Resolved word run",
|
||||
"categories": ["backchannel"]
|
||||
}
|
||||
],
|
||||
"overlap_groups": [
|
||||
{
|
||||
"id": 1,
|
||||
"start": 1.25,
|
||||
"end": 4.0,
|
||||
"segments": ["eric.json#0", "mike.json#0"],
|
||||
"speakers": ["Eric Rakestraw", "Mike Brown"],
|
||||
"class": "unknown",
|
||||
"resolution": "unresolved"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
The `seriatim-minimal` schema emits minimal metadata and compact ordered segments:
|
||||
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"application": "seriatim",
|
||||
"version": "dev",
|
||||
"output_schema": "seriatim-minimal"
|
||||
},
|
||||
"segments": [
|
||||
{
|
||||
"id": 1,
|
||||
"start": 1.25,
|
||||
"end": 3.5,
|
||||
"speaker": "Eric Rakestraw",
|
||||
"text": "Hello there."
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
Minimal output intentionally omits categories, overlap groups, source/provenance fields, and pipeline configuration metadata.
|
||||
|
||||
Intermediate output intentionally omits overlap groups and source/provenance fields, but keeps optional `categories` and minimal metadata.
|
||||
|
||||
Segments are sorted deterministically by:
|
||||
|
||||
```text
|
||||
(start, end, source, source_segment_index/source_ref, speaker)
|
||||
```
|
||||
|
||||
Final segment IDs are assigned after sorting and start at `1`.
|
||||
|
||||
The public Go output contract is available from:
|
||||
|
||||
```go
|
||||
import "gitea.maximumdirect.net/eric/seriatim/schema"
|
||||
```
|
||||
|
||||
The same package embeds machine-readable JSON Schemas in `schema/full-output.schema.json`, `schema/intermediate-output.schema.json`, and `schema/minimal-output.schema.json`. The default `validate-output` postprocessor validates the selected output shape and verifies final segment IDs are present, sequential, and start at `1`.
|
||||
|
||||
## Overlap Detection
|
||||
|
||||
The default postprocessing pipeline detects overlapping segment groups.
|
||||
|
||||
Overlap behavior:
|
||||
|
||||
- A strict timing overlap is required: `next.start < current_group_end`.
|
||||
- Segments that only touch at a boundary are not grouped.
|
||||
- Groups require at least two distinct speakers.
|
||||
- Transitive overlaps are grouped together.
|
||||
- Segments in detected groups receive `overlap_group_id`.
|
||||
- `overlap_groups[].segments` contains stable references in `source#source_segment_index` format.
|
||||
- `class` is currently `unknown`.
|
||||
- `resolution` is `unresolved` until `resolve-overlaps` replaces the group.
|
||||
|
||||
## Overlap Resolution
|
||||
|
||||
The default postprocessing pipeline runs `detect-overlaps`, then `resolve-overlaps`, then `backchannel`, then `filler`, then `resolve-danglers`, then `coalesce`, then a second `detect-overlaps` pass.
|
||||
|
||||
For each detected overlap group, `resolve-overlaps` uses preserved WhisperX word timing to build smaller word-run replacement segments:
|
||||
|
||||
- The resolution window expands the detected overlap group by `--coalesce-gap` seconds on both sides.
|
||||
- Nearby same-speaker context segments are included when they intersect the expanded window and their start or end is within `--coalesce-gap` of the original overlap boundary.
|
||||
- Once a segment is selected for replacement, all timed words from that segment participate in word-run construction; the window controls segment selection, not per-word clipping.
|
||||
- Context segments that are part of another detected overlap group are not pulled into the current group.
|
||||
- Untimed words are included in replacement text in original word order when nearby timed words create a replacement run.
|
||||
- Untimed words do not affect replacement segment start/end times or word-run gap splitting.
|
||||
- Words for the same speaker are merged into one run when the gap between adjacent words is no greater than `SERIATIM_OVERLAP_WORD_RUN_GAP`.
|
||||
- The default word-run gap is `1.0` seconds.
|
||||
- Set `SERIATIM_OVERLAP_WORD_RUN_GAP` to a positive number of seconds to override the default.
|
||||
- Near-start replacement word runs are reordered so shorter segments come first when adjacent starts are within `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW`.
|
||||
- The default word-run reorder window is `1.0` seconds.
|
||||
- Set `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW` to a positive number of seconds to override the default.
|
||||
- Replacement segment text is built by joining word text with single spaces.
|
||||
- Replacement segments include `source_ref` and `derived_from`.
|
||||
- Replacement segments omit `source_segment_index` because they are derived from one or more original segments.
|
||||
- Resolved overlap groups are removed before the second detection pass.
|
||||
- Replacement segments are left without `overlap_group_id` until the second detection pass annotates any remaining overlap.
|
||||
- If a speaker has no usable word timing in a group, that speaker's original segment is kept.
|
||||
- If no speakers in a group have usable word timing, the original group and annotations remain unchanged.
|
||||
|
||||
## Backchannels
|
||||
|
||||
The default pipeline runs `backchannel` before `coalesce`. It tags short acknowledgement segments with:
|
||||
|
||||
```json
|
||||
"categories": ["backchannel"]
|
||||
```
|
||||
|
||||
Backchannel matching is case-insensitive, ignores punctuation for matching and word-count purposes, trims surrounding whitespace, and requires a matching acknowledgement phrase, no more than three whitespace-delimited words, and duration no greater than `SERIATIM_BACKCHANNEL_MAX_DURATION` seconds. The default maximum duration is `2.0` seconds.
|
||||
|
||||
## Fillers
|
||||
|
||||
The default pipeline runs `filler` after `backchannel` and before `coalesce`. It tags short filler utterances with:
|
||||
|
||||
```json
|
||||
"categories": ["filler"]
|
||||
```
|
||||
|
||||
Filler matching is case-insensitive, ignores punctuation for matching and word-count purposes, trims surrounding whitespace, and requires only filler tokens such as `um`, `uh`, `er`, `erm`, `ah`, `eh`, `hmm`, `mm`, or repeated combinations of those tokens. Matching segments must contain no more than three whitespace-delimited words and have duration no greater than `SERIATIM_FILLER_MAX_DURATION` seconds. The default maximum duration is `1.25` seconds.
|
||||
|
||||
## Dangler Resolution
|
||||
|
||||
The default pipeline runs `resolve-danglers` before `coalesce` and before the second overlap detection pass. It repairs short derived fragments when they share provenance with a nearby segment:
|
||||
|
||||
- Dangling-end fragments have no more than two words and end in punctuation.
|
||||
- Dangling-start fragments have no more than two words.
|
||||
- Matching uses same-speaker segments with any shared `derived_from` value.
|
||||
- Merged segments use `source_ref` values such as `resolve-danglers:1`, keep the target segment's transcript position, and union `derived_from`.
|
||||
|
||||
## Coalescing
|
||||
|
||||
The default pipeline runs `coalesce` after `resolve-danglers` and before the second overlap detection pass. It merges adjacent same-speaker segments in the transcript's current order when `next.start - current.end <= --coalesce-gap`.
|
||||
|
||||
Coalesced segments use `source_ref` values such as `coalesce:1`, include `derived_from`, and omit `source_segment_index`.
|
||||
|
||||
Different-speaker backchannel and filler segments do not block coalescing of surrounding same-speaker segments. Same-speaker backchannel and filler segments are merged normally when they are within `--coalesce-gap`. When same-speaker segments are coalesced, any `backchannel` or `filler` category from the merged inputs is dropped from the coalesced segment.
|
||||
|
||||
## Autocorrect
|
||||
|
||||
Autocorrect is included in the default postprocessing pipeline. If `--autocorrect` is omitted, the module leaves transcript text unchanged and records a skip event in the optional report.
|
||||
|
||||
Enable corrections by passing `--autocorrect`:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim merge \
|
||||
--input-file input.json \
|
||||
--autocorrect autocorrect.yml \
|
||||
--output-file merged.json
|
||||
```
|
||||
|
||||
`autocorrect.yml` format:
|
||||
|
||||
```yaml
|
||||
autocorrect:
|
||||
- target: "Hrank"
|
||||
match:
|
||||
- "hrank"
|
||||
- "Frank"
|
||||
|
||||
- target: "Mike Brown"
|
||||
match:
|
||||
- "Mike Pat"
|
||||
```
|
||||
|
||||
Matching behavior:
|
||||
|
||||
- Matching is case-sensitive.
|
||||
- Matches apply only to whole tokens, not substrings inside larger words.
|
||||
- Punctuation and whitespace can surround a match.
|
||||
- Multi-word and hyphenated matches are supported.
|
||||
- Duplicate match strings are invalid, including duplicates across separate rules.
|
||||
|
||||
## Current Limitations
|
||||
|
||||
- Only JSON input is supported.
|
||||
- Overlap resolution depends on WhisperX word timing; groups without usable word timing remain unresolved.
|
||||
- Alternate output formats are not implemented yet.
|
||||
|
||||
## Release Builds
|
||||
|
||||
Local builds record version metadata as `dev`. Release builds should inject the release version with `ldflags`:
|
||||
|
||||
```sh
|
||||
go build -ldflags "-X gitea.maximumdirect.net/eric/seriatim/internal/buildinfo.Version=v1.0.0" ./cmd/seriatim
|
||||
```
|
||||
## Commands
|
||||
|
||||
- `merge`: merge one or more input transcript JSON files.
|
||||
- `trim`: keep/remove segment IDs from an existing seriatim artifact.
|
||||
- `normalize`: canonicalize transcript-like JSON into a seriatim artifact.
|
||||
- `render`: render an existing seriatim artifact as Markdown.
|
||||
|
||||
## Documentation
|
||||
|
||||
- CLI reference: [docs/cli.md](docs/cli.md)
|
||||
- Configuration reference: [docs/config.md](docs/config.md)
|
||||
- Operations guide: [docs/operations.md](docs/operations.md)
|
||||
- Troubleshooting: [docs/troubleshooting.md](docs/troubleshooting.md)
|
||||
- Integration references:
|
||||
- [docs/integrations/whisperx-json.md](docs/integrations/whisperx-json.md)
|
||||
- [docs/integrations/output-schemas.md](docs/integrations/output-schemas.md)
|
||||
- Development policies:
|
||||
- [docs/policy/architecture.md](docs/policy/architecture.md)
|
||||
- [docs/policy/development.md](docs/policy/development.md)
|
||||
- [docs/policy/documentation.md](docs/policy/documentation.md)
|
||||
- Internal implementation references:
|
||||
- [docs/internal/pipeline.md](docs/internal/pipeline.md)
|
||||
- [docs/internal/artifacts.md](docs/internal/artifacts.md)
|
||||
- [docs/internal/modules.md](docs/internal/modules.md)
|
||||
- Public JSON schema files:
|
||||
- [schema/minimal-output.schema.json](schema/minimal-output.schema.json)
|
||||
- [schema/intermediate-output.schema.json](schema/intermediate-output.schema.json)
|
||||
- [schema/full-output.schema.json](schema/full-output.schema.json)
|
||||
- Synthetic examples: [examples/README.md](examples/README.md)
|
||||
|
||||
524
architecture.md
524
architecture.md
@@ -1,524 +0,0 @@
|
||||
# seriatim Architecture
|
||||
|
||||
`seriatim` is a deterministic transcript utility for:
|
||||
|
||||
- merging multiple per-speaker transcript inputs into a single chronologically ordered diarized transcript, and
|
||||
- projecting existing seriatim transcript artifacts through deterministic segment-ID trimming, and
|
||||
- canonicalizing external transcript-style JSON inputs into standard seriatim output schemas.
|
||||
|
||||
The initial use case is merging independently transcribed speaker audio tracks from the same recorded session, such as a weekly tabletop RPG session. The architecture should also support meetings, podcasts, interviews, and other multi-speaker events.
|
||||
|
||||
`seriatim` is implemented in Go.
|
||||
|
||||
## Goals
|
||||
|
||||
`seriatim` should:
|
||||
|
||||
1. Validate runtime configuration before performing transcript processing.
|
||||
2. Support multiple input methods and formats through input readers.
|
||||
3. Normalize raw per-speaker transcripts into a canonical internal model.
|
||||
4. Apply deterministic preprocessing modules to canonical per-speaker transcripts.
|
||||
5. Merge all segments into a deterministic global chronological order.
|
||||
6. Apply deterministic postprocessing modules to the merged transcript.
|
||||
7. Preserve word-level timing data when available.
|
||||
8. Detect and annotate overlapping speech regions.
|
||||
9. Emit one or more output artifacts through output writers.
|
||||
10. Produce report data for validation findings, corrections, and transformations.
|
||||
11. Support artifact-level transcript projection commands that operate on existing seriatim output.
|
||||
|
||||
## Non-goals
|
||||
|
||||
The 1.0 release does not attempt to:
|
||||
|
||||
- Perform transcription.
|
||||
- Perform audio diarization.
|
||||
- Use an LLM.
|
||||
- Summarize transcript content.
|
||||
- Infer speaker identity from audio or text.
|
||||
- Fully resolve every crosstalk case.
|
||||
- Load arbitrary third-party code as dynamic plugins.
|
||||
|
||||
The application supports runtime composition of built-in modules by canonical module name. Arbitrary external plugin loading can be considered later.
|
||||
|
||||
## Core Assumption
|
||||
|
||||
The merge algorithm assumes that all input transcript timestamps are measured against the same session clock.
|
||||
|
||||
This is expected when each speaker has a separate recording that preserves silence and starts at the same session recording time. If input files have independent local timelines, `seriatim` cannot safely merge them without a separate alignment step.
|
||||
|
||||
## Pipeline Overview
|
||||
|
||||
The internal pipeline is:
|
||||
|
||||
```text
|
||||
configuration check
|
||||
-> input
|
||||
-> preprocessing
|
||||
-> merge
|
||||
-> postprocessing
|
||||
-> output
|
||||
```
|
||||
|
||||
Each stage has an explicit data contract. Input and output stages perform I/O. Processing stages should be deterministic transformations over in-memory models and should record report events for validation findings, corrections, and transformations.
|
||||
|
||||
`merge` runs this pipeline. `trim` and `normalize` are intentionally separate from this pipeline and operate at the artifact layer.
|
||||
|
||||
## Stage Contracts
|
||||
|
||||
### 1. Configuration Check
|
||||
|
||||
The configuration stage validates all CLI flags, environment variables, module names, input paths, output paths, and module-specific options before transcript data is processed.
|
||||
|
||||
Configuration validation should fail fast for:
|
||||
|
||||
- Missing required input.
|
||||
- Unknown module names.
|
||||
- Unknown input or output formats.
|
||||
- Ambiguous speaker mappings.
|
||||
- Invalid correction policies.
|
||||
- Invalid timing thresholds.
|
||||
- Invalid output paths.
|
||||
|
||||
The configuration stage produces an application config value that is passed through the pipeline.
|
||||
|
||||
### 2. Input Stage
|
||||
|
||||
The input stage converts external inputs into raw transcript documents with source metadata.
|
||||
|
||||
The current input method is one or more JSON files passed with repeated `--input-file` flags:
|
||||
|
||||
```text
|
||||
seriatim merge --input-file eric.json --input-file mike.json --output-file merged.json
|
||||
```
|
||||
|
||||
Future input methods may include:
|
||||
|
||||
- A `.tar.gz` bundle.
|
||||
- A URI.
|
||||
- A directory.
|
||||
|
||||
Future input formats may include:
|
||||
|
||||
- JSON.
|
||||
- SRT.
|
||||
- VTT.
|
||||
|
||||
Input readers should be selected from an explicit registry. A reader is responsible for loading external data and returning raw transcript documents, not for canonical normalization.
|
||||
|
||||
### 3. Preprocessing Stage
|
||||
|
||||
The preprocessing stage applies zero or more modules before global merge.
|
||||
|
||||
Preprocessing starts with raw transcript documents from input readers and must end with canonical per-speaker transcripts. Some preprocessing modules operate on raw transcripts, some perform raw-to-canonical normalization, and some operate only on canonical transcripts.
|
||||
|
||||
Preprocessing modules are selected at runtime with a comma-separated list of canonical module names:
|
||||
|
||||
```text
|
||||
--preprocessing-modules validate-raw,normalize-speakers,trim-text
|
||||
```
|
||||
|
||||
Modules run in the exact order provided. Unknown module names are configuration errors.
|
||||
|
||||
Potential preprocessing modules include:
|
||||
|
||||
- Structural raw transcript validation.
|
||||
- Semantic transcript validation.
|
||||
- Raw-to-canonical transcript normalization.
|
||||
- Speaker name normalization based on input filename.
|
||||
- Timing validation and deterministic correction.
|
||||
- Text trimming.
|
||||
|
||||
Preprocessing should not depend on global chronological ordering across speakers. Modules that need the globally merged transcript belong in postprocessing.
|
||||
|
||||
Each preprocessing module must declare the model state it requires and the model state it produces. For example, `validate-raw` requires raw transcripts and produces raw transcripts, while `normalize-speakers` requires raw transcripts and produces canonical transcripts. Configuration validation should reject module orders that cannot type-check.
|
||||
|
||||
### 4. Merge Stage
|
||||
|
||||
The merge stage extracts all canonical segments from the preprocessed per-speaker transcripts and sorts them into a single deterministic chronological sequence.
|
||||
|
||||
The recommended sort key is:
|
||||
|
||||
```text
|
||||
(start, end, source, source_segment_index, speaker)
|
||||
```
|
||||
|
||||
The exact tie-breaker must be documented and stable across runs.
|
||||
|
||||
The merge stage should assign temporary internal references if needed, but it should not assign final output IDs until after all order-affecting postprocessing is complete.
|
||||
|
||||
### 5. Postprocessing Stage
|
||||
|
||||
The postprocessing stage applies zero or more modules to the merged transcript.
|
||||
|
||||
Postprocessing modules are selected at runtime with a comma-separated list of canonical module names:
|
||||
|
||||
```text
|
||||
--postprocessing-modules detect-overlaps,resolve-overlaps,backchannel,filler,resolve-danglers,coalesce,detect-overlaps,autocorrect,assign-ids,validate-output
|
||||
```
|
||||
|
||||
Modules run in the exact order provided. Unknown module names are configuration errors.
|
||||
|
||||
Potential postprocessing modules include:
|
||||
|
||||
- Overlap group detection.
|
||||
- Overlap group refinement.
|
||||
- Same-speaker segment coalescing.
|
||||
- Deterministic grammar cleanup.
|
||||
- Word replacement from `autocorrect.yml`.
|
||||
- Final segment ID assignment.
|
||||
- Output model validation.
|
||||
|
||||
Any module that can reorder, split, merge, drop, or create segments must run before final ID assignment.
|
||||
|
||||
### 6. Output Stage
|
||||
|
||||
The output stage emits one or more artifacts from the final transcript and report model.
|
||||
|
||||
The current output format is JSON, specified with:
|
||||
|
||||
```text
|
||||
--output-file merged.json
|
||||
```
|
||||
|
||||
The current named JSON schemas are:
|
||||
|
||||
- `seriatim-minimal`
|
||||
- `seriatim-intermediate`
|
||||
- `seriatim-full`
|
||||
|
||||
The current runtime default selection is `seriatim-intermediate`, but default selection may change over time. Consumers that depend on a specific schema should request it explicitly.
|
||||
|
||||
Future output formats may include:
|
||||
|
||||
- Markdown.
|
||||
- SRT.
|
||||
- VTT.
|
||||
- Validation reports.
|
||||
- Overlap reports.
|
||||
|
||||
Output writers should be selected from an explicit registry and should consume the final transcript model read-only. Multiple output writers may run for a single invocation.
|
||||
|
||||
### 7. Artifact Projection Stage (`trim` command)
|
||||
|
||||
`trim` is an artifact-level command that reads an existing seriatim output artifact and emits a projected artifact containing a segment-ID subset.
|
||||
|
||||
Design constraints:
|
||||
|
||||
- `trim` runs after `merge`, not as a merge postprocessor.
|
||||
- `trim` validates the input artifact against supported seriatim output schemas.
|
||||
- `trim` performs deterministic keep/remove selection by segment ID.
|
||||
- `trim` renumbers retained IDs to `1..N` in transcript order.
|
||||
- `trim` validates the final output against the selected output schema before writing.
|
||||
- `trim` records audit metadata in report output.
|
||||
|
||||
`trim` is intentionally separate from merge postprocessing because it consumes already-emitted public artifacts. This separation keeps merge semantics stable and avoids rerunning merge-only transforms on projected artifacts.
|
||||
|
||||
`trim` must not rerun merge postprocessors such as `resolve-overlaps`, `coalesce`, or `autocorrect`.
|
||||
|
||||
### 8. Artifact Canonicalization Stage (`normalize` command)
|
||||
|
||||
`normalize` is an artifact-level command that reads transcript-like JSON and emits a standard seriatim output artifact in a selected schema.
|
||||
|
||||
Design constraints:
|
||||
|
||||
- `normalize` runs outside the merge pipeline and does not invoke merge preprocessing or postprocessing modules.
|
||||
- `normalize` accepts two input shapes: object-with-`segments` and bare segment arrays.
|
||||
- `normalize` applies deterministic repair rules for missing/irregular `start`, `end`, and `speaker`, and drops segments with missing/empty `text`.
|
||||
- `normalize` sorts segments deterministically by chronological keys and stable input-index tie-breakers.
|
||||
- `normalize` assigns fresh sequential output IDs (`1..N`) after sorting.
|
||||
- `normalize` validates final output against the selected schema before writing.
|
||||
- `normalize` writes optional deterministic report diagnostics when `--report-file` is requested.
|
||||
|
||||
`normalize` is intended for canonicalizing external transcript outputs (including Audita-style bare arrays) into seriatim contracts, not for running merge-time language or overlap transformations.
|
||||
|
||||
`normalize` must not run merge postprocessors such as overlap detection, overlap resolution, coalescing, or autocorrect.
|
||||
|
||||
## Module Classification
|
||||
|
||||
Modules should be classified by their contract and allowed effects.
|
||||
|
||||
| Class | Input | Output | Allowed effects |
|
||||
| --- | --- | --- | --- |
|
||||
| `InputReader` | External source spec | Raw transcript documents | Reads external data |
|
||||
| `Validator` | Raw, canonical, merged, or final model | Same model plus report events | Observes only |
|
||||
| `Normalizer` | Raw model | Canonical model | Converts representation |
|
||||
| `Corrector` | Canonical model | Canonical model plus report events | Deterministic mutation |
|
||||
| `Annotator` | Canonical or merged model | Same model plus annotations | Adds metadata |
|
||||
| `Transformer` | Canonical or merged model | Updated model plus report events | May reorder, split, merge, drop, or create segments |
|
||||
| `OutputWriter` | Final transcript and report | External artifact | Writes output |
|
||||
|
||||
This classification should guide Go interfaces and package boundaries. It should also determine where a module is allowed to run.
|
||||
|
||||
## Runtime Module Composition
|
||||
|
||||
The application supports runtime composition of built-in modules.
|
||||
|
||||
Module names are canonical strings registered at startup. CLI flags refer to those names. The configuration stage resolves names into module instances before the pipeline runs.
|
||||
|
||||
Example:
|
||||
|
||||
```text
|
||||
seriatim merge \
|
||||
--input-file eric.json \
|
||||
--input-file mike.json \
|
||||
--speakers speakers.yml \
|
||||
--autocorrect autocorrect.yml \
|
||||
--preprocessing-modules validate-raw,normalize-speakers,trim-text \
|
||||
--postprocessing-modules detect-overlaps,resolve-overlaps,backchannel,filler,resolve-danglers,coalesce,detect-overlaps,autocorrect,assign-ids,validate-output \
|
||||
--output-modules json \
|
||||
--output-schema seriatim-intermediate \
|
||||
--output-file merged.json \
|
||||
--report-file report.json
|
||||
```
|
||||
|
||||
Composition rules:
|
||||
|
||||
- Module order is exactly the order specified by the user.
|
||||
- An empty module list is valid when the stage supports zero modules.
|
||||
- Unknown module names are fatal configuration errors.
|
||||
- Module-specific options are read from the validated application config.
|
||||
- A module must declare which pipeline stage and model type it supports.
|
||||
- Modules should be deterministic for the same inputs, config, and application version.
|
||||
- Modules should not perform I/O unless their class explicitly allows it.
|
||||
|
||||
Some modules may be recommended defaults. Defaults should be explicit in documentation and should be equivalent to passing the corresponding module list.
|
||||
|
||||
## Go Interface Sketch
|
||||
|
||||
The exact implementation may evolve, but the core interfaces should resemble:
|
||||
|
||||
```go
|
||||
type InputReader interface {
|
||||
Name() string
|
||||
Read(ctx context.Context, spec InputSpec, cfg Config) ([]RawTranscript, []ReportEvent, error)
|
||||
}
|
||||
|
||||
type Preprocessor interface {
|
||||
Name() string
|
||||
Requires() ModelState
|
||||
Produces() ModelState
|
||||
Process(ctx context.Context, in PreprocessState, cfg Config) (PreprocessState, []ReportEvent, error)
|
||||
}
|
||||
|
||||
type Merger interface {
|
||||
Merge(ctx context.Context, in []CanonicalTranscript, cfg Config) (MergedTranscript, []ReportEvent, error)
|
||||
}
|
||||
|
||||
type Postprocessor interface {
|
||||
Name() string
|
||||
Process(ctx context.Context, in MergedTranscript, cfg Config) (MergedTranscript, []ReportEvent, error)
|
||||
}
|
||||
|
||||
type OutputWriter interface {
|
||||
Name() string
|
||||
Write(ctx context.Context, out any, report Report, cfg Config) ([]ReportEvent, error)
|
||||
}
|
||||
```
|
||||
|
||||
`PreprocessState` should carry either raw transcripts, canonical transcripts, or both during migration between representations. The pipeline should validate that the ordered preprocessing list transitions from raw input state to canonical output state exactly once before merge.
|
||||
|
||||
The interfaces should favor value returns over hidden mutation. If pointer-based implementations are chosen for performance, mutation boundaries must still be clear and tested.
|
||||
|
||||
## Canonical Internal Model
|
||||
|
||||
The canonical model should be richer than the final output schema.
|
||||
|
||||
Canonical segment fields should include:
|
||||
|
||||
- Temporary internal reference.
|
||||
- Source identifier.
|
||||
- Source segment index.
|
||||
- Canonical speaker.
|
||||
- Start time.
|
||||
- End time.
|
||||
- Text.
|
||||
- Word-level timing data, if available.
|
||||
- Raw diarization labels, if useful for reporting.
|
||||
- Validation and correction metadata, if needed internally.
|
||||
|
||||
The final output model can omit internal-only fields, but the report should retain enough provenance to diagnose corrections and transformations.
|
||||
|
||||
## Validation Strategy
|
||||
|
||||
Validation occurs at multiple boundaries:
|
||||
|
||||
- Configuration validation before processing.
|
||||
- Raw input structural validation after input loading.
|
||||
- Raw input semantic validation before normalization or correction.
|
||||
- Canonical model validation after normalization and preprocessing.
|
||||
- Merged model validation after merge and postprocessing.
|
||||
- Final output schema validation before writing artifacts.
|
||||
|
||||
Structural validation answers whether data has the required shape and types.
|
||||
|
||||
Semantic validation answers whether the data is plausible and internally consistent.
|
||||
|
||||
Correctable issues should be deterministic and reportable. Fatal issues should stop the run with a non-zero exit code.
|
||||
|
||||
Examples of correctable issues:
|
||||
|
||||
- Leading or trailing whitespace.
|
||||
- Segment `end < start`, when configured correction policy allows deterministic repair.
|
||||
- Missing word speaker labels when canonical speaker is known.
|
||||
- Raw diarization labels that should be replaced with the canonical speaker.
|
||||
|
||||
Examples of fatal issues:
|
||||
|
||||
- Input file is not valid JSON.
|
||||
- Required transcript fields are missing.
|
||||
- Speaker map does not identify a canonical speaker for an input.
|
||||
- Unknown module name.
|
||||
- Output fails final schema validation.
|
||||
|
||||
## Overlap Handling
|
||||
|
||||
Overlap detection should create overlap groups rather than only pairwise annotations.
|
||||
|
||||
Two adjacent sorted segments overlap when:
|
||||
|
||||
```text
|
||||
next.start < current_group_end
|
||||
```
|
||||
|
||||
This supports transitive overlap groups:
|
||||
|
||||
```text
|
||||
A: 10.0-14.0
|
||||
B: 12.0-13.0
|
||||
C: 13.5-15.0
|
||||
```
|
||||
|
||||
These belong to one overlap group spanning `10.0-15.0`.
|
||||
|
||||
Overlap groups should record:
|
||||
|
||||
- Overlap group ID.
|
||||
- Group start time.
|
||||
- Group end time.
|
||||
- Segment references.
|
||||
- Speakers involved.
|
||||
- Classification, if known.
|
||||
- Resolution status.
|
||||
|
||||
Initial classifications may include:
|
||||
|
||||
- `unknown`
|
||||
- `minor_overlap`
|
||||
- `handoff`
|
||||
- `backchannel`
|
||||
- `crosstalk`
|
||||
|
||||
The `resolve-overlaps` module uses preserved word-level timing to replace detected overlap-group segments with smaller word-run segments when usable timing is available. Resolution expands each overlap window by the configured coalesce gap so nearby same-speaker context can be absorbed into the replacement runs. Once a segment is selected for replacement, all timed words from that segment participate in word-run construction so text is not clipped at the window boundary. Groups without usable word timing remain unresolved for later passes or human review.
|
||||
|
||||
Overlap resolution should be non-destructive. Original segment text, timing, and source metadata must remain recoverable.
|
||||
|
||||
## Final ID Assignment
|
||||
|
||||
Final segment IDs should be assigned by an explicit postprocessing module after every transformation that can affect segment order.
|
||||
|
||||
Final IDs should be sequential integers starting from `1`.
|
||||
|
||||
Final IDs should reflect final chronological order.
|
||||
|
||||
Before final ID assignment, modules should reference segments using stable internal references rather than final output IDs.
|
||||
|
||||
## Output Invariants
|
||||
|
||||
A valid merged transcript should satisfy:
|
||||
|
||||
- Every segment has a unique integer ID.
|
||||
- Segment IDs begin at `1`.
|
||||
- Segment IDs increase in final chronological order.
|
||||
- Every segment has a canonical speaker.
|
||||
- Every segment has a source.
|
||||
- Every segment has `start >= 0`.
|
||||
- Every segment has `end >= start`.
|
||||
- The segments array is sorted deterministically.
|
||||
- Any `overlap_group_id` on a segment refers to an existing overlap group.
|
||||
- Every overlap group references at least two segments.
|
||||
- Every referenced segment exists.
|
||||
- Output validates against the selected output schema.
|
||||
|
||||
For full-schema trim output, overlap groups are recomputed from retained segments so overlap annotations and group references remain internally consistent after projection.
|
||||
|
||||
## Determinism Requirements
|
||||
|
||||
Given the same inputs, config, and application version, `seriatim` should produce byte-stable JSON output where practical.
|
||||
|
||||
To support this:
|
||||
|
||||
- Sort input specs deterministically unless explicit input order is meaningful.
|
||||
- Use stable sort keys.
|
||||
- Assign final IDs only after final ordering.
|
||||
- Avoid Go map iteration order affecting output.
|
||||
- Emit JSON through structs with stable field ordering.
|
||||
- Record application version in output metadata.
|
||||
- Record enabled module names and module order in output metadata or report data.
|
||||
|
||||
Trim-specific determinism requirements:
|
||||
|
||||
- Selector normalization and retained IDs are deterministic.
|
||||
- Old-to-new ID mapping in trim reports is emitted in deterministic order.
|
||||
- Full-schema overlap recomputation is deterministic for the same input artifact and selector.
|
||||
|
||||
Normalize-specific determinism requirements:
|
||||
|
||||
- Input-shape detection is deterministic.
|
||||
- Segment ordering is deterministic for identical input data.
|
||||
- Output IDs are always reassigned sequentially after deterministic sorting.
|
||||
- Normalize diagnostic reports are deterministic for identical inputs and configuration.
|
||||
|
||||
## Go Package Layout
|
||||
|
||||
```text
|
||||
cmd/seriatim/ CLI entrypoint
|
||||
internal/config/ CLI/env/config loading and validation
|
||||
internal/pipeline/ Pipeline orchestration and module registry
|
||||
internal/builtin/ Built-in pipeline modules
|
||||
internal/artifact/ Conversion from internal model to public output schema
|
||||
internal/normalize/ Normalize input parsing, validation, deterministic sorting, schema conversion, and diagnostics
|
||||
internal/trim/ Artifact parsing, trim selection, schema conversion, overlap recomputation for full schema
|
||||
internal/buildinfo/ Build-time version metadata
|
||||
internal/speaker/ Speaker map parsing and lookup
|
||||
internal/model/ Canonical and merged transcript models
|
||||
internal/overlap/ Overlap detection and refinement helpers
|
||||
internal/autocorrect/ Word replacement rules
|
||||
internal/report/ Report model and event accumulation
|
||||
schema/ Public output contract and JSON Schema validation
|
||||
```
|
||||
|
||||
Package boundaries should follow data ownership. Shared models belong in `internal/model`; stage-specific behavior belongs in the relevant stage package.
|
||||
|
||||
For trim:
|
||||
|
||||
- `internal/trim` contains pure transformation logic over artifact structs.
|
||||
- CLI command code handles only flag parsing, file I/O, and report emission.
|
||||
- Transform logic is deterministic and pure except for command-layer I/O.
|
||||
|
||||
For normalize:
|
||||
|
||||
- `internal/normalize` contains parsing/validation and deterministic schema conversion logic.
|
||||
- CLI command code handles flag parsing and delegates execution.
|
||||
- Normalize remains artifact-level and does not compose merge pipeline modules.
|
||||
|
||||
## Default Modules
|
||||
|
||||
The default pipeline is equivalent to explicit module lists.
|
||||
|
||||
Recommended default preprocessing modules:
|
||||
|
||||
```text
|
||||
validate-raw,normalize-speakers,trim-text
|
||||
```
|
||||
|
||||
Recommended default postprocessing modules:
|
||||
|
||||
```text
|
||||
detect-overlaps,resolve-overlaps,backchannel,filler,resolve-danglers,coalesce,detect-overlaps,autocorrect,assign-ids,validate-output
|
||||
```
|
||||
|
||||
The default output module is:
|
||||
|
||||
```text
|
||||
json
|
||||
```
|
||||
220
docs/cli.md
Normal file
220
docs/cli.md
Normal file
@@ -0,0 +1,220 @@
|
||||
# CLI Reference
|
||||
|
||||
## Shortest useful command
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim merge \
|
||||
--input-file speaker-a.json \
|
||||
--input-file speaker-b.json \
|
||||
--output-file merged.json
|
||||
```
|
||||
|
||||
## Command overview
|
||||
|
||||
| Command | Purpose |
|
||||
| --- | --- |
|
||||
| `merge` | Merge one or more raw transcript JSON inputs into one seriatim artifact. |
|
||||
| `trim` | Keep or remove segment IDs from an existing seriatim artifact. |
|
||||
| `normalize` | Canonicalize transcript-like JSON into a seriatim artifact. |
|
||||
| `render` | Render an existing seriatim artifact as Markdown. |
|
||||
|
||||
Root usage:
|
||||
|
||||
```text
|
||||
seriatim [command]
|
||||
```
|
||||
|
||||
## Global flags
|
||||
|
||||
| Flag | Description |
|
||||
| --- | --- |
|
||||
| `-h, --help` | Show help. |
|
||||
| `-v, --version` | Show build version. |
|
||||
|
||||
## `merge`
|
||||
|
||||
Usage:
|
||||
|
||||
```text
|
||||
seriatim merge [flags]
|
||||
```
|
||||
|
||||
Flags:
|
||||
|
||||
| Flag | Required | Default | Description |
|
||||
| --- | --- | --- | --- |
|
||||
| `--input-file stringArray` | Yes, repeat at least once | none | Input transcript JSON file(s). |
|
||||
| `--output-file string` | Yes | none | Output transcript JSON file path. |
|
||||
| `--report-file string` | No | none | Optional report JSON path. |
|
||||
| `--speakers string` | No | none | Speaker-map YAML file. |
|
||||
| `--autocorrect string` | No | none | Autocorrect YAML file. |
|
||||
| `--input-reader string` | No | `json-files` | Input reader module name. |
|
||||
| `--output-modules string` | No | `json` | Comma-separated output module names. |
|
||||
| `--output-schema string` | No | `seriatim-intermediate` | Output schema name: `seriatim-minimal`, `seriatim-intermediate`, `seriatim-full`. |
|
||||
| `--preprocessing-modules string` | No | `validate-raw,normalize-speakers,trim-text` | Comma-separated preprocessing module names, run in order. |
|
||||
| `--postprocessing-modules string` | No | `detect-overlaps,resolve-overlaps,backchannel,filler,resolve-danglers,coalesce,detect-overlaps,autocorrect,assign-ids,validate-output` | Comma-separated postprocessing module names, run in order. |
|
||||
| `--coalesce-gap string` | No | `3.0` | Non-negative seconds for coalescing and overlap-resolution context. |
|
||||
|
||||
`merge` behavior and validation:
|
||||
|
||||
- Unknown input reader, preprocessing module, postprocessing module, or output module fails the command.
|
||||
- Preprocessing order must satisfy module state requirements (`raw` -> `canonical`); invalid order fails.
|
||||
- Input files are validated, deduplicated, normalized, then sorted for deterministic processing.
|
||||
- Optional report output is written only when `--report-file` is set.
|
||||
- When `--output-schema` is omitted, schema resolution is: `SERIATIM_OUTPUT_SCHEMA` -> default `seriatim-intermediate`.
|
||||
|
||||
## `trim`
|
||||
|
||||
Usage:
|
||||
|
||||
```text
|
||||
seriatim trim [flags]
|
||||
```
|
||||
|
||||
Flags:
|
||||
|
||||
| Flag | Required | Default | Description |
|
||||
| --- | --- | --- | --- |
|
||||
| `--input-file string` | Yes | none | Input seriatim artifact JSON file. |
|
||||
| `--output-file string` | Yes | none | Output transcript JSON file path. |
|
||||
| `--keep string` | Exactly one of `--keep` / `--remove` | none | Segment ID selector to keep. |
|
||||
| `--remove string` | Exactly one of `--keep` / `--remove` | none | Segment ID selector to remove. |
|
||||
| `--output-schema string` | No | preserve input artifact schema | Output schema override: `seriatim-minimal`, `seriatim-intermediate`, `seriatim-full`. |
|
||||
| `--report-file string` | No | none | Optional report JSON path. |
|
||||
| `--allow-empty` | No | `false` | Allow output with zero segments. |
|
||||
|
||||
Selector rules:
|
||||
|
||||
- IDs must be positive integers.
|
||||
- Single IDs and inclusive ranges are supported: `1`, `1-10`.
|
||||
- Comma-separated selectors are supported: `1-10,15,20-25`.
|
||||
- Whitespace around commas and hyphens is allowed.
|
||||
- Descending ranges (example `10-1`) are invalid.
|
||||
- Duplicates and overlapping ranges are normalized as a union.
|
||||
|
||||
`trim` behavior:
|
||||
|
||||
- Input must already be a valid seriatim artifact (not raw merge input JSON).
|
||||
- Output keeps transcript order from input and renumbers retained segment IDs sequentially.
|
||||
- If `--output-schema` is omitted, the input artifact schema is preserved.
|
||||
- `trim` never runs merge preprocessing/postprocessing modules.
|
||||
|
||||
## `normalize`
|
||||
|
||||
Usage:
|
||||
|
||||
```text
|
||||
seriatim normalize [flags]
|
||||
```
|
||||
|
||||
Flags:
|
||||
|
||||
| Flag | Required | Default | Description |
|
||||
| --- | --- | --- | --- |
|
||||
| `--input-file string` | Yes | none | Input transcript JSON file. |
|
||||
| `--output-file string` | Yes | none | Output transcript JSON file path. |
|
||||
| `--output-schema string` | No | `seriatim-intermediate` | Output schema name: `seriatim-minimal`, `seriatim-intermediate`, `seriatim-full`. |
|
||||
| `--output-modules string` | No | `json` | Comma-separated output module names (`json` only). |
|
||||
| `--report-file string` | No | none | Optional report JSON path. |
|
||||
|
||||
`normalize` input shapes:
|
||||
|
||||
- Object with top-level `segments` array.
|
||||
- Bare top-level segment array.
|
||||
|
||||
`normalize` behavior:
|
||||
|
||||
- Sorts deterministically and reassigns output IDs sequentially from `1`.
|
||||
- Fills missing/blank speakers with `Unknown_Speaker`.
|
||||
- Repairs/sanitizes timing fields deterministically; rejects invalid repaired timing.
|
||||
- Drops segments with missing or blank text.
|
||||
- Does not run merge modules.
|
||||
- When `--output-schema` is omitted, schema resolution is: `SERIATIM_OUTPUT_SCHEMA` -> default `seriatim-intermediate`.
|
||||
|
||||
## `render`
|
||||
|
||||
Usage:
|
||||
|
||||
```text
|
||||
seriatim render [flags]
|
||||
```
|
||||
|
||||
Flags:
|
||||
|
||||
| Flag | Required | Default | Description |
|
||||
| --- | --- | --- | --- |
|
||||
| `--input-file string` | Yes | none | Input seriatim artifact JSON file. |
|
||||
| `--output-file string` | Yes | none | Rendered output file path. |
|
||||
| `--format string` | Yes | none | Output format. Current supported value: `markdown`. |
|
||||
| `--title string` | No | `Transcript` | Markdown document title. |
|
||||
| `--include-timestamps` | No | `true` | Include `[HH:MM:SS–HH:MM:SS]` per segment. |
|
||||
| `--include-segment-ids` | No | `false` | Include `[#id]` marker per segment. |
|
||||
| `--include-metadata` | No | `false` | Include artifact metadata block near the top. |
|
||||
|
||||
`render` behavior:
|
||||
|
||||
- Input must be a valid existing seriatim output artifact (`seriatim-minimal`, `seriatim-intermediate`, or `seriatim-full`).
|
||||
- Raw WhisperX-style JSON is rejected.
|
||||
- `render` does not execute merge/trim/normalize transformations.
|
||||
- `render` has no `--report-file` output in the current implementation.
|
||||
- Markdown output is deterministic for the same input artifact and render flags.
|
||||
- Category names are not printed directly; `background`, `backchannel`, and `filler` only influence italics.
|
||||
|
||||
## Common workflows
|
||||
|
||||
Merge with a speaker map and report output:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim merge \
|
||||
--input-file examples/minimal-merge/input-alice.json \
|
||||
--input-file examples/minimal-merge/input-bob.json \
|
||||
--speakers examples/minimal-merge/speakers.yml \
|
||||
--output-file /tmp/seriatim-example-merge.json \
|
||||
--report-file /tmp/seriatim-example-merge-report.json
|
||||
```
|
||||
|
||||
Trim to a segment subset:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim trim \
|
||||
--input-file examples/trim/input-full.json \
|
||||
--output-file /tmp/seriatim-example-trim.json \
|
||||
--keep "1-2"
|
||||
```
|
||||
|
||||
Normalize an external transcript JSON file:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim normalize \
|
||||
--input-file examples/normalize/object-with-segments.json \
|
||||
--output-file /tmp/seriatim-example-normalize-object.json
|
||||
```
|
||||
|
||||
Render an existing artifact as Markdown:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim render \
|
||||
--input-file examples/render/input-intermediate.json \
|
||||
--output-file /tmp/seriatim-example-render.md \
|
||||
--format markdown
|
||||
```
|
||||
|
||||
## Exit and errors
|
||||
|
||||
- Commands return exit code `0` on success.
|
||||
- On error, the CLI prints one error line to stderr and exits with status `1`.
|
||||
- Cobra usage text is silenced on runtime errors; use `--help` for command usage.
|
||||
|
||||
## Related docs
|
||||
|
||||
- Configuration reference: [config.md](config.md)
|
||||
- Operations guide: [operations.md](operations.md)
|
||||
- Troubleshooting: [troubleshooting.md](troubleshooting.md)
|
||||
- Integration notes:
|
||||
- [integrations/whisperx-json.md](integrations/whisperx-json.md)
|
||||
- [integrations/output-schemas.md](integrations/output-schemas.md)
|
||||
- Synthetic examples: [../examples/README.md](../examples/README.md)
|
||||
- Public output schemas:
|
||||
- [../schema/minimal-output.schema.json](../schema/minimal-output.schema.json)
|
||||
- [../schema/intermediate-output.schema.json](../schema/intermediate-output.schema.json)
|
||||
- [../schema/full-output.schema.json](../schema/full-output.schema.json)
|
||||
184
docs/config.md
Normal file
184
docs/config.md
Normal file
@@ -0,0 +1,184 @@
|
||||
# Configuration Reference
|
||||
|
||||
## Configuration surfaces
|
||||
|
||||
seriatim has no central JSON/TOML/YAML application config file.
|
||||
|
||||
Runtime configuration comes from:
|
||||
|
||||
1. CLI flags
|
||||
2. Environment variables (`SERIATIM_*`)
|
||||
3. Optional YAML rule files referenced by CLI flags (`--speakers`, `--autocorrect`)
|
||||
|
||||
## Output schema precedence
|
||||
|
||||
For `merge` and `normalize`:
|
||||
|
||||
1. `--output-schema` flag (when explicitly set)
|
||||
2. `SERIATIM_OUTPUT_SCHEMA`
|
||||
3. default `seriatim-intermediate`
|
||||
|
||||
For `trim`:
|
||||
|
||||
- If `--output-schema` is omitted, output preserves the input artifact schema.
|
||||
- If `--output-schema` is set, it must be one of `seriatim-minimal`, `seriatim-intermediate`, `seriatim-full`.
|
||||
|
||||
## Render format and defaults
|
||||
|
||||
`render` requires `--input-file`, `--output-file`, and `--format`.
|
||||
Current supported format value is `markdown`.
|
||||
|
||||
Render defaults:
|
||||
|
||||
- `--title`: `Transcript`
|
||||
- `--include-timestamps`: `true`
|
||||
- `--include-segment-ids`: `false`
|
||||
- `--include-metadata`: `false`
|
||||
|
||||
## Merge module defaults
|
||||
|
||||
Default merge module selections:
|
||||
|
||||
- `--input-reader`: `json-files`
|
||||
- `--preprocessing-modules`: `validate-raw,normalize-speakers,trim-text`
|
||||
- `--postprocessing-modules`: `detect-overlaps,resolve-overlaps,backchannel,filler,resolve-danglers,coalesce,detect-overlaps,autocorrect,assign-ids,validate-output`
|
||||
- `--output-modules`: `json`
|
||||
|
||||
Module-list notes:
|
||||
|
||||
- Lists are comma-separated.
|
||||
- Empty module names are invalid.
|
||||
- Unknown module names fail the command.
|
||||
- Preprocessing order must satisfy state requirements.
|
||||
|
||||
## Environment variables
|
||||
|
||||
| Variable | Default | Used by | Rules |
|
||||
| --- | --- | --- | --- |
|
||||
| `SERIATIM_OUTPUT_SCHEMA` | `seriatim-intermediate` | `merge`, `normalize` | Must be `seriatim-minimal`, `seriatim-intermediate`, or `seriatim-full`. Ignored when `--output-schema` is explicitly set. |
|
||||
| `SERIATIM_OVERLAP_WORD_RUN_GAP` | `1.0` | `merge` | Positive float (`> 0`). |
|
||||
| `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW` | `1.0` | `merge` | Positive float (`> 0`). |
|
||||
| `SERIATIM_BACKCHANNEL_MAX_DURATION` | `2.0` | `merge` | Positive float (`> 0`). |
|
||||
| `SERIATIM_FILLER_MAX_DURATION` | `1.25` | `merge` | Positive float (`> 0`). |
|
||||
|
||||
Additional merge threshold flag:
|
||||
|
||||
- `--coalesce-gap` defaults to `3.0` and must be a non-negative float (`>= 0`).
|
||||
|
||||
## `speakers.yml`
|
||||
|
||||
Purpose:
|
||||
|
||||
- Maps each merge input filename basename to a canonical speaker label.
|
||||
|
||||
Top-level key:
|
||||
|
||||
- `match` (array of ordered rules)
|
||||
|
||||
Rule fields:
|
||||
|
||||
- `speaker` (required, non-empty)
|
||||
- `match` (required, non-empty array of non-empty strings)
|
||||
|
||||
Example:
|
||||
|
||||
```yaml
|
||||
match:
|
||||
- speaker: "Alice"
|
||||
match:
|
||||
- "alice_track"
|
||||
- "alice"
|
||||
|
||||
- speaker: "Bob"
|
||||
match:
|
||||
- "bob_track"
|
||||
```
|
||||
|
||||
Behavior:
|
||||
|
||||
- Matching is case-insensitive.
|
||||
- Matching is against basename only (not full path).
|
||||
- First matching rule wins.
|
||||
- Duplicate `speaker` values are invalid.
|
||||
- If any input file has no match, merge fails.
|
||||
|
||||
## `autocorrect.yml`
|
||||
|
||||
Purpose:
|
||||
|
||||
- Applies ordered token-level text replacements during merge `autocorrect` postprocessing.
|
||||
|
||||
Top-level key:
|
||||
|
||||
- `autocorrect` (array of rules)
|
||||
|
||||
Rule fields:
|
||||
|
||||
- `target` (required, non-empty)
|
||||
- `match` (required, non-empty array of non-empty strings)
|
||||
|
||||
Example:
|
||||
|
||||
```yaml
|
||||
autocorrect:
|
||||
- target: "Godfrey"
|
||||
match:
|
||||
- "God-free"
|
||||
|
||||
- target: "Mike Brown"
|
||||
match:
|
||||
- "Mike Pat"
|
||||
```
|
||||
|
||||
Behavior:
|
||||
|
||||
- Match strings are case-sensitive.
|
||||
- Replacements are whole-token only (no substring replacement inside larger tokens).
|
||||
- Duplicate match strings within one rule are invalid.
|
||||
- Duplicate match strings across different rules are invalid.
|
||||
- If `--autocorrect` is not provided, the autocorrect module is skipped.
|
||||
|
||||
## Path and validation rules
|
||||
|
||||
All commands:
|
||||
|
||||
- `--input-file` paths must exist and must be files.
|
||||
- Output/report parent directories must already exist.
|
||||
- Paths are normalized before use.
|
||||
|
||||
`merge`:
|
||||
|
||||
- Requires at least one `--input-file`.
|
||||
- Rejects duplicate `--input-file` paths.
|
||||
- Sorts normalized input file paths for deterministic execution.
|
||||
- `--speakers` and `--autocorrect` are optional, but when set they must point to existing files.
|
||||
|
||||
`trim`:
|
||||
|
||||
- Requires exactly one of `--keep` or `--remove`.
|
||||
- `--keep` and `--remove` are mutually exclusive.
|
||||
- Validates optional `--output-schema` when provided.
|
||||
|
||||
`normalize`:
|
||||
|
||||
- Validates `--output-schema` through the same schema set as `merge`.
|
||||
- Currently accepts only `json` in `--output-modules`.
|
||||
|
||||
`render`:
|
||||
|
||||
- Requires `--input-file`, `--output-file`, and `--format`.
|
||||
- Validates `--format` as `markdown`.
|
||||
|
||||
## Related docs
|
||||
|
||||
- CLI reference: [cli.md](cli.md)
|
||||
- Operations guide: [operations.md](operations.md)
|
||||
- Troubleshooting: [troubleshooting.md](troubleshooting.md)
|
||||
- YAML example files:
|
||||
- [../examples/speakers.yml](../examples/speakers.yml)
|
||||
- [../examples/autocorrect.yml](../examples/autocorrect.yml)
|
||||
- Synthetic examples: [../examples/README.md](../examples/README.md)
|
||||
- Public output schemas:
|
||||
- [../schema/minimal-output.schema.json](../schema/minimal-output.schema.json)
|
||||
- [../schema/intermediate-output.schema.json](../schema/intermediate-output.schema.json)
|
||||
- [../schema/full-output.schema.json](../schema/full-output.schema.json)
|
||||
69
docs/integrations/output-schemas.md
Normal file
69
docs/integrations/output-schemas.md
Normal file
@@ -0,0 +1,69 @@
|
||||
# Output Schemas
|
||||
|
||||
## Scope
|
||||
|
||||
seriatim emits one of three public JSON output contracts:
|
||||
|
||||
- `seriatim-minimal`
|
||||
- `seriatim-intermediate`
|
||||
- `seriatim-full`
|
||||
|
||||
These are used by `merge`, `trim`, and `normalize`, and are accepted as input
|
||||
by `render`.
|
||||
|
||||
## Schema roles
|
||||
|
||||
`seriatim-minimal`:
|
||||
|
||||
- compact metadata plus ordered transcript segments
|
||||
- no source/provenance fields
|
||||
- no overlap groups
|
||||
|
||||
`seriatim-intermediate`:
|
||||
|
||||
- compact metadata plus ordered segments
|
||||
- includes optional segment `categories`
|
||||
- no source/provenance fields
|
||||
- no overlap groups
|
||||
|
||||
`seriatim-full`:
|
||||
|
||||
- full metadata (`input_reader`, module lists, input files, output modules)
|
||||
- source/provenance fields on segments
|
||||
- overlap-group data
|
||||
- version metadata populated from build info (`internal/buildinfo`)
|
||||
|
||||
## Semantic invariants
|
||||
|
||||
All schema outputs enforce:
|
||||
|
||||
- segment IDs are sequential starting at `1`
|
||||
- segment timing uses `end >= start`
|
||||
|
||||
Full schema also enforces overlap-group timing (`end >= start`).
|
||||
|
||||
## Validation APIs
|
||||
|
||||
Go package: `gitea.maximumdirect.net/eric/seriatim/schema`
|
||||
|
||||
Key validators:
|
||||
|
||||
- `schema.ValidateMinimalTranscript`
|
||||
- `schema.ValidateIntermediateTranscript`
|
||||
- `schema.ValidateTranscript`
|
||||
- `schema.ValidateMinimalJSON`
|
||||
- `schema.ValidateIntermediateJSON`
|
||||
- `schema.ValidateJSON`
|
||||
|
||||
## Machine-readable schema files
|
||||
|
||||
- [../../schema/minimal-output.schema.json](../../schema/minimal-output.schema.json)
|
||||
- [../../schema/intermediate-output.schema.json](../../schema/intermediate-output.schema.json)
|
||||
- [../../schema/full-output.schema.json](../../schema/full-output.schema.json)
|
||||
|
||||
## Related docs and examples
|
||||
|
||||
- CLI reference: [../cli.md](../cli.md)
|
||||
- Artifact internals: [../internal/artifacts.md](../internal/artifacts.md)
|
||||
- Trim example input artifact:
|
||||
- [../../examples/trim/input-full.json](../../examples/trim/input-full.json)
|
||||
85
docs/integrations/whisperx-json.md
Normal file
85
docs/integrations/whisperx-json.md
Normal file
@@ -0,0 +1,85 @@
|
||||
# WhisperX-Like JSON Input
|
||||
|
||||
## Scope
|
||||
|
||||
This document covers the implemented JSON subset consumed by `seriatim merge`.
|
||||
It does not describe full WhisperX output.
|
||||
No explicit WhisperX version is encoded in the repository.
|
||||
|
||||
## Supported top-level shape
|
||||
|
||||
Merge expects a JSON object with top-level `segments` array:
|
||||
|
||||
```json
|
||||
{
|
||||
"segments": [
|
||||
{
|
||||
"start": 0.0,
|
||||
"end": 1.2,
|
||||
"text": "hello"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
## Supported segment fields
|
||||
|
||||
Required per segment:
|
||||
|
||||
- `start` (number, `>= 0`)
|
||||
- `end` (number, `>= start`)
|
||||
- `text` (string)
|
||||
|
||||
Optional per segment:
|
||||
|
||||
- `words` (array)
|
||||
|
||||
## Supported word fields
|
||||
|
||||
Required when a word object is present:
|
||||
|
||||
- `word` (string)
|
||||
|
||||
Optional word timing fields:
|
||||
|
||||
- `start` (number)
|
||||
- `end` (number)
|
||||
|
||||
Timing rules:
|
||||
|
||||
- if both `start` and `end` are present, they must be numeric and `end >= start`
|
||||
- if either timing field is missing, the word is accepted but not used as a
|
||||
timing anchor for overlap resolution
|
||||
|
||||
Additional optional word fields:
|
||||
|
||||
- `score` (number)
|
||||
- `speaker` (string)
|
||||
|
||||
## Validation and failure behavior
|
||||
|
||||
Merge fails for:
|
||||
|
||||
- malformed JSON
|
||||
- missing top-level `segments`
|
||||
- non-array `segments`
|
||||
- missing required segment fields
|
||||
- wrong field types
|
||||
- negative segment/word start times
|
||||
- segment/word end before start
|
||||
|
||||
Word timing missing from a word does not fail merge; it emits a warning event
|
||||
in the optional report.
|
||||
|
||||
## Overlap-resolution impact
|
||||
|
||||
- overlap resolution uses timed words when available
|
||||
- untimed words are kept in replacement text but do not provide timing anchors
|
||||
|
||||
## Related docs and examples
|
||||
|
||||
- CLI reference: [../cli.md](../cli.md)
|
||||
- Configuration reference: [../config.md](../config.md)
|
||||
- Minimal merge example inputs:
|
||||
- [../../examples/minimal-merge/input-alice.json](../../examples/minimal-merge/input-alice.json)
|
||||
- [../../examples/minimal-merge/input-bob.json](../../examples/minimal-merge/input-bob.json)
|
||||
216
docs/internal/artifacts.md
Normal file
216
docs/internal/artifacts.md
Normal file
@@ -0,0 +1,216 @@
|
||||
# Artifact Internals
|
||||
|
||||
## Purpose
|
||||
|
||||
Describes implemented artifact parsing, conversion, validation, and render-model
|
||||
normalization internals.
|
||||
|
||||
## Artifact contracts
|
||||
|
||||
Public contracts live in `schema/`:
|
||||
|
||||
- full: `schema.Transcript`
|
||||
- intermediate: `schema.IntermediateTranscript`
|
||||
- minimal: `schema.MinimalTranscript`
|
||||
|
||||
Machine-readable schemas:
|
||||
|
||||
- `schema/full-output.schema.json`
|
||||
- `schema/intermediate-output.schema.json`
|
||||
- `schema/minimal-output.schema.json`
|
||||
|
||||
## Shared output-artifact parser
|
||||
|
||||
`internal/artifact/output_artifact.go` provides schema-aware parsing for
|
||||
existing seriatim output artifacts.
|
||||
|
||||
Behavior:
|
||||
|
||||
- accepts only valid full, intermediate, or minimal seriatim output artifacts
|
||||
- validates through `schema` semantic + JSON schema checks
|
||||
- rejects malformed JSON
|
||||
- rejects raw WhisperX-style JSON and other non-seriatim shapes
|
||||
|
||||
Consumers:
|
||||
|
||||
- `internal/trim` artifact-level trim flow
|
||||
- `internal/render` artifact-level render flow
|
||||
|
||||
## Merge conversion behavior
|
||||
|
||||
`internal/artifact/transcript.go` converts `model.MergedTranscript` to public
|
||||
contracts:
|
||||
|
||||
- full schema preserves source/provenance, overlap groups, and metadata module
|
||||
lists
|
||||
- intermediate schema emits segment timing/text/speaker with optional
|
||||
categories and compact metadata
|
||||
- minimal schema emits compact segment timing/text/speaker and compact metadata
|
||||
|
||||
Schema selection uses `internal/artifact.SelectedFromMerged`:
|
||||
|
||||
- `seriatim-full` -> `artifact.FromMerged`
|
||||
- `seriatim-intermediate` -> `artifact.IntermediateFromMerged`
|
||||
- `seriatim-minimal` -> `artifact.MinimalFromMerged`
|
||||
- unknown/empty -> intermediate fallback
|
||||
|
||||
## Trim internals
|
||||
|
||||
`internal/trim` handles artifact-level projection and does not execute merge
|
||||
pipeline modules.
|
||||
|
||||
Run layer (`run.go`):
|
||||
|
||||
1. Parse selector from validated config.
|
||||
2. Read and parse input artifact JSON.
|
||||
3. Apply trim projection through schema-aware artifact handling.
|
||||
4. Resolve output schema (preserve input schema unless overridden).
|
||||
5. Validate output artifact.
|
||||
6. Write output JSON.
|
||||
7. Optionally write report JSON with `trim-audit`.
|
||||
|
||||
Apply layer (`apply.go`):
|
||||
|
||||
- one shared projection policy for selector mode, input ID validation, selected
|
||||
ID existence checks, keep/remove filtering, removed IDs, and old-to-new ID
|
||||
mappings
|
||||
- schema-specific segment reconstruction for full/intermediate/minimal outputs
|
||||
- overlap-group recomputation only for full-schema outputs
|
||||
|
||||
Artifact conversion layer (`artifact.go`):
|
||||
|
||||
- schema-preserving trim application
|
||||
- supported schema conversions:
|
||||
- full -> intermediate/minimal
|
||||
- intermediate -> minimal
|
||||
- minimal -> intermediate
|
||||
- rejected conversion:
|
||||
- intermediate/minimal -> full
|
||||
|
||||
Trim invariants:
|
||||
|
||||
- selected IDs must exist in input
|
||||
- input IDs must be positive, unique, sequential
|
||||
- retained segment order follows input transcript order
|
||||
- output IDs are reassigned to `1..N`
|
||||
|
||||
## Normalize internals
|
||||
|
||||
`internal/normalize` canonicalizes transcript-like JSON input into a selected
|
||||
public schema.
|
||||
|
||||
Parse layer (`parse.go`):
|
||||
|
||||
- accepts object-with-`segments` or bare segment array
|
||||
- repairs missing timing deterministically
|
||||
- swaps inverted timing
|
||||
- fills missing/blank speaker with `Unknown_Speaker`
|
||||
- drops missing/blank text segments
|
||||
|
||||
Build layer (`build.go`):
|
||||
|
||||
- sorts deterministically by `(start, end, input_index, speaker)`
|
||||
- reassigns output IDs sequentially
|
||||
- builds minimal/intermediate/full output shape
|
||||
- validates selected output schema before write
|
||||
|
||||
Run layer (`normalize.go`):
|
||||
|
||||
- writes output JSON
|
||||
- optionally writes report with `normalize-audit`
|
||||
|
||||
Normalize invariant:
|
||||
|
||||
- report events do not embed transcript text
|
||||
|
||||
## Render internals
|
||||
|
||||
`internal/render` is an artifact-level, downstream-only renderer.
|
||||
|
||||
Model normalization (`normalize.go`):
|
||||
|
||||
- converts full/intermediate/minimal artifacts into a common render model
|
||||
- preserves segment order and segment IDs
|
||||
- normalizes per-segment fields to ID, start, end, speaker, text, categories
|
||||
- emits empty categories slice when categories are absent in input
|
||||
|
||||
Renderer registry (`registry.go`):
|
||||
|
||||
- resolves renderers by public format name
|
||||
- currently registers `markdown`
|
||||
|
||||
Markdown renderer (`markdown.go`):
|
||||
|
||||
- writes title header `# {title}`
|
||||
- renders optional `[HH:MM:SS–HH:MM:SS]` timestamps
|
||||
- renders optional `[#id]` segment references
|
||||
- renders `**speaker:** text`
|
||||
- italicizes text when categories include `background`, `backchannel`, or
|
||||
`filler`
|
||||
- ignores unknown categories
|
||||
- optionally includes metadata summary block
|
||||
|
||||
Run layer (`run.go`):
|
||||
|
||||
1. Read input artifact JSON.
|
||||
2. Parse via shared output-artifact parser.
|
||||
3. Normalize to render model.
|
||||
4. Resolve renderer by `--format`.
|
||||
5. Render text output.
|
||||
6. Write output file.
|
||||
|
||||
Render invariants:
|
||||
|
||||
- does not run merge/trim/normalize modules
|
||||
- does not expose report output
|
||||
- deterministic for identical input artifact and render flags
|
||||
|
||||
## Validation behavior
|
||||
|
||||
`schema/output.go` validates both structure and semantics:
|
||||
|
||||
- embedded JSON Schema validation via `jsonschema/v6`
|
||||
- semantic checks for sequential segment IDs starting at `1`
|
||||
- semantic checks for non-inverted segment timing (`end >= start`)
|
||||
- full schema overlap-group timing checks (`group.end >= group.start`)
|
||||
|
||||
## Boundaries
|
||||
|
||||
- CLI flag semantics belong to `docs/cli.md`.
|
||||
- Runtime config/env surfaces belong to `docs/config.md`.
|
||||
- This document describes internal conversion/validation behavior only.
|
||||
|
||||
## Failure behavior
|
||||
|
||||
Representative failure classes:
|
||||
|
||||
- malformed or unsupported input JSON shape
|
||||
- schema validation failure for parsed artifact or built output
|
||||
- unsupported schema conversion path (trim)
|
||||
- selector or input-ID consistency errors (trim)
|
||||
- unsupported renderer format (render)
|
||||
- output/report file write failures from command paths
|
||||
|
||||
## Tests to inspect before changes
|
||||
|
||||
- `schema/output_test.go`
|
||||
- `internal/artifact/transcript_test.go`
|
||||
- `internal/artifact/output_artifact_test.go`
|
||||
- `internal/trim/selector_test.go`
|
||||
- `internal/trim/artifact_test.go`
|
||||
- `internal/trim/apply_test.go`
|
||||
- `internal/normalize/parse_test.go`
|
||||
- `internal/render/normalize_test.go`
|
||||
- `internal/render/markdown_test.go`
|
||||
- `internal/render/registry_test.go`
|
||||
- `internal/cli/trim_test.go`
|
||||
- `internal/cli/normalize_test.go`
|
||||
- `internal/cli/render_test.go`
|
||||
|
||||
## Invariants
|
||||
|
||||
- Public artifacts are validated through `schema` before acceptance.
|
||||
- Segment IDs in emitted artifacts are sequential and deterministic.
|
||||
- Internal-only fields are not emitted in minimal/intermediate contracts.
|
||||
- Trim, normalize, and render stay artifact-level and do not execute merge
|
||||
modules.
|
||||
120
docs/internal/modules.md
Normal file
120
docs/internal/modules.md
Normal file
@@ -0,0 +1,120 @@
|
||||
# Built-In Modules
|
||||
|
||||
## Purpose
|
||||
|
||||
Describes implemented built-in module behavior and boundaries in
|
||||
`internal/builtin`.
|
||||
|
||||
## Implemented module set
|
||||
|
||||
Input reader:
|
||||
|
||||
- `json-files`
|
||||
|
||||
Preprocessing:
|
||||
|
||||
- `validate-raw`
|
||||
- `normalize-speakers`
|
||||
- `trim-text`
|
||||
|
||||
Merger:
|
||||
|
||||
- `chronological-merge`
|
||||
|
||||
Postprocessing:
|
||||
|
||||
- `detect-overlaps`
|
||||
- `resolve-overlaps`
|
||||
- `backchannel`
|
||||
- `filler`
|
||||
- `resolve-danglers`
|
||||
- `coalesce`
|
||||
- `autocorrect`
|
||||
- `assign-ids`
|
||||
- `validate-output`
|
||||
|
||||
Output writer:
|
||||
|
||||
- `json`
|
||||
|
||||
## Inputs, outputs, and side effects
|
||||
|
||||
- `json-files`: reads JSON files from `cfg.InputFiles`, parses supported
|
||||
segment/word fields, emits warnings for untimed words.
|
||||
- `validate-raw`: validates raw source/timing invariants.
|
||||
- `normalize-speakers`: converts raw transcripts to canonical segments,
|
||||
optionally resolving speakers from `cfg.SpeakersFile`.
|
||||
- `trim-text`: trims canonical segment text whitespace.
|
||||
- `chronological-merge`: flattens canonical segments and applies deterministic
|
||||
sort (`model.SegmentLess`).
|
||||
- `detect-overlaps`: annotates overlap groups.
|
||||
- `resolve-overlaps`: rewrites overlap groups using timed words and thresholds.
|
||||
- `backchannel`/`filler`: classify short utterances using duration thresholds.
|
||||
- `resolve-danglers`: merges dangling derived fragments.
|
||||
- `coalesce`: merges adjacent same-speaker segments within configured gap.
|
||||
- `autocorrect`: applies YAML replacement rules when configured.
|
||||
- `assign-ids`: assigns final sequential IDs.
|
||||
- `validate-output`: validates selected public artifact shape.
|
||||
- `json`: writes artifact JSON to `cfg.OutputFile` through shared deterministic
|
||||
JSON file writing.
|
||||
|
||||
Filesystem side effects are limited to:
|
||||
|
||||
- reading configured input/YAML files
|
||||
- writing configured output artifact
|
||||
|
||||
## Config fields used
|
||||
|
||||
Primary module inputs from `config.Config`:
|
||||
|
||||
- file paths: `InputFiles`, `SpeakersFile`, `AutocorrectFile`, `OutputFile`
|
||||
- schema/modules: `OutputSchema`, `OutputModules`
|
||||
- overlap/coalesce thresholds: `OverlapWordRunGap`,
|
||||
`WordRunReorderWindow`, `CoalesceGap`
|
||||
- category thresholds: `BackchannelMaxDuration`, `FillerMaxDuration`
|
||||
|
||||
## Ordering constraints
|
||||
|
||||
- Preprocessing must satisfy state contracts from `raw` to `canonical`.
|
||||
- Order-sensitive transforms should run before `assign-ids`.
|
||||
- `validate-output` should run after final ID assignment and output-shape
|
||||
mutations.
|
||||
- Default configuration includes a second `detect-overlaps` pass after
|
||||
transformations.
|
||||
|
||||
## Boundaries
|
||||
|
||||
- Modules implement behavior; CLI/config parsing remains outside modules.
|
||||
- Modules communicate through explicit model contracts and report events.
|
||||
- Output modules operate on final artifacts and do not re-run transform logic.
|
||||
|
||||
## Failure behavior
|
||||
|
||||
Representative failures:
|
||||
|
||||
- invalid input JSON shape or typed field errors (`json-files`)
|
||||
- invalid YAML or unmatched speaker map entries
|
||||
- unknown module names during registry resolution
|
||||
- invalid ordering/state transitions in preprocessing chain
|
||||
- validation failure in `validate-output`
|
||||
- output write failure in `json` writer
|
||||
|
||||
## Tests to inspect before changes
|
||||
|
||||
- `internal/builtin/preprocess_test.go`
|
||||
- `internal/builtin/postprocess_test.go`
|
||||
- `internal/overlap/resolve_test.go`
|
||||
- `internal/overlap/detect_test.go`
|
||||
- `internal/coalesce/coalesce_test.go`
|
||||
- `internal/danglers/danglers_test.go`
|
||||
- `internal/backchannel/backchannel_test.go`
|
||||
- `internal/filler/filler_test.go`
|
||||
- `internal/autocorrect/autocorrect_test.go`
|
||||
- `internal/cli/merge_test.go`
|
||||
|
||||
## Invariants
|
||||
|
||||
- Modules are selected by canonical name through the registry.
|
||||
- Execution is sequential and deterministic for a fixed configuration.
|
||||
- `assign-ids` defines final public segment IDs.
|
||||
- `validate-output` enforces public artifact contracts through `schema`.
|
||||
103
docs/internal/pipeline.md
Normal file
103
docs/internal/pipeline.md
Normal file
@@ -0,0 +1,103 @@
|
||||
# Pipeline Internals
|
||||
|
||||
## Purpose
|
||||
|
||||
Describes implemented merge pipeline orchestration in `internal/pipeline`.
|
||||
|
||||
## Inputs and outputs
|
||||
|
||||
Input:
|
||||
|
||||
- `config.Config`
|
||||
- registry-resolved modules from `internal/builtin`
|
||||
|
||||
Output:
|
||||
|
||||
- selected public artifact written by output writer modules
|
||||
- optional report JSON when `cfg.ReportFile` is set
|
||||
|
||||
## Stage contracts
|
||||
|
||||
The runner executes these contracts in order:
|
||||
|
||||
1. `InputReader`: external inputs -> `[]model.RawTranscript`
|
||||
2. `Preprocessor`: `PreprocessState` transformations (`raw` -> `canonical`)
|
||||
3. `Merger`: canonical transcripts -> `model.MergedTranscript`
|
||||
4. `Postprocessor`: merged transcript transformations
|
||||
5. `OutputWriter`: serialized artifact writes
|
||||
|
||||
`PreprocessState` must end in `StateCanonical` before merge.
|
||||
|
||||
## Registry resolution
|
||||
|
||||
`resolvePlan` maps configured names to modules:
|
||||
|
||||
- input reader: `cfg.InputReader`
|
||||
- preprocessors: `cfg.PreprocessingModules`
|
||||
- postprocessors: `cfg.PostprocessingModules`
|
||||
- output writers: `cfg.OutputModules`
|
||||
- merger: single registered merger
|
||||
|
||||
Unknown names fail fast with contextual errors.
|
||||
|
||||
## Execution order and reporting
|
||||
|
||||
- Modules run sequentially in configured order.
|
||||
- Events returned by modules are appended in execution order.
|
||||
- Report metadata includes input reader, input files, and module lists.
|
||||
- Output writer events are appended before optional report write.
|
||||
|
||||
## Config fields used
|
||||
|
||||
Runner-level fields:
|
||||
|
||||
- `InputReader`
|
||||
- `InputFiles`
|
||||
- `PreprocessingModules`
|
||||
- `PostprocessingModules`
|
||||
- `OutputModules`
|
||||
- `OutputSchema` (via `artifact.SelectedFromMerged`)
|
||||
- `ReportFile`
|
||||
|
||||
Module-specific settings are consumed inside builtin modules (for example
|
||||
coalesce gap and overlap thresholds).
|
||||
|
||||
## Adapters used
|
||||
|
||||
- Input adapters: registered `InputReader` implementations (default `json-files`).
|
||||
- Output adapters: registered `OutputWriter` implementations (default `json`).
|
||||
- Report adapter: `report.WriteJSON` when `cfg.ReportFile` is provided.
|
||||
|
||||
## Boundaries
|
||||
|
||||
- Pipeline does not parse CLI flags.
|
||||
- Pipeline does not normalize raw CLI strings.
|
||||
- Pipeline delegates conversion to public output contracts to `internal/artifact`.
|
||||
- Artifact-level commands `trim`, `normalize`, and `render` are outside this
|
||||
pipeline.
|
||||
|
||||
## Failure behavior
|
||||
|
||||
Pipeline returns errors from:
|
||||
|
||||
- registry resolution (unknown modules, missing merger)
|
||||
- invalid preprocessing state transitions
|
||||
- module read/process/merge/write failures
|
||||
- optional report write failure
|
||||
|
||||
No retry/resume state is stored.
|
||||
|
||||
## Tests to inspect before changes
|
||||
|
||||
- `internal/pipeline/runner_test.go`
|
||||
- `internal/builtin/preprocess_test.go`
|
||||
- `internal/builtin/postprocess_test.go`
|
||||
- `internal/cli/merge_test.go`
|
||||
|
||||
## Invariants
|
||||
|
||||
- Sequential deterministic execution order.
|
||||
- Preprocessing state must type-check from `raw` to `canonical`.
|
||||
- Module selection is explicit by canonical names.
|
||||
- Report event order reflects actual execution order.
|
||||
- Output artifact selection is schema-driven via `internal/artifact`.
|
||||
158
docs/operations.md
Normal file
158
docs/operations.md
Normal file
@@ -0,0 +1,158 @@
|
||||
# Operations Guide
|
||||
|
||||
## Scope
|
||||
|
||||
This document covers runtime operation of the implemented CLI commands:
|
||||
|
||||
- `merge`
|
||||
- `trim`
|
||||
- `normalize`
|
||||
- `render`
|
||||
|
||||
## Runtime model
|
||||
|
||||
seriatim is a single-process, filesystem-only CLI.
|
||||
|
||||
- Each invocation reads input files, processes in memory, and writes output files.
|
||||
- There is no daemon, queue, database, resume checkpoint, remote storage, or background worker.
|
||||
- On error, the command exits non-zero; there is no built-in retry/resume flow.
|
||||
|
||||
## Filesystem expectations
|
||||
|
||||
All commands require accessible local files and existing parent directories for outputs.
|
||||
|
||||
- Input paths must exist and must be files.
|
||||
- Output/report parent directories must already exist.
|
||||
- Output and report files are created with `os.Create`, so existing files at those paths are overwritten.
|
||||
|
||||
Command-specific expectations:
|
||||
|
||||
- `merge`: requires at least one `--input-file`; optional `--speakers` and `--autocorrect` paths must exist when provided.
|
||||
- `trim`: input must be an existing valid seriatim artifact JSON file.
|
||||
- `normalize`: input must be a JSON object with `segments` or a top-level segment array.
|
||||
- `render`: input must be an existing valid seriatim artifact JSON file.
|
||||
|
||||
## Normal workflow
|
||||
|
||||
### Merge
|
||||
|
||||
1. Provide one or more `--input-file` values.
|
||||
2. Optionally provide `--speakers`, `--autocorrect`, and `--report-file`.
|
||||
3. Provide `--output-file`.
|
||||
4. Run command.
|
||||
|
||||
Example:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim merge \
|
||||
--input-file speaker-a.json \
|
||||
--input-file speaker-b.json \
|
||||
--output-file merged.json \
|
||||
--report-file merge-report.json
|
||||
```
|
||||
|
||||
### Trim
|
||||
|
||||
1. Provide existing artifact with `--input-file`.
|
||||
2. Select segments with exactly one of `--keep` or `--remove`.
|
||||
3. Provide `--output-file`.
|
||||
4. Optionally provide `--output-schema`, `--allow-empty`, and `--report-file`.
|
||||
|
||||
Example:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim trim \
|
||||
--input-file merged.json \
|
||||
--output-file trimmed.json \
|
||||
--keep "1-20,25"
|
||||
```
|
||||
|
||||
### Normalize
|
||||
|
||||
1. Provide `--input-file` containing supported JSON shape.
|
||||
2. Provide `--output-file`.
|
||||
3. Optionally provide `--output-schema`, `--output-modules`, and `--report-file`.
|
||||
|
||||
Example:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim normalize \
|
||||
--input-file external.json \
|
||||
--output-file normalized.json \
|
||||
--report-file normalize-report.json
|
||||
```
|
||||
|
||||
### Render
|
||||
|
||||
1. Provide existing seriatim artifact with `--input-file`.
|
||||
2. Provide `--output-file`.
|
||||
3. Provide `--format markdown`.
|
||||
4. Optionally provide `--title`, `--include-timestamps`, `--include-segment-ids`, and `--include-metadata`.
|
||||
|
||||
Example:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim render \
|
||||
--input-file examples/render/input-intermediate.json \
|
||||
--output-file /tmp/seriatim-example-render.md \
|
||||
--format markdown
|
||||
```
|
||||
|
||||
## Output and report artifacts
|
||||
|
||||
Primary outputs:
|
||||
|
||||
- `merge`, `trim`, `normalize`: `--output-file` writes JSON transcript artifact in the selected schema.
|
||||
- `render`: `--output-file` writes presentation Markdown.
|
||||
|
||||
Optional report output:
|
||||
|
||||
- `--report-file` writes deterministic JSON report events.
|
||||
- `merge` report metadata records reader/modules and event sequence.
|
||||
- `trim` report includes a `trim-audit` event with mode/selector/counts and old-to-new ID mapping.
|
||||
- `normalize` report includes a `normalize-audit` event with input shape, repair stats, and output selection details.
|
||||
- `render` has no report output in the current implementation.
|
||||
|
||||
## Failure and retry behavior
|
||||
|
||||
Failure behavior:
|
||||
|
||||
- Errors are printed once to stderr by the root command and exit status is `1`.
|
||||
- There is no partial-state recovery mechanism.
|
||||
|
||||
Retry guidance:
|
||||
|
||||
1. Fix the reported input/config/path issue.
|
||||
2. Re-run the same command.
|
||||
3. If a prior run created a partial or unwanted output/report file, remove it and rerun.
|
||||
|
||||
Operational notes:
|
||||
|
||||
- With identical inputs/config/version, `merge` behavior is deterministic and input files are sorted before processing.
|
||||
- With identical input artifact and render flags, `render` output is deterministic.
|
||||
|
||||
## Cleanup
|
||||
|
||||
seriatim does not manage retention.
|
||||
|
||||
- Remove unneeded output/report artifacts manually.
|
||||
- No cache, state directory, or lock files are maintained by the application.
|
||||
|
||||
## Privacy considerations
|
||||
|
||||
Transcript artifacts and reports are local files and may contain sensitive conversational data.
|
||||
|
||||
- Store outputs in controlled directories with appropriate OS permissions.
|
||||
- Share report files carefully; they include file paths and processing diagnostics.
|
||||
- Normalize report events intentionally avoid embedding transcript text, but output artifacts contain transcript content.
|
||||
- Rendered Markdown is human-readable transcript content and should be handled as sensitive output when applicable.
|
||||
|
||||
## Related docs
|
||||
|
||||
- CLI reference: [cli.md](cli.md)
|
||||
- Configuration reference: [config.md](config.md)
|
||||
- Troubleshooting: [troubleshooting.md](troubleshooting.md)
|
||||
- Integration notes:
|
||||
- [integrations/whisperx-json.md](integrations/whisperx-json.md)
|
||||
- [integrations/output-schemas.md](integrations/output-schemas.md)
|
||||
- Synthetic examples: [../examples/README.md](../examples/README.md)
|
||||
220
docs/policy/architecture.md
Normal file
220
docs/policy/architecture.md
Normal file
@@ -0,0 +1,220 @@
|
||||
# Architecture Policy
|
||||
|
||||
## Purpose
|
||||
|
||||
This document defines seriatim's development architecture and invariants for
|
||||
maintainers and automated coding agents. It describes how the implemented
|
||||
system is intended to be built and changed. It is not a user manual, CLI
|
||||
reference, config reference, or roadmap.
|
||||
|
||||
Keep this document aligned with [documentation policy](documentation.md). It
|
||||
must describe current behavior only; planned or speculative work belongs under
|
||||
`docs/roadmap/`.
|
||||
|
||||
## Project Shape
|
||||
|
||||
seriatim is a Go CLI for transcript artifact processing. The implemented
|
||||
commands are `merge`, `trim`, `normalize`, and `render`.
|
||||
|
||||
`merge` reads one or more JSON transcript files, optionally maps input files to
|
||||
canonical speakers, runs a registry-selected preprocessing chain, merges
|
||||
canonical segments into deterministic chronological order, runs a
|
||||
registry-selected postprocessing chain, validates the selected output schema,
|
||||
and writes JSON output plus an optional JSON report.
|
||||
|
||||
`trim`, `normalize`, and `render` are artifact-level commands outside the merge
|
||||
pipeline. `trim` reads an existing seriatim output artifact and projects it by
|
||||
segment ID. `normalize` reads transcript-like JSON and emits one of seriatim's
|
||||
supported output schemas. `render` reads an existing seriatim output artifact
|
||||
and emits human-readable Markdown. None of these commands runs merge
|
||||
preprocessing or postprocessing modules.
|
||||
|
||||
The supported public output schemas are `seriatim-minimal`,
|
||||
`seriatim-intermediate`, and `seriatim-full`. For command and flag details, use
|
||||
[CLI reference](../cli.md) and [configuration reference](../config.md).
|
||||
|
||||
## Core Design Principles
|
||||
|
||||
- Keep a hexagonal architecture boundary. Domain models, stage contracts, and
|
||||
deterministic transformations must stay separate from CLI parsing,
|
||||
filesystem access, config loading, reporting, and other external adapters.
|
||||
- Keep stages and modules composable. Built-in modules are selected by
|
||||
canonical registry names and implement explicit interfaces for their pipeline
|
||||
role.
|
||||
- Preserve deterministic behavior. Given the same inputs, configuration, and
|
||||
version, output ordering, segment IDs, schema validation, and report event
|
||||
ordering should remain stable.
|
||||
- Current command execution is sequential. There is no scheduler, worker pool,
|
||||
or concurrent module execution in the implemented pipeline. Any concurrency
|
||||
added later must be bounded, observable, and must not make output handling
|
||||
nondeterministic.
|
||||
- Prefer the Go standard library. Third-party dependencies should remain narrow
|
||||
and justified, such as Cobra for CLI structure, YAML parsing, and JSON Schema
|
||||
validation.
|
||||
- Document current behavior. Architecture, user, and internal docs must not
|
||||
describe planned features as implemented behavior.
|
||||
|
||||
## Architectural Boundaries
|
||||
|
||||
Core transcript data belongs in `internal/model` and public artifact contracts
|
||||
belong in `schema`. Conversion from internal merged data to public JSON shapes
|
||||
belongs at the artifact boundary, not inside CLI code or transformation
|
||||
packages.
|
||||
|
||||
Pipeline orchestration belongs in `internal/pipeline`. It resolves registered
|
||||
modules, validates preprocessing state transitions, executes stages in order,
|
||||
collects report events, converts the final transcript, and writes optional
|
||||
reports. Built-in adapters and modules are registered from `internal/builtin`.
|
||||
|
||||
CLI code in `internal/cli` should parse flags, build validated config values,
|
||||
and delegate. `merge` delegates to `pipeline.Run`; `trim`, `normalize`, and
|
||||
`render` perform artifact-level orchestration and delegate deterministic
|
||||
parsing, validation, and transformation work to their internal packages.
|
||||
|
||||
Config loading and validation belongs in `internal/config`. Filesystem reads and
|
||||
writes are adapter concerns and should not spread into pure transformation
|
||||
helpers. Existing built-in modules that load configured YAML files must keep
|
||||
that I/O narrow and explicit.
|
||||
|
||||
Reports belong in `internal/report`. Modules and commands should emit concise
|
||||
events for validation findings, corrections, and transformations without
|
||||
turning report messages into a duplicate output artifact.
|
||||
|
||||
Tests and samples are supporting evidence for behavior. Tests should verify
|
||||
stable contracts and edge cases; samples should remain valid examples, not
|
||||
hidden architecture dependencies.
|
||||
|
||||
## Modules or Stages
|
||||
|
||||
The merge pipeline has these implemented stages:
|
||||
|
||||
- `InputReader`: reads configured external input into raw transcripts.
|
||||
- `Preprocessor`: transforms `PreprocessState` from raw to canonical state.
|
||||
- `Merger`: combines canonical transcripts into one merged transcript.
|
||||
- `Postprocessor`: transforms or annotates the merged transcript.
|
||||
- `OutputWriter`: writes the selected output artifact.
|
||||
|
||||
Modules must keep narrow responsibilities, declare their stage through the
|
||||
interface they implement, and use explicit config values. Preprocessors must
|
||||
declare `Requires()` and `Produces()` states; the runner rejects invalid
|
||||
raw/canonical ordering before processing completes.
|
||||
|
||||
Modules run in the configured order. Order-affecting modules must run before
|
||||
`assign-ids`, and `validate-output` must see final IDs that match the selected
|
||||
schema. Accepted and rejected transformations should be deterministic and, when
|
||||
observable, recorded through report events.
|
||||
|
||||
Transformation helpers should avoid hidden global state. Shared caches, such as
|
||||
compiled JSON schemas, must be protected and must not affect output ordering.
|
||||
|
||||
## State, Inputs, and Outputs
|
||||
|
||||
seriatim is file-based. It reads JSON inputs and optional YAML rule files, then
|
||||
writes JSON transcript artifacts and optional JSON reports.
|
||||
|
||||
The implemented application has no durable database, daemon state, resume
|
||||
state, remote storage, or background job state. Runtime state is held in memory
|
||||
for the current command invocation and serialized only through requested output
|
||||
and report files.
|
||||
|
||||
Input file paths are normalized and validated during config construction.
|
||||
`merge` sorts input file paths before processing, then uses stable segment sort
|
||||
keys. `trim` preserves transcript order while renumbering retained IDs.
|
||||
`normalize` sorts by implemented deterministic keys and assigns fresh IDs.
|
||||
|
||||
## Configuration and CLI Boundaries
|
||||
|
||||
The CLI surface is an adapter over validated config structs. Cobra command code
|
||||
should stay thin: parse flags, account for flag/default precedence, call config
|
||||
constructors, and delegate.
|
||||
|
||||
Config constructors validate required paths, output parent directories, module
|
||||
lists, selected schemas, mutually exclusive trim selector options, and supported
|
||||
environment-derived settings. Module name validation is split between config
|
||||
where command-specific names are fixed and the pipeline registry where module
|
||||
composition is resolved.
|
||||
|
||||
Do not duplicate full CLI or config reference material here. Use
|
||||
[CLI reference](../cli.md) and [configuration reference](../config.md) for
|
||||
canonical user-facing details.
|
||||
|
||||
## Errors, Logging, and Diagnostics
|
||||
|
||||
Commands return errors instead of printing inside deep logic. The root command
|
||||
silences Cobra usage/error output, and `cmd/seriatim/main.go` prints one error
|
||||
to stderr and exits with status `1`.
|
||||
|
||||
Validation failures should fail fast with contextual errors. Correctable
|
||||
conditions should be deterministic and, where reports are requested, reflected
|
||||
as report events. Optional reports contain metadata and ordered events; they are
|
||||
not required for command success unless the report file itself cannot be
|
||||
written.
|
||||
|
||||
The implemented code does not use a logging subsystem. Diagnostics are returned
|
||||
as errors or written to optional report JSON. Normalize report events avoid
|
||||
embedding transcript text; keep that privacy-oriented behavior when changing
|
||||
normalize diagnostics.
|
||||
|
||||
## Testing Expectations
|
||||
|
||||
`go test ./...` is the repository-wide check. There is currently no Makefile,
|
||||
taskfile, linter config, or dedicated documentation check.
|
||||
|
||||
When changing config or CLI behavior, inspect `internal/config` and
|
||||
`internal/cli` tests. When changing pipeline composition or stage contracts,
|
||||
inspect `internal/pipeline` and `internal/builtin` tests. When changing
|
||||
correction or annotation modules, inspect the package tests for overlap,
|
||||
coalesce, danglers, backchannel, filler, and autocorrect behavior.
|
||||
|
||||
When changing artifact-level commands, inspect `internal/trim`,
|
||||
`internal/normalize`, `internal/render`, and their CLI tests. When changing
|
||||
public output shape or schema validation, inspect `schema` and
|
||||
`internal/artifact` tests. Report and diagnostic changes should be covered
|
||||
through the command or package tests that emit the affected events.
|
||||
|
||||
## Dependency Policy
|
||||
|
||||
Prefer the Go standard library for parsing, data transformation, concurrency
|
||||
primitives, filesystem work, and testing wherever it is reasonable.
|
||||
|
||||
Third-party dependencies must be narrow, justified, and preferably de facto
|
||||
standard for their purpose. Existing examples include Cobra for CLI structure,
|
||||
`gopkg.in/yaml.v3` for YAML files, and `jsonschema/v6` for validating embedded
|
||||
public JSON schemas. Avoid broad framework dependencies for behavior that is
|
||||
already simple and local.
|
||||
|
||||
## Documentation Expectations
|
||||
|
||||
Architecture docs must stay aligned with [documentation policy](documentation.md).
|
||||
Current-behavior docs must not become aspirational. If code and docs disagree,
|
||||
fix the inaccurate current-behavior doc or put planned work under
|
||||
`docs/roadmap/`.
|
||||
|
||||
Prefer links to canonical docs instead of repeating full CLI, config, schema, or
|
||||
operations reference material. Keep examples real, tested where practical, and
|
||||
free of secrets or private transcript data.
|
||||
|
||||
## Architectural Invariants
|
||||
|
||||
- Keep core/domain logic separate from CLI, config, filesystem, reporting, and
|
||||
other adapter concerns.
|
||||
- Centralize default configuration values as constants defined in internal/config/config.go.
|
||||
- Keep modules narrowly scoped, explicitly configured, and composable by
|
||||
registry name.
|
||||
- Preserve deterministic ordering, final segment ID assignment, and schema
|
||||
validation before output acceptance.
|
||||
- Keep `trim`, `normalize`, and `render` artifact-level; do not run merge
|
||||
modules from those commands.
|
||||
- Keep public output schemas validated through `schema`.
|
||||
- Keep optional reports ordered, concise, and diagnostic.
|
||||
- Avoid broad dependencies without a concrete maintainability benefit.
|
||||
- Do not document unimplemented behavior outside `docs/roadmap/`.
|
||||
|
||||
## Non-Goals
|
||||
|
||||
The implemented application does not perform transcription, audio diarization,
|
||||
speaker inference from audio or text, summarization, daemon operation, remote
|
||||
storage, dynamic external plugin loading, or concurrent pipeline execution.
|
||||
|
||||
The architecture policy is not a package-by-package reference, CLI manual,
|
||||
config reference, schema reference, or roadmap.
|
||||
113
docs/policy/development.md
Normal file
113
docs/policy/development.md
Normal file
@@ -0,0 +1,113 @@
|
||||
# Development Policy
|
||||
|
||||
## Purpose
|
||||
|
||||
This document defines contributor workflow for maintainers and coding agents.
|
||||
It complements [architecture policy](architecture.md) and
|
||||
[documentation policy](documentation.md).
|
||||
|
||||
## Repository layout
|
||||
|
||||
- `cmd/seriatim/`: process entrypoint.
|
||||
- `internal/cli/`: Cobra commands and flag wiring.
|
||||
- `internal/config/`: option normalization and validation.
|
||||
- `internal/pipeline/`: orchestration interfaces, registry, runner.
|
||||
- `internal/builtin/`: implemented input/pre/post/output modules and merger.
|
||||
- `internal/artifact/`: conversion from internal merged model to public shapes.
|
||||
- `internal/trim/`: artifact-level trim logic.
|
||||
- `internal/normalize/`: artifact-level normalize parsing/building.
|
||||
- `internal/render/`: artifact-level rendering and renderer registry.
|
||||
- `internal/*` domain packages: overlap, coalesce, danglers, filler,
|
||||
backchannel, speaker, autocorrect, report, model.
|
||||
- `schema/`: public structs plus embedded JSON Schemas and validation.
|
||||
- `docs/`: policy, user docs, roadmap, and internal docs.
|
||||
|
||||
## Local checks
|
||||
|
||||
Primary repository check:
|
||||
|
||||
```sh
|
||||
go test ./...
|
||||
```
|
||||
|
||||
Useful manual checks for CLI-facing changes:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim --help
|
||||
go run ./cmd/seriatim merge --help
|
||||
go run ./cmd/seriatim trim --help
|
||||
go run ./cmd/seriatim normalize --help
|
||||
go run ./cmd/seriatim render --help
|
||||
```
|
||||
|
||||
Current toolchain note:
|
||||
|
||||
- There is no Makefile.
|
||||
- There is no taskfile.
|
||||
- There is no committed linter configuration.
|
||||
- There is no automated documentation checker.
|
||||
|
||||
## Coding conventions
|
||||
|
||||
- Keep core behavior deterministic for identical inputs/config/version.
|
||||
- Keep CLI command functions thin: parse flags, construct config, delegate.
|
||||
- Keep validation in `internal/config` and package-specific validators.
|
||||
- Return errors from deep logic; do not print inside internal packages.
|
||||
- Preserve clear package boundaries between adapters and domain transforms.
|
||||
- Define configuration defaults as constants in internal/config/config.go.
|
||||
|
||||
## Dependency policy
|
||||
|
||||
Prefer the Go standard library first.
|
||||
|
||||
Third-party dependencies should stay narrow and justified. Current direct
|
||||
runtime dependencies are:
|
||||
|
||||
- `github.com/spf13/cobra` for CLI structure.
|
||||
- `gopkg.in/yaml.v3` for YAML rule files.
|
||||
- `github.com/santhosh-tekuri/jsonschema/v6` for public schema validation.
|
||||
|
||||
## Adding CLI flags
|
||||
|
||||
1. Add the flag in the relevant `internal/cli/*.go` command.
|
||||
2. Thread the raw value through `config.*Options`.
|
||||
3. Add normalization/validation in `internal/config/config.go`.
|
||||
4. Update or add CLI/config tests.
|
||||
5. Update canonical docs (`docs/cli.md`, `docs/config.md`) if user-visible.
|
||||
|
||||
## Adding config fields or environment variables
|
||||
|
||||
1. Add field(s) to the relevant config struct(s).
|
||||
2. Parse and validate in `internal/config/config.go`.
|
||||
3. Add tests in `internal/config/config_test.go`.
|
||||
4. Thread validated values into consuming modules.
|
||||
5. Update `docs/config.md` and related docs.
|
||||
|
||||
## Adding modules or pipeline behavior
|
||||
|
||||
1. Implement the module in the appropriate package (often `internal/builtin`).
|
||||
2. Expose a stable module name via `Name()`.
|
||||
3. Register it in `internal/builtin/registry.go`.
|
||||
4. Ensure preprocessing modules declare correct `Requires()`/`Produces()`
|
||||
states.
|
||||
5. Add/adjust tests in module packages and `internal/cli/merge_test.go`.
|
||||
6. Document internal behavior changes in `docs/internal/`.
|
||||
|
||||
## Schema and artifact changes
|
||||
|
||||
1. Update public structs and validation logic in `schema/`.
|
||||
2. Update embedded JSON Schema files (`schema/*.schema.json`) if contract
|
||||
changes.
|
||||
3. Update conversion behavior in `internal/artifact`, `internal/trim`, and/or
|
||||
`internal/normalize` as needed.
|
||||
4. Add tests in `schema/`, `internal/artifact/`, `internal/trim/`,
|
||||
`internal/normalize/`, and CLI tests.
|
||||
5. Update user and internal docs that reference output contracts.
|
||||
|
||||
## Documentation expectations
|
||||
|
||||
- Outside `docs/roadmap/`, document only implemented behavior.
|
||||
- Keep canonical homes: CLI in `docs/cli.md`, config in `docs/config.md`,
|
||||
operations in `docs/operations.md`, troubleshooting in
|
||||
`docs/troubleshooting.md`, internals in `docs/internal/`.
|
||||
- When behavior changes, update docs in the same change.
|
||||
356
docs/policy/documentation.md
Normal file
356
docs/policy/documentation.md
Normal file
@@ -0,0 +1,356 @@
|
||||
# Go Project Documentation Policy
|
||||
|
||||
## Purpose
|
||||
|
||||
Project documentation must help four audiences:
|
||||
|
||||
1. users who need to run the application;
|
||||
2. administrators/operators who need to configure and operate it;
|
||||
3. developers who need to understand and change it safely;
|
||||
4. LLM coding agents that need clear scope, boundaries, and invariants.
|
||||
|
||||
Docs should be accurate, concise, task-oriented, and organized by audience. Prefer links to canonical docs over repetition.
|
||||
|
||||
## Core Rules
|
||||
|
||||
### 1. Keep docs concise
|
||||
|
||||
Each document should cover a defined scope and only the essentials for that scope.
|
||||
|
||||
Avoid:
|
||||
- long background explanations;
|
||||
- repeated reference material;
|
||||
- implementation detail in user-facing docs;
|
||||
- aspirational language outside roadmap docs;
|
||||
- verbose examples where one minimal example is clearer.
|
||||
|
||||
### 2. Document only implemented behavior outside roadmap files
|
||||
|
||||
Unimplemented, planned, aspirational, experimental, or future work may be described only under:
|
||||
|
||||
- `docs/roadmap/`
|
||||
|
||||
No other documentation file, including `README.md`, should describe code, features, modules, stages, commands, config fields, or behaviors that do not currently exist.
|
||||
|
||||
If a feature is partial, non-roadmap docs may describe only the implemented portion and its current boundary.
|
||||
|
||||
### 3. Use canonical homes
|
||||
|
||||
Each type of information should have one canonical location.
|
||||
|
||||
Canonical homes:
|
||||
|
||||
- project purpose and quickstart: `README.md`
|
||||
- development principles: `docs/policy/architecture.md`
|
||||
- configuration reference: `docs/config.md`
|
||||
- CLI reference: `docs/cli.md`
|
||||
- operations and recovery: `docs/operations.md`
|
||||
- troubleshooting: `docs/troubleshooting.md`
|
||||
- implemented internals: `docs/internal/`
|
||||
- future work: `docs/roadmap/`
|
||||
- contributor workflow: `docs/policy/development.md`
|
||||
- copyable examples: `examples/`
|
||||
|
||||
Other files should summarize briefly and link to the canonical source.
|
||||
|
||||
### 4. Keep examples real
|
||||
|
||||
Examples should be valid, maintained, and free of secrets.
|
||||
|
||||
Where practical:
|
||||
- example configs should load successfully;
|
||||
- example commands should match real CLI syntax;
|
||||
- important examples should be covered by tests.
|
||||
|
||||
## Documentation Profiles
|
||||
|
||||
All projects require:
|
||||
|
||||
- `README.md`
|
||||
- `docs/policy/architecture.md`
|
||||
|
||||
Additional docs depend on the project.
|
||||
|
||||
### Small library
|
||||
|
||||
Recommended:
|
||||
- `docs/policy/development.md`, if contributor conventions are non-obvious
|
||||
|
||||
### Simple CLI
|
||||
|
||||
Required:
|
||||
- `docs/cli.md`
|
||||
|
||||
Recommended:
|
||||
- `docs/policy/development.md`
|
||||
|
||||
### Config-driven CLI
|
||||
|
||||
Required:
|
||||
- `docs/cli.md`
|
||||
- `docs/config.md`
|
||||
|
||||
Recommended:
|
||||
- `examples/`
|
||||
- `docs/policy/development.md`
|
||||
|
||||
### Stateful or operator-facing application
|
||||
|
||||
Required:
|
||||
- `docs/cli.md`, if CLI-based
|
||||
- `docs/config.md`, if config-driven
|
||||
- `docs/operations.md`
|
||||
|
||||
Recommended:
|
||||
- `docs/troubleshooting.md`
|
||||
- `examples/`
|
||||
- `docs/policy/development.md`
|
||||
|
||||
### Modular, staged, service-oriented, or orchestration application
|
||||
|
||||
Required:
|
||||
- `docs/cli.md`, if CLI-based
|
||||
- `docs/config.md`, if config-driven
|
||||
- `docs/operations.md`
|
||||
- `docs/internal/`
|
||||
- `docs/policy/development.md`
|
||||
|
||||
Recommended:
|
||||
- `docs/troubleshooting.md`
|
||||
- validated examples under `examples/`
|
||||
|
||||
## Required Documents
|
||||
|
||||
### README.md
|
||||
|
||||
**Audience:** users, administrators, operators
|
||||
|
||||
The README is the outward-facing project orientation page.
|
||||
|
||||
It should include, in order:
|
||||
|
||||
1. concise description;
|
||||
2. elevator pitch;
|
||||
3. shortest useful command or usage example;
|
||||
4. links to targeted docs.
|
||||
|
||||
The README should be short. It is not a manual.
|
||||
|
||||
The “shortest useful command” means the simplest command that performs the project’s core use case. (It does not mean `app --help`.)
|
||||
|
||||
### docs/policy/architecture.md
|
||||
|
||||
**Audience:** developers, LLM coding agents
|
||||
|
||||
`docs/policy/architecture.md` is required for every project.
|
||||
|
||||
It is an inward-facing development policy document. It should describe how the project is intended to be built and changed.
|
||||
|
||||
It should include:
|
||||
|
||||
- project shape;
|
||||
- core design principles;
|
||||
- package and boundary philosophy;
|
||||
- state/persistence philosophy, if applicable;
|
||||
- external integration philosophy, if applicable;
|
||||
- error-handling and logging principles;
|
||||
- testing expectations;
|
||||
- documentation expectations;
|
||||
- architectural invariants;
|
||||
- explicit non-goals, if useful.
|
||||
|
||||
For small projects, this file may be brief. It may simply state that the project is intentionally narrow, monolithic, and dependency-light.
|
||||
|
||||
### docs/policy/development.md
|
||||
|
||||
**Audience:** developers, LLM coding agents
|
||||
|
||||
Required for projects maintained by humans and LLM coding agents.
|
||||
|
||||
It should include:
|
||||
|
||||
- repository layout;
|
||||
- build/test commands;
|
||||
- coding conventions;
|
||||
- dependency policy;
|
||||
- how to add config fields;
|
||||
- how to add CLI flags;
|
||||
- how to add stages/modules/adapters, if applicable;
|
||||
- how to update examples;
|
||||
- documentation update expectations.
|
||||
|
||||
### docs/config.md
|
||||
|
||||
**Audience:** administrators, operators, advanced users
|
||||
|
||||
Required for applications with configuration files.
|
||||
|
||||
It should include, in order:
|
||||
|
||||
1. config file locations and discovery precedence;
|
||||
2. minimal working config;
|
||||
3. production-oriented config;
|
||||
4. full configuration reference;
|
||||
5. secrets handling, if applicable;
|
||||
6. links to maintained examples.
|
||||
|
||||
The full configuration reference should be canonical.
|
||||
|
||||
### docs/cli.md
|
||||
|
||||
**Audience:** users, administrators, operators
|
||||
|
||||
Required for CLI applications.
|
||||
|
||||
It should include, in order:
|
||||
|
||||
1. shortest useful command;
|
||||
2. command overview;
|
||||
3. complete flag reference;
|
||||
4. common workflows;
|
||||
5. diagnostic or recovery commands, if applicable.
|
||||
|
||||
Explain when commands are useful, not just their syntax.
|
||||
|
||||
### docs/operations.md
|
||||
|
||||
**Audience:** administrators, operators
|
||||
|
||||
Required for applications that maintain state, support resume behavior, run multiple stages, write durable artifacts, use remote storage, or require recovery procedures.
|
||||
|
||||
It should cover:
|
||||
|
||||
- normal workflow;
|
||||
- filesystem layout;
|
||||
- remote storage layout, if applicable;
|
||||
- logs and manifests;
|
||||
- resume/retry behavior;
|
||||
- cleanup behavior;
|
||||
- archive/backup behavior;
|
||||
- safe recovery procedures;
|
||||
- operational caveats.
|
||||
|
||||
### docs/troubleshooting.md
|
||||
|
||||
**Audience:** administrators, operators
|
||||
|
||||
Recommended once recurring failure modes exist.
|
||||
|
||||
Each entry should include:
|
||||
|
||||
- symptom;
|
||||
- likely cause;
|
||||
- diagnostic command or inspection step;
|
||||
- safe fix;
|
||||
- relevant links.
|
||||
|
||||
### docs/internal/
|
||||
|
||||
**Audience:** developers, LLM coding agents
|
||||
|
||||
Required for modular, staged, service-oriented, or orchestration projects.
|
||||
|
||||
This directory describes implemented internal components. It is not the roadmap.
|
||||
|
||||
Use one file per major component where useful.
|
||||
|
||||
Each component doc should include:
|
||||
|
||||
1. purpose;
|
||||
2. inputs and outputs;
|
||||
3. boundaries;
|
||||
4. config fields used;
|
||||
5. external adapters used;
|
||||
6. state or manifest behavior, if applicable;
|
||||
7. skip/resume behavior, if applicable;
|
||||
8. failure behavior;
|
||||
9. tests to inspect before changing;
|
||||
10. architectural invariants.
|
||||
|
||||
### docs/roadmap/
|
||||
|
||||
**Audience:** maintainers, developers, LLM coding agents
|
||||
|
||||
This is the only place for planned, future, aspirational, experimental, or unimplemented work.
|
||||
|
||||
Roadmap docs should clearly distinguish:
|
||||
|
||||
- proposed work;
|
||||
- accepted plans;
|
||||
- deferred ideas;
|
||||
- rejected ideas;
|
||||
- implementation prompts or task breakdowns, if useful.
|
||||
|
||||
Roadmap docs should not be confused with current behavior.
|
||||
|
||||
### docs/integrations/
|
||||
|
||||
**Audience:** developers, LLM coding agents
|
||||
|
||||
Required for projects that depend on external CLIs, APIs, services, protocols, or file formats where the integration contract is important to maintain.
|
||||
|
||||
This directory contains concise, versioned reference notes for external integration contracts. It should document only the parts of the external system that this project actually uses.
|
||||
|
||||
Use one file per integration where useful.
|
||||
|
||||
## Examples Directory
|
||||
|
||||
Projects with non-trivial configuration or workflows should include `examples/`.
|
||||
|
||||
Useful examples include:
|
||||
|
||||
- minimal working config;
|
||||
- production-oriented config;
|
||||
- full annotated config;
|
||||
- local development config;
|
||||
- remote/object-storage config;
|
||||
- minimal session/input file.
|
||||
|
||||
Examples should be valid, maintained, tested when practical, and linked from relevant docs.
|
||||
|
||||
## Security and Privacy
|
||||
|
||||
Docs and examples must not include:
|
||||
|
||||
- real API keys;
|
||||
- tokens;
|
||||
- passwords;
|
||||
- private keys;
|
||||
- private environment dumps;
|
||||
- sensitive user data;
|
||||
- raw private transcripts;
|
||||
- private infrastructure details unless intentionally public.
|
||||
|
||||
Document secret-handling mechanisms, not actual secret values.
|
||||
|
||||
## Maintenance Rules
|
||||
|
||||
When docs change, verify the affected behavior.
|
||||
|
||||
Where practical:
|
||||
|
||||
- load example config files in tests;
|
||||
- test CLI examples or command parser behavior;
|
||||
- validate documented flags against real flags;
|
||||
- remove stale references;
|
||||
- update links after renames;
|
||||
- keep roadmap content out of non-roadmap docs.
|
||||
|
||||
If documentation and code disagree, fix the documentation and/or open a roadmap item; do not leave aspirational behavior in current-behavior docs.
|
||||
|
||||
Documentation is complete only when it matches the current code.
|
||||
|
||||
## Documentation Change Checklist
|
||||
|
||||
Before merging documentation changes, verify:
|
||||
|
||||
- README is concise and orientation-focused.
|
||||
- `docs/policy/architecture.md` describes development principles.
|
||||
- Future work appears only under `docs/roadmap/`.
|
||||
- User-facing docs avoid unnecessary internals.
|
||||
- Developer-facing docs preserve boundaries and invariants.
|
||||
- Config examples match the schema.
|
||||
- CLI examples match real commands and flags.
|
||||
- Defaults appear in the canonical config reference.
|
||||
- No secrets or private data are included.
|
||||
- Links are accurate.
|
||||
116
docs/troubleshooting.md
Normal file
116
docs/troubleshooting.md
Normal file
@@ -0,0 +1,116 @@
|
||||
# Troubleshooting
|
||||
|
||||
Each entry includes symptom, likely cause, inspection step, and safe fix.
|
||||
|
||||
## Missing required flags
|
||||
|
||||
- Symptom: command fails with messages like `--input-file is required`, `--output-file is required`, `--format is required`, or `exactly one of --keep or --remove is required`.
|
||||
- Likely cause: one or more required flags were omitted.
|
||||
- Inspection: run help for the failing command:
|
||||
- `go run ./cmd/seriatim merge --help`
|
||||
- `go run ./cmd/seriatim trim --help`
|
||||
- `go run ./cmd/seriatim normalize --help`
|
||||
- `go run ./cmd/seriatim render --help`
|
||||
- Safe fix: provide all required flags; for `trim`, provide exactly one selector mode (`--keep` or `--remove`).
|
||||
|
||||
## Invalid output or report path
|
||||
|
||||
- Symptom: errors like `--output-file parent directory ...` or `--report-file parent directory ...`.
|
||||
- Likely cause: parent directory does not exist, is not a directory, or the target path is unusable.
|
||||
- Inspection: verify parent path and permissions:
|
||||
- `dirname <path>`
|
||||
- `ls -ld <parent-dir>`
|
||||
- Safe fix: create or fix the parent directory and rerun. Use a file path (not a directory path) for output/report targets.
|
||||
|
||||
## Invalid merge input JSON
|
||||
|
||||
- Symptom: merge fails with messages like `parse input file`, `must contain top-level segments array`, `segment 0 missing numeric start`, or `segment 0 words must be an array`.
|
||||
- Likely cause: malformed JSON or unsupported/missing fields in a merge input file.
|
||||
- Inspection: validate JSON and required segment fields (`start`, `end`, `text`):
|
||||
- `jq . <input-file>`
|
||||
- Safe fix: correct JSON structure and segment/word field types, then rerun `merge`.
|
||||
|
||||
## Invalid normalize input shape
|
||||
|
||||
- Symptom: normalize fails with messages like `must contain a "segments" field`, `"segments" must be an array`, or `top-level object with "segments" or a top-level segment array`.
|
||||
- Likely cause: normalize input is neither supported object-with-segments nor top-level segment array.
|
||||
- Inspection: inspect top-level JSON shape:
|
||||
- `jq 'type' <input-file>`
|
||||
- `jq 'keys' <input-file>` (for object input)
|
||||
- Safe fix: reshape input into one supported form and rerun `normalize`.
|
||||
|
||||
## Invalid render input artifact
|
||||
|
||||
- Symptom: render fails with messages like `input JSON is malformed` or `input JSON is not a valid seriatim output artifact`.
|
||||
- Likely cause: input is malformed JSON or not one of the supported seriatim output schemas.
|
||||
- Inspection:
|
||||
- `jq . <input-file>`
|
||||
- compare input shape against `schema/minimal-output.schema.json`, `schema/intermediate-output.schema.json`, and `schema/full-output.schema.json`
|
||||
- Safe fix: render only a valid existing seriatim artifact (`seriatim-minimal`, `seriatim-intermediate`, or `seriatim-full`).
|
||||
|
||||
## Invalid speaker map or autocorrect YAML
|
||||
|
||||
- Symptom: merge fails with errors such as `must contain at least one match rule`, `must include speaker`, `must include target`, or duplicate match/speaker validation failures.
|
||||
- Likely cause: YAML rule file structure/content does not match expected contract.
|
||||
- Inspection: check YAML validity and required top-level keys:
|
||||
- `speakers.yml` requires top-level `match` rules
|
||||
- `autocorrect.yml` requires top-level `autocorrect` rules
|
||||
- Safe fix: correct YAML structure and rule content, then rerun `merge`.
|
||||
|
||||
## Unknown module names
|
||||
|
||||
- Symptom: errors like `unknown input reader`, `unknown preprocessing module`, `unknown postprocessing module`, or `unknown output module`.
|
||||
- Likely cause: module name typo or unsupported module in flag lists.
|
||||
- Inspection: compare provided module names against defaults in CLI help and config docs.
|
||||
- Safe fix: use implemented module names only or remove unsupported modules from comma-separated lists.
|
||||
|
||||
## Invalid format or schema values
|
||||
|
||||
- Symptom:
|
||||
- render: `--format must be "markdown"`
|
||||
- merge/normalize/trim: `--output-schema must be one of ...`
|
||||
- Likely cause: unsupported `--format` or `--output-schema` value.
|
||||
- Inspection:
|
||||
- command flags
|
||||
- `echo "$SERIATIM_OUTPUT_SCHEMA"` (for merge/normalize defaults)
|
||||
- Safe fix:
|
||||
- render: use `--format markdown`
|
||||
- output schema: use `seriatim-minimal`, `seriatim-intermediate`, or `seriatim-full`
|
||||
|
||||
## Invalid trim selector
|
||||
|
||||
- Symptom: trim fails with messages like `invalid selector ... malformed element`, `segment ID must be positive`, or descending-range errors.
|
||||
- Likely cause: selector syntax is invalid.
|
||||
- Inspection: verify selector format:
|
||||
- single ID: `7`
|
||||
- range: `1-10`
|
||||
- list: `1-10,15,20-25`
|
||||
- Safe fix: correct selector syntax and rerun `trim`.
|
||||
|
||||
## Artifact or schema validation failures
|
||||
|
||||
- Symptom: errors such as `validate-output: ...`, `input JSON is not a valid seriatim output artifact`, or related schema-validation errors.
|
||||
- Likely cause:
|
||||
- merge module order/config produced an invalid output artifact, or
|
||||
- trim/render input is not a valid seriatim output artifact.
|
||||
- Inspection:
|
||||
- for merge: inspect customized module ordering flags
|
||||
- for trim/render: validate input against schema files in `schema/`
|
||||
- Safe fix:
|
||||
- merge: restore a valid postprocessing order ending with assigned IDs before output validation
|
||||
- trim/render: provide a valid seriatim artifact as input
|
||||
|
||||
## Report write failure
|
||||
|
||||
- Symptom: errors like `write --report-file ...` or file-create failures when report writing is requested.
|
||||
- Likely cause: report path is not writable or points to an invalid target.
|
||||
- Inspection:
|
||||
- `ls -ld <report-parent-dir>`
|
||||
- verify `--report-file` is a file path, not a directory
|
||||
- Safe fix: choose a writable file path under an existing directory and rerun.
|
||||
|
||||
## Related docs
|
||||
|
||||
- CLI reference: [cli.md](cli.md)
|
||||
- Configuration reference: [config.md](config.md)
|
||||
- Operations guide: [operations.md](operations.md)
|
||||
79
examples/README.md
Normal file
79
examples/README.md
Normal file
@@ -0,0 +1,79 @@
|
||||
# Examples
|
||||
|
||||
These are small synthetic, copyable example assets for the implemented CLI
|
||||
commands. This directory is the canonical examples home for documentation.
|
||||
|
||||
## Merge example
|
||||
|
||||
Inputs:
|
||||
|
||||
- `minimal-merge/input-alice.json`
|
||||
- `minimal-merge/input-bob.json`
|
||||
- `minimal-merge/speakers.yml`
|
||||
|
||||
Run:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim merge \
|
||||
--input-file examples/minimal-merge/input-alice.json \
|
||||
--input-file examples/minimal-merge/input-bob.json \
|
||||
--speakers examples/minimal-merge/speakers.yml \
|
||||
--output-file /tmp/seriatim-example-merge.json
|
||||
```
|
||||
|
||||
## Normalize examples
|
||||
|
||||
Object-with-segments input:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim normalize \
|
||||
--input-file examples/normalize/object-with-segments.json \
|
||||
--output-file /tmp/seriatim-example-normalize-object.json
|
||||
```
|
||||
|
||||
Bare-array input:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim normalize \
|
||||
--input-file examples/normalize/bare-segments-array.json \
|
||||
--output-file /tmp/seriatim-example-normalize-array.json
|
||||
```
|
||||
|
||||
## Trim example
|
||||
|
||||
Input artifact:
|
||||
|
||||
- `trim/input-full.json`
|
||||
|
||||
Run:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim trim \
|
||||
--input-file examples/trim/input-full.json \
|
||||
--output-file /tmp/seriatim-example-trim.json \
|
||||
--keep "1-2"
|
||||
```
|
||||
|
||||
## Render example
|
||||
|
||||
Input artifact:
|
||||
|
||||
- `render/input-intermediate.json`
|
||||
|
||||
Expected Markdown output shape:
|
||||
|
||||
- `render/output-markdown.md`
|
||||
|
||||
Run:
|
||||
|
||||
```sh
|
||||
go run ./cmd/seriatim render \
|
||||
--input-file examples/render/input-intermediate.json \
|
||||
--output-file /tmp/seriatim-example-render.md \
|
||||
--format markdown
|
||||
```
|
||||
|
||||
## YAML rule examples
|
||||
|
||||
- `speakers.yml`
|
||||
- `autocorrect.yml`
|
||||
8
examples/autocorrect.yml
Normal file
8
examples/autocorrect.yml
Normal file
@@ -0,0 +1,8 @@
|
||||
autocorrect:
|
||||
- target: "General Kenobi"
|
||||
match:
|
||||
- "General Kenobi."
|
||||
|
||||
- target: "Okay"
|
||||
match:
|
||||
- "Okay."
|
||||
14
examples/minimal-merge/input-alice.json
Normal file
14
examples/minimal-merge/input-alice.json
Normal file
@@ -0,0 +1,14 @@
|
||||
{
|
||||
"segments": [
|
||||
{
|
||||
"start": 0.0,
|
||||
"end": 1.2,
|
||||
"text": " Hello there. "
|
||||
},
|
||||
{
|
||||
"start": 2.6,
|
||||
"end": 3.1,
|
||||
"text": "Okay."
|
||||
}
|
||||
]
|
||||
}
|
||||
9
examples/minimal-merge/input-bob.json
Normal file
9
examples/minimal-merge/input-bob.json
Normal file
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"segments": [
|
||||
{
|
||||
"start": 1.3,
|
||||
"end": 2.4,
|
||||
"text": "General Kenobi."
|
||||
}
|
||||
]
|
||||
}
|
||||
8
examples/minimal-merge/speakers.yml
Normal file
8
examples/minimal-merge/speakers.yml
Normal file
@@ -0,0 +1,8 @@
|
||||
match:
|
||||
- speaker: "Alice Example"
|
||||
match:
|
||||
- "alice"
|
||||
|
||||
- speaker: "Bob Example"
|
||||
match:
|
||||
- "bob"
|
||||
13
examples/normalize/bare-segments-array.json
Normal file
13
examples/normalize/bare-segments-array.json
Normal file
@@ -0,0 +1,13 @@
|
||||
[
|
||||
{
|
||||
"start": 2.5,
|
||||
"end": 3.0,
|
||||
"speaker": "Bob",
|
||||
"text": "later"
|
||||
},
|
||||
{
|
||||
"end": 2.0,
|
||||
"speaker": "",
|
||||
"text": "no start uses end"
|
||||
}
|
||||
]
|
||||
19
examples/normalize/object-with-segments.json
Normal file
19
examples/normalize/object-with-segments.json
Normal file
@@ -0,0 +1,19 @@
|
||||
{
|
||||
"segments": [
|
||||
{
|
||||
"id": 7,
|
||||
"start": 2.0,
|
||||
"end": 2.5,
|
||||
"speaker": "Bob",
|
||||
"text": "second"
|
||||
},
|
||||
{
|
||||
"id": 1,
|
||||
"start": 1.0,
|
||||
"end": 1.3,
|
||||
"speaker": "Alice",
|
||||
"text": "first",
|
||||
"categories": ["backchannel"]
|
||||
}
|
||||
]
|
||||
}
|
||||
33
examples/render/input-intermediate.json
Normal file
33
examples/render/input-intermediate.json
Normal file
@@ -0,0 +1,33 @@
|
||||
{
|
||||
"metadata": {
|
||||
"application": "seriatim",
|
||||
"version": "v-test",
|
||||
"output_schema": "seriatim-intermediate"
|
||||
},
|
||||
"segments": [
|
||||
{
|
||||
"id": 1,
|
||||
"start": 1,
|
||||
"end": 4,
|
||||
"speaker": "Eric Rakestraw",
|
||||
"text": "Hello there."
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"start": 5,
|
||||
"end": 8,
|
||||
"speaker": "Mike Brown",
|
||||
"text": "Welcome back, everyone."
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"start": 9,
|
||||
"end": 10,
|
||||
"speaker": "Eric Rakestraw",
|
||||
"text": "Yeah.",
|
||||
"categories": [
|
||||
"backchannel"
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
7
examples/render/output-markdown.md
Normal file
7
examples/render/output-markdown.md
Normal file
@@ -0,0 +1,7 @@
|
||||
# Transcript
|
||||
|
||||
[00:00:01–00:00:04] **Eric Rakestraw:** Hello there.
|
||||
|
||||
[00:00:05–00:00:08] **Mike Brown:** Welcome back, everyone.
|
||||
|
||||
[00:00:09–00:00:10] **Eric Rakestraw:** *Yeah.*
|
||||
8
examples/speakers.yml
Normal file
8
examples/speakers.yml
Normal file
@@ -0,0 +1,8 @@
|
||||
match:
|
||||
- speaker: "Alice Example"
|
||||
match:
|
||||
- "alice"
|
||||
|
||||
- speaker: "Bob Example"
|
||||
match:
|
||||
- "bob"
|
||||
64
examples/trim/input-full.json
Normal file
64
examples/trim/input-full.json
Normal file
@@ -0,0 +1,64 @@
|
||||
{
|
||||
"metadata": {
|
||||
"application": "seriatim",
|
||||
"version": "dev",
|
||||
"input_reader": "json-files",
|
||||
"input_files": [
|
||||
"examples/minimal-merge/input-alice.json",
|
||||
"examples/minimal-merge/input-bob.json"
|
||||
],
|
||||
"preprocessing_modules": [
|
||||
"validate-raw",
|
||||
"normalize-speakers",
|
||||
"trim-text"
|
||||
],
|
||||
"postprocessing_modules": [
|
||||
"detect-overlaps",
|
||||
"resolve-overlaps",
|
||||
"backchannel",
|
||||
"filler",
|
||||
"resolve-danglers",
|
||||
"coalesce",
|
||||
"detect-overlaps",
|
||||
"autocorrect",
|
||||
"assign-ids",
|
||||
"validate-output"
|
||||
],
|
||||
"output_modules": [
|
||||
"json"
|
||||
]
|
||||
},
|
||||
"segments": [
|
||||
{
|
||||
"id": 1,
|
||||
"source": "examples/minimal-merge/input-alice.json",
|
||||
"source_segment_index": 0,
|
||||
"speaker": "Alice Example",
|
||||
"start": 0,
|
||||
"end": 1.2,
|
||||
"text": "Hello there."
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"source": "examples/minimal-merge/input-bob.json",
|
||||
"source_segment_index": 0,
|
||||
"speaker": "Bob Example",
|
||||
"start": 1.3,
|
||||
"end": 2.4,
|
||||
"text": "General Kenobi."
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"source": "examples/minimal-merge/input-alice.json",
|
||||
"source_segment_index": 1,
|
||||
"speaker": "Alice Example",
|
||||
"start": 2.6,
|
||||
"end": 3.1,
|
||||
"text": "Okay.",
|
||||
"categories": [
|
||||
"backchannel"
|
||||
]
|
||||
}
|
||||
],
|
||||
"overlap_groups": []
|
||||
}
|
||||
178
internal/artifact/output_artifact.go
Normal file
178
internal/artifact/output_artifact.go
Normal file
@@ -0,0 +1,178 @@
|
||||
package artifact
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/schema"
|
||||
)
|
||||
|
||||
const (
|
||||
OutputSchemaMinimal = schema.OutputSchemaMinimal
|
||||
OutputSchemaIntermediate = schema.OutputSchemaIntermediate
|
||||
OutputSchemaFull = schema.OutputSchemaFull
|
||||
)
|
||||
|
||||
// OutputArtifact stores a parsed seriatim output artifact of one supported schema.
|
||||
type OutputArtifact struct {
|
||||
Schema string
|
||||
Full *schema.Transcript
|
||||
Intermediate *schema.IntermediateTranscript
|
||||
Minimal *schema.MinimalTranscript
|
||||
}
|
||||
|
||||
// ParseOutputArtifactJSON parses and validates serialized seriatim output JSON.
|
||||
func ParseOutputArtifactJSON(data []byte) (OutputArtifact, error) {
|
||||
var decoded any
|
||||
if err := json.Unmarshal(data, &decoded); err != nil {
|
||||
return OutputArtifact{}, fmt.Errorf("input JSON is malformed: %w", err)
|
||||
}
|
||||
|
||||
var full schema.Transcript
|
||||
if err := json.Unmarshal(data, &full); err == nil {
|
||||
if err := schema.ValidateTranscript(full); err == nil {
|
||||
return OutputArtifact{
|
||||
Schema: OutputSchemaFull,
|
||||
Full: &full,
|
||||
}, nil
|
||||
}
|
||||
}
|
||||
|
||||
var intermediate schema.IntermediateTranscript
|
||||
if err := json.Unmarshal(data, &intermediate); err == nil {
|
||||
if err := schema.ValidateIntermediateTranscript(intermediate); err == nil {
|
||||
return OutputArtifact{
|
||||
Schema: OutputSchemaIntermediate,
|
||||
Intermediate: &intermediate,
|
||||
}, nil
|
||||
}
|
||||
}
|
||||
|
||||
var minimal schema.MinimalTranscript
|
||||
if err := json.Unmarshal(data, &minimal); err == nil {
|
||||
if err := schema.ValidateMinimalTranscript(minimal); err == nil {
|
||||
return OutputArtifact{
|
||||
Schema: OutputSchemaMinimal,
|
||||
Minimal: &minimal,
|
||||
}, nil
|
||||
}
|
||||
}
|
||||
|
||||
return OutputArtifact{}, fmt.Errorf("input JSON is not a valid seriatim output artifact")
|
||||
}
|
||||
|
||||
// Value returns the output payload value for serialization.
|
||||
func (artifact OutputArtifact) Value() any {
|
||||
switch artifact.Schema {
|
||||
case OutputSchemaFull:
|
||||
if artifact.Full == nil {
|
||||
return schema.Transcript{}
|
||||
}
|
||||
return *artifact.Full
|
||||
case OutputSchemaIntermediate:
|
||||
if artifact.Intermediate == nil {
|
||||
return schema.IntermediateTranscript{}
|
||||
}
|
||||
return *artifact.Intermediate
|
||||
case OutputSchemaMinimal:
|
||||
if artifact.Minimal == nil {
|
||||
return schema.MinimalTranscript{}
|
||||
}
|
||||
return *artifact.Minimal
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// SegmentCount returns the number of segments in the output artifact.
|
||||
func (artifact OutputArtifact) SegmentCount() int {
|
||||
switch artifact.Schema {
|
||||
case OutputSchemaFull:
|
||||
if artifact.Full == nil {
|
||||
return 0
|
||||
}
|
||||
return len(artifact.Full.Segments)
|
||||
case OutputSchemaIntermediate:
|
||||
if artifact.Intermediate == nil {
|
||||
return 0
|
||||
}
|
||||
return len(artifact.Intermediate.Segments)
|
||||
case OutputSchemaMinimal:
|
||||
if artifact.Minimal == nil {
|
||||
return 0
|
||||
}
|
||||
return len(artifact.Minimal.Segments)
|
||||
default:
|
||||
return 0
|
||||
}
|
||||
}
|
||||
|
||||
// Application returns output artifact metadata application name.
|
||||
func (artifact OutputArtifact) Application() string {
|
||||
switch artifact.Schema {
|
||||
case OutputSchemaFull:
|
||||
if artifact.Full == nil {
|
||||
return ""
|
||||
}
|
||||
return artifact.Full.Metadata.Application
|
||||
case OutputSchemaIntermediate:
|
||||
if artifact.Intermediate == nil {
|
||||
return ""
|
||||
}
|
||||
return artifact.Intermediate.Metadata.Application
|
||||
case OutputSchemaMinimal:
|
||||
if artifact.Minimal == nil {
|
||||
return ""
|
||||
}
|
||||
return artifact.Minimal.Metadata.Application
|
||||
default:
|
||||
return ""
|
||||
}
|
||||
}
|
||||
|
||||
// Version returns output artifact metadata version.
|
||||
func (artifact OutputArtifact) Version() string {
|
||||
switch artifact.Schema {
|
||||
case OutputSchemaFull:
|
||||
if artifact.Full == nil {
|
||||
return ""
|
||||
}
|
||||
return artifact.Full.Metadata.Version
|
||||
case OutputSchemaIntermediate:
|
||||
if artifact.Intermediate == nil {
|
||||
return ""
|
||||
}
|
||||
return artifact.Intermediate.Metadata.Version
|
||||
case OutputSchemaMinimal:
|
||||
if artifact.Minimal == nil {
|
||||
return ""
|
||||
}
|
||||
return artifact.Minimal.Metadata.Version
|
||||
default:
|
||||
return ""
|
||||
}
|
||||
}
|
||||
|
||||
// FullPayload returns the full-schema payload when present.
|
||||
func (artifact OutputArtifact) FullPayload() (*schema.Transcript, error) {
|
||||
if artifact.Full == nil {
|
||||
return nil, fmt.Errorf("full artifact payload is missing")
|
||||
}
|
||||
return artifact.Full, nil
|
||||
}
|
||||
|
||||
// IntermediatePayload returns the intermediate-schema payload when present.
|
||||
func (artifact OutputArtifact) IntermediatePayload() (*schema.IntermediateTranscript, error) {
|
||||
if artifact.Intermediate == nil {
|
||||
return nil, fmt.Errorf("intermediate artifact payload is missing")
|
||||
}
|
||||
return artifact.Intermediate, nil
|
||||
}
|
||||
|
||||
// MinimalPayload returns the minimal-schema payload when present.
|
||||
func (artifact OutputArtifact) MinimalPayload() (*schema.MinimalTranscript, error) {
|
||||
if artifact.Minimal == nil {
|
||||
return nil, fmt.Errorf("minimal artifact payload is missing")
|
||||
}
|
||||
return artifact.Minimal, nil
|
||||
}
|
||||
134
internal/artifact/output_artifact_test.go
Normal file
134
internal/artifact/output_artifact_test.go
Normal file
@@ -0,0 +1,134 @@
|
||||
package artifact
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/schema"
|
||||
)
|
||||
|
||||
func TestParseOutputArtifactJSONParsesFullIntermediateAndMinimal(t *testing.T) {
|
||||
t.Run("full", func(t *testing.T) {
|
||||
first := 0
|
||||
value := schema.Transcript{
|
||||
Metadata: schema.Metadata{
|
||||
Application: "seriatim",
|
||||
Version: "v-test",
|
||||
InputReader: "json-files",
|
||||
InputFiles: []string{"input.json"},
|
||||
PreprocessingModules: []string{"validate-raw"},
|
||||
PostprocessingModules: []string{"assign-ids", "validate-output"},
|
||||
OutputModules: []string{"json"},
|
||||
},
|
||||
Segments: []schema.Segment{
|
||||
{
|
||||
ID: 1,
|
||||
Source: "input.json",
|
||||
SourceSegmentIndex: &first,
|
||||
Speaker: "Alice",
|
||||
Start: 1,
|
||||
End: 2,
|
||||
Text: "hello",
|
||||
Categories: []string{"backchannel"},
|
||||
},
|
||||
},
|
||||
OverlapGroups: []schema.OverlapGroup{},
|
||||
}
|
||||
|
||||
parsed := mustParseOutputArtifact(t, value)
|
||||
if parsed.Schema != OutputSchemaFull {
|
||||
t.Fatalf("schema = %q, want %q", parsed.Schema, OutputSchemaFull)
|
||||
}
|
||||
if parsed.Full == nil {
|
||||
t.Fatal("expected full payload")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("intermediate", func(t *testing.T) {
|
||||
value := schema.IntermediateTranscript{
|
||||
Metadata: schema.IntermediateMetadata{
|
||||
Application: "seriatim",
|
||||
Version: "v-test",
|
||||
OutputSchema: OutputSchemaIntermediate,
|
||||
},
|
||||
Segments: []schema.IntermediateSegment{
|
||||
{ID: 1, Start: 1, End: 2, Speaker: "Alice", Text: "hello", Categories: []string{"filler"}},
|
||||
},
|
||||
}
|
||||
|
||||
parsed := mustParseOutputArtifact(t, value)
|
||||
if parsed.Schema != OutputSchemaIntermediate {
|
||||
t.Fatalf("schema = %q, want %q", parsed.Schema, OutputSchemaIntermediate)
|
||||
}
|
||||
if parsed.Intermediate == nil {
|
||||
t.Fatal("expected intermediate payload")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("minimal", func(t *testing.T) {
|
||||
value := schema.MinimalTranscript{
|
||||
Metadata: schema.MinimalMetadata{
|
||||
Application: "seriatim",
|
||||
Version: "v-test",
|
||||
OutputSchema: OutputSchemaMinimal,
|
||||
},
|
||||
Segments: []schema.MinimalSegment{
|
||||
{ID: 1, Start: 1, End: 2, Speaker: "Alice", Text: "hello"},
|
||||
},
|
||||
}
|
||||
|
||||
parsed := mustParseOutputArtifact(t, value)
|
||||
if parsed.Schema != OutputSchemaMinimal {
|
||||
t.Fatalf("schema = %q, want %q", parsed.Schema, OutputSchemaMinimal)
|
||||
}
|
||||
if parsed.Minimal == nil {
|
||||
t.Fatal("expected minimal payload")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestParseOutputArtifactJSONRejectsMalformedJSON(t *testing.T) {
|
||||
_, err := ParseOutputArtifactJSON([]byte(`{"metadata":`))
|
||||
if err == nil {
|
||||
t.Fatal("expected malformed JSON error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "input JSON is malformed") {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseOutputArtifactJSONRejectsRawWhisperXLikeInput(t *testing.T) {
|
||||
data := []byte(`{
|
||||
"segments": [
|
||||
{
|
||||
"id": 0,
|
||||
"start": 0.1,
|
||||
"end": 1.2,
|
||||
"text": "hello",
|
||||
"words": [{"word":"hello","start":0.1,"end":0.8}]
|
||||
}
|
||||
]
|
||||
}`)
|
||||
|
||||
_, err := ParseOutputArtifactJSON(data)
|
||||
if err == nil {
|
||||
t.Fatal("expected artifact validation error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "not a valid seriatim output artifact") {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func mustParseOutputArtifact(t *testing.T, value any) OutputArtifact {
|
||||
t.Helper()
|
||||
data, err := json.Marshal(value)
|
||||
if err != nil {
|
||||
t.Fatalf("marshal: %v", err)
|
||||
}
|
||||
parsed, err := ParseOutputArtifactJSON(data)
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
return parsed
|
||||
}
|
||||
@@ -2,10 +2,9 @@ package builtin
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"os"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/jsonfile"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
||||
)
|
||||
|
||||
@@ -20,15 +19,7 @@ func (jsonOutputWriter) Write(ctx context.Context, out any, rpt report.Report, c
|
||||
return nil, err
|
||||
}
|
||||
|
||||
file, err := os.Create(cfg.OutputFile)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
enc := json.NewEncoder(file)
|
||||
enc.SetIndent("", " ")
|
||||
if err := enc.Encode(out); err != nil {
|
||||
if err := jsonfile.Write(cfg.OutputFile, out); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
|
||||
31
internal/cli/flags.go
Normal file
31
internal/cli/flags.go
Normal file
@@ -0,0 +1,31 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
)
|
||||
|
||||
func addOutputFileFlag(cmd *cobra.Command, target *string) {
|
||||
cmd.Flags().StringVar(target, "output-file", "", "output transcript JSON file")
|
||||
}
|
||||
|
||||
func addReportFileFlag(cmd *cobra.Command, target *string) {
|
||||
cmd.Flags().StringVar(target, "report-file", "", "optional report JSON file")
|
||||
}
|
||||
|
||||
func addOutputModulesFlag(cmd *cobra.Command, target *string) {
|
||||
cmd.Flags().StringVar(target, "output-modules", config.DefaultOutputModules, "comma-separated output modules")
|
||||
}
|
||||
|
||||
func addMergeOutputSchemaFlag(cmd *cobra.Command, target *string) {
|
||||
cmd.Flags().StringVar(target, "output-schema", config.DefaultOutputSchema, "output JSON schema: seriatim-minimal, seriatim-intermediate (default), or seriatim-full")
|
||||
}
|
||||
|
||||
func addNormalizeOutputSchemaFlag(cmd *cobra.Command, target *string) {
|
||||
cmd.Flags().StringVar(target, "output-schema", config.DefaultOutputSchema, "output JSON schema: seriatim-minimal, seriatim-intermediate, or seriatim-full")
|
||||
}
|
||||
|
||||
func addTrimOutputSchemaFlag(cmd *cobra.Command, target *string) {
|
||||
cmd.Flags().StringVar(target, "output-schema", "", "optional output JSON schema override: seriatim-minimal, seriatim-intermediate, or seriatim-full")
|
||||
}
|
||||
@@ -31,13 +31,13 @@ func newMergeCommand() *cobra.Command {
|
||||
|
||||
flags := cmd.Flags()
|
||||
flags.StringArrayVar(&opts.InputFiles, "input-file", nil, "input transcript file; may be repeated")
|
||||
flags.StringVar(&opts.OutputFile, "output-file", "", "output transcript JSON file")
|
||||
flags.StringVar(&opts.ReportFile, "report-file", "", "optional report JSON file")
|
||||
addOutputFileFlag(cmd, &opts.OutputFile)
|
||||
addReportFileFlag(cmd, &opts.ReportFile)
|
||||
flags.StringVar(&opts.SpeakersFile, "speakers", "", "speaker map file")
|
||||
flags.StringVar(&opts.AutocorrectFile, "autocorrect", "", "autocorrect rules file")
|
||||
flags.StringVar(&opts.InputReader, "input-reader", config.DefaultInputReader, "input reader module")
|
||||
flags.StringVar(&opts.OutputModules, "output-modules", config.DefaultOutputModules, "comma-separated output modules")
|
||||
flags.StringVar(&opts.OutputSchema, "output-schema", config.DefaultOutputSchema, "output JSON schema: seriatim-minimal, seriatim-intermediate (default), or seriatim-full")
|
||||
addOutputModulesFlag(cmd, &opts.OutputModules)
|
||||
addMergeOutputSchemaFlag(cmd, &opts.OutputSchema)
|
||||
flags.StringVar(&opts.PreprocessingModules, "preprocessing-modules", config.DefaultPreprocessingModules, "comma-separated preprocessing modules")
|
||||
flags.StringVar(&opts.PostprocessingModules, "postprocessing-modules", config.DefaultPostprocessingModules, "comma-separated postprocessing modules")
|
||||
flags.StringVar(&opts.CoalesceGap, "coalesce-gap", config.DefaultCoalesceGapValue, "maximum same-speaker gap in seconds for coalesce")
|
||||
|
||||
@@ -30,10 +30,10 @@ func newNormalizeCommand() *cobra.Command {
|
||||
|
||||
flags := cmd.Flags()
|
||||
flags.StringVar(&opts.InputFile, "input-file", "", "input transcript JSON file")
|
||||
flags.StringVar(&opts.OutputFile, "output-file", "", "output transcript JSON file")
|
||||
flags.StringVar(&opts.ReportFile, "report-file", "", "optional report JSON file")
|
||||
flags.StringVar(&opts.OutputSchema, "output-schema", config.DefaultOutputSchema, "output JSON schema: seriatim-minimal, seriatim-intermediate, or seriatim-full")
|
||||
flags.StringVar(&opts.OutputModules, "output-modules", config.DefaultOutputModules, "comma-separated output modules")
|
||||
addOutputFileFlag(cmd, &opts.OutputFile)
|
||||
addReportFileFlag(cmd, &opts.ReportFile)
|
||||
addNormalizeOutputSchemaFlag(cmd, &opts.OutputSchema)
|
||||
addOutputModulesFlag(cmd, &opts.OutputModules)
|
||||
|
||||
return cmd
|
||||
}
|
||||
|
||||
39
internal/cli/render.go
Normal file
39
internal/cli/render.go
Normal file
@@ -0,0 +1,39 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/render"
|
||||
)
|
||||
|
||||
func newRenderCommand() *cobra.Command {
|
||||
opts := config.RenderOptions{
|
||||
Title: config.DefaultRenderTitle,
|
||||
IncludeTimestamps: true,
|
||||
}
|
||||
|
||||
cmd := &cobra.Command{
|
||||
Use: "render",
|
||||
Short: "Render a seriatim transcript artifact into human-readable output",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
cfg, err := config.NewRenderConfig(opts)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
return render.Run(cmd.Context(), cfg)
|
||||
},
|
||||
}
|
||||
|
||||
flags := cmd.Flags()
|
||||
flags.StringVar(&opts.InputFile, "input-file", "", "input seriatim transcript artifact JSON file")
|
||||
flags.StringVar(&opts.OutputFile, "output-file", "", "rendered output file path")
|
||||
flags.StringVar(&opts.Format, "format", "", "output format (markdown)")
|
||||
flags.StringVar(&opts.Title, "title", config.DefaultRenderTitle, "document title")
|
||||
flags.BoolVar(&opts.IncludeTimestamps, "include-timestamps", true, "include segment timestamps")
|
||||
flags.BoolVar(&opts.IncludeSegmentIDs, "include-segment-ids", false, "include segment IDs")
|
||||
flags.BoolVar(&opts.IncludeMetadata, "include-metadata", false, "include artifact metadata")
|
||||
|
||||
return cmd
|
||||
}
|
||||
274
internal/cli/render_test.go
Normal file
274
internal/cli/render_test.go
Normal file
@@ -0,0 +1,274 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
)
|
||||
|
||||
func TestRenderCommandIsRecognized(t *testing.T) {
|
||||
cmd := NewRootCommand()
|
||||
cmd.SetArgs([]string{"render", "--help"})
|
||||
if err := cmd.Execute(); err != nil {
|
||||
t.Fatalf("render command should be recognized: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRootHelpIncludesRender(t *testing.T) {
|
||||
cmd := NewRootCommand()
|
||||
var out bytes.Buffer
|
||||
cmd.SetOut(&out)
|
||||
cmd.SetErr(&out)
|
||||
cmd.SetArgs([]string{"--help"})
|
||||
if err := cmd.Execute(); err != nil {
|
||||
t.Fatalf("help failed: %v", err)
|
||||
}
|
||||
if !strings.Contains(out.String(), "render") {
|
||||
t.Fatalf("root help missing render command:\n%s", out.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestRenderEndToEndMarkdownOutput(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeJSONFile(t, dir, "input.json", `{
|
||||
"metadata": {
|
||||
"application": "seriatim",
|
||||
"version": "v-test",
|
||||
"output_schema": "seriatim-intermediate"
|
||||
},
|
||||
"segments": [
|
||||
{"id": 1, "start": 1, "end": 4, "speaker": "Eric", "text": "Hello there."},
|
||||
{"id": 2, "start": 5, "end": 8, "speaker": "Mike", "text": "Yeah.", "categories": ["backchannel"]}
|
||||
]
|
||||
}`)
|
||||
output := writeJSONFile(t, dir, "output.md", "")
|
||||
|
||||
err := executeRender(
|
||||
"--input-file", input,
|
||||
"--output-file", output,
|
||||
"--format", config.RenderFormatMarkdown,
|
||||
"--title", "Transcript",
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("render failed: %v", err)
|
||||
}
|
||||
|
||||
data := readFile(t, output)
|
||||
if !strings.Contains(data, "# Transcript") {
|
||||
t.Fatalf("missing title:\n%s", data)
|
||||
}
|
||||
if !strings.Contains(data, "[00:00:01–00:00:04] **Eric:** Hello there.") {
|
||||
t.Fatalf("missing first segment:\n%s", data)
|
||||
}
|
||||
if !strings.Contains(data, "[00:00:05–00:00:08] **Mike:** *Yeah.*") {
|
||||
t.Fatalf("missing italicized backchannel segment:\n%s", data)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRenderWorksWithRequiredFlagsOnly(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeJSONFile(t, dir, "input.json", `{
|
||||
"metadata": {
|
||||
"application": "seriatim",
|
||||
"version": "v-test",
|
||||
"output_schema": "seriatim-minimal"
|
||||
},
|
||||
"segments": [
|
||||
{"id": 1, "start": 1, "end": 2, "speaker": "Eric", "text": "Hello there."}
|
||||
]
|
||||
}`)
|
||||
output := writeJSONFile(t, dir, "output.md", "")
|
||||
|
||||
err := executeRender(
|
||||
"--input-file", input,
|
||||
"--output-file", output,
|
||||
"--format", config.RenderFormatMarkdown,
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("render with required flags failed: %v", err)
|
||||
}
|
||||
|
||||
data := readFile(t, output)
|
||||
if !strings.Contains(data, "# Transcript") {
|
||||
t.Fatalf("missing default title:\n%s", data)
|
||||
}
|
||||
if !strings.Contains(data, "[00:00:01–00:00:02] **Eric:** Hello there.") {
|
||||
t.Fatalf("missing rendered segment:\n%s", data)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRenderRejectsUnsupportedFormat(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeJSONFile(t, dir, "input.json", `{"metadata":{"application":"seriatim","version":"v-test","output_schema":"seriatim-minimal"},"segments":[]}`)
|
||||
output := writeJSONFile(t, dir, "output.md", "")
|
||||
|
||||
err := executeRender(
|
||||
"--input-file", input,
|
||||
"--output-file", output,
|
||||
"--format", "txt",
|
||||
)
|
||||
if err == nil {
|
||||
t.Fatal("expected format error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "--format must be") {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRenderRejectsMalformedAndRawInput(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
output := writeJSONFile(t, dir, "output.md", "")
|
||||
|
||||
malformed := writeJSONFile(t, dir, "malformed.json", `{"metadata":`)
|
||||
err := executeRender(
|
||||
"--input-file", malformed,
|
||||
"--output-file", output,
|
||||
"--format", config.RenderFormatMarkdown,
|
||||
)
|
||||
if err == nil {
|
||||
t.Fatal("expected malformed input error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "input JSON is malformed") {
|
||||
t.Fatalf("unexpected malformed input error: %v", err)
|
||||
}
|
||||
|
||||
raw := writeJSONFile(t, dir, "raw.json", `{"segments":[{"id":0,"start":0.1,"end":1.1,"text":"hello","words":[{"word":"hello"}]}]}`)
|
||||
err = executeRender(
|
||||
"--input-file", raw,
|
||||
"--output-file", output,
|
||||
"--format", config.RenderFormatMarkdown,
|
||||
)
|
||||
if err == nil {
|
||||
t.Fatal("expected artifact validation error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "not a valid seriatim output artifact") {
|
||||
t.Fatalf("unexpected raw input error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRenderSupportsMinimalIntermediateAndFullInputs(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
content string
|
||||
}{
|
||||
{
|
||||
name: "minimal",
|
||||
content: `{
|
||||
"metadata": {
|
||||
"application": "seriatim",
|
||||
"version": "v-test",
|
||||
"output_schema": "seriatim-minimal"
|
||||
},
|
||||
"segments": [{"id":1,"start":1,"end":2,"speaker":"A","text":"one"}]
|
||||
}`,
|
||||
},
|
||||
{
|
||||
name: "intermediate",
|
||||
content: `{
|
||||
"metadata": {
|
||||
"application": "seriatim",
|
||||
"version": "v-test",
|
||||
"output_schema": "seriatim-intermediate"
|
||||
},
|
||||
"segments": [{"id":1,"start":1,"end":2,"speaker":"A","text":"one","categories":["filler"]}]
|
||||
}`,
|
||||
},
|
||||
{
|
||||
name: "full",
|
||||
content: `{
|
||||
"metadata": {
|
||||
"application": "seriatim",
|
||||
"version": "v-test",
|
||||
"input_reader": "json-files",
|
||||
"input_files": ["input.json"],
|
||||
"preprocessing_modules": [],
|
||||
"postprocessing_modules": [],
|
||||
"output_modules": ["json"]
|
||||
},
|
||||
"segments": [{
|
||||
"id":1,
|
||||
"source":"input.json",
|
||||
"source_segment_index":0,
|
||||
"speaker":"A",
|
||||
"start":1,
|
||||
"end":2,
|
||||
"text":"one"
|
||||
}],
|
||||
"overlap_groups": []
|
||||
}`,
|
||||
},
|
||||
}
|
||||
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeJSONFile(t, dir, "input.json", test.content)
|
||||
output := writeJSONFile(t, dir, "output.md", "")
|
||||
|
||||
err := executeRender(
|
||||
"--input-file", input,
|
||||
"--output-file", output,
|
||||
"--format", config.RenderFormatMarkdown,
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("render failed: %v", err)
|
||||
}
|
||||
data := readFile(t, output)
|
||||
if !strings.Contains(data, "**A:**") {
|
||||
t.Fatalf("missing rendered segment for %s input:\n%s", test.name, data)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRenderEmptyTranscriptIsDeterministic(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeJSONFile(t, dir, "input.json", `{
|
||||
"metadata": {
|
||||
"application": "seriatim",
|
||||
"version": "v-test",
|
||||
"output_schema": "seriatim-minimal"
|
||||
},
|
||||
"segments": []
|
||||
}`)
|
||||
output := writeJSONFile(t, dir, "output.md", "")
|
||||
|
||||
run := func() string {
|
||||
err := executeRender(
|
||||
"--input-file", input,
|
||||
"--output-file", output,
|
||||
"--format", config.RenderFormatMarkdown,
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("render failed: %v", err)
|
||||
}
|
||||
return readFile(t, output)
|
||||
}
|
||||
|
||||
first := run()
|
||||
second := run()
|
||||
if first != second {
|
||||
t.Fatalf("empty transcript render is not deterministic:\nfirst:\n%s\nsecond:\n%s", first, second)
|
||||
}
|
||||
if first != "# Transcript\n" {
|
||||
t.Fatalf("unexpected empty transcript output:\n%s", first)
|
||||
}
|
||||
}
|
||||
|
||||
func executeRender(args ...string) error {
|
||||
cmd := NewRootCommand()
|
||||
cmd.SetArgs(append([]string{"render"}, args...))
|
||||
return cmd.Execute()
|
||||
}
|
||||
|
||||
func readFile(t *testing.T, path string) string {
|
||||
t.Helper()
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("read %s: %v", path, err)
|
||||
}
|
||||
return string(data)
|
||||
}
|
||||
@@ -10,7 +10,7 @@ import (
|
||||
func NewRootCommand() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "seriatim",
|
||||
Short: "Merge, trim, and normalize transcript artifacts",
|
||||
Short: "Merge, trim, normalize, and render transcript artifacts",
|
||||
Version: buildinfo.Version,
|
||||
SilenceErrors: true,
|
||||
SilenceUsage: true,
|
||||
@@ -18,6 +18,7 @@ func NewRootCommand() *cobra.Command {
|
||||
|
||||
cmd.AddCommand(newMergeCommand())
|
||||
cmd.AddCommand(newNormalizeCommand())
|
||||
cmd.AddCommand(newRenderCommand())
|
||||
cmd.AddCommand(newTrimCommand())
|
||||
return cmd
|
||||
}
|
||||
|
||||
@@ -1,41 +1,12 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"sort"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
||||
triminternal "gitea.maximumdirect.net/eric/seriatim/internal/trim"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/trim"
|
||||
)
|
||||
|
||||
type trimAuditReport struct {
|
||||
Operation string `json:"operation"`
|
||||
InputFile string `json:"input_file"`
|
||||
OutputFile string `json:"output_file"`
|
||||
InputSchema string `json:"input_schema"`
|
||||
OutputSchema string `json:"output_schema"`
|
||||
Mode string `json:"mode"`
|
||||
Selector string `json:"selector"`
|
||||
SelectedIDs []int `json:"selected_ids"`
|
||||
AllowEmpty bool `json:"allow_empty"`
|
||||
InputSegmentCount int `json:"input_segment_count"`
|
||||
RetainedSegmentCount int `json:"retained_segment_count"`
|
||||
RemovedSegmentCount int `json:"removed_segment_count"`
|
||||
RemovedInputIDs []int `json:"removed_input_ids"`
|
||||
OldToNewIDMapping []trimIDMapping `json:"old_to_new_id_mapping"`
|
||||
OverlapGroupsRecomputed bool `json:"overlap_groups_recomputed"`
|
||||
}
|
||||
|
||||
type trimIDMapping struct {
|
||||
OldID int `json:"old_id"`
|
||||
NewID int `json:"new_id"`
|
||||
}
|
||||
|
||||
func newTrimCommand() *cobra.Command {
|
||||
var opts config.TrimOptions
|
||||
|
||||
@@ -53,139 +24,18 @@ func newTrimCommand() *cobra.Command {
|
||||
return err
|
||||
}
|
||||
|
||||
selector, err := triminternal.ParseSelector(cfg.Selector)
|
||||
if err != nil {
|
||||
return fmt.Errorf("invalid selector %q: %w", cfg.Selector, err)
|
||||
}
|
||||
|
||||
data, err := os.ReadFile(cfg.InputFile)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read --input-file %q: %w", cfg.InputFile, err)
|
||||
}
|
||||
|
||||
artifact, err := triminternal.ParseArtifactJSON(data)
|
||||
if err != nil {
|
||||
return fmt.Errorf("--input-file %q: %w", cfg.InputFile, err)
|
||||
}
|
||||
inputSegmentCount := artifact.SegmentCount()
|
||||
inputSchema := artifact.Schema
|
||||
|
||||
mode := triminternal.ModeKeep
|
||||
if cfg.Mode == "remove" {
|
||||
mode = triminternal.ModeRemove
|
||||
}
|
||||
|
||||
trimmed, err := triminternal.ApplyArtifact(artifact, triminternal.Options{
|
||||
Mode: mode,
|
||||
Selector: selector,
|
||||
AllowEmpty: cfg.AllowEmpty,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
outputSchema := artifact.Schema
|
||||
if cfg.OutputSchema != "" {
|
||||
outputSchema = cfg.OutputSchema
|
||||
}
|
||||
|
||||
outputArtifact, err := triminternal.ConvertArtifact(trimmed.Artifact, outputSchema)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if err := triminternal.ValidateArtifact(outputArtifact); err != nil {
|
||||
return fmt.Errorf("validate trimmed output: %w", err)
|
||||
}
|
||||
|
||||
if err := writeOutputJSON(cfg.OutputFile, outputArtifact.Value()); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if cfg.ReportFile != "" {
|
||||
audit := trimAuditReport{
|
||||
Operation: "trim",
|
||||
InputFile: cfg.InputFile,
|
||||
OutputFile: cfg.OutputFile,
|
||||
InputSchema: inputSchema,
|
||||
OutputSchema: outputArtifact.Schema,
|
||||
Mode: cfg.Mode,
|
||||
Selector: cfg.Selector,
|
||||
SelectedIDs: selector.IDs(),
|
||||
AllowEmpty: cfg.AllowEmpty,
|
||||
InputSegmentCount: inputSegmentCount,
|
||||
RetainedSegmentCount: len(trimmed.OldToNewID),
|
||||
RemovedSegmentCount: len(trimmed.RemovedIDs),
|
||||
RemovedInputIDs: append([]int(nil), trimmed.RemovedIDs...),
|
||||
OldToNewIDMapping: orderedIDMapping(trimmed.OldToNewID),
|
||||
OverlapGroupsRecomputed: trimmed.OverlapGroupsRecomputed,
|
||||
}
|
||||
auditJSON, err := json.Marshal(audit)
|
||||
if err != nil {
|
||||
return fmt.Errorf("marshal trim audit report: %w", err)
|
||||
}
|
||||
|
||||
rpt := report.Report{
|
||||
Metadata: report.Metadata{
|
||||
Application: outputArtifact.Application(),
|
||||
Version: outputArtifact.Version(),
|
||||
InputReader: "trim-artifact",
|
||||
InputFiles: []string{cfg.InputFile},
|
||||
OutputModules: []string{"json"},
|
||||
},
|
||||
Events: []report.Event{
|
||||
report.Info("trim", "trim", fmt.Sprintf("trimmed %d input segment(s) into %d output segment(s) with mode=%s", inputSegmentCount, outputArtifact.SegmentCount(), cfg.Mode)),
|
||||
report.Info("trim", "trim-audit", string(auditJSON)),
|
||||
report.Info("trim", "validate-output", fmt.Sprintf("validated %d output segment(s)", outputArtifact.SegmentCount())),
|
||||
report.Info("output", "json", "wrote transcript JSON"),
|
||||
},
|
||||
}
|
||||
if err := report.WriteJSON(cfg.ReportFile, rpt); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
return nil
|
||||
return trim.Run(cmd.Context(), cfg)
|
||||
},
|
||||
}
|
||||
|
||||
flags := cmd.Flags()
|
||||
flags.StringVar(&opts.InputFile, "input-file", "", "input seriatim transcript artifact JSON file")
|
||||
flags.StringVar(&opts.OutputFile, "output-file", "", "output transcript JSON file")
|
||||
flags.StringVar(&opts.ReportFile, "report-file", "", "optional report JSON file")
|
||||
addOutputFileFlag(cmd, &opts.OutputFile)
|
||||
addReportFileFlag(cmd, &opts.ReportFile)
|
||||
flags.StringVar(&opts.Keep, "keep", "", "segment ID selector to keep (for example: 1-10,15)")
|
||||
flags.StringVar(&opts.Remove, "remove", "", "segment ID selector to remove (for example: 1-10,15)")
|
||||
flags.StringVar(&opts.OutputSchema, "output-schema", "", "optional output JSON schema override: seriatim-minimal, seriatim-intermediate, or seriatim-full")
|
||||
addTrimOutputSchemaFlag(cmd, &opts.OutputSchema)
|
||||
flags.BoolVar(&opts.AllowEmpty, "allow-empty", false, "allow trimming to an empty transcript")
|
||||
|
||||
return cmd
|
||||
}
|
||||
|
||||
func writeOutputJSON(path string, value any) error {
|
||||
file, err := os.Create(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
enc := json.NewEncoder(file)
|
||||
enc.SetIndent("", " ")
|
||||
return enc.Encode(value)
|
||||
}
|
||||
|
||||
func orderedIDMapping(mapping map[int]int) []trimIDMapping {
|
||||
keys := make([]int, 0, len(mapping))
|
||||
for oldID := range mapping {
|
||||
keys = append(keys, oldID)
|
||||
}
|
||||
sort.Ints(keys)
|
||||
|
||||
pairs := make([]trimIDMapping, 0, len(keys))
|
||||
for _, oldID := range keys {
|
||||
pairs = append(pairs, trimIDMapping{
|
||||
OldID: oldID,
|
||||
NewID: mapping[oldID],
|
||||
})
|
||||
}
|
||||
return pairs
|
||||
}
|
||||
|
||||
@@ -12,6 +12,29 @@ import (
|
||||
"gitea.maximumdirect.net/eric/seriatim/schema"
|
||||
)
|
||||
|
||||
type trimAuditReport struct {
|
||||
Operation string `json:"operation"`
|
||||
InputFile string `json:"input_file"`
|
||||
OutputFile string `json:"output_file"`
|
||||
InputSchema string `json:"input_schema"`
|
||||
OutputSchema string `json:"output_schema"`
|
||||
Mode string `json:"mode"`
|
||||
Selector string `json:"selector"`
|
||||
SelectedIDs []int `json:"selected_ids"`
|
||||
AllowEmpty bool `json:"allow_empty"`
|
||||
InputSegmentCount int `json:"input_segment_count"`
|
||||
RetainedSegmentCount int `json:"retained_segment_count"`
|
||||
RemovedSegmentCount int `json:"removed_segment_count"`
|
||||
RemovedInputIDs []int `json:"removed_input_ids"`
|
||||
OldToNewIDMapping []trimIDMapping `json:"old_to_new_id_mapping"`
|
||||
OverlapGroupsRecomputed bool `json:"overlap_groups_recomputed"`
|
||||
}
|
||||
|
||||
type trimIDMapping struct {
|
||||
OldID int `json:"old_id"`
|
||||
NewID int `json:"new_id"`
|
||||
}
|
||||
|
||||
func TestTrimKeepModeEndToEnd(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTrimFullFixture(t, dir, "input.json")
|
||||
|
||||
@@ -160,13 +160,7 @@ func (r run) coalescedSegment(id int) model.Segment {
|
||||
}
|
||||
|
||||
func segmentRef(segment model.Segment) string {
|
||||
if segment.SourceSegmentIndex != nil {
|
||||
return fmt.Sprintf("%s#%d", segment.Source, *segment.SourceSegmentIndex)
|
||||
}
|
||||
if segment.SourceRef != "" {
|
||||
return segment.SourceRef
|
||||
}
|
||||
return segment.Source
|
||||
return model.SegmentReference(segment)
|
||||
}
|
||||
|
||||
func isSkippableInterjection(segment model.Segment) bool {
|
||||
|
||||
@@ -8,12 +8,16 @@ import (
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/schema"
|
||||
)
|
||||
|
||||
const (
|
||||
DefaultInputReader = "json-files"
|
||||
DefaultOutputModules = "json"
|
||||
DefaultOutputSchema = OutputSchemaIntermediate
|
||||
DefaultRenderTitle = "Transcript"
|
||||
RenderFormatMarkdown = "markdown"
|
||||
DefaultPreprocessingModules = "validate-raw,normalize-speakers,trim-text"
|
||||
DefaultPostprocessingModules = "detect-overlaps,resolve-overlaps,backchannel,filler,resolve-danglers,coalesce,detect-overlaps,autocorrect,assign-ids,validate-output"
|
||||
DefaultOverlapWordRunGap = 1.0
|
||||
@@ -27,9 +31,9 @@ const (
|
||||
WordRunReorderWindowEnv = "SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW"
|
||||
BackchannelMaxDurationEnv = "SERIATIM_BACKCHANNEL_MAX_DURATION"
|
||||
FillerMaxDurationEnv = "SERIATIM_FILLER_MAX_DURATION"
|
||||
OutputSchemaMinimal = "seriatim-minimal"
|
||||
OutputSchemaIntermediate = "seriatim-intermediate"
|
||||
OutputSchemaFull = "seriatim-full"
|
||||
OutputSchemaMinimal = schema.OutputSchemaMinimal
|
||||
OutputSchemaIntermediate = schema.OutputSchemaIntermediate
|
||||
OutputSchemaFull = schema.OutputSchemaFull
|
||||
)
|
||||
|
||||
// MergeOptions captures raw CLI option values before validation.
|
||||
@@ -67,6 +71,17 @@ type NormalizeOptions struct {
|
||||
OutputModules string
|
||||
}
|
||||
|
||||
// RenderOptions captures raw CLI option values before validation.
|
||||
type RenderOptions struct {
|
||||
InputFile string
|
||||
OutputFile string
|
||||
Format string
|
||||
Title string
|
||||
IncludeTimestamps bool
|
||||
IncludeSegmentIDs bool
|
||||
IncludeMetadata bool
|
||||
}
|
||||
|
||||
// Config is the validated runtime configuration for a merge invocation.
|
||||
type Config struct {
|
||||
InputFiles []string
|
||||
@@ -106,6 +121,17 @@ type NormalizeConfig struct {
|
||||
OutputModules []string
|
||||
}
|
||||
|
||||
// RenderConfig is the validated runtime configuration for a render invocation.
|
||||
type RenderConfig struct {
|
||||
InputFile string
|
||||
OutputFile string
|
||||
Format string
|
||||
Title string
|
||||
IncludeTimestamps bool
|
||||
IncludeSegmentIDs bool
|
||||
IncludeMetadata bool
|
||||
}
|
||||
|
||||
// NewMergeConfig validates raw merge options and returns normalized config.
|
||||
func NewMergeConfig(opts MergeOptions) (Config, error) {
|
||||
cfg := Config{
|
||||
@@ -210,11 +236,8 @@ func NewMergeConfig(opts MergeOptions) (Config, error) {
|
||||
|
||||
// NewTrimConfig validates raw trim options and returns normalized config.
|
||||
func NewTrimConfig(opts TrimOptions) (TrimConfig, error) {
|
||||
inputFile := filepath.Clean(strings.TrimSpace(opts.InputFile))
|
||||
if strings.TrimSpace(opts.InputFile) == "" {
|
||||
return TrimConfig{}, errors.New("--input-file is required")
|
||||
}
|
||||
if err := requireFile(inputFile, "--input-file"); err != nil {
|
||||
inputFile, err := normalizeSingleInputFile(opts.InputFile, "--input-file")
|
||||
if err != nil {
|
||||
return TrimConfig{}, err
|
||||
}
|
||||
|
||||
@@ -223,12 +246,9 @@ func NewTrimConfig(opts TrimOptions) (TrimConfig, error) {
|
||||
return TrimConfig{}, err
|
||||
}
|
||||
|
||||
reportFile := ""
|
||||
if strings.TrimSpace(opts.ReportFile) != "" {
|
||||
reportFile, err = normalizeOutputPath(opts.ReportFile, "--report-file")
|
||||
if err != nil {
|
||||
return TrimConfig{}, err
|
||||
}
|
||||
reportFile, err := normalizeOptionalOutputPath(opts.ReportFile, "--report-file")
|
||||
if err != nil {
|
||||
return TrimConfig{}, err
|
||||
}
|
||||
|
||||
keep := strings.TrimSpace(opts.Keep)
|
||||
@@ -267,11 +287,8 @@ func NewTrimConfig(opts TrimOptions) (TrimConfig, error) {
|
||||
|
||||
// NewNormalizeConfig validates raw normalize options and returns normalized config.
|
||||
func NewNormalizeConfig(opts NormalizeOptions) (NormalizeConfig, error) {
|
||||
inputFile := filepath.Clean(strings.TrimSpace(opts.InputFile))
|
||||
if strings.TrimSpace(opts.InputFile) == "" {
|
||||
return NormalizeConfig{}, errors.New("--input-file is required")
|
||||
}
|
||||
if err := requireFile(inputFile, "--input-file"); err != nil {
|
||||
inputFile, err := normalizeSingleInputFile(opts.InputFile, "--input-file")
|
||||
if err != nil {
|
||||
return NormalizeConfig{}, err
|
||||
}
|
||||
|
||||
@@ -280,12 +297,9 @@ func NewNormalizeConfig(opts NormalizeOptions) (NormalizeConfig, error) {
|
||||
return NormalizeConfig{}, err
|
||||
}
|
||||
|
||||
reportFile := ""
|
||||
if strings.TrimSpace(opts.ReportFile) != "" {
|
||||
reportFile, err = normalizeOutputPath(opts.ReportFile, "--report-file")
|
||||
if err != nil {
|
||||
return NormalizeConfig{}, err
|
||||
}
|
||||
reportFile, err := normalizeOptionalOutputPath(opts.ReportFile, "--report-file")
|
||||
if err != nil {
|
||||
return NormalizeConfig{}, err
|
||||
}
|
||||
|
||||
outputSchema, err := resolveOutputSchema(opts.OutputSchema)
|
||||
@@ -313,6 +327,42 @@ func NewNormalizeConfig(opts NormalizeOptions) (NormalizeConfig, error) {
|
||||
}, nil
|
||||
}
|
||||
|
||||
// NewRenderConfig validates raw render options and returns normalized config.
|
||||
func NewRenderConfig(opts RenderOptions) (RenderConfig, error) {
|
||||
inputFile, err := normalizeSingleInputFile(opts.InputFile, "--input-file")
|
||||
if err != nil {
|
||||
return RenderConfig{}, err
|
||||
}
|
||||
|
||||
outputFile, err := normalizeOutputPath(opts.OutputFile, "--output-file")
|
||||
if err != nil {
|
||||
return RenderConfig{}, err
|
||||
}
|
||||
|
||||
format := strings.TrimSpace(opts.Format)
|
||||
if format == "" {
|
||||
return RenderConfig{}, errors.New("--format is required")
|
||||
}
|
||||
if err := validateRenderFormat(format); err != nil {
|
||||
return RenderConfig{}, err
|
||||
}
|
||||
|
||||
title := strings.TrimSpace(opts.Title)
|
||||
if title == "" {
|
||||
title = DefaultRenderTitle
|
||||
}
|
||||
|
||||
return RenderConfig{
|
||||
InputFile: inputFile,
|
||||
OutputFile: outputFile,
|
||||
Format: format,
|
||||
Title: title,
|
||||
IncludeTimestamps: opts.IncludeTimestamps,
|
||||
IncludeSegmentIDs: opts.IncludeSegmentIDs,
|
||||
IncludeMetadata: opts.IncludeMetadata,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func parseModuleList(value string) ([]string, error) {
|
||||
value = strings.TrimSpace(value)
|
||||
if value == "" {
|
||||
@@ -332,12 +382,12 @@ func parseModuleList(value string) ([]string, error) {
|
||||
}
|
||||
|
||||
func validateOutputSchema(value string) error {
|
||||
switch value {
|
||||
case OutputSchemaMinimal, OutputSchemaIntermediate, OutputSchemaFull:
|
||||
if schema.ValidOutputSchemaName(value) {
|
||||
return nil
|
||||
default:
|
||||
return fmt.Errorf("--output-schema must be one of %q, %q, or %q", OutputSchemaMinimal, OutputSchemaIntermediate, OutputSchemaFull)
|
||||
}
|
||||
|
||||
names := schema.OutputSchemaNames()
|
||||
return fmt.Errorf("--output-schema must be one of %q, %q, or %q", names[0], names[1], names[2])
|
||||
}
|
||||
|
||||
func resolveOutputSchema(value string) (string, error) {
|
||||
@@ -381,6 +431,26 @@ func normalizeInputFiles(paths []string) ([]string, error) {
|
||||
return normalized, nil
|
||||
}
|
||||
|
||||
func normalizeSingleInputFile(path string, flag string) (string, error) {
|
||||
path = strings.TrimSpace(path)
|
||||
if path == "" {
|
||||
return "", fmt.Errorf("%s is required", flag)
|
||||
}
|
||||
|
||||
clean := filepath.Clean(path)
|
||||
if err := requireFile(clean, flag); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return clean, nil
|
||||
}
|
||||
|
||||
func normalizeOptionalOutputPath(path string, flag string) (string, error) {
|
||||
if strings.TrimSpace(path) == "" {
|
||||
return "", nil
|
||||
}
|
||||
return normalizeOutputPath(path, flag)
|
||||
}
|
||||
|
||||
func normalizeOutputPath(path string, flag string) (string, error) {
|
||||
path = strings.TrimSpace(path)
|
||||
if path == "" {
|
||||
@@ -475,3 +545,12 @@ func validateNormalizeOutputModules(modules []string) error {
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func validateRenderFormat(format string) error {
|
||||
switch format {
|
||||
case RenderFormatMarkdown:
|
||||
return nil
|
||||
default:
|
||||
return fmt.Errorf("--format must be %q", RenderFormatMarkdown)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -538,15 +538,9 @@ func TestCoalesceGapUsesValidOverride(t *testing.T) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "merged.json")
|
||||
|
||||
cfg, err := NewMergeConfig(MergeOptions{
|
||||
InputFiles: []string{input},
|
||||
OutputFile: output,
|
||||
InputReader: DefaultInputReader,
|
||||
OutputModules: DefaultOutputModules,
|
||||
PreprocessingModules: DefaultPreprocessingModules,
|
||||
PostprocessingModules: DefaultPostprocessingModules,
|
||||
CoalesceGap: "1.5",
|
||||
})
|
||||
opts := validMergeOptions(input, output)
|
||||
opts.CoalesceGap = "1.5"
|
||||
cfg, err := NewMergeConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
@@ -560,15 +554,9 @@ func TestCoalesceGapAllowsZero(t *testing.T) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "merged.json")
|
||||
|
||||
cfg, err := NewMergeConfig(MergeOptions{
|
||||
InputFiles: []string{input},
|
||||
OutputFile: output,
|
||||
InputReader: DefaultInputReader,
|
||||
OutputModules: DefaultOutputModules,
|
||||
PreprocessingModules: DefaultPreprocessingModules,
|
||||
PostprocessingModules: DefaultPostprocessingModules,
|
||||
CoalesceGap: "0",
|
||||
})
|
||||
opts := validMergeOptions(input, output)
|
||||
opts.CoalesceGap = "0"
|
||||
cfg, err := NewMergeConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
@@ -593,15 +581,9 @@ func TestCoalesceGapRejectsInvalidOverride(t *testing.T) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "merged.json")
|
||||
|
||||
_, err := NewMergeConfig(MergeOptions{
|
||||
InputFiles: []string{input},
|
||||
OutputFile: output,
|
||||
InputReader: DefaultInputReader,
|
||||
OutputModules: DefaultOutputModules,
|
||||
PreprocessingModules: DefaultPreprocessingModules,
|
||||
PostprocessingModules: DefaultPostprocessingModules,
|
||||
CoalesceGap: test.value,
|
||||
})
|
||||
opts := validMergeOptions(input, output)
|
||||
opts.CoalesceGap = test.value
|
||||
_, err := NewMergeConfig(opts)
|
||||
if err == nil {
|
||||
t.Fatal("expected error")
|
||||
}
|
||||
@@ -639,20 +621,16 @@ func TestNewTrimConfigRequiresExactlyOneSelectorFlag(t *testing.T) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "trimmed.json")
|
||||
|
||||
_, err := NewTrimConfig(TrimOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
})
|
||||
opts := validTrimOptions(input, output)
|
||||
opts.Keep = ""
|
||||
_, err := NewTrimConfig(opts)
|
||||
if err == nil || !strings.Contains(err.Error(), "exactly one of --keep or --remove is required") {
|
||||
t.Fatalf("expected missing selector error, got %v", err)
|
||||
}
|
||||
|
||||
_, err = NewTrimConfig(TrimOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
Keep: "1",
|
||||
Remove: "2",
|
||||
})
|
||||
opts = validTrimOptions(input, output)
|
||||
opts.Remove = "2"
|
||||
_, err = NewTrimConfig(opts)
|
||||
if err == nil || !strings.Contains(err.Error(), "mutually exclusive") {
|
||||
t.Fatalf("expected mutually exclusive selector error, got %v", err)
|
||||
}
|
||||
@@ -664,14 +642,13 @@ func TestNewTrimConfigAcceptsOutputSchemaOverride(t *testing.T) {
|
||||
output := filepath.Join(dir, "trimmed.json")
|
||||
reportPath := filepath.Join(dir, "report.json")
|
||||
|
||||
cfg, err := NewTrimConfig(TrimOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
ReportFile: reportPath,
|
||||
Remove: "3-5",
|
||||
OutputSchema: OutputSchemaMinimal,
|
||||
AllowEmpty: true,
|
||||
})
|
||||
opts := validTrimOptions(input, output)
|
||||
opts.Keep = ""
|
||||
opts.Remove = "3-5"
|
||||
opts.ReportFile = reportPath
|
||||
opts.OutputSchema = OutputSchemaMinimal
|
||||
opts.AllowEmpty = true
|
||||
cfg, err := NewTrimConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
@@ -692,17 +669,30 @@ func TestNewTrimConfigAcceptsOutputSchemaOverride(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewTrimConfigTreatsWhitespaceReportFileAsOmitted(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "trimmed.json")
|
||||
|
||||
opts := validTrimOptions(input, output)
|
||||
opts.ReportFile = " \t "
|
||||
cfg, err := NewTrimConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
if cfg.ReportFile != "" {
|
||||
t.Fatalf("report file = %q, want empty", cfg.ReportFile)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewTrimConfigRejectsInvalidOutputSchemaOverride(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "trimmed.json")
|
||||
|
||||
_, err := NewTrimConfig(TrimOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
Keep: "1",
|
||||
OutputSchema: "compact",
|
||||
})
|
||||
opts := validTrimOptions(input, output)
|
||||
opts.OutputSchema = "compact"
|
||||
_, err := NewTrimConfig(opts)
|
||||
if err == nil {
|
||||
t.Fatal("expected output schema validation error")
|
||||
}
|
||||
@@ -731,10 +721,8 @@ func TestNewNormalizeConfigRequiresOutputFile(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
|
||||
_, err := NewNormalizeConfig(NormalizeOptions{
|
||||
InputFile: input,
|
||||
OutputModules: DefaultOutputModules,
|
||||
})
|
||||
opts := validNormalizeOptions(input, "")
|
||||
_, err := NewNormalizeConfig(opts)
|
||||
if err == nil {
|
||||
t.Fatal("expected output-file required error")
|
||||
}
|
||||
@@ -749,11 +737,8 @@ func TestNewNormalizeConfigResolvesOutputSchemaDefaultAndEnv(t *testing.T) {
|
||||
output := filepath.Join(dir, "normalized.json")
|
||||
|
||||
t.Setenv(OutputSchemaEnv, "")
|
||||
cfg, err := NewNormalizeConfig(NormalizeOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
OutputModules: DefaultOutputModules,
|
||||
})
|
||||
opts := validNormalizeOptions(input, output)
|
||||
cfg, err := NewNormalizeConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
@@ -762,11 +747,7 @@ func TestNewNormalizeConfigResolvesOutputSchemaDefaultAndEnv(t *testing.T) {
|
||||
}
|
||||
|
||||
t.Setenv(OutputSchemaEnv, OutputSchemaMinimal)
|
||||
cfg, err = NewNormalizeConfig(NormalizeOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
OutputModules: DefaultOutputModules,
|
||||
})
|
||||
cfg, err = NewNormalizeConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
@@ -780,12 +761,9 @@ func TestNewNormalizeConfigRejectsInvalidOutputSchema(t *testing.T) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "normalized.json")
|
||||
|
||||
_, err := NewNormalizeConfig(NormalizeOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
OutputSchema: "compact",
|
||||
OutputModules: DefaultOutputModules,
|
||||
})
|
||||
opts := validNormalizeOptions(input, output)
|
||||
opts.OutputSchema = "compact"
|
||||
_, err := NewNormalizeConfig(opts)
|
||||
if err == nil {
|
||||
t.Fatal("expected output schema error")
|
||||
}
|
||||
@@ -799,11 +777,9 @@ func TestNewNormalizeConfigRejectsUnknownOutputModule(t *testing.T) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "normalized.json")
|
||||
|
||||
_, err := NewNormalizeConfig(NormalizeOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
OutputModules: "json,yaml",
|
||||
})
|
||||
opts := validNormalizeOptions(input, output)
|
||||
opts.OutputModules = "json,yaml"
|
||||
_, err := NewNormalizeConfig(opts)
|
||||
if err == nil {
|
||||
t.Fatal("expected output module error")
|
||||
}
|
||||
@@ -812,6 +788,153 @@ func TestNewNormalizeConfigRejectsUnknownOutputModule(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewNormalizeConfigTreatsWhitespaceReportFileAsOmitted(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "normalized.json")
|
||||
|
||||
opts := validNormalizeOptions(input, output)
|
||||
opts.ReportFile = "\n\t "
|
||||
cfg, err := NewNormalizeConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
if cfg.ReportFile != "" {
|
||||
t.Fatalf("report file = %q, want empty", cfg.ReportFile)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewRenderConfigRequiresInputOutputAndFormat(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "rendered.md")
|
||||
|
||||
_, err := NewRenderConfig(RenderOptions{
|
||||
OutputFile: output,
|
||||
Format: RenderFormatMarkdown,
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), "--input-file is required") {
|
||||
t.Fatalf("expected input-file required error, got %v", err)
|
||||
}
|
||||
|
||||
_, err = NewRenderConfig(RenderOptions{
|
||||
InputFile: input,
|
||||
Format: RenderFormatMarkdown,
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), "--output-file is required") {
|
||||
t.Fatalf("expected output-file required error, got %v", err)
|
||||
}
|
||||
|
||||
_, err = NewRenderConfig(RenderOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), "--format is required") {
|
||||
t.Fatalf("expected format required error, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewRenderConfigRejectsUnknownFormat(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "rendered.md")
|
||||
|
||||
opts := validRenderOptions(input, output)
|
||||
opts.Format = "txt"
|
||||
_, err := NewRenderConfig(opts)
|
||||
if err == nil {
|
||||
t.Fatal("expected format validation error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "--format must be") {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewRenderConfigAppliesDefaultsAndFlags(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "rendered.md")
|
||||
|
||||
cfg, err := NewRenderConfig(validRenderOptions(input, output))
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
if cfg.Title != DefaultRenderTitle {
|
||||
t.Fatalf("title = %q, want %q", cfg.Title, DefaultRenderTitle)
|
||||
}
|
||||
if !cfg.IncludeTimestamps {
|
||||
t.Fatal("include timestamps should default true")
|
||||
}
|
||||
if cfg.IncludeSegmentIDs {
|
||||
t.Fatal("include segment IDs should default false")
|
||||
}
|
||||
if cfg.IncludeMetadata {
|
||||
t.Fatal("include metadata should default false")
|
||||
}
|
||||
|
||||
opts := validRenderOptions(input, output)
|
||||
opts.Title = "Meeting Notes"
|
||||
opts.IncludeTimestamps = false
|
||||
opts.IncludeSegmentIDs = true
|
||||
opts.IncludeMetadata = true
|
||||
cfg, err = NewRenderConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
if cfg.Title != "Meeting Notes" {
|
||||
t.Fatalf("title = %q, want Meeting Notes", cfg.Title)
|
||||
}
|
||||
if cfg.IncludeTimestamps {
|
||||
t.Fatal("include timestamps should be false")
|
||||
}
|
||||
if !cfg.IncludeSegmentIDs {
|
||||
t.Fatal("include segment IDs should be true")
|
||||
}
|
||||
if !cfg.IncludeMetadata {
|
||||
t.Fatal("include metadata should be true")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewRenderConfigRejectsMissingAndDirectoryInputFile(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
output := filepath.Join(dir, "rendered.md")
|
||||
|
||||
missingInput := filepath.Join(dir, "missing.json")
|
||||
_, err := NewRenderConfig(validRenderOptions(missingInput, output))
|
||||
if err == nil {
|
||||
t.Fatal("expected missing input-file error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "--input-file") {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
|
||||
inputDir := filepath.Join(dir, "input-dir")
|
||||
if err := os.MkdirAll(inputDir, 0o700); err != nil {
|
||||
t.Fatalf("mkdir input dir: %v", err)
|
||||
}
|
||||
_, err = NewRenderConfig(validRenderOptions(inputDir, output))
|
||||
if err == nil {
|
||||
t.Fatal("expected directory input-file error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "is a directory, not a file") {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewRenderConfigRejectsMissingOutputParent(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "missing-parent", "rendered.md")
|
||||
|
||||
_, err := NewRenderConfig(validRenderOptions(input, output))
|
||||
if err == nil {
|
||||
t.Fatal("expected output parent directory error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "--output-file parent directory") {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func assertPositiveFloatEnvValidation(t *testing.T, envName string) {
|
||||
t.Helper()
|
||||
|
||||
@@ -832,14 +955,7 @@ func assertPositiveFloatEnvValidation(t *testing.T, envName string) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "merged.json")
|
||||
|
||||
_, err := NewMergeConfig(MergeOptions{
|
||||
InputFiles: []string{input},
|
||||
OutputFile: output,
|
||||
InputReader: DefaultInputReader,
|
||||
OutputModules: DefaultOutputModules,
|
||||
PreprocessingModules: DefaultPreprocessingModules,
|
||||
PostprocessingModules: DefaultPostprocessingModules,
|
||||
})
|
||||
_, err := NewMergeConfig(validMergeOptions(input, output))
|
||||
if err == nil {
|
||||
t.Fatal("expected error")
|
||||
}
|
||||
@@ -850,6 +966,45 @@ func assertPositiveFloatEnvValidation(t *testing.T, envName string) {
|
||||
}
|
||||
}
|
||||
|
||||
func validMergeOptions(inputFile string, outputFile string) MergeOptions {
|
||||
return MergeOptions{
|
||||
InputFiles: []string{inputFile},
|
||||
OutputFile: outputFile,
|
||||
InputReader: DefaultInputReader,
|
||||
OutputModules: DefaultOutputModules,
|
||||
PreprocessingModules: DefaultPreprocessingModules,
|
||||
PostprocessingModules: DefaultPostprocessingModules,
|
||||
}
|
||||
}
|
||||
|
||||
func validTrimOptions(inputFile string, outputFile string) TrimOptions {
|
||||
return TrimOptions{
|
||||
InputFile: inputFile,
|
||||
OutputFile: outputFile,
|
||||
Keep: "1",
|
||||
}
|
||||
}
|
||||
|
||||
func validNormalizeOptions(inputFile string, outputFile string) NormalizeOptions {
|
||||
return NormalizeOptions{
|
||||
InputFile: inputFile,
|
||||
OutputFile: outputFile,
|
||||
OutputModules: DefaultOutputModules,
|
||||
}
|
||||
}
|
||||
|
||||
func validRenderOptions(inputFile string, outputFile string) RenderOptions {
|
||||
return RenderOptions{
|
||||
InputFile: inputFile,
|
||||
OutputFile: outputFile,
|
||||
Format: RenderFormatMarkdown,
|
||||
Title: DefaultRenderTitle,
|
||||
IncludeTimestamps: true,
|
||||
IncludeSegmentIDs: false,
|
||||
IncludeMetadata: false,
|
||||
}
|
||||
}
|
||||
|
||||
func writeTempFile(t *testing.T, dir string, name string) string {
|
||||
t.Helper()
|
||||
|
||||
|
||||
28
internal/jsonfile/jsonfile.go
Normal file
28
internal/jsonfile/jsonfile.go
Normal file
@@ -0,0 +1,28 @@
|
||||
package jsonfile
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
)
|
||||
|
||||
// Write creates or truncates path and writes deterministic indented JSON.
|
||||
func Write(path string, value any) (err error) {
|
||||
file, err := os.Create(path)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create %q: %w", path, err)
|
||||
}
|
||||
defer func() {
|
||||
closeErr := file.Close()
|
||||
if err == nil && closeErr != nil {
|
||||
err = fmt.Errorf("close %q: %w", path, closeErr)
|
||||
}
|
||||
}()
|
||||
|
||||
encoder := json.NewEncoder(file)
|
||||
encoder.SetIndent("", " ")
|
||||
if err := encoder.Encode(value); err != nil {
|
||||
return fmt.Errorf("encode %q: %w", path, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
69
internal/jsonfile/jsonfile_test.go
Normal file
69
internal/jsonfile/jsonfile_test.go
Normal file
@@ -0,0 +1,69 @@
|
||||
package jsonfile
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestWriteFormatsWithTwoSpaceIndentAndTrailingNewline(t *testing.T) {
|
||||
type payload struct {
|
||||
Name string `json:"name"`
|
||||
Items []int `json:"items"`
|
||||
}
|
||||
|
||||
path := filepath.Join(t.TempDir(), "out.json")
|
||||
value := payload{
|
||||
Name: "alpha",
|
||||
Items: []int{1, 2},
|
||||
}
|
||||
|
||||
if err := Write(path, value); err != nil {
|
||||
t.Fatalf("write failed: %v", err)
|
||||
}
|
||||
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("read output: %v", err)
|
||||
}
|
||||
|
||||
got := string(data)
|
||||
want := "{\n \"name\": \"alpha\",\n \"items\": [\n 1,\n 2\n ]\n}\n"
|
||||
if got != want {
|
||||
t.Fatalf("formatted JSON mismatch\nwant:\n%s\ngot:\n%s", want, got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriteProducesValidJSON(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "out.json")
|
||||
|
||||
value := map[string]any{
|
||||
"application": "seriatim",
|
||||
"segments": []map[string]any{
|
||||
{
|
||||
"id": 1,
|
||||
"speaker": "A",
|
||||
"text": "hello",
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
if err := Write(path, value); err != nil {
|
||||
t.Fatalf("write failed: %v", err)
|
||||
}
|
||||
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("read output: %v", err)
|
||||
}
|
||||
if !strings.HasSuffix(string(data), "\n") {
|
||||
t.Fatalf("output missing trailing newline: %q", string(data))
|
||||
}
|
||||
|
||||
var decoded map[string]any
|
||||
if err := json.Unmarshal(data, &decoded); err != nil {
|
||||
t.Fatalf("output is not valid JSON: %v", err)
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,7 @@
|
||||
package model
|
||||
|
||||
import "fmt"
|
||||
|
||||
// RawTranscript is a loaded input document before canonical normalization.
|
||||
type RawTranscript struct {
|
||||
Source string `json:"source"`
|
||||
@@ -61,6 +63,17 @@ type Segment struct {
|
||||
OverlapGroupID int `json:"overlap_group_id,omitempty"`
|
||||
}
|
||||
|
||||
// SegmentReference returns the best available external reference for a segment.
|
||||
func SegmentReference(segment Segment) string {
|
||||
if segment.Source != "" && segment.SourceSegmentIndex != nil {
|
||||
return fmt.Sprintf("%s#%d", segment.Source, *segment.SourceSegmentIndex)
|
||||
}
|
||||
if segment.SourceRef != "" {
|
||||
return segment.SourceRef
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// Word preserves optional word-level timing data.
|
||||
type Word struct {
|
||||
Text string `json:"text"`
|
||||
|
||||
41
internal/model/model_test.go
Normal file
41
internal/model/model_test.go
Normal file
@@ -0,0 +1,41 @@
|
||||
package model
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestSegmentReferenceUsesSourceAndIndexWhenAvailable(t *testing.T) {
|
||||
index := 3
|
||||
segment := Segment{
|
||||
Source: "input.json",
|
||||
SourceSegmentIndex: &index,
|
||||
SourceRef: "word-run:1:2:3",
|
||||
}
|
||||
|
||||
got := SegmentReference(segment)
|
||||
want := "input.json#3"
|
||||
if got != want {
|
||||
t.Fatalf("reference = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSegmentReferenceFallsBackToSourceRef(t *testing.T) {
|
||||
segment := Segment{
|
||||
Source: "input.json",
|
||||
SourceRef: "coalesce:2",
|
||||
}
|
||||
|
||||
got := SegmentReference(segment)
|
||||
want := "coalesce:2"
|
||||
if got != want {
|
||||
t.Fatalf("reference = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSegmentReferenceReturnsEmptyWhenNoReferenceFieldsPresent(t *testing.T) {
|
||||
segment := Segment{
|
||||
Source: "input.json",
|
||||
}
|
||||
|
||||
if got := SegmentReference(segment); got != "" {
|
||||
t.Fatalf("reference = %q, want empty", got)
|
||||
}
|
||||
}
|
||||
@@ -4,12 +4,12 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/artifact"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/buildinfo"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/jsonfile"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
||||
)
|
||||
|
||||
@@ -47,7 +47,7 @@ func Run(ctx context.Context, cfg config.NormalizeConfig) error {
|
||||
return err
|
||||
}
|
||||
|
||||
if err := writeOutputJSON(cfg.OutputFile, built.Output); err != nil {
|
||||
if err := jsonfile.Write(cfg.OutputFile, built.Output); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -118,18 +118,3 @@ func Run(ctx context.Context, cfg config.NormalizeConfig) error {
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func writeOutputJSON(path string, value any) error {
|
||||
file, err := os.Create(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
encoder := json.NewEncoder(file)
|
||||
encoder.SetIndent("", " ")
|
||||
if err := encoder.Encode(value); err != nil {
|
||||
return fmt.Errorf("encode normalize output JSON: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
package overlap
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/model"
|
||||
@@ -121,13 +120,7 @@ func distinctSpeakers(segments []model.Segment, indices []int) []string {
|
||||
|
||||
// SegmentRef returns the stable overlap reference for a segment.
|
||||
func SegmentRef(segment model.Segment) string {
|
||||
if segment.SourceSegmentIndex != nil {
|
||||
return fmt.Sprintf("%s#%d", segment.Source, *segment.SourceSegmentIndex)
|
||||
}
|
||||
if segment.SourceRef != "" {
|
||||
return segment.SourceRef
|
||||
}
|
||||
return segment.Source
|
||||
return model.SegmentReference(segment)
|
||||
}
|
||||
|
||||
func clearExisting(in *model.MergedTranscript) {
|
||||
|
||||
97
internal/render/markdown.go
Normal file
97
internal/render/markdown.go
Normal file
@@ -0,0 +1,97 @@
|
||||
package render
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// MarkdownRenderer renders transcript artifacts as Markdown.
|
||||
type MarkdownRenderer struct{}
|
||||
|
||||
// Render renders the transcript into deterministic Markdown.
|
||||
func (MarkdownRenderer) Render(transcript Transcript, opts Options) (string, error) {
|
||||
var lines []string
|
||||
|
||||
title := strings.TrimSpace(opts.Title)
|
||||
if title == "" {
|
||||
title = "Transcript"
|
||||
}
|
||||
lines = append(lines, "# "+escapeMarkdownInline(title), "")
|
||||
|
||||
if opts.IncludeMetadata {
|
||||
lines = append(lines,
|
||||
fmt.Sprintf("- Application: %s", escapeMarkdownInline(transcript.Metadata.Application)),
|
||||
fmt.Sprintf("- Version: %s", escapeMarkdownInline(transcript.Metadata.Version)),
|
||||
fmt.Sprintf("- Output schema: %s", escapeMarkdownInline(transcript.Schema)),
|
||||
"",
|
||||
)
|
||||
}
|
||||
|
||||
for _, segment := range transcript.Segments {
|
||||
parts := make([]string, 0, 4)
|
||||
if opts.IncludeTimestamps {
|
||||
parts = append(parts, fmt.Sprintf("[%s–%s]", formatTimestamp(segment.Start), formatTimestamp(segment.End)))
|
||||
}
|
||||
if opts.IncludeSegmentIDs {
|
||||
parts = append(parts, fmt.Sprintf("[#%d]", segment.ID))
|
||||
}
|
||||
|
||||
text := escapeMarkdownInline(segment.Text)
|
||||
if shouldItalicize(segment.Categories) {
|
||||
text = "*" + text + "*"
|
||||
}
|
||||
parts = append(parts, fmt.Sprintf("**%s:** %s", escapeMarkdownInline(segment.Speaker), text))
|
||||
lines = append(lines, strings.Join(parts, " "))
|
||||
lines = append(lines, "")
|
||||
}
|
||||
|
||||
output := strings.Join(lines, "\n")
|
||||
if !strings.HasSuffix(output, "\n") {
|
||||
output += "\n"
|
||||
}
|
||||
return output, nil
|
||||
}
|
||||
|
||||
func escapeMarkdownInline(value string) string {
|
||||
replacer := strings.NewReplacer(
|
||||
`\`, `\\`,
|
||||
"`", "\\`",
|
||||
"*", "\\*",
|
||||
"_", "\\_",
|
||||
"{", "\\{",
|
||||
"}", "\\}",
|
||||
"[", "\\[",
|
||||
"]", "\\]",
|
||||
"(", "\\(",
|
||||
")", "\\)",
|
||||
"#", "\\#",
|
||||
"+", "\\+",
|
||||
"!", "\\!",
|
||||
"|", "\\|",
|
||||
"<", "\\<",
|
||||
">", "\\>",
|
||||
)
|
||||
return replacer.Replace(value)
|
||||
}
|
||||
|
||||
func shouldItalicize(categories []string) bool {
|
||||
for _, category := range categories {
|
||||
switch category {
|
||||
case "background", "backchannel", "filler":
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func formatTimestamp(seconds float64) string {
|
||||
total := int(math.Round(seconds))
|
||||
if total < 0 {
|
||||
total = 0
|
||||
}
|
||||
hours := total / 3600
|
||||
minutes := (total % 3600) / 60
|
||||
remainder := total % 60
|
||||
return fmt.Sprintf("%02d:%02d:%02d", hours, minutes, remainder)
|
||||
}
|
||||
227
internal/render/markdown_test.go
Normal file
227
internal/render/markdown_test.go
Normal file
@@ -0,0 +1,227 @@
|
||||
package render
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestMarkdownRendererDefaultTranscriptShape(t *testing.T) {
|
||||
transcript := Transcript{
|
||||
Schema: "seriatim-intermediate",
|
||||
Metadata: Metadata{
|
||||
Application: "seriatim",
|
||||
Version: "v-test",
|
||||
},
|
||||
Segments: []Segment{
|
||||
{ID: 1, Start: 1, End: 4, Speaker: "Eric", Text: "Hello there."},
|
||||
{ID: 2, Start: 5, End: 8, Speaker: "Mike", Text: "Welcome back, everyone."},
|
||||
},
|
||||
}
|
||||
|
||||
output, err := MarkdownRenderer{}.Render(transcript, Options{
|
||||
Title: "Transcript",
|
||||
IncludeTimestamps: true,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("render markdown: %v", err)
|
||||
}
|
||||
|
||||
if !strings.Contains(output, "# Transcript") {
|
||||
t.Fatalf("expected title in output:\n%s", output)
|
||||
}
|
||||
if !strings.Contains(output, "[00:00:01–00:00:04] **Eric:** Hello there.") {
|
||||
t.Fatalf("expected first segment in output:\n%s", output)
|
||||
}
|
||||
if !strings.Contains(output, "[00:00:05–00:00:08] **Mike:** Welcome back, everyone.") {
|
||||
t.Fatalf("expected second segment in output:\n%s", output)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkdownRendererWithoutTimestamps(t *testing.T) {
|
||||
transcript := Transcript{
|
||||
Segments: []Segment{
|
||||
{ID: 1, Start: 1, End: 4, Speaker: "Eric", Text: "Hello."},
|
||||
},
|
||||
}
|
||||
|
||||
output, err := MarkdownRenderer{}.Render(transcript, Options{
|
||||
Title: "Transcript",
|
||||
IncludeTimestamps: false,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("render markdown: %v", err)
|
||||
}
|
||||
if strings.Contains(output, "[00:00:01") {
|
||||
t.Fatalf("timestamps should be omitted:\n%s", output)
|
||||
}
|
||||
if !strings.Contains(output, "**Eric:** Hello.") {
|
||||
t.Fatalf("expected speaker/text line:\n%s", output)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkdownRendererWithSegmentIDs(t *testing.T) {
|
||||
transcript := Transcript{
|
||||
Segments: []Segment{
|
||||
{ID: 17, Start: 1, End: 4, Speaker: "Eric", Text: "Hello."},
|
||||
},
|
||||
}
|
||||
|
||||
output, err := MarkdownRenderer{}.Render(transcript, Options{
|
||||
Title: "Transcript",
|
||||
IncludeTimestamps: true,
|
||||
IncludeSegmentIDs: true,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("render markdown: %v", err)
|
||||
}
|
||||
if !strings.Contains(output, "[#17]") {
|
||||
t.Fatalf("expected segment ID in output:\n%s", output)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkdownRendererMetadataOnlyWhenRequested(t *testing.T) {
|
||||
transcript := Transcript{
|
||||
Schema: "seriatim-full",
|
||||
Metadata: Metadata{
|
||||
Application: "seriatim",
|
||||
Version: "v-test",
|
||||
},
|
||||
}
|
||||
|
||||
withMetadata, err := MarkdownRenderer{}.Render(transcript, Options{
|
||||
Title: "Transcript",
|
||||
IncludeMetadata: true,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("render with metadata: %v", err)
|
||||
}
|
||||
if !strings.Contains(withMetadata, "- Application: seriatim") {
|
||||
t.Fatalf("expected metadata block:\n%s", withMetadata)
|
||||
}
|
||||
|
||||
withoutMetadata, err := MarkdownRenderer{}.Render(transcript, Options{Title: "Transcript"})
|
||||
if err != nil {
|
||||
t.Fatalf("render without metadata: %v", err)
|
||||
}
|
||||
if strings.Contains(withoutMetadata, "- Application: seriatim") {
|
||||
t.Fatalf("metadata should be omitted:\n%s", withoutMetadata)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkdownRendererEscapesUserProvidedMarkdown(t *testing.T) {
|
||||
transcript := Transcript{
|
||||
Schema: "seriatim-intermediate",
|
||||
Metadata: Metadata{
|
||||
Application: "seriatim *cli*",
|
||||
Version: "v[test]",
|
||||
},
|
||||
Segments: []Segment{
|
||||
{
|
||||
ID: 1,
|
||||
Start: 1,
|
||||
End: 2,
|
||||
Speaker: "Dr. *A_[1]",
|
||||
Text: "Use *literal* [link](target) and `code` \\ slash!",
|
||||
},
|
||||
{
|
||||
ID: 2,
|
||||
Start: 2,
|
||||
End: 3,
|
||||
Speaker: "Narrator",
|
||||
Text: "_aside_ with | pipe",
|
||||
Categories: []string{"background"},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
output, err := MarkdownRenderer{}.Render(transcript, Options{
|
||||
Title: "# Planning [notes]",
|
||||
IncludeTimestamps: false,
|
||||
IncludeMetadata: true,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("render markdown: %v", err)
|
||||
}
|
||||
|
||||
assertContains(t, output, "# \\# Planning \\[notes\\]")
|
||||
assertContains(t, output, "- Application: seriatim \\*cli\\*")
|
||||
assertContains(t, output, "- Version: v\\[test\\]")
|
||||
assertContains(t, output, "**Dr. \\*A\\_\\[1\\]:** Use \\*literal\\* \\[link\\]\\(target\\) and \\`code\\` \\\\ slash\\!")
|
||||
assertContains(t, output, "**Narrator:** *\\_aside\\_ with \\| pipe*")
|
||||
}
|
||||
|
||||
func TestMarkdownRendererCategoryHintItalicsAndUnknownCategories(t *testing.T) {
|
||||
transcript := Transcript{
|
||||
Segments: []Segment{
|
||||
{ID: 1, Start: 1, End: 2, Speaker: "A", Text: "bg", Categories: []string{"background"}},
|
||||
{ID: 2, Start: 2, End: 3, Speaker: "B", Text: "bc", Categories: []string{"backchannel"}},
|
||||
{ID: 3, Start: 3, End: 4, Speaker: "C", Text: "fill", Categories: []string{"filler"}},
|
||||
{ID: 4, Start: 4, End: 5, Speaker: "D", Text: "plain", Categories: []string{"unknown-tag"}},
|
||||
},
|
||||
}
|
||||
|
||||
output, err := MarkdownRenderer{}.Render(transcript, Options{
|
||||
Title: "Transcript",
|
||||
IncludeTimestamps: false,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("render markdown: %v", err)
|
||||
}
|
||||
if !strings.Contains(output, "**A:** *bg*") {
|
||||
t.Fatalf("expected background italics:\n%s", output)
|
||||
}
|
||||
if !strings.Contains(output, "**B:** *bc*") {
|
||||
t.Fatalf("expected backchannel italics:\n%s", output)
|
||||
}
|
||||
if !strings.Contains(output, "**C:** *fill*") {
|
||||
t.Fatalf("expected filler italics:\n%s", output)
|
||||
}
|
||||
if !strings.Contains(output, "**D:** plain") {
|
||||
t.Fatalf("expected unknown category to be ignored:\n%s", output)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkdownRendererIsDeterministic(t *testing.T) {
|
||||
transcript := Transcript{
|
||||
Schema: "seriatim-intermediate",
|
||||
Metadata: Metadata{
|
||||
Application: "seriatim",
|
||||
Version: "v-test",
|
||||
},
|
||||
Segments: []Segment{
|
||||
{ID: 1, Start: 1.2, End: 4.4, Speaker: "Eric", Text: "Hello there.", Categories: []string{"unknown-tag"}},
|
||||
{ID: 2, Start: 65.1, End: 68.8, Speaker: "Mike", Text: "Yeah.", Categories: []string{"backchannel"}},
|
||||
},
|
||||
}
|
||||
opts := Options{
|
||||
Title: "Transcript",
|
||||
IncludeTimestamps: true,
|
||||
IncludeSegmentIDs: true,
|
||||
IncludeMetadata: true,
|
||||
}
|
||||
|
||||
first, err := MarkdownRenderer{}.Render(transcript, opts)
|
||||
if err != nil {
|
||||
t.Fatalf("first render failed: %v", err)
|
||||
}
|
||||
second, err := MarkdownRenderer{}.Render(transcript, opts)
|
||||
if err != nil {
|
||||
t.Fatalf("second render failed: %v", err)
|
||||
}
|
||||
if first != second {
|
||||
t.Fatalf("render output is not deterministic:\nfirst:\n%s\nsecond:\n%s", first, second)
|
||||
}
|
||||
if !strings.Contains(first, "[00:00:01–00:00:04] [#1] **Eric:** Hello there.") {
|
||||
t.Fatalf("expected HH:MM:SS timestamp formatting:\n%s", first)
|
||||
}
|
||||
if !strings.Contains(first, "[00:01:05–00:01:09] [#2] **Mike:** *Yeah.*") {
|
||||
t.Fatalf("expected HH:MM:SS timestamp formatting:\n%s", first)
|
||||
}
|
||||
}
|
||||
|
||||
func assertContains(t *testing.T, value string, want string) {
|
||||
t.Helper()
|
||||
if !strings.Contains(value, want) {
|
||||
t.Fatalf("expected output to contain %q:\n%s", want, value)
|
||||
}
|
||||
}
|
||||
24
internal/render/model.go
Normal file
24
internal/render/model.go
Normal file
@@ -0,0 +1,24 @@
|
||||
package render
|
||||
|
||||
// Transcript is the render-normalized transcript model used by renderers.
|
||||
type Transcript struct {
|
||||
Schema string
|
||||
Metadata Metadata
|
||||
Segments []Segment
|
||||
}
|
||||
|
||||
// Metadata is the render-relevant artifact metadata.
|
||||
type Metadata struct {
|
||||
Application string
|
||||
Version string
|
||||
}
|
||||
|
||||
// Segment is a normalized render segment.
|
||||
type Segment struct {
|
||||
ID int
|
||||
Start float64
|
||||
End float64
|
||||
Speaker string
|
||||
Text string
|
||||
Categories []string
|
||||
}
|
||||
96
internal/render/normalize.go
Normal file
96
internal/render/normalize.go
Normal file
@@ -0,0 +1,96 @@
|
||||
package render
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/artifact"
|
||||
)
|
||||
|
||||
// FromOutputArtifact converts a parsed output artifact into the internal render model.
|
||||
func FromOutputArtifact(input artifact.OutputArtifact) (Transcript, error) {
|
||||
switch input.Schema {
|
||||
case artifact.OutputSchemaFull:
|
||||
payload, err := input.FullPayload()
|
||||
if err != nil {
|
||||
return Transcript{}, err
|
||||
}
|
||||
segments := make([]Segment, len(payload.Segments))
|
||||
for index, segment := range payload.Segments {
|
||||
segments[index] = Segment{
|
||||
ID: segment.ID,
|
||||
Start: segment.Start,
|
||||
End: segment.End,
|
||||
Speaker: segment.Speaker,
|
||||
Text: segment.Text,
|
||||
Categories: normalizeCategories(segment.Categories),
|
||||
}
|
||||
}
|
||||
return Transcript{
|
||||
Schema: input.Schema,
|
||||
Metadata: Metadata{
|
||||
Application: payload.Metadata.Application,
|
||||
Version: payload.Metadata.Version,
|
||||
},
|
||||
Segments: segments,
|
||||
}, nil
|
||||
case artifact.OutputSchemaIntermediate:
|
||||
payload, err := input.IntermediatePayload()
|
||||
if err != nil {
|
||||
return Transcript{}, err
|
||||
}
|
||||
segments := make([]Segment, len(payload.Segments))
|
||||
for index, segment := range payload.Segments {
|
||||
segments[index] = Segment{
|
||||
ID: segment.ID,
|
||||
Start: segment.Start,
|
||||
End: segment.End,
|
||||
Speaker: segment.Speaker,
|
||||
Text: segment.Text,
|
||||
Categories: normalizeCategories(segment.Categories),
|
||||
}
|
||||
}
|
||||
return Transcript{
|
||||
Schema: input.Schema,
|
||||
Metadata: Metadata{
|
||||
Application: payload.Metadata.Application,
|
||||
Version: payload.Metadata.Version,
|
||||
},
|
||||
Segments: segments,
|
||||
}, nil
|
||||
case artifact.OutputSchemaMinimal:
|
||||
payload, err := input.MinimalPayload()
|
||||
if err != nil {
|
||||
return Transcript{}, err
|
||||
}
|
||||
segments := make([]Segment, len(payload.Segments))
|
||||
for index, segment := range payload.Segments {
|
||||
segments[index] = Segment{
|
||||
ID: segment.ID,
|
||||
Start: segment.Start,
|
||||
End: segment.End,
|
||||
Speaker: segment.Speaker,
|
||||
Text: segment.Text,
|
||||
Categories: []string{},
|
||||
}
|
||||
}
|
||||
return Transcript{
|
||||
Schema: input.Schema,
|
||||
Metadata: Metadata{
|
||||
Application: payload.Metadata.Application,
|
||||
Version: payload.Metadata.Version,
|
||||
},
|
||||
Segments: segments,
|
||||
}, nil
|
||||
default:
|
||||
return Transcript{}, fmt.Errorf("unsupported artifact schema %q", input.Schema)
|
||||
}
|
||||
}
|
||||
|
||||
func normalizeCategories(categories []string) []string {
|
||||
if categories == nil {
|
||||
return []string{}
|
||||
}
|
||||
out := make([]string, len(categories))
|
||||
copy(out, categories)
|
||||
return out
|
||||
}
|
||||
139
internal/render/normalize_test.go
Normal file
139
internal/render/normalize_test.go
Normal file
@@ -0,0 +1,139 @@
|
||||
package render
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/artifact"
|
||||
"gitea.maximumdirect.net/eric/seriatim/schema"
|
||||
)
|
||||
|
||||
func TestFromOutputArtifactNormalizesSupportedSchemas(t *testing.T) {
|
||||
t.Run("full", func(t *testing.T) {
|
||||
sourceIndex := 0
|
||||
input := schema.Transcript{
|
||||
Metadata: schema.Metadata{
|
||||
Application: "seriatim",
|
||||
Version: "v-test",
|
||||
InputReader: "json-files",
|
||||
InputFiles: []string{"a.json"},
|
||||
PreprocessingModules: []string{"validate-raw"},
|
||||
PostprocessingModules: []string{"assign-ids", "validate-output"},
|
||||
OutputModules: []string{"json"},
|
||||
},
|
||||
Segments: []schema.Segment{
|
||||
{
|
||||
ID: 1,
|
||||
Source: "a.json",
|
||||
SourceSegmentIndex: &sourceIndex,
|
||||
Speaker: "Alice",
|
||||
Start: 1,
|
||||
End: 2,
|
||||
Text: "hello",
|
||||
Categories: []string{"background"},
|
||||
},
|
||||
},
|
||||
OverlapGroups: []schema.OverlapGroup{},
|
||||
}
|
||||
model := mustNormalizeOutputArtifact(t, input)
|
||||
if model.Schema != artifact.OutputSchemaFull {
|
||||
t.Fatalf("schema = %q, want %q", model.Schema, artifact.OutputSchemaFull)
|
||||
}
|
||||
if len(model.Segments) != 1 {
|
||||
t.Fatalf("segment count = %d, want 1", len(model.Segments))
|
||||
}
|
||||
if model.Segments[0].ID != 1 || model.Segments[0].Speaker != "Alice" || model.Segments[0].Text != "hello" {
|
||||
t.Fatalf("unexpected segment: %#v", model.Segments[0])
|
||||
}
|
||||
if len(model.Segments[0].Categories) != 1 || model.Segments[0].Categories[0] != "background" {
|
||||
t.Fatalf("categories = %#v, want [background]", model.Segments[0].Categories)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("intermediate", func(t *testing.T) {
|
||||
input := schema.IntermediateTranscript{
|
||||
Metadata: schema.IntermediateMetadata{
|
||||
Application: "seriatim",
|
||||
Version: "v-test",
|
||||
OutputSchema: artifact.OutputSchemaIntermediate,
|
||||
},
|
||||
Segments: []schema.IntermediateSegment{
|
||||
{ID: 1, Start: 1, End: 2, Speaker: "Alice", Text: "one", Categories: []string{}},
|
||||
{ID: 2, Start: 2, End: 3, Speaker: "Bob", Text: "two"},
|
||||
},
|
||||
}
|
||||
model := mustNormalizeOutputArtifact(t, input)
|
||||
if model.Schema != artifact.OutputSchemaIntermediate {
|
||||
t.Fatalf("schema = %q, want %q", model.Schema, artifact.OutputSchemaIntermediate)
|
||||
}
|
||||
if len(model.Segments[0].Categories) != 0 {
|
||||
t.Fatalf("segment[0] categories = %#v, want empty slice", model.Segments[0].Categories)
|
||||
}
|
||||
if len(model.Segments[1].Categories) != 0 {
|
||||
t.Fatalf("segment[1] categories = %#v, want empty slice", model.Segments[1].Categories)
|
||||
}
|
||||
if model.Segments[0].Categories == nil || model.Segments[1].Categories == nil {
|
||||
t.Fatal("expected non-nil empty categories slices")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("minimal", func(t *testing.T) {
|
||||
input := schema.MinimalTranscript{
|
||||
Metadata: schema.MinimalMetadata{
|
||||
Application: "seriatim",
|
||||
Version: "v-test",
|
||||
OutputSchema: artifact.OutputSchemaMinimal,
|
||||
},
|
||||
Segments: []schema.MinimalSegment{
|
||||
{ID: 1, Start: 1, End: 2, Speaker: "Alice", Text: "one"},
|
||||
},
|
||||
}
|
||||
model := mustNormalizeOutputArtifact(t, input)
|
||||
if model.Schema != artifact.OutputSchemaMinimal {
|
||||
t.Fatalf("schema = %q, want %q", model.Schema, artifact.OutputSchemaMinimal)
|
||||
}
|
||||
if len(model.Segments[0].Categories) != 0 {
|
||||
t.Fatalf("categories = %#v, want empty slice", model.Segments[0].Categories)
|
||||
}
|
||||
if model.Segments[0].Categories == nil {
|
||||
t.Fatal("expected non-nil empty categories slice")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestFromOutputArtifactRejectsMalformedAndRawInput(t *testing.T) {
|
||||
_, err := artifact.ParseOutputArtifactJSON([]byte(`{"metadata":`))
|
||||
if err == nil {
|
||||
t.Fatal("expected malformed JSON error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "input JSON is malformed") {
|
||||
t.Fatalf("unexpected malformed error: %v", err)
|
||||
}
|
||||
|
||||
rawWhisper := []byte(`{"segments":[{"id":0,"start":0.1,"end":1.2,"text":"hello","words":[{"word":"hello"}]}]}`)
|
||||
_, err = artifact.ParseOutputArtifactJSON(rawWhisper)
|
||||
if err == nil {
|
||||
t.Fatal("expected raw input artifact error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "not a valid seriatim output artifact") {
|
||||
t.Fatalf("unexpected raw input error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func mustNormalizeOutputArtifact(t *testing.T, value any) Transcript {
|
||||
t.Helper()
|
||||
data, err := json.Marshal(value)
|
||||
if err != nil {
|
||||
t.Fatalf("marshal: %v", err)
|
||||
}
|
||||
parsed, err := artifact.ParseOutputArtifactJSON(data)
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
model, err := FromOutputArtifact(parsed)
|
||||
if err != nil {
|
||||
t.Fatalf("normalize: %v", err)
|
||||
}
|
||||
return model
|
||||
}
|
||||
41
internal/render/registry.go
Normal file
41
internal/render/registry.go
Normal file
@@ -0,0 +1,41 @@
|
||||
package render
|
||||
|
||||
import "fmt"
|
||||
|
||||
const FormatMarkdown = "markdown"
|
||||
|
||||
// Options configures rendering behavior across formats.
|
||||
type Options struct {
|
||||
Title string
|
||||
IncludeTimestamps bool
|
||||
IncludeSegmentIDs bool
|
||||
IncludeMetadata bool
|
||||
}
|
||||
|
||||
// Renderer turns a normalized render model into text output.
|
||||
type Renderer interface {
|
||||
Render(transcript Transcript, opts Options) (string, error)
|
||||
}
|
||||
|
||||
// Registry resolves renderers by public format name.
|
||||
type Registry struct {
|
||||
renderers map[string]Renderer
|
||||
}
|
||||
|
||||
// NewRegistry returns a renderer registry with built-in renderers.
|
||||
func NewRegistry() Registry {
|
||||
return Registry{
|
||||
renderers: map[string]Renderer{
|
||||
FormatMarkdown: MarkdownRenderer{},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// Resolve resolves a renderer by format name.
|
||||
func (registry Registry) Resolve(format string) (Renderer, error) {
|
||||
renderer, ok := registry.renderers[format]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("unsupported --format %q", format)
|
||||
}
|
||||
return renderer, nil
|
||||
}
|
||||
22
internal/render/registry_test.go
Normal file
22
internal/render/registry_test.go
Normal file
@@ -0,0 +1,22 @@
|
||||
package render
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestRegistryResolvesMarkdownRenderer(t *testing.T) {
|
||||
registry := NewRegistry()
|
||||
renderer, err := registry.Resolve(FormatMarkdown)
|
||||
if err != nil {
|
||||
t.Fatalf("resolve markdown renderer: %v", err)
|
||||
}
|
||||
if renderer == nil {
|
||||
t.Fatal("expected renderer")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegistryRejectsUnknownRenderer(t *testing.T) {
|
||||
registry := NewRegistry()
|
||||
_, err := registry.Resolve("txt")
|
||||
if err == nil {
|
||||
t.Fatal("expected unsupported format error")
|
||||
}
|
||||
}
|
||||
72
internal/render/run.go
Normal file
72
internal/render/run.go
Normal file
@@ -0,0 +1,72 @@
|
||||
package render
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/artifact"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
)
|
||||
|
||||
// Run executes artifact-level render orchestration.
|
||||
func Run(ctx context.Context, cfg config.RenderConfig) error {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
data, err := os.ReadFile(cfg.InputFile)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read --input-file %q: %w", cfg.InputFile, err)
|
||||
}
|
||||
|
||||
inputArtifact, err := artifact.ParseOutputArtifactJSON(data)
|
||||
if err != nil {
|
||||
return fmt.Errorf("--input-file %q: %w", cfg.InputFile, err)
|
||||
}
|
||||
|
||||
model, err := FromOutputArtifact(inputArtifact)
|
||||
if err != nil {
|
||||
return fmt.Errorf("normalize artifact for render: %w", err)
|
||||
}
|
||||
|
||||
registry := NewRegistry()
|
||||
renderer, err := registry.Resolve(cfg.Format)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
rendered, err := renderer.Render(model, Options{
|
||||
Title: cfg.Title,
|
||||
IncludeTimestamps: cfg.IncludeTimestamps,
|
||||
IncludeSegmentIDs: cfg.IncludeSegmentIDs,
|
||||
IncludeMetadata: cfg.IncludeMetadata,
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("render %q output: %w", cfg.Format, err)
|
||||
}
|
||||
|
||||
if err := writeFile(cfg.OutputFile, rendered); err != nil {
|
||||
return fmt.Errorf("write --output-file %q: %w", cfg.OutputFile, err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func writeFile(path string, content string) (err error) {
|
||||
file, err := os.Create(path)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create %q: %w", path, err)
|
||||
}
|
||||
defer func() {
|
||||
closeErr := file.Close()
|
||||
if err == nil && closeErr != nil {
|
||||
err = fmt.Errorf("close %q: %w", path, closeErr)
|
||||
}
|
||||
}()
|
||||
|
||||
if _, err := file.WriteString(content); err != nil {
|
||||
return fmt.Errorf("write %q: %w", path, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -1,9 +1,6 @@
|
||||
package report
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"os"
|
||||
)
|
||||
import "gitea.maximumdirect.net/eric/seriatim/internal/jsonfile"
|
||||
|
||||
// Severity classifies report events.
|
||||
type Severity string
|
||||
@@ -62,13 +59,5 @@ func Warning(stage string, module string, message string) Event {
|
||||
|
||||
// WriteJSON writes a deterministic JSON report.
|
||||
func WriteJSON(path string, rpt Report) error {
|
||||
file, err := os.Create(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
enc := json.NewEncoder(file)
|
||||
enc.SetIndent("", " ")
|
||||
return enc.Encode(rpt)
|
||||
return jsonfile.Write(path, rpt)
|
||||
}
|
||||
|
||||
@@ -44,54 +44,29 @@ type MinimalResult struct {
|
||||
RemovedIDs []int
|
||||
}
|
||||
|
||||
type projection struct {
|
||||
retainedIndexes []int
|
||||
oldToNewID map[int]int
|
||||
removedIDs []int
|
||||
}
|
||||
|
||||
// Apply trims a full seriatim output transcript by segment ID.
|
||||
func Apply(input schema.Transcript, opts Options) (Result, error) {
|
||||
if err := validateMode(opts.Mode); err != nil {
|
||||
return Result{}, err
|
||||
}
|
||||
|
||||
selected := opts.Selector.IDs()
|
||||
if len(selected) == 0 {
|
||||
return Result{}, fmt.Errorf("selector cannot be empty")
|
||||
}
|
||||
|
||||
inputIDs := make([]int, len(input.Segments))
|
||||
for index, segment := range input.Segments {
|
||||
inputIDs[index] = segment.ID
|
||||
}
|
||||
|
||||
idIndex, err := validateInputIDs(inputIDs)
|
||||
proj, err := projectSegmentIDs(inputIDs, opts)
|
||||
if err != nil {
|
||||
return Result{}, err
|
||||
}
|
||||
|
||||
if err := validateSelectedIDsExist(selected, idIndex); err != nil {
|
||||
return Result{}, err
|
||||
}
|
||||
|
||||
kept := make([]schema.Segment, 0, len(input.Segments))
|
||||
removed := make([]int, 0, len(input.Segments))
|
||||
oldToNew := make(map[int]int, len(input.Segments))
|
||||
for _, segment := range input.Segments {
|
||||
keep := opts.Mode == ModeKeep && opts.Selector.Contains(segment.ID)
|
||||
if opts.Mode == ModeRemove {
|
||||
keep = !opts.Selector.Contains(segment.ID)
|
||||
}
|
||||
|
||||
if !keep {
|
||||
removed = append(removed, segment.ID)
|
||||
continue
|
||||
}
|
||||
|
||||
rewritten := copySegment(segment)
|
||||
rewritten.ID = len(kept) + 1
|
||||
kept := make([]schema.Segment, len(proj.retainedIndexes))
|
||||
for outputIndex, inputIndex := range proj.retainedIndexes {
|
||||
rewritten := copySegment(input.Segments[inputIndex])
|
||||
rewritten.ID = outputIndex + 1
|
||||
rewritten.OverlapGroupID = 0
|
||||
kept = append(kept, rewritten)
|
||||
oldToNew[segment.ID] = rewritten.ID
|
||||
}
|
||||
|
||||
if len(kept) == 0 && !opts.AllowEmpty {
|
||||
return Result{}, fmt.Errorf("trim operation produced an empty transcript; set AllowEmpty to true to permit this")
|
||||
kept[outputIndex] = rewritten
|
||||
}
|
||||
|
||||
kept, groups := recomputeOverlapGroups(kept)
|
||||
@@ -104,62 +79,35 @@ func Apply(input schema.Transcript, opts Options) (Result, error) {
|
||||
out.OverlapGroups = groups
|
||||
return Result{
|
||||
Transcript: out,
|
||||
OldToNewID: oldToNew,
|
||||
RemovedIDs: removed,
|
||||
OldToNewID: proj.oldToNewID,
|
||||
RemovedIDs: proj.removedIDs,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// ApplyIntermediate trims an intermediate seriatim output transcript by
|
||||
// segment ID.
|
||||
func ApplyIntermediate(input schema.IntermediateTranscript, opts Options) (IntermediateResult, error) {
|
||||
if err := validateMode(opts.Mode); err != nil {
|
||||
return IntermediateResult{}, err
|
||||
}
|
||||
|
||||
selected := opts.Selector.IDs()
|
||||
if len(selected) == 0 {
|
||||
return IntermediateResult{}, fmt.Errorf("selector cannot be empty")
|
||||
}
|
||||
|
||||
inputIDs := make([]int, len(input.Segments))
|
||||
for index, segment := range input.Segments {
|
||||
inputIDs[index] = segment.ID
|
||||
}
|
||||
idIndex, err := validateInputIDs(inputIDs)
|
||||
proj, err := projectSegmentIDs(inputIDs, opts)
|
||||
if err != nil {
|
||||
return IntermediateResult{}, err
|
||||
}
|
||||
if err := validateSelectedIDsExist(selected, idIndex); err != nil {
|
||||
return IntermediateResult{}, err
|
||||
}
|
||||
|
||||
kept := make([]schema.IntermediateSegment, 0, len(input.Segments))
|
||||
removed := make([]int, 0, len(input.Segments))
|
||||
oldToNew := make(map[int]int, len(input.Segments))
|
||||
for _, segment := range input.Segments {
|
||||
keep := opts.Mode == ModeKeep && opts.Selector.Contains(segment.ID)
|
||||
if opts.Mode == ModeRemove {
|
||||
keep = !opts.Selector.Contains(segment.ID)
|
||||
}
|
||||
if !keep {
|
||||
removed = append(removed, segment.ID)
|
||||
continue
|
||||
}
|
||||
|
||||
kept := make([]schema.IntermediateSegment, len(proj.retainedIndexes))
|
||||
for outputIndex, inputIndex := range proj.retainedIndexes {
|
||||
segment := input.Segments[inputIndex]
|
||||
rewritten := schema.IntermediateSegment{
|
||||
ID: len(kept) + 1,
|
||||
ID: outputIndex + 1,
|
||||
Start: segment.Start,
|
||||
End: segment.End,
|
||||
Speaker: segment.Speaker,
|
||||
Text: segment.Text,
|
||||
Categories: append([]string(nil), segment.Categories...),
|
||||
}
|
||||
kept = append(kept, rewritten)
|
||||
oldToNew[segment.ID] = rewritten.ID
|
||||
}
|
||||
|
||||
if len(kept) == 0 && !opts.AllowEmpty {
|
||||
return IntermediateResult{}, fmt.Errorf("trim operation produced an empty transcript; set AllowEmpty to true to permit this")
|
||||
kept[outputIndex] = rewritten
|
||||
}
|
||||
|
||||
return IntermediateResult{
|
||||
@@ -171,60 +119,33 @@ func ApplyIntermediate(input schema.IntermediateTranscript, opts Options) (Inter
|
||||
},
|
||||
Segments: kept,
|
||||
},
|
||||
OldToNewID: oldToNew,
|
||||
RemovedIDs: removed,
|
||||
OldToNewID: proj.oldToNewID,
|
||||
RemovedIDs: proj.removedIDs,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// ApplyMinimal trims a minimal seriatim output transcript by segment ID.
|
||||
func ApplyMinimal(input schema.MinimalTranscript, opts Options) (MinimalResult, error) {
|
||||
if err := validateMode(opts.Mode); err != nil {
|
||||
return MinimalResult{}, err
|
||||
}
|
||||
|
||||
selected := opts.Selector.IDs()
|
||||
if len(selected) == 0 {
|
||||
return MinimalResult{}, fmt.Errorf("selector cannot be empty")
|
||||
}
|
||||
|
||||
inputIDs := make([]int, len(input.Segments))
|
||||
for index, segment := range input.Segments {
|
||||
inputIDs[index] = segment.ID
|
||||
}
|
||||
idIndex, err := validateInputIDs(inputIDs)
|
||||
proj, err := projectSegmentIDs(inputIDs, opts)
|
||||
if err != nil {
|
||||
return MinimalResult{}, err
|
||||
}
|
||||
if err := validateSelectedIDsExist(selected, idIndex); err != nil {
|
||||
return MinimalResult{}, err
|
||||
}
|
||||
|
||||
kept := make([]schema.MinimalSegment, 0, len(input.Segments))
|
||||
removed := make([]int, 0, len(input.Segments))
|
||||
oldToNew := make(map[int]int, len(input.Segments))
|
||||
for _, segment := range input.Segments {
|
||||
keep := opts.Mode == ModeKeep && opts.Selector.Contains(segment.ID)
|
||||
if opts.Mode == ModeRemove {
|
||||
keep = !opts.Selector.Contains(segment.ID)
|
||||
}
|
||||
if !keep {
|
||||
removed = append(removed, segment.ID)
|
||||
continue
|
||||
}
|
||||
|
||||
kept := make([]schema.MinimalSegment, len(proj.retainedIndexes))
|
||||
for outputIndex, inputIndex := range proj.retainedIndexes {
|
||||
segment := input.Segments[inputIndex]
|
||||
rewritten := schema.MinimalSegment{
|
||||
ID: len(kept) + 1,
|
||||
ID: outputIndex + 1,
|
||||
Start: segment.Start,
|
||||
End: segment.End,
|
||||
Speaker: segment.Speaker,
|
||||
Text: segment.Text,
|
||||
}
|
||||
kept = append(kept, rewritten)
|
||||
oldToNew[segment.ID] = rewritten.ID
|
||||
}
|
||||
|
||||
if len(kept) == 0 && !opts.AllowEmpty {
|
||||
return MinimalResult{}, fmt.Errorf("trim operation produced an empty transcript; set AllowEmpty to true to permit this")
|
||||
kept[outputIndex] = rewritten
|
||||
}
|
||||
|
||||
return MinimalResult{
|
||||
@@ -236,11 +157,53 @@ func ApplyMinimal(input schema.MinimalTranscript, opts Options) (MinimalResult,
|
||||
},
|
||||
Segments: kept,
|
||||
},
|
||||
OldToNewID: oldToNew,
|
||||
RemovedIDs: removed,
|
||||
OldToNewID: proj.oldToNewID,
|
||||
RemovedIDs: proj.removedIDs,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func projectSegmentIDs(ids []int, opts Options) (projection, error) {
|
||||
if err := validateMode(opts.Mode); err != nil {
|
||||
return projection{}, err
|
||||
}
|
||||
|
||||
selected := opts.Selector.IDs()
|
||||
if len(selected) == 0 {
|
||||
return projection{}, fmt.Errorf("selector cannot be empty")
|
||||
}
|
||||
|
||||
idIndex, err := validateInputIDs(ids)
|
||||
if err != nil {
|
||||
return projection{}, err
|
||||
}
|
||||
if err := validateSelectedIDsExist(selected, idIndex); err != nil {
|
||||
return projection{}, err
|
||||
}
|
||||
|
||||
result := projection{
|
||||
retainedIndexes: make([]int, 0, len(ids)),
|
||||
oldToNewID: make(map[int]int, len(ids)),
|
||||
removedIDs: make([]int, 0, len(ids)),
|
||||
}
|
||||
for index, id := range ids {
|
||||
keep := opts.Mode == ModeKeep && opts.Selector.Contains(id)
|
||||
if opts.Mode == ModeRemove {
|
||||
keep = !opts.Selector.Contains(id)
|
||||
}
|
||||
if !keep {
|
||||
result.removedIDs = append(result.removedIDs, id)
|
||||
continue
|
||||
}
|
||||
result.retainedIndexes = append(result.retainedIndexes, index)
|
||||
result.oldToNewID[id] = len(result.retainedIndexes)
|
||||
}
|
||||
|
||||
if len(result.retainedIndexes) == 0 && !opts.AllowEmpty {
|
||||
return projection{}, fmt.Errorf("trim operation produced an empty transcript; set AllowEmpty to true to permit this")
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func validateMode(mode Mode) error {
|
||||
switch mode {
|
||||
case ModeKeep, ModeRemove:
|
||||
|
||||
@@ -399,6 +399,106 @@ func TestApplyMinimalDoesNotIncludeOverlapGroups(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplySelectorPolicyIsSharedAcrossSchemas(t *testing.T) {
|
||||
type testCase struct {
|
||||
name string
|
||||
opts Options
|
||||
wantTexts []string
|
||||
wantOldToNew map[int]int
|
||||
wantRemoved []int
|
||||
wantSegmentCount int
|
||||
wantErrorSubstring string
|
||||
}
|
||||
|
||||
cases := []testCase{
|
||||
{
|
||||
name: "keep preserves input order regardless of selector order",
|
||||
opts: Options{Mode: ModeKeep, Selector: mustParseSelector(t, "4,1,3")},
|
||||
wantTexts: []string{"alpha", "gamma", "delta"},
|
||||
wantOldToNew: map[int]int{1: 1, 3: 2, 4: 3},
|
||||
wantRemoved: []int{2},
|
||||
wantSegmentCount: 3,
|
||||
},
|
||||
{
|
||||
name: "remove reports deterministic renumbering metadata",
|
||||
opts: Options{Mode: ModeRemove, Selector: mustParseSelector(t, "2,4")},
|
||||
wantTexts: []string{"alpha", "gamma"},
|
||||
wantOldToNew: map[int]int{1: 1, 3: 2},
|
||||
wantRemoved: []int{2, 4},
|
||||
wantSegmentCount: 2,
|
||||
},
|
||||
{
|
||||
name: "missing selected id returns error",
|
||||
opts: Options{Mode: ModeKeep, Selector: mustParseSelector(t, "9")},
|
||||
wantErrorSubstring: "does not exist",
|
||||
},
|
||||
{
|
||||
name: "empty selector returns error",
|
||||
opts: Options{Mode: ModeKeep, Selector: Selector{}},
|
||||
wantErrorSubstring: "selector cannot be empty",
|
||||
},
|
||||
{
|
||||
name: "invalid mode returns error",
|
||||
opts: Options{Mode: Mode("bad"), Selector: mustParseSelector(t, "1")},
|
||||
wantErrorSubstring: `invalid trim mode "bad"`,
|
||||
},
|
||||
{
|
||||
name: "empty output blocked when allow empty is false",
|
||||
opts: Options{Mode: ModeRemove, Selector: mustParseSelector(t, "1-4")},
|
||||
wantErrorSubstring: "empty transcript",
|
||||
},
|
||||
{
|
||||
name: "empty output allowed when allow empty is true",
|
||||
opts: Options{Mode: ModeRemove, Selector: mustParseSelector(t, "1-4"), AllowEmpty: true},
|
||||
wantTexts: []string{},
|
||||
wantOldToNew: map[int]int{},
|
||||
wantRemoved: []int{1, 2, 3, 4},
|
||||
wantSegmentCount: 0,
|
||||
},
|
||||
}
|
||||
|
||||
for _, test := range cases {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
fullInput := fullTranscriptFixture()
|
||||
intermediateInput := intermediateFixture()
|
||||
minimalInput := minimalFixture()
|
||||
|
||||
fullResult, fullErr := Apply(fullInput, test.opts)
|
||||
intermediateResult, intermediateErr := ApplyIntermediate(intermediateInput, test.opts)
|
||||
minimalResult, minimalErr := ApplyMinimal(minimalInput, test.opts)
|
||||
|
||||
if test.wantErrorSubstring != "" {
|
||||
assertErrorContains(t, fullErr, test.wantErrorSubstring)
|
||||
assertErrorContains(t, intermediateErr, test.wantErrorSubstring)
|
||||
assertErrorContains(t, minimalErr, test.wantErrorSubstring)
|
||||
return
|
||||
}
|
||||
if fullErr != nil {
|
||||
t.Fatalf("apply full failed: %v", fullErr)
|
||||
}
|
||||
if intermediateErr != nil {
|
||||
t.Fatalf("apply intermediate failed: %v", intermediateErr)
|
||||
}
|
||||
if minimalErr != nil {
|
||||
t.Fatalf("apply minimal failed: %v", minimalErr)
|
||||
}
|
||||
|
||||
assertIntSlice(t, extractFullIDs(fullResult.Transcript.Segments), extractSequentialIDs(test.wantSegmentCount))
|
||||
assertIntSlice(t, extractIntermediateIDs(intermediateResult.Transcript.Segments), extractSequentialIDs(test.wantSegmentCount))
|
||||
assertIntSlice(t, extractMinimalIDs(minimalResult.Transcript.Segments), extractSequentialIDs(test.wantSegmentCount))
|
||||
assertStringSlice(t, extractFullTexts(fullResult.Transcript.Segments), test.wantTexts)
|
||||
assertStringSlice(t, extractIntermediateTexts(intermediateResult.Transcript.Segments), test.wantTexts)
|
||||
assertStringSlice(t, extractMinimalTexts(minimalResult.Transcript.Segments), test.wantTexts)
|
||||
assertIntMap(t, fullResult.OldToNewID, test.wantOldToNew)
|
||||
assertIntMap(t, intermediateResult.OldToNewID, test.wantOldToNew)
|
||||
assertIntMap(t, minimalResult.OldToNewID, test.wantOldToNew)
|
||||
assertIntSlice(t, fullResult.RemovedIDs, test.wantRemoved)
|
||||
assertIntSlice(t, intermediateResult.RemovedIDs, test.wantRemoved)
|
||||
assertIntSlice(t, minimalResult.RemovedIDs, test.wantRemoved)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyOutputInvariantsValidAfterRenumberAndOverlapRecompute(t *testing.T) {
|
||||
input := overlapTranscriptFixture()
|
||||
selector := mustParseSelector(t, "2,1")
|
||||
@@ -666,3 +766,108 @@ func equalStringSlices(got []string, want []string) bool {
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func assertErrorContains(t *testing.T, err error, substring string) {
|
||||
t.Helper()
|
||||
if err == nil {
|
||||
t.Fatalf("expected error containing %q", substring)
|
||||
}
|
||||
if !strings.Contains(err.Error(), substring) {
|
||||
t.Fatalf("error %q does not contain %q", err.Error(), substring)
|
||||
}
|
||||
}
|
||||
|
||||
func assertStringSlice(t *testing.T, got []string, want []string) {
|
||||
t.Helper()
|
||||
if !equalStringSlices(got, want) {
|
||||
t.Fatalf("slice = %v, want %v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func extractSequentialIDs(count int) []int {
|
||||
ids := make([]int, count)
|
||||
for index := range ids {
|
||||
ids[index] = index + 1
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
func extractFullIDs(segments []schema.Segment) []int {
|
||||
ids := make([]int, len(segments))
|
||||
for index, segment := range segments {
|
||||
ids[index] = segment.ID
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
func extractIntermediateIDs(segments []schema.IntermediateSegment) []int {
|
||||
ids := make([]int, len(segments))
|
||||
for index, segment := range segments {
|
||||
ids[index] = segment.ID
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
func extractMinimalIDs(segments []schema.MinimalSegment) []int {
|
||||
ids := make([]int, len(segments))
|
||||
for index, segment := range segments {
|
||||
ids[index] = segment.ID
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
func extractFullTexts(segments []schema.Segment) []string {
|
||||
texts := make([]string, len(segments))
|
||||
for index, segment := range segments {
|
||||
texts[index] = segment.Text
|
||||
}
|
||||
return texts
|
||||
}
|
||||
|
||||
func extractIntermediateTexts(segments []schema.IntermediateSegment) []string {
|
||||
texts := make([]string, len(segments))
|
||||
for index, segment := range segments {
|
||||
texts[index] = segment.Text
|
||||
}
|
||||
return texts
|
||||
}
|
||||
|
||||
func extractMinimalTexts(segments []schema.MinimalSegment) []string {
|
||||
texts := make([]string, len(segments))
|
||||
for index, segment := range segments {
|
||||
texts[index] = segment.Text
|
||||
}
|
||||
return texts
|
||||
}
|
||||
|
||||
func intermediateFixture() schema.IntermediateTranscript {
|
||||
return schema.IntermediateTranscript{
|
||||
Metadata: schema.IntermediateMetadata{
|
||||
Application: "seriatim",
|
||||
Version: "v-test",
|
||||
OutputSchema: schema.OutputSchemaIntermediate,
|
||||
},
|
||||
Segments: []schema.IntermediateSegment{
|
||||
{ID: 1, Start: 1, End: 2, Speaker: "Alice", Text: "alpha", Categories: []string{"word-run"}},
|
||||
{ID: 2, Start: 2, End: 3, Speaker: "Bob", Text: "beta", Categories: []string{"filler", "backchannel"}},
|
||||
{ID: 3, Start: 3, End: 4, Speaker: "Carol", Text: "gamma", Categories: []string{"normal"}},
|
||||
{ID: 4, Start: 4, End: 5, Speaker: "Dan", Text: "delta", Categories: []string{"normal"}},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func minimalFixture() schema.MinimalTranscript {
|
||||
return schema.MinimalTranscript{
|
||||
Metadata: schema.MinimalMetadata{
|
||||
Application: "seriatim",
|
||||
Version: "v-test",
|
||||
OutputSchema: schema.OutputSchemaMinimal,
|
||||
},
|
||||
Segments: []schema.MinimalSegment{
|
||||
{ID: 1, Start: 1, End: 2, Speaker: "Alice", Text: "alpha"},
|
||||
{ID: 2, Start: 2, End: 3, Speaker: "Bob", Text: "beta"},
|
||||
{ID: 3, Start: 3, End: 4, Speaker: "Carol", Text: "gamma"},
|
||||
{ID: 4, Start: 4, End: 5, Speaker: "Dan", Text: "delta"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,16 +1,16 @@
|
||||
package trim
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
|
||||
artifactpkg "gitea.maximumdirect.net/eric/seriatim/internal/artifact"
|
||||
"gitea.maximumdirect.net/eric/seriatim/schema"
|
||||
)
|
||||
|
||||
const (
|
||||
SchemaMinimal = "seriatim-minimal"
|
||||
SchemaIntermediate = "seriatim-intermediate"
|
||||
SchemaFull = "seriatim-full"
|
||||
SchemaMinimal = artifactpkg.OutputSchemaMinimal
|
||||
SchemaIntermediate = artifactpkg.OutputSchemaIntermediate
|
||||
SchemaFull = artifactpkg.OutputSchemaFull
|
||||
)
|
||||
|
||||
// Artifact stores a parsed seriatim output artifact of one supported schema.
|
||||
@@ -31,62 +31,39 @@ type ApplyArtifactResult struct {
|
||||
|
||||
// ParseArtifactJSON parses and validates a serialized seriatim output artifact.
|
||||
func ParseArtifactJSON(data []byte) (Artifact, error) {
|
||||
var decoded any
|
||||
if err := json.Unmarshal(data, &decoded); err != nil {
|
||||
return Artifact{}, fmt.Errorf("input JSON is malformed: %w", err)
|
||||
parsed, err := artifactpkg.ParseOutputArtifactJSON(data)
|
||||
if err != nil {
|
||||
return Artifact{}, err
|
||||
}
|
||||
|
||||
var full schema.Transcript
|
||||
if err := json.Unmarshal(data, &full); err == nil {
|
||||
if err := schema.ValidateTranscript(full); err == nil {
|
||||
return Artifact{
|
||||
Schema: SchemaFull,
|
||||
Full: &full,
|
||||
}, nil
|
||||
}
|
||||
}
|
||||
|
||||
var intermediate schema.IntermediateTranscript
|
||||
if err := json.Unmarshal(data, &intermediate); err == nil {
|
||||
if err := schema.ValidateIntermediateTranscript(intermediate); err == nil {
|
||||
return Artifact{
|
||||
Schema: SchemaIntermediate,
|
||||
Intermediate: &intermediate,
|
||||
}, nil
|
||||
}
|
||||
}
|
||||
|
||||
var minimal schema.MinimalTranscript
|
||||
if err := json.Unmarshal(data, &minimal); err == nil {
|
||||
if err := schema.ValidateMinimalTranscript(minimal); err == nil {
|
||||
return Artifact{
|
||||
Schema: SchemaMinimal,
|
||||
Minimal: &minimal,
|
||||
}, nil
|
||||
}
|
||||
}
|
||||
|
||||
return Artifact{}, fmt.Errorf("input JSON is not a valid seriatim output artifact")
|
||||
return Artifact{
|
||||
Schema: parsed.Schema,
|
||||
Full: parsed.Full,
|
||||
Intermediate: parsed.Intermediate,
|
||||
Minimal: parsed.Minimal,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// ValidateArtifact validates an artifact against its declared schema.
|
||||
func ValidateArtifact(artifact Artifact) error {
|
||||
switch artifact.Schema {
|
||||
case SchemaFull:
|
||||
if artifact.Full == nil {
|
||||
return fmt.Errorf("full artifact payload is missing")
|
||||
payload, err := artifact.fullPayload()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return schema.ValidateTranscript(*artifact.Full)
|
||||
return schema.ValidateTranscript(*payload)
|
||||
case SchemaIntermediate:
|
||||
if artifact.Intermediate == nil {
|
||||
return fmt.Errorf("intermediate artifact payload is missing")
|
||||
payload, err := artifact.intermediatePayload()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return schema.ValidateIntermediateTranscript(*artifact.Intermediate)
|
||||
return schema.ValidateIntermediateTranscript(*payload)
|
||||
case SchemaMinimal:
|
||||
if artifact.Minimal == nil {
|
||||
return fmt.Errorf("minimal artifact payload is missing")
|
||||
payload, err := artifact.minimalPayload()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return schema.ValidateMinimalTranscript(*artifact.Minimal)
|
||||
return schema.ValidateMinimalTranscript(*payload)
|
||||
default:
|
||||
return fmt.Errorf("unsupported artifact schema %q", artifact.Schema)
|
||||
}
|
||||
@@ -188,10 +165,11 @@ func (artifact Artifact) Version() string {
|
||||
func ApplyArtifact(input Artifact, opts Options) (ApplyArtifactResult, error) {
|
||||
switch input.Schema {
|
||||
case SchemaFull:
|
||||
if input.Full == nil {
|
||||
return ApplyArtifactResult{}, fmt.Errorf("full artifact payload is missing")
|
||||
payload, err := input.fullPayload()
|
||||
if err != nil {
|
||||
return ApplyArtifactResult{}, err
|
||||
}
|
||||
result, err := Apply(*input.Full, opts)
|
||||
result, err := Apply(*payload, opts)
|
||||
if err != nil {
|
||||
return ApplyArtifactResult{}, err
|
||||
}
|
||||
@@ -206,10 +184,11 @@ func ApplyArtifact(input Artifact, opts Options) (ApplyArtifactResult, error) {
|
||||
OverlapGroupsRecomputed: true,
|
||||
}, nil
|
||||
case SchemaIntermediate:
|
||||
if input.Intermediate == nil {
|
||||
return ApplyArtifactResult{}, fmt.Errorf("intermediate artifact payload is missing")
|
||||
payload, err := input.intermediatePayload()
|
||||
if err != nil {
|
||||
return ApplyArtifactResult{}, err
|
||||
}
|
||||
result, err := ApplyIntermediate(*input.Intermediate, opts)
|
||||
result, err := ApplyIntermediate(*payload, opts)
|
||||
if err != nil {
|
||||
return ApplyArtifactResult{}, err
|
||||
}
|
||||
@@ -224,10 +203,11 @@ func ApplyArtifact(input Artifact, opts Options) (ApplyArtifactResult, error) {
|
||||
OverlapGroupsRecomputed: false,
|
||||
}, nil
|
||||
case SchemaMinimal:
|
||||
if input.Minimal == nil {
|
||||
return ApplyArtifactResult{}, fmt.Errorf("minimal artifact payload is missing")
|
||||
payload, err := input.minimalPayload()
|
||||
if err != nil {
|
||||
return ApplyArtifactResult{}, err
|
||||
}
|
||||
result, err := ApplyMinimal(*input.Minimal, opts)
|
||||
result, err := ApplyMinimal(*payload, opts)
|
||||
if err != nil {
|
||||
return ApplyArtifactResult{}, err
|
||||
}
|
||||
@@ -254,18 +234,19 @@ func ConvertArtifact(input Artifact, outputSchema string) (Artifact, error) {
|
||||
|
||||
switch input.Schema {
|
||||
case SchemaFull:
|
||||
if input.Full == nil {
|
||||
return Artifact{}, fmt.Errorf("full artifact payload is missing")
|
||||
payload, err := input.fullPayload()
|
||||
if err != nil {
|
||||
return Artifact{}, err
|
||||
}
|
||||
switch outputSchema {
|
||||
case SchemaIntermediate:
|
||||
out := intermediateFromFull(*input.Full)
|
||||
out := intermediateFromFull(*payload)
|
||||
return Artifact{
|
||||
Schema: SchemaIntermediate,
|
||||
Intermediate: &out,
|
||||
}, nil
|
||||
case SchemaMinimal:
|
||||
out := minimalFromFull(*input.Full)
|
||||
out := minimalFromFull(*payload)
|
||||
return Artifact{
|
||||
Schema: SchemaMinimal,
|
||||
Minimal: &out,
|
||||
@@ -274,12 +255,13 @@ func ConvertArtifact(input Artifact, outputSchema string) (Artifact, error) {
|
||||
return Artifact{}, fmt.Errorf("unsupported output schema %q", outputSchema)
|
||||
}
|
||||
case SchemaIntermediate:
|
||||
if input.Intermediate == nil {
|
||||
return Artifact{}, fmt.Errorf("intermediate artifact payload is missing")
|
||||
payload, err := input.intermediatePayload()
|
||||
if err != nil {
|
||||
return Artifact{}, err
|
||||
}
|
||||
switch outputSchema {
|
||||
case SchemaMinimal:
|
||||
out := minimalFromIntermediate(*input.Intermediate)
|
||||
out := minimalFromIntermediate(*payload)
|
||||
return Artifact{
|
||||
Schema: SchemaMinimal,
|
||||
Minimal: &out,
|
||||
@@ -290,12 +272,13 @@ func ConvertArtifact(input Artifact, outputSchema string) (Artifact, error) {
|
||||
return Artifact{}, fmt.Errorf("unsupported output schema %q", outputSchema)
|
||||
}
|
||||
case SchemaMinimal:
|
||||
if input.Minimal == nil {
|
||||
return Artifact{}, fmt.Errorf("minimal artifact payload is missing")
|
||||
payload, err := input.minimalPayload()
|
||||
if err != nil {
|
||||
return Artifact{}, err
|
||||
}
|
||||
switch outputSchema {
|
||||
case SchemaIntermediate:
|
||||
out := intermediateFromMinimal(*input.Minimal)
|
||||
out := intermediateFromMinimal(*payload)
|
||||
return Artifact{
|
||||
Schema: SchemaIntermediate,
|
||||
Intermediate: &out,
|
||||
@@ -310,6 +293,27 @@ func ConvertArtifact(input Artifact, outputSchema string) (Artifact, error) {
|
||||
}
|
||||
}
|
||||
|
||||
func (artifact Artifact) fullPayload() (*schema.Transcript, error) {
|
||||
if artifact.Full == nil {
|
||||
return nil, fmt.Errorf("full artifact payload is missing")
|
||||
}
|
||||
return artifact.Full, nil
|
||||
}
|
||||
|
||||
func (artifact Artifact) intermediatePayload() (*schema.IntermediateTranscript, error) {
|
||||
if artifact.Intermediate == nil {
|
||||
return nil, fmt.Errorf("intermediate artifact payload is missing")
|
||||
}
|
||||
return artifact.Intermediate, nil
|
||||
}
|
||||
|
||||
func (artifact Artifact) minimalPayload() (*schema.MinimalTranscript, error) {
|
||||
if artifact.Minimal == nil {
|
||||
return nil, fmt.Errorf("minimal artifact payload is missing")
|
||||
}
|
||||
return artifact.Minimal, nil
|
||||
}
|
||||
|
||||
func intermediateFromFull(input schema.Transcript) schema.IntermediateTranscript {
|
||||
segments := make([]schema.IntermediateSegment, len(input.Segments))
|
||||
for index, segment := range input.Segments {
|
||||
|
||||
@@ -128,6 +128,61 @@ func TestConvertArtifactMinimalToFullFails(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidateArtifactRejectsMissingPayloads(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
artifact Artifact
|
||||
want string
|
||||
}{
|
||||
{
|
||||
name: "full",
|
||||
artifact: Artifact{Schema: SchemaFull},
|
||||
want: "full artifact payload is missing",
|
||||
},
|
||||
{
|
||||
name: "intermediate",
|
||||
artifact: Artifact{Schema: SchemaIntermediate},
|
||||
want: "intermediate artifact payload is missing",
|
||||
},
|
||||
{
|
||||
name: "minimal",
|
||||
artifact: Artifact{Schema: SchemaMinimal},
|
||||
want: "minimal artifact payload is missing",
|
||||
},
|
||||
}
|
||||
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
err := ValidateArtifact(test.artifact)
|
||||
assertErrorContains(t, err, test.want)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyArtifactRejectsMissingPayload(t *testing.T) {
|
||||
_, err := ApplyArtifact(Artifact{Schema: SchemaFull}, Options{})
|
||||
assertErrorContains(t, err, "full artifact payload is missing")
|
||||
}
|
||||
|
||||
func TestConvertArtifactRejectsMissingPayloadWhenConversionRequested(t *testing.T) {
|
||||
_, err := ConvertArtifact(Artifact{Schema: SchemaFull}, SchemaMinimal)
|
||||
assertErrorContains(t, err, "full artifact payload is missing")
|
||||
}
|
||||
|
||||
func TestConvertArtifactSameSchemaDoesNotRequirePayload(t *testing.T) {
|
||||
artifact := Artifact{Schema: SchemaFull}
|
||||
converted, err := ConvertArtifact(artifact, SchemaFull)
|
||||
if err != nil {
|
||||
t.Fatalf("convert failed: %v", err)
|
||||
}
|
||||
if converted.Schema != SchemaFull {
|
||||
t.Fatalf("schema = %q, want %q", converted.Schema, SchemaFull)
|
||||
}
|
||||
if converted.Full != nil {
|
||||
t.Fatalf("full payload = %#v, want nil", converted.Full)
|
||||
}
|
||||
}
|
||||
|
||||
func mustMarshalJSON(t *testing.T, value any) []byte {
|
||||
t.Helper()
|
||||
data, err := json.Marshal(value)
|
||||
|
||||
156
internal/trim/run.go
Normal file
156
internal/trim/run.go
Normal file
@@ -0,0 +1,156 @@
|
||||
package trim
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"sort"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/jsonfile"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
||||
)
|
||||
|
||||
type auditReport struct {
|
||||
Operation string `json:"operation"`
|
||||
InputFile string `json:"input_file"`
|
||||
OutputFile string `json:"output_file"`
|
||||
InputSchema string `json:"input_schema"`
|
||||
OutputSchema string `json:"output_schema"`
|
||||
Mode string `json:"mode"`
|
||||
Selector string `json:"selector"`
|
||||
SelectedIDs []int `json:"selected_ids"`
|
||||
AllowEmpty bool `json:"allow_empty"`
|
||||
InputSegmentCount int `json:"input_segment_count"`
|
||||
RetainedSegmentCount int `json:"retained_segment_count"`
|
||||
RemovedSegmentCount int `json:"removed_segment_count"`
|
||||
RemovedInputIDs []int `json:"removed_input_ids"`
|
||||
OldToNewIDMapping []idMapping `json:"old_to_new_id_mapping"`
|
||||
OverlapGroupsRecomputed bool `json:"overlap_groups_recomputed"`
|
||||
}
|
||||
|
||||
type idMapping struct {
|
||||
OldID int `json:"old_id"`
|
||||
NewID int `json:"new_id"`
|
||||
}
|
||||
|
||||
// Run executes artifact-level trim orchestration.
|
||||
func Run(ctx context.Context, cfg config.TrimConfig) error {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
selector, err := ParseSelector(cfg.Selector)
|
||||
if err != nil {
|
||||
return fmt.Errorf("invalid selector %q: %w", cfg.Selector, err)
|
||||
}
|
||||
|
||||
data, err := os.ReadFile(cfg.InputFile)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read --input-file %q: %w", cfg.InputFile, err)
|
||||
}
|
||||
|
||||
artifact, err := ParseArtifactJSON(data)
|
||||
if err != nil {
|
||||
return fmt.Errorf("--input-file %q: %w", cfg.InputFile, err)
|
||||
}
|
||||
inputSegmentCount := artifact.SegmentCount()
|
||||
inputSchema := artifact.Schema
|
||||
|
||||
mode := ModeKeep
|
||||
if cfg.Mode == "remove" {
|
||||
mode = ModeRemove
|
||||
}
|
||||
|
||||
trimmed, err := ApplyArtifact(artifact, Options{
|
||||
Mode: mode,
|
||||
Selector: selector,
|
||||
AllowEmpty: cfg.AllowEmpty,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
outputSchema := artifact.Schema
|
||||
if cfg.OutputSchema != "" {
|
||||
outputSchema = cfg.OutputSchema
|
||||
}
|
||||
|
||||
outputArtifact, err := ConvertArtifact(trimmed.Artifact, outputSchema)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if err := ValidateArtifact(outputArtifact); err != nil {
|
||||
return fmt.Errorf("validate trimmed output: %w", err)
|
||||
}
|
||||
|
||||
if err := jsonfile.Write(cfg.OutputFile, outputArtifact.Value()); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if cfg.ReportFile == "" {
|
||||
return nil
|
||||
}
|
||||
|
||||
audit := auditReport{
|
||||
Operation: "trim",
|
||||
InputFile: cfg.InputFile,
|
||||
OutputFile: cfg.OutputFile,
|
||||
InputSchema: inputSchema,
|
||||
OutputSchema: outputArtifact.Schema,
|
||||
Mode: cfg.Mode,
|
||||
Selector: cfg.Selector,
|
||||
SelectedIDs: selector.IDs(),
|
||||
AllowEmpty: cfg.AllowEmpty,
|
||||
InputSegmentCount: inputSegmentCount,
|
||||
RetainedSegmentCount: len(trimmed.OldToNewID),
|
||||
RemovedSegmentCount: len(trimmed.RemovedIDs),
|
||||
RemovedInputIDs: append([]int(nil), trimmed.RemovedIDs...),
|
||||
OldToNewIDMapping: orderedIDMapping(trimmed.OldToNewID),
|
||||
OverlapGroupsRecomputed: trimmed.OverlapGroupsRecomputed,
|
||||
}
|
||||
auditJSON, err := json.Marshal(audit)
|
||||
if err != nil {
|
||||
return fmt.Errorf("marshal trim audit report: %w", err)
|
||||
}
|
||||
|
||||
rpt := report.Report{
|
||||
Metadata: report.Metadata{
|
||||
Application: outputArtifact.Application(),
|
||||
Version: outputArtifact.Version(),
|
||||
InputReader: "trim-artifact",
|
||||
InputFiles: []string{cfg.InputFile},
|
||||
OutputModules: []string{"json"},
|
||||
},
|
||||
Events: []report.Event{
|
||||
report.Info("trim", "trim", fmt.Sprintf("trimmed %d input segment(s) into %d output segment(s) with mode=%s", inputSegmentCount, outputArtifact.SegmentCount(), cfg.Mode)),
|
||||
report.Info("trim", "trim-audit", string(auditJSON)),
|
||||
report.Info("trim", "validate-output", fmt.Sprintf("validated %d output segment(s)", outputArtifact.SegmentCount())),
|
||||
report.Info("output", "json", "wrote transcript JSON"),
|
||||
},
|
||||
}
|
||||
if err := report.WriteJSON(cfg.ReportFile, rpt); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func orderedIDMapping(mapping map[int]int) []idMapping {
|
||||
keys := make([]int, 0, len(mapping))
|
||||
for oldID := range mapping {
|
||||
keys = append(keys, oldID)
|
||||
}
|
||||
sort.Ints(keys)
|
||||
|
||||
pairs := make([]idMapping, 0, len(keys))
|
||||
for _, oldID := range keys {
|
||||
pairs = append(pairs, idMapping{
|
||||
OldID: oldID,
|
||||
NewID: mapping[oldID],
|
||||
})
|
||||
}
|
||||
return pairs
|
||||
}
|
||||
28
internal/trim/run_test.go
Normal file
28
internal/trim/run_test.go
Normal file
@@ -0,0 +1,28 @@
|
||||
package trim
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
)
|
||||
|
||||
func TestRunReturnsContextErrorBeforeWork(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
|
||||
err := Run(ctx, config.TrimConfig{
|
||||
InputFile: filepath.Join(dir, "input.json"),
|
||||
OutputFile: filepath.Join(dir, "output.json"),
|
||||
Mode: "keep",
|
||||
Selector: "1",
|
||||
OutputSchema: "",
|
||||
AllowEmpty: false,
|
||||
})
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("error = %v, want context.Canceled", err)
|
||||
}
|
||||
}
|
||||
@@ -14,6 +14,10 @@ import (
|
||||
var schemaFS embed.FS
|
||||
|
||||
const (
|
||||
OutputSchemaMinimal = "seriatim-minimal"
|
||||
OutputSchemaIntermediate = "seriatim-intermediate"
|
||||
OutputSchemaFull = "seriatim-full"
|
||||
|
||||
fullOutputSchemaPath = "full-output.schema.json"
|
||||
intermediateOutputSchemaPath = "intermediate-output.schema.json"
|
||||
minimalOutputSchemaPath = "minimal-output.schema.json"
|
||||
@@ -115,6 +119,25 @@ type OverlapGroup struct {
|
||||
Resolution string `json:"resolution"`
|
||||
}
|
||||
|
||||
// ValidOutputSchemaName reports whether value is a supported output schema name.
|
||||
func ValidOutputSchemaName(value string) bool {
|
||||
switch value {
|
||||
case OutputSchemaMinimal, OutputSchemaIntermediate, OutputSchemaFull:
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// OutputSchemaNames returns supported output schema names in validation order.
|
||||
func OutputSchemaNames() []string {
|
||||
return []string{
|
||||
OutputSchemaMinimal,
|
||||
OutputSchemaIntermediate,
|
||||
OutputSchemaFull,
|
||||
}
|
||||
}
|
||||
|
||||
// ValidateTranscript validates a full transcript against the public JSON
|
||||
// schema and seriatim-specific semantic rules.
|
||||
func ValidateTranscript(transcript Transcript) error {
|
||||
@@ -228,15 +251,17 @@ func outputSchema(schemaPath string) (*jsonschema.Schema, error) {
|
||||
}
|
||||
|
||||
func validateSemantics(transcript Transcript) error {
|
||||
segments := make([]segmentSemantics, len(transcript.Segments))
|
||||
for index, segment := range transcript.Segments {
|
||||
wantID := index + 1
|
||||
if segment.ID != wantID {
|
||||
return fmt.Errorf("segment %d has id %d; want %d", index, segment.ID, wantID)
|
||||
}
|
||||
if segment.End < segment.Start {
|
||||
return fmt.Errorf("segment %d has end %.3f before start %.3f", index, segment.End, segment.Start)
|
||||
segments[index] = segmentSemantics{
|
||||
id: segment.ID,
|
||||
start: segment.Start,
|
||||
end: segment.End,
|
||||
}
|
||||
}
|
||||
if err := validateSegmentSemantics(segments); err != nil {
|
||||
return err
|
||||
}
|
||||
for index, group := range transcript.OverlapGroups {
|
||||
if group.End < group.Start {
|
||||
return fmt.Errorf("overlap_group %d has end %.3f before start %.3f", index, group.End, group.Start)
|
||||
@@ -246,26 +271,43 @@ func validateSemantics(transcript Transcript) error {
|
||||
}
|
||||
|
||||
func validateIntermediateSemantics(transcript IntermediateTranscript) error {
|
||||
segments := make([]segmentSemantics, len(transcript.Segments))
|
||||
for index, segment := range transcript.Segments {
|
||||
wantID := index + 1
|
||||
if segment.ID != wantID {
|
||||
return fmt.Errorf("segment %d has id %d; want %d", index, segment.ID, wantID)
|
||||
}
|
||||
if segment.End < segment.Start {
|
||||
return fmt.Errorf("segment %d has end %.3f before start %.3f", index, segment.End, segment.Start)
|
||||
segments[index] = segmentSemantics{
|
||||
id: segment.ID,
|
||||
start: segment.Start,
|
||||
end: segment.End,
|
||||
}
|
||||
}
|
||||
return nil
|
||||
return validateSegmentSemantics(segments)
|
||||
}
|
||||
|
||||
func validateMinimalSemantics(transcript MinimalTranscript) error {
|
||||
segments := make([]segmentSemantics, len(transcript.Segments))
|
||||
for index, segment := range transcript.Segments {
|
||||
wantID := index + 1
|
||||
if segment.ID != wantID {
|
||||
return fmt.Errorf("segment %d has id %d; want %d", index, segment.ID, wantID)
|
||||
segments[index] = segmentSemantics{
|
||||
id: segment.ID,
|
||||
start: segment.Start,
|
||||
end: segment.End,
|
||||
}
|
||||
if segment.End < segment.Start {
|
||||
return fmt.Errorf("segment %d has end %.3f before start %.3f", index, segment.End, segment.Start)
|
||||
}
|
||||
return validateSegmentSemantics(segments)
|
||||
}
|
||||
|
||||
type segmentSemantics struct {
|
||||
id int
|
||||
start float64
|
||||
end float64
|
||||
}
|
||||
|
||||
func validateSegmentSemantics(segments []segmentSemantics) error {
|
||||
for index, segment := range segments {
|
||||
wantID := index + 1
|
||||
if segment.id != wantID {
|
||||
return fmt.Errorf("segment %d has id %d; want %d", index, segment.id, wantID)
|
||||
}
|
||||
if segment.end < segment.start {
|
||||
return fmt.Errorf("segment %d has end %.3f before start %.3f", index, segment.end, segment.start)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
|
||||
@@ -5,6 +5,43 @@ import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestValidOutputSchemaName(t *testing.T) {
|
||||
valid := []string{
|
||||
OutputSchemaMinimal,
|
||||
OutputSchemaIntermediate,
|
||||
OutputSchemaFull,
|
||||
}
|
||||
for _, name := range valid {
|
||||
if !ValidOutputSchemaName(name) {
|
||||
t.Fatalf("expected %q to be valid", name)
|
||||
}
|
||||
}
|
||||
|
||||
invalid := []string{"", "compact", "minimal", "seriatim"}
|
||||
for _, name := range invalid {
|
||||
if ValidOutputSchemaName(name) {
|
||||
t.Fatalf("expected %q to be invalid", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestOutputSchemaNames(t *testing.T) {
|
||||
names := OutputSchemaNames()
|
||||
want := []string{
|
||||
OutputSchemaMinimal,
|
||||
OutputSchemaIntermediate,
|
||||
OutputSchemaFull,
|
||||
}
|
||||
if len(names) != len(want) {
|
||||
t.Fatalf("len(names) = %d, want %d", len(names), len(want))
|
||||
}
|
||||
for index := range want {
|
||||
if names[index] != want[index] {
|
||||
t.Fatalf("names[%d] = %q, want %q", index, names[index], want[index])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidateTranscriptAcceptsValidTranscript(t *testing.T) {
|
||||
transcript := validTranscript()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user