Compare commits
11 Commits
e5173c78fe
...
f40d4add91
| Author | SHA1 | Date | |
|---|---|---|---|
| f40d4add91 | |||
| 16bb12face | |||
| f18e2428dc | |||
| 3b64e784a1 | |||
| 3744d229a2 | |||
| 9bbe1fb7f1 | |||
| b7a66f6cc4 | |||
| c8efdb53d3 | |||
| ab4b252b08 | |||
| e9028e08a4 | |||
| 332884f887 |
14
README.md
14
README.md
@@ -27,19 +27,19 @@ go run ./cmd/seriatim merge \
|
||||
- Configuration reference: [docs/config.md](docs/config.md)
|
||||
- Operations guide: [docs/operations.md](docs/operations.md)
|
||||
- Troubleshooting: [docs/troubleshooting.md](docs/troubleshooting.md)
|
||||
- Integrations:
|
||||
- Integration references:
|
||||
- [docs/integrations/whisperx-json.md](docs/integrations/whisperx-json.md)
|
||||
- [docs/integrations/output-schemas.md](docs/integrations/output-schemas.md)
|
||||
- Development architecture policy: [docs/policy/architecture.md](docs/policy/architecture.md)
|
||||
- Contributor workflow: [docs/policy/development.md](docs/policy/development.md)
|
||||
- Documentation policy: [docs/policy/documentation.md](docs/policy/documentation.md)
|
||||
- Internal implementation docs:
|
||||
- Development policies:
|
||||
- [docs/policy/architecture.md](docs/policy/architecture.md)
|
||||
- [docs/policy/development.md](docs/policy/development.md)
|
||||
- [docs/policy/documentation.md](docs/policy/documentation.md)
|
||||
- Internal implementation references:
|
||||
- [docs/internal/pipeline.md](docs/internal/pipeline.md)
|
||||
- [docs/internal/artifacts.md](docs/internal/artifacts.md)
|
||||
- [docs/internal/modules.md](docs/internal/modules.md)
|
||||
- Public JSON schemas:
|
||||
- Public JSON schema files:
|
||||
- [schema/minimal-output.schema.json](schema/minimal-output.schema.json)
|
||||
- [schema/intermediate-output.schema.json](schema/intermediate-output.schema.json)
|
||||
- [schema/full-output.schema.json](schema/full-output.schema.json)
|
||||
- Synthetic examples: [examples/README.md](examples/README.md)
|
||||
- Documentation roadmap: [docs/roadmap/documentation.md](docs/roadmap/documentation.md)
|
||||
|
||||
@@ -179,4 +179,3 @@ go run ./cmd/seriatim normalize \
|
||||
- [../schema/minimal-output.schema.json](../schema/minimal-output.schema.json)
|
||||
- [../schema/intermediate-output.schema.json](../schema/intermediate-output.schema.json)
|
||||
- [../schema/full-output.schema.json](../schema/full-output.schema.json)
|
||||
- Documentation roadmap: [roadmap/documentation.md](roadmap/documentation.md)
|
||||
|
||||
@@ -165,4 +165,3 @@ All commands:
|
||||
- [../schema/minimal-output.schema.json](../schema/minimal-output.schema.json)
|
||||
- [../schema/intermediate-output.schema.json](../schema/intermediate-output.schema.json)
|
||||
- [../schema/full-output.schema.json](../schema/full-output.schema.json)
|
||||
- Documentation roadmap: [roadmap/documentation.md](roadmap/documentation.md)
|
||||
|
||||
@@ -51,29 +51,43 @@ Unknown/empty selection falls back to intermediate conversion.
|
||||
|
||||
## Trim internals
|
||||
|
||||
`internal/trim` is artifact-level projection, not merge reprocessing.
|
||||
`internal/trim` handles artifact-level projection and does not execute merge
|
||||
pipeline modules.
|
||||
|
||||
Core flow:
|
||||
Run layer (`run.go`):
|
||||
|
||||
1. Parse selector (`internal/trim/selector.go`).
|
||||
2. Parse input artifact and detect schema (`ParseArtifactJSON`).
|
||||
3. Apply keep/remove projection with sequential ID renumbering.
|
||||
4. Recompute overlap groups only for full-schema artifacts.
|
||||
5. Optionally convert output schema when supported.
|
||||
6. Validate output artifact before write.
|
||||
1. Parse selector from validated config.
|
||||
2. Read and parse input artifact JSON.
|
||||
3. Apply trim projection through schema-aware artifact handling.
|
||||
4. Resolve output schema (preserve input schema unless overridden).
|
||||
5. Validate output artifact.
|
||||
6. Write output JSON.
|
||||
7. Optionally write report JSON with `trim-audit`.
|
||||
|
||||
Schema-conversion limits:
|
||||
Apply layer (`apply.go`):
|
||||
|
||||
- full -> intermediate/minimal supported.
|
||||
- intermediate -> minimal supported.
|
||||
- minimal -> intermediate supported.
|
||||
- intermediate/minimal -> full is rejected.
|
||||
- one shared projection policy for selector mode, input ID validation, selected
|
||||
ID existence checks, keep/remove filtering, removed IDs, and old-to-new ID
|
||||
mappings
|
||||
- schema-specific segment reconstruction for full/intermediate/minimal outputs
|
||||
- overlap-group recomputation only for full-schema outputs
|
||||
|
||||
Artifact layer (`artifact.go`):
|
||||
|
||||
- schema detection for full/intermediate/minimal artifacts
|
||||
- schema-preserving trim application
|
||||
- supported schema conversions:
|
||||
- full -> intermediate/minimal
|
||||
- intermediate -> minimal
|
||||
- minimal -> intermediate
|
||||
- rejected conversion:
|
||||
- intermediate/minimal -> full
|
||||
|
||||
Trim invariants:
|
||||
|
||||
- selected IDs must exist in input.
|
||||
- input IDs must be positive, unique, sequential.
|
||||
- retained order follows input transcript order.
|
||||
- retained segment order follows input transcript order.
|
||||
- output IDs are reassigned to `1..N`.
|
||||
|
||||
## Normalize internals
|
||||
|
||||
@@ -55,7 +55,8 @@ Output writer:
|
||||
- `autocorrect`: applies YAML replacement rules when configured.
|
||||
- `assign-ids`: assigns final sequential IDs.
|
||||
- `validate-output`: validates selected public artifact shape.
|
||||
- `json`: writes artifact JSON to `cfg.OutputFile`.
|
||||
- `json`: writes artifact JSON to `cfg.OutputFile` through shared deterministic
|
||||
JSON file writing.
|
||||
|
||||
Filesystem side effects are limited to:
|
||||
|
||||
|
||||
@@ -28,7 +28,7 @@ Reviewed documentation and policy:
|
||||
- `docs/internal/modules.md`
|
||||
- `docs/integrations/output-schemas.md`
|
||||
- `docs/integrations/whisperx-json.md`
|
||||
- `docs/roadmap/documentation.md`
|
||||
- `docs/roadmap/cleanup.md`
|
||||
|
||||
Reviewed implementation areas:
|
||||
|
||||
|
||||
@@ -2,10 +2,9 @@ package builtin
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"os"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/jsonfile"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
||||
)
|
||||
|
||||
@@ -20,15 +19,7 @@ func (jsonOutputWriter) Write(ctx context.Context, out any, rpt report.Report, c
|
||||
return nil, err
|
||||
}
|
||||
|
||||
file, err := os.Create(cfg.OutputFile)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
enc := json.NewEncoder(file)
|
||||
enc.SetIndent("", " ")
|
||||
if err := enc.Encode(out); err != nil {
|
||||
if err := jsonfile.Write(cfg.OutputFile, out); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
|
||||
31
internal/cli/flags.go
Normal file
31
internal/cli/flags.go
Normal file
@@ -0,0 +1,31 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
)
|
||||
|
||||
func addOutputFileFlag(cmd *cobra.Command, target *string) {
|
||||
cmd.Flags().StringVar(target, "output-file", "", "output transcript JSON file")
|
||||
}
|
||||
|
||||
func addReportFileFlag(cmd *cobra.Command, target *string) {
|
||||
cmd.Flags().StringVar(target, "report-file", "", "optional report JSON file")
|
||||
}
|
||||
|
||||
func addOutputModulesFlag(cmd *cobra.Command, target *string) {
|
||||
cmd.Flags().StringVar(target, "output-modules", config.DefaultOutputModules, "comma-separated output modules")
|
||||
}
|
||||
|
||||
func addMergeOutputSchemaFlag(cmd *cobra.Command, target *string) {
|
||||
cmd.Flags().StringVar(target, "output-schema", config.DefaultOutputSchema, "output JSON schema: seriatim-minimal, seriatim-intermediate (default), or seriatim-full")
|
||||
}
|
||||
|
||||
func addNormalizeOutputSchemaFlag(cmd *cobra.Command, target *string) {
|
||||
cmd.Flags().StringVar(target, "output-schema", config.DefaultOutputSchema, "output JSON schema: seriatim-minimal, seriatim-intermediate, or seriatim-full")
|
||||
}
|
||||
|
||||
func addTrimOutputSchemaFlag(cmd *cobra.Command, target *string) {
|
||||
cmd.Flags().StringVar(target, "output-schema", "", "optional output JSON schema override: seriatim-minimal, seriatim-intermediate, or seriatim-full")
|
||||
}
|
||||
@@ -31,13 +31,13 @@ func newMergeCommand() *cobra.Command {
|
||||
|
||||
flags := cmd.Flags()
|
||||
flags.StringArrayVar(&opts.InputFiles, "input-file", nil, "input transcript file; may be repeated")
|
||||
flags.StringVar(&opts.OutputFile, "output-file", "", "output transcript JSON file")
|
||||
flags.StringVar(&opts.ReportFile, "report-file", "", "optional report JSON file")
|
||||
addOutputFileFlag(cmd, &opts.OutputFile)
|
||||
addReportFileFlag(cmd, &opts.ReportFile)
|
||||
flags.StringVar(&opts.SpeakersFile, "speakers", "", "speaker map file")
|
||||
flags.StringVar(&opts.AutocorrectFile, "autocorrect", "", "autocorrect rules file")
|
||||
flags.StringVar(&opts.InputReader, "input-reader", config.DefaultInputReader, "input reader module")
|
||||
flags.StringVar(&opts.OutputModules, "output-modules", config.DefaultOutputModules, "comma-separated output modules")
|
||||
flags.StringVar(&opts.OutputSchema, "output-schema", config.DefaultOutputSchema, "output JSON schema: seriatim-minimal, seriatim-intermediate (default), or seriatim-full")
|
||||
addOutputModulesFlag(cmd, &opts.OutputModules)
|
||||
addMergeOutputSchemaFlag(cmd, &opts.OutputSchema)
|
||||
flags.StringVar(&opts.PreprocessingModules, "preprocessing-modules", config.DefaultPreprocessingModules, "comma-separated preprocessing modules")
|
||||
flags.StringVar(&opts.PostprocessingModules, "postprocessing-modules", config.DefaultPostprocessingModules, "comma-separated postprocessing modules")
|
||||
flags.StringVar(&opts.CoalesceGap, "coalesce-gap", config.DefaultCoalesceGapValue, "maximum same-speaker gap in seconds for coalesce")
|
||||
|
||||
@@ -30,10 +30,10 @@ func newNormalizeCommand() *cobra.Command {
|
||||
|
||||
flags := cmd.Flags()
|
||||
flags.StringVar(&opts.InputFile, "input-file", "", "input transcript JSON file")
|
||||
flags.StringVar(&opts.OutputFile, "output-file", "", "output transcript JSON file")
|
||||
flags.StringVar(&opts.ReportFile, "report-file", "", "optional report JSON file")
|
||||
flags.StringVar(&opts.OutputSchema, "output-schema", config.DefaultOutputSchema, "output JSON schema: seriatim-minimal, seriatim-intermediate, or seriatim-full")
|
||||
flags.StringVar(&opts.OutputModules, "output-modules", config.DefaultOutputModules, "comma-separated output modules")
|
||||
addOutputFileFlag(cmd, &opts.OutputFile)
|
||||
addReportFileFlag(cmd, &opts.ReportFile)
|
||||
addNormalizeOutputSchemaFlag(cmd, &opts.OutputSchema)
|
||||
addOutputModulesFlag(cmd, &opts.OutputModules)
|
||||
|
||||
return cmd
|
||||
}
|
||||
|
||||
@@ -1,41 +1,12 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"sort"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
||||
triminternal "gitea.maximumdirect.net/eric/seriatim/internal/trim"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/trim"
|
||||
)
|
||||
|
||||
type trimAuditReport struct {
|
||||
Operation string `json:"operation"`
|
||||
InputFile string `json:"input_file"`
|
||||
OutputFile string `json:"output_file"`
|
||||
InputSchema string `json:"input_schema"`
|
||||
OutputSchema string `json:"output_schema"`
|
||||
Mode string `json:"mode"`
|
||||
Selector string `json:"selector"`
|
||||
SelectedIDs []int `json:"selected_ids"`
|
||||
AllowEmpty bool `json:"allow_empty"`
|
||||
InputSegmentCount int `json:"input_segment_count"`
|
||||
RetainedSegmentCount int `json:"retained_segment_count"`
|
||||
RemovedSegmentCount int `json:"removed_segment_count"`
|
||||
RemovedInputIDs []int `json:"removed_input_ids"`
|
||||
OldToNewIDMapping []trimIDMapping `json:"old_to_new_id_mapping"`
|
||||
OverlapGroupsRecomputed bool `json:"overlap_groups_recomputed"`
|
||||
}
|
||||
|
||||
type trimIDMapping struct {
|
||||
OldID int `json:"old_id"`
|
||||
NewID int `json:"new_id"`
|
||||
}
|
||||
|
||||
func newTrimCommand() *cobra.Command {
|
||||
var opts config.TrimOptions
|
||||
|
||||
@@ -53,139 +24,18 @@ func newTrimCommand() *cobra.Command {
|
||||
return err
|
||||
}
|
||||
|
||||
selector, err := triminternal.ParseSelector(cfg.Selector)
|
||||
if err != nil {
|
||||
return fmt.Errorf("invalid selector %q: %w", cfg.Selector, err)
|
||||
}
|
||||
|
||||
data, err := os.ReadFile(cfg.InputFile)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read --input-file %q: %w", cfg.InputFile, err)
|
||||
}
|
||||
|
||||
artifact, err := triminternal.ParseArtifactJSON(data)
|
||||
if err != nil {
|
||||
return fmt.Errorf("--input-file %q: %w", cfg.InputFile, err)
|
||||
}
|
||||
inputSegmentCount := artifact.SegmentCount()
|
||||
inputSchema := artifact.Schema
|
||||
|
||||
mode := triminternal.ModeKeep
|
||||
if cfg.Mode == "remove" {
|
||||
mode = triminternal.ModeRemove
|
||||
}
|
||||
|
||||
trimmed, err := triminternal.ApplyArtifact(artifact, triminternal.Options{
|
||||
Mode: mode,
|
||||
Selector: selector,
|
||||
AllowEmpty: cfg.AllowEmpty,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
outputSchema := artifact.Schema
|
||||
if cfg.OutputSchema != "" {
|
||||
outputSchema = cfg.OutputSchema
|
||||
}
|
||||
|
||||
outputArtifact, err := triminternal.ConvertArtifact(trimmed.Artifact, outputSchema)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if err := triminternal.ValidateArtifact(outputArtifact); err != nil {
|
||||
return fmt.Errorf("validate trimmed output: %w", err)
|
||||
}
|
||||
|
||||
if err := writeOutputJSON(cfg.OutputFile, outputArtifact.Value()); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if cfg.ReportFile != "" {
|
||||
audit := trimAuditReport{
|
||||
Operation: "trim",
|
||||
InputFile: cfg.InputFile,
|
||||
OutputFile: cfg.OutputFile,
|
||||
InputSchema: inputSchema,
|
||||
OutputSchema: outputArtifact.Schema,
|
||||
Mode: cfg.Mode,
|
||||
Selector: cfg.Selector,
|
||||
SelectedIDs: selector.IDs(),
|
||||
AllowEmpty: cfg.AllowEmpty,
|
||||
InputSegmentCount: inputSegmentCount,
|
||||
RetainedSegmentCount: len(trimmed.OldToNewID),
|
||||
RemovedSegmentCount: len(trimmed.RemovedIDs),
|
||||
RemovedInputIDs: append([]int(nil), trimmed.RemovedIDs...),
|
||||
OldToNewIDMapping: orderedIDMapping(trimmed.OldToNewID),
|
||||
OverlapGroupsRecomputed: trimmed.OverlapGroupsRecomputed,
|
||||
}
|
||||
auditJSON, err := json.Marshal(audit)
|
||||
if err != nil {
|
||||
return fmt.Errorf("marshal trim audit report: %w", err)
|
||||
}
|
||||
|
||||
rpt := report.Report{
|
||||
Metadata: report.Metadata{
|
||||
Application: outputArtifact.Application(),
|
||||
Version: outputArtifact.Version(),
|
||||
InputReader: "trim-artifact",
|
||||
InputFiles: []string{cfg.InputFile},
|
||||
OutputModules: []string{"json"},
|
||||
},
|
||||
Events: []report.Event{
|
||||
report.Info("trim", "trim", fmt.Sprintf("trimmed %d input segment(s) into %d output segment(s) with mode=%s", inputSegmentCount, outputArtifact.SegmentCount(), cfg.Mode)),
|
||||
report.Info("trim", "trim-audit", string(auditJSON)),
|
||||
report.Info("trim", "validate-output", fmt.Sprintf("validated %d output segment(s)", outputArtifact.SegmentCount())),
|
||||
report.Info("output", "json", "wrote transcript JSON"),
|
||||
},
|
||||
}
|
||||
if err := report.WriteJSON(cfg.ReportFile, rpt); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
return nil
|
||||
return trim.Run(cmd.Context(), cfg)
|
||||
},
|
||||
}
|
||||
|
||||
flags := cmd.Flags()
|
||||
flags.StringVar(&opts.InputFile, "input-file", "", "input seriatim transcript artifact JSON file")
|
||||
flags.StringVar(&opts.OutputFile, "output-file", "", "output transcript JSON file")
|
||||
flags.StringVar(&opts.ReportFile, "report-file", "", "optional report JSON file")
|
||||
addOutputFileFlag(cmd, &opts.OutputFile)
|
||||
addReportFileFlag(cmd, &opts.ReportFile)
|
||||
flags.StringVar(&opts.Keep, "keep", "", "segment ID selector to keep (for example: 1-10,15)")
|
||||
flags.StringVar(&opts.Remove, "remove", "", "segment ID selector to remove (for example: 1-10,15)")
|
||||
flags.StringVar(&opts.OutputSchema, "output-schema", "", "optional output JSON schema override: seriatim-minimal, seriatim-intermediate, or seriatim-full")
|
||||
addTrimOutputSchemaFlag(cmd, &opts.OutputSchema)
|
||||
flags.BoolVar(&opts.AllowEmpty, "allow-empty", false, "allow trimming to an empty transcript")
|
||||
|
||||
return cmd
|
||||
}
|
||||
|
||||
func writeOutputJSON(path string, value any) error {
|
||||
file, err := os.Create(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
enc := json.NewEncoder(file)
|
||||
enc.SetIndent("", " ")
|
||||
return enc.Encode(value)
|
||||
}
|
||||
|
||||
func orderedIDMapping(mapping map[int]int) []trimIDMapping {
|
||||
keys := make([]int, 0, len(mapping))
|
||||
for oldID := range mapping {
|
||||
keys = append(keys, oldID)
|
||||
}
|
||||
sort.Ints(keys)
|
||||
|
||||
pairs := make([]trimIDMapping, 0, len(keys))
|
||||
for _, oldID := range keys {
|
||||
pairs = append(pairs, trimIDMapping{
|
||||
OldID: oldID,
|
||||
NewID: mapping[oldID],
|
||||
})
|
||||
}
|
||||
return pairs
|
||||
}
|
||||
|
||||
@@ -12,6 +12,29 @@ import (
|
||||
"gitea.maximumdirect.net/eric/seriatim/schema"
|
||||
)
|
||||
|
||||
type trimAuditReport struct {
|
||||
Operation string `json:"operation"`
|
||||
InputFile string `json:"input_file"`
|
||||
OutputFile string `json:"output_file"`
|
||||
InputSchema string `json:"input_schema"`
|
||||
OutputSchema string `json:"output_schema"`
|
||||
Mode string `json:"mode"`
|
||||
Selector string `json:"selector"`
|
||||
SelectedIDs []int `json:"selected_ids"`
|
||||
AllowEmpty bool `json:"allow_empty"`
|
||||
InputSegmentCount int `json:"input_segment_count"`
|
||||
RetainedSegmentCount int `json:"retained_segment_count"`
|
||||
RemovedSegmentCount int `json:"removed_segment_count"`
|
||||
RemovedInputIDs []int `json:"removed_input_ids"`
|
||||
OldToNewIDMapping []trimIDMapping `json:"old_to_new_id_mapping"`
|
||||
OverlapGroupsRecomputed bool `json:"overlap_groups_recomputed"`
|
||||
}
|
||||
|
||||
type trimIDMapping struct {
|
||||
OldID int `json:"old_id"`
|
||||
NewID int `json:"new_id"`
|
||||
}
|
||||
|
||||
func TestTrimKeepModeEndToEnd(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTrimFullFixture(t, dir, "input.json")
|
||||
|
||||
@@ -160,13 +160,7 @@ func (r run) coalescedSegment(id int) model.Segment {
|
||||
}
|
||||
|
||||
func segmentRef(segment model.Segment) string {
|
||||
if segment.SourceSegmentIndex != nil {
|
||||
return fmt.Sprintf("%s#%d", segment.Source, *segment.SourceSegmentIndex)
|
||||
}
|
||||
if segment.SourceRef != "" {
|
||||
return segment.SourceRef
|
||||
}
|
||||
return segment.Source
|
||||
return model.SegmentReference(segment)
|
||||
}
|
||||
|
||||
func isSkippableInterjection(segment model.Segment) bool {
|
||||
|
||||
@@ -8,6 +8,8 @@ import (
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/schema"
|
||||
)
|
||||
|
||||
const (
|
||||
@@ -27,9 +29,9 @@ const (
|
||||
WordRunReorderWindowEnv = "SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW"
|
||||
BackchannelMaxDurationEnv = "SERIATIM_BACKCHANNEL_MAX_DURATION"
|
||||
FillerMaxDurationEnv = "SERIATIM_FILLER_MAX_DURATION"
|
||||
OutputSchemaMinimal = "seriatim-minimal"
|
||||
OutputSchemaIntermediate = "seriatim-intermediate"
|
||||
OutputSchemaFull = "seriatim-full"
|
||||
OutputSchemaMinimal = schema.OutputSchemaMinimal
|
||||
OutputSchemaIntermediate = schema.OutputSchemaIntermediate
|
||||
OutputSchemaFull = schema.OutputSchemaFull
|
||||
)
|
||||
|
||||
// MergeOptions captures raw CLI option values before validation.
|
||||
@@ -210,11 +212,8 @@ func NewMergeConfig(opts MergeOptions) (Config, error) {
|
||||
|
||||
// NewTrimConfig validates raw trim options and returns normalized config.
|
||||
func NewTrimConfig(opts TrimOptions) (TrimConfig, error) {
|
||||
inputFile := filepath.Clean(strings.TrimSpace(opts.InputFile))
|
||||
if strings.TrimSpace(opts.InputFile) == "" {
|
||||
return TrimConfig{}, errors.New("--input-file is required")
|
||||
}
|
||||
if err := requireFile(inputFile, "--input-file"); err != nil {
|
||||
inputFile, err := normalizeSingleInputFile(opts.InputFile, "--input-file")
|
||||
if err != nil {
|
||||
return TrimConfig{}, err
|
||||
}
|
||||
|
||||
@@ -223,12 +222,9 @@ func NewTrimConfig(opts TrimOptions) (TrimConfig, error) {
|
||||
return TrimConfig{}, err
|
||||
}
|
||||
|
||||
reportFile := ""
|
||||
if strings.TrimSpace(opts.ReportFile) != "" {
|
||||
reportFile, err = normalizeOutputPath(opts.ReportFile, "--report-file")
|
||||
if err != nil {
|
||||
return TrimConfig{}, err
|
||||
}
|
||||
reportFile, err := normalizeOptionalOutputPath(opts.ReportFile, "--report-file")
|
||||
if err != nil {
|
||||
return TrimConfig{}, err
|
||||
}
|
||||
|
||||
keep := strings.TrimSpace(opts.Keep)
|
||||
@@ -267,11 +263,8 @@ func NewTrimConfig(opts TrimOptions) (TrimConfig, error) {
|
||||
|
||||
// NewNormalizeConfig validates raw normalize options and returns normalized config.
|
||||
func NewNormalizeConfig(opts NormalizeOptions) (NormalizeConfig, error) {
|
||||
inputFile := filepath.Clean(strings.TrimSpace(opts.InputFile))
|
||||
if strings.TrimSpace(opts.InputFile) == "" {
|
||||
return NormalizeConfig{}, errors.New("--input-file is required")
|
||||
}
|
||||
if err := requireFile(inputFile, "--input-file"); err != nil {
|
||||
inputFile, err := normalizeSingleInputFile(opts.InputFile, "--input-file")
|
||||
if err != nil {
|
||||
return NormalizeConfig{}, err
|
||||
}
|
||||
|
||||
@@ -280,12 +273,9 @@ func NewNormalizeConfig(opts NormalizeOptions) (NormalizeConfig, error) {
|
||||
return NormalizeConfig{}, err
|
||||
}
|
||||
|
||||
reportFile := ""
|
||||
if strings.TrimSpace(opts.ReportFile) != "" {
|
||||
reportFile, err = normalizeOutputPath(opts.ReportFile, "--report-file")
|
||||
if err != nil {
|
||||
return NormalizeConfig{}, err
|
||||
}
|
||||
reportFile, err := normalizeOptionalOutputPath(opts.ReportFile, "--report-file")
|
||||
if err != nil {
|
||||
return NormalizeConfig{}, err
|
||||
}
|
||||
|
||||
outputSchema, err := resolveOutputSchema(opts.OutputSchema)
|
||||
@@ -332,12 +322,12 @@ func parseModuleList(value string) ([]string, error) {
|
||||
}
|
||||
|
||||
func validateOutputSchema(value string) error {
|
||||
switch value {
|
||||
case OutputSchemaMinimal, OutputSchemaIntermediate, OutputSchemaFull:
|
||||
if schema.ValidOutputSchemaName(value) {
|
||||
return nil
|
||||
default:
|
||||
return fmt.Errorf("--output-schema must be one of %q, %q, or %q", OutputSchemaMinimal, OutputSchemaIntermediate, OutputSchemaFull)
|
||||
}
|
||||
|
||||
names := schema.OutputSchemaNames()
|
||||
return fmt.Errorf("--output-schema must be one of %q, %q, or %q", names[0], names[1], names[2])
|
||||
}
|
||||
|
||||
func resolveOutputSchema(value string) (string, error) {
|
||||
@@ -381,6 +371,26 @@ func normalizeInputFiles(paths []string) ([]string, error) {
|
||||
return normalized, nil
|
||||
}
|
||||
|
||||
func normalizeSingleInputFile(path string, flag string) (string, error) {
|
||||
path = strings.TrimSpace(path)
|
||||
if path == "" {
|
||||
return "", fmt.Errorf("%s is required", flag)
|
||||
}
|
||||
|
||||
clean := filepath.Clean(path)
|
||||
if err := requireFile(clean, flag); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return clean, nil
|
||||
}
|
||||
|
||||
func normalizeOptionalOutputPath(path string, flag string) (string, error) {
|
||||
if strings.TrimSpace(path) == "" {
|
||||
return "", nil
|
||||
}
|
||||
return normalizeOutputPath(path, flag)
|
||||
}
|
||||
|
||||
func normalizeOutputPath(path string, flag string) (string, error) {
|
||||
path = strings.TrimSpace(path)
|
||||
if path == "" {
|
||||
|
||||
@@ -538,15 +538,9 @@ func TestCoalesceGapUsesValidOverride(t *testing.T) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "merged.json")
|
||||
|
||||
cfg, err := NewMergeConfig(MergeOptions{
|
||||
InputFiles: []string{input},
|
||||
OutputFile: output,
|
||||
InputReader: DefaultInputReader,
|
||||
OutputModules: DefaultOutputModules,
|
||||
PreprocessingModules: DefaultPreprocessingModules,
|
||||
PostprocessingModules: DefaultPostprocessingModules,
|
||||
CoalesceGap: "1.5",
|
||||
})
|
||||
opts := validMergeOptions(input, output)
|
||||
opts.CoalesceGap = "1.5"
|
||||
cfg, err := NewMergeConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
@@ -560,15 +554,9 @@ func TestCoalesceGapAllowsZero(t *testing.T) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "merged.json")
|
||||
|
||||
cfg, err := NewMergeConfig(MergeOptions{
|
||||
InputFiles: []string{input},
|
||||
OutputFile: output,
|
||||
InputReader: DefaultInputReader,
|
||||
OutputModules: DefaultOutputModules,
|
||||
PreprocessingModules: DefaultPreprocessingModules,
|
||||
PostprocessingModules: DefaultPostprocessingModules,
|
||||
CoalesceGap: "0",
|
||||
})
|
||||
opts := validMergeOptions(input, output)
|
||||
opts.CoalesceGap = "0"
|
||||
cfg, err := NewMergeConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
@@ -593,15 +581,9 @@ func TestCoalesceGapRejectsInvalidOverride(t *testing.T) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "merged.json")
|
||||
|
||||
_, err := NewMergeConfig(MergeOptions{
|
||||
InputFiles: []string{input},
|
||||
OutputFile: output,
|
||||
InputReader: DefaultInputReader,
|
||||
OutputModules: DefaultOutputModules,
|
||||
PreprocessingModules: DefaultPreprocessingModules,
|
||||
PostprocessingModules: DefaultPostprocessingModules,
|
||||
CoalesceGap: test.value,
|
||||
})
|
||||
opts := validMergeOptions(input, output)
|
||||
opts.CoalesceGap = test.value
|
||||
_, err := NewMergeConfig(opts)
|
||||
if err == nil {
|
||||
t.Fatal("expected error")
|
||||
}
|
||||
@@ -639,20 +621,16 @@ func TestNewTrimConfigRequiresExactlyOneSelectorFlag(t *testing.T) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "trimmed.json")
|
||||
|
||||
_, err := NewTrimConfig(TrimOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
})
|
||||
opts := validTrimOptions(input, output)
|
||||
opts.Keep = ""
|
||||
_, err := NewTrimConfig(opts)
|
||||
if err == nil || !strings.Contains(err.Error(), "exactly one of --keep or --remove is required") {
|
||||
t.Fatalf("expected missing selector error, got %v", err)
|
||||
}
|
||||
|
||||
_, err = NewTrimConfig(TrimOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
Keep: "1",
|
||||
Remove: "2",
|
||||
})
|
||||
opts = validTrimOptions(input, output)
|
||||
opts.Remove = "2"
|
||||
_, err = NewTrimConfig(opts)
|
||||
if err == nil || !strings.Contains(err.Error(), "mutually exclusive") {
|
||||
t.Fatalf("expected mutually exclusive selector error, got %v", err)
|
||||
}
|
||||
@@ -664,14 +642,13 @@ func TestNewTrimConfigAcceptsOutputSchemaOverride(t *testing.T) {
|
||||
output := filepath.Join(dir, "trimmed.json")
|
||||
reportPath := filepath.Join(dir, "report.json")
|
||||
|
||||
cfg, err := NewTrimConfig(TrimOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
ReportFile: reportPath,
|
||||
Remove: "3-5",
|
||||
OutputSchema: OutputSchemaMinimal,
|
||||
AllowEmpty: true,
|
||||
})
|
||||
opts := validTrimOptions(input, output)
|
||||
opts.Keep = ""
|
||||
opts.Remove = "3-5"
|
||||
opts.ReportFile = reportPath
|
||||
opts.OutputSchema = OutputSchemaMinimal
|
||||
opts.AllowEmpty = true
|
||||
cfg, err := NewTrimConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
@@ -692,17 +669,30 @@ func TestNewTrimConfigAcceptsOutputSchemaOverride(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewTrimConfigTreatsWhitespaceReportFileAsOmitted(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "trimmed.json")
|
||||
|
||||
opts := validTrimOptions(input, output)
|
||||
opts.ReportFile = " \t "
|
||||
cfg, err := NewTrimConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
if cfg.ReportFile != "" {
|
||||
t.Fatalf("report file = %q, want empty", cfg.ReportFile)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewTrimConfigRejectsInvalidOutputSchemaOverride(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "trimmed.json")
|
||||
|
||||
_, err := NewTrimConfig(TrimOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
Keep: "1",
|
||||
OutputSchema: "compact",
|
||||
})
|
||||
opts := validTrimOptions(input, output)
|
||||
opts.OutputSchema = "compact"
|
||||
_, err := NewTrimConfig(opts)
|
||||
if err == nil {
|
||||
t.Fatal("expected output schema validation error")
|
||||
}
|
||||
@@ -731,10 +721,8 @@ func TestNewNormalizeConfigRequiresOutputFile(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
|
||||
_, err := NewNormalizeConfig(NormalizeOptions{
|
||||
InputFile: input,
|
||||
OutputModules: DefaultOutputModules,
|
||||
})
|
||||
opts := validNormalizeOptions(input, "")
|
||||
_, err := NewNormalizeConfig(opts)
|
||||
if err == nil {
|
||||
t.Fatal("expected output-file required error")
|
||||
}
|
||||
@@ -749,11 +737,8 @@ func TestNewNormalizeConfigResolvesOutputSchemaDefaultAndEnv(t *testing.T) {
|
||||
output := filepath.Join(dir, "normalized.json")
|
||||
|
||||
t.Setenv(OutputSchemaEnv, "")
|
||||
cfg, err := NewNormalizeConfig(NormalizeOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
OutputModules: DefaultOutputModules,
|
||||
})
|
||||
opts := validNormalizeOptions(input, output)
|
||||
cfg, err := NewNormalizeConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
@@ -762,11 +747,7 @@ func TestNewNormalizeConfigResolvesOutputSchemaDefaultAndEnv(t *testing.T) {
|
||||
}
|
||||
|
||||
t.Setenv(OutputSchemaEnv, OutputSchemaMinimal)
|
||||
cfg, err = NewNormalizeConfig(NormalizeOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
OutputModules: DefaultOutputModules,
|
||||
})
|
||||
cfg, err = NewNormalizeConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
@@ -780,12 +761,9 @@ func TestNewNormalizeConfigRejectsInvalidOutputSchema(t *testing.T) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "normalized.json")
|
||||
|
||||
_, err := NewNormalizeConfig(NormalizeOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
OutputSchema: "compact",
|
||||
OutputModules: DefaultOutputModules,
|
||||
})
|
||||
opts := validNormalizeOptions(input, output)
|
||||
opts.OutputSchema = "compact"
|
||||
_, err := NewNormalizeConfig(opts)
|
||||
if err == nil {
|
||||
t.Fatal("expected output schema error")
|
||||
}
|
||||
@@ -799,11 +777,9 @@ func TestNewNormalizeConfigRejectsUnknownOutputModule(t *testing.T) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "normalized.json")
|
||||
|
||||
_, err := NewNormalizeConfig(NormalizeOptions{
|
||||
InputFile: input,
|
||||
OutputFile: output,
|
||||
OutputModules: "json,yaml",
|
||||
})
|
||||
opts := validNormalizeOptions(input, output)
|
||||
opts.OutputModules = "json,yaml"
|
||||
_, err := NewNormalizeConfig(opts)
|
||||
if err == nil {
|
||||
t.Fatal("expected output module error")
|
||||
}
|
||||
@@ -812,6 +788,22 @@ func TestNewNormalizeConfigRejectsUnknownOutputModule(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewNormalizeConfigTreatsWhitespaceReportFileAsOmitted(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "normalized.json")
|
||||
|
||||
opts := validNormalizeOptions(input, output)
|
||||
opts.ReportFile = "\n\t "
|
||||
cfg, err := NewNormalizeConfig(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("config failed: %v", err)
|
||||
}
|
||||
if cfg.ReportFile != "" {
|
||||
t.Fatalf("report file = %q, want empty", cfg.ReportFile)
|
||||
}
|
||||
}
|
||||
|
||||
func assertPositiveFloatEnvValidation(t *testing.T, envName string) {
|
||||
t.Helper()
|
||||
|
||||
@@ -832,14 +824,7 @@ func assertPositiveFloatEnvValidation(t *testing.T, envName string) {
|
||||
input := writeTempFile(t, dir, "input.json")
|
||||
output := filepath.Join(dir, "merged.json")
|
||||
|
||||
_, err := NewMergeConfig(MergeOptions{
|
||||
InputFiles: []string{input},
|
||||
OutputFile: output,
|
||||
InputReader: DefaultInputReader,
|
||||
OutputModules: DefaultOutputModules,
|
||||
PreprocessingModules: DefaultPreprocessingModules,
|
||||
PostprocessingModules: DefaultPostprocessingModules,
|
||||
})
|
||||
_, err := NewMergeConfig(validMergeOptions(input, output))
|
||||
if err == nil {
|
||||
t.Fatal("expected error")
|
||||
}
|
||||
@@ -850,6 +835,33 @@ func assertPositiveFloatEnvValidation(t *testing.T, envName string) {
|
||||
}
|
||||
}
|
||||
|
||||
func validMergeOptions(inputFile string, outputFile string) MergeOptions {
|
||||
return MergeOptions{
|
||||
InputFiles: []string{inputFile},
|
||||
OutputFile: outputFile,
|
||||
InputReader: DefaultInputReader,
|
||||
OutputModules: DefaultOutputModules,
|
||||
PreprocessingModules: DefaultPreprocessingModules,
|
||||
PostprocessingModules: DefaultPostprocessingModules,
|
||||
}
|
||||
}
|
||||
|
||||
func validTrimOptions(inputFile string, outputFile string) TrimOptions {
|
||||
return TrimOptions{
|
||||
InputFile: inputFile,
|
||||
OutputFile: outputFile,
|
||||
Keep: "1",
|
||||
}
|
||||
}
|
||||
|
||||
func validNormalizeOptions(inputFile string, outputFile string) NormalizeOptions {
|
||||
return NormalizeOptions{
|
||||
InputFile: inputFile,
|
||||
OutputFile: outputFile,
|
||||
OutputModules: DefaultOutputModules,
|
||||
}
|
||||
}
|
||||
|
||||
func writeTempFile(t *testing.T, dir string, name string) string {
|
||||
t.Helper()
|
||||
|
||||
|
||||
28
internal/jsonfile/jsonfile.go
Normal file
28
internal/jsonfile/jsonfile.go
Normal file
@@ -0,0 +1,28 @@
|
||||
package jsonfile
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
)
|
||||
|
||||
// Write creates or truncates path and writes deterministic indented JSON.
|
||||
func Write(path string, value any) (err error) {
|
||||
file, err := os.Create(path)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create %q: %w", path, err)
|
||||
}
|
||||
defer func() {
|
||||
closeErr := file.Close()
|
||||
if err == nil && closeErr != nil {
|
||||
err = fmt.Errorf("close %q: %w", path, closeErr)
|
||||
}
|
||||
}()
|
||||
|
||||
encoder := json.NewEncoder(file)
|
||||
encoder.SetIndent("", " ")
|
||||
if err := encoder.Encode(value); err != nil {
|
||||
return fmt.Errorf("encode %q: %w", path, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
69
internal/jsonfile/jsonfile_test.go
Normal file
69
internal/jsonfile/jsonfile_test.go
Normal file
@@ -0,0 +1,69 @@
|
||||
package jsonfile
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestWriteFormatsWithTwoSpaceIndentAndTrailingNewline(t *testing.T) {
|
||||
type payload struct {
|
||||
Name string `json:"name"`
|
||||
Items []int `json:"items"`
|
||||
}
|
||||
|
||||
path := filepath.Join(t.TempDir(), "out.json")
|
||||
value := payload{
|
||||
Name: "alpha",
|
||||
Items: []int{1, 2},
|
||||
}
|
||||
|
||||
if err := Write(path, value); err != nil {
|
||||
t.Fatalf("write failed: %v", err)
|
||||
}
|
||||
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("read output: %v", err)
|
||||
}
|
||||
|
||||
got := string(data)
|
||||
want := "{\n \"name\": \"alpha\",\n \"items\": [\n 1,\n 2\n ]\n}\n"
|
||||
if got != want {
|
||||
t.Fatalf("formatted JSON mismatch\nwant:\n%s\ngot:\n%s", want, got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriteProducesValidJSON(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "out.json")
|
||||
|
||||
value := map[string]any{
|
||||
"application": "seriatim",
|
||||
"segments": []map[string]any{
|
||||
{
|
||||
"id": 1,
|
||||
"speaker": "A",
|
||||
"text": "hello",
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
if err := Write(path, value); err != nil {
|
||||
t.Fatalf("write failed: %v", err)
|
||||
}
|
||||
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("read output: %v", err)
|
||||
}
|
||||
if !strings.HasSuffix(string(data), "\n") {
|
||||
t.Fatalf("output missing trailing newline: %q", string(data))
|
||||
}
|
||||
|
||||
var decoded map[string]any
|
||||
if err := json.Unmarshal(data, &decoded); err != nil {
|
||||
t.Fatalf("output is not valid JSON: %v", err)
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,7 @@
|
||||
package model
|
||||
|
||||
import "fmt"
|
||||
|
||||
// RawTranscript is a loaded input document before canonical normalization.
|
||||
type RawTranscript struct {
|
||||
Source string `json:"source"`
|
||||
@@ -61,6 +63,17 @@ type Segment struct {
|
||||
OverlapGroupID int `json:"overlap_group_id,omitempty"`
|
||||
}
|
||||
|
||||
// SegmentReference returns the best available external reference for a segment.
|
||||
func SegmentReference(segment Segment) string {
|
||||
if segment.Source != "" && segment.SourceSegmentIndex != nil {
|
||||
return fmt.Sprintf("%s#%d", segment.Source, *segment.SourceSegmentIndex)
|
||||
}
|
||||
if segment.SourceRef != "" {
|
||||
return segment.SourceRef
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// Word preserves optional word-level timing data.
|
||||
type Word struct {
|
||||
Text string `json:"text"`
|
||||
|
||||
41
internal/model/model_test.go
Normal file
41
internal/model/model_test.go
Normal file
@@ -0,0 +1,41 @@
|
||||
package model
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestSegmentReferenceUsesSourceAndIndexWhenAvailable(t *testing.T) {
|
||||
index := 3
|
||||
segment := Segment{
|
||||
Source: "input.json",
|
||||
SourceSegmentIndex: &index,
|
||||
SourceRef: "word-run:1:2:3",
|
||||
}
|
||||
|
||||
got := SegmentReference(segment)
|
||||
want := "input.json#3"
|
||||
if got != want {
|
||||
t.Fatalf("reference = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSegmentReferenceFallsBackToSourceRef(t *testing.T) {
|
||||
segment := Segment{
|
||||
Source: "input.json",
|
||||
SourceRef: "coalesce:2",
|
||||
}
|
||||
|
||||
got := SegmentReference(segment)
|
||||
want := "coalesce:2"
|
||||
if got != want {
|
||||
t.Fatalf("reference = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSegmentReferenceReturnsEmptyWhenNoReferenceFieldsPresent(t *testing.T) {
|
||||
segment := Segment{
|
||||
Source: "input.json",
|
||||
}
|
||||
|
||||
if got := SegmentReference(segment); got != "" {
|
||||
t.Fatalf("reference = %q, want empty", got)
|
||||
}
|
||||
}
|
||||
@@ -4,12 +4,12 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/artifact"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/buildinfo"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/jsonfile"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
||||
)
|
||||
|
||||
@@ -47,7 +47,7 @@ func Run(ctx context.Context, cfg config.NormalizeConfig) error {
|
||||
return err
|
||||
}
|
||||
|
||||
if err := writeOutputJSON(cfg.OutputFile, built.Output); err != nil {
|
||||
if err := jsonfile.Write(cfg.OutputFile, built.Output); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -118,18 +118,3 @@ func Run(ctx context.Context, cfg config.NormalizeConfig) error {
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func writeOutputJSON(path string, value any) error {
|
||||
file, err := os.Create(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
encoder := json.NewEncoder(file)
|
||||
encoder.SetIndent("", " ")
|
||||
if err := encoder.Encode(value); err != nil {
|
||||
return fmt.Errorf("encode normalize output JSON: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
package overlap
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/model"
|
||||
@@ -121,13 +120,7 @@ func distinctSpeakers(segments []model.Segment, indices []int) []string {
|
||||
|
||||
// SegmentRef returns the stable overlap reference for a segment.
|
||||
func SegmentRef(segment model.Segment) string {
|
||||
if segment.SourceSegmentIndex != nil {
|
||||
return fmt.Sprintf("%s#%d", segment.Source, *segment.SourceSegmentIndex)
|
||||
}
|
||||
if segment.SourceRef != "" {
|
||||
return segment.SourceRef
|
||||
}
|
||||
return segment.Source
|
||||
return model.SegmentReference(segment)
|
||||
}
|
||||
|
||||
func clearExisting(in *model.MergedTranscript) {
|
||||
|
||||
@@ -1,9 +1,6 @@
|
||||
package report
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"os"
|
||||
)
|
||||
import "gitea.maximumdirect.net/eric/seriatim/internal/jsonfile"
|
||||
|
||||
// Severity classifies report events.
|
||||
type Severity string
|
||||
@@ -62,13 +59,5 @@ func Warning(stage string, module string, message string) Event {
|
||||
|
||||
// WriteJSON writes a deterministic JSON report.
|
||||
func WriteJSON(path string, rpt Report) error {
|
||||
file, err := os.Create(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
enc := json.NewEncoder(file)
|
||||
enc.SetIndent("", " ")
|
||||
return enc.Encode(rpt)
|
||||
return jsonfile.Write(path, rpt)
|
||||
}
|
||||
|
||||
@@ -44,54 +44,29 @@ type MinimalResult struct {
|
||||
RemovedIDs []int
|
||||
}
|
||||
|
||||
type projection struct {
|
||||
retainedIndexes []int
|
||||
oldToNewID map[int]int
|
||||
removedIDs []int
|
||||
}
|
||||
|
||||
// Apply trims a full seriatim output transcript by segment ID.
|
||||
func Apply(input schema.Transcript, opts Options) (Result, error) {
|
||||
if err := validateMode(opts.Mode); err != nil {
|
||||
return Result{}, err
|
||||
}
|
||||
|
||||
selected := opts.Selector.IDs()
|
||||
if len(selected) == 0 {
|
||||
return Result{}, fmt.Errorf("selector cannot be empty")
|
||||
}
|
||||
|
||||
inputIDs := make([]int, len(input.Segments))
|
||||
for index, segment := range input.Segments {
|
||||
inputIDs[index] = segment.ID
|
||||
}
|
||||
|
||||
idIndex, err := validateInputIDs(inputIDs)
|
||||
proj, err := projectSegmentIDs(inputIDs, opts)
|
||||
if err != nil {
|
||||
return Result{}, err
|
||||
}
|
||||
|
||||
if err := validateSelectedIDsExist(selected, idIndex); err != nil {
|
||||
return Result{}, err
|
||||
}
|
||||
|
||||
kept := make([]schema.Segment, 0, len(input.Segments))
|
||||
removed := make([]int, 0, len(input.Segments))
|
||||
oldToNew := make(map[int]int, len(input.Segments))
|
||||
for _, segment := range input.Segments {
|
||||
keep := opts.Mode == ModeKeep && opts.Selector.Contains(segment.ID)
|
||||
if opts.Mode == ModeRemove {
|
||||
keep = !opts.Selector.Contains(segment.ID)
|
||||
}
|
||||
|
||||
if !keep {
|
||||
removed = append(removed, segment.ID)
|
||||
continue
|
||||
}
|
||||
|
||||
rewritten := copySegment(segment)
|
||||
rewritten.ID = len(kept) + 1
|
||||
kept := make([]schema.Segment, len(proj.retainedIndexes))
|
||||
for outputIndex, inputIndex := range proj.retainedIndexes {
|
||||
rewritten := copySegment(input.Segments[inputIndex])
|
||||
rewritten.ID = outputIndex + 1
|
||||
rewritten.OverlapGroupID = 0
|
||||
kept = append(kept, rewritten)
|
||||
oldToNew[segment.ID] = rewritten.ID
|
||||
}
|
||||
|
||||
if len(kept) == 0 && !opts.AllowEmpty {
|
||||
return Result{}, fmt.Errorf("trim operation produced an empty transcript; set AllowEmpty to true to permit this")
|
||||
kept[outputIndex] = rewritten
|
||||
}
|
||||
|
||||
kept, groups := recomputeOverlapGroups(kept)
|
||||
@@ -104,62 +79,35 @@ func Apply(input schema.Transcript, opts Options) (Result, error) {
|
||||
out.OverlapGroups = groups
|
||||
return Result{
|
||||
Transcript: out,
|
||||
OldToNewID: oldToNew,
|
||||
RemovedIDs: removed,
|
||||
OldToNewID: proj.oldToNewID,
|
||||
RemovedIDs: proj.removedIDs,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// ApplyIntermediate trims an intermediate seriatim output transcript by
|
||||
// segment ID.
|
||||
func ApplyIntermediate(input schema.IntermediateTranscript, opts Options) (IntermediateResult, error) {
|
||||
if err := validateMode(opts.Mode); err != nil {
|
||||
return IntermediateResult{}, err
|
||||
}
|
||||
|
||||
selected := opts.Selector.IDs()
|
||||
if len(selected) == 0 {
|
||||
return IntermediateResult{}, fmt.Errorf("selector cannot be empty")
|
||||
}
|
||||
|
||||
inputIDs := make([]int, len(input.Segments))
|
||||
for index, segment := range input.Segments {
|
||||
inputIDs[index] = segment.ID
|
||||
}
|
||||
idIndex, err := validateInputIDs(inputIDs)
|
||||
proj, err := projectSegmentIDs(inputIDs, opts)
|
||||
if err != nil {
|
||||
return IntermediateResult{}, err
|
||||
}
|
||||
if err := validateSelectedIDsExist(selected, idIndex); err != nil {
|
||||
return IntermediateResult{}, err
|
||||
}
|
||||
|
||||
kept := make([]schema.IntermediateSegment, 0, len(input.Segments))
|
||||
removed := make([]int, 0, len(input.Segments))
|
||||
oldToNew := make(map[int]int, len(input.Segments))
|
||||
for _, segment := range input.Segments {
|
||||
keep := opts.Mode == ModeKeep && opts.Selector.Contains(segment.ID)
|
||||
if opts.Mode == ModeRemove {
|
||||
keep = !opts.Selector.Contains(segment.ID)
|
||||
}
|
||||
if !keep {
|
||||
removed = append(removed, segment.ID)
|
||||
continue
|
||||
}
|
||||
|
||||
kept := make([]schema.IntermediateSegment, len(proj.retainedIndexes))
|
||||
for outputIndex, inputIndex := range proj.retainedIndexes {
|
||||
segment := input.Segments[inputIndex]
|
||||
rewritten := schema.IntermediateSegment{
|
||||
ID: len(kept) + 1,
|
||||
ID: outputIndex + 1,
|
||||
Start: segment.Start,
|
||||
End: segment.End,
|
||||
Speaker: segment.Speaker,
|
||||
Text: segment.Text,
|
||||
Categories: append([]string(nil), segment.Categories...),
|
||||
}
|
||||
kept = append(kept, rewritten)
|
||||
oldToNew[segment.ID] = rewritten.ID
|
||||
}
|
||||
|
||||
if len(kept) == 0 && !opts.AllowEmpty {
|
||||
return IntermediateResult{}, fmt.Errorf("trim operation produced an empty transcript; set AllowEmpty to true to permit this")
|
||||
kept[outputIndex] = rewritten
|
||||
}
|
||||
|
||||
return IntermediateResult{
|
||||
@@ -171,60 +119,33 @@ func ApplyIntermediate(input schema.IntermediateTranscript, opts Options) (Inter
|
||||
},
|
||||
Segments: kept,
|
||||
},
|
||||
OldToNewID: oldToNew,
|
||||
RemovedIDs: removed,
|
||||
OldToNewID: proj.oldToNewID,
|
||||
RemovedIDs: proj.removedIDs,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// ApplyMinimal trims a minimal seriatim output transcript by segment ID.
|
||||
func ApplyMinimal(input schema.MinimalTranscript, opts Options) (MinimalResult, error) {
|
||||
if err := validateMode(opts.Mode); err != nil {
|
||||
return MinimalResult{}, err
|
||||
}
|
||||
|
||||
selected := opts.Selector.IDs()
|
||||
if len(selected) == 0 {
|
||||
return MinimalResult{}, fmt.Errorf("selector cannot be empty")
|
||||
}
|
||||
|
||||
inputIDs := make([]int, len(input.Segments))
|
||||
for index, segment := range input.Segments {
|
||||
inputIDs[index] = segment.ID
|
||||
}
|
||||
idIndex, err := validateInputIDs(inputIDs)
|
||||
proj, err := projectSegmentIDs(inputIDs, opts)
|
||||
if err != nil {
|
||||
return MinimalResult{}, err
|
||||
}
|
||||
if err := validateSelectedIDsExist(selected, idIndex); err != nil {
|
||||
return MinimalResult{}, err
|
||||
}
|
||||
|
||||
kept := make([]schema.MinimalSegment, 0, len(input.Segments))
|
||||
removed := make([]int, 0, len(input.Segments))
|
||||
oldToNew := make(map[int]int, len(input.Segments))
|
||||
for _, segment := range input.Segments {
|
||||
keep := opts.Mode == ModeKeep && opts.Selector.Contains(segment.ID)
|
||||
if opts.Mode == ModeRemove {
|
||||
keep = !opts.Selector.Contains(segment.ID)
|
||||
}
|
||||
if !keep {
|
||||
removed = append(removed, segment.ID)
|
||||
continue
|
||||
}
|
||||
|
||||
kept := make([]schema.MinimalSegment, len(proj.retainedIndexes))
|
||||
for outputIndex, inputIndex := range proj.retainedIndexes {
|
||||
segment := input.Segments[inputIndex]
|
||||
rewritten := schema.MinimalSegment{
|
||||
ID: len(kept) + 1,
|
||||
ID: outputIndex + 1,
|
||||
Start: segment.Start,
|
||||
End: segment.End,
|
||||
Speaker: segment.Speaker,
|
||||
Text: segment.Text,
|
||||
}
|
||||
kept = append(kept, rewritten)
|
||||
oldToNew[segment.ID] = rewritten.ID
|
||||
}
|
||||
|
||||
if len(kept) == 0 && !opts.AllowEmpty {
|
||||
return MinimalResult{}, fmt.Errorf("trim operation produced an empty transcript; set AllowEmpty to true to permit this")
|
||||
kept[outputIndex] = rewritten
|
||||
}
|
||||
|
||||
return MinimalResult{
|
||||
@@ -236,11 +157,53 @@ func ApplyMinimal(input schema.MinimalTranscript, opts Options) (MinimalResult,
|
||||
},
|
||||
Segments: kept,
|
||||
},
|
||||
OldToNewID: oldToNew,
|
||||
RemovedIDs: removed,
|
||||
OldToNewID: proj.oldToNewID,
|
||||
RemovedIDs: proj.removedIDs,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func projectSegmentIDs(ids []int, opts Options) (projection, error) {
|
||||
if err := validateMode(opts.Mode); err != nil {
|
||||
return projection{}, err
|
||||
}
|
||||
|
||||
selected := opts.Selector.IDs()
|
||||
if len(selected) == 0 {
|
||||
return projection{}, fmt.Errorf("selector cannot be empty")
|
||||
}
|
||||
|
||||
idIndex, err := validateInputIDs(ids)
|
||||
if err != nil {
|
||||
return projection{}, err
|
||||
}
|
||||
if err := validateSelectedIDsExist(selected, idIndex); err != nil {
|
||||
return projection{}, err
|
||||
}
|
||||
|
||||
result := projection{
|
||||
retainedIndexes: make([]int, 0, len(ids)),
|
||||
oldToNewID: make(map[int]int, len(ids)),
|
||||
removedIDs: make([]int, 0, len(ids)),
|
||||
}
|
||||
for index, id := range ids {
|
||||
keep := opts.Mode == ModeKeep && opts.Selector.Contains(id)
|
||||
if opts.Mode == ModeRemove {
|
||||
keep = !opts.Selector.Contains(id)
|
||||
}
|
||||
if !keep {
|
||||
result.removedIDs = append(result.removedIDs, id)
|
||||
continue
|
||||
}
|
||||
result.retainedIndexes = append(result.retainedIndexes, index)
|
||||
result.oldToNewID[id] = len(result.retainedIndexes)
|
||||
}
|
||||
|
||||
if len(result.retainedIndexes) == 0 && !opts.AllowEmpty {
|
||||
return projection{}, fmt.Errorf("trim operation produced an empty transcript; set AllowEmpty to true to permit this")
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func validateMode(mode Mode) error {
|
||||
switch mode {
|
||||
case ModeKeep, ModeRemove:
|
||||
|
||||
@@ -399,6 +399,106 @@ func TestApplyMinimalDoesNotIncludeOverlapGroups(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplySelectorPolicyIsSharedAcrossSchemas(t *testing.T) {
|
||||
type testCase struct {
|
||||
name string
|
||||
opts Options
|
||||
wantTexts []string
|
||||
wantOldToNew map[int]int
|
||||
wantRemoved []int
|
||||
wantSegmentCount int
|
||||
wantErrorSubstring string
|
||||
}
|
||||
|
||||
cases := []testCase{
|
||||
{
|
||||
name: "keep preserves input order regardless of selector order",
|
||||
opts: Options{Mode: ModeKeep, Selector: mustParseSelector(t, "4,1,3")},
|
||||
wantTexts: []string{"alpha", "gamma", "delta"},
|
||||
wantOldToNew: map[int]int{1: 1, 3: 2, 4: 3},
|
||||
wantRemoved: []int{2},
|
||||
wantSegmentCount: 3,
|
||||
},
|
||||
{
|
||||
name: "remove reports deterministic renumbering metadata",
|
||||
opts: Options{Mode: ModeRemove, Selector: mustParseSelector(t, "2,4")},
|
||||
wantTexts: []string{"alpha", "gamma"},
|
||||
wantOldToNew: map[int]int{1: 1, 3: 2},
|
||||
wantRemoved: []int{2, 4},
|
||||
wantSegmentCount: 2,
|
||||
},
|
||||
{
|
||||
name: "missing selected id returns error",
|
||||
opts: Options{Mode: ModeKeep, Selector: mustParseSelector(t, "9")},
|
||||
wantErrorSubstring: "does not exist",
|
||||
},
|
||||
{
|
||||
name: "empty selector returns error",
|
||||
opts: Options{Mode: ModeKeep, Selector: Selector{}},
|
||||
wantErrorSubstring: "selector cannot be empty",
|
||||
},
|
||||
{
|
||||
name: "invalid mode returns error",
|
||||
opts: Options{Mode: Mode("bad"), Selector: mustParseSelector(t, "1")},
|
||||
wantErrorSubstring: `invalid trim mode "bad"`,
|
||||
},
|
||||
{
|
||||
name: "empty output blocked when allow empty is false",
|
||||
opts: Options{Mode: ModeRemove, Selector: mustParseSelector(t, "1-4")},
|
||||
wantErrorSubstring: "empty transcript",
|
||||
},
|
||||
{
|
||||
name: "empty output allowed when allow empty is true",
|
||||
opts: Options{Mode: ModeRemove, Selector: mustParseSelector(t, "1-4"), AllowEmpty: true},
|
||||
wantTexts: []string{},
|
||||
wantOldToNew: map[int]int{},
|
||||
wantRemoved: []int{1, 2, 3, 4},
|
||||
wantSegmentCount: 0,
|
||||
},
|
||||
}
|
||||
|
||||
for _, test := range cases {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
fullInput := fullTranscriptFixture()
|
||||
intermediateInput := intermediateFixture()
|
||||
minimalInput := minimalFixture()
|
||||
|
||||
fullResult, fullErr := Apply(fullInput, test.opts)
|
||||
intermediateResult, intermediateErr := ApplyIntermediate(intermediateInput, test.opts)
|
||||
minimalResult, minimalErr := ApplyMinimal(minimalInput, test.opts)
|
||||
|
||||
if test.wantErrorSubstring != "" {
|
||||
assertErrorContains(t, fullErr, test.wantErrorSubstring)
|
||||
assertErrorContains(t, intermediateErr, test.wantErrorSubstring)
|
||||
assertErrorContains(t, minimalErr, test.wantErrorSubstring)
|
||||
return
|
||||
}
|
||||
if fullErr != nil {
|
||||
t.Fatalf("apply full failed: %v", fullErr)
|
||||
}
|
||||
if intermediateErr != nil {
|
||||
t.Fatalf("apply intermediate failed: %v", intermediateErr)
|
||||
}
|
||||
if minimalErr != nil {
|
||||
t.Fatalf("apply minimal failed: %v", minimalErr)
|
||||
}
|
||||
|
||||
assertIntSlice(t, extractFullIDs(fullResult.Transcript.Segments), extractSequentialIDs(test.wantSegmentCount))
|
||||
assertIntSlice(t, extractIntermediateIDs(intermediateResult.Transcript.Segments), extractSequentialIDs(test.wantSegmentCount))
|
||||
assertIntSlice(t, extractMinimalIDs(minimalResult.Transcript.Segments), extractSequentialIDs(test.wantSegmentCount))
|
||||
assertStringSlice(t, extractFullTexts(fullResult.Transcript.Segments), test.wantTexts)
|
||||
assertStringSlice(t, extractIntermediateTexts(intermediateResult.Transcript.Segments), test.wantTexts)
|
||||
assertStringSlice(t, extractMinimalTexts(minimalResult.Transcript.Segments), test.wantTexts)
|
||||
assertIntMap(t, fullResult.OldToNewID, test.wantOldToNew)
|
||||
assertIntMap(t, intermediateResult.OldToNewID, test.wantOldToNew)
|
||||
assertIntMap(t, minimalResult.OldToNewID, test.wantOldToNew)
|
||||
assertIntSlice(t, fullResult.RemovedIDs, test.wantRemoved)
|
||||
assertIntSlice(t, intermediateResult.RemovedIDs, test.wantRemoved)
|
||||
assertIntSlice(t, minimalResult.RemovedIDs, test.wantRemoved)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyOutputInvariantsValidAfterRenumberAndOverlapRecompute(t *testing.T) {
|
||||
input := overlapTranscriptFixture()
|
||||
selector := mustParseSelector(t, "2,1")
|
||||
@@ -666,3 +766,108 @@ func equalStringSlices(got []string, want []string) bool {
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func assertErrorContains(t *testing.T, err error, substring string) {
|
||||
t.Helper()
|
||||
if err == nil {
|
||||
t.Fatalf("expected error containing %q", substring)
|
||||
}
|
||||
if !strings.Contains(err.Error(), substring) {
|
||||
t.Fatalf("error %q does not contain %q", err.Error(), substring)
|
||||
}
|
||||
}
|
||||
|
||||
func assertStringSlice(t *testing.T, got []string, want []string) {
|
||||
t.Helper()
|
||||
if !equalStringSlices(got, want) {
|
||||
t.Fatalf("slice = %v, want %v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func extractSequentialIDs(count int) []int {
|
||||
ids := make([]int, count)
|
||||
for index := range ids {
|
||||
ids[index] = index + 1
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
func extractFullIDs(segments []schema.Segment) []int {
|
||||
ids := make([]int, len(segments))
|
||||
for index, segment := range segments {
|
||||
ids[index] = segment.ID
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
func extractIntermediateIDs(segments []schema.IntermediateSegment) []int {
|
||||
ids := make([]int, len(segments))
|
||||
for index, segment := range segments {
|
||||
ids[index] = segment.ID
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
func extractMinimalIDs(segments []schema.MinimalSegment) []int {
|
||||
ids := make([]int, len(segments))
|
||||
for index, segment := range segments {
|
||||
ids[index] = segment.ID
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
func extractFullTexts(segments []schema.Segment) []string {
|
||||
texts := make([]string, len(segments))
|
||||
for index, segment := range segments {
|
||||
texts[index] = segment.Text
|
||||
}
|
||||
return texts
|
||||
}
|
||||
|
||||
func extractIntermediateTexts(segments []schema.IntermediateSegment) []string {
|
||||
texts := make([]string, len(segments))
|
||||
for index, segment := range segments {
|
||||
texts[index] = segment.Text
|
||||
}
|
||||
return texts
|
||||
}
|
||||
|
||||
func extractMinimalTexts(segments []schema.MinimalSegment) []string {
|
||||
texts := make([]string, len(segments))
|
||||
for index, segment := range segments {
|
||||
texts[index] = segment.Text
|
||||
}
|
||||
return texts
|
||||
}
|
||||
|
||||
func intermediateFixture() schema.IntermediateTranscript {
|
||||
return schema.IntermediateTranscript{
|
||||
Metadata: schema.IntermediateMetadata{
|
||||
Application: "seriatim",
|
||||
Version: "v-test",
|
||||
OutputSchema: schema.OutputSchemaIntermediate,
|
||||
},
|
||||
Segments: []schema.IntermediateSegment{
|
||||
{ID: 1, Start: 1, End: 2, Speaker: "Alice", Text: "alpha", Categories: []string{"word-run"}},
|
||||
{ID: 2, Start: 2, End: 3, Speaker: "Bob", Text: "beta", Categories: []string{"filler", "backchannel"}},
|
||||
{ID: 3, Start: 3, End: 4, Speaker: "Carol", Text: "gamma", Categories: []string{"normal"}},
|
||||
{ID: 4, Start: 4, End: 5, Speaker: "Dan", Text: "delta", Categories: []string{"normal"}},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func minimalFixture() schema.MinimalTranscript {
|
||||
return schema.MinimalTranscript{
|
||||
Metadata: schema.MinimalMetadata{
|
||||
Application: "seriatim",
|
||||
Version: "v-test",
|
||||
OutputSchema: schema.OutputSchemaMinimal,
|
||||
},
|
||||
Segments: []schema.MinimalSegment{
|
||||
{ID: 1, Start: 1, End: 2, Speaker: "Alice", Text: "alpha"},
|
||||
{ID: 2, Start: 2, End: 3, Speaker: "Bob", Text: "beta"},
|
||||
{ID: 3, Start: 3, End: 4, Speaker: "Carol", Text: "gamma"},
|
||||
{ID: 4, Start: 4, End: 5, Speaker: "Dan", Text: "delta"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -8,9 +8,9 @@ import (
|
||||
)
|
||||
|
||||
const (
|
||||
SchemaMinimal = "seriatim-minimal"
|
||||
SchemaIntermediate = "seriatim-intermediate"
|
||||
SchemaFull = "seriatim-full"
|
||||
SchemaMinimal = schema.OutputSchemaMinimal
|
||||
SchemaIntermediate = schema.OutputSchemaIntermediate
|
||||
SchemaFull = schema.OutputSchemaFull
|
||||
)
|
||||
|
||||
// Artifact stores a parsed seriatim output artifact of one supported schema.
|
||||
|
||||
156
internal/trim/run.go
Normal file
156
internal/trim/run.go
Normal file
@@ -0,0 +1,156 @@
|
||||
package trim
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"sort"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/jsonfile"
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
||||
)
|
||||
|
||||
type auditReport struct {
|
||||
Operation string `json:"operation"`
|
||||
InputFile string `json:"input_file"`
|
||||
OutputFile string `json:"output_file"`
|
||||
InputSchema string `json:"input_schema"`
|
||||
OutputSchema string `json:"output_schema"`
|
||||
Mode string `json:"mode"`
|
||||
Selector string `json:"selector"`
|
||||
SelectedIDs []int `json:"selected_ids"`
|
||||
AllowEmpty bool `json:"allow_empty"`
|
||||
InputSegmentCount int `json:"input_segment_count"`
|
||||
RetainedSegmentCount int `json:"retained_segment_count"`
|
||||
RemovedSegmentCount int `json:"removed_segment_count"`
|
||||
RemovedInputIDs []int `json:"removed_input_ids"`
|
||||
OldToNewIDMapping []idMapping `json:"old_to_new_id_mapping"`
|
||||
OverlapGroupsRecomputed bool `json:"overlap_groups_recomputed"`
|
||||
}
|
||||
|
||||
type idMapping struct {
|
||||
OldID int `json:"old_id"`
|
||||
NewID int `json:"new_id"`
|
||||
}
|
||||
|
||||
// Run executes artifact-level trim orchestration.
|
||||
func Run(ctx context.Context, cfg config.TrimConfig) error {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
selector, err := ParseSelector(cfg.Selector)
|
||||
if err != nil {
|
||||
return fmt.Errorf("invalid selector %q: %w", cfg.Selector, err)
|
||||
}
|
||||
|
||||
data, err := os.ReadFile(cfg.InputFile)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read --input-file %q: %w", cfg.InputFile, err)
|
||||
}
|
||||
|
||||
artifact, err := ParseArtifactJSON(data)
|
||||
if err != nil {
|
||||
return fmt.Errorf("--input-file %q: %w", cfg.InputFile, err)
|
||||
}
|
||||
inputSegmentCount := artifact.SegmentCount()
|
||||
inputSchema := artifact.Schema
|
||||
|
||||
mode := ModeKeep
|
||||
if cfg.Mode == "remove" {
|
||||
mode = ModeRemove
|
||||
}
|
||||
|
||||
trimmed, err := ApplyArtifact(artifact, Options{
|
||||
Mode: mode,
|
||||
Selector: selector,
|
||||
AllowEmpty: cfg.AllowEmpty,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
outputSchema := artifact.Schema
|
||||
if cfg.OutputSchema != "" {
|
||||
outputSchema = cfg.OutputSchema
|
||||
}
|
||||
|
||||
outputArtifact, err := ConvertArtifact(trimmed.Artifact, outputSchema)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if err := ValidateArtifact(outputArtifact); err != nil {
|
||||
return fmt.Errorf("validate trimmed output: %w", err)
|
||||
}
|
||||
|
||||
if err := jsonfile.Write(cfg.OutputFile, outputArtifact.Value()); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if cfg.ReportFile == "" {
|
||||
return nil
|
||||
}
|
||||
|
||||
audit := auditReport{
|
||||
Operation: "trim",
|
||||
InputFile: cfg.InputFile,
|
||||
OutputFile: cfg.OutputFile,
|
||||
InputSchema: inputSchema,
|
||||
OutputSchema: outputArtifact.Schema,
|
||||
Mode: cfg.Mode,
|
||||
Selector: cfg.Selector,
|
||||
SelectedIDs: selector.IDs(),
|
||||
AllowEmpty: cfg.AllowEmpty,
|
||||
InputSegmentCount: inputSegmentCount,
|
||||
RetainedSegmentCount: len(trimmed.OldToNewID),
|
||||
RemovedSegmentCount: len(trimmed.RemovedIDs),
|
||||
RemovedInputIDs: append([]int(nil), trimmed.RemovedIDs...),
|
||||
OldToNewIDMapping: orderedIDMapping(trimmed.OldToNewID),
|
||||
OverlapGroupsRecomputed: trimmed.OverlapGroupsRecomputed,
|
||||
}
|
||||
auditJSON, err := json.Marshal(audit)
|
||||
if err != nil {
|
||||
return fmt.Errorf("marshal trim audit report: %w", err)
|
||||
}
|
||||
|
||||
rpt := report.Report{
|
||||
Metadata: report.Metadata{
|
||||
Application: outputArtifact.Application(),
|
||||
Version: outputArtifact.Version(),
|
||||
InputReader: "trim-artifact",
|
||||
InputFiles: []string{cfg.InputFile},
|
||||
OutputModules: []string{"json"},
|
||||
},
|
||||
Events: []report.Event{
|
||||
report.Info("trim", "trim", fmt.Sprintf("trimmed %d input segment(s) into %d output segment(s) with mode=%s", inputSegmentCount, outputArtifact.SegmentCount(), cfg.Mode)),
|
||||
report.Info("trim", "trim-audit", string(auditJSON)),
|
||||
report.Info("trim", "validate-output", fmt.Sprintf("validated %d output segment(s)", outputArtifact.SegmentCount())),
|
||||
report.Info("output", "json", "wrote transcript JSON"),
|
||||
},
|
||||
}
|
||||
if err := report.WriteJSON(cfg.ReportFile, rpt); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func orderedIDMapping(mapping map[int]int) []idMapping {
|
||||
keys := make([]int, 0, len(mapping))
|
||||
for oldID := range mapping {
|
||||
keys = append(keys, oldID)
|
||||
}
|
||||
sort.Ints(keys)
|
||||
|
||||
pairs := make([]idMapping, 0, len(keys))
|
||||
for _, oldID := range keys {
|
||||
pairs = append(pairs, idMapping{
|
||||
OldID: oldID,
|
||||
NewID: mapping[oldID],
|
||||
})
|
||||
}
|
||||
return pairs
|
||||
}
|
||||
28
internal/trim/run_test.go
Normal file
28
internal/trim/run_test.go
Normal file
@@ -0,0 +1,28 @@
|
||||
package trim
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||
)
|
||||
|
||||
func TestRunReturnsContextErrorBeforeWork(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
|
||||
err := Run(ctx, config.TrimConfig{
|
||||
InputFile: filepath.Join(dir, "input.json"),
|
||||
OutputFile: filepath.Join(dir, "output.json"),
|
||||
Mode: "keep",
|
||||
Selector: "1",
|
||||
OutputSchema: "",
|
||||
AllowEmpty: false,
|
||||
})
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("error = %v, want context.Canceled", err)
|
||||
}
|
||||
}
|
||||
@@ -14,6 +14,10 @@ import (
|
||||
var schemaFS embed.FS
|
||||
|
||||
const (
|
||||
OutputSchemaMinimal = "seriatim-minimal"
|
||||
OutputSchemaIntermediate = "seriatim-intermediate"
|
||||
OutputSchemaFull = "seriatim-full"
|
||||
|
||||
fullOutputSchemaPath = "full-output.schema.json"
|
||||
intermediateOutputSchemaPath = "intermediate-output.schema.json"
|
||||
minimalOutputSchemaPath = "minimal-output.schema.json"
|
||||
@@ -115,6 +119,25 @@ type OverlapGroup struct {
|
||||
Resolution string `json:"resolution"`
|
||||
}
|
||||
|
||||
// ValidOutputSchemaName reports whether value is a supported output schema name.
|
||||
func ValidOutputSchemaName(value string) bool {
|
||||
switch value {
|
||||
case OutputSchemaMinimal, OutputSchemaIntermediate, OutputSchemaFull:
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// OutputSchemaNames returns supported output schema names in validation order.
|
||||
func OutputSchemaNames() []string {
|
||||
return []string{
|
||||
OutputSchemaMinimal,
|
||||
OutputSchemaIntermediate,
|
||||
OutputSchemaFull,
|
||||
}
|
||||
}
|
||||
|
||||
// ValidateTranscript validates a full transcript against the public JSON
|
||||
// schema and seriatim-specific semantic rules.
|
||||
func ValidateTranscript(transcript Transcript) error {
|
||||
@@ -228,15 +251,17 @@ func outputSchema(schemaPath string) (*jsonschema.Schema, error) {
|
||||
}
|
||||
|
||||
func validateSemantics(transcript Transcript) error {
|
||||
segments := make([]segmentSemantics, len(transcript.Segments))
|
||||
for index, segment := range transcript.Segments {
|
||||
wantID := index + 1
|
||||
if segment.ID != wantID {
|
||||
return fmt.Errorf("segment %d has id %d; want %d", index, segment.ID, wantID)
|
||||
}
|
||||
if segment.End < segment.Start {
|
||||
return fmt.Errorf("segment %d has end %.3f before start %.3f", index, segment.End, segment.Start)
|
||||
segments[index] = segmentSemantics{
|
||||
id: segment.ID,
|
||||
start: segment.Start,
|
||||
end: segment.End,
|
||||
}
|
||||
}
|
||||
if err := validateSegmentSemantics(segments); err != nil {
|
||||
return err
|
||||
}
|
||||
for index, group := range transcript.OverlapGroups {
|
||||
if group.End < group.Start {
|
||||
return fmt.Errorf("overlap_group %d has end %.3f before start %.3f", index, group.End, group.Start)
|
||||
@@ -246,26 +271,43 @@ func validateSemantics(transcript Transcript) error {
|
||||
}
|
||||
|
||||
func validateIntermediateSemantics(transcript IntermediateTranscript) error {
|
||||
segments := make([]segmentSemantics, len(transcript.Segments))
|
||||
for index, segment := range transcript.Segments {
|
||||
wantID := index + 1
|
||||
if segment.ID != wantID {
|
||||
return fmt.Errorf("segment %d has id %d; want %d", index, segment.ID, wantID)
|
||||
}
|
||||
if segment.End < segment.Start {
|
||||
return fmt.Errorf("segment %d has end %.3f before start %.3f", index, segment.End, segment.Start)
|
||||
segments[index] = segmentSemantics{
|
||||
id: segment.ID,
|
||||
start: segment.Start,
|
||||
end: segment.End,
|
||||
}
|
||||
}
|
||||
return nil
|
||||
return validateSegmentSemantics(segments)
|
||||
}
|
||||
|
||||
func validateMinimalSemantics(transcript MinimalTranscript) error {
|
||||
segments := make([]segmentSemantics, len(transcript.Segments))
|
||||
for index, segment := range transcript.Segments {
|
||||
wantID := index + 1
|
||||
if segment.ID != wantID {
|
||||
return fmt.Errorf("segment %d has id %d; want %d", index, segment.ID, wantID)
|
||||
segments[index] = segmentSemantics{
|
||||
id: segment.ID,
|
||||
start: segment.Start,
|
||||
end: segment.End,
|
||||
}
|
||||
if segment.End < segment.Start {
|
||||
return fmt.Errorf("segment %d has end %.3f before start %.3f", index, segment.End, segment.Start)
|
||||
}
|
||||
return validateSegmentSemantics(segments)
|
||||
}
|
||||
|
||||
type segmentSemantics struct {
|
||||
id int
|
||||
start float64
|
||||
end float64
|
||||
}
|
||||
|
||||
func validateSegmentSemantics(segments []segmentSemantics) error {
|
||||
for index, segment := range segments {
|
||||
wantID := index + 1
|
||||
if segment.id != wantID {
|
||||
return fmt.Errorf("segment %d has id %d; want %d", index, segment.id, wantID)
|
||||
}
|
||||
if segment.end < segment.start {
|
||||
return fmt.Errorf("segment %d has end %.3f before start %.3f", index, segment.end, segment.start)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
|
||||
@@ -5,6 +5,43 @@ import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestValidOutputSchemaName(t *testing.T) {
|
||||
valid := []string{
|
||||
OutputSchemaMinimal,
|
||||
OutputSchemaIntermediate,
|
||||
OutputSchemaFull,
|
||||
}
|
||||
for _, name := range valid {
|
||||
if !ValidOutputSchemaName(name) {
|
||||
t.Fatalf("expected %q to be valid", name)
|
||||
}
|
||||
}
|
||||
|
||||
invalid := []string{"", "compact", "minimal", "seriatim"}
|
||||
for _, name := range invalid {
|
||||
if ValidOutputSchemaName(name) {
|
||||
t.Fatalf("expected %q to be invalid", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestOutputSchemaNames(t *testing.T) {
|
||||
names := OutputSchemaNames()
|
||||
want := []string{
|
||||
OutputSchemaMinimal,
|
||||
OutputSchemaIntermediate,
|
||||
OutputSchemaFull,
|
||||
}
|
||||
if len(names) != len(want) {
|
||||
t.Fatalf("len(names) = %d, want %d", len(names), len(want))
|
||||
}
|
||||
for index := range want {
|
||||
if names[index] != want[index] {
|
||||
t.Fatalf("names[%d] = %q, want %q", index, names[index], want[index])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidateTranscriptAcceptsValidTranscript(t *testing.T) {
|
||||
transcript := validTranscript()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user