Compare commits
11 Commits
e5173c78fe
...
f40d4add91
| Author | SHA1 | Date | |
|---|---|---|---|
| f40d4add91 | |||
| 16bb12face | |||
| f18e2428dc | |||
| 3b64e784a1 | |||
| 3744d229a2 | |||
| 9bbe1fb7f1 | |||
| b7a66f6cc4 | |||
| c8efdb53d3 | |||
| ab4b252b08 | |||
| e9028e08a4 | |||
| 332884f887 |
14
README.md
14
README.md
@@ -27,19 +27,19 @@ go run ./cmd/seriatim merge \
|
|||||||
- Configuration reference: [docs/config.md](docs/config.md)
|
- Configuration reference: [docs/config.md](docs/config.md)
|
||||||
- Operations guide: [docs/operations.md](docs/operations.md)
|
- Operations guide: [docs/operations.md](docs/operations.md)
|
||||||
- Troubleshooting: [docs/troubleshooting.md](docs/troubleshooting.md)
|
- Troubleshooting: [docs/troubleshooting.md](docs/troubleshooting.md)
|
||||||
- Integrations:
|
- Integration references:
|
||||||
- [docs/integrations/whisperx-json.md](docs/integrations/whisperx-json.md)
|
- [docs/integrations/whisperx-json.md](docs/integrations/whisperx-json.md)
|
||||||
- [docs/integrations/output-schemas.md](docs/integrations/output-schemas.md)
|
- [docs/integrations/output-schemas.md](docs/integrations/output-schemas.md)
|
||||||
- Development architecture policy: [docs/policy/architecture.md](docs/policy/architecture.md)
|
- Development policies:
|
||||||
- Contributor workflow: [docs/policy/development.md](docs/policy/development.md)
|
- [docs/policy/architecture.md](docs/policy/architecture.md)
|
||||||
- Documentation policy: [docs/policy/documentation.md](docs/policy/documentation.md)
|
- [docs/policy/development.md](docs/policy/development.md)
|
||||||
- Internal implementation docs:
|
- [docs/policy/documentation.md](docs/policy/documentation.md)
|
||||||
|
- Internal implementation references:
|
||||||
- [docs/internal/pipeline.md](docs/internal/pipeline.md)
|
- [docs/internal/pipeline.md](docs/internal/pipeline.md)
|
||||||
- [docs/internal/artifacts.md](docs/internal/artifacts.md)
|
- [docs/internal/artifacts.md](docs/internal/artifacts.md)
|
||||||
- [docs/internal/modules.md](docs/internal/modules.md)
|
- [docs/internal/modules.md](docs/internal/modules.md)
|
||||||
- Public JSON schemas:
|
- Public JSON schema files:
|
||||||
- [schema/minimal-output.schema.json](schema/minimal-output.schema.json)
|
- [schema/minimal-output.schema.json](schema/minimal-output.schema.json)
|
||||||
- [schema/intermediate-output.schema.json](schema/intermediate-output.schema.json)
|
- [schema/intermediate-output.schema.json](schema/intermediate-output.schema.json)
|
||||||
- [schema/full-output.schema.json](schema/full-output.schema.json)
|
- [schema/full-output.schema.json](schema/full-output.schema.json)
|
||||||
- Synthetic examples: [examples/README.md](examples/README.md)
|
- Synthetic examples: [examples/README.md](examples/README.md)
|
||||||
- Documentation roadmap: [docs/roadmap/documentation.md](docs/roadmap/documentation.md)
|
|
||||||
|
|||||||
@@ -179,4 +179,3 @@ go run ./cmd/seriatim normalize \
|
|||||||
- [../schema/minimal-output.schema.json](../schema/minimal-output.schema.json)
|
- [../schema/minimal-output.schema.json](../schema/minimal-output.schema.json)
|
||||||
- [../schema/intermediate-output.schema.json](../schema/intermediate-output.schema.json)
|
- [../schema/intermediate-output.schema.json](../schema/intermediate-output.schema.json)
|
||||||
- [../schema/full-output.schema.json](../schema/full-output.schema.json)
|
- [../schema/full-output.schema.json](../schema/full-output.schema.json)
|
||||||
- Documentation roadmap: [roadmap/documentation.md](roadmap/documentation.md)
|
|
||||||
|
|||||||
@@ -165,4 +165,3 @@ All commands:
|
|||||||
- [../schema/minimal-output.schema.json](../schema/minimal-output.schema.json)
|
- [../schema/minimal-output.schema.json](../schema/minimal-output.schema.json)
|
||||||
- [../schema/intermediate-output.schema.json](../schema/intermediate-output.schema.json)
|
- [../schema/intermediate-output.schema.json](../schema/intermediate-output.schema.json)
|
||||||
- [../schema/full-output.schema.json](../schema/full-output.schema.json)
|
- [../schema/full-output.schema.json](../schema/full-output.schema.json)
|
||||||
- Documentation roadmap: [roadmap/documentation.md](roadmap/documentation.md)
|
|
||||||
|
|||||||
@@ -51,29 +51,43 @@ Unknown/empty selection falls back to intermediate conversion.
|
|||||||
|
|
||||||
## Trim internals
|
## Trim internals
|
||||||
|
|
||||||
`internal/trim` is artifact-level projection, not merge reprocessing.
|
`internal/trim` handles artifact-level projection and does not execute merge
|
||||||
|
pipeline modules.
|
||||||
|
|
||||||
Core flow:
|
Run layer (`run.go`):
|
||||||
|
|
||||||
1. Parse selector (`internal/trim/selector.go`).
|
1. Parse selector from validated config.
|
||||||
2. Parse input artifact and detect schema (`ParseArtifactJSON`).
|
2. Read and parse input artifact JSON.
|
||||||
3. Apply keep/remove projection with sequential ID renumbering.
|
3. Apply trim projection through schema-aware artifact handling.
|
||||||
4. Recompute overlap groups only for full-schema artifacts.
|
4. Resolve output schema (preserve input schema unless overridden).
|
||||||
5. Optionally convert output schema when supported.
|
5. Validate output artifact.
|
||||||
6. Validate output artifact before write.
|
6. Write output JSON.
|
||||||
|
7. Optionally write report JSON with `trim-audit`.
|
||||||
|
|
||||||
Schema-conversion limits:
|
Apply layer (`apply.go`):
|
||||||
|
|
||||||
- full -> intermediate/minimal supported.
|
- one shared projection policy for selector mode, input ID validation, selected
|
||||||
- intermediate -> minimal supported.
|
ID existence checks, keep/remove filtering, removed IDs, and old-to-new ID
|
||||||
- minimal -> intermediate supported.
|
mappings
|
||||||
- intermediate/minimal -> full is rejected.
|
- schema-specific segment reconstruction for full/intermediate/minimal outputs
|
||||||
|
- overlap-group recomputation only for full-schema outputs
|
||||||
|
|
||||||
|
Artifact layer (`artifact.go`):
|
||||||
|
|
||||||
|
- schema detection for full/intermediate/minimal artifacts
|
||||||
|
- schema-preserving trim application
|
||||||
|
- supported schema conversions:
|
||||||
|
- full -> intermediate/minimal
|
||||||
|
- intermediate -> minimal
|
||||||
|
- minimal -> intermediate
|
||||||
|
- rejected conversion:
|
||||||
|
- intermediate/minimal -> full
|
||||||
|
|
||||||
Trim invariants:
|
Trim invariants:
|
||||||
|
|
||||||
- selected IDs must exist in input.
|
- selected IDs must exist in input.
|
||||||
- input IDs must be positive, unique, sequential.
|
- input IDs must be positive, unique, sequential.
|
||||||
- retained order follows input transcript order.
|
- retained segment order follows input transcript order.
|
||||||
- output IDs are reassigned to `1..N`.
|
- output IDs are reassigned to `1..N`.
|
||||||
|
|
||||||
## Normalize internals
|
## Normalize internals
|
||||||
|
|||||||
@@ -55,7 +55,8 @@ Output writer:
|
|||||||
- `autocorrect`: applies YAML replacement rules when configured.
|
- `autocorrect`: applies YAML replacement rules when configured.
|
||||||
- `assign-ids`: assigns final sequential IDs.
|
- `assign-ids`: assigns final sequential IDs.
|
||||||
- `validate-output`: validates selected public artifact shape.
|
- `validate-output`: validates selected public artifact shape.
|
||||||
- `json`: writes artifact JSON to `cfg.OutputFile`.
|
- `json`: writes artifact JSON to `cfg.OutputFile` through shared deterministic
|
||||||
|
JSON file writing.
|
||||||
|
|
||||||
Filesystem side effects are limited to:
|
Filesystem side effects are limited to:
|
||||||
|
|
||||||
|
|||||||
@@ -28,7 +28,7 @@ Reviewed documentation and policy:
|
|||||||
- `docs/internal/modules.md`
|
- `docs/internal/modules.md`
|
||||||
- `docs/integrations/output-schemas.md`
|
- `docs/integrations/output-schemas.md`
|
||||||
- `docs/integrations/whisperx-json.md`
|
- `docs/integrations/whisperx-json.md`
|
||||||
- `docs/roadmap/documentation.md`
|
- `docs/roadmap/cleanup.md`
|
||||||
|
|
||||||
Reviewed implementation areas:
|
Reviewed implementation areas:
|
||||||
|
|
||||||
|
|||||||
@@ -2,10 +2,9 @@ package builtin
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
"encoding/json"
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||||
|
"gitea.maximumdirect.net/eric/seriatim/internal/jsonfile"
|
||||||
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -20,15 +19,7 @@ func (jsonOutputWriter) Write(ctx context.Context, out any, rpt report.Report, c
|
|||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
file, err := os.Create(cfg.OutputFile)
|
if err := jsonfile.Write(cfg.OutputFile, out); err != nil {
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
defer file.Close()
|
|
||||||
|
|
||||||
enc := json.NewEncoder(file)
|
|
||||||
enc.SetIndent("", " ")
|
|
||||||
if err := enc.Encode(out); err != nil {
|
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
31
internal/cli/flags.go
Normal file
31
internal/cli/flags.go
Normal file
@@ -0,0 +1,31 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"github.com/spf13/cobra"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||||
|
)
|
||||||
|
|
||||||
|
func addOutputFileFlag(cmd *cobra.Command, target *string) {
|
||||||
|
cmd.Flags().StringVar(target, "output-file", "", "output transcript JSON file")
|
||||||
|
}
|
||||||
|
|
||||||
|
func addReportFileFlag(cmd *cobra.Command, target *string) {
|
||||||
|
cmd.Flags().StringVar(target, "report-file", "", "optional report JSON file")
|
||||||
|
}
|
||||||
|
|
||||||
|
func addOutputModulesFlag(cmd *cobra.Command, target *string) {
|
||||||
|
cmd.Flags().StringVar(target, "output-modules", config.DefaultOutputModules, "comma-separated output modules")
|
||||||
|
}
|
||||||
|
|
||||||
|
func addMergeOutputSchemaFlag(cmd *cobra.Command, target *string) {
|
||||||
|
cmd.Flags().StringVar(target, "output-schema", config.DefaultOutputSchema, "output JSON schema: seriatim-minimal, seriatim-intermediate (default), or seriatim-full")
|
||||||
|
}
|
||||||
|
|
||||||
|
func addNormalizeOutputSchemaFlag(cmd *cobra.Command, target *string) {
|
||||||
|
cmd.Flags().StringVar(target, "output-schema", config.DefaultOutputSchema, "output JSON schema: seriatim-minimal, seriatim-intermediate, or seriatim-full")
|
||||||
|
}
|
||||||
|
|
||||||
|
func addTrimOutputSchemaFlag(cmd *cobra.Command, target *string) {
|
||||||
|
cmd.Flags().StringVar(target, "output-schema", "", "optional output JSON schema override: seriatim-minimal, seriatim-intermediate, or seriatim-full")
|
||||||
|
}
|
||||||
@@ -31,13 +31,13 @@ func newMergeCommand() *cobra.Command {
|
|||||||
|
|
||||||
flags := cmd.Flags()
|
flags := cmd.Flags()
|
||||||
flags.StringArrayVar(&opts.InputFiles, "input-file", nil, "input transcript file; may be repeated")
|
flags.StringArrayVar(&opts.InputFiles, "input-file", nil, "input transcript file; may be repeated")
|
||||||
flags.StringVar(&opts.OutputFile, "output-file", "", "output transcript JSON file")
|
addOutputFileFlag(cmd, &opts.OutputFile)
|
||||||
flags.StringVar(&opts.ReportFile, "report-file", "", "optional report JSON file")
|
addReportFileFlag(cmd, &opts.ReportFile)
|
||||||
flags.StringVar(&opts.SpeakersFile, "speakers", "", "speaker map file")
|
flags.StringVar(&opts.SpeakersFile, "speakers", "", "speaker map file")
|
||||||
flags.StringVar(&opts.AutocorrectFile, "autocorrect", "", "autocorrect rules file")
|
flags.StringVar(&opts.AutocorrectFile, "autocorrect", "", "autocorrect rules file")
|
||||||
flags.StringVar(&opts.InputReader, "input-reader", config.DefaultInputReader, "input reader module")
|
flags.StringVar(&opts.InputReader, "input-reader", config.DefaultInputReader, "input reader module")
|
||||||
flags.StringVar(&opts.OutputModules, "output-modules", config.DefaultOutputModules, "comma-separated output modules")
|
addOutputModulesFlag(cmd, &opts.OutputModules)
|
||||||
flags.StringVar(&opts.OutputSchema, "output-schema", config.DefaultOutputSchema, "output JSON schema: seriatim-minimal, seriatim-intermediate (default), or seriatim-full")
|
addMergeOutputSchemaFlag(cmd, &opts.OutputSchema)
|
||||||
flags.StringVar(&opts.PreprocessingModules, "preprocessing-modules", config.DefaultPreprocessingModules, "comma-separated preprocessing modules")
|
flags.StringVar(&opts.PreprocessingModules, "preprocessing-modules", config.DefaultPreprocessingModules, "comma-separated preprocessing modules")
|
||||||
flags.StringVar(&opts.PostprocessingModules, "postprocessing-modules", config.DefaultPostprocessingModules, "comma-separated postprocessing modules")
|
flags.StringVar(&opts.PostprocessingModules, "postprocessing-modules", config.DefaultPostprocessingModules, "comma-separated postprocessing modules")
|
||||||
flags.StringVar(&opts.CoalesceGap, "coalesce-gap", config.DefaultCoalesceGapValue, "maximum same-speaker gap in seconds for coalesce")
|
flags.StringVar(&opts.CoalesceGap, "coalesce-gap", config.DefaultCoalesceGapValue, "maximum same-speaker gap in seconds for coalesce")
|
||||||
|
|||||||
@@ -30,10 +30,10 @@ func newNormalizeCommand() *cobra.Command {
|
|||||||
|
|
||||||
flags := cmd.Flags()
|
flags := cmd.Flags()
|
||||||
flags.StringVar(&opts.InputFile, "input-file", "", "input transcript JSON file")
|
flags.StringVar(&opts.InputFile, "input-file", "", "input transcript JSON file")
|
||||||
flags.StringVar(&opts.OutputFile, "output-file", "", "output transcript JSON file")
|
addOutputFileFlag(cmd, &opts.OutputFile)
|
||||||
flags.StringVar(&opts.ReportFile, "report-file", "", "optional report JSON file")
|
addReportFileFlag(cmd, &opts.ReportFile)
|
||||||
flags.StringVar(&opts.OutputSchema, "output-schema", config.DefaultOutputSchema, "output JSON schema: seriatim-minimal, seriatim-intermediate, or seriatim-full")
|
addNormalizeOutputSchemaFlag(cmd, &opts.OutputSchema)
|
||||||
flags.StringVar(&opts.OutputModules, "output-modules", config.DefaultOutputModules, "comma-separated output modules")
|
addOutputModulesFlag(cmd, &opts.OutputModules)
|
||||||
|
|
||||||
return cmd
|
return cmd
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,41 +1,12 @@
|
|||||||
package cli
|
package cli
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"encoding/json"
|
|
||||||
"fmt"
|
|
||||||
"os"
|
|
||||||
"sort"
|
|
||||||
|
|
||||||
"github.com/spf13/cobra"
|
"github.com/spf13/cobra"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||||
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
"gitea.maximumdirect.net/eric/seriatim/internal/trim"
|
||||||
triminternal "gitea.maximumdirect.net/eric/seriatim/internal/trim"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
type trimAuditReport struct {
|
|
||||||
Operation string `json:"operation"`
|
|
||||||
InputFile string `json:"input_file"`
|
|
||||||
OutputFile string `json:"output_file"`
|
|
||||||
InputSchema string `json:"input_schema"`
|
|
||||||
OutputSchema string `json:"output_schema"`
|
|
||||||
Mode string `json:"mode"`
|
|
||||||
Selector string `json:"selector"`
|
|
||||||
SelectedIDs []int `json:"selected_ids"`
|
|
||||||
AllowEmpty bool `json:"allow_empty"`
|
|
||||||
InputSegmentCount int `json:"input_segment_count"`
|
|
||||||
RetainedSegmentCount int `json:"retained_segment_count"`
|
|
||||||
RemovedSegmentCount int `json:"removed_segment_count"`
|
|
||||||
RemovedInputIDs []int `json:"removed_input_ids"`
|
|
||||||
OldToNewIDMapping []trimIDMapping `json:"old_to_new_id_mapping"`
|
|
||||||
OverlapGroupsRecomputed bool `json:"overlap_groups_recomputed"`
|
|
||||||
}
|
|
||||||
|
|
||||||
type trimIDMapping struct {
|
|
||||||
OldID int `json:"old_id"`
|
|
||||||
NewID int `json:"new_id"`
|
|
||||||
}
|
|
||||||
|
|
||||||
func newTrimCommand() *cobra.Command {
|
func newTrimCommand() *cobra.Command {
|
||||||
var opts config.TrimOptions
|
var opts config.TrimOptions
|
||||||
|
|
||||||
@@ -53,139 +24,18 @@ func newTrimCommand() *cobra.Command {
|
|||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
selector, err := triminternal.ParseSelector(cfg.Selector)
|
return trim.Run(cmd.Context(), cfg)
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("invalid selector %q: %w", cfg.Selector, err)
|
|
||||||
}
|
|
||||||
|
|
||||||
data, err := os.ReadFile(cfg.InputFile)
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("read --input-file %q: %w", cfg.InputFile, err)
|
|
||||||
}
|
|
||||||
|
|
||||||
artifact, err := triminternal.ParseArtifactJSON(data)
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("--input-file %q: %w", cfg.InputFile, err)
|
|
||||||
}
|
|
||||||
inputSegmentCount := artifact.SegmentCount()
|
|
||||||
inputSchema := artifact.Schema
|
|
||||||
|
|
||||||
mode := triminternal.ModeKeep
|
|
||||||
if cfg.Mode == "remove" {
|
|
||||||
mode = triminternal.ModeRemove
|
|
||||||
}
|
|
||||||
|
|
||||||
trimmed, err := triminternal.ApplyArtifact(artifact, triminternal.Options{
|
|
||||||
Mode: mode,
|
|
||||||
Selector: selector,
|
|
||||||
AllowEmpty: cfg.AllowEmpty,
|
|
||||||
})
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
|
|
||||||
outputSchema := artifact.Schema
|
|
||||||
if cfg.OutputSchema != "" {
|
|
||||||
outputSchema = cfg.OutputSchema
|
|
||||||
}
|
|
||||||
|
|
||||||
outputArtifact, err := triminternal.ConvertArtifact(trimmed.Artifact, outputSchema)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
|
|
||||||
if err := triminternal.ValidateArtifact(outputArtifact); err != nil {
|
|
||||||
return fmt.Errorf("validate trimmed output: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
if err := writeOutputJSON(cfg.OutputFile, outputArtifact.Value()); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
|
|
||||||
if cfg.ReportFile != "" {
|
|
||||||
audit := trimAuditReport{
|
|
||||||
Operation: "trim",
|
|
||||||
InputFile: cfg.InputFile,
|
|
||||||
OutputFile: cfg.OutputFile,
|
|
||||||
InputSchema: inputSchema,
|
|
||||||
OutputSchema: outputArtifact.Schema,
|
|
||||||
Mode: cfg.Mode,
|
|
||||||
Selector: cfg.Selector,
|
|
||||||
SelectedIDs: selector.IDs(),
|
|
||||||
AllowEmpty: cfg.AllowEmpty,
|
|
||||||
InputSegmentCount: inputSegmentCount,
|
|
||||||
RetainedSegmentCount: len(trimmed.OldToNewID),
|
|
||||||
RemovedSegmentCount: len(trimmed.RemovedIDs),
|
|
||||||
RemovedInputIDs: append([]int(nil), trimmed.RemovedIDs...),
|
|
||||||
OldToNewIDMapping: orderedIDMapping(trimmed.OldToNewID),
|
|
||||||
OverlapGroupsRecomputed: trimmed.OverlapGroupsRecomputed,
|
|
||||||
}
|
|
||||||
auditJSON, err := json.Marshal(audit)
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("marshal trim audit report: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
rpt := report.Report{
|
|
||||||
Metadata: report.Metadata{
|
|
||||||
Application: outputArtifact.Application(),
|
|
||||||
Version: outputArtifact.Version(),
|
|
||||||
InputReader: "trim-artifact",
|
|
||||||
InputFiles: []string{cfg.InputFile},
|
|
||||||
OutputModules: []string{"json"},
|
|
||||||
},
|
|
||||||
Events: []report.Event{
|
|
||||||
report.Info("trim", "trim", fmt.Sprintf("trimmed %d input segment(s) into %d output segment(s) with mode=%s", inputSegmentCount, outputArtifact.SegmentCount(), cfg.Mode)),
|
|
||||||
report.Info("trim", "trim-audit", string(auditJSON)),
|
|
||||||
report.Info("trim", "validate-output", fmt.Sprintf("validated %d output segment(s)", outputArtifact.SegmentCount())),
|
|
||||||
report.Info("output", "json", "wrote transcript JSON"),
|
|
||||||
},
|
|
||||||
}
|
|
||||||
if err := report.WriteJSON(cfg.ReportFile, rpt); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return nil
|
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
flags := cmd.Flags()
|
flags := cmd.Flags()
|
||||||
flags.StringVar(&opts.InputFile, "input-file", "", "input seriatim transcript artifact JSON file")
|
flags.StringVar(&opts.InputFile, "input-file", "", "input seriatim transcript artifact JSON file")
|
||||||
flags.StringVar(&opts.OutputFile, "output-file", "", "output transcript JSON file")
|
addOutputFileFlag(cmd, &opts.OutputFile)
|
||||||
flags.StringVar(&opts.ReportFile, "report-file", "", "optional report JSON file")
|
addReportFileFlag(cmd, &opts.ReportFile)
|
||||||
flags.StringVar(&opts.Keep, "keep", "", "segment ID selector to keep (for example: 1-10,15)")
|
flags.StringVar(&opts.Keep, "keep", "", "segment ID selector to keep (for example: 1-10,15)")
|
||||||
flags.StringVar(&opts.Remove, "remove", "", "segment ID selector to remove (for example: 1-10,15)")
|
flags.StringVar(&opts.Remove, "remove", "", "segment ID selector to remove (for example: 1-10,15)")
|
||||||
flags.StringVar(&opts.OutputSchema, "output-schema", "", "optional output JSON schema override: seriatim-minimal, seriatim-intermediate, or seriatim-full")
|
addTrimOutputSchemaFlag(cmd, &opts.OutputSchema)
|
||||||
flags.BoolVar(&opts.AllowEmpty, "allow-empty", false, "allow trimming to an empty transcript")
|
flags.BoolVar(&opts.AllowEmpty, "allow-empty", false, "allow trimming to an empty transcript")
|
||||||
|
|
||||||
return cmd
|
return cmd
|
||||||
}
|
}
|
||||||
|
|
||||||
func writeOutputJSON(path string, value any) error {
|
|
||||||
file, err := os.Create(path)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
defer file.Close()
|
|
||||||
|
|
||||||
enc := json.NewEncoder(file)
|
|
||||||
enc.SetIndent("", " ")
|
|
||||||
return enc.Encode(value)
|
|
||||||
}
|
|
||||||
|
|
||||||
func orderedIDMapping(mapping map[int]int) []trimIDMapping {
|
|
||||||
keys := make([]int, 0, len(mapping))
|
|
||||||
for oldID := range mapping {
|
|
||||||
keys = append(keys, oldID)
|
|
||||||
}
|
|
||||||
sort.Ints(keys)
|
|
||||||
|
|
||||||
pairs := make([]trimIDMapping, 0, len(keys))
|
|
||||||
for _, oldID := range keys {
|
|
||||||
pairs = append(pairs, trimIDMapping{
|
|
||||||
OldID: oldID,
|
|
||||||
NewID: mapping[oldID],
|
|
||||||
})
|
|
||||||
}
|
|
||||||
return pairs
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -12,6 +12,29 @@ import (
|
|||||||
"gitea.maximumdirect.net/eric/seriatim/schema"
|
"gitea.maximumdirect.net/eric/seriatim/schema"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
type trimAuditReport struct {
|
||||||
|
Operation string `json:"operation"`
|
||||||
|
InputFile string `json:"input_file"`
|
||||||
|
OutputFile string `json:"output_file"`
|
||||||
|
InputSchema string `json:"input_schema"`
|
||||||
|
OutputSchema string `json:"output_schema"`
|
||||||
|
Mode string `json:"mode"`
|
||||||
|
Selector string `json:"selector"`
|
||||||
|
SelectedIDs []int `json:"selected_ids"`
|
||||||
|
AllowEmpty bool `json:"allow_empty"`
|
||||||
|
InputSegmentCount int `json:"input_segment_count"`
|
||||||
|
RetainedSegmentCount int `json:"retained_segment_count"`
|
||||||
|
RemovedSegmentCount int `json:"removed_segment_count"`
|
||||||
|
RemovedInputIDs []int `json:"removed_input_ids"`
|
||||||
|
OldToNewIDMapping []trimIDMapping `json:"old_to_new_id_mapping"`
|
||||||
|
OverlapGroupsRecomputed bool `json:"overlap_groups_recomputed"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type trimIDMapping struct {
|
||||||
|
OldID int `json:"old_id"`
|
||||||
|
NewID int `json:"new_id"`
|
||||||
|
}
|
||||||
|
|
||||||
func TestTrimKeepModeEndToEnd(t *testing.T) {
|
func TestTrimKeepModeEndToEnd(t *testing.T) {
|
||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
input := writeTrimFullFixture(t, dir, "input.json")
|
input := writeTrimFullFixture(t, dir, "input.json")
|
||||||
|
|||||||
@@ -160,13 +160,7 @@ func (r run) coalescedSegment(id int) model.Segment {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func segmentRef(segment model.Segment) string {
|
func segmentRef(segment model.Segment) string {
|
||||||
if segment.SourceSegmentIndex != nil {
|
return model.SegmentReference(segment)
|
||||||
return fmt.Sprintf("%s#%d", segment.Source, *segment.SourceSegmentIndex)
|
|
||||||
}
|
|
||||||
if segment.SourceRef != "" {
|
|
||||||
return segment.SourceRef
|
|
||||||
}
|
|
||||||
return segment.Source
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func isSkippableInterjection(segment model.Segment) bool {
|
func isSkippableInterjection(segment model.Segment) bool {
|
||||||
|
|||||||
@@ -8,6 +8,8 @@ import (
|
|||||||
"sort"
|
"sort"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/seriatim/schema"
|
||||||
)
|
)
|
||||||
|
|
||||||
const (
|
const (
|
||||||
@@ -27,9 +29,9 @@ const (
|
|||||||
WordRunReorderWindowEnv = "SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW"
|
WordRunReorderWindowEnv = "SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW"
|
||||||
BackchannelMaxDurationEnv = "SERIATIM_BACKCHANNEL_MAX_DURATION"
|
BackchannelMaxDurationEnv = "SERIATIM_BACKCHANNEL_MAX_DURATION"
|
||||||
FillerMaxDurationEnv = "SERIATIM_FILLER_MAX_DURATION"
|
FillerMaxDurationEnv = "SERIATIM_FILLER_MAX_DURATION"
|
||||||
OutputSchemaMinimal = "seriatim-minimal"
|
OutputSchemaMinimal = schema.OutputSchemaMinimal
|
||||||
OutputSchemaIntermediate = "seriatim-intermediate"
|
OutputSchemaIntermediate = schema.OutputSchemaIntermediate
|
||||||
OutputSchemaFull = "seriatim-full"
|
OutputSchemaFull = schema.OutputSchemaFull
|
||||||
)
|
)
|
||||||
|
|
||||||
// MergeOptions captures raw CLI option values before validation.
|
// MergeOptions captures raw CLI option values before validation.
|
||||||
@@ -210,11 +212,8 @@ func NewMergeConfig(opts MergeOptions) (Config, error) {
|
|||||||
|
|
||||||
// NewTrimConfig validates raw trim options and returns normalized config.
|
// NewTrimConfig validates raw trim options and returns normalized config.
|
||||||
func NewTrimConfig(opts TrimOptions) (TrimConfig, error) {
|
func NewTrimConfig(opts TrimOptions) (TrimConfig, error) {
|
||||||
inputFile := filepath.Clean(strings.TrimSpace(opts.InputFile))
|
inputFile, err := normalizeSingleInputFile(opts.InputFile, "--input-file")
|
||||||
if strings.TrimSpace(opts.InputFile) == "" {
|
if err != nil {
|
||||||
return TrimConfig{}, errors.New("--input-file is required")
|
|
||||||
}
|
|
||||||
if err := requireFile(inputFile, "--input-file"); err != nil {
|
|
||||||
return TrimConfig{}, err
|
return TrimConfig{}, err
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -223,12 +222,9 @@ func NewTrimConfig(opts TrimOptions) (TrimConfig, error) {
|
|||||||
return TrimConfig{}, err
|
return TrimConfig{}, err
|
||||||
}
|
}
|
||||||
|
|
||||||
reportFile := ""
|
reportFile, err := normalizeOptionalOutputPath(opts.ReportFile, "--report-file")
|
||||||
if strings.TrimSpace(opts.ReportFile) != "" {
|
if err != nil {
|
||||||
reportFile, err = normalizeOutputPath(opts.ReportFile, "--report-file")
|
return TrimConfig{}, err
|
||||||
if err != nil {
|
|
||||||
return TrimConfig{}, err
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
keep := strings.TrimSpace(opts.Keep)
|
keep := strings.TrimSpace(opts.Keep)
|
||||||
@@ -267,11 +263,8 @@ func NewTrimConfig(opts TrimOptions) (TrimConfig, error) {
|
|||||||
|
|
||||||
// NewNormalizeConfig validates raw normalize options and returns normalized config.
|
// NewNormalizeConfig validates raw normalize options and returns normalized config.
|
||||||
func NewNormalizeConfig(opts NormalizeOptions) (NormalizeConfig, error) {
|
func NewNormalizeConfig(opts NormalizeOptions) (NormalizeConfig, error) {
|
||||||
inputFile := filepath.Clean(strings.TrimSpace(opts.InputFile))
|
inputFile, err := normalizeSingleInputFile(opts.InputFile, "--input-file")
|
||||||
if strings.TrimSpace(opts.InputFile) == "" {
|
if err != nil {
|
||||||
return NormalizeConfig{}, errors.New("--input-file is required")
|
|
||||||
}
|
|
||||||
if err := requireFile(inputFile, "--input-file"); err != nil {
|
|
||||||
return NormalizeConfig{}, err
|
return NormalizeConfig{}, err
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -280,12 +273,9 @@ func NewNormalizeConfig(opts NormalizeOptions) (NormalizeConfig, error) {
|
|||||||
return NormalizeConfig{}, err
|
return NormalizeConfig{}, err
|
||||||
}
|
}
|
||||||
|
|
||||||
reportFile := ""
|
reportFile, err := normalizeOptionalOutputPath(opts.ReportFile, "--report-file")
|
||||||
if strings.TrimSpace(opts.ReportFile) != "" {
|
if err != nil {
|
||||||
reportFile, err = normalizeOutputPath(opts.ReportFile, "--report-file")
|
return NormalizeConfig{}, err
|
||||||
if err != nil {
|
|
||||||
return NormalizeConfig{}, err
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
outputSchema, err := resolveOutputSchema(opts.OutputSchema)
|
outputSchema, err := resolveOutputSchema(opts.OutputSchema)
|
||||||
@@ -332,12 +322,12 @@ func parseModuleList(value string) ([]string, error) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func validateOutputSchema(value string) error {
|
func validateOutputSchema(value string) error {
|
||||||
switch value {
|
if schema.ValidOutputSchemaName(value) {
|
||||||
case OutputSchemaMinimal, OutputSchemaIntermediate, OutputSchemaFull:
|
|
||||||
return nil
|
return nil
|
||||||
default:
|
|
||||||
return fmt.Errorf("--output-schema must be one of %q, %q, or %q", OutputSchemaMinimal, OutputSchemaIntermediate, OutputSchemaFull)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
names := schema.OutputSchemaNames()
|
||||||
|
return fmt.Errorf("--output-schema must be one of %q, %q, or %q", names[0], names[1], names[2])
|
||||||
}
|
}
|
||||||
|
|
||||||
func resolveOutputSchema(value string) (string, error) {
|
func resolveOutputSchema(value string) (string, error) {
|
||||||
@@ -381,6 +371,26 @@ func normalizeInputFiles(paths []string) ([]string, error) {
|
|||||||
return normalized, nil
|
return normalized, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func normalizeSingleInputFile(path string, flag string) (string, error) {
|
||||||
|
path = strings.TrimSpace(path)
|
||||||
|
if path == "" {
|
||||||
|
return "", fmt.Errorf("%s is required", flag)
|
||||||
|
}
|
||||||
|
|
||||||
|
clean := filepath.Clean(path)
|
||||||
|
if err := requireFile(clean, flag); err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return clean, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func normalizeOptionalOutputPath(path string, flag string) (string, error) {
|
||||||
|
if strings.TrimSpace(path) == "" {
|
||||||
|
return "", nil
|
||||||
|
}
|
||||||
|
return normalizeOutputPath(path, flag)
|
||||||
|
}
|
||||||
|
|
||||||
func normalizeOutputPath(path string, flag string) (string, error) {
|
func normalizeOutputPath(path string, flag string) (string, error) {
|
||||||
path = strings.TrimSpace(path)
|
path = strings.TrimSpace(path)
|
||||||
if path == "" {
|
if path == "" {
|
||||||
|
|||||||
@@ -538,15 +538,9 @@ func TestCoalesceGapUsesValidOverride(t *testing.T) {
|
|||||||
input := writeTempFile(t, dir, "input.json")
|
input := writeTempFile(t, dir, "input.json")
|
||||||
output := filepath.Join(dir, "merged.json")
|
output := filepath.Join(dir, "merged.json")
|
||||||
|
|
||||||
cfg, err := NewMergeConfig(MergeOptions{
|
opts := validMergeOptions(input, output)
|
||||||
InputFiles: []string{input},
|
opts.CoalesceGap = "1.5"
|
||||||
OutputFile: output,
|
cfg, err := NewMergeConfig(opts)
|
||||||
InputReader: DefaultInputReader,
|
|
||||||
OutputModules: DefaultOutputModules,
|
|
||||||
PreprocessingModules: DefaultPreprocessingModules,
|
|
||||||
PostprocessingModules: DefaultPostprocessingModules,
|
|
||||||
CoalesceGap: "1.5",
|
|
||||||
})
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("config failed: %v", err)
|
t.Fatalf("config failed: %v", err)
|
||||||
}
|
}
|
||||||
@@ -560,15 +554,9 @@ func TestCoalesceGapAllowsZero(t *testing.T) {
|
|||||||
input := writeTempFile(t, dir, "input.json")
|
input := writeTempFile(t, dir, "input.json")
|
||||||
output := filepath.Join(dir, "merged.json")
|
output := filepath.Join(dir, "merged.json")
|
||||||
|
|
||||||
cfg, err := NewMergeConfig(MergeOptions{
|
opts := validMergeOptions(input, output)
|
||||||
InputFiles: []string{input},
|
opts.CoalesceGap = "0"
|
||||||
OutputFile: output,
|
cfg, err := NewMergeConfig(opts)
|
||||||
InputReader: DefaultInputReader,
|
|
||||||
OutputModules: DefaultOutputModules,
|
|
||||||
PreprocessingModules: DefaultPreprocessingModules,
|
|
||||||
PostprocessingModules: DefaultPostprocessingModules,
|
|
||||||
CoalesceGap: "0",
|
|
||||||
})
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("config failed: %v", err)
|
t.Fatalf("config failed: %v", err)
|
||||||
}
|
}
|
||||||
@@ -593,15 +581,9 @@ func TestCoalesceGapRejectsInvalidOverride(t *testing.T) {
|
|||||||
input := writeTempFile(t, dir, "input.json")
|
input := writeTempFile(t, dir, "input.json")
|
||||||
output := filepath.Join(dir, "merged.json")
|
output := filepath.Join(dir, "merged.json")
|
||||||
|
|
||||||
_, err := NewMergeConfig(MergeOptions{
|
opts := validMergeOptions(input, output)
|
||||||
InputFiles: []string{input},
|
opts.CoalesceGap = test.value
|
||||||
OutputFile: output,
|
_, err := NewMergeConfig(opts)
|
||||||
InputReader: DefaultInputReader,
|
|
||||||
OutputModules: DefaultOutputModules,
|
|
||||||
PreprocessingModules: DefaultPreprocessingModules,
|
|
||||||
PostprocessingModules: DefaultPostprocessingModules,
|
|
||||||
CoalesceGap: test.value,
|
|
||||||
})
|
|
||||||
if err == nil {
|
if err == nil {
|
||||||
t.Fatal("expected error")
|
t.Fatal("expected error")
|
||||||
}
|
}
|
||||||
@@ -639,20 +621,16 @@ func TestNewTrimConfigRequiresExactlyOneSelectorFlag(t *testing.T) {
|
|||||||
input := writeTempFile(t, dir, "input.json")
|
input := writeTempFile(t, dir, "input.json")
|
||||||
output := filepath.Join(dir, "trimmed.json")
|
output := filepath.Join(dir, "trimmed.json")
|
||||||
|
|
||||||
_, err := NewTrimConfig(TrimOptions{
|
opts := validTrimOptions(input, output)
|
||||||
InputFile: input,
|
opts.Keep = ""
|
||||||
OutputFile: output,
|
_, err := NewTrimConfig(opts)
|
||||||
})
|
|
||||||
if err == nil || !strings.Contains(err.Error(), "exactly one of --keep or --remove is required") {
|
if err == nil || !strings.Contains(err.Error(), "exactly one of --keep or --remove is required") {
|
||||||
t.Fatalf("expected missing selector error, got %v", err)
|
t.Fatalf("expected missing selector error, got %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
_, err = NewTrimConfig(TrimOptions{
|
opts = validTrimOptions(input, output)
|
||||||
InputFile: input,
|
opts.Remove = "2"
|
||||||
OutputFile: output,
|
_, err = NewTrimConfig(opts)
|
||||||
Keep: "1",
|
|
||||||
Remove: "2",
|
|
||||||
})
|
|
||||||
if err == nil || !strings.Contains(err.Error(), "mutually exclusive") {
|
if err == nil || !strings.Contains(err.Error(), "mutually exclusive") {
|
||||||
t.Fatalf("expected mutually exclusive selector error, got %v", err)
|
t.Fatalf("expected mutually exclusive selector error, got %v", err)
|
||||||
}
|
}
|
||||||
@@ -664,14 +642,13 @@ func TestNewTrimConfigAcceptsOutputSchemaOverride(t *testing.T) {
|
|||||||
output := filepath.Join(dir, "trimmed.json")
|
output := filepath.Join(dir, "trimmed.json")
|
||||||
reportPath := filepath.Join(dir, "report.json")
|
reportPath := filepath.Join(dir, "report.json")
|
||||||
|
|
||||||
cfg, err := NewTrimConfig(TrimOptions{
|
opts := validTrimOptions(input, output)
|
||||||
InputFile: input,
|
opts.Keep = ""
|
||||||
OutputFile: output,
|
opts.Remove = "3-5"
|
||||||
ReportFile: reportPath,
|
opts.ReportFile = reportPath
|
||||||
Remove: "3-5",
|
opts.OutputSchema = OutputSchemaMinimal
|
||||||
OutputSchema: OutputSchemaMinimal,
|
opts.AllowEmpty = true
|
||||||
AllowEmpty: true,
|
cfg, err := NewTrimConfig(opts)
|
||||||
})
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("config failed: %v", err)
|
t.Fatalf("config failed: %v", err)
|
||||||
}
|
}
|
||||||
@@ -692,17 +669,30 @@ func TestNewTrimConfigAcceptsOutputSchemaOverride(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestNewTrimConfigTreatsWhitespaceReportFileAsOmitted(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
input := writeTempFile(t, dir, "input.json")
|
||||||
|
output := filepath.Join(dir, "trimmed.json")
|
||||||
|
|
||||||
|
opts := validTrimOptions(input, output)
|
||||||
|
opts.ReportFile = " \t "
|
||||||
|
cfg, err := NewTrimConfig(opts)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("config failed: %v", err)
|
||||||
|
}
|
||||||
|
if cfg.ReportFile != "" {
|
||||||
|
t.Fatalf("report file = %q, want empty", cfg.ReportFile)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestNewTrimConfigRejectsInvalidOutputSchemaOverride(t *testing.T) {
|
func TestNewTrimConfigRejectsInvalidOutputSchemaOverride(t *testing.T) {
|
||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
input := writeTempFile(t, dir, "input.json")
|
input := writeTempFile(t, dir, "input.json")
|
||||||
output := filepath.Join(dir, "trimmed.json")
|
output := filepath.Join(dir, "trimmed.json")
|
||||||
|
|
||||||
_, err := NewTrimConfig(TrimOptions{
|
opts := validTrimOptions(input, output)
|
||||||
InputFile: input,
|
opts.OutputSchema = "compact"
|
||||||
OutputFile: output,
|
_, err := NewTrimConfig(opts)
|
||||||
Keep: "1",
|
|
||||||
OutputSchema: "compact",
|
|
||||||
})
|
|
||||||
if err == nil {
|
if err == nil {
|
||||||
t.Fatal("expected output schema validation error")
|
t.Fatal("expected output schema validation error")
|
||||||
}
|
}
|
||||||
@@ -731,10 +721,8 @@ func TestNewNormalizeConfigRequiresOutputFile(t *testing.T) {
|
|||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
input := writeTempFile(t, dir, "input.json")
|
input := writeTempFile(t, dir, "input.json")
|
||||||
|
|
||||||
_, err := NewNormalizeConfig(NormalizeOptions{
|
opts := validNormalizeOptions(input, "")
|
||||||
InputFile: input,
|
_, err := NewNormalizeConfig(opts)
|
||||||
OutputModules: DefaultOutputModules,
|
|
||||||
})
|
|
||||||
if err == nil {
|
if err == nil {
|
||||||
t.Fatal("expected output-file required error")
|
t.Fatal("expected output-file required error")
|
||||||
}
|
}
|
||||||
@@ -749,11 +737,8 @@ func TestNewNormalizeConfigResolvesOutputSchemaDefaultAndEnv(t *testing.T) {
|
|||||||
output := filepath.Join(dir, "normalized.json")
|
output := filepath.Join(dir, "normalized.json")
|
||||||
|
|
||||||
t.Setenv(OutputSchemaEnv, "")
|
t.Setenv(OutputSchemaEnv, "")
|
||||||
cfg, err := NewNormalizeConfig(NormalizeOptions{
|
opts := validNormalizeOptions(input, output)
|
||||||
InputFile: input,
|
cfg, err := NewNormalizeConfig(opts)
|
||||||
OutputFile: output,
|
|
||||||
OutputModules: DefaultOutputModules,
|
|
||||||
})
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("config failed: %v", err)
|
t.Fatalf("config failed: %v", err)
|
||||||
}
|
}
|
||||||
@@ -762,11 +747,7 @@ func TestNewNormalizeConfigResolvesOutputSchemaDefaultAndEnv(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
t.Setenv(OutputSchemaEnv, OutputSchemaMinimal)
|
t.Setenv(OutputSchemaEnv, OutputSchemaMinimal)
|
||||||
cfg, err = NewNormalizeConfig(NormalizeOptions{
|
cfg, err = NewNormalizeConfig(opts)
|
||||||
InputFile: input,
|
|
||||||
OutputFile: output,
|
|
||||||
OutputModules: DefaultOutputModules,
|
|
||||||
})
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("config failed: %v", err)
|
t.Fatalf("config failed: %v", err)
|
||||||
}
|
}
|
||||||
@@ -780,12 +761,9 @@ func TestNewNormalizeConfigRejectsInvalidOutputSchema(t *testing.T) {
|
|||||||
input := writeTempFile(t, dir, "input.json")
|
input := writeTempFile(t, dir, "input.json")
|
||||||
output := filepath.Join(dir, "normalized.json")
|
output := filepath.Join(dir, "normalized.json")
|
||||||
|
|
||||||
_, err := NewNormalizeConfig(NormalizeOptions{
|
opts := validNormalizeOptions(input, output)
|
||||||
InputFile: input,
|
opts.OutputSchema = "compact"
|
||||||
OutputFile: output,
|
_, err := NewNormalizeConfig(opts)
|
||||||
OutputSchema: "compact",
|
|
||||||
OutputModules: DefaultOutputModules,
|
|
||||||
})
|
|
||||||
if err == nil {
|
if err == nil {
|
||||||
t.Fatal("expected output schema error")
|
t.Fatal("expected output schema error")
|
||||||
}
|
}
|
||||||
@@ -799,11 +777,9 @@ func TestNewNormalizeConfigRejectsUnknownOutputModule(t *testing.T) {
|
|||||||
input := writeTempFile(t, dir, "input.json")
|
input := writeTempFile(t, dir, "input.json")
|
||||||
output := filepath.Join(dir, "normalized.json")
|
output := filepath.Join(dir, "normalized.json")
|
||||||
|
|
||||||
_, err := NewNormalizeConfig(NormalizeOptions{
|
opts := validNormalizeOptions(input, output)
|
||||||
InputFile: input,
|
opts.OutputModules = "json,yaml"
|
||||||
OutputFile: output,
|
_, err := NewNormalizeConfig(opts)
|
||||||
OutputModules: "json,yaml",
|
|
||||||
})
|
|
||||||
if err == nil {
|
if err == nil {
|
||||||
t.Fatal("expected output module error")
|
t.Fatal("expected output module error")
|
||||||
}
|
}
|
||||||
@@ -812,6 +788,22 @@ func TestNewNormalizeConfigRejectsUnknownOutputModule(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestNewNormalizeConfigTreatsWhitespaceReportFileAsOmitted(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
input := writeTempFile(t, dir, "input.json")
|
||||||
|
output := filepath.Join(dir, "normalized.json")
|
||||||
|
|
||||||
|
opts := validNormalizeOptions(input, output)
|
||||||
|
opts.ReportFile = "\n\t "
|
||||||
|
cfg, err := NewNormalizeConfig(opts)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("config failed: %v", err)
|
||||||
|
}
|
||||||
|
if cfg.ReportFile != "" {
|
||||||
|
t.Fatalf("report file = %q, want empty", cfg.ReportFile)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func assertPositiveFloatEnvValidation(t *testing.T, envName string) {
|
func assertPositiveFloatEnvValidation(t *testing.T, envName string) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
@@ -832,14 +824,7 @@ func assertPositiveFloatEnvValidation(t *testing.T, envName string) {
|
|||||||
input := writeTempFile(t, dir, "input.json")
|
input := writeTempFile(t, dir, "input.json")
|
||||||
output := filepath.Join(dir, "merged.json")
|
output := filepath.Join(dir, "merged.json")
|
||||||
|
|
||||||
_, err := NewMergeConfig(MergeOptions{
|
_, err := NewMergeConfig(validMergeOptions(input, output))
|
||||||
InputFiles: []string{input},
|
|
||||||
OutputFile: output,
|
|
||||||
InputReader: DefaultInputReader,
|
|
||||||
OutputModules: DefaultOutputModules,
|
|
||||||
PreprocessingModules: DefaultPreprocessingModules,
|
|
||||||
PostprocessingModules: DefaultPostprocessingModules,
|
|
||||||
})
|
|
||||||
if err == nil {
|
if err == nil {
|
||||||
t.Fatal("expected error")
|
t.Fatal("expected error")
|
||||||
}
|
}
|
||||||
@@ -850,6 +835,33 @@ func assertPositiveFloatEnvValidation(t *testing.T, envName string) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func validMergeOptions(inputFile string, outputFile string) MergeOptions {
|
||||||
|
return MergeOptions{
|
||||||
|
InputFiles: []string{inputFile},
|
||||||
|
OutputFile: outputFile,
|
||||||
|
InputReader: DefaultInputReader,
|
||||||
|
OutputModules: DefaultOutputModules,
|
||||||
|
PreprocessingModules: DefaultPreprocessingModules,
|
||||||
|
PostprocessingModules: DefaultPostprocessingModules,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func validTrimOptions(inputFile string, outputFile string) TrimOptions {
|
||||||
|
return TrimOptions{
|
||||||
|
InputFile: inputFile,
|
||||||
|
OutputFile: outputFile,
|
||||||
|
Keep: "1",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func validNormalizeOptions(inputFile string, outputFile string) NormalizeOptions {
|
||||||
|
return NormalizeOptions{
|
||||||
|
InputFile: inputFile,
|
||||||
|
OutputFile: outputFile,
|
||||||
|
OutputModules: DefaultOutputModules,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func writeTempFile(t *testing.T, dir string, name string) string {
|
func writeTempFile(t *testing.T, dir string, name string) string {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
|
|||||||
28
internal/jsonfile/jsonfile.go
Normal file
28
internal/jsonfile/jsonfile.go
Normal file
@@ -0,0 +1,28 @@
|
|||||||
|
package jsonfile
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Write creates or truncates path and writes deterministic indented JSON.
|
||||||
|
func Write(path string, value any) (err error) {
|
||||||
|
file, err := os.Create(path)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("create %q: %w", path, err)
|
||||||
|
}
|
||||||
|
defer func() {
|
||||||
|
closeErr := file.Close()
|
||||||
|
if err == nil && closeErr != nil {
|
||||||
|
err = fmt.Errorf("close %q: %w", path, closeErr)
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
|
||||||
|
encoder := json.NewEncoder(file)
|
||||||
|
encoder.SetIndent("", " ")
|
||||||
|
if err := encoder.Encode(value); err != nil {
|
||||||
|
return fmt.Errorf("encode %q: %w", path, err)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
69
internal/jsonfile/jsonfile_test.go
Normal file
69
internal/jsonfile/jsonfile_test.go
Normal file
@@ -0,0 +1,69 @@
|
|||||||
|
package jsonfile
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/json"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestWriteFormatsWithTwoSpaceIndentAndTrailingNewline(t *testing.T) {
|
||||||
|
type payload struct {
|
||||||
|
Name string `json:"name"`
|
||||||
|
Items []int `json:"items"`
|
||||||
|
}
|
||||||
|
|
||||||
|
path := filepath.Join(t.TempDir(), "out.json")
|
||||||
|
value := payload{
|
||||||
|
Name: "alpha",
|
||||||
|
Items: []int{1, 2},
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := Write(path, value); err != nil {
|
||||||
|
t.Fatalf("write failed: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
data, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read output: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
got := string(data)
|
||||||
|
want := "{\n \"name\": \"alpha\",\n \"items\": [\n 1,\n 2\n ]\n}\n"
|
||||||
|
if got != want {
|
||||||
|
t.Fatalf("formatted JSON mismatch\nwant:\n%s\ngot:\n%s", want, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestWriteProducesValidJSON(t *testing.T) {
|
||||||
|
path := filepath.Join(t.TempDir(), "out.json")
|
||||||
|
|
||||||
|
value := map[string]any{
|
||||||
|
"application": "seriatim",
|
||||||
|
"segments": []map[string]any{
|
||||||
|
{
|
||||||
|
"id": 1,
|
||||||
|
"speaker": "A",
|
||||||
|
"text": "hello",
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := Write(path, value); err != nil {
|
||||||
|
t.Fatalf("write failed: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
data, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read output: %v", err)
|
||||||
|
}
|
||||||
|
if !strings.HasSuffix(string(data), "\n") {
|
||||||
|
t.Fatalf("output missing trailing newline: %q", string(data))
|
||||||
|
}
|
||||||
|
|
||||||
|
var decoded map[string]any
|
||||||
|
if err := json.Unmarshal(data, &decoded); err != nil {
|
||||||
|
t.Fatalf("output is not valid JSON: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,5 +1,7 @@
|
|||||||
package model
|
package model
|
||||||
|
|
||||||
|
import "fmt"
|
||||||
|
|
||||||
// RawTranscript is a loaded input document before canonical normalization.
|
// RawTranscript is a loaded input document before canonical normalization.
|
||||||
type RawTranscript struct {
|
type RawTranscript struct {
|
||||||
Source string `json:"source"`
|
Source string `json:"source"`
|
||||||
@@ -61,6 +63,17 @@ type Segment struct {
|
|||||||
OverlapGroupID int `json:"overlap_group_id,omitempty"`
|
OverlapGroupID int `json:"overlap_group_id,omitempty"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// SegmentReference returns the best available external reference for a segment.
|
||||||
|
func SegmentReference(segment Segment) string {
|
||||||
|
if segment.Source != "" && segment.SourceSegmentIndex != nil {
|
||||||
|
return fmt.Sprintf("%s#%d", segment.Source, *segment.SourceSegmentIndex)
|
||||||
|
}
|
||||||
|
if segment.SourceRef != "" {
|
||||||
|
return segment.SourceRef
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
// Word preserves optional word-level timing data.
|
// Word preserves optional word-level timing data.
|
||||||
type Word struct {
|
type Word struct {
|
||||||
Text string `json:"text"`
|
Text string `json:"text"`
|
||||||
|
|||||||
41
internal/model/model_test.go
Normal file
41
internal/model/model_test.go
Normal file
@@ -0,0 +1,41 @@
|
|||||||
|
package model
|
||||||
|
|
||||||
|
import "testing"
|
||||||
|
|
||||||
|
func TestSegmentReferenceUsesSourceAndIndexWhenAvailable(t *testing.T) {
|
||||||
|
index := 3
|
||||||
|
segment := Segment{
|
||||||
|
Source: "input.json",
|
||||||
|
SourceSegmentIndex: &index,
|
||||||
|
SourceRef: "word-run:1:2:3",
|
||||||
|
}
|
||||||
|
|
||||||
|
got := SegmentReference(segment)
|
||||||
|
want := "input.json#3"
|
||||||
|
if got != want {
|
||||||
|
t.Fatalf("reference = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSegmentReferenceFallsBackToSourceRef(t *testing.T) {
|
||||||
|
segment := Segment{
|
||||||
|
Source: "input.json",
|
||||||
|
SourceRef: "coalesce:2",
|
||||||
|
}
|
||||||
|
|
||||||
|
got := SegmentReference(segment)
|
||||||
|
want := "coalesce:2"
|
||||||
|
if got != want {
|
||||||
|
t.Fatalf("reference = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSegmentReferenceReturnsEmptyWhenNoReferenceFieldsPresent(t *testing.T) {
|
||||||
|
segment := Segment{
|
||||||
|
Source: "input.json",
|
||||||
|
}
|
||||||
|
|
||||||
|
if got := SegmentReference(segment); got != "" {
|
||||||
|
t.Fatalf("reference = %q, want empty", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -4,12 +4,12 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
"fmt"
|
"fmt"
|
||||||
"os"
|
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/seriatim/internal/artifact"
|
"gitea.maximumdirect.net/eric/seriatim/internal/artifact"
|
||||||
"gitea.maximumdirect.net/eric/seriatim/internal/buildinfo"
|
"gitea.maximumdirect.net/eric/seriatim/internal/buildinfo"
|
||||||
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||||
|
"gitea.maximumdirect.net/eric/seriatim/internal/jsonfile"
|
||||||
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -47,7 +47,7 @@ func Run(ctx context.Context, cfg config.NormalizeConfig) error {
|
|||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
if err := writeOutputJSON(cfg.OutputFile, built.Output); err != nil {
|
if err := jsonfile.Write(cfg.OutputFile, built.Output); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -118,18 +118,3 @@ func Run(ctx context.Context, cfg config.NormalizeConfig) error {
|
|||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func writeOutputJSON(path string, value any) error {
|
|
||||||
file, err := os.Create(path)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
defer file.Close()
|
|
||||||
|
|
||||||
encoder := json.NewEncoder(file)
|
|
||||||
encoder.SetIndent("", " ")
|
|
||||||
if err := encoder.Encode(value); err != nil {
|
|
||||||
return fmt.Errorf("encode normalize output JSON: %w", err)
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -1,7 +1,6 @@
|
|||||||
package overlap
|
package overlap
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
|
||||||
"sort"
|
"sort"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/seriatim/internal/model"
|
"gitea.maximumdirect.net/eric/seriatim/internal/model"
|
||||||
@@ -121,13 +120,7 @@ func distinctSpeakers(segments []model.Segment, indices []int) []string {
|
|||||||
|
|
||||||
// SegmentRef returns the stable overlap reference for a segment.
|
// SegmentRef returns the stable overlap reference for a segment.
|
||||||
func SegmentRef(segment model.Segment) string {
|
func SegmentRef(segment model.Segment) string {
|
||||||
if segment.SourceSegmentIndex != nil {
|
return model.SegmentReference(segment)
|
||||||
return fmt.Sprintf("%s#%d", segment.Source, *segment.SourceSegmentIndex)
|
|
||||||
}
|
|
||||||
if segment.SourceRef != "" {
|
|
||||||
return segment.SourceRef
|
|
||||||
}
|
|
||||||
return segment.Source
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func clearExisting(in *model.MergedTranscript) {
|
func clearExisting(in *model.MergedTranscript) {
|
||||||
|
|||||||
@@ -1,9 +1,6 @@
|
|||||||
package report
|
package report
|
||||||
|
|
||||||
import (
|
import "gitea.maximumdirect.net/eric/seriatim/internal/jsonfile"
|
||||||
"encoding/json"
|
|
||||||
"os"
|
|
||||||
)
|
|
||||||
|
|
||||||
// Severity classifies report events.
|
// Severity classifies report events.
|
||||||
type Severity string
|
type Severity string
|
||||||
@@ -62,13 +59,5 @@ func Warning(stage string, module string, message string) Event {
|
|||||||
|
|
||||||
// WriteJSON writes a deterministic JSON report.
|
// WriteJSON writes a deterministic JSON report.
|
||||||
func WriteJSON(path string, rpt Report) error {
|
func WriteJSON(path string, rpt Report) error {
|
||||||
file, err := os.Create(path)
|
return jsonfile.Write(path, rpt)
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
defer file.Close()
|
|
||||||
|
|
||||||
enc := json.NewEncoder(file)
|
|
||||||
enc.SetIndent("", " ")
|
|
||||||
return enc.Encode(rpt)
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -44,54 +44,29 @@ type MinimalResult struct {
|
|||||||
RemovedIDs []int
|
RemovedIDs []int
|
||||||
}
|
}
|
||||||
|
|
||||||
|
type projection struct {
|
||||||
|
retainedIndexes []int
|
||||||
|
oldToNewID map[int]int
|
||||||
|
removedIDs []int
|
||||||
|
}
|
||||||
|
|
||||||
// Apply trims a full seriatim output transcript by segment ID.
|
// Apply trims a full seriatim output transcript by segment ID.
|
||||||
func Apply(input schema.Transcript, opts Options) (Result, error) {
|
func Apply(input schema.Transcript, opts Options) (Result, error) {
|
||||||
if err := validateMode(opts.Mode); err != nil {
|
|
||||||
return Result{}, err
|
|
||||||
}
|
|
||||||
|
|
||||||
selected := opts.Selector.IDs()
|
|
||||||
if len(selected) == 0 {
|
|
||||||
return Result{}, fmt.Errorf("selector cannot be empty")
|
|
||||||
}
|
|
||||||
|
|
||||||
inputIDs := make([]int, len(input.Segments))
|
inputIDs := make([]int, len(input.Segments))
|
||||||
for index, segment := range input.Segments {
|
for index, segment := range input.Segments {
|
||||||
inputIDs[index] = segment.ID
|
inputIDs[index] = segment.ID
|
||||||
}
|
}
|
||||||
|
proj, err := projectSegmentIDs(inputIDs, opts)
|
||||||
idIndex, err := validateInputIDs(inputIDs)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return Result{}, err
|
return Result{}, err
|
||||||
}
|
}
|
||||||
|
|
||||||
if err := validateSelectedIDsExist(selected, idIndex); err != nil {
|
kept := make([]schema.Segment, len(proj.retainedIndexes))
|
||||||
return Result{}, err
|
for outputIndex, inputIndex := range proj.retainedIndexes {
|
||||||
}
|
rewritten := copySegment(input.Segments[inputIndex])
|
||||||
|
rewritten.ID = outputIndex + 1
|
||||||
kept := make([]schema.Segment, 0, len(input.Segments))
|
|
||||||
removed := make([]int, 0, len(input.Segments))
|
|
||||||
oldToNew := make(map[int]int, len(input.Segments))
|
|
||||||
for _, segment := range input.Segments {
|
|
||||||
keep := opts.Mode == ModeKeep && opts.Selector.Contains(segment.ID)
|
|
||||||
if opts.Mode == ModeRemove {
|
|
||||||
keep = !opts.Selector.Contains(segment.ID)
|
|
||||||
}
|
|
||||||
|
|
||||||
if !keep {
|
|
||||||
removed = append(removed, segment.ID)
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
rewritten := copySegment(segment)
|
|
||||||
rewritten.ID = len(kept) + 1
|
|
||||||
rewritten.OverlapGroupID = 0
|
rewritten.OverlapGroupID = 0
|
||||||
kept = append(kept, rewritten)
|
kept[outputIndex] = rewritten
|
||||||
oldToNew[segment.ID] = rewritten.ID
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(kept) == 0 && !opts.AllowEmpty {
|
|
||||||
return Result{}, fmt.Errorf("trim operation produced an empty transcript; set AllowEmpty to true to permit this")
|
|
||||||
}
|
}
|
||||||
|
|
||||||
kept, groups := recomputeOverlapGroups(kept)
|
kept, groups := recomputeOverlapGroups(kept)
|
||||||
@@ -104,62 +79,35 @@ func Apply(input schema.Transcript, opts Options) (Result, error) {
|
|||||||
out.OverlapGroups = groups
|
out.OverlapGroups = groups
|
||||||
return Result{
|
return Result{
|
||||||
Transcript: out,
|
Transcript: out,
|
||||||
OldToNewID: oldToNew,
|
OldToNewID: proj.oldToNewID,
|
||||||
RemovedIDs: removed,
|
RemovedIDs: proj.removedIDs,
|
||||||
}, nil
|
}, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// ApplyIntermediate trims an intermediate seriatim output transcript by
|
// ApplyIntermediate trims an intermediate seriatim output transcript by
|
||||||
// segment ID.
|
// segment ID.
|
||||||
func ApplyIntermediate(input schema.IntermediateTranscript, opts Options) (IntermediateResult, error) {
|
func ApplyIntermediate(input schema.IntermediateTranscript, opts Options) (IntermediateResult, error) {
|
||||||
if err := validateMode(opts.Mode); err != nil {
|
|
||||||
return IntermediateResult{}, err
|
|
||||||
}
|
|
||||||
|
|
||||||
selected := opts.Selector.IDs()
|
|
||||||
if len(selected) == 0 {
|
|
||||||
return IntermediateResult{}, fmt.Errorf("selector cannot be empty")
|
|
||||||
}
|
|
||||||
|
|
||||||
inputIDs := make([]int, len(input.Segments))
|
inputIDs := make([]int, len(input.Segments))
|
||||||
for index, segment := range input.Segments {
|
for index, segment := range input.Segments {
|
||||||
inputIDs[index] = segment.ID
|
inputIDs[index] = segment.ID
|
||||||
}
|
}
|
||||||
idIndex, err := validateInputIDs(inputIDs)
|
proj, err := projectSegmentIDs(inputIDs, opts)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return IntermediateResult{}, err
|
return IntermediateResult{}, err
|
||||||
}
|
}
|
||||||
if err := validateSelectedIDsExist(selected, idIndex); err != nil {
|
|
||||||
return IntermediateResult{}, err
|
|
||||||
}
|
|
||||||
|
|
||||||
kept := make([]schema.IntermediateSegment, 0, len(input.Segments))
|
|
||||||
removed := make([]int, 0, len(input.Segments))
|
|
||||||
oldToNew := make(map[int]int, len(input.Segments))
|
|
||||||
for _, segment := range input.Segments {
|
|
||||||
keep := opts.Mode == ModeKeep && opts.Selector.Contains(segment.ID)
|
|
||||||
if opts.Mode == ModeRemove {
|
|
||||||
keep = !opts.Selector.Contains(segment.ID)
|
|
||||||
}
|
|
||||||
if !keep {
|
|
||||||
removed = append(removed, segment.ID)
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
|
kept := make([]schema.IntermediateSegment, len(proj.retainedIndexes))
|
||||||
|
for outputIndex, inputIndex := range proj.retainedIndexes {
|
||||||
|
segment := input.Segments[inputIndex]
|
||||||
rewritten := schema.IntermediateSegment{
|
rewritten := schema.IntermediateSegment{
|
||||||
ID: len(kept) + 1,
|
ID: outputIndex + 1,
|
||||||
Start: segment.Start,
|
Start: segment.Start,
|
||||||
End: segment.End,
|
End: segment.End,
|
||||||
Speaker: segment.Speaker,
|
Speaker: segment.Speaker,
|
||||||
Text: segment.Text,
|
Text: segment.Text,
|
||||||
Categories: append([]string(nil), segment.Categories...),
|
Categories: append([]string(nil), segment.Categories...),
|
||||||
}
|
}
|
||||||
kept = append(kept, rewritten)
|
kept[outputIndex] = rewritten
|
||||||
oldToNew[segment.ID] = rewritten.ID
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(kept) == 0 && !opts.AllowEmpty {
|
|
||||||
return IntermediateResult{}, fmt.Errorf("trim operation produced an empty transcript; set AllowEmpty to true to permit this")
|
|
||||||
}
|
}
|
||||||
|
|
||||||
return IntermediateResult{
|
return IntermediateResult{
|
||||||
@@ -171,60 +119,33 @@ func ApplyIntermediate(input schema.IntermediateTranscript, opts Options) (Inter
|
|||||||
},
|
},
|
||||||
Segments: kept,
|
Segments: kept,
|
||||||
},
|
},
|
||||||
OldToNewID: oldToNew,
|
OldToNewID: proj.oldToNewID,
|
||||||
RemovedIDs: removed,
|
RemovedIDs: proj.removedIDs,
|
||||||
}, nil
|
}, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// ApplyMinimal trims a minimal seriatim output transcript by segment ID.
|
// ApplyMinimal trims a minimal seriatim output transcript by segment ID.
|
||||||
func ApplyMinimal(input schema.MinimalTranscript, opts Options) (MinimalResult, error) {
|
func ApplyMinimal(input schema.MinimalTranscript, opts Options) (MinimalResult, error) {
|
||||||
if err := validateMode(opts.Mode); err != nil {
|
|
||||||
return MinimalResult{}, err
|
|
||||||
}
|
|
||||||
|
|
||||||
selected := opts.Selector.IDs()
|
|
||||||
if len(selected) == 0 {
|
|
||||||
return MinimalResult{}, fmt.Errorf("selector cannot be empty")
|
|
||||||
}
|
|
||||||
|
|
||||||
inputIDs := make([]int, len(input.Segments))
|
inputIDs := make([]int, len(input.Segments))
|
||||||
for index, segment := range input.Segments {
|
for index, segment := range input.Segments {
|
||||||
inputIDs[index] = segment.ID
|
inputIDs[index] = segment.ID
|
||||||
}
|
}
|
||||||
idIndex, err := validateInputIDs(inputIDs)
|
proj, err := projectSegmentIDs(inputIDs, opts)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return MinimalResult{}, err
|
return MinimalResult{}, err
|
||||||
}
|
}
|
||||||
if err := validateSelectedIDsExist(selected, idIndex); err != nil {
|
|
||||||
return MinimalResult{}, err
|
|
||||||
}
|
|
||||||
|
|
||||||
kept := make([]schema.MinimalSegment, 0, len(input.Segments))
|
|
||||||
removed := make([]int, 0, len(input.Segments))
|
|
||||||
oldToNew := make(map[int]int, len(input.Segments))
|
|
||||||
for _, segment := range input.Segments {
|
|
||||||
keep := opts.Mode == ModeKeep && opts.Selector.Contains(segment.ID)
|
|
||||||
if opts.Mode == ModeRemove {
|
|
||||||
keep = !opts.Selector.Contains(segment.ID)
|
|
||||||
}
|
|
||||||
if !keep {
|
|
||||||
removed = append(removed, segment.ID)
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
|
kept := make([]schema.MinimalSegment, len(proj.retainedIndexes))
|
||||||
|
for outputIndex, inputIndex := range proj.retainedIndexes {
|
||||||
|
segment := input.Segments[inputIndex]
|
||||||
rewritten := schema.MinimalSegment{
|
rewritten := schema.MinimalSegment{
|
||||||
ID: len(kept) + 1,
|
ID: outputIndex + 1,
|
||||||
Start: segment.Start,
|
Start: segment.Start,
|
||||||
End: segment.End,
|
End: segment.End,
|
||||||
Speaker: segment.Speaker,
|
Speaker: segment.Speaker,
|
||||||
Text: segment.Text,
|
Text: segment.Text,
|
||||||
}
|
}
|
||||||
kept = append(kept, rewritten)
|
kept[outputIndex] = rewritten
|
||||||
oldToNew[segment.ID] = rewritten.ID
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(kept) == 0 && !opts.AllowEmpty {
|
|
||||||
return MinimalResult{}, fmt.Errorf("trim operation produced an empty transcript; set AllowEmpty to true to permit this")
|
|
||||||
}
|
}
|
||||||
|
|
||||||
return MinimalResult{
|
return MinimalResult{
|
||||||
@@ -236,11 +157,53 @@ func ApplyMinimal(input schema.MinimalTranscript, opts Options) (MinimalResult,
|
|||||||
},
|
},
|
||||||
Segments: kept,
|
Segments: kept,
|
||||||
},
|
},
|
||||||
OldToNewID: oldToNew,
|
OldToNewID: proj.oldToNewID,
|
||||||
RemovedIDs: removed,
|
RemovedIDs: proj.removedIDs,
|
||||||
}, nil
|
}, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func projectSegmentIDs(ids []int, opts Options) (projection, error) {
|
||||||
|
if err := validateMode(opts.Mode); err != nil {
|
||||||
|
return projection{}, err
|
||||||
|
}
|
||||||
|
|
||||||
|
selected := opts.Selector.IDs()
|
||||||
|
if len(selected) == 0 {
|
||||||
|
return projection{}, fmt.Errorf("selector cannot be empty")
|
||||||
|
}
|
||||||
|
|
||||||
|
idIndex, err := validateInputIDs(ids)
|
||||||
|
if err != nil {
|
||||||
|
return projection{}, err
|
||||||
|
}
|
||||||
|
if err := validateSelectedIDsExist(selected, idIndex); err != nil {
|
||||||
|
return projection{}, err
|
||||||
|
}
|
||||||
|
|
||||||
|
result := projection{
|
||||||
|
retainedIndexes: make([]int, 0, len(ids)),
|
||||||
|
oldToNewID: make(map[int]int, len(ids)),
|
||||||
|
removedIDs: make([]int, 0, len(ids)),
|
||||||
|
}
|
||||||
|
for index, id := range ids {
|
||||||
|
keep := opts.Mode == ModeKeep && opts.Selector.Contains(id)
|
||||||
|
if opts.Mode == ModeRemove {
|
||||||
|
keep = !opts.Selector.Contains(id)
|
||||||
|
}
|
||||||
|
if !keep {
|
||||||
|
result.removedIDs = append(result.removedIDs, id)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
result.retainedIndexes = append(result.retainedIndexes, index)
|
||||||
|
result.oldToNewID[id] = len(result.retainedIndexes)
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(result.retainedIndexes) == 0 && !opts.AllowEmpty {
|
||||||
|
return projection{}, fmt.Errorf("trim operation produced an empty transcript; set AllowEmpty to true to permit this")
|
||||||
|
}
|
||||||
|
return result, nil
|
||||||
|
}
|
||||||
|
|
||||||
func validateMode(mode Mode) error {
|
func validateMode(mode Mode) error {
|
||||||
switch mode {
|
switch mode {
|
||||||
case ModeKeep, ModeRemove:
|
case ModeKeep, ModeRemove:
|
||||||
|
|||||||
@@ -399,6 +399,106 @@ func TestApplyMinimalDoesNotIncludeOverlapGroups(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestApplySelectorPolicyIsSharedAcrossSchemas(t *testing.T) {
|
||||||
|
type testCase struct {
|
||||||
|
name string
|
||||||
|
opts Options
|
||||||
|
wantTexts []string
|
||||||
|
wantOldToNew map[int]int
|
||||||
|
wantRemoved []int
|
||||||
|
wantSegmentCount int
|
||||||
|
wantErrorSubstring string
|
||||||
|
}
|
||||||
|
|
||||||
|
cases := []testCase{
|
||||||
|
{
|
||||||
|
name: "keep preserves input order regardless of selector order",
|
||||||
|
opts: Options{Mode: ModeKeep, Selector: mustParseSelector(t, "4,1,3")},
|
||||||
|
wantTexts: []string{"alpha", "gamma", "delta"},
|
||||||
|
wantOldToNew: map[int]int{1: 1, 3: 2, 4: 3},
|
||||||
|
wantRemoved: []int{2},
|
||||||
|
wantSegmentCount: 3,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "remove reports deterministic renumbering metadata",
|
||||||
|
opts: Options{Mode: ModeRemove, Selector: mustParseSelector(t, "2,4")},
|
||||||
|
wantTexts: []string{"alpha", "gamma"},
|
||||||
|
wantOldToNew: map[int]int{1: 1, 3: 2},
|
||||||
|
wantRemoved: []int{2, 4},
|
||||||
|
wantSegmentCount: 2,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "missing selected id returns error",
|
||||||
|
opts: Options{Mode: ModeKeep, Selector: mustParseSelector(t, "9")},
|
||||||
|
wantErrorSubstring: "does not exist",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "empty selector returns error",
|
||||||
|
opts: Options{Mode: ModeKeep, Selector: Selector{}},
|
||||||
|
wantErrorSubstring: "selector cannot be empty",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "invalid mode returns error",
|
||||||
|
opts: Options{Mode: Mode("bad"), Selector: mustParseSelector(t, "1")},
|
||||||
|
wantErrorSubstring: `invalid trim mode "bad"`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "empty output blocked when allow empty is false",
|
||||||
|
opts: Options{Mode: ModeRemove, Selector: mustParseSelector(t, "1-4")},
|
||||||
|
wantErrorSubstring: "empty transcript",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "empty output allowed when allow empty is true",
|
||||||
|
opts: Options{Mode: ModeRemove, Selector: mustParseSelector(t, "1-4"), AllowEmpty: true},
|
||||||
|
wantTexts: []string{},
|
||||||
|
wantOldToNew: map[int]int{},
|
||||||
|
wantRemoved: []int{1, 2, 3, 4},
|
||||||
|
wantSegmentCount: 0,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, test := range cases {
|
||||||
|
t.Run(test.name, func(t *testing.T) {
|
||||||
|
fullInput := fullTranscriptFixture()
|
||||||
|
intermediateInput := intermediateFixture()
|
||||||
|
minimalInput := minimalFixture()
|
||||||
|
|
||||||
|
fullResult, fullErr := Apply(fullInput, test.opts)
|
||||||
|
intermediateResult, intermediateErr := ApplyIntermediate(intermediateInput, test.opts)
|
||||||
|
minimalResult, minimalErr := ApplyMinimal(minimalInput, test.opts)
|
||||||
|
|
||||||
|
if test.wantErrorSubstring != "" {
|
||||||
|
assertErrorContains(t, fullErr, test.wantErrorSubstring)
|
||||||
|
assertErrorContains(t, intermediateErr, test.wantErrorSubstring)
|
||||||
|
assertErrorContains(t, minimalErr, test.wantErrorSubstring)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if fullErr != nil {
|
||||||
|
t.Fatalf("apply full failed: %v", fullErr)
|
||||||
|
}
|
||||||
|
if intermediateErr != nil {
|
||||||
|
t.Fatalf("apply intermediate failed: %v", intermediateErr)
|
||||||
|
}
|
||||||
|
if minimalErr != nil {
|
||||||
|
t.Fatalf("apply minimal failed: %v", minimalErr)
|
||||||
|
}
|
||||||
|
|
||||||
|
assertIntSlice(t, extractFullIDs(fullResult.Transcript.Segments), extractSequentialIDs(test.wantSegmentCount))
|
||||||
|
assertIntSlice(t, extractIntermediateIDs(intermediateResult.Transcript.Segments), extractSequentialIDs(test.wantSegmentCount))
|
||||||
|
assertIntSlice(t, extractMinimalIDs(minimalResult.Transcript.Segments), extractSequentialIDs(test.wantSegmentCount))
|
||||||
|
assertStringSlice(t, extractFullTexts(fullResult.Transcript.Segments), test.wantTexts)
|
||||||
|
assertStringSlice(t, extractIntermediateTexts(intermediateResult.Transcript.Segments), test.wantTexts)
|
||||||
|
assertStringSlice(t, extractMinimalTexts(minimalResult.Transcript.Segments), test.wantTexts)
|
||||||
|
assertIntMap(t, fullResult.OldToNewID, test.wantOldToNew)
|
||||||
|
assertIntMap(t, intermediateResult.OldToNewID, test.wantOldToNew)
|
||||||
|
assertIntMap(t, minimalResult.OldToNewID, test.wantOldToNew)
|
||||||
|
assertIntSlice(t, fullResult.RemovedIDs, test.wantRemoved)
|
||||||
|
assertIntSlice(t, intermediateResult.RemovedIDs, test.wantRemoved)
|
||||||
|
assertIntSlice(t, minimalResult.RemovedIDs, test.wantRemoved)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestApplyOutputInvariantsValidAfterRenumberAndOverlapRecompute(t *testing.T) {
|
func TestApplyOutputInvariantsValidAfterRenumberAndOverlapRecompute(t *testing.T) {
|
||||||
input := overlapTranscriptFixture()
|
input := overlapTranscriptFixture()
|
||||||
selector := mustParseSelector(t, "2,1")
|
selector := mustParseSelector(t, "2,1")
|
||||||
@@ -666,3 +766,108 @@ func equalStringSlices(got []string, want []string) bool {
|
|||||||
}
|
}
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func assertErrorContains(t *testing.T, err error, substring string) {
|
||||||
|
t.Helper()
|
||||||
|
if err == nil {
|
||||||
|
t.Fatalf("expected error containing %q", substring)
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), substring) {
|
||||||
|
t.Fatalf("error %q does not contain %q", err.Error(), substring)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func assertStringSlice(t *testing.T, got []string, want []string) {
|
||||||
|
t.Helper()
|
||||||
|
if !equalStringSlices(got, want) {
|
||||||
|
t.Fatalf("slice = %v, want %v", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func extractSequentialIDs(count int) []int {
|
||||||
|
ids := make([]int, count)
|
||||||
|
for index := range ids {
|
||||||
|
ids[index] = index + 1
|
||||||
|
}
|
||||||
|
return ids
|
||||||
|
}
|
||||||
|
|
||||||
|
func extractFullIDs(segments []schema.Segment) []int {
|
||||||
|
ids := make([]int, len(segments))
|
||||||
|
for index, segment := range segments {
|
||||||
|
ids[index] = segment.ID
|
||||||
|
}
|
||||||
|
return ids
|
||||||
|
}
|
||||||
|
|
||||||
|
func extractIntermediateIDs(segments []schema.IntermediateSegment) []int {
|
||||||
|
ids := make([]int, len(segments))
|
||||||
|
for index, segment := range segments {
|
||||||
|
ids[index] = segment.ID
|
||||||
|
}
|
||||||
|
return ids
|
||||||
|
}
|
||||||
|
|
||||||
|
func extractMinimalIDs(segments []schema.MinimalSegment) []int {
|
||||||
|
ids := make([]int, len(segments))
|
||||||
|
for index, segment := range segments {
|
||||||
|
ids[index] = segment.ID
|
||||||
|
}
|
||||||
|
return ids
|
||||||
|
}
|
||||||
|
|
||||||
|
func extractFullTexts(segments []schema.Segment) []string {
|
||||||
|
texts := make([]string, len(segments))
|
||||||
|
for index, segment := range segments {
|
||||||
|
texts[index] = segment.Text
|
||||||
|
}
|
||||||
|
return texts
|
||||||
|
}
|
||||||
|
|
||||||
|
func extractIntermediateTexts(segments []schema.IntermediateSegment) []string {
|
||||||
|
texts := make([]string, len(segments))
|
||||||
|
for index, segment := range segments {
|
||||||
|
texts[index] = segment.Text
|
||||||
|
}
|
||||||
|
return texts
|
||||||
|
}
|
||||||
|
|
||||||
|
func extractMinimalTexts(segments []schema.MinimalSegment) []string {
|
||||||
|
texts := make([]string, len(segments))
|
||||||
|
for index, segment := range segments {
|
||||||
|
texts[index] = segment.Text
|
||||||
|
}
|
||||||
|
return texts
|
||||||
|
}
|
||||||
|
|
||||||
|
func intermediateFixture() schema.IntermediateTranscript {
|
||||||
|
return schema.IntermediateTranscript{
|
||||||
|
Metadata: schema.IntermediateMetadata{
|
||||||
|
Application: "seriatim",
|
||||||
|
Version: "v-test",
|
||||||
|
OutputSchema: schema.OutputSchemaIntermediate,
|
||||||
|
},
|
||||||
|
Segments: []schema.IntermediateSegment{
|
||||||
|
{ID: 1, Start: 1, End: 2, Speaker: "Alice", Text: "alpha", Categories: []string{"word-run"}},
|
||||||
|
{ID: 2, Start: 2, End: 3, Speaker: "Bob", Text: "beta", Categories: []string{"filler", "backchannel"}},
|
||||||
|
{ID: 3, Start: 3, End: 4, Speaker: "Carol", Text: "gamma", Categories: []string{"normal"}},
|
||||||
|
{ID: 4, Start: 4, End: 5, Speaker: "Dan", Text: "delta", Categories: []string{"normal"}},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func minimalFixture() schema.MinimalTranscript {
|
||||||
|
return schema.MinimalTranscript{
|
||||||
|
Metadata: schema.MinimalMetadata{
|
||||||
|
Application: "seriatim",
|
||||||
|
Version: "v-test",
|
||||||
|
OutputSchema: schema.OutputSchemaMinimal,
|
||||||
|
},
|
||||||
|
Segments: []schema.MinimalSegment{
|
||||||
|
{ID: 1, Start: 1, End: 2, Speaker: "Alice", Text: "alpha"},
|
||||||
|
{ID: 2, Start: 2, End: 3, Speaker: "Bob", Text: "beta"},
|
||||||
|
{ID: 3, Start: 3, End: 4, Speaker: "Carol", Text: "gamma"},
|
||||||
|
{ID: 4, Start: 4, End: 5, Speaker: "Dan", Text: "delta"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -8,9 +8,9 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
const (
|
const (
|
||||||
SchemaMinimal = "seriatim-minimal"
|
SchemaMinimal = schema.OutputSchemaMinimal
|
||||||
SchemaIntermediate = "seriatim-intermediate"
|
SchemaIntermediate = schema.OutputSchemaIntermediate
|
||||||
SchemaFull = "seriatim-full"
|
SchemaFull = schema.OutputSchemaFull
|
||||||
)
|
)
|
||||||
|
|
||||||
// Artifact stores a parsed seriatim output artifact of one supported schema.
|
// Artifact stores a parsed seriatim output artifact of one supported schema.
|
||||||
|
|||||||
156
internal/trim/run.go
Normal file
156
internal/trim/run.go
Normal file
@@ -0,0 +1,156 @@
|
|||||||
|
package trim
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"sort"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||||
|
"gitea.maximumdirect.net/eric/seriatim/internal/jsonfile"
|
||||||
|
"gitea.maximumdirect.net/eric/seriatim/internal/report"
|
||||||
|
)
|
||||||
|
|
||||||
|
type auditReport struct {
|
||||||
|
Operation string `json:"operation"`
|
||||||
|
InputFile string `json:"input_file"`
|
||||||
|
OutputFile string `json:"output_file"`
|
||||||
|
InputSchema string `json:"input_schema"`
|
||||||
|
OutputSchema string `json:"output_schema"`
|
||||||
|
Mode string `json:"mode"`
|
||||||
|
Selector string `json:"selector"`
|
||||||
|
SelectedIDs []int `json:"selected_ids"`
|
||||||
|
AllowEmpty bool `json:"allow_empty"`
|
||||||
|
InputSegmentCount int `json:"input_segment_count"`
|
||||||
|
RetainedSegmentCount int `json:"retained_segment_count"`
|
||||||
|
RemovedSegmentCount int `json:"removed_segment_count"`
|
||||||
|
RemovedInputIDs []int `json:"removed_input_ids"`
|
||||||
|
OldToNewIDMapping []idMapping `json:"old_to_new_id_mapping"`
|
||||||
|
OverlapGroupsRecomputed bool `json:"overlap_groups_recomputed"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type idMapping struct {
|
||||||
|
OldID int `json:"old_id"`
|
||||||
|
NewID int `json:"new_id"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// Run executes artifact-level trim orchestration.
|
||||||
|
func Run(ctx context.Context, cfg config.TrimConfig) error {
|
||||||
|
if err := ctx.Err(); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
selector, err := ParseSelector(cfg.Selector)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("invalid selector %q: %w", cfg.Selector, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
data, err := os.ReadFile(cfg.InputFile)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("read --input-file %q: %w", cfg.InputFile, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
artifact, err := ParseArtifactJSON(data)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("--input-file %q: %w", cfg.InputFile, err)
|
||||||
|
}
|
||||||
|
inputSegmentCount := artifact.SegmentCount()
|
||||||
|
inputSchema := artifact.Schema
|
||||||
|
|
||||||
|
mode := ModeKeep
|
||||||
|
if cfg.Mode == "remove" {
|
||||||
|
mode = ModeRemove
|
||||||
|
}
|
||||||
|
|
||||||
|
trimmed, err := ApplyArtifact(artifact, Options{
|
||||||
|
Mode: mode,
|
||||||
|
Selector: selector,
|
||||||
|
AllowEmpty: cfg.AllowEmpty,
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
outputSchema := artifact.Schema
|
||||||
|
if cfg.OutputSchema != "" {
|
||||||
|
outputSchema = cfg.OutputSchema
|
||||||
|
}
|
||||||
|
|
||||||
|
outputArtifact, err := ConvertArtifact(trimmed.Artifact, outputSchema)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := ValidateArtifact(outputArtifact); err != nil {
|
||||||
|
return fmt.Errorf("validate trimmed output: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := jsonfile.Write(cfg.OutputFile, outputArtifact.Value()); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
if cfg.ReportFile == "" {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
audit := auditReport{
|
||||||
|
Operation: "trim",
|
||||||
|
InputFile: cfg.InputFile,
|
||||||
|
OutputFile: cfg.OutputFile,
|
||||||
|
InputSchema: inputSchema,
|
||||||
|
OutputSchema: outputArtifact.Schema,
|
||||||
|
Mode: cfg.Mode,
|
||||||
|
Selector: cfg.Selector,
|
||||||
|
SelectedIDs: selector.IDs(),
|
||||||
|
AllowEmpty: cfg.AllowEmpty,
|
||||||
|
InputSegmentCount: inputSegmentCount,
|
||||||
|
RetainedSegmentCount: len(trimmed.OldToNewID),
|
||||||
|
RemovedSegmentCount: len(trimmed.RemovedIDs),
|
||||||
|
RemovedInputIDs: append([]int(nil), trimmed.RemovedIDs...),
|
||||||
|
OldToNewIDMapping: orderedIDMapping(trimmed.OldToNewID),
|
||||||
|
OverlapGroupsRecomputed: trimmed.OverlapGroupsRecomputed,
|
||||||
|
}
|
||||||
|
auditJSON, err := json.Marshal(audit)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("marshal trim audit report: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
rpt := report.Report{
|
||||||
|
Metadata: report.Metadata{
|
||||||
|
Application: outputArtifact.Application(),
|
||||||
|
Version: outputArtifact.Version(),
|
||||||
|
InputReader: "trim-artifact",
|
||||||
|
InputFiles: []string{cfg.InputFile},
|
||||||
|
OutputModules: []string{"json"},
|
||||||
|
},
|
||||||
|
Events: []report.Event{
|
||||||
|
report.Info("trim", "trim", fmt.Sprintf("trimmed %d input segment(s) into %d output segment(s) with mode=%s", inputSegmentCount, outputArtifact.SegmentCount(), cfg.Mode)),
|
||||||
|
report.Info("trim", "trim-audit", string(auditJSON)),
|
||||||
|
report.Info("trim", "validate-output", fmt.Sprintf("validated %d output segment(s)", outputArtifact.SegmentCount())),
|
||||||
|
report.Info("output", "json", "wrote transcript JSON"),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
if err := report.WriteJSON(cfg.ReportFile, rpt); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func orderedIDMapping(mapping map[int]int) []idMapping {
|
||||||
|
keys := make([]int, 0, len(mapping))
|
||||||
|
for oldID := range mapping {
|
||||||
|
keys = append(keys, oldID)
|
||||||
|
}
|
||||||
|
sort.Ints(keys)
|
||||||
|
|
||||||
|
pairs := make([]idMapping, 0, len(keys))
|
||||||
|
for _, oldID := range keys {
|
||||||
|
pairs = append(pairs, idMapping{
|
||||||
|
OldID: oldID,
|
||||||
|
NewID: mapping[oldID],
|
||||||
|
})
|
||||||
|
}
|
||||||
|
return pairs
|
||||||
|
}
|
||||||
28
internal/trim/run_test.go
Normal file
28
internal/trim/run_test.go
Normal file
@@ -0,0 +1,28 @@
|
|||||||
|
package trim
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"path/filepath"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/seriatim/internal/config"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestRunReturnsContextErrorBeforeWork(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
ctx, cancel := context.WithCancel(context.Background())
|
||||||
|
cancel()
|
||||||
|
|
||||||
|
err := Run(ctx, config.TrimConfig{
|
||||||
|
InputFile: filepath.Join(dir, "input.json"),
|
||||||
|
OutputFile: filepath.Join(dir, "output.json"),
|
||||||
|
Mode: "keep",
|
||||||
|
Selector: "1",
|
||||||
|
OutputSchema: "",
|
||||||
|
AllowEmpty: false,
|
||||||
|
})
|
||||||
|
if !errors.Is(err, context.Canceled) {
|
||||||
|
t.Fatalf("error = %v, want context.Canceled", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -14,6 +14,10 @@ import (
|
|||||||
var schemaFS embed.FS
|
var schemaFS embed.FS
|
||||||
|
|
||||||
const (
|
const (
|
||||||
|
OutputSchemaMinimal = "seriatim-minimal"
|
||||||
|
OutputSchemaIntermediate = "seriatim-intermediate"
|
||||||
|
OutputSchemaFull = "seriatim-full"
|
||||||
|
|
||||||
fullOutputSchemaPath = "full-output.schema.json"
|
fullOutputSchemaPath = "full-output.schema.json"
|
||||||
intermediateOutputSchemaPath = "intermediate-output.schema.json"
|
intermediateOutputSchemaPath = "intermediate-output.schema.json"
|
||||||
minimalOutputSchemaPath = "minimal-output.schema.json"
|
minimalOutputSchemaPath = "minimal-output.schema.json"
|
||||||
@@ -115,6 +119,25 @@ type OverlapGroup struct {
|
|||||||
Resolution string `json:"resolution"`
|
Resolution string `json:"resolution"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ValidOutputSchemaName reports whether value is a supported output schema name.
|
||||||
|
func ValidOutputSchemaName(value string) bool {
|
||||||
|
switch value {
|
||||||
|
case OutputSchemaMinimal, OutputSchemaIntermediate, OutputSchemaFull:
|
||||||
|
return true
|
||||||
|
default:
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// OutputSchemaNames returns supported output schema names in validation order.
|
||||||
|
func OutputSchemaNames() []string {
|
||||||
|
return []string{
|
||||||
|
OutputSchemaMinimal,
|
||||||
|
OutputSchemaIntermediate,
|
||||||
|
OutputSchemaFull,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ValidateTranscript validates a full transcript against the public JSON
|
// ValidateTranscript validates a full transcript against the public JSON
|
||||||
// schema and seriatim-specific semantic rules.
|
// schema and seriatim-specific semantic rules.
|
||||||
func ValidateTranscript(transcript Transcript) error {
|
func ValidateTranscript(transcript Transcript) error {
|
||||||
@@ -228,15 +251,17 @@ func outputSchema(schemaPath string) (*jsonschema.Schema, error) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func validateSemantics(transcript Transcript) error {
|
func validateSemantics(transcript Transcript) error {
|
||||||
|
segments := make([]segmentSemantics, len(transcript.Segments))
|
||||||
for index, segment := range transcript.Segments {
|
for index, segment := range transcript.Segments {
|
||||||
wantID := index + 1
|
segments[index] = segmentSemantics{
|
||||||
if segment.ID != wantID {
|
id: segment.ID,
|
||||||
return fmt.Errorf("segment %d has id %d; want %d", index, segment.ID, wantID)
|
start: segment.Start,
|
||||||
}
|
end: segment.End,
|
||||||
if segment.End < segment.Start {
|
|
||||||
return fmt.Errorf("segment %d has end %.3f before start %.3f", index, segment.End, segment.Start)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if err := validateSegmentSemantics(segments); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
for index, group := range transcript.OverlapGroups {
|
for index, group := range transcript.OverlapGroups {
|
||||||
if group.End < group.Start {
|
if group.End < group.Start {
|
||||||
return fmt.Errorf("overlap_group %d has end %.3f before start %.3f", index, group.End, group.Start)
|
return fmt.Errorf("overlap_group %d has end %.3f before start %.3f", index, group.End, group.Start)
|
||||||
@@ -246,26 +271,43 @@ func validateSemantics(transcript Transcript) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func validateIntermediateSemantics(transcript IntermediateTranscript) error {
|
func validateIntermediateSemantics(transcript IntermediateTranscript) error {
|
||||||
|
segments := make([]segmentSemantics, len(transcript.Segments))
|
||||||
for index, segment := range transcript.Segments {
|
for index, segment := range transcript.Segments {
|
||||||
wantID := index + 1
|
segments[index] = segmentSemantics{
|
||||||
if segment.ID != wantID {
|
id: segment.ID,
|
||||||
return fmt.Errorf("segment %d has id %d; want %d", index, segment.ID, wantID)
|
start: segment.Start,
|
||||||
}
|
end: segment.End,
|
||||||
if segment.End < segment.Start {
|
|
||||||
return fmt.Errorf("segment %d has end %.3f before start %.3f", index, segment.End, segment.Start)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return nil
|
return validateSegmentSemantics(segments)
|
||||||
}
|
}
|
||||||
|
|
||||||
func validateMinimalSemantics(transcript MinimalTranscript) error {
|
func validateMinimalSemantics(transcript MinimalTranscript) error {
|
||||||
|
segments := make([]segmentSemantics, len(transcript.Segments))
|
||||||
for index, segment := range transcript.Segments {
|
for index, segment := range transcript.Segments {
|
||||||
wantID := index + 1
|
segments[index] = segmentSemantics{
|
||||||
if segment.ID != wantID {
|
id: segment.ID,
|
||||||
return fmt.Errorf("segment %d has id %d; want %d", index, segment.ID, wantID)
|
start: segment.Start,
|
||||||
|
end: segment.End,
|
||||||
}
|
}
|
||||||
if segment.End < segment.Start {
|
}
|
||||||
return fmt.Errorf("segment %d has end %.3f before start %.3f", index, segment.End, segment.Start)
|
return validateSegmentSemantics(segments)
|
||||||
|
}
|
||||||
|
|
||||||
|
type segmentSemantics struct {
|
||||||
|
id int
|
||||||
|
start float64
|
||||||
|
end float64
|
||||||
|
}
|
||||||
|
|
||||||
|
func validateSegmentSemantics(segments []segmentSemantics) error {
|
||||||
|
for index, segment := range segments {
|
||||||
|
wantID := index + 1
|
||||||
|
if segment.id != wantID {
|
||||||
|
return fmt.Errorf("segment %d has id %d; want %d", index, segment.id, wantID)
|
||||||
|
}
|
||||||
|
if segment.end < segment.start {
|
||||||
|
return fmt.Errorf("segment %d has end %.3f before start %.3f", index, segment.end, segment.start)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
|
|||||||
@@ -5,6 +5,43 @@ import (
|
|||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
func TestValidOutputSchemaName(t *testing.T) {
|
||||||
|
valid := []string{
|
||||||
|
OutputSchemaMinimal,
|
||||||
|
OutputSchemaIntermediate,
|
||||||
|
OutputSchemaFull,
|
||||||
|
}
|
||||||
|
for _, name := range valid {
|
||||||
|
if !ValidOutputSchemaName(name) {
|
||||||
|
t.Fatalf("expected %q to be valid", name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
invalid := []string{"", "compact", "minimal", "seriatim"}
|
||||||
|
for _, name := range invalid {
|
||||||
|
if ValidOutputSchemaName(name) {
|
||||||
|
t.Fatalf("expected %q to be invalid", name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestOutputSchemaNames(t *testing.T) {
|
||||||
|
names := OutputSchemaNames()
|
||||||
|
want := []string{
|
||||||
|
OutputSchemaMinimal,
|
||||||
|
OutputSchemaIntermediate,
|
||||||
|
OutputSchemaFull,
|
||||||
|
}
|
||||||
|
if len(names) != len(want) {
|
||||||
|
t.Fatalf("len(names) = %d, want %d", len(names), len(want))
|
||||||
|
}
|
||||||
|
for index := range want {
|
||||||
|
if names[index] != want[index] {
|
||||||
|
t.Fatalf("names[%d] = %q, want %q", index, names[index], want[index])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestValidateTranscriptAcceptsValidTranscript(t *testing.T) {
|
func TestValidateTranscriptAcceptsValidTranscript(t *testing.T) {
|
||||||
transcript := validTranscript()
|
transcript := validTranscript()
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user