package source import ( "encoding/json" "strings" "testing" ) func TestValidateDocumentValid(t *testing.T) { doc := validDocument() if err := ValidateDocument(doc); err != nil { t.Fatalf("ValidateDocument() error = %v, want nil", err) } } func TestValidateDocumentNil(t *testing.T) { err := ValidateDocument(nil) if err == nil { t.Fatal("ValidateDocument() error = nil, want error") } if err.Error() != "source document must not be nil" { t.Fatalf("ValidateDocument() error = %q", err.Error()) } } func TestValidateDocumentMissingFields(t *testing.T) { tests := []struct { name string mutate func(*SourceDocument) wantErr string }{ { name: "id", mutate: func(doc *SourceDocument) { doc.ID = " \t" }, wantErr: "source document id must not be empty", }, { name: "id surrounding whitespace", mutate: func(doc *SourceDocument) { doc.ID = " source-1 " }, wantErr: "source document id \" source-1 \" must not contain leading or trailing whitespace", }, { name: "kind", mutate: func(doc *SourceDocument) { doc.Kind = "" }, wantErr: "source document kind must not be empty", }, { name: "format", mutate: func(doc *SourceDocument) { doc.Format = "\n" }, wantErr: "source document format must not be empty", }, { name: "digest", mutate: func(doc *SourceDocument) { doc.Digest = "" }, wantErr: "source document digest must not be empty", }, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { doc := validDocument() tt.mutate(doc) err := ValidateDocument(doc) if err == nil { t.Fatal("ValidateDocument() error = nil, want error") } if err.Error() != tt.wantErr { t.Fatalf("ValidateDocument() error = %q, want %q", err.Error(), tt.wantErr) } }) } } func TestValidateDocumentEmptyUnits(t *testing.T) { doc := validDocument() doc.Units = nil err := ValidateDocument(doc) if err == nil { t.Fatal("ValidateDocument() error = nil, want error") } if err.Error() != "source document units must not be empty" { t.Fatalf("ValidateDocument() error = %q", err.Error()) } } func TestValidateDocumentMissingUnitFields(t *testing.T) { tests := []struct { name string mutate func(*SourceDocument) wantErr string }{ { name: "id", mutate: func(doc *SourceDocument) { doc.Units[1].ID = 0 }, wantErr: "source unit[1].id must be positive", }, { name: "kind", mutate: func(doc *SourceDocument) { doc.Units[1].Kind = " " }, wantErr: "source unit[1].kind must not be empty", }, { name: "text", mutate: func(doc *SourceDocument) { doc.Units[1].Text = "\n\t" }, wantErr: "source unit[1].text must not be empty", }, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { doc := validDocument() tt.mutate(doc) err := ValidateDocument(doc) if err == nil { t.Fatal("ValidateDocument() error = nil, want error") } if err.Error() != tt.wantErr { t.Fatalf("ValidateDocument() error = %q, want %q", err.Error(), tt.wantErr) } }) } } func TestValidateDocumentDuplicateUnitIDs(t *testing.T) { doc := validDocument() doc.Units[1].ID = 1 err := ValidateDocument(doc) if err == nil { t.Fatal("ValidateDocument() error = nil, want error") } if err.Error() != "source unit id 1 is duplicated" { t.Fatalf("ValidateDocument() error = %q", err.Error()) } } func TestValidateDocumentUnitReferences(t *testing.T) { tests := []struct { name string mutate func(*SourceDocument) wantErr string }{ { name: "missing", mutate: func(doc *SourceDocument) { doc.Units[0].Ref = SourceRef{} }, wantErr: "source unit[0].ref: source ref source_id must not be empty", }, { name: "foreign source", mutate: func(doc *SourceDocument) { doc.Units[0].Ref.SourceID = "source-2" }, wantErr: "source unit[0].ref: source ref source_id \"source-2\" does not match document id \"source-1\"", }, { name: "non-self range", mutate: func(doc *SourceDocument) { doc.Units[0].Ref.StartUnitID = 2 doc.Units[0].Ref.EndUnitID = 2 }, wantErr: "source unit[0].ref must identify source unit id 1", }, { name: "reversed range", mutate: func(doc *SourceDocument) { doc.Units[0].Ref.StartUnitID = 2 doc.Units[0].Ref.EndUnitID = 1 }, wantErr: "source unit[0].ref: source ref start_unit_id 2 appears after end_unit_id 1", }, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { doc := validDocument() tt.mutate(doc) err := ValidateDocument(doc) if err == nil { t.Fatal("ValidateDocument() error = nil, want unit reference error") } if err.Error() != tt.wantErr { t.Fatalf("ValidateDocument() error = %q, want %q", err.Error(), tt.wantErr) } }) } } func TestDigestDocumentIsDeterministicAndIncludesUnitReference(t *testing.T) { doc := validDocument() doc.Metadata = map[string]any{"second": "value", "first": true} first, err := DigestDocument(doc) if err != nil { t.Fatalf("DigestDocument() error = %v, want nil", err) } reordered := validDocument() reordered.Metadata = map[string]any{"first": true, "second": "value"} second, err := DigestDocument(reordered) if err != nil { t.Fatalf("DigestDocument(reordered) error = %v, want nil", err) } if first != second { t.Fatalf("digests = %q and %q, want deterministic map ordering", first, second) } changed := validDocument() changed.Metadata = map[string]any{"first": true, "second": "value"} changed.Units[0].Ref.SourceID = "different-source" changedDigest, err := DigestDocument(changed) if err != nil { t.Fatalf("DigestDocument(changed) error = %v, want nil", err) } if first == changedDigest { t.Fatalf("digest = %q after reference change, want different digest", changedDigest) } } func TestDigestChunkIsDeterministicAndIncludesReference(t *testing.T) { doc := validDocument() chunk := Chunk{ ID: "chunk-1", SourceID: doc.ID, Index: 0, Ref: SourceRef{SourceID: doc.ID, StartUnitID: 1, EndUnitID: 2}, Content: []byte("chunk content"), MediaType: "text/plain", Units: doc.Units, Metadata: map[string]any{"second": "value", "first": true}, } first, err := DigestChunk(chunk) if err != nil { t.Fatalf("DigestChunk() error = %v, want nil", err) } chunk.Metadata = map[string]any{"first": true, "second": "value"} second, err := DigestChunk(chunk) if err != nil { t.Fatalf("DigestChunk(reordered metadata) error = %v, want nil", err) } if first != second { t.Fatalf("digests = %q and %q, want deterministic map ordering", first, second) } chunk.Ref.EndUnitID = 1 changed, err := DigestChunk(chunk) if err != nil { t.Fatalf("DigestChunk(changed ref) error = %v, want nil", err) } if first == changed { t.Fatalf("digest = %q after reference change, want different digest", changed) } } func TestDigestChunkIncludesAnnotationScopes(t *testing.T) { doc := validDocument() chunk := Chunk{ ID: "chunk-1", SourceID: doc.ID, Ref: SourceRef{SourceID: doc.ID, StartUnitID: 1, EndUnitID: 2}, Content: []byte("content"), MediaType: "text/plain", Units: doc.Units, Annotations: ChunkAnnotations{"scope": json.RawMessage(`{"value":1}`)}, PlanAnnotations: ChunkAnnotations{"scope": json.RawMessage(`{"value":2}`)}, } base, err := DigestChunk(chunk) if err != nil { t.Fatalf("DigestChunk() error = %v", err) } chunk.Annotations["scope"] = json.RawMessage(`{"value":3}`) rangeChanged, _ := DigestChunk(chunk) chunk.Annotations["scope"] = json.RawMessage(`{"value":1}`) chunk.PlanAnnotations["scope"] = json.RawMessage(`{"value":3}`) planChanged, _ := DigestChunk(chunk) if base == rangeChanged || base == planChanged || rangeChanged == planChanged { t.Fatalf("annotation scope digests did not change distinctly: %q %q %q", base, rangeChanged, planChanged) } } func TestValidateRefValid(t *testing.T) { doc := validDocument() ref := SourceRef{ SourceID: "source-1", StartUnitID: 1, EndUnitID: 2, } if err := ValidateRef(doc, ref); err != nil { t.Fatalf("ValidateRef() error = %v, want nil", err) } } func TestValidateRefSourceIDMismatch(t *testing.T) { doc := validDocument() ref := SourceRef{ SourceID: "source-2", StartUnitID: 1, EndUnitID: 2, } err := ValidateRef(doc, ref) if err == nil { t.Fatal("ValidateRef() error = nil, want error") } if err.Error() != "source ref source_id \"source-2\" does not match document id \"source-1\"" { t.Fatalf("ValidateRef() error = %q", err.Error()) } } func TestValidateRefMissingUnitIDs(t *testing.T) { tests := []struct { name string ref SourceRef wantErr string }{ { name: "missing source id", ref: SourceRef{StartUnitID: 1, EndUnitID: 2}, wantErr: "source ref source_id must not be empty", }, { name: "source id surrounding whitespace", ref: SourceRef{SourceID: " source-1 ", StartUnitID: 1, EndUnitID: 2}, wantErr: "source ref source_id \" source-1 \" must not contain leading or trailing whitespace", }, { name: "missing start id", ref: SourceRef{SourceID: "source-1", EndUnitID: 2}, wantErr: "source ref start_unit_id must be positive", }, { name: "missing end id", ref: SourceRef{SourceID: "source-1", StartUnitID: 1}, wantErr: "source ref end_unit_id must be positive", }, { name: "unknown start id", ref: SourceRef{SourceID: "source-1", StartUnitID: 9, EndUnitID: 2}, wantErr: "source ref start_unit_id 9 was not found", }, { name: "unknown end id", ref: SourceRef{SourceID: "source-1", StartUnitID: 1, EndUnitID: 9}, wantErr: "source ref end_unit_id 9 was not found", }, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { err := ValidateRef(validDocument(), tt.ref) if err == nil { t.Fatal("ValidateRef() error = nil, want error") } if err.Error() != tt.wantErr { t.Fatalf("ValidateRef() error = %q, want %q", err.Error(), tt.wantErr) } }) } } func TestValidateRefReversedUnitOrder(t *testing.T) { doc := validDocument() ref := SourceRef{ SourceID: "source-1", StartUnitID: 2, EndUnitID: 1, } err := ValidateRef(doc, ref) if err == nil { t.Fatal("ValidateRef() error = nil, want error") } if !strings.Contains(err.Error(), "appears after") { t.Fatalf("ValidateRef() error = %q, want reversed order error", err.Error()) } } func TestUnitIndex(t *testing.T) { doc := validDocument() index, ok := UnitIndex(doc, 2) if !ok { t.Fatal("UnitIndex() ok = false, want true") } if index != 1 { t.Fatalf("UnitIndex() index = %d, want 1", index) } index, ok = UnitIndex(doc, 9) if ok { t.Fatal("UnitIndex() ok = true, want false") } if index != 0 { t.Fatalf("UnitIndex() index = %d, want 0", index) } } func validDocument() *SourceDocument { return &SourceDocument{ ID: "source-1", Kind: "document", Format: "text/plain", Digest: "sha256:abc123", Units: []SourceUnit{ { ID: 1, Kind: "paragraph", Text: "First unit.", Ref: SourceRef{SourceID: "source-1", StartUnitID: 1, EndUnitID: 1}, }, { ID: 2, Kind: "paragraph", Text: "Second unit.", Ref: SourceRef{SourceID: "source-1", StartUnitID: 2, EndUnitID: 2}, }, }, } }