diff --git a/docs/adr/0015-separate-process-warnings-from-quality-diagnostics.md b/docs/adr/0015-separate-process-warnings-from-quality-diagnostics.md new file mode 100644 index 00000000..80340636 --- /dev/null +++ b/docs/adr/0015-separate-process-warnings-from-quality-diagnostics.md @@ -0,0 +1,67 @@ +# ADR-0015: Separate process warnings from quality diagnostics + +**Status:** Accepted +**Date:** 2026-08-27 + +## Context + +Notarius currently represents process degradation, incomplete validation, +extraction-quality doubt, and routine normalization with one flat warning +record. That makes ordinary successful runs noisy, loses the framework context +needed to explain a finding, and gives `warning_count` no stable operational +meaning. It also permits output encoders to add a warning after the durable +warning file has already been written. + +The application needs one bounded diagnostic model that preserves exact +occurrence counts while retaining only safe, representative samples. Fresh and +resumed logical runs must present the same groups. The model must not alter +validation decisions, retry budgets, rejected-output behavior, or process exit +policy. + +## Decision + +Warnings are reserved for a completed run that advanced under an allowed +process-level degradation or incomplete-work policy. Extraction-quality signals +are advisories, and routine accepted transformations are observations. A +non-degraded successful run therefore has zero actionable warnings. + +Modules and validators own a diagnostic's disposition, category, reason code, +scope, and safe message. The framework adds pipeline origin, including stage, +step, lane, module, validator, and chunk context where applicable. It then +aggregates deterministically by disposition, category, reason code, and full +origin. Chunk context remains on representative samples so equivalent findings +across chunks aggregate together. + +Diagnostics carry exact occurrence counts, at most three distinct samples, and +numeric omitted-sample metadata. Producers and validators are bounded to 64 +local groups. Final actionable warning groups are bounded without truncation; +the non-warning collection may truncate represented groups while preserving an +exact total occurrence count and explicit truncation metadata. + +The public contracts will be versioned: grouped actionable warnings use +`notarius.warnings.v2`, grouped advisories and observations use +`notarius.diagnostics.v1`, and the run receipt uses +`notarius.run-result.v2`. Successful output encoders return logical files or +an error; they do not add post-encoding warnings. + +## Alternatives considered + +- Keep one warning list and filter only CLI output. This would leave durable + consumers with the same semantically mixed, unbounded contract. +- Map reason codes to severity in a central framework registry. This would + split module-owned meaning between synchronized policy tables and make new + diagnostic meaning implicit. +- Preserve local omission warning records. They inflate visible group counts + and lose exact occurrence semantics. +- Keep output-encoder warnings. A one-pass encoder cannot include those + records consistently in files it has already serialized; a two-phase encoder + protocol is deferred until a demonstrated need exists. + +## Consequences + +The framework gains validated diagnostic primitives, local collection, +origin-aware aggregation, and versioned durable presentation. Existing warning +transport remains temporarily while producers migrate. Current architecture, +operator, integration, and internal documentation will describe the behavior +only as each implementation step lands; this accepted decision does not claim +that the migration is complete. diff --git a/docs/roadmap/implementation.md b/docs/roadmap/implementation.md index 9a0ede8c..2d89bc18 100644 --- a/docs/roadmap/implementation.md +++ b/docs/roadmap/implementation.md @@ -193,7 +193,7 @@ Remove `OutputResult.Warnings`. An output encoder either returns its complete logical files or returns an error. It cannot discover a warning after the warning file has already been serialized. -## Stage 1 — Record The Decision And Add Core Diagnostic Primitives +## Stage 1 ✅ — Record The Decision And Add Core Diagnostic Primitives ### Goal diff --git a/internal/framework/contracts/diagnostics.go b/internal/framework/contracts/diagnostics.go new file mode 100644 index 00000000..d65effbf --- /dev/null +++ b/internal/framework/contracts/diagnostics.go @@ -0,0 +1,348 @@ +package contracts + +import ( + "errors" + "fmt" + "strings" + "unicode/utf8" +) + +const ( + MaxDiagnosticReasonCodeBytes = 128 + MaxDiagnosticScopeBytes = 512 + MaxDiagnosticMessageBytes = 4 * 1024 + MaxDiagnosticSamples = 3 + MaxProducerDiagnosticGroups = 64 +) + +// DiagnosticDisposition identifies the operator significance of a producer +// finding. Warnings are reserved for process-level degradation or incomplete +// configured work. +type DiagnosticDisposition string + +const ( + DiagnosticDispositionWarning DiagnosticDisposition = "warning" + DiagnosticDispositionAdvisory DiagnosticDisposition = "advisory" + DiagnosticDispositionObservation DiagnosticDisposition = "observation" +) + +// DiagnosticCategory gives a stable, bounded classification for a producer +// finding. +type DiagnosticCategory string + +const ( + DiagnosticCategoryConfiguration DiagnosticCategory = "configuration" + DiagnosticCategoryDegradation DiagnosticCategory = "degradation" + DiagnosticCategoryValidationIncomplete DiagnosticCategory = "validation_incomplete" + DiagnosticCategoryFallback DiagnosticCategory = "fallback" + DiagnosticCategoryDataQuality DiagnosticCategory = "data_quality" + DiagnosticCategoryNormalization DiagnosticCategory = "normalization" +) + +// DiagnosticOriginStage identifies the framework operation that promoted a +// diagnostic. It is framework-owned rather than producer-owned. +type DiagnosticOriginStage string + +const ( + DiagnosticOriginStageReferences DiagnosticOriginStage = "references" + DiagnosticOriginStageChunk DiagnosticOriginStage = "chunk" + DiagnosticOriginStageExtract DiagnosticOriginStage = "extract" + DiagnosticOriginStageMerge DiagnosticOriginStage = "merge" + DiagnosticOriginStageNormalize DiagnosticOriginStage = "normalize" +) + +// DiagnosticSample is a bounded, safe example of a diagnostic occurrence. +// Chunk identity is attached by the framework when it promotes a producer +// diagnostic into a final group. +type DiagnosticSample struct { + Scope string `json:"scope"` + Message string `json:"message"` + ChunkID string `json:"chunk_id,omitempty"` + ChunkIndex *int `json:"chunk_index,omitempty"` +} + +// ProducerDiagnostic is the locally grouped form returned by one producer or +// validator. It intentionally has no pipeline origin. +type ProducerDiagnostic struct { + Disposition DiagnosticDisposition `json:"disposition"` + Category DiagnosticCategory `json:"category"` + ReasonCode string `json:"reason_code"` + OccurrenceCount int `json:"occurrence_count"` + Samples []DiagnosticSample `json:"samples"` + OmittedSampleCount int `json:"omitted_sample_count"` +} + +// DiagnosticOrigin is framework-owned context used to distinguish findings +// from different pipeline locations during final aggregation. +type DiagnosticOrigin struct { + Stage DiagnosticOriginStage `json:"stage"` + StepID string `json:"step_id,omitempty"` + LaneID string `json:"lane_id,omitempty"` + ModuleKey string `json:"module_key,omitempty"` + ValidatorKey string `json:"validator_key,omitempty"` +} + +// DiagnosticGroup is a producer diagnostic after framework origin enrichment. +type DiagnosticGroup struct { + Disposition DiagnosticDisposition `json:"disposition"` + Category DiagnosticCategory `json:"category"` + ReasonCode string `json:"reason_code"` + Origin DiagnosticOrigin `json:"origin"` + OccurrenceCount int `json:"occurrence_count"` + Samples []DiagnosticSample `json:"samples"` + OmittedSampleCount int `json:"omitted_sample_count"` +} + +// DiagnosticCollection is the grouped collection supplied to later durable +// and presentation boundaries. Global aggregation policy is applied by the +// framework before it reaches those boundaries. +type DiagnosticCollection struct { + Groups []DiagnosticGroup `json:"groups"` + Truncated bool `json:"truncated"` + UnrepresentedOccurrenceCount int `json:"unrepresented_occurrence_count"` +} + +// Validate checks a producer-local diagnostic against the public safety and +// classification contract. +func (diagnostic ProducerDiagnostic) Validate() error { + return validateDiagnostic( + diagnostic.Disposition, + diagnostic.Category, + diagnostic.ReasonCode, + diagnostic.OccurrenceCount, + diagnostic.Samples, + diagnostic.OmittedSampleCount, + false, + ) +} + +// ValidateProducerDiagnostics validates the complete set returned by one +// producer or validator result. +func ValidateProducerDiagnostics(diagnostics []ProducerDiagnostic) error { + if len(diagnostics) > MaxProducerDiagnosticGroups { + return errors.New("producer diagnostics exceed maximum group count") + } + for index, diagnostic := range diagnostics { + if err := diagnostic.Validate(); err != nil { + return fmt.Errorf("producer diagnostic %d: %w", index, err) + } + } + return nil +} + +// Validate checks framework-owned origin fields. +func (origin DiagnosticOrigin) Validate() error { + switch origin.Stage { + case DiagnosticOriginStageReferences, DiagnosticOriginStageChunk, DiagnosticOriginStageExtract, DiagnosticOriginStageMerge, DiagnosticOriginStageNormalize: + default: + return errors.New("diagnostic origin stage is invalid") + } + for _, field := range []struct { + name string + value string + }{ + {name: "diagnostic origin step ID", value: origin.StepID}, + {name: "diagnostic origin lane ID", value: origin.LaneID}, + {name: "diagnostic origin module key", value: origin.ModuleKey}, + {name: "diagnostic origin validator key", value: origin.ValidatorKey}, + } { + if field.value != "" && (!utf8.ValidString(field.value) || strings.TrimSpace(field.value) == "") { + return fmt.Errorf("%s must be valid nonblank UTF-8 when present", field.name) + } + } + return nil +} + +// Validate checks a final origin-enriched group. +func (group DiagnosticGroup) Validate() error { + if err := group.Origin.Validate(); err != nil { + return err + } + return validateDiagnostic( + group.Disposition, + group.Category, + group.ReasonCode, + group.OccurrenceCount, + group.Samples, + group.OmittedSampleCount, + true, + ) +} + +// Validate checks the collection shape without imposing later global +// aggregation limits. +func (collection DiagnosticCollection) Validate() error { + if collection.UnrepresentedOccurrenceCount < 0 { + return errors.New("diagnostic collection unrepresented occurrence count must not be negative") + } + if !collection.Truncated && collection.UnrepresentedOccurrenceCount != 0 { + return errors.New("diagnostic collection has unrepresented occurrences without truncation") + } + seen := make(map[diagnosticGroupKey]struct{}, len(collection.Groups)) + for index, group := range collection.Groups { + if err := group.Validate(); err != nil { + return fmt.Errorf("diagnostic group %d: %w", index, err) + } + key := diagnosticGroupKeyFromGroup(group) + if _, exists := seen[key]; exists { + return errors.New("diagnostic collection contains duplicate group identity") + } + seen[key] = struct{}{} + } + return nil +} + +// CloneProducerDiagnostics returns independent diagnostic slice ownership. +func CloneProducerDiagnostics(diagnostics []ProducerDiagnostic) []ProducerDiagnostic { + if len(diagnostics) == 0 { + return nil + } + cloned := make([]ProducerDiagnostic, len(diagnostics)) + for index, diagnostic := range diagnostics { + cloned[index] = cloneProducerDiagnostic(diagnostic) + } + return cloned +} + +// CloneDiagnosticCollection returns independent collection ownership. +func CloneDiagnosticCollection(collection DiagnosticCollection) DiagnosticCollection { + collection.Groups = make([]DiagnosticGroup, len(collection.Groups)) + for index, group := range collection.Groups { + collection.Groups[index] = cloneDiagnosticGroup(group) + } + return collection +} + +func validateDiagnostic(disposition DiagnosticDisposition, category DiagnosticCategory, reasonCode string, occurrenceCount int, samples []DiagnosticSample, omittedSampleCount int, allowChunkContext bool) error { + if !diagnosticCategoryAllowed(disposition, category) { + return errors.New("diagnostic disposition and category combination is invalid") + } + if err := validateDiagnosticText(reasonCode, MaxDiagnosticReasonCodeBytes, "diagnostic reason code"); err != nil { + return err + } + if occurrenceCount <= 0 { + return errors.New("diagnostic occurrence count must be positive") + } + if len(samples) == 0 { + return errors.New("diagnostic samples must not be empty") + } + if len(samples) > MaxDiagnosticSamples { + return errors.New("diagnostic samples exceed maximum count") + } + seen := make(map[diagnosticSampleKey]struct{}, len(samples)) + for index, sample := range samples { + if err := validateDiagnosticSample(sample, allowChunkContext); err != nil { + return fmt.Errorf("diagnostic sample %d: %w", index, err) + } + key := diagnosticSampleKeyFromSample(sample) + if _, exists := seen[key]; exists { + return errors.New("diagnostic samples must be distinct") + } + seen[key] = struct{}{} + } + if occurrenceCount < len(samples) { + return errors.New("diagnostic occurrence count is smaller than sample count") + } + if omittedSampleCount != occurrenceCount-len(samples) { + return errors.New("diagnostic omitted sample count is inconsistent") + } + return nil +} + +func diagnosticCategoryAllowed(disposition DiagnosticDisposition, category DiagnosticCategory) bool { + switch disposition { + case DiagnosticDispositionWarning: + return category == DiagnosticCategoryConfiguration || category == DiagnosticCategoryDegradation || category == DiagnosticCategoryValidationIncomplete || category == DiagnosticCategoryFallback + case DiagnosticDispositionAdvisory: + return category == DiagnosticCategoryDataQuality + case DiagnosticDispositionObservation: + return category == DiagnosticCategoryNormalization + default: + return false + } +} + +func validateDiagnosticSample(sample DiagnosticSample, allowChunkContext bool) error { + if err := validateDiagnosticText(sample.Scope, MaxDiagnosticScopeBytes, "diagnostic sample scope"); err != nil { + return err + } + if err := validateDiagnosticText(sample.Message, MaxDiagnosticMessageBytes, "diagnostic sample message"); err != nil { + return err + } + if !allowChunkContext && (sample.ChunkID != "" || sample.ChunkIndex != nil) { + return errors.New("producer diagnostic sample must not include framework chunk context") + } + if sample.ChunkID != "" && (!utf8.ValidString(sample.ChunkID) || strings.TrimSpace(sample.ChunkID) == "") { + return errors.New("diagnostic sample chunk ID must be valid nonblank UTF-8 when present") + } + if sample.ChunkIndex != nil && *sample.ChunkIndex < 0 { + return errors.New("diagnostic sample chunk index must not be negative") + } + return nil +} + +func validateDiagnosticText(value string, maximum int, name string) error { + if !utf8.ValidString(value) { + return fmt.Errorf("%s must be valid UTF-8", name) + } + if strings.TrimSpace(value) == "" { + return fmt.Errorf("%s must not be blank", name) + } + if len(value) > maximum { + return fmt.Errorf("%s exceeds maximum length", name) + } + return nil +} + +func cloneProducerDiagnostic(diagnostic ProducerDiagnostic) ProducerDiagnostic { + diagnostic.Samples = cloneDiagnosticSamples(diagnostic.Samples) + return diagnostic +} + +func cloneDiagnosticGroup(group DiagnosticGroup) DiagnosticGroup { + group.Samples = cloneDiagnosticSamples(group.Samples) + return group +} + +func cloneDiagnosticSamples(samples []DiagnosticSample) []DiagnosticSample { + if len(samples) == 0 { + return nil + } + cloned := make([]DiagnosticSample, len(samples)) + for index, sample := range samples { + if sample.ChunkIndex != nil { + chunkIndex := *sample.ChunkIndex + sample.ChunkIndex = &chunkIndex + } + cloned[index] = sample + } + return cloned +} + +type diagnosticSampleKey struct { + scope string + message string + chunkID string + chunkIndex int + hasChunkIndex bool +} + +func diagnosticSampleKeyFromSample(sample DiagnosticSample) diagnosticSampleKey { + key := diagnosticSampleKey{scope: sample.Scope, message: sample.Message, chunkID: sample.ChunkID} + if sample.ChunkIndex != nil { + key.chunkIndex = *sample.ChunkIndex + key.hasChunkIndex = true + } + return key +} + +type diagnosticGroupKey struct { + disposition DiagnosticDisposition + category DiagnosticCategory + reasonCode string + origin DiagnosticOrigin +} + +func diagnosticGroupKeyFromGroup(group DiagnosticGroup) diagnosticGroupKey { + return diagnosticGroupKey{disposition: group.Disposition, category: group.Category, reasonCode: group.ReasonCode, origin: group.Origin} +} diff --git a/internal/framework/contracts/diagnostics_test.go b/internal/framework/contracts/diagnostics_test.go new file mode 100644 index 00000000..09177ce0 --- /dev/null +++ b/internal/framework/contracts/diagnostics_test.go @@ -0,0 +1,139 @@ +package contracts + +import ( + "strings" + "testing" +) + +func TestProducerDiagnosticValidationAcceptsClassificationMatrix(t *testing.T) { + for _, test := range []struct { + name string + disposition DiagnosticDisposition + category DiagnosticCategory + }{ + {name: "configuration warning", disposition: DiagnosticDispositionWarning, category: DiagnosticCategoryConfiguration}, + {name: "degradation warning", disposition: DiagnosticDispositionWarning, category: DiagnosticCategoryDegradation}, + {name: "incomplete validation warning", disposition: DiagnosticDispositionWarning, category: DiagnosticCategoryValidationIncomplete}, + {name: "fallback warning", disposition: DiagnosticDispositionWarning, category: DiagnosticCategoryFallback}, + {name: "quality advisory", disposition: DiagnosticDispositionAdvisory, category: DiagnosticCategoryDataQuality}, + {name: "normalization observation", disposition: DiagnosticDispositionObservation, category: DiagnosticCategoryNormalization}, + } { + t.Run(test.name, func(t *testing.T) { + diagnostic := validProducerDiagnostic() + diagnostic.Disposition = test.disposition + diagnostic.Category = test.category + if err := diagnostic.Validate(); err != nil { + t.Fatalf("Validate() error = %v", err) + } + }) + } +} + +func TestProducerDiagnosticValidationRejectsInvalidFieldsAndCounts(t *testing.T) { + tooLongReason := strings.Repeat("r", MaxDiagnosticReasonCodeBytes+1) + tooLongScope := strings.Repeat("s", MaxDiagnosticScopeBytes+1) + tooLongMessage := strings.Repeat("m", MaxDiagnosticMessageBytes+1) + for _, test := range []struct { + name string + mutate func(*ProducerDiagnostic) + }{ + {name: "invalid classification", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Category = DiagnosticCategoryDataQuality }}, + {name: "blank reason", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.ReasonCode = " \t" }}, + {name: "invalid reason UTF-8", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.ReasonCode = string([]byte{0xff}) }}, + {name: "oversized reason", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.ReasonCode = tooLongReason }}, + {name: "blank scope", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Samples[0].Scope = "\n" }}, + {name: "invalid scope UTF-8", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Samples[0].Scope = string([]byte{0xff}) }}, + {name: "oversized scope", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Samples[0].Scope = tooLongScope }}, + {name: "blank message", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Samples[0].Message = " " }}, + {name: "invalid message UTF-8", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Samples[0].Message = string([]byte{0xff}) }}, + {name: "oversized message", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Samples[0].Message = tooLongMessage }}, + {name: "zero occurrences", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.OccurrenceCount = 0 }}, + {name: "missing samples", mutate: func(diagnostic *ProducerDiagnostic) { + diagnostic.Samples = nil + diagnostic.OmittedSampleCount = diagnostic.OccurrenceCount + }}, + {name: "too many samples", mutate: func(diagnostic *ProducerDiagnostic) { + diagnostic.OccurrenceCount = 4 + diagnostic.Samples = []DiagnosticSample{{Scope: "one", Message: "one"}, {Scope: "two", Message: "two"}, {Scope: "three", Message: "three"}, {Scope: "four", Message: "four"}} + diagnostic.OmittedSampleCount = 0 + }}, + {name: "duplicate samples", mutate: func(diagnostic *ProducerDiagnostic) { + diagnostic.OccurrenceCount = 2 + diagnostic.Samples = []DiagnosticSample{{Scope: "scope", Message: "message"}, {Scope: "scope", Message: "message"}} + diagnostic.OmittedSampleCount = 0 + }}, + {name: "inconsistent omission", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.OmittedSampleCount = 1 }}, + {name: "producer chunk context", mutate: func(diagnostic *ProducerDiagnostic) { chunkIndex := 0; diagnostic.Samples[0].ChunkIndex = &chunkIndex }}, + } { + t.Run(test.name, func(t *testing.T) { + diagnostic := validProducerDiagnostic() + test.mutate(&diagnostic) + if err := diagnostic.Validate(); err == nil { + t.Fatal("Validate() error = nil, want invalid diagnostic error") + } + }) + } +} + +func TestValidateProducerDiagnosticsEnforcesLocalGroupBound(t *testing.T) { + diagnostics := make([]ProducerDiagnostic, MaxProducerDiagnosticGroups) + for index := range diagnostics { + diagnostics[index] = validProducerDiagnostic() + diagnostics[index].ReasonCode = "reason-" + string(rune('a'+index)) + } + if err := ValidateProducerDiagnostics(diagnostics); err != nil { + t.Fatalf("ValidateProducerDiagnostics() error = %v", err) + } + diagnostics = append(diagnostics, validProducerDiagnostic()) + if err := ValidateProducerDiagnostics(diagnostics); err == nil { + t.Fatal("ValidateProducerDiagnostics() error = nil, want excessive-group error") + } +} + +func TestDiagnosticGroupValidationPreservesChunkIndexZero(t *testing.T) { + chunkIndex := 0 + group := DiagnosticGroup{ + Disposition: DiagnosticDispositionAdvisory, + Category: DiagnosticCategoryDataQuality, + ReasonCode: "unresolved", + Origin: DiagnosticOrigin{Stage: DiagnosticOriginStageExtract, StepID: "extract", LaneID: "spells", ModuleKey: "dnd/spells", ValidatorKey: "dnd/spells/source-relatedness"}, + OccurrenceCount: 1, + Samples: []DiagnosticSample{{Scope: "spells[0]", Message: "Spell was not found", ChunkID: "chunk-1", ChunkIndex: &chunkIndex}}, + } + if err := group.Validate(); err != nil { + t.Fatalf("Validate() error = %v", err) + } + collection := DiagnosticCollection{Groups: []DiagnosticGroup{group}} + if err := collection.Validate(); err != nil { + t.Fatalf("DiagnosticCollection.Validate() error = %v", err) + } +} + +func TestDiagnosticGroupRejectsRepeatedSampleWithEqualChunkIndex(t *testing.T) { + firstIndex := 0 + secondIndex := 0 + group := DiagnosticGroup{ + Disposition: DiagnosticDispositionAdvisory, + Category: DiagnosticCategoryDataQuality, + ReasonCode: "unresolved", + Origin: DiagnosticOrigin{Stage: DiagnosticOriginStageExtract, StepID: "extract", LaneID: "spells", ModuleKey: "dnd/spells"}, + OccurrenceCount: 2, + Samples: []DiagnosticSample{ + {Scope: "spells[0]", Message: "Spell was not found", ChunkID: "chunk-1", ChunkIndex: &firstIndex}, + {Scope: "spells[0]", Message: "Spell was not found", ChunkID: "chunk-1", ChunkIndex: &secondIndex}, + }, + } + if err := group.Validate(); err == nil { + t.Fatal("Validate() error = nil, want duplicate sample error") + } +} + +func validProducerDiagnostic() ProducerDiagnostic { + return ProducerDiagnostic{ + Disposition: DiagnosticDispositionWarning, + Category: DiagnosticCategoryConfiguration, + ReasonCode: "empty_reference", + OccurrenceCount: 1, + Samples: []DiagnosticSample{{Scope: "references.glossary", Message: "Reference is empty"}}, + } +} diff --git a/internal/framework/diagnostics/collector.go b/internal/framework/diagnostics/collector.go new file mode 100644 index 00000000..71a69923 --- /dev/null +++ b/internal/framework/diagnostics/collector.go @@ -0,0 +1,93 @@ +// Package diagnostics provides bounded local grouping for producer and +// validator diagnostic results. +package diagnostics + +import ( + "errors" + "fmt" + + "gitea.maximumdirect.net/eric/notarius/internal/framework/contracts" +) + +// Collector merges local producer diagnostics by their semantic identity. Its +// zero value is ready for use. +type Collector struct { + diagnostics []contracts.ProducerDiagnostic + indices map[key]int +} + +// NewCollector returns an empty local diagnostic collector. +func NewCollector() *Collector { + return &Collector{} +} + +// Add validates and merges one producer diagnostic. Every occurrence remains +// counted, while the first three distinct samples in input order are retained. +func (collector *Collector) Add(diagnostic contracts.ProducerDiagnostic) error { + if err := diagnostic.Validate(); err != nil { + return fmt.Errorf("producer diagnostic: %w", err) + } + if collector.indices == nil { + collector.indices = make(map[key]int) + } + diagnosticKey := key{disposition: diagnostic.Disposition, category: diagnostic.Category, reasonCode: diagnostic.ReasonCode} + index, exists := collector.indices[diagnosticKey] + if !exists { + if len(collector.diagnostics) >= contracts.MaxProducerDiagnosticGroups { + return errors.New("producer diagnostics exceed maximum group count") + } + collector.indices[diagnosticKey] = len(collector.diagnostics) + collector.diagnostics = append(collector.diagnostics, contracts.CloneProducerDiagnostics([]contracts.ProducerDiagnostic{diagnostic})[0]) + return nil + } + + current := &collector.diagnostics[index] + if diagnostic.OccurrenceCount > maximumInt()-current.OccurrenceCount { + return errors.New("producer diagnostic occurrence count overflow") + } + current.OccurrenceCount += diagnostic.OccurrenceCount + for _, sample := range diagnostic.Samples { + if len(current.Samples) == contracts.MaxDiagnosticSamples || containsSample(current.Samples, sample) { + continue + } + current.Samples = append(current.Samples, cloneSample(sample)) + } + current.OmittedSampleCount = current.OccurrenceCount - len(current.Samples) + return nil +} + +// Diagnostics returns an independently owned snapshot in first-occurrence +// order. +func (collector *Collector) Diagnostics() []contracts.ProducerDiagnostic { + if collector == nil { + return nil + } + return contracts.CloneProducerDiagnostics(collector.diagnostics) +} + +func containsSample(samples []contracts.DiagnosticSample, candidate contracts.DiagnosticSample) bool { + for _, sample := range samples { + if sample.Scope == candidate.Scope && sample.Message == candidate.Message { + return true + } + } + return false +} + +func cloneSample(sample contracts.DiagnosticSample) contracts.DiagnosticSample { + if sample.ChunkIndex != nil { + chunkIndex := *sample.ChunkIndex + sample.ChunkIndex = &chunkIndex + } + return sample +} + +func maximumInt() int { + return int(^uint(0) >> 1) +} + +type key struct { + disposition contracts.DiagnosticDisposition + category contracts.DiagnosticCategory + reasonCode string +} diff --git a/internal/framework/diagnostics/collector_test.go b/internal/framework/diagnostics/collector_test.go new file mode 100644 index 00000000..b8d6b97d --- /dev/null +++ b/internal/framework/diagnostics/collector_test.go @@ -0,0 +1,103 @@ +package diagnostics + +import ( + "testing" + + "gitea.maximumdirect.net/eric/notarius/internal/framework/contracts" +) + +func TestCollectorCountsOccurrencesAndRetainsDistinctSamplesInOrder(t *testing.T) { + collector := NewCollector() + for _, diagnostic := range []contracts.ProducerDiagnostic{ + advisory("one", "first"), + advisory("one", "first"), + advisory("two", "second"), + advisory("three", "third"), + advisory("four", "fourth"), + } { + if err := collector.Add(diagnostic); err != nil { + t.Fatalf("Add() error = %v", err) + } + } + + diagnostics := collector.Diagnostics() + if len(diagnostics) != 1 { + t.Fatalf("group count = %d, want 1", len(diagnostics)) + } + group := diagnostics[0] + if group.OccurrenceCount != 5 || group.OmittedSampleCount != 2 { + t.Fatalf("group counts = %#v, want five occurrences and two omitted samples", group) + } + if got := []string{group.Samples[0].Scope, group.Samples[1].Scope, group.Samples[2].Scope}; !equalStrings(got, []string{"one", "two", "three"}) { + t.Fatalf("sample order = %#v, want first three distinct samples", got) + } +} + +func TestCollectorSeparatesGroupsAndRejectsInvalidOrExcessiveGroups(t *testing.T) { + collector := NewCollector() + if err := collector.Add(advisory("one", "first")); err != nil { + t.Fatalf("Add() error = %v", err) + } + warning := advisory("two", "second") + warning.Disposition = contracts.DiagnosticDispositionWarning + warning.Category = contracts.DiagnosticCategoryFallback + if err := collector.Add(warning); err != nil { + t.Fatalf("Add() error = %v", err) + } + if got := len(collector.Diagnostics()); got != 2 { + t.Fatalf("group count = %d, want 2", got) + } + + invalid := advisory("bad", "bad") + invalid.ReasonCode = "" + if err := collector.Add(invalid); err == nil { + t.Fatal("Add() error = nil, want invalid diagnostic error") + } + + limited := NewCollector() + for index := 0; index < contracts.MaxProducerDiagnosticGroups; index++ { + diagnostic := advisory("scope", "message") + diagnostic.ReasonCode = "reason-" + string(rune('a'+index)) + if err := limited.Add(diagnostic); err != nil { + t.Fatalf("Add(%d) error = %v", index, err) + } + } + if err := limited.Add(advisory("overflow", "overflow")); err == nil { + t.Fatal("Add() error = nil, want local group limit error") + } +} + +func TestCollectorReturnsIndependentSnapshots(t *testing.T) { + collector := NewCollector() + if err := collector.Add(advisory("scope", "message")); err != nil { + t.Fatalf("Add() error = %v", err) + } + first := collector.Diagnostics() + first[0].Samples[0].Message = "changed" + second := collector.Diagnostics() + if second[0].Samples[0].Message != "message" { + t.Fatalf("collector snapshot changed = %#v", second) + } +} + +func advisory(scope string, message string) contracts.ProducerDiagnostic { + return contracts.ProducerDiagnostic{ + Disposition: contracts.DiagnosticDispositionAdvisory, + Category: contracts.DiagnosticCategoryDataQuality, + ReasonCode: "source_unrelated", + OccurrenceCount: 1, + Samples: []contracts.DiagnosticSample{{Scope: scope, Message: message}}, + } +} + +func equalStrings(left []string, right []string) bool { + if len(left) != len(right) { + return false + } + for index := range left { + if left[index] != right[index] { + return false + } + } + return true +}