Files
audita/internal/core/chunking/sections.go

377 lines
11 KiB
Go

package chunking
import (
"fmt"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
)
// Section represents a contiguous chunk of transcript segments with metadata.
type Section struct {
// Index is the 0-based section index within the chunked transcript
Index int `json:"section_index"`
// StartSegmentID is the ID of the first segment in this section
StartSegmentID int `json:"start_segment_id"`
// EndSegmentID is the ID of the last segment in this section
EndSegmentID int `json:"end_segment_id"`
// EstimatedTokens is the approximate token count for this section
EstimatedTokens int `json:"estimated_tokens"`
// Segments contains the segments in this section, in order
Segments []schema.Segment `json:"segments"`
}
// ChunkingConfig holds configuration for transcript chunking.
type ChunkingConfig struct {
// MaxSectionTokens is the maximum allowed tokens per section
MaxSectionTokens int
// MinSectionTokens is the minimum desired tokens per section (for balancing)
MinSectionTokens int
// TargetSections is an optional target number of sections
// If nil, chunking will optimize for max token bounds only
TargetSections *int
}
// Chunker performs deterministic chunking of normalized transcript segments.
type Chunker struct {
config ChunkingConfig
estimator TokenEstimator
}
// NewChunker creates a new chunker with the given configuration.
func NewChunker(config ChunkingConfig) *Chunker {
return &Chunker{
config: config,
estimator: NewSimpleTokenEstimator(),
}
}
// NewChunkerWithEstimator creates a new chunker with a custom estimator.
func NewChunkerWithEstimator(config ChunkingConfig, estimator TokenEstimator) *Chunker {
return &Chunker{
config: config,
estimator: estimator,
}
}
// ChunkTranscript divides a normalized transcript into contiguous token-bounded sections.
// Returns an error if the target section count is impossible under the constraints.
// The input transcript is never mutated.
func (c *Chunker) ChunkTranscript(transcript *schema.Transcript) ([]Section, error) {
if transcript == nil || len(transcript.Segments) == 0 {
return []Section{}, nil
}
// Validate configuration
if c.config.MaxSectionTokens <= 0 {
return nil, fmt.Errorf("max_section_tokens must be positive, got %d", c.config.MaxSectionTokens)
}
if c.config.MinSectionTokens < 0 {
return nil, fmt.Errorf("min_section_tokens must be non-negative, got %d", c.config.MinSectionTokens)
}
if c.config.MinSectionTokens > c.config.MaxSectionTokens {
return nil, fmt.Errorf("min_section_tokens (%d) cannot exceed max_section_tokens (%d)",
c.config.MinSectionTokens, c.config.MaxSectionTokens)
}
// Calculate token counts for each segment (deterministic)
segmentTokens := make([]int, len(transcript.Segments))
for i, seg := range transcript.Segments {
segmentTokens[i] = c.estimator.EstimateTokens(seg.Text)
}
// If target sections specified, use target-aware chunking
if c.config.TargetSections != nil {
target := *c.config.TargetSections
if target <= 0 {
return nil, fmt.Errorf("target_sections must be positive, got %d", target)
}
// Validate target is achievable
if err := c.validateTargetSections(target, segmentTokens); err != nil {
return nil, err
}
// Use target-aware chunking strategy
return c.chunkWithTarget(transcript.Segments, segmentTokens, target)
}
// No target: just respect max token bounds
return c.createSectionsUnderMax(transcript.Segments, segmentTokens)
}
// validateTargetSections checks if the target section count is achievable.
func (c *Chunker) validateTargetSections(target int, segmentTokens []int) error {
// Maximum possible sections: limited by segment count
maxPossibleSections := len(segmentTokens)
if target > maxPossibleSections {
return fmt.Errorf(
"target_sections (%d) is impossible: cannot have more sections than segments (%d)",
target, maxPossibleSections)
}
// Minimum possible sections: each segment must fit within max bounds
minPossibleSections := c.calculateMinPossibleSections(segmentTokens)
if target < minPossibleSections {
return fmt.Errorf(
"target_sections (%d) is impossible: need at least %d sections to respect max_section_tokens (%d)",
target, minPossibleSections, c.config.MaxSectionTokens)
}
return nil
}
// chunkWithTarget creates sections aiming for a specific target count.
func (c *Chunker) chunkWithTarget(segments []schema.Segment, segmentTokens []int, target int) ([]Section, error) {
// Strategy:
// 1. First, create sections respecting max bounds (greedy packing)
// 2. If we have more sections than target, try to merge adjacent sections
// 3. If we have fewer sections than target, split sections where possible
// Step 1: Create initial sections respecting max bounds
sections, err := c.createSectionsUnderMax(segments, segmentTokens)
if err != nil {
return nil, err
}
// Step 2: Adjust to reach target
if len(sections) > target {
// Need to merge sections
sections = c.mergeSectionsToTarget(sections, target)
} else if len(sections) < target {
// Need to split sections - split the largest sections first
sections = c.splitSectionsToTarget(sections, target, segmentTokens)
}
// Verify we achieved the target
if len(sections) != target {
return nil, fmt.Errorf(
"target_sections (%d) is impossible under current constraints (got %d sections)",
target, len(sections))
}
// Reindex sections
for i := range sections {
sections[i].Index = i
}
return sections, nil
}
// mergeSectionsToTarget merges adjacent sections to reduce count to target.
func (c *Chunker) mergeSectionsToTarget(sections []Section, target int) []Section {
for len(sections) > target {
// Find the best pair to merge (smallest combined section that stays under max)
bestIdx := -1
bestTokens := -1
for i := 0; i < len(sections)-1; i++ {
combinedTokens := sections[i].EstimatedTokens + sections[i+1].EstimatedTokens
// Can only merge if combined is under max
if combinedTokens <= c.config.MaxSectionTokens {
// Prefer smaller combined sections to keep sections balanced
if bestIdx == -1 || combinedTokens < bestTokens {
bestIdx = i
bestTokens = combinedTokens
}
}
}
// No valid merge found
if bestIdx == -1 {
break
}
// Merge sections[bestIdx] and sections[bestIdx+1]
merged := c.mergeTwoSections(sections[bestIdx], sections[bestIdx+1])
// Replace the two sections with the merged one
newSections := make([]Section, 0, len(sections)-1)
newSections = append(newSections, sections[:bestIdx]...)
newSections = append(newSections, merged)
if bestIdx+2 < len(sections) {
newSections = append(newSections, sections[bestIdx+2:]...)
}
sections = newSections
}
return sections
}
// splitSectionsToTarget splits sections to increase count to target.
func (c *Chunker) splitSectionsToTarget(sections []Section, target int, segmentTokens []int) []Section {
// We can only split sections that have multiple segments
// Split the largest sections first
for len(sections) < target {
// Find the best section to split (largest section with multiple segments)
bestIdx := -1
bestTokenCount := -1
for i, sec := range sections {
if len(sec.Segments) > 1 {
if sec.EstimatedTokens > bestTokenCount {
bestIdx = i
bestTokenCount = sec.EstimatedTokens
}
}
}
// No splittable section found
if bestIdx == -1 {
break
}
// Split the section into two
sec := sections[bestIdx]
splitPoint := len(sec.Segments) / 2
// Ensure each half has at least one segment
if splitPoint < 1 || splitPoint >= len(sec.Segments) {
break
}
// Calculate token counts for each half
leftTokens := 0
for i := 0; i < splitPoint; i++ {
leftTokens += c.estimator.EstimateTokens(sec.Segments[i].Text)
}
rightTokens := sec.EstimatedTokens - leftTokens
// Create two new sections
leftSection := Section{
StartSegmentID: sec.Segments[0].ID,
EndSegmentID: sec.Segments[splitPoint-1].ID,
EstimatedTokens: leftTokens,
Segments: sec.Segments[:splitPoint],
}
rightSection := Section{
StartSegmentID: sec.Segments[splitPoint].ID,
EndSegmentID: sec.Segments[len(sec.Segments)-1].ID,
EstimatedTokens: rightTokens,
Segments: sec.Segments[splitPoint:],
}
// Replace old section with two new ones
newSections := make([]Section, 0, len(sections)+1)
newSections = append(newSections, sections[:bestIdx]...)
newSections = append(newSections, leftSection, rightSection)
if bestIdx+1 < len(sections) {
newSections = append(newSections, sections[bestIdx+1:]...)
}
sections = newSections
}
return sections
}
// createSectionsUnderMax creates initial sections respecting max token bounds.
func (c *Chunker) createSectionsUnderMax(segments []schema.Segment, segmentTokens []int) ([]Section, error) {
var sections []Section
var currentSegments []schema.Segment
currentTokens := 0
for i, seg := range segments {
tokens := segmentTokens[i]
// If a single segment exceeds max tokens, it must be its own section
if tokens > c.config.MaxSectionTokens {
// Flush current section if any
if len(currentSegments) > 0 {
sections = append(sections, c.buildSection(len(sections), currentSegments, currentTokens))
currentSegments = nil
currentTokens = 0
}
// Add the oversized segment as its own section
sections = append(sections, c.buildSection(len(sections), []schema.Segment{seg}, tokens))
continue
}
// Check if adding this segment would exceed max
if currentTokens+tokens > c.config.MaxSectionTokens && len(currentSegments) > 0 {
// Finish current section
sections = append(sections, c.buildSection(len(sections), currentSegments, currentTokens))
currentSegments = []schema.Segment{seg}
currentTokens = tokens
} else {
// Add to current section
currentSegments = append(currentSegments, seg)
currentTokens += tokens
}
}
// Don't forget the last section
if len(currentSegments) > 0 {
sections = append(sections, c.buildSection(len(sections), currentSegments, currentTokens))
}
return sections, nil
}
// buildSection creates a Section from segments.
func (c *Chunker) buildSection(index int, segments []schema.Segment, tokens int) Section {
return Section{
Index: index,
StartSegmentID: segments[0].ID,
EndSegmentID: segments[len(segments)-1].ID,
EstimatedTokens: tokens,
Segments: segments,
}
}
// calculateMinPossibleSections calculates the minimum number of sections needed
// to ensure no section exceeds max tokens.
func (c *Chunker) calculateMinPossibleSections(segmentTokens []int) int {
sections := 0
currentTokens := 0
for _, tokens := range segmentTokens {
if tokens > c.config.MaxSectionTokens {
// Each oversized segment needs its own section
if currentTokens > 0 {
sections++
currentTokens = 0
}
sections++
} else if currentTokens+tokens > c.config.MaxSectionTokens {
sections++
currentTokens = tokens
} else {
currentTokens += tokens
}
}
if currentTokens > 0 {
sections++
}
return sections
}
// mergeTwoSections combines two adjacent sections.
func (c *Chunker) mergeTwoSections(a, b Section) Section {
mergedSegments := make([]schema.Segment, 0, len(a.Segments)+len(b.Segments))
mergedSegments = append(mergedSegments, a.Segments...)
mergedSegments = append(mergedSegments, b.Segments...)
return Section{
Index: a.Index, // Will be reassigned
StartSegmentID: a.StartSegmentID,
EndSegmentID: b.EndSegmentID,
EstimatedTokens: a.EstimatedTokens + b.EstimatedTokens,
Segments: mergedSegments,
}
}