Add token-bounded transcript chunking
This commit is contained in:
376
internal/core/chunking/sections.go
Normal file
376
internal/core/chunking/sections.go
Normal file
@@ -0,0 +1,376 @@
|
||||
package chunking
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
|
||||
)
|
||||
|
||||
// Section represents a contiguous chunk of transcript segments with metadata.
|
||||
type Section struct {
|
||||
// Index is the 0-based section index within the chunked transcript
|
||||
Index int `json:"section_index"`
|
||||
|
||||
// StartSegmentID is the ID of the first segment in this section
|
||||
StartSegmentID int `json:"start_segment_id"`
|
||||
|
||||
// EndSegmentID is the ID of the last segment in this section
|
||||
EndSegmentID int `json:"end_segment_id"`
|
||||
|
||||
// EstimatedTokens is the approximate token count for this section
|
||||
EstimatedTokens int `json:"estimated_tokens"`
|
||||
|
||||
// Segments contains the segments in this section, in order
|
||||
Segments []schema.Segment `json:"segments"`
|
||||
}
|
||||
|
||||
// ChunkingConfig holds configuration for transcript chunking.
|
||||
type ChunkingConfig struct {
|
||||
// MaxSectionTokens is the maximum allowed tokens per section
|
||||
MaxSectionTokens int
|
||||
|
||||
// MinSectionTokens is the minimum desired tokens per section (for balancing)
|
||||
MinSectionTokens int
|
||||
|
||||
// TargetSections is an optional target number of sections
|
||||
// If nil, chunking will optimize for max token bounds only
|
||||
TargetSections *int
|
||||
}
|
||||
|
||||
// Chunker performs deterministic chunking of normalized transcript segments.
|
||||
type Chunker struct {
|
||||
config ChunkingConfig
|
||||
estimator TokenEstimator
|
||||
}
|
||||
|
||||
// NewChunker creates a new chunker with the given configuration.
|
||||
func NewChunker(config ChunkingConfig) *Chunker {
|
||||
return &Chunker{
|
||||
config: config,
|
||||
estimator: NewSimpleTokenEstimator(),
|
||||
}
|
||||
}
|
||||
|
||||
// NewChunkerWithEstimator creates a new chunker with a custom estimator.
|
||||
func NewChunkerWithEstimator(config ChunkingConfig, estimator TokenEstimator) *Chunker {
|
||||
return &Chunker{
|
||||
config: config,
|
||||
estimator: estimator,
|
||||
}
|
||||
}
|
||||
|
||||
// ChunkTranscript divides a normalized transcript into contiguous token-bounded sections.
|
||||
// Returns an error if the target section count is impossible under the constraints.
|
||||
// The input transcript is never mutated.
|
||||
func (c *Chunker) ChunkTranscript(transcript *schema.Transcript) ([]Section, error) {
|
||||
if transcript == nil || len(transcript.Segments) == 0 {
|
||||
return []Section{}, nil
|
||||
}
|
||||
|
||||
// Validate configuration
|
||||
if c.config.MaxSectionTokens <= 0 {
|
||||
return nil, fmt.Errorf("max_section_tokens must be positive, got %d", c.config.MaxSectionTokens)
|
||||
}
|
||||
|
||||
if c.config.MinSectionTokens < 0 {
|
||||
return nil, fmt.Errorf("min_section_tokens must be non-negative, got %d", c.config.MinSectionTokens)
|
||||
}
|
||||
|
||||
if c.config.MinSectionTokens > c.config.MaxSectionTokens {
|
||||
return nil, fmt.Errorf("min_section_tokens (%d) cannot exceed max_section_tokens (%d)",
|
||||
c.config.MinSectionTokens, c.config.MaxSectionTokens)
|
||||
}
|
||||
|
||||
// Calculate token counts for each segment (deterministic)
|
||||
segmentTokens := make([]int, len(transcript.Segments))
|
||||
for i, seg := range transcript.Segments {
|
||||
segmentTokens[i] = c.estimator.EstimateTokens(seg.Text)
|
||||
}
|
||||
|
||||
// If target sections specified, use target-aware chunking
|
||||
if c.config.TargetSections != nil {
|
||||
target := *c.config.TargetSections
|
||||
|
||||
if target <= 0 {
|
||||
return nil, fmt.Errorf("target_sections must be positive, got %d", target)
|
||||
}
|
||||
|
||||
// Validate target is achievable
|
||||
if err := c.validateTargetSections(target, segmentTokens); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// Use target-aware chunking strategy
|
||||
return c.chunkWithTarget(transcript.Segments, segmentTokens, target)
|
||||
}
|
||||
|
||||
// No target: just respect max token bounds
|
||||
return c.createSectionsUnderMax(transcript.Segments, segmentTokens)
|
||||
}
|
||||
|
||||
// validateTargetSections checks if the target section count is achievable.
|
||||
func (c *Chunker) validateTargetSections(target int, segmentTokens []int) error {
|
||||
// Maximum possible sections: limited by segment count
|
||||
maxPossibleSections := len(segmentTokens)
|
||||
|
||||
if target > maxPossibleSections {
|
||||
return fmt.Errorf(
|
||||
"target_sections (%d) is impossible: cannot have more sections than segments (%d)",
|
||||
target, maxPossibleSections)
|
||||
}
|
||||
|
||||
// Minimum possible sections: each segment must fit within max bounds
|
||||
minPossibleSections := c.calculateMinPossibleSections(segmentTokens)
|
||||
|
||||
if target < minPossibleSections {
|
||||
return fmt.Errorf(
|
||||
"target_sections (%d) is impossible: need at least %d sections to respect max_section_tokens (%d)",
|
||||
target, minPossibleSections, c.config.MaxSectionTokens)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// chunkWithTarget creates sections aiming for a specific target count.
|
||||
func (c *Chunker) chunkWithTarget(segments []schema.Segment, segmentTokens []int, target int) ([]Section, error) {
|
||||
// Strategy:
|
||||
// 1. First, create sections respecting max bounds (greedy packing)
|
||||
// 2. If we have more sections than target, try to merge adjacent sections
|
||||
// 3. If we have fewer sections than target, split sections where possible
|
||||
|
||||
// Step 1: Create initial sections respecting max bounds
|
||||
sections, err := c.createSectionsUnderMax(segments, segmentTokens)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// Step 2: Adjust to reach target
|
||||
if len(sections) > target {
|
||||
// Need to merge sections
|
||||
sections = c.mergeSectionsToTarget(sections, target)
|
||||
} else if len(sections) < target {
|
||||
// Need to split sections - split the largest sections first
|
||||
sections = c.splitSectionsToTarget(sections, target, segmentTokens)
|
||||
}
|
||||
|
||||
// Verify we achieved the target
|
||||
if len(sections) != target {
|
||||
return nil, fmt.Errorf(
|
||||
"target_sections (%d) is impossible under current constraints (got %d sections)",
|
||||
target, len(sections))
|
||||
}
|
||||
|
||||
// Reindex sections
|
||||
for i := range sections {
|
||||
sections[i].Index = i
|
||||
}
|
||||
|
||||
return sections, nil
|
||||
}
|
||||
|
||||
// mergeSectionsToTarget merges adjacent sections to reduce count to target.
|
||||
func (c *Chunker) mergeSectionsToTarget(sections []Section, target int) []Section {
|
||||
for len(sections) > target {
|
||||
// Find the best pair to merge (smallest combined section that stays under max)
|
||||
bestIdx := -1
|
||||
bestTokens := -1
|
||||
|
||||
for i := 0; i < len(sections)-1; i++ {
|
||||
combinedTokens := sections[i].EstimatedTokens + sections[i+1].EstimatedTokens
|
||||
|
||||
// Can only merge if combined is under max
|
||||
if combinedTokens <= c.config.MaxSectionTokens {
|
||||
// Prefer smaller combined sections to keep sections balanced
|
||||
if bestIdx == -1 || combinedTokens < bestTokens {
|
||||
bestIdx = i
|
||||
bestTokens = combinedTokens
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// No valid merge found
|
||||
if bestIdx == -1 {
|
||||
break
|
||||
}
|
||||
|
||||
// Merge sections[bestIdx] and sections[bestIdx+1]
|
||||
merged := c.mergeTwoSections(sections[bestIdx], sections[bestIdx+1])
|
||||
|
||||
// Replace the two sections with the merged one
|
||||
newSections := make([]Section, 0, len(sections)-1)
|
||||
newSections = append(newSections, sections[:bestIdx]...)
|
||||
newSections = append(newSections, merged)
|
||||
if bestIdx+2 < len(sections) {
|
||||
newSections = append(newSections, sections[bestIdx+2:]...)
|
||||
}
|
||||
sections = newSections
|
||||
}
|
||||
|
||||
return sections
|
||||
}
|
||||
|
||||
// splitSectionsToTarget splits sections to increase count to target.
|
||||
func (c *Chunker) splitSectionsToTarget(sections []Section, target int, segmentTokens []int) []Section {
|
||||
// We can only split sections that have multiple segments
|
||||
// Split the largest sections first
|
||||
|
||||
for len(sections) < target {
|
||||
// Find the best section to split (largest section with multiple segments)
|
||||
bestIdx := -1
|
||||
bestTokenCount := -1
|
||||
|
||||
for i, sec := range sections {
|
||||
if len(sec.Segments) > 1 {
|
||||
if sec.EstimatedTokens > bestTokenCount {
|
||||
bestIdx = i
|
||||
bestTokenCount = sec.EstimatedTokens
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// No splittable section found
|
||||
if bestIdx == -1 {
|
||||
break
|
||||
}
|
||||
|
||||
// Split the section into two
|
||||
sec := sections[bestIdx]
|
||||
splitPoint := len(sec.Segments) / 2
|
||||
|
||||
// Ensure each half has at least one segment
|
||||
if splitPoint < 1 || splitPoint >= len(sec.Segments) {
|
||||
break
|
||||
}
|
||||
|
||||
// Calculate token counts for each half
|
||||
leftTokens := 0
|
||||
for i := 0; i < splitPoint; i++ {
|
||||
leftTokens += c.estimator.EstimateTokens(sec.Segments[i].Text)
|
||||
}
|
||||
rightTokens := sec.EstimatedTokens - leftTokens
|
||||
|
||||
// Create two new sections
|
||||
leftSection := Section{
|
||||
StartSegmentID: sec.Segments[0].ID,
|
||||
EndSegmentID: sec.Segments[splitPoint-1].ID,
|
||||
EstimatedTokens: leftTokens,
|
||||
Segments: sec.Segments[:splitPoint],
|
||||
}
|
||||
|
||||
rightSection := Section{
|
||||
StartSegmentID: sec.Segments[splitPoint].ID,
|
||||
EndSegmentID: sec.Segments[len(sec.Segments)-1].ID,
|
||||
EstimatedTokens: rightTokens,
|
||||
Segments: sec.Segments[splitPoint:],
|
||||
}
|
||||
|
||||
// Replace old section with two new ones
|
||||
newSections := make([]Section, 0, len(sections)+1)
|
||||
newSections = append(newSections, sections[:bestIdx]...)
|
||||
newSections = append(newSections, leftSection, rightSection)
|
||||
if bestIdx+1 < len(sections) {
|
||||
newSections = append(newSections, sections[bestIdx+1:]...)
|
||||
}
|
||||
sections = newSections
|
||||
}
|
||||
|
||||
return sections
|
||||
}
|
||||
|
||||
// createSectionsUnderMax creates initial sections respecting max token bounds.
|
||||
func (c *Chunker) createSectionsUnderMax(segments []schema.Segment, segmentTokens []int) ([]Section, error) {
|
||||
var sections []Section
|
||||
var currentSegments []schema.Segment
|
||||
currentTokens := 0
|
||||
|
||||
for i, seg := range segments {
|
||||
tokens := segmentTokens[i]
|
||||
|
||||
// If a single segment exceeds max tokens, it must be its own section
|
||||
if tokens > c.config.MaxSectionTokens {
|
||||
// Flush current section if any
|
||||
if len(currentSegments) > 0 {
|
||||
sections = append(sections, c.buildSection(len(sections), currentSegments, currentTokens))
|
||||
currentSegments = nil
|
||||
currentTokens = 0
|
||||
}
|
||||
// Add the oversized segment as its own section
|
||||
sections = append(sections, c.buildSection(len(sections), []schema.Segment{seg}, tokens))
|
||||
continue
|
||||
}
|
||||
|
||||
// Check if adding this segment would exceed max
|
||||
if currentTokens+tokens > c.config.MaxSectionTokens && len(currentSegments) > 0 {
|
||||
// Finish current section
|
||||
sections = append(sections, c.buildSection(len(sections), currentSegments, currentTokens))
|
||||
currentSegments = []schema.Segment{seg}
|
||||
currentTokens = tokens
|
||||
} else {
|
||||
// Add to current section
|
||||
currentSegments = append(currentSegments, seg)
|
||||
currentTokens += tokens
|
||||
}
|
||||
}
|
||||
|
||||
// Don't forget the last section
|
||||
if len(currentSegments) > 0 {
|
||||
sections = append(sections, c.buildSection(len(sections), currentSegments, currentTokens))
|
||||
}
|
||||
|
||||
return sections, nil
|
||||
}
|
||||
|
||||
// buildSection creates a Section from segments.
|
||||
func (c *Chunker) buildSection(index int, segments []schema.Segment, tokens int) Section {
|
||||
return Section{
|
||||
Index: index,
|
||||
StartSegmentID: segments[0].ID,
|
||||
EndSegmentID: segments[len(segments)-1].ID,
|
||||
EstimatedTokens: tokens,
|
||||
Segments: segments,
|
||||
}
|
||||
}
|
||||
|
||||
// calculateMinPossibleSections calculates the minimum number of sections needed
|
||||
// to ensure no section exceeds max tokens.
|
||||
func (c *Chunker) calculateMinPossibleSections(segmentTokens []int) int {
|
||||
sections := 0
|
||||
currentTokens := 0
|
||||
|
||||
for _, tokens := range segmentTokens {
|
||||
if tokens > c.config.MaxSectionTokens {
|
||||
// Each oversized segment needs its own section
|
||||
if currentTokens > 0 {
|
||||
sections++
|
||||
currentTokens = 0
|
||||
}
|
||||
sections++
|
||||
} else if currentTokens+tokens > c.config.MaxSectionTokens {
|
||||
sections++
|
||||
currentTokens = tokens
|
||||
} else {
|
||||
currentTokens += tokens
|
||||
}
|
||||
}
|
||||
|
||||
if currentTokens > 0 {
|
||||
sections++
|
||||
}
|
||||
|
||||
return sections
|
||||
}
|
||||
|
||||
// mergeTwoSections combines two adjacent sections.
|
||||
func (c *Chunker) mergeTwoSections(a, b Section) Section {
|
||||
mergedSegments := make([]schema.Segment, 0, len(a.Segments)+len(b.Segments))
|
||||
mergedSegments = append(mergedSegments, a.Segments...)
|
||||
mergedSegments = append(mergedSegments, b.Segments...)
|
||||
|
||||
return Section{
|
||||
Index: a.Index, // Will be reassigned
|
||||
StartSegmentID: a.StartSegmentID,
|
||||
EndSegmentID: b.EndSegmentID,
|
||||
EstimatedTokens: a.EstimatedTokens + b.EstimatedTokens,
|
||||
Segments: mergedSegments,
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user