Moved the current implementation to src/audita_prototype, moved current test suite to tests/audita_prototype, and started a new, more modular application skeleton in src/audita
This commit is contained in:
27
tests/test_framework_chunking.py
Normal file
27
tests/test_framework_chunking.py
Normal file
@@ -0,0 +1,27 @@
|
||||
from audita.core.chunking import TokenEstimatorProtocol, chunk_transcript
|
||||
from audita.core.schemas import parse_transcript_json
|
||||
|
||||
|
||||
class FakeEstimator(TokenEstimatorProtocol):
|
||||
def estimate_json(self, value):
|
||||
if len(value) == 1:
|
||||
return 4
|
||||
return len(value) * 4
|
||||
|
||||
|
||||
def test_chunk_transcript_batches_sections_by_token_limit():
|
||||
transcript = parse_transcript_json(
|
||||
"""
|
||||
[
|
||||
{"id": 1, "speaker": "A", "start": 0.0, "end": 1.0, "text": "one"},
|
||||
{"id": 2, "speaker": "A", "start": 1.0, "end": 2.0, "text": "two"},
|
||||
{"id": 3, "speaker": "A", "start": 2.0, "end": 3.0, "text": "three"}
|
||||
]
|
||||
"""
|
||||
)
|
||||
|
||||
sections = chunk_transcript(transcript, max_section_tokens=8, estimator=FakeEstimator())
|
||||
|
||||
assert len(sections) == 2
|
||||
assert [segment.segment.id for segment in sections[0].segments] == [1, 2]
|
||||
assert [segment.segment.id for segment in sections[1].segments] == [3]
|
||||
Reference in New Issue
Block a user