89 lines
5.1 KiB
Go
89 lines
5.1 KiB
Go
package spoken_word
|
|
|
|
import (
|
|
"encoding/json"
|
|
"fmt"
|
|
|
|
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
|
|
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
|
|
"gitea.maximumdirect.net/eric/audita/internal/framework/promptcontext"
|
|
)
|
|
|
|
type promptSegment struct {
|
|
ID int `json:"id"`
|
|
Speaker string `json:"speaker"`
|
|
Start float64 `json:"start"`
|
|
End float64 `json:"end"`
|
|
Text string `json:"text"`
|
|
Categories []string `json:"categories,omitempty"`
|
|
}
|
|
|
|
type promptTranscriptSection struct {
|
|
SectionIndex int `json:"section_index"`
|
|
Segments []promptSegment `json:"segments"`
|
|
}
|
|
|
|
// BuildProposalMessages constrains corrections to conservative dysfluency
|
|
// cleanup with strict semantic preservation.
|
|
func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
|
|
glossaryJSON, err := json.MarshalIndent(glossary, "", " ")
|
|
if err != nil {
|
|
return nil, fmt.Errorf("marshal glossary prompt context: %w", err)
|
|
}
|
|
|
|
sectionPayload := promptTranscriptSection{
|
|
SectionIndex: sectionIndex,
|
|
Segments: make([]promptSegment, 0),
|
|
}
|
|
if transcript != nil {
|
|
for _, s := range transcript.Segments {
|
|
sectionPayload.Segments = append(sectionPayload.Segments, promptSegment{
|
|
ID: s.ID,
|
|
Speaker: s.Speaker,
|
|
Start: s.Start,
|
|
End: s.End,
|
|
Text: s.Text,
|
|
Categories: append([]string(nil), s.Categories...),
|
|
})
|
|
}
|
|
}
|
|
sectionJSON, err := json.MarshalIndent(sectionPayload, "", " ")
|
|
if err != nil {
|
|
return nil, fmt.Errorf("marshal transcript prompt context: %w", err)
|
|
}
|
|
|
|
system := "You are Audita, a conservative spoken-word cleanup assistant. Identify only low-risk cleanup of repeated words or short phrases, filler words, hesitation artifacts, and similar dysfluencies that commonly appear in spoken English transcripts. Preserve substantive meaning, named entities, and transcript content."
|
|
user := "Review this transcript section and return only spoken-word cleanup corrections that should be applied.\n\n" +
|
|
"Rules:\n" +
|
|
"- Approve only conservative cleanup of repeated words, repeated short phrases, filler words, hesitation artifacts, and similar spoken dysfluencies.\n" +
|
|
"- You may collapse adjacent repetition such as \"I I think\" to \"I think\" or remove filler spans such as \"you know\" or \"uh\" when local context supports that cleanup.\n" +
|
|
"- Do not collapse repeated words or short phrases when the repetition plausibly expresses urgency, excitement, insistence, or deliberate rhetorical emphasis rather than dysfluency.\n" +
|
|
"- Phrases such as \"Help! Help! Help!\", \"Stop! Stop! Stop!\", \"No! No! No!\", \"Yes! Yes! Yes!\", and \"Go! Go! Go!\" are often intentional emphasis and should usually be preserved.\n" +
|
|
"- Only collapse repetition when local context supports it as accidental spoken repetition, hesitation, or verbal restart.\n" +
|
|
"- You may include low-risk punctuation, spacing, or capitalization cleanup when it is part of removing a dysfluency, such as removing ellipses or hesitation punctuation that no longer belongs after the cleanup.\n" +
|
|
"- Do not paraphrase, summarize, reorder ideas, replace content with different wording, or make substantive semantic edits.\n" +
|
|
"- Do not change clear content words just because a different phrasing reads better.\n" +
|
|
"- Do not convert uncertain statements into certain statements.\n" +
|
|
"- Treat glossary names and aliases as protected spellings and context.\n" +
|
|
"- Do not replace, Anglicize, normalize, lowercase, or otherwise alter protected glossary names or aliases that already appear correctly in the transcript.\n" +
|
|
"- Preserve canonical glossary capitalization for protected names and aliases, even if they look unusual.\n" +
|
|
"- If a segment includes categories, treat them as additional transcript context.\n" +
|
|
"- Use the exact id from the input segment.\n" +
|
|
"- For returned corrections, original_text must be only the exact text span that needs replacement, not the full segment text unless the whole segment is the replacement span.\n" +
|
|
"- Choose an original_text span that appears exactly once in the current segment text.\n" +
|
|
"- corrected_text must be only the replacement text for that span, not the full corrected segment text unless the whole segment is the replacement span.\n" +
|
|
"- Each returned correction must contain only id, original_text, corrected_text, and confidence.\n" +
|
|
"- Do not return corrections where original_text and corrected_text are identical.\n" +
|
|
"- Do not return speaker, start, or end fields.\n" +
|
|
"- Return only changed segments; do not return entries for unchanged segments.\n" +
|
|
"- confidence must be between 0.0 and 1.0.\n" +
|
|
"- If no corrections are needed, return an empty corrections list.\n\n" +
|
|
promptcontext.TranscriptDescriptionBlock(transcriptDescription) +
|
|
fmt.Sprintf("Protected glossary/context:\n%s\n\nTranscript section:\n%s", string(glossaryJSON), string(sectionJSON))
|
|
|
|
return []contracts.LLMMessage{
|
|
{Role: "system", Content: system},
|
|
{Role: "user", Content: user},
|
|
}, nil
|
|
}
|