More bugfixes in the glossary protection guard

This commit is contained in:
2026-04-23 10:51:47 -05:00
parent 671188aaba
commit dee6ae4067
3 changed files with 161 additions and 20 deletions

View File

@@ -30,43 +30,72 @@ class ProtectedVocabulary:
return cls(terms_by_folded=terms_by_folded, pattern=pattern)
def violation_reason(self, before: str, after: str) -> Optional[str]:
before_terms = self._terms(before)
after_terms = self._terms(after)
before_occurrences = self._occurrences_by_identity(before)
after_occurrences = self._occurrences_by_identity(after)
if before_terms:
after_identities = [term.identity for term in after_terms]
for before_term in before_terms:
if before_term.identity not in after_identities:
return "correction changes protected glossary term usage"
reason = self._validate_identity_preservation(before_occurrences, after_occurrences)
if reason is not None:
return reason
return self._validate_capitalization_transitions(before_occurrences, after_occurrences)
for after_term in after_terms:
if after_term.text != after_term.canonical:
return "correction changes protected glossary term capitalization"
return None
def _terms(self, text: str) -> List["_ProtectedTerm"]:
def _occurrences(self, text: str) -> List["_ProtectedOccurrence"]:
if self.pattern is None:
return []
terms = []
occurrences = []
for match in self.pattern.finditer(text):
matched_text = match.group(0)
definition = self.terms_by_folded[matched_text.casefold()]
terms.append(
_ProtectedTerm(
occurrences.append(
_ProtectedOccurrence(
text=matched_text,
identity=definition.identity,
canonical=definition.canonical,
)
)
return terms
return occurrences
def _occurrences_by_identity(self, text: str) -> Dict[int, List["_ProtectedOccurrence"]]:
occurrences_by_identity: Dict[int, List["_ProtectedOccurrence"]] = {}
for occurrence in self._occurrences(text):
occurrences_by_identity.setdefault(occurrence.identity, []).append(occurrence)
return occurrences_by_identity
def _validate_identity_preservation(
self,
before_occurrences: Dict[int, List["_ProtectedOccurrence"]],
after_occurrences: Dict[int, List["_ProtectedOccurrence"]],
) -> Optional[str]:
for identity, before_items in before_occurrences.items():
if len(after_occurrences.get(identity, [])) < len(before_items):
return "correction changes protected glossary term usage"
return None
def _validate_capitalization_transitions(
self,
before_occurrences: Dict[int, List["_ProtectedOccurrence"]],
after_occurrences: Dict[int, List["_ProtectedOccurrence"]],
) -> Optional[str]:
for identity, after_items in after_occurrences.items():
before_items = before_occurrences.get(identity, [])
before_count = len(before_items)
for index, after_item in enumerate(after_items):
if index < before_count:
before_item = before_items[index]
if after_item.text == before_item.text:
continue
if after_item.text == after_item.canonical:
continue
return "correction changes protected glossary term capitalization"
if after_item.text != after_item.canonical:
return "correction changes protected glossary term capitalization"
return None
def contains_term(self, text: str) -> bool:
return bool(self._terms(text))
return bool(self._occurrences(text))
@dataclass(frozen=True)
class _ProtectedTerm:
class _ProtectedOccurrence:
text: str
identity: int
canonical: str

View File

@@ -1042,6 +1042,74 @@ def test_grammar_stage_can_correct_toward_protected_term(tmp_path):
assert list((tmp_path / "work").iterdir()) == []
def test_grammar_stage_allows_quote_wrapping_with_unchanged_lowercase_protected_term(tmp_path):
transcript = parse_source_transcript_json(
"""
[
{
"speaker": "Eric",
"start": 0.0,
"end": 1.0,
"text": "When you say that, Popov will say, when I was in that room with the jesters, I just knew that Godfrey and Lyra came directly from Loviator herself. They're really powerful."
}
]
"""
)
glossary = parse_glossary_yaml(
"""
glossary:
- name: "Popov"
category: npc
summary: "Popov is an allied NPC."
- name: "Jesters"
aliases:
- "Jester"
category: faction
summary: "The Jesters are a faction."
- name: "Godfrey"
category: npc
summary: "Godfrey is an NPC."
- name: "Lyra"
category: npc
summary: "Lyra is an NPC."
- name: "Loviator"
category: deity
summary: "Loviator is a deity."
"""
)
original_text = (
"When you say that, Popov will say, when I was in that room with the jesters, "
"I just knew that Godfrey and Lyra came directly from Loviator herself. They're really powerful."
)
corrected_text = (
'When you say that, Popov will say, "When I was in that room with the jesters, '
'I just knew that Godfrey and Lyra came directly from Loviator herself. '
'They\'re really powerful."'
)
grammar_correction = CorrectionCandidate(
id=1,
original_text=original_text,
corrected_text=corrected_text,
confidence=0.95,
)
fake_client = FakeLLMClient(
[
CorrectionSet(corrections=[]),
CorrectionSet(corrections=[grammar_correction]),
]
)
revised = process_transcript(
transcript,
glossary,
_config(tmp_path),
llm_client=fake_client,
)
assert revised[0].text == corrected_text
assert list((tmp_path / "work").iterdir()) == []
def test_grammar_stage_cannot_change_away_from_protected_term(tmp_path):
transcript = parse_source_transcript_json(
"""

View File

@@ -25,6 +25,12 @@ def _vocabulary():
- name: "Godfrey"
category: npc
summary: "Godfrey is an NPC."
- name: "Lyra"
category: npc
summary: "Lyra is an NPC."
- name: "Loviator"
category: deity
summary: "Loviator is a deity."
"""
)
return ProtectedVocabulary.from_glossary(glossary)
@@ -66,6 +72,12 @@ def test_protection_allows_canonical_capitalization():
assert vocabulary.violation_reason("hrank moves.", "Hrank moves.") is None
def test_protection_allows_unchanged_noncanonical_protected_term():
vocabulary = _vocabulary()
assert vocabulary.violation_reason("jesters advance.", "jesters advance.") is None
def test_protection_allows_correction_toward_protected_term():
vocabulary = _vocabulary()
@@ -98,6 +110,15 @@ def test_protection_blocks_noncanonical_introduced_protected_term():
)
def test_protection_blocks_changed_noncanonical_variant():
vocabulary = _vocabulary()
assert (
vocabulary.violation_reason("jesters advance.", "JESTERS advance.")
== "correction changes protected glossary term capitalization"
)
def test_protection_allows_inferred_name_plural():
vocabulary = _vocabulary()
@@ -135,6 +156,20 @@ def test_protection_allows_punctuation_around_protected_term():
assert vocabulary.violation_reason("Popov, moves.", "Popov. Moves.") is None
def test_protection_allows_quote_wrapping_sentence_with_unchanged_lowercase_protected_term():
vocabulary = _vocabulary()
before = (
"When you say that, Popov will say, when I was in that room with the jesters, "
"I just knew that Godfrey and Lyra came directly from Loviator herself."
)
after = (
'When you say that, Popov will say, "When I was in that room with the jesters, '
'I just knew that Godfrey and Lyra came directly from Loviator herself."'
)
assert vocabulary.violation_reason(before, after) is None
def test_protection_does_not_match_terms_inside_larger_words():
vocabulary = _vocabulary()
@@ -145,3 +180,12 @@ def test_protection_applies_to_aliases():
vocabulary = _vocabulary()
assert vocabulary.violation_reason("Greenfield waits.", "greenfield waits.") is not None
def test_protection_blocks_removing_preexisting_protected_occurrence():
vocabulary = _vocabulary()
assert (
vocabulary.violation_reason("Jesters flank the Jesters.", "Jesters flank the gestures.")
== "correction changes protected glossary term usage"
)