test/app/core/text_cleanup.py

200 lines
11 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters!

This file contains ambiguous Unicode characters that may be confused with others in your current locale. If your use case is intentional and legitimate, you can safely ignore this warning. Use the Escape button to highlight these characters.

# -*- coding: utf-8 -*-
"""Conservative ASR text deduplication and filler cleanup."""
from __future__ import annotations
import re
from app.core.text_cleanup_lexicon import DISCOURSE_FILLER_TOKENS
from app.core.text_cleanup_lexicon import FILLER_TOKENS
from app.core.text_cleanup_lexicon import NUMERIC_STUTTER_CHARS
from app.core.text_cleanup_lexicon import REPEAT_COLLAPSIBLE_DISCOURSE_TOKENS
from app.core.text_cleanup_lexicon import SAFE_DOUBLE_WORDS
from app.core.text_cleanup_lexicon import TERMINAL_FILLER_TOKENS
_PREFIX_STUTTER_CHARS = "这那离超和跟在对把将又还上先后了的"
_PREFIX_STUTTER_PATTERN = re.compile(rf"([{re.escape(_PREFIX_STUTTER_CHARS)}])\1(?!\1)([\u4e00-\u9fffA-Za-z]{{1,3}})")
_REPEATED_PHRASE_PATTERN = re.compile(r"([\u4e00-\u9fffA-Za-z]{2,4})\1")
_WORD_STUTTER_PREFIX_PATTERN = re.compile(r"([\u4e00-\u9fff])\1{1,5}(?=\1[\u4e00-\u9fff])")
_LONG_CHAR_REPEAT_PATTERN = re.compile(r"([\u4e00-\u9fff])\1{2,}")
_INTERNAL_DOUBLE_CHAR_REPEAT_PATTERN = re.compile(r"(?<=[\u4e00-\u9fff])([\u4e00-\u9fff])\1(?=[\u4e00-\u9fff])")
_SHORT_LEADING_STUTTER_TOKEN_PATTERN = re.compile(r"(^|[,。!?;:、,\s])([\u4e00-\u9fff])\2([\u4e00-\u9fff])(?=($|[,。!?;:、,\s]))")
_SHORT_CONFIRMATION_STUTTER_PATTERN = re.compile(r"(^|[,。!?;:、,\s])([\u4e00-\u9fff])\2([\u4e00-\u9fff])(?=(是吧|对吧|对吗|对不对))")
_REPEATED_SEPARATED_PHRASE_PATTERN = re.compile(r"([\u4e00-\u9fffA-Za-z]{1,6})([,、,\s]+)\1(?:\2\1)*")
_REPEATED_SENTENCE_CLAUSE_PATTERN = re.compile(r"([\u4e00-\u9fffA-Za-z0-9]{2,12})([。!?;:]+)(?:\s*\1\2)+")
_REPEATED_SHORT_SENTENCE_CLAUSE_PATTERN = re.compile(
r"([\u4e00-\u9fff])([。!?;:]+)(?:\s*\1\2){2,}(?:\s*\1)?(?=$|[,。!?;:、,\s])"
)
_BOUNDARY_OVERLAP_MIN_CHARS = 2
_BOUNDARY_OVERLAP_MAX_CHARS = 12
_STANDALONE_FILLER_PATTERN = re.compile(rf"(^|[,。!?;:、,\s])({'|'.join(map(re.escape, FILLER_TOKENS))})(?=($|[,。!?;:、,\s]))")
_FILLER_ONLY_PATTERN = re.compile(rf"^[\s,。!?;:、,]*(?:{'|'.join(map(re.escape, FILLER_TOKENS))}[\s,。!?;:、,]*)+$")
_LEADING_FILLER_PREFIX_PATTERN = re.compile(rf"^(?:{'|'.join(map(re.escape, FILLER_TOKENS))})[,、,\s]*")
_INLINE_FILLER_PATTERN = re.compile(r"(?<=[\u4e00-\u9fffA-Za-z0-9])(嗯|呃|啊|(?<!金)额(?!度))(?=[\u4e00-\u9fffA-Za-z0-9])")
_REPEATED_DISCOURSE_FILLER_PATTERN = re.compile(
rf"({'|'.join(map(re.escape, REPEAT_COLLAPSIBLE_DISCOURSE_TOKENS))})(?:[,、,\s]*\1)+"
)
_STANDALONE_DISCOURSE_FILLER_PATTERN = re.compile(
rf"(^|[,。!?;:、,\s])({'|'.join(map(re.escape, DISCOURSE_FILLER_TOKENS))})(?=($|[,。!?;:、,\s]))"
)
_BRIDGE_FILLER_PATTERN = re.compile(r"(和|跟)(?:这个|那个)(和|跟)")
_REPEATED_TOPIC_WITH_DEICTIC_PATTERN = re.compile(r"([\u4e00-\u9fffA-Za-z]{2,6})(这个|那个)\1(?=[这那])")
_TERMINAL_FILLER_PATTERN = re.compile(
rf"(?<=[\u4e00-\u9fffA-Za-z0-9%])(?:{'|'.join(map(re.escape, TERMINAL_FILLER_TOKENS))})(?=($|[,。!?;:、,\s]))"
)
_POST_PUNCT_TERMINAL_FILLER_PATTERN = re.compile(
rf"(?<=[。!?;:,、,])(?:{'|'.join(map(re.escape, TERMINAL_FILLER_TOKENS))})(?=($|[,。!?;:、,\s]))"
)
_CURRENCY_STUTTER_PATTERN = re.compile(r"[¥¥]\s*\d{1,2}\s*[¥¥]\s*(\d{3,})(?=元?|[的档手]|$)")
_CURRENCY_SYMBOL_PATTERN = re.compile(r"[¥¥]\s*(\d+(?:\.\d+)?)")
_HOUSEHOLD_COLLECTION_GLUE_PATTERN = re.compile(r"(收集到)(\d{5,6})(户(?=的(?:一个)?清单))")
_NUMERIC_TOKEN_CHARS = r"0-9零〇○O一幺二两三四五六七八九十百千万亿点"
_REPEATED_NUMERIC_CLAUSE_PATTERN = re.compile(
rf"(?<![A-Za-z0-9])([{_NUMERIC_TOKEN_CHARS}]{{1,8}})([,。!?;:、,\s]+)(?:\1\2){{2,}}"
)
def _is_numeric_like(text: str) -> bool:
return bool(text) and all(character.isdigit() or character in "零〇○O一幺二两三四五六七八九十百千万亿点" for character in text)
def _has_meaningful_overlap(text: str) -> bool:
return bool(text and any(not character.isspace() and character not in ",。!?;:、,.!?;:" for character in text))
def _collapse_internal_double_char(match: re.Match[str]) -> str:
start_index = match.start()
source_text = match.string
pair_text = source_text[start_index:start_index + 2]
if match.group(1) in NUMERIC_STUTTER_CHARS:
return pair_text
if pair_text in SAFE_DOUBLE_WORDS:
return pair_text
return match.group(1)
def _remove_filler_words(text: str) -> str:
normalized_text = str(text or "").strip()
if not normalized_text:
return normalized_text
if _FILLER_ONLY_PATTERN.fullmatch(normalized_text):
return ""
compact_text = _LEADING_FILLER_PREFIX_PATTERN.sub("", normalized_text)
compact_text = _INLINE_FILLER_PATTERN.sub("", compact_text)
compact_text = _BRIDGE_FILLER_PATTERN.sub(lambda match: match.group(1) if match.group(1) == match.group(2) else match.group(2), compact_text)
compact_text = _REPEATED_TOPIC_WITH_DEICTIC_PATTERN.sub(lambda match: match.group(1), compact_text)
compact_text = _REPEATED_DISCOURSE_FILLER_PATTERN.sub(lambda match: match.group(1), compact_text)
compact_text = _STANDALONE_FILLER_PATTERN.sub(lambda match: match.group(1), compact_text)
compact_text = _STANDALONE_DISCOURSE_FILLER_PATTERN.sub(lambda match: match.group(1), compact_text)
compact_text = _TERMINAL_FILLER_PATTERN.sub("", compact_text)
compact_text = _POST_PUNCT_TERMINAL_FILLER_PATTERN.sub("", compact_text)
compact_text = re.sub(r"([。!?;:])[。!?;:]+", r"\1", compact_text)
compact_text = re.sub(r"([。!?;:])[,、,]+", r"\1", compact_text)
compact_text = re.sub(r"[,、,\s]{2,}", ",", compact_text)
compact_text = re.sub(r"^[,、,\s]+", "", compact_text)
compact_text = re.sub(r"[,、,\s]+([。!?;:])", r"\1", compact_text)
compact_text = re.sub(r"[,、,\s]+$", "", compact_text)
if not compact_text or re.fullmatch(r"[\s,。!?;:、,]*", compact_text):
return ""
return compact_text
def _repair_household_collection_count(match: re.Match[str]) -> str:
prefix, digits, suffix = match.groups()
following_context = match.string[match.end():match.end() + 100]
# Meeting reports often say "collected N households, estimated M households,
# close to 50%". If the glued count has a plausible suffix matching that
# ratio, keep the suffix and drop the noisy leading recognition artifact.
if "50" in following_context:
reference_counts = [
int(value)
for value in re.findall(r"(?:大概有|约|预计|测算[^,。!?;:]{0,10}?有)(\d{1,4})户", following_context)
]
for suffix_len in range(min(4, len(digits) - 1), 1, -1):
candidate = int(digits[-suffix_len:])
if candidate <= 0:
continue
if any(0.35 <= reference_count / candidate <= 0.65 for reference_count in reference_counts):
return f"{prefix}{candidate}{suffix}"
return match.group(0)
def _repair_numeric_asr_artifacts(text: str) -> str:
repaired_text = _CURRENCY_STUTTER_PATTERN.sub(lambda match: f"{match.group(1)}元", text)
repaired_text = _CURRENCY_SYMBOL_PATTERN.sub(lambda match: f"{match.group(1)}元", repaired_text)
repaired_text = _HOUSEHOLD_COLLECTION_GLUE_PATTERN.sub(_repair_household_collection_count, repaired_text)
return repaired_text
def _collapse_repeated_numeric_clauses(text: str) -> str:
return _REPEATED_NUMERIC_CLAUSE_PATTERN.sub(lambda match: f"{match.group(1)}{match.group(2)}", text)
def _collapse_repeated_short_sentence_clauses(text: str) -> str:
return _REPEATED_SHORT_SENTENCE_CLAUSE_PATTERN.sub(lambda match: f"{match.group(1)}{match.group(2)}", text)
def deduplicate_asr_text(text: str) -> str:
normalized_text = str(text or "").strip()
if not normalized_text:
return normalized_text
previous_text = None
while previous_text != normalized_text:
previous_text = normalized_text
normalized_text = _PREFIX_STUTTER_PATTERN.sub(lambda match: f"{match.group(1)}{match.group(2)}", normalized_text)
def _collapse_repeated_phrase(match: re.Match[str]) -> str:
phrase = match.group(1)
if _is_numeric_like(phrase):
return match.group(0)
return phrase
normalized_text = _REPEATED_PHRASE_PATTERN.sub(_collapse_repeated_phrase, normalized_text)
normalized_text = _WORD_STUTTER_PREFIX_PATTERN.sub(lambda match: match.group(0) if match.group(1) in NUMERIC_STUTTER_CHARS else "", normalized_text)
normalized_text = _LONG_CHAR_REPEAT_PATTERN.sub(lambda match: match.group(0) if match.group(1) in NUMERIC_STUTTER_CHARS else match.group(1), normalized_text)
normalized_text = _INTERNAL_DOUBLE_CHAR_REPEAT_PATTERN.sub(_collapse_internal_double_char, normalized_text)
normalized_text = _SHORT_LEADING_STUTTER_TOKEN_PATTERN.sub(lambda match: f"{match.group(1)}{match.group(2)}{match.group(3)}", normalized_text)
def _collapse_confirmation(match: re.Match[str]) -> str:
repeated_pair = f"{match.group(2)}{match.group(2)}"
if match.group(2) in NUMERIC_STUTTER_CHARS or repeated_pair in SAFE_DOUBLE_WORDS:
return match.group(0)
return f"{match.group(1)}{match.group(2)}{match.group(3)}"
normalized_text = _SHORT_CONFIRMATION_STUTTER_PATTERN.sub(_collapse_confirmation, normalized_text)
def _collapse_separated(match: re.Match[str]) -> str:
phrase = match.group(1)
if phrase.isascii() and len(phrase) == 1:
return match.group(0)
if _is_numeric_like(phrase):
return match.group(0)
return phrase
normalized_text = _REPEATED_SEPARATED_PHRASE_PATTERN.sub(_collapse_separated, normalized_text)
normalized_text = _REPEATED_SENTENCE_CLAUSE_PATTERN.sub(lambda match: match.group(0) if _is_numeric_like(match.group(1)) else f"{match.group(1)}{match.group(2)}", normalized_text)
normalized_text = _collapse_repeated_short_sentence_clauses(normalized_text)
normalized_text = _collapse_repeated_numeric_clauses(normalized_text)
normalized_text = _remove_filler_words(normalized_text)
normalized_text = _repair_numeric_asr_artifacts(normalized_text)
return normalized_text
def trim_segment_boundary_overlap(previous_text: str, current_text: str) -> tuple[str, bool]:
normalized_previous = str(previous_text or "").strip()
normalized_current = str(current_text or "").strip()
if not normalized_previous or not normalized_current:
return normalized_current, False
max_overlap_chars = min(len(normalized_previous), len(normalized_current), _BOUNDARY_OVERLAP_MAX_CHARS)
for overlap_chars in range(max_overlap_chars, _BOUNDARY_OVERLAP_MIN_CHARS - 1, -1):
overlap_suffix = normalized_previous[-overlap_chars:]
overlap_prefix = normalized_current[:overlap_chars]
if overlap_suffix != overlap_prefix:
continue
if not _has_meaningful_overlap(overlap_prefix):
continue
return normalized_current[overlap_chars:].lstrip(), True
return normalized_current, False