200 lines
11 KiB
Python
200 lines
11 KiB
Python
# -*- coding: utf-8 -*-
|
||
"""Conservative ASR text deduplication and filler cleanup."""
|
||
|
||
from __future__ import annotations
|
||
|
||
import re
|
||
|
||
from app.core.text_cleanup_lexicon import DISCOURSE_FILLER_TOKENS
|
||
from app.core.text_cleanup_lexicon import FILLER_TOKENS
|
||
from app.core.text_cleanup_lexicon import NUMERIC_STUTTER_CHARS
|
||
from app.core.text_cleanup_lexicon import REPEAT_COLLAPSIBLE_DISCOURSE_TOKENS
|
||
from app.core.text_cleanup_lexicon import SAFE_DOUBLE_WORDS
|
||
from app.core.text_cleanup_lexicon import TERMINAL_FILLER_TOKENS
|
||
|
||
_PREFIX_STUTTER_CHARS = "这那离超和跟在对把将又还上先后了的"
|
||
_PREFIX_STUTTER_PATTERN = re.compile(rf"([{re.escape(_PREFIX_STUTTER_CHARS)}])\1(?!\1)([\u4e00-\u9fffA-Za-z]{{1,3}})")
|
||
_REPEATED_PHRASE_PATTERN = re.compile(r"([\u4e00-\u9fffA-Za-z]{2,4})\1")
|
||
_WORD_STUTTER_PREFIX_PATTERN = re.compile(r"([\u4e00-\u9fff])\1{1,5}(?=\1[\u4e00-\u9fff])")
|
||
_LONG_CHAR_REPEAT_PATTERN = re.compile(r"([\u4e00-\u9fff])\1{2,}")
|
||
_INTERNAL_DOUBLE_CHAR_REPEAT_PATTERN = re.compile(r"(?<=[\u4e00-\u9fff])([\u4e00-\u9fff])\1(?=[\u4e00-\u9fff])")
|
||
_SHORT_LEADING_STUTTER_TOKEN_PATTERN = re.compile(r"(^|[,。!?;:、,\s])([\u4e00-\u9fff])\2([\u4e00-\u9fff])(?=($|[,。!?;:、,\s]))")
|
||
_SHORT_CONFIRMATION_STUTTER_PATTERN = re.compile(r"(^|[,。!?;:、,\s])([\u4e00-\u9fff])\2([\u4e00-\u9fff])(?=(是吧|对吧|对吗|对不对))")
|
||
_REPEATED_SEPARATED_PHRASE_PATTERN = re.compile(r"([\u4e00-\u9fffA-Za-z]{1,6})([,、,\s]+)\1(?:\2\1)*")
|
||
_REPEATED_SENTENCE_CLAUSE_PATTERN = re.compile(r"([\u4e00-\u9fffA-Za-z0-9]{2,12})([。!?;:]+)(?:\s*\1\2)+")
|
||
_REPEATED_SHORT_SENTENCE_CLAUSE_PATTERN = re.compile(
|
||
r"([\u4e00-\u9fff])([。!?;:]+)(?:\s*\1\2){2,}(?:\s*\1)?(?=$|[,。!?;:、,\s])"
|
||
)
|
||
_BOUNDARY_OVERLAP_MIN_CHARS = 2
|
||
_BOUNDARY_OVERLAP_MAX_CHARS = 12
|
||
_STANDALONE_FILLER_PATTERN = re.compile(rf"(^|[,。!?;:、,\s])({'|'.join(map(re.escape, FILLER_TOKENS))})(?=($|[,。!?;:、,\s]))")
|
||
_FILLER_ONLY_PATTERN = re.compile(rf"^[\s,。!?;:、,]*(?:{'|'.join(map(re.escape, FILLER_TOKENS))}[\s,。!?;:、,]*)+$")
|
||
_LEADING_FILLER_PREFIX_PATTERN = re.compile(rf"^(?:{'|'.join(map(re.escape, FILLER_TOKENS))})[,、,\s]*")
|
||
_INLINE_FILLER_PATTERN = re.compile(r"(?<=[\u4e00-\u9fffA-Za-z0-9])(嗯|呃|啊|(?<!金)额(?!度))(?=[\u4e00-\u9fffA-Za-z0-9])")
|
||
_REPEATED_DISCOURSE_FILLER_PATTERN = re.compile(
|
||
rf"({'|'.join(map(re.escape, REPEAT_COLLAPSIBLE_DISCOURSE_TOKENS))})(?:[,、,\s]*\1)+"
|
||
)
|
||
_STANDALONE_DISCOURSE_FILLER_PATTERN = re.compile(
|
||
rf"(^|[,。!?;:、,\s])({'|'.join(map(re.escape, DISCOURSE_FILLER_TOKENS))})(?=($|[,。!?;:、,\s]))"
|
||
)
|
||
_BRIDGE_FILLER_PATTERN = re.compile(r"(和|跟)(?:这个|那个)(和|跟)")
|
||
_REPEATED_TOPIC_WITH_DEICTIC_PATTERN = re.compile(r"([\u4e00-\u9fffA-Za-z]{2,6})(这个|那个)\1(?=[这那])")
|
||
_TERMINAL_FILLER_PATTERN = re.compile(
|
||
rf"(?<=[\u4e00-\u9fffA-Za-z0-9%])(?:{'|'.join(map(re.escape, TERMINAL_FILLER_TOKENS))})(?=($|[,。!?;:、,\s]))"
|
||
)
|
||
_POST_PUNCT_TERMINAL_FILLER_PATTERN = re.compile(
|
||
rf"(?<=[。!?;:,、,])(?:{'|'.join(map(re.escape, TERMINAL_FILLER_TOKENS))})(?=($|[,。!?;:、,\s]))"
|
||
)
|
||
_CURRENCY_STUTTER_PATTERN = re.compile(r"[¥¥]\s*\d{1,2}\s*[¥¥]\s*(\d{3,})(?=元?|[的档手]|$)")
|
||
_CURRENCY_SYMBOL_PATTERN = re.compile(r"[¥¥]\s*(\d+(?:\.\d+)?)")
|
||
_HOUSEHOLD_COLLECTION_GLUE_PATTERN = re.compile(r"(收集到)(\d{5,6})(户(?=的(?:一个)?清单))")
|
||
_NUMERIC_TOKEN_CHARS = r"0-9零〇○O一幺二两三四五六七八九十百千万亿点"
|
||
_REPEATED_NUMERIC_CLAUSE_PATTERN = re.compile(
|
||
rf"(?<![A-Za-z0-9])([{_NUMERIC_TOKEN_CHARS}]{{1,8}})([,。!?;:、,\s]+)(?:\1\2){{2,}}"
|
||
)
|
||
|
||
|
||
def _is_numeric_like(text: str) -> bool:
|
||
return bool(text) and all(character.isdigit() or character in "零〇○O一幺二两三四五六七八九十百千万亿点" for character in text)
|
||
|
||
|
||
def _has_meaningful_overlap(text: str) -> bool:
|
||
return bool(text and any(not character.isspace() and character not in ",。!?;:、,.!?;:" for character in text))
|
||
|
||
|
||
def _collapse_internal_double_char(match: re.Match[str]) -> str:
|
||
start_index = match.start()
|
||
source_text = match.string
|
||
pair_text = source_text[start_index:start_index + 2]
|
||
if match.group(1) in NUMERIC_STUTTER_CHARS:
|
||
return pair_text
|
||
if pair_text in SAFE_DOUBLE_WORDS:
|
||
return pair_text
|
||
return match.group(1)
|
||
|
||
|
||
def _remove_filler_words(text: str) -> str:
|
||
normalized_text = str(text or "").strip()
|
||
if not normalized_text:
|
||
return normalized_text
|
||
if _FILLER_ONLY_PATTERN.fullmatch(normalized_text):
|
||
return ""
|
||
compact_text = _LEADING_FILLER_PREFIX_PATTERN.sub("", normalized_text)
|
||
compact_text = _INLINE_FILLER_PATTERN.sub("", compact_text)
|
||
compact_text = _BRIDGE_FILLER_PATTERN.sub(lambda match: match.group(1) if match.group(1) == match.group(2) else match.group(2), compact_text)
|
||
compact_text = _REPEATED_TOPIC_WITH_DEICTIC_PATTERN.sub(lambda match: match.group(1), compact_text)
|
||
compact_text = _REPEATED_DISCOURSE_FILLER_PATTERN.sub(lambda match: match.group(1), compact_text)
|
||
compact_text = _STANDALONE_FILLER_PATTERN.sub(lambda match: match.group(1), compact_text)
|
||
compact_text = _STANDALONE_DISCOURSE_FILLER_PATTERN.sub(lambda match: match.group(1), compact_text)
|
||
compact_text = _TERMINAL_FILLER_PATTERN.sub("", compact_text)
|
||
compact_text = _POST_PUNCT_TERMINAL_FILLER_PATTERN.sub("", compact_text)
|
||
compact_text = re.sub(r"([。!?;:])[。!?;:]+", r"\1", compact_text)
|
||
compact_text = re.sub(r"([。!?;:])[,、,]+", r"\1", compact_text)
|
||
compact_text = re.sub(r"[,、,\s]{2,}", ",", compact_text)
|
||
compact_text = re.sub(r"^[,、,\s]+", "", compact_text)
|
||
compact_text = re.sub(r"[,、,\s]+([。!?;:])", r"\1", compact_text)
|
||
compact_text = re.sub(r"[,、,\s]+$", "", compact_text)
|
||
if not compact_text or re.fullmatch(r"[\s,。!?;:、,]*", compact_text):
|
||
return ""
|
||
return compact_text
|
||
|
||
|
||
def _repair_household_collection_count(match: re.Match[str]) -> str:
|
||
prefix, digits, suffix = match.groups()
|
||
following_context = match.string[match.end():match.end() + 100]
|
||
|
||
# Meeting reports often say "collected N households, estimated M households,
|
||
# close to 50%". If the glued count has a plausible suffix matching that
|
||
# ratio, keep the suffix and drop the noisy leading recognition artifact.
|
||
if "50" in following_context:
|
||
reference_counts = [
|
||
int(value)
|
||
for value in re.findall(r"(?:大概有|约|预计|测算[^,。!?;:]{0,10}?有)(\d{1,4})户", following_context)
|
||
]
|
||
for suffix_len in range(min(4, len(digits) - 1), 1, -1):
|
||
candidate = int(digits[-suffix_len:])
|
||
if candidate <= 0:
|
||
continue
|
||
if any(0.35 <= reference_count / candidate <= 0.65 for reference_count in reference_counts):
|
||
return f"{prefix}{candidate}{suffix}"
|
||
|
||
return match.group(0)
|
||
|
||
|
||
def _repair_numeric_asr_artifacts(text: str) -> str:
|
||
repaired_text = _CURRENCY_STUTTER_PATTERN.sub(lambda match: f"{match.group(1)}元", text)
|
||
repaired_text = _CURRENCY_SYMBOL_PATTERN.sub(lambda match: f"{match.group(1)}元", repaired_text)
|
||
repaired_text = _HOUSEHOLD_COLLECTION_GLUE_PATTERN.sub(_repair_household_collection_count, repaired_text)
|
||
return repaired_text
|
||
|
||
|
||
def _collapse_repeated_numeric_clauses(text: str) -> str:
|
||
return _REPEATED_NUMERIC_CLAUSE_PATTERN.sub(lambda match: f"{match.group(1)}{match.group(2)}", text)
|
||
|
||
|
||
def _collapse_repeated_short_sentence_clauses(text: str) -> str:
|
||
return _REPEATED_SHORT_SENTENCE_CLAUSE_PATTERN.sub(lambda match: f"{match.group(1)}{match.group(2)}", text)
|
||
|
||
|
||
def deduplicate_asr_text(text: str) -> str:
|
||
normalized_text = str(text or "").strip()
|
||
if not normalized_text:
|
||
return normalized_text
|
||
previous_text = None
|
||
while previous_text != normalized_text:
|
||
previous_text = normalized_text
|
||
normalized_text = _PREFIX_STUTTER_PATTERN.sub(lambda match: f"{match.group(1)}{match.group(2)}", normalized_text)
|
||
|
||
def _collapse_repeated_phrase(match: re.Match[str]) -> str:
|
||
phrase = match.group(1)
|
||
if _is_numeric_like(phrase):
|
||
return match.group(0)
|
||
return phrase
|
||
|
||
normalized_text = _REPEATED_PHRASE_PATTERN.sub(_collapse_repeated_phrase, normalized_text)
|
||
normalized_text = _WORD_STUTTER_PREFIX_PATTERN.sub(lambda match: match.group(0) if match.group(1) in NUMERIC_STUTTER_CHARS else "", normalized_text)
|
||
normalized_text = _LONG_CHAR_REPEAT_PATTERN.sub(lambda match: match.group(0) if match.group(1) in NUMERIC_STUTTER_CHARS else match.group(1), normalized_text)
|
||
normalized_text = _INTERNAL_DOUBLE_CHAR_REPEAT_PATTERN.sub(_collapse_internal_double_char, normalized_text)
|
||
normalized_text = _SHORT_LEADING_STUTTER_TOKEN_PATTERN.sub(lambda match: f"{match.group(1)}{match.group(2)}{match.group(3)}", normalized_text)
|
||
|
||
def _collapse_confirmation(match: re.Match[str]) -> str:
|
||
repeated_pair = f"{match.group(2)}{match.group(2)}"
|
||
if match.group(2) in NUMERIC_STUTTER_CHARS or repeated_pair in SAFE_DOUBLE_WORDS:
|
||
return match.group(0)
|
||
return f"{match.group(1)}{match.group(2)}{match.group(3)}"
|
||
|
||
normalized_text = _SHORT_CONFIRMATION_STUTTER_PATTERN.sub(_collapse_confirmation, normalized_text)
|
||
|
||
def _collapse_separated(match: re.Match[str]) -> str:
|
||
phrase = match.group(1)
|
||
if phrase.isascii() and len(phrase) == 1:
|
||
return match.group(0)
|
||
if _is_numeric_like(phrase):
|
||
return match.group(0)
|
||
return phrase
|
||
|
||
normalized_text = _REPEATED_SEPARATED_PHRASE_PATTERN.sub(_collapse_separated, normalized_text)
|
||
normalized_text = _REPEATED_SENTENCE_CLAUSE_PATTERN.sub(lambda match: match.group(0) if _is_numeric_like(match.group(1)) else f"{match.group(1)}{match.group(2)}", normalized_text)
|
||
normalized_text = _collapse_repeated_short_sentence_clauses(normalized_text)
|
||
normalized_text = _collapse_repeated_numeric_clauses(normalized_text)
|
||
normalized_text = _remove_filler_words(normalized_text)
|
||
normalized_text = _repair_numeric_asr_artifacts(normalized_text)
|
||
return normalized_text
|
||
|
||
|
||
def trim_segment_boundary_overlap(previous_text: str, current_text: str) -> tuple[str, bool]:
|
||
normalized_previous = str(previous_text or "").strip()
|
||
normalized_current = str(current_text or "").strip()
|
||
if not normalized_previous or not normalized_current:
|
||
return normalized_current, False
|
||
max_overlap_chars = min(len(normalized_previous), len(normalized_current), _BOUNDARY_OVERLAP_MAX_CHARS)
|
||
for overlap_chars in range(max_overlap_chars, _BOUNDARY_OVERLAP_MIN_CHARS - 1, -1):
|
||
overlap_suffix = normalized_previous[-overlap_chars:]
|
||
overlap_prefix = normalized_current[:overlap_chars]
|
||
if overlap_suffix != overlap_prefix:
|
||
continue
|
||
if not _has_meaningful_overlap(overlap_prefix):
|
||
continue
|
||
return normalized_current[overlap_chars:].lstrip(), True
|
||
return normalized_current, False
|