# -*- coding: utf-8 -*- """Conservative ASR text deduplication and filler cleanup.""" from __future__ import annotations import re from app.core.text_cleanup_lexicon import DISCOURSE_FILLER_TOKENS from app.core.text_cleanup_lexicon import FILLER_TOKENS from app.core.text_cleanup_lexicon import NUMERIC_STUTTER_CHARS from app.core.text_cleanup_lexicon import REPEAT_COLLAPSIBLE_DISCOURSE_TOKENS from app.core.text_cleanup_lexicon import SAFE_DOUBLE_WORDS from app.core.text_cleanup_lexicon import TERMINAL_FILLER_TOKENS _PREFIX_STUTTER_CHARS = "这那离超和跟在对把将又还上先后了的" _PREFIX_STUTTER_PATTERN = re.compile(rf"([{re.escape(_PREFIX_STUTTER_CHARS)}])\1(?!\1)([\u4e00-\u9fffA-Za-z]{{1,3}})") _REPEATED_PHRASE_PATTERN = re.compile(r"([\u4e00-\u9fffA-Za-z]{2,4})\1") _WORD_STUTTER_PREFIX_PATTERN = re.compile(r"([\u4e00-\u9fff])\1{1,5}(?=\1[\u4e00-\u9fff])") _LONG_CHAR_REPEAT_PATTERN = re.compile(r"([\u4e00-\u9fff])\1{2,}") _INTERNAL_DOUBLE_CHAR_REPEAT_PATTERN = re.compile(r"(?<=[\u4e00-\u9fff])([\u4e00-\u9fff])\1(?=[\u4e00-\u9fff])") _SHORT_LEADING_STUTTER_TOKEN_PATTERN = re.compile(r"(^|[,。!?;:、,\s])([\u4e00-\u9fff])\2([\u4e00-\u9fff])(?=($|[,。!?;:、,\s]))") _SHORT_CONFIRMATION_STUTTER_PATTERN = re.compile(r"(^|[,。!?;:、,\s])([\u4e00-\u9fff])\2([\u4e00-\u9fff])(?=(是吧|对吧|对吗|对不对))") _REPEATED_SEPARATED_PHRASE_PATTERN = re.compile(r"([\u4e00-\u9fffA-Za-z]{1,6})([,、,\s]+)\1(?:\2\1)*") _REPEATED_SENTENCE_CLAUSE_PATTERN = re.compile(r"([\u4e00-\u9fffA-Za-z0-9]{2,12})([。!?;:]+)(?:\s*\1\2)+") _REPEATED_SHORT_SENTENCE_CLAUSE_PATTERN = re.compile( r"([\u4e00-\u9fff])([。!?;:]+)(?:\s*\1\2){2,}(?:\s*\1)?(?=$|[,。!?;:、,\s])" ) _BOUNDARY_OVERLAP_MIN_CHARS = 2 _BOUNDARY_OVERLAP_MAX_CHARS = 12 _STANDALONE_FILLER_PATTERN = re.compile(rf"(^|[,。!?;:、,\s])({'|'.join(map(re.escape, FILLER_TOKENS))})(?=($|[,。!?;:、,\s]))") _FILLER_ONLY_PATTERN = re.compile(rf"^[\s,。!?;:、,]*(?:{'|'.join(map(re.escape, FILLER_TOKENS))}[\s,。!?;:、,]*)+$") _LEADING_FILLER_PREFIX_PATTERN = re.compile(rf"^(?:{'|'.join(map(re.escape, FILLER_TOKENS))})[,、,\s]*") _INLINE_FILLER_PATTERN = re.compile(r"(?<=[\u4e00-\u9fffA-Za-z0-9])(嗯|呃|啊|(? bool: return bool(text) and all(character.isdigit() or character in "零〇○O一幺二两三四五六七八九十百千万亿点" for character in text) def _has_meaningful_overlap(text: str) -> bool: return bool(text and any(not character.isspace() and character not in ",。!?;:、,.!?;:" for character in text)) def _collapse_internal_double_char(match: re.Match[str]) -> str: start_index = match.start() source_text = match.string pair_text = source_text[start_index:start_index + 2] if match.group(1) in NUMERIC_STUTTER_CHARS: return pair_text if pair_text in SAFE_DOUBLE_WORDS: return pair_text return match.group(1) def _remove_filler_words(text: str) -> str: normalized_text = str(text or "").strip() if not normalized_text: return normalized_text if _FILLER_ONLY_PATTERN.fullmatch(normalized_text): return "" compact_text = _LEADING_FILLER_PREFIX_PATTERN.sub("", normalized_text) compact_text = _INLINE_FILLER_PATTERN.sub("", compact_text) compact_text = _BRIDGE_FILLER_PATTERN.sub(lambda match: match.group(1) if match.group(1) == match.group(2) else match.group(2), compact_text) compact_text = _REPEATED_TOPIC_WITH_DEICTIC_PATTERN.sub(lambda match: match.group(1), compact_text) compact_text = _REPEATED_DISCOURSE_FILLER_PATTERN.sub(lambda match: match.group(1), compact_text) compact_text = _STANDALONE_FILLER_PATTERN.sub(lambda match: match.group(1), compact_text) compact_text = _STANDALONE_DISCOURSE_FILLER_PATTERN.sub(lambda match: match.group(1), compact_text) compact_text = _TERMINAL_FILLER_PATTERN.sub("", compact_text) compact_text = _POST_PUNCT_TERMINAL_FILLER_PATTERN.sub("", compact_text) compact_text = re.sub(r"([。!?;:])[。!?;:]+", r"\1", compact_text) compact_text = re.sub(r"([。!?;:])[,、,]+", r"\1", compact_text) compact_text = re.sub(r"[,、,\s]{2,}", ",", compact_text) compact_text = re.sub(r"^[,、,\s]+", "", compact_text) compact_text = re.sub(r"[,、,\s]+([。!?;:])", r"\1", compact_text) compact_text = re.sub(r"[,、,\s]+$", "", compact_text) if not compact_text or re.fullmatch(r"[\s,。!?;:、,]*", compact_text): return "" return compact_text def _repair_household_collection_count(match: re.Match[str]) -> str: prefix, digits, suffix = match.groups() following_context = match.string[match.end():match.end() + 100] # Meeting reports often say "collected N households, estimated M households, # close to 50%". If the glued count has a plausible suffix matching that # ratio, keep the suffix and drop the noisy leading recognition artifact. if "50" in following_context: reference_counts = [ int(value) for value in re.findall(r"(?:大概有|约|预计|测算[^,。!?;:]{0,10}?有)(\d{1,4})户", following_context) ] for suffix_len in range(min(4, len(digits) - 1), 1, -1): candidate = int(digits[-suffix_len:]) if candidate <= 0: continue if any(0.35 <= reference_count / candidate <= 0.65 for reference_count in reference_counts): return f"{prefix}{candidate}{suffix}" return match.group(0) def _repair_numeric_asr_artifacts(text: str) -> str: repaired_text = _CURRENCY_STUTTER_PATTERN.sub(lambda match: f"{match.group(1)}元", text) repaired_text = _CURRENCY_SYMBOL_PATTERN.sub(lambda match: f"{match.group(1)}元", repaired_text) repaired_text = _HOUSEHOLD_COLLECTION_GLUE_PATTERN.sub(_repair_household_collection_count, repaired_text) return repaired_text def _collapse_repeated_numeric_clauses(text: str) -> str: return _REPEATED_NUMERIC_CLAUSE_PATTERN.sub(lambda match: f"{match.group(1)}{match.group(2)}", text) def _collapse_repeated_short_sentence_clauses(text: str) -> str: return _REPEATED_SHORT_SENTENCE_CLAUSE_PATTERN.sub(lambda match: f"{match.group(1)}{match.group(2)}", text) def deduplicate_asr_text(text: str) -> str: normalized_text = str(text or "").strip() if not normalized_text: return normalized_text previous_text = None while previous_text != normalized_text: previous_text = normalized_text normalized_text = _PREFIX_STUTTER_PATTERN.sub(lambda match: f"{match.group(1)}{match.group(2)}", normalized_text) def _collapse_repeated_phrase(match: re.Match[str]) -> str: phrase = match.group(1) if _is_numeric_like(phrase): return match.group(0) return phrase normalized_text = _REPEATED_PHRASE_PATTERN.sub(_collapse_repeated_phrase, normalized_text) normalized_text = _WORD_STUTTER_PREFIX_PATTERN.sub(lambda match: match.group(0) if match.group(1) in NUMERIC_STUTTER_CHARS else "", normalized_text) normalized_text = _LONG_CHAR_REPEAT_PATTERN.sub(lambda match: match.group(0) if match.group(1) in NUMERIC_STUTTER_CHARS else match.group(1), normalized_text) normalized_text = _INTERNAL_DOUBLE_CHAR_REPEAT_PATTERN.sub(_collapse_internal_double_char, normalized_text) normalized_text = _SHORT_LEADING_STUTTER_TOKEN_PATTERN.sub(lambda match: f"{match.group(1)}{match.group(2)}{match.group(3)}", normalized_text) def _collapse_confirmation(match: re.Match[str]) -> str: repeated_pair = f"{match.group(2)}{match.group(2)}" if match.group(2) in NUMERIC_STUTTER_CHARS or repeated_pair in SAFE_DOUBLE_WORDS: return match.group(0) return f"{match.group(1)}{match.group(2)}{match.group(3)}" normalized_text = _SHORT_CONFIRMATION_STUTTER_PATTERN.sub(_collapse_confirmation, normalized_text) def _collapse_separated(match: re.Match[str]) -> str: phrase = match.group(1) if phrase.isascii() and len(phrase) == 1: return match.group(0) if _is_numeric_like(phrase): return match.group(0) return phrase normalized_text = _REPEATED_SEPARATED_PHRASE_PATTERN.sub(_collapse_separated, normalized_text) normalized_text = _REPEATED_SENTENCE_CLAUSE_PATTERN.sub(lambda match: match.group(0) if _is_numeric_like(match.group(1)) else f"{match.group(1)}{match.group(2)}", normalized_text) normalized_text = _collapse_repeated_short_sentence_clauses(normalized_text) normalized_text = _collapse_repeated_numeric_clauses(normalized_text) normalized_text = _remove_filler_words(normalized_text) normalized_text = _repair_numeric_asr_artifacts(normalized_text) return normalized_text def trim_segment_boundary_overlap(previous_text: str, current_text: str) -> tuple[str, bool]: normalized_previous = str(previous_text or "").strip() normalized_current = str(current_text or "").strip() if not normalized_previous or not normalized_current: return normalized_current, False max_overlap_chars = min(len(normalized_previous), len(normalized_current), _BOUNDARY_OVERLAP_MAX_CHARS) for overlap_chars in range(max_overlap_chars, _BOUNDARY_OVERLAP_MIN_CHARS - 1, -1): overlap_suffix = normalized_previous[-overlap_chars:] overlap_prefix = normalized_current[:overlap_chars] if overlap_suffix != overlap_prefix: continue if not _has_meaningful_overlap(overlap_prefix): continue return normalized_current[overlap_chars:].lstrip(), True return normalized_current, False