test/app/utils/text_processing.py

60 lines
1.6 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters!

This file contains ambiguous Unicode characters that may be confused with others in your current locale. If your use case is intentional and legitimate, you can safely ignore this warning. Use the Escape button to highlight these characters.

# -*- coding: utf-8 -*-
"""
基于itntext的ITN(逆文本标准化)工具模块
使用itntext库提供高质量的中文ITN处理
"""
import logging
logger = logging.getLogger(__name__)
# itntext导入 - 延迟导入以避免初始化问题
_itntext_normalizer = None
def _get_normalizer():
"""获取itntext标准化器实例(单例模式)"""
global _itntext_normalizer
if _itntext_normalizer is None:
try:
from itntext import Normalizer
_itntext_normalizer = Normalizer(lang="zh", operator="itn")
logger.info("itntext ITN模块初始化成功")
except ImportError as e:
logger.error(f"导入itntext失败: {e}")
raise ImportError("请安装itntext库: pip install itntext")
except Exception as e:
logger.error(f"初始化itntext失败: {e}")
raise
return _itntext_normalizer
def apply_itn_to_text(text: str) -> str:
"""
对文本应用逆文本标准化(ITN)
使用itntext库进行高质量的中文ITN处理
Args:
text: 语音识别结果文本
Returns:
应用ITN后的文本
"""
if not text or not text.strip():
return text
try:
normalizer = _get_normalizer()
result = normalizer.normalize(text)
logger.debug(f"ITN处理: '{text}' -> '{result}'")
return result
except Exception as e:
logger.warning(f"ITN处理失败: {text}, 错误: {str(e)}")
return text
def normalize_asr_text(text: str, enable_itn: bool) -> str:
if not enable_itn:
return text
return apply_itn_to_text(text)