60 lines
1.6 KiB
Python
60 lines
1.6 KiB
Python
# -*- coding: utf-8 -*-
|
||
"""
|
||
基于itntext的ITN(逆文本标准化)工具模块
|
||
使用itntext库提供高质量的中文ITN处理
|
||
"""
|
||
|
||
import logging
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
# itntext导入 - 延迟导入以避免初始化问题
|
||
_itntext_normalizer = None
|
||
|
||
|
||
def _get_normalizer():
|
||
"""获取itntext标准化器实例(单例模式)"""
|
||
global _itntext_normalizer
|
||
if _itntext_normalizer is None:
|
||
try:
|
||
from itntext import Normalizer
|
||
_itntext_normalizer = Normalizer(lang="zh", operator="itn")
|
||
logger.info("itntext ITN模块初始化成功")
|
||
except ImportError as e:
|
||
logger.error(f"导入itntext失败: {e}")
|
||
raise ImportError("请安装itntext库: pip install itntext")
|
||
except Exception as e:
|
||
logger.error(f"初始化itntext失败: {e}")
|
||
raise
|
||
return _itntext_normalizer
|
||
|
||
|
||
def apply_itn_to_text(text: str) -> str:
|
||
"""
|
||
对文本应用逆文本标准化(ITN)
|
||
使用itntext库进行高质量的中文ITN处理
|
||
|
||
Args:
|
||
text: 语音识别结果文本
|
||
|
||
Returns:
|
||
应用ITN后的文本
|
||
"""
|
||
if not text or not text.strip():
|
||
return text
|
||
|
||
try:
|
||
normalizer = _get_normalizer()
|
||
result = normalizer.normalize(text)
|
||
logger.debug(f"ITN处理: '{text}' -> '{result}'")
|
||
return result
|
||
except Exception as e:
|
||
logger.warning(f"ITN处理失败: {text}, 错误: {str(e)}")
|
||
return text
|
||
|
||
|
||
def normalize_asr_text(text: str, enable_itn: bool) -> str:
|
||
if not enable_itn:
|
||
return text
|
||
return apply_itn_to_text(text)
|