From 2daa939b1f9077b488c41e87d48a16f9b114de0e Mon Sep 17 00:00:00 2001 From: Bifang <915779419@qq.com> Date: Wed, 23 Sep 2026 11:09:09 +0800 Subject: [PATCH] Add FunASR streaming model download config --- FUNASR_README.md | 10 +++- README.md | 7 ++- model_manifest.json | 62 ++++++++++++++++++++---- pyproject.toml | 1 + scripts/auxiliary_server.py | 3 +- scripts/download_models.py | 13 +++++ scripts/download_models_qwen_legacy.py | 38 ++++++++++++--- scripts/model_manifest.py | 25 ++++++++++ scripts/run_backend.py | 56 ++++++++++++--------- tests/qwen_legacy_model_manifest_test.py | 9 ++-- 10 files changed, 177 insertions(+), 47 deletions(-) create mode 100644 scripts/download_models.py diff --git a/FUNASR_README.md b/FUNASR_README.md index 115f6de..34b3ab9 100644 --- a/FUNASR_README.md +++ b/FUNASR_README.md @@ -3,7 +3,12 @@ The frontend and backend start separately. The backend launcher loads three required local assets: streaming ASR, FSMN VAD, and CAM++ speaker verification. It starts the CAM++ model service and the WebSocket service and stops both -together. No model is downloaded by the launcher. +together. No model is downloaded by the launcher. Use the shared model manifest +and downloader to prepare the three required snapshots: + +~~~powershell +python scripts/download_models.py --funasr-runtime +~~~ ## Model directories @@ -11,7 +16,7 @@ Put the assets under models/ or set MODEL_DIR in .env. The default FunASR names resolve to these local directories: - models/iic/speech_paraformer-large_asr_nat-zh-cn-16k-common-vocab8404-online -- models/iic/speech_fsmn_vad_zh-cn-16k-common-pytorch (or the damo/ variant) +- models/damo/speech_fsmn_vad_zh-cn-16k-common-pytorch - models/iic/speech_campplus_sv_zh-cn_16k-common (or the damo/ variant) If your directories have different names, set FUNASR_ASR_MODEL and @@ -28,6 +33,7 @@ then install the project dependencies: cd D:\github-project\ASR\Asr-demo python -m pip install -r requirements-funasr.txt python -m pip install -r requirements-auxiliary.txt +python scripts/download_models.py --funasr-runtime if (-not (Test-Path .env)) { Copy-Item .env.funasr.example .env } ~~~ diff --git a/README.md b/README.md index d1e2c28..bd5195e 100644 --- a/README.md +++ b/README.md @@ -7,7 +7,12 @@ The browser UI starts as a separate service. Install a torch/torchaudio build for the host, then install requirements-funasr.txt and requirements-auxiliary.txt. Copy -.env.funasr.example to .env if no .env exists, and check MODEL_DIR. +.env.funasr.example to .env if no .env exists, then download the three +required FunASR models with the shared project manifest: + +~~~powershell +python scripts/download_models.py --funasr-runtime +~~~ Start the backend (CAM++ model service plus WebSocket) in one terminal: diff --git a/model_manifest.json b/model_manifest.json index a96f58e..db14a5c 100644 --- a/model_manifest.json +++ b/model_manifest.json @@ -10,6 +10,24 @@ "alias": "0.6b", "directory": "Qwen/Qwen3-ASR-0.6B", "description": "Qwen3-ASR 0.6B,GPU 默认轻量模型" + }, + "iic/speech_paraformer-large_asr_nat-zh-cn-16k-common-vocab8404-online": { + "alias": "paraformer-zh-streaming", + "directory": "iic/speech_paraformer-large_asr_nat-zh-cn-16k-common-vocab8404-online", + "description": "FunASR Paraformer streaming ASR used by the realtime backend", + "required_files": [ + "configuration.json", + "config.yaml" + ], + "any_files": [ + "*.pb", + "*.pt", + "*.onnx", + "*.bin", + "*.model", + "*.safetensors" + ], + "min_total_size_bytes": 1000000 } }, "auxiliary_models": { @@ -19,8 +37,13 @@ "kind": "vad", "description": "FunASR FSMN VAD", "revision": "v2.0.2", - "required_files": ["configuration.json", "config.yaml", "model.pb"], - "min_total_size_bytes": 1000000 + "required_files": [ + "configuration.json", + "config.yaml", + "model.pb" + ], + "min_total_size_bytes": 1000000, + "funasr_alias": "fsmn-vad" }, "iic/speech_campplus_speaker-diarization_common": { "alias": "diarization", @@ -43,7 +66,11 @@ "kind": "speaker_verification", "description": "Configured CAM++ speaker verification", "revision": "v2.0.2", - "required_files": ["configuration.json", "config.yaml", "campplus_cn_common.bin"], + "required_files": [ + "configuration.json", + "config.yaml", + "campplus_cn_common.bin" + ], "min_total_size_bytes": 10000000 }, "iic/speech_eres2netv2_sv_zh-cn_16k-common": { @@ -51,8 +78,12 @@ "directory": "iic/speech_eres2netv2_sv_zh-cn_16k-common", "kind": "realtime_speaker_verification", "description": "Realtime speaker verification", - "required_files": ["configuration.json"], - "any_files": ["*"], + "required_files": [ + "configuration.json" + ], + "any_files": [ + "*" + ], "min_total_size_bytes": 10000000 }, "damo/speech_campplus_sv_zh-cn_16k-common": { @@ -60,7 +91,11 @@ "directory": "damo/speech_campplus_sv_zh-cn_16k-common", "kind": "speaker_verification", "description": "CAM++ speaker verification dependency", - "required_files": ["configuration.json", "config.yaml", "campplus_cn_common.bin"], + "required_files": [ + "configuration.json", + "config.yaml", + "campplus_cn_common.bin" + ], "min_total_size_bytes": 10000000 }, "damo/speech_campplus-transformer_scl_zh-cn_16k-common": { @@ -68,7 +103,11 @@ "directory": "damo/speech_campplus-transformer_scl_zh-cn_16k-common", "kind": "speaker_transformer", "description": "CAM++ Transformer dependency", - "required_files": ["configuration.json", "campplus_cn_encoder.pt", "transformer_backend.pt"], + "required_files": [ + "configuration.json", + "campplus_cn_encoder.pt", + "transformer_backend.pt" + ], "min_total_size_bytes": 10000000 }, "Qwen/Qwen3-ForcedAligner-0.6B": { @@ -76,8 +115,13 @@ "directory": "Qwen/Qwen3-ForcedAligner-0.6B", "kind": "forced_aligner", "description": "Qwen3 word-level forced aligner", - "required_files": ["config.json"], - "any_files": ["*.safetensors", "*.bin"], + "required_files": [ + "config.json" + ], + "any_files": [ + "*.safetensors", + "*.bin" + ], "min_total_size_bytes": 500000000 } } diff --git a/pyproject.toml b/pyproject.toml index f0c63ad..b2bbe73 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -19,6 +19,7 @@ dependencies = [ [project.scripts] funasr-realtime-demo = "scripts.run_funasr_demo:main" +funasr-download-models = "scripts.download_models:main" [tool.setuptools] packages = ["scripts", "realtime_websocket"] diff --git a/scripts/auxiliary_server.py b/scripts/auxiliary_server.py index 2131614..b6198c6 100644 --- a/scripts/auxiliary_server.py +++ b/scripts/auxiliary_server.py @@ -197,7 +197,8 @@ class AuxiliaryRuntime: ) raise RuntimeError( "Auxiliary core model is missing or failed to load: " + details - + ". Set MODEL_DIR or CAM_MODEL_PATH to the local CAM++ asset." + + ". Run python scripts/download_models.py --funasr-runtime, " + "or set MODEL_DIR/CAM_MODEL_PATH to the local CAM++ asset." ) def _find_model(self, kind: str) -> Any: diff --git a/scripts/download_models.py b/scripts/download_models.py new file mode 100644 index 0000000..885073c --- /dev/null +++ b/scripts/download_models.py @@ -0,0 +1,13 @@ +#!/usr/bin/env python3 +"""Download models using the project's shared model manifest.""" + +from __future__ import annotations + +try: + from .download_models_qwen_legacy import main +except ImportError: + from download_models_qwen_legacy import main + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/download_models_qwen_legacy.py b/scripts/download_models_qwen_legacy.py index e3159c1..91fe8c0 100644 --- a/scripts/download_models_qwen_legacy.py +++ b/scripts/download_models_qwen_legacy.py @@ -73,7 +73,7 @@ def download_model( from modelscope.hub.snapshot_download import snapshot_download except ImportError as exc: raise RuntimeError( - "ModelScope is required for downloading; install requirements-download.txt first" + "ModelScope is required for downloading; install requirements-funasr.txt first" ) from exc model_path.parent.mkdir(parents=True, exist_ok=True) @@ -176,17 +176,41 @@ def main() -> int: action="store_true", help="Only download/check VAD, speaker, diarization, and aligner assets", ) + auxiliary_group.add_argument( + "--funasr-runtime", + action="store_true", + help="Download/check only streaming Paraformer, FSMN-VAD, and CAM++", + ) args = parser.parse_args() manifest = load_manifest() - models_dir = args.models_dir.resolve() + models_dir = args.models_dir + if not models_dir.is_absolute(): + models_dir = Path(__file__).resolve().parents[1] / models_dir + models_dir = models_dir.resolve() cache_dir = args.cache_dir.resolve() if args.cache_dir else None selected_assets: list[tuple[str, dict[str, object]]] = [] - if not args.auxiliary_only: - model_id = resolve_model_id(args.model, manifest) - selected_assets.append((model_id, manifest["models"][model_id])) - if not args.skip_auxiliary: - selected_assets.extend(auxiliary_models(manifest).items()) + if args.funasr_runtime: + # The realtime branch needs the streaming ASR plus its VAD and CAM++ only. + asr_id = resolve_model_id("paraformer-zh-streaming", manifest) + selected_assets.append((asr_id, manifest["models"][asr_id])) + assets = auxiliary_models(manifest) + vad_id = next( + model_id for model_id, config in assets.items() + if config.get("kind") == "vad" + ) + cam_id = next( + model_id for model_id, config in assets.items() + if config.get("kind") == "speaker_verification" + and model_id.startswith("iic/") + ) + selected_assets.extend((model_id, assets[model_id]) for model_id in (vad_id, cam_id)) + else: + if not args.auxiliary_only: + model_id = resolve_model_id(args.model, manifest) + selected_assets.append((model_id, manifest["models"][model_id])) + if not args.skip_auxiliary: + selected_assets.extend(auxiliary_models(manifest).items()) missing: list[tuple[str, Path, dict[str, object]]] = [] for model_id, config in selected_assets: diff --git a/scripts/model_manifest.py b/scripts/model_manifest.py index d4e7d18..91f9701 100644 --- a/scripts/model_manifest.py +++ b/scripts/model_manifest.py @@ -52,3 +52,28 @@ def auxiliary_models(manifest: dict[str, Any]) -> dict[str, dict[str, Any]]: if not isinstance(models, dict): raise ValueError("model_manifest.json 的 auxiliary_models 必须是对象") return {str(model_id): config for model_id, config in models.items() if isinstance(config, dict)} + +def resolve_auxiliary_model_id( + model: str | None, + manifest: dict[str, Any], + kind: str | None = None, +) -> str: + """Resolve a configured auxiliary asset by ID, alias, or FunASR alias.""" + assets = auxiliary_models(manifest) + requested = (model or "").strip() + if requested in assets and (kind is None or assets[requested].get("kind") == kind): + return requested + + for model_id, config in assets.items(): + if kind is not None and config.get("kind") != kind: + continue + aliases = [config.get("alias"), config.get("funasr_alias")] + if requested.lower() in { + str(alias).lower() for alias in aliases if isinstance(alias, str) + }: + return model_id + expected = ", ".join( + model_id for model_id, config in assets.items() + if kind is None or config.get("kind") == kind + ) + raise ValueError(f"Unsupported auxiliary model '{model}'; configured assets: {expected}") diff --git a/scripts/run_backend.py b/scripts/run_backend.py index 436f318..5af1c14 100644 --- a/scripts/run_backend.py +++ b/scripts/run_backend.py @@ -18,35 +18,41 @@ PROJECT_ROOT = Path(__file__).resolve().parents[1] if str(PROJECT_ROOT) not in sys.path: sys.path.insert(0, str(PROJECT_ROOT)) -from scripts.model_manifest import auxiliary_models, load_manifest, model_directory +from scripts.model_manifest import ( + auxiliary_models, + load_manifest, + model_directory, + resolve_auxiliary_model_id, + resolve_model_id, +) load_dotenv(PROJECT_ROOT / ".env") -# FunASR's published short names resolve to these ModelScope local directories. -LOCAL_MODEL_NAMES = { - "paraformer-zh-streaming": ( - "iic/speech_paraformer-large_asr_nat-zh-cn-16k-common-vocab8404-online", - ), - "fsmn-vad": ( - "iic/speech_fsmn_vad_zh-cn-16k-common-pytorch", - "damo/speech_fsmn_vad_zh-cn-16k-common-pytorch", - ), -} - -def local_model(requested: str, models_dir: Path) -> Path: - """Resolve a model ID or path beneath MODEL_DIR without a network fallback.""" +def local_model(requested: str, models_dir: Path, kind: str) -> Path: + """Resolve ASR/VAD IDs through the shared model manifest.""" name = requested.strip() - direct = Path(name) - candidates = [direct] if direct.is_absolute() else [ - PROJECT_ROOT / direct, models_dir / direct, - *(models_dir / alias for alias in LOCAL_MODEL_NAMES.get(name, ())), - ] - for candidate in candidates: + configured_path = Path(name) + direct_candidates = ( + [configured_path] + if configured_path.is_absolute() + else [PROJECT_ROOT / configured_path, models_dir / configured_path] + ) + for candidate in direct_candidates: if candidate.is_dir() and (candidate / "configuration.json").is_file(): return candidate.resolve() - checked = ", ".join(str(path) for path in candidates) - raise FileNotFoundError(f"Local model '{name}' was not found; checked: {checked}") + + manifest = load_manifest() + if kind == "asr": + model_id = resolve_model_id(name, manifest) + else: + model_id = resolve_auxiliary_model_id(name, manifest, kind=kind) + candidate = model_directory(model_id, manifest, models_dir) + if candidate.is_dir() and (candidate / "configuration.json").is_file(): + return candidate.resolve() + raise FileNotFoundError( + f"Local {kind.upper()} model '{name}' is missing; expected: {candidate}" + ) def local_cam_model(models_dir: Path) -> Path: @@ -115,8 +121,10 @@ def main() -> None: if not models_dir.is_absolute(): models_dir = PROJECT_ROOT / models_dir models_dir = models_dir.resolve() - asr = local_model(os.getenv("FUNASR_ASR_MODEL", "paraformer-zh-streaming"), models_dir) - vad = local_model(os.getenv("FUNASR_VAD_MODEL", "fsmn-vad"), models_dir) + asr = local_model( + os.getenv("FUNASR_ASR_MODEL", "paraformer-zh-streaming"), models_dir, "asr" + ) + vad = local_model(os.getenv("FUNASR_VAD_MODEL", "fsmn-vad"), models_dir, "vad") cam = local_cam_model(models_dir) print(f"Local models: ASR={asr}; VAD={vad}; CAM++={cam}", flush=True) diff --git a/tests/qwen_legacy_model_manifest_test.py b/tests/qwen_legacy_model_manifest_test.py index b8a404b..18c1621 100644 --- a/tests/qwen_legacy_model_manifest_test.py +++ b/tests/qwen_legacy_model_manifest_test.py @@ -22,10 +22,13 @@ class ModelManifestTests(unittest.TestCase): self.assertEqual(resolve_model_id("1.7b", self.manifest), "Qwen/Qwen3-ASR-1.7B") self.assertEqual(resolve_model_id("0.6b", self.manifest), "Qwen/Qwen3-ASR-0.6B") - def test_manifest_has_only_asr_models(self) -> None: + def test_manifest_contains_legacy_and_funasr_asr_models(self) -> None: + self.assertTrue({"Qwen/Qwen3-ASR-0.6B", "Qwen/Qwen3-ASR-1.7B"}.issubset( + set(self.manifest["models"]) + )) self.assertEqual( - set(self.manifest["models"]), - {"Qwen/Qwen3-ASR-0.6B", "Qwen/Qwen3-ASR-1.7B"}, + resolve_model_id("paraformer-zh-streaming", self.manifest), + "iic/speech_paraformer-large_asr_nat-zh-cn-16k-common-vocab8404-online", ) def test_manifest_has_auxiliary_runtime_assets(self) -> None: