From 69b95c050e4bb23cc3d8c3d4f0fbbd90ca2a5083 Mon Sep 17 00:00:00 2001 From: LauraGPT <18321252+LauraGPT@users.noreply.github.com> Date: Sun, 30 Aug 2026 17:02:22 +0000 Subject: [PATCH] docs(site): publish FunClip MOSS clipping guide Signed-off-by: LauraGPT <18321252+LauraGPT@users.noreply.github.com> --- .../product-site/content/legacy-manifest.json | 8 +- web-pages/product-site/data/deployments.json | 2 + .../funclip-v2-2-0-moss-speaker-clipping.html | 87 +++++++++++++++++++ web-pages/product-site/legacy/blog/index.html | 6 +- .../funclip-v2-2-0-moss-speaker-clipping.html | 87 +++++++++++++++++++ .../product-site/legacy/en/blog/index.html | 6 +- web-pages/product-site/legacy/sitemap.xml | 2 + .../tests/browser/product-site.spec.ts | 24 +++-- web-pages/product-site/tests/test_legacy.py | 50 +++++++++++ web-pages/product-site/tests/test_output.py | 51 +++++++++-- 10 files changed, 299 insertions(+), 24 deletions(-) create mode 100644 web-pages/product-site/legacy/blog/funclip-v2-2-0-moss-speaker-clipping.html create mode 100644 web-pages/product-site/legacy/en/blog/funclip-v2-2-0-moss-speaker-clipping.html diff --git a/web-pages/product-site/content/legacy-manifest.json b/web-pages/product-site/content/legacy-manifest.json index 409843b05..1615cc60d 100644 --- a/web-pages/product-site/content/legacy-manifest.json +++ b/web-pages/product-site/content/legacy-manifest.json @@ -21,8 +21,9 @@ "blog/funasr-vs-faster-whisper-chinese.html": "bfe9bb8017be80c7e7f4587726f43f6064f1dc4c65e39788f836fdfb0c9789f7", "blog/funasr-vs-whisper-benchmark.html": "b7b49adf24d20570abb09b733ce03d4a50a4a0e98e746b4a9320f453e01cce84", "blog/funclip-v2-1-0-video-clipping-release.html": "88f6c44e5332d1746c4db0fc97d755ef82f12e46d8f1c1c0ad9351152ff9dfc1", + "blog/funclip-v2-2-0-moss-speaker-clipping.html": "7127fb3d6493500d1e44e2168427b3e9adf9ec9bcd840d9da8fd5a1323af9709", "blog/generate-subtitles-srt-vtt-from-audio-video.html": "f1133235673d441cc654310581f59a319f182314ab7c0b0824af87da9f4a0591", - "blog/index.html": "fbfa2611c97cc5426d78450078df36fb1818da1a5eeaef718e6a18fa9fd3d287", + "blog/index.html": "0698ce3d08c4b61c1e1e3f94142d3564581d162765081437229d2c4f3fff5ff7", "blog/japanese-speech-recognition.html": "399f5ce84e68ac00854bdf70b52b1cde6795efca7c800fd6492035a37a7c1b68", "blog/lightweight-speech-recognition-cpu.html": "d6270066222ed5baef0df108e6a4155f4a87bb5482169f23f218794d371f3281", "blog/punctuation-restoration-python.html": "6cd26e03bf75b4343afd7b351c2681ebd513b058fb549c5888802be9ff3a1229", @@ -59,8 +60,9 @@ "en/blog/funasr-vs-faster-whisper-chinese.html": "abf31d3827dfba9b31ccd4e76229de9e4bfff00ab8bcee33bd718c88249630f4", "en/blog/funasr-vs-whisper-benchmark.html": "367d4a8a1cc09ac932c80cdad065127925c5e5f942a806ace683c16dc1132769", "en/blog/funclip-v2-1-0-video-clipping-release.html": "229baf59adf2c3290541d9b3c8a6243992406ba84b712714e9d14599205cb1d4", + "en/blog/funclip-v2-2-0-moss-speaker-clipping.html": "6456c699b4837e5c1fc3741cda40a58a1d0caed8498351445e2dd6314dd89b3c", "en/blog/generate-subtitles-srt-vtt-from-audio-video.html": "1530d4b9e94820a0b60d801e8d1c19b7aeb031b2f2b76c092657392e47706c6d", - "en/blog/index.html": "526253689a12ead8058ef9f45b9cdc37254f6507ab6d23511a742d88fddb95c2", + "en/blog/index.html": "fc46e28078b49173d9f124afc51bed070b94b3738850b077a7fffcdaa965bc3a", "en/blog/japanese-speech-recognition.html": "c476adcc2be1ed7c19e2345cc91b8ee6a0e04efd794b9b6476dd5d9b3a2c8b04", "en/blog/lightweight-speech-recognition-cpu.html": "0e02c0e0853b12613d32c250f593c05c2052f18231207b63ee01d6e6600a86f7", "en/blog/punctuation-restoration-python.html": "d6e30049f48d9dc6e6bc22eb567830013e4161eca51215616738b675dd1ffc40", @@ -106,7 +108,7 @@ "models.html": "724c88917c9934ffa66948cf3eb583fed930dd689d691b7d21fa22ee8332e606", "quickstart.html": "a543bf07c2cfc9a69ef3831f15c9e9d3e571a826a73cf7e4bdc8e9eb7b910254", "robots.txt": "1305b0f680b3e9f73b01301dc474423230dffab85731a379a694f30c73c07f2d", - "sitemap.xml": "bcc17d763cd1b0d317b3af8e52cc40fde58c4a80f3fb5664a12b5503ecf2fc19", + "sitemap.xml": "ed91fc0211b48458b77c56c1e600d055c013a6bb4b461821c3ec78dcff4ac26a", "static/liveplayer/liveplayer-component.min.js": "ea10fda35574cc3a77a3e7187095467d1a409009a2d39690051d33d0e7055d30", "static/liveplayer/liveplayer-lib.min.js": "ad67b4e1188c586ec218674a6db9968daf71ed4fbc262258f061c1f931d54250", "static/offline/index.html": "868781c621ddabe5d3f088d6b739e0c08e6cdc150587af6515cc323852018e62", diff --git a/web-pages/product-site/data/deployments.json b/web-pages/product-site/data/deployments.json index 873aa24e7..952c97497 100644 --- a/web-pages/product-site/data/deployments.json +++ b/web-pages/product-site/data/deployments.json @@ -98,6 +98,8 @@ {"label": "merged native SGLang Omni MOSS runtime and H100 benchmark", "url": "https://github.com/sgl-project/sglang-omni/pull/914"}, {"label": "FunASR ecosystem integration guide", "url": "https://github.com/modelscope/FunASR/blob/main/docs/moss_transcribe_diarize.md"}, {"label": "FunASR AutoModel adapter merge and exact-head CI", "url": "https://github.com/modelscope/FunASR/pull/3558"}, + {"label": "FunClip v2.2.0 MOSS speaker-clipping guide", "url": "https://www.funasr.com/en/blog/funclip-v2-2-0-moss-speaker-clipping.html"}, + {"label": "FunClip v2.2.0 release and checksums", "url": "https://github.com/modelscope/FunClip/releases/tag/v2.2.0"}, {"label": "moss-transcribe.cpp third-party C++17 and GGUF runtime", "url": "https://github.com/localai-org/moss-transcribe.cpp"}, {"label": "LocalAI third-party moss-transcribe-cpp backend", "url": "https://github.com/mudler/LocalAI"} ], diff --git a/web-pages/product-site/legacy/blog/funclip-v2-2-0-moss-speaker-clipping.html b/web-pages/product-site/legacy/blog/funclip-v2-2-0-moss-speaker-clipping.html new file mode 100644 index 000000000..fe29da7cc --- /dev/null +++ b/web-pages/product-site/legacy/blog/funclip-v2-2-0-moss-speaker-clipping.html @@ -0,0 +1,87 @@ + + + + + + FunClip v2.2.0:MOSS 长音频说话人识别与视频剪辑 | FunASR + + + + + + + + + + + + + + + +
+

FunClip v2.2.0:用 MOSS 做长音频说话人识别与视频剪辑

+

2026-08-31 · FunClip Release

+ FunClip 本地视频、字幕识别和智能剪辑界面 +

FunClip v2.2.0 新增一条可选的 MOSS 路径:把长音频交给 vLLM 服务,FunASR 将模型输出归一化为文本、说话人身份和时间段,FunClip 再生成 SRT 或按 spkS01spkS02 剪辑。

+

MOSS-Transcribe-Diarize 是 OpenMOSS 维护的第三方模型,不属于 FunASR 或 FunClip。集成固定使用 OpenMOSS-Team/MOSS-Transcribe-Diarize revision e8681d68e7042738ffca8ac8212bc8fcb1131ab8,并明确保留模型归属与支持边界。

+ +

数据路径

+ + + + +
阶段职责
vLLM加载固定 MOSS revision,通过 /v1/audio/transcriptions 返回带时间与说话人标记的 JSON。
FunASR 1.4.9+解析转写结果并生成统一的 texttimestampsentence_info
FunClip 2.2.0渲染说话人 SRT、按说话人选段,或把时间段交给现有音视频剪辑流程。
+ +

1. 启动固定 revision 的 vLLM 服务

+
python -m venv .venv-moss
+. .venv-moss/bin/activate
+pip install -U vllm
+
+vllm serve OpenMOSS-Team/MOSS-Transcribe-Diarize \
+  --revision e8681d68e7042738ffca8ac8212bc8fcb1131ab8 \
+  --served-model-name moss-transcribe-diarize \
+  --trust-remote-code --host 127.0.0.1 --port 8898
+

先用真实音频检查服务契约。当前经过验证的格式是 response_format=json

+
curl -fsS http://127.0.0.1:8898/v1/audio/transcriptions \
+  -F file=@sample.wav \
+  -F model=moss-transcribe-diarize \
+  -F response_format=json \
+  -F max_completion_tokens=8192
+ +

2. 启动 FunClip

+
python -m pip install -U -r requirements.txt
+python funclip/launch.py \
+  --model moss \
+  --moss-backend vllm \
+  --moss-base-url http://127.0.0.1:8898/v1 \
+  --moss-max-tokens 8192
+

远端服务需要 bearer token 时,把凭据放进 MOSS_API_KEY 环境变量;FunClip 不要求把 token 写进命令行或仓库。

+ +

能力与边界

+ +

完整生产部署、健康检查和容量边界见 MOSS 双语部署指南

+ +

下载与校验 v2.2.0

+
FunClip-2.2.0.tar.gz
+SHA256 994c5d9cf392b74b36284d526eca8bada1560a3e7825ab7baa9c673a1b4ef216
+
+FunClip-2.2.0.zip
+SHA256 4f5a7d33d9ea65467f29b55b15ed5be18e64de7e57e2f9f36fe51a23b40557e7
+

FunClip v2.2.0 Release 下载归档和 SHA256SUMS,不要从不明镜像复制二进制或凭据。

+ +
发布内容对应 commit c205bf32a8b11226ff5e8acb9a3c7a1f00cd3b06。仓库测试结果为 84 passed、1 skipped;H100 + vLLM 两说话人样例返回 S01/S02、有效 SRT,并成功按 S02 剪辑。这个验证证明发布链路可用,不代表所有语言、音频条件或并发规模。
+

先按部署指南跑通一段真实双人音频,再进入 FunClip 处理自己的视频。

查看 FunClip v2.2.0 发布与下载
+
+ + + diff --git a/web-pages/product-site/legacy/blog/index.html b/web-pages/product-site/legacy/blog/index.html index 7925da337..9627c3f37 100644 --- a/web-pages/product-site/legacy/blog/index.html +++ b/web-pages/product-site/legacy/blog/index.html @@ -42,13 +42,13 @@ .post-card p{margin:0;color:var(--text-soft);font-size:0.9rem} .post-card .date{color:var(--text-muted);font-size:0.8rem} .launch-feature{max-width:800px;margin:104px auto 0;padding:0 24px}.launch-feature a{display:grid;grid-template-columns:1fr auto;align-items:center;gap:20px;border:1px solid #bbf7d0;border-left:4px solid #047857;border-radius:8px;background:#f0fdf4;padding:20px 22px;color:var(--text)}.launch-feature a:hover{border-color:#047857;text-decoration:none}.launch-feature .date{font-size:.8rem;color:#047857;font-weight:600}.launch-feature h2{border:0;padding:0;margin:3px 0 5px;font-size:1.2rem}.launch-feature p{margin:0;font-size:.9rem}.launch-feature .action{font-weight:700;color:#1d4ed8;white-space:nowrap} -.previous-release{padding-top:16px}.previous-release .container{display:grid;grid-template-columns:repeat(3,minmax(0,1fr));gap:16px}.previous-release .post-card{margin-bottom:0} +.previous-release{padding-top:16px}.previous-release .container{display:grid;grid-template-columns:repeat(4,minmax(0,1fr));gap:16px}.previous-release .post-card{margin-bottom:0} footer{background:#0f172a;color:#94a3b8;padding:28px 0;text-align:center;font-size:0.8rem} footer a{color:#94a3b8} @media(max-width:900px){.nav-links{display:none}.nav .container{gap:16px}.nav-logo{margin-right:auto}.launch-feature a{grid-template-columns:1fr}.launch-feature .action{white-space:normal}.previous-release .container{grid-template-columns:1fr}} -
2026-08-28 · 正式 Release

FunASR v1.4.5:更轻的 Python 推理与九平台 llama.cpp 运行时

默认安装不再硬依赖 torchaudio,可选 kaldi-native-fbank,并在一个页面提供 PyPI 包、九平台运行时与 SHA-256 清单。

查看安装与部署指南 →
-
2026-08-21 · 正式 Release

FunASR v1.4.3:可选 Silero VAD 与大规模固定 K 说话人聚类

新增毫秒级 Silero VAD 适配器,降低已知说话人数的大规模聚类内存压力,并提供 12 个带 SHA-256 的发布资产。

2026-07-31 · 正式 Release

FunASR v1.4.0:更完整的 PyPI 安装包与更安全的 AutoModel 参数

修复 SenseVoice 与 RWKV-BAT 包数据,提前拒绝 vda_model 误拼写,并提供 12 个带 SHA-256 的发布资产。

2026-08-03 · 生态集成

Subtitle Edit 5.2:Fun-ASR Nano / SenseVoice 本地视频字幕

在 Windows、macOS 与 Linux 的成熟字幕工作区中直接选择本地 ASR 引擎,支持 Q4、Q8、F16。

+
2026-08-31 · FunClip v2.2.0

MOSS 多说话人转写进入 FunClip:从一句话到带说话人的片段

通过经真实双说话人音频验证的 vLLM 路径,把第三方 MOSS 模型的段级时间戳和说话人标签直接用于 SRT 与说话人片段导出。

查看部署与剪辑边界 →
+
2026-08-28 · 正式 Release

FunASR v1.4.5:更轻的 Python 推理与九平台 llama.cpp 运行时

默认安装不再硬依赖 torchaudio,可选 kaldi-native-fbank,并提供 PyPI 包、九平台运行时与 SHA-256 清单。

2026-08-21 · 正式 Release

FunASR v1.4.3:可选 Silero VAD 与大规模固定 K 说话人聚类

新增毫秒级 Silero VAD 适配器,降低已知说话人数的大规模聚类内存压力,并提供 12 个带 SHA-256 的发布资产。

2026-07-31 · 正式 Release

FunASR v1.4.0:更完整的 PyPI 安装包与更安全的 AutoModel 参数

修复 SenseVoice 与 RWKV-BAT 包数据,提前拒绝 vda_model 误拼写,并提供 12 个带 SHA-256 的发布资产。

2026-08-03 · 生态集成

Subtitle Edit 5.2:Fun-ASR Nano / SenseVoice 本地视频字幕

在 Windows、macOS 与 Linux 的成熟字幕工作区中直接选择本地 ASR 引擎,支持 Q4、Q8、F16。

FunASR 技术博客

2026-06-23

语音识别带时间戳(字级 timestamps)Python 实战:每个字精确到毫秒

FunASR Paraformer 原生字级时间戳:每个字带 [起始毫秒,结束毫秒],一次调用即有。可做逐字高亮、点击跳转、字幕对齐。附真实实测+配对代码。

2026-06-23

标点恢复 Python 实战:给无标点文本/ASR 结果自动加标点

FunASR ct-punc 开源标点恢复,中英双语,3 行 Python 补全 ,。?(英文还做句首大写),也能一行挂到 ASR 上。附真实实测。

2026-06-22

中文语音识别(普通话)Python 实战:用 FunASR 又快又准

FunASR 专为中文打造:默认旗舰 Fun-ASR-Nano(CER 8.06%),CPU 用 SenseVoice(7.81%)/ Paraformer(10.18%,时间戳/热词),都远好于 Whisper(~20%)。3 行 Python。

2026-06-22

自托管语音转文字:Google / AWS / Azure 云语音 API 的免费开源替代

开源免费(MIT)、本地推理、不按分钟计费、数据不出内网、中文强,OpenAI 兼容改 base_url 即可迁移。附实测代码 + FunASR vs 云 API 对比 + 成本分析。

2026-06-22

Python 语音活动检测(VAD):检测语音、去静音、按停顿切分音频

FunASR fsmn-vad 3 行代码返回每段语音的起止毫秒。实测 13s 录音 0.12s 切成 2 段、去除 18% 静音;可去静音、切分长录音、给 Whisper 等做预处理防幻觉。

2026-06-22

日语语音识别:SenseVoice 一个模型搞定日语转写+标点+情感

同一段日语音频实测:SenseVoice 写对同音字 転売、自动加标点,Whisper-small 误为 天売、无标点。原生 ja 支持 + 自动语种识别 + 情感事件,3 行 Python。

2026-06-23

自托管替代 Deepgram / AssemblyAI

开源免费、自托管、音频不出本地、中文领先,自带 OpenAI 兼容 API——客户端只改 base_url 即可替代按分钟付费的云 STT。

2026-06-23

选哪个 FunASR 模型?Nano vs MLT-Nano vs SenseVoice vs Paraformer

模型选型表 + 场景代码:中英日及中文方言/口音选旗舰 Nano,31 语种选独立 MLT-Nano,CPU 再选 SenseVoice 或 Paraformer。

2026-06-22

轻量语音识别:CPU 上约 250MB 跑中文 ASR

单二进制 + 254MB q8 模型,无需 GPU/Python,CPU 0.16s,中文 CER 7.99%——比 whisper.cpp small 还小且准 3 倍。

2026-06-21

粤语语音识别:SenseVoice 原生支持粤语口语(Whisper 会转成普通话)

同一段粤语音频实测:SenseVoice 保留 呢/唔/嘅,Whisper 转成普通话书面语。原生 yue 支持 + 自动语种识别,3 行 Python。

2026-06-21

FunASR vs faster-whisper:中文与粤语实测对比

粤语被 faster-whisper 误判为普通话、日语同音字错;SenseVoice 原生支持粤语+语种识别,中文 CER 低约 2.7 倍。实测。

2026-06-20

FunASR 跑进 llama.cpp:中文 ASR 的 whisper.cpp 替代品(CPU/零 Python)

单个自包含二进制、内置 VAD、吃任意音频,下载即用转写中文;中文 CPU 上比 whisper.cpp 准约 2.7 倍。3 步实测。

2026-06-18

Python 语音转文字:用 FunASR 本地免费转写音频

几行 Python 把音频转成文本,带时间戳/说话人/批量;本地、免费、无 API key、中文强。

2026-06-18

用 FunASR 自动生成字幕:音频/视频一键出 SRT 和 VTT

一行命令出 SRT,Python 同时导出 VTT,带说话人和真实时间戳;本地、免费、中文强。

2026-06-18

自托管 OpenAI Whisper API 替代:FunASR 起兼容 /v1/audio/transcriptions 服务

funasr-server 暴露 OpenAI 兼容接口,OpenAI SDK 只改 base_url 就能用;本地、免费、隐私、中文更准。

2026-06-18

用 FunASR 命令行转写音频:文本/JSON/SRT 字幕

一行命令出文字/字幕/JSON,--spk 带说话人;还能 funasr-server 起 OpenAI 兼容 API。

2026-06-17

用 FunASR 转写超长音频:1 小时一次搞定

Whisper 限 30 秒,FunASR 内置 VAD 一次吃下任意时长;实测 13 分钟 4.3 秒转完(186x)。

2026-06-17

用 FunASR 实现实时流式语音识别(边说边出字)

600ms 级低延迟流式 ASR:分块+cache 边说边出字,含 2-pass(流式+离线)最佳实践。

2026-06-17

超越转写:用 SenseVoice 识别语言、情感与声学事件

一次非自回归前向同时输出转写+语种+情感+音频事件,Whisper 做不到的四合一。

2026-06-17

用 FunASR 做说话人分离:谁在何时说了什么

一次 generate 调用同时输出转写+说话人标签+时间戳,替代 pyannote+Whisper,无需 HF 授权。

2026-06-16

FunASR vs Whisper 实测对比:谁更快更准

184 中文文件 H100 实测:SenseVoice 169.6x、CER 7.81%,完整速度+准确率数据。

2026-06-16

Fun-ASR-Nano 使用指南:800M 端到端语音识别大模型

主力旗舰,中英日 + 7 大中文方言/26 种口音,热词/流式/说话人分离;31 语种请用 MLT-Nano。

2026-06-16

SenseVoice 部署指南:五语种识别、情感与音频事件

3 行代码跑通多语言识别,含语种/情感/事件检测、VAD、GPU/CPU。

更多:快速上手 · 模型

diff --git a/web-pages/product-site/legacy/en/blog/funclip-v2-2-0-moss-speaker-clipping.html b/web-pages/product-site/legacy/en/blog/funclip-v2-2-0-moss-speaker-clipping.html new file mode 100644 index 000000000..b19deb146 --- /dev/null +++ b/web-pages/product-site/legacy/en/blog/funclip-v2-2-0-moss-speaker-clipping.html @@ -0,0 +1,87 @@ + + + + + + FunClip v2.2.0: MOSS Speaker-Aware Video Clipping | FunASR + + + + + + + + + + + + + + + +
+

FunClip v2.2.0: Long-Form Speaker-Aware Video Clipping with MOSS

+

August 31, 2026 · FunClip Release

+ FunClip local video, subtitle recognition, and intelligent clipping interface +

FunClip v2.2.0 adds an opt-in MOSS path: send long audio to a vLLM service, normalize text, speaker identities, and time ranges through FunASR, then generate SRT or clip by spkS01, spkS02, and later speaker IDs.

+

MOSS-Transcribe-Diarize is a third-party model maintained by OpenMOSS, not a FunASR or FunClip-owned checkpoint. The integration pins OpenMOSS-Team/MOSS-Transcribe-Diarize revision e8681d68e7042738ffca8ac8212bc8fcb1131ab8 and keeps ownership and support boundaries explicit.

+ +

Data path

+ + + + +
StageResponsibility
vLLMLoads the pinned MOSS revision and serves timestamped speaker markup through /v1/audio/transcriptions.
FunASR 1.4.9+Parses the response into common text, timestamp, and sentence_info fields.
FunClip 2.2.0Renders speaker-labeled SRT and sends selected speaker ranges into the existing audio/video clipping workflow.
+ +

1. Start vLLM at the pinned revision

+
python -m venv .venv-moss
+. .venv-moss/bin/activate
+pip install -U vllm
+
+vllm serve OpenMOSS-Team/MOSS-Transcribe-Diarize \
+  --revision e8681d68e7042738ffca8ac8212bc8fcb1131ab8 \
+  --served-model-name moss-transcribe-diarize \
+  --trust-remote-code --host 127.0.0.1 --port 8898
+

Verify the service with real audio before starting FunClip. The tested contract uses response_format=json:

+
curl -fsS http://127.0.0.1:8898/v1/audio/transcriptions \
+  -F file=@sample.wav \
+  -F model=moss-transcribe-diarize \
+  -F response_format=json \
+  -F max_completion_tokens=8192
+ +

2. Start FunClip

+
python -m pip install -U -r requirements.txt
+python funclip/launch.py \
+  --model moss \
+  --moss-backend vllm \
+  --moss-base-url http://127.0.0.1:8898/v1 \
+  --moss-max-tokens 8192
+

For an authenticated remote service, put the bearer credential in the MOSS_API_KEY environment variable. FunClip does not require putting the token in the command line or repository.

+ +

Capabilities and boundaries

+ +

See the bilingual MOSS production guide for health checks, runtime choices, and capacity boundaries.

+ +

Download and verify v2.2.0

+
FunClip-2.2.0.tar.gz
+SHA256 994c5d9cf392b74b36284d526eca8bada1560a3e7825ab7baa9c673a1b4ef216
+
+FunClip-2.2.0.zip
+SHA256 4f5a7d33d9ea65467f29b55b15ed5be18e64de7e57e2f9f36fe51a23b40557e7
+

Download the archives and SHA256SUMS from the FunClip v2.2.0 release.

+ +
Release content resolves to commit c205bf32a8b11226ff5e8acb9a3c7a1f00cd3b06. The repository suite completed with 84 passed and 1 skipped. A live H100 + vLLM two-speaker run returned S01/S02, valid SRT, and the expected S02 clip. This verifies the release path, not every language, audio condition, or production concurrency level.
+

Run one real two-speaker recording through the production guide, then bring the same service into FunClip.

Open the FunClip v2.2.0 release and downloads
+
+ + + diff --git a/web-pages/product-site/legacy/en/blog/index.html b/web-pages/product-site/legacy/en/blog/index.html index fe315c379..2191d5a2b 100644 --- a/web-pages/product-site/legacy/en/blog/index.html +++ b/web-pages/product-site/legacy/en/blog/index.html @@ -42,13 +42,13 @@ .post-card p{margin:0;color:var(--text-soft);font-size:0.9rem} .post-card .date{color:var(--text-muted);font-size:0.8rem} .launch-feature{max-width:800px;margin:104px auto 0;padding:0 24px}.launch-feature a{display:grid;grid-template-columns:1fr auto;align-items:center;gap:20px;border:1px solid #bbf7d0;border-left:4px solid #047857;border-radius:8px;background:#f0fdf4;padding:20px 22px;color:var(--text)}.launch-feature a:hover{border-color:#047857;text-decoration:none}.launch-feature .date{font-size:.8rem;color:#047857;font-weight:600}.launch-feature h2{border:0;padding:0;margin:3px 0 5px;font-size:1.2rem}.launch-feature p{margin:0;font-size:.9rem}.launch-feature .action{font-weight:700;color:#1d4ed8;white-space:nowrap} -.previous-release{padding-top:16px}.previous-release .container{display:grid;grid-template-columns:repeat(3,minmax(0,1fr));gap:16px}.previous-release .post-card{margin-bottom:0} +.previous-release{padding-top:16px}.previous-release .container{display:grid;grid-template-columns:repeat(4,minmax(0,1fr));gap:16px}.previous-release .post-card{margin-bottom:0} footer{background:#0f172a;color:#94a3b8;padding:28px 0;text-align:center;font-size:0.8rem} footer a{color:#94a3b8} @media(max-width:900px){.nav-links{display:none}.nav .container{gap:16px}.nav-logo{margin-right:auto}.launch-feature a{grid-template-columns:1fr}.launch-feature .action{white-space:normal}.previous-release .container{grid-template-columns:1fr}} -
August 28, 2026 · Stable release

FunASR v1.4.5: Lighter Python Inference and Nine llama.cpp Runtimes

No hard torchaudio dependency in the default install, optional kaldi-native-fbank, and one page for PyPI packages, nine runtimes, and SHA-256 checksums.

Read the install and deployment guide →
-
August 21, 2026 · Stable release

FunASR v1.4.3: Optional Silero VAD and Fixed-K Speaker Clustering

Millisecond Silero VAD segments, lower-memory clustering for large known-speaker workloads, and 12 SHA-256-addressed release assets.

July 31, 2026 · Stable release

FunASR v1.4.0: Complete PyPI Packages and Safer AutoModel Arguments

SenseVoice and RWKV-BAT package-data fixes, an early vda_model typo guard, and 12 SHA-256-addressed release assets.

August 3, 2026 · Ecosystem

Subtitle Edit 5.2: Local Video Subtitles with Fun-ASR Nano or SenseVoice

Select a local ASR engine inside a mature subtitle workspace on Windows, macOS, or Linux, with Q4, Q8, and F16 models.

+
August 31, 2026 · FunClip v2.2.0

MOSS Multi-Speaker Transcription in FunClip: From One Prompt to Speaker Clips

Use the vLLM path validated on real two-speaker audio to turn the third-party MOSS model's segment timestamps and speaker labels into SRT and speaker-specific clip exports.

Read the deployment and clipping boundaries →
+
August 28, 2026 · Stable release

FunASR v1.4.5: Lighter Python Inference and Nine llama.cpp Runtimes

No hard torchaudio dependency in the default install, optional kaldi-native-fbank, plus PyPI packages, nine runtimes, and SHA-256 checksums.

August 21, 2026 · Stable release

FunASR v1.4.3: Optional Silero VAD and Fixed-K Speaker Clustering

Millisecond Silero VAD segments, lower-memory clustering for large known-speaker workloads, and 12 SHA-256-addressed release assets.

July 31, 2026 · Stable release

FunASR v1.4.0: Complete PyPI Packages and Safer AutoModel Arguments

SenseVoice and RWKV-BAT package-data fixes, an early vda_model typo guard, and 12 SHA-256-addressed release assets.

August 3, 2026 · Ecosystem

Subtitle Edit 5.2: Local Video Subtitles with Fun-ASR Nano or SenseVoice

Select a local ASR engine inside a mature subtitle workspace on Windows, macOS, or Linux, with Q4, Q8, and F16 models.

FunASR Blog

2026-06-23

Speech-to-Text with Word/Character-Level Timestamps in Python

FunASR Paraformer gives native character timestamps: every char has [start_ms, end_ms] in one call. Build word highlighting, click-to-seek, subtitle alignment. Real output + pairing code.

2026-06-23

Punctuation Restoration in Python — Add Punctuation to Text / ASR Output

FunASR ct-punc: open-source bilingual (zh+en) punctuation restoration. 3 lines of Python to add ,。? (and capitalize English), or attach to ASR in one line. Real output.

2026-06-22

Chinese (Mandarin) Speech Recognition in Python — Fast & Accurate with FunASR

Purpose-built for Chinese: default flagship Fun-ASR-Nano (CER 8.06%), with SenseVoice (7.81%) / Paraformer (10.18%, timestamps/hotwords) for CPU — all far better than Whisper (~20%).

2026-06-22

Self-Hosted Speech-to-Text — Free Alternative to Google / AWS / Azure Cloud Speech APIs

Open-source (MIT), local inference, no per-minute billing, audio stays on your network, strong Chinese, OpenAI-compatible (migrate via base_url). Runnable code + FunASR vs cloud comparison + cost analysis.

2026-06-22

Voice Activity Detection in Python — Detect Speech, Remove Silence, Split Audio

FunASR fsmn-vad returns millisecond speech spans in 3 lines. Measured: a 13s clip split into 2 segments in 0.12s, 18% silence removed. Trim silence, split long audio, preprocess Whisper to cut hallucinations.

2026-06-22

Japanese Speech Recognition in Python — SenseVoice: Transcription + Punctuation + Emotion in One Pass

Real side-by-side on one Japanese clip: SenseVoice writes 転売 correctly and adds punctuation; Whisper-small gives 天売 with none. Native ja support + auto language ID + emotion/events, 3 lines of Python.

2026-06-23

Self-Hosted Deepgram / AssemblyAI Alternative

Open-source, free, self-hosted, audio stays local, Chinese-leading, with an OpenAI-compatible API — replace per-minute cloud STT by changing base_url.

2026-06-23

Which FunASR Model? Nano vs MLT-Nano vs SenseVoice vs Paraformer

A decision table with code: flagship Nano for zh/en/ja plus Chinese dialects/accents, separate MLT-Nano for 31 languages, and CPU choices SenseVoice or Paraformer.

2026-06-22

Lightweight Speech Recognition: Chinese ASR in ~250MB on CPU

One binary + a 254MB q8 model, no GPU/Python, 0.16s on CPU, 7.99% CER — smaller than whisper.cpp small and ~3x more accurate.

2026-06-21

Cantonese Speech Recognition in Python — SenseVoice Keeps Real Cantonese (Whisper Doesn't)

Real side-by-side on one Cantonese clip: SenseVoice keeps 呢/唔/嘅, Whisper rewrites to Mandarin. Native yue support + auto language ID, 3 lines of Python.

2026-06-21

FunASR vs faster-whisper: Chinese & Cantonese Compared

faster-whisper mislabels Cantonese as Mandarin + Japanese homophone errors; SenseVoice handles Cantonese natively, ~2.7x lower CER on Chinese.

2026-06-20

FunASR on llama.cpp — a whisper.cpp Alternative for Chinese ASR (CPU, no Python)

One self-contained binary, built-in VAD, any audio — download and transcribe Chinese; ~2.7x more accurate than whisper.cpp on CPU.

2026-06-18

Speech to Text in Python with FunASR

Transcribe audio in a few lines of Python — timestamps, speakers, batching. Local, free, no API key.

2026-06-18

Auto-Generate Subtitles (SRT & VTT) from Audio or Video with FunASR

One command for SRT, Python for VTT too, with speaker labels and real timestamps. Local, free, strong on Chinese.

2026-06-18

Self-Hosted OpenAI Whisper API Alternative with FunASR

funasr-server exposes an OpenAI-compatible /v1/audio/transcriptions; the OpenAI SDK works by changing only base_url. Local, free, private.

2026-06-18

Transcribe Audio from the Command Line with FunASR

One command -> text/SRT/JSON, --spk for speakers; plus funasr-server for an OpenAI-compatible API.

2026-06-17

Transcribe Long Audio with FunASR: Hours in One Call

Whisper caps at 30s; FunASR ingests any length via built-in VAD - 13 min in 4.3s (186x).

2026-06-17

Real-Time Streaming Speech-to-Text with FunASR

~600ms low-latency streaming ASR with chunks + cache, plus the 2-pass (streaming + offline) best practice.

2026-06-17

Beyond Transcription: Language, Emotion & Audio Events with SenseVoice

Transcription + language + emotion + audio events in one pass — the 4-in-1 Whisper cannot do.

2026-06-17

Speaker Diarization with FunASR: Who Spoke When

Transcription + speaker labels + timestamps in one generate() call. A pyannote+Whisper alternative, no HF gated access.

2026-06-16

FunASR vs Whisper: Real Chinese ASR Benchmark

Measured on 184 Chinese files (H100): SenseVoice 169.6x, 7.81% CER — full speed & accuracy data.

2026-06-16

Fun-ASR-Nano Guide: 800M End-to-End ASR LLM

Flagship for zh/en/ja plus 7 Chinese dialect groups and 26 accents; choose MLT-Nano for 31 languages.

2026-06-16

SenseVoice Deployment Guide: Five-Language ASR, Emotion and Audio Events

Multilingual ASR in 3 lines, with language/emotion/event tags, VAD, GPU/CPU.

More: Quickstart · Models

diff --git a/web-pages/product-site/legacy/sitemap.xml b/web-pages/product-site/legacy/sitemap.xml index ae14af506..d9ea3c623 100644 --- a/web-pages/product-site/legacy/sitemap.xml +++ b/web-pages/product-site/legacy/sitemap.xml @@ -12,6 +12,8 @@ https://www.funasr.com/en/ecosystem.html0.8weekly https://www.funasr.com/blog/0.9weekly https://www.funasr.com/en/blog/0.9weekly + https://www.funasr.com/blog/funclip-v2-2-0-moss-speaker-clipping.html2026-08-310.9monthly + https://www.funasr.com/en/blog/funclip-v2-2-0-moss-speaker-clipping.html2026-08-310.9monthly https://www.funasr.com/blog/funasr-v1-4-5-pypi-llama-cpp-release.html2026-08-280.9monthly https://www.funasr.com/en/blog/funasr-v1-4-5-pypi-llama-cpp-release.html2026-08-280.9monthly https://www.funasr.com/blog/funasr-v1-4-0-pypi-release.html2026-07-310.9monthly diff --git a/web-pages/product-site/tests/browser/product-site.spec.ts b/web-pages/product-site/tests/browser/product-site.spec.ts index d2364f7ff..2dae89fda 100644 --- a/web-pages/product-site/tests/browser/product-site.spec.ts +++ b/web-pages/product-site/tests/browser/product-site.spec.ts @@ -373,18 +373,20 @@ for (const viewport of [ { name: 'mobile', width: 390, height: 844 }, { name: 'desktop', width: 1440, height: 900 }, ]) { - test(`v1.4.5 release discovery is stable at ${viewport.name}`, async ({ page }, testInfo) => { + test(`FunClip v2.2.0 MOSS release discovery is stable at ${viewport.name}`, async ({ page }, testInfo) => { await page.setViewportSize(viewport); for (const release of [ { language: 'zh', index: '/blog/', - article: '/blog/funasr-v1-4-5-pypi-llama-cpp-release.html', + article: '/blog/funclip-v2-2-0-moss-speaker-clipping.html', + previous: '/blog/funasr-v1-4-5-pypi-llama-cpp-release.html', }, { language: 'en', index: '/en/blog/', - article: '/en/blog/funasr-v1-4-5-pypi-llama-cpp-release.html', + article: '/en/blog/funclip-v2-2-0-moss-speaker-clipping.html', + previous: '/en/blog/funasr-v1-4-5-pypi-llama-cpp-release.html', }, ]) { await page.goto(release.index); @@ -392,18 +394,22 @@ for (const viewport of [ page.locator(`.launch-feature a[href="${release.article}"]`), ).toBeVisible(); const history = page.locator('.previous-release .post-card'); - await expect(history).toHaveCount(3); + await expect(history).toHaveCount(4); + await expect( + page.locator(`.previous-release a[href="${release.previous}"]`), + ).toBeVisible(); const indexLayout = await history.evaluateAll((cards) => ({ overflow: document.documentElement.scrollWidth - document.documentElement.clientWidth, rows: new Set(cards.map((card) => Math.round(card.getBoundingClientRect().top))).size, })); expect(indexLayout.overflow).toBeLessThanOrEqual(1); - expect(indexLayout.rows).toBe(viewport.name === 'mobile' ? 3 : 1); + expect(indexLayout.rows).toBe(viewport.name === 'mobile' ? 4 : 1); await page.goto(release.article); - await expect(page.locator('h1')).toContainText('FunASR v1.4.5'); - await expect(page.getByText('funasr[knf]==1.4.5', { exact: false })).toBeVisible(); - await expect(page.getByText('runtime-llamacpp-v0.2.1', { exact: false })).toBeVisible(); + await expect(page.locator('h1')).toContainText('FunClip v2.2.0'); + await expect(page.getByText('OpenMOSS-Team/MOSS-Transcribe-Diarize', { exact: false }).first()).toBeVisible(); + await expect(page.getByText('/v1/audio/transcriptions', { exact: false }).first()).toBeVisible(); + await expect(page.locator('img[src="/img/funclip-v2-1-0-interface.jpg"]')).toBeVisible(); const articleLayout = await page.evaluate(() => { const navigation = document.querySelector('nav.nav'); @@ -419,7 +425,7 @@ for (const viewport of [ await page.screenshot({ path: testInfo.outputPath( - `v1.4.5-release-${release.language}-${viewport.name}.png`, + `funclip-v2.2.0-moss-${release.language}-${viewport.name}.png`, ), fullPage: true, }); diff --git a/web-pages/product-site/tests/test_legacy.py b/web-pages/product-site/tests/test_legacy.py index f920e8869..7274fe4ba 100644 --- a/web-pages/product-site/tests/test_legacy.py +++ b/web-pages/product-site/tests/test_legacy.py @@ -421,3 +421,53 @@ def test_public_pages_do_not_overstate_sensevoice_language_or_speed_claims(): violations.append(path.relative_to(LEGACY).as_posix()) assert violations == [] + + +def test_funclip_v220_moss_release_pages_are_bilingual_indexed_and_verifiable(): + slug = 'funclip-v2-2-0-moss-speaker-clipping.html' + pages = { + 'zh': LEGACY / 'blog' / slug, + 'en': LEGACY / 'en' / 'blog' / slug, + } + language_contracts = { + 'zh': ('第三方模型', '段级时间戳', '不接外部 VAD 或说话人模型'), + 'en': ('third-party model', 'segment-level timestamps', 'no external VAD or speaker model'), + } + + for language, path in pages.items(): + text = path.read_text(encoding='utf-8') + soup = BeautifulSoup(text, 'html.parser') + hrefs = {link.get('href') for link in soup.select('a[href]')} + + assert 'FunClip v2.2.0' in text + assert 'OpenMOSS-Team/MOSS-Transcribe-Diarize' in text + assert 'e8681d68e7042738ffca8ac8212bc8fcb1131ab8' in text + assert '/v1/audio/transcriptions' in text + assert 'response_format=json' in text + assert '994c5d9cf392b74b36284d526eca8bada1560a3e7825ab7baa9c673a1b4ef216' in text + assert '4f5a7d33d9ea65467f29b55b15ed5be18e64de7e57e2f9f36fe51a23b40557e7' in text + assert all(marker in text for marker in language_contracts[language]) + assert { + 'https://github.com/modelscope/FunClip/releases/tag/v2.2.0', + '/deploy/moss-transcribe-diarize.html' + if language == 'zh' + else '/en/deploy/moss-transcribe-diarize.html', + } <= hrefs + + route = f'/{"" if language == "zh" else "en/"}blog/{slug}' + peer = f'/{"en/" if language == "zh" else ""}blog/{slug}' + assert soup.select_one('link[rel="canonical"]')['href'].endswith(route) + assert soup.select_one(f'link[rel="alternate"][href$="{peer}"]') + image = soup.select_one('article img[src="/img/funclip-v2-1-0-interface.jpg"]') + assert image + metadata = json.loads(soup.select_one('script[type="application/ld+json"]').string) + assert metadata['datePublished'] == '2026-08-31' + assert metadata['dateModified'] == '2026-08-31' + + zh_index = (LEGACY / 'blog' / 'index.html').read_text(encoding='utf-8') + en_index = (LEGACY / 'en' / 'blog' / 'index.html').read_text(encoding='utf-8') + sitemap = (LEGACY / 'sitemap.xml').read_text(encoding='utf-8') + assert f'/blog/{slug}' in zh_index + assert f'/en/blog/{slug}' in en_index + assert f'https://www.funasr.com/blog/{slug}' in sitemap + assert f'https://www.funasr.com/en/blog/{slug}' in sitemap diff --git a/web-pages/product-site/tests/test_output.py b/web-pages/product-site/tests/test_output.py index 32a7b0ce8..e470f7797 100644 --- a/web-pages/product-site/tests/test_output.py +++ b/web-pages/product-site/tests/test_output.py @@ -609,22 +609,61 @@ def test_v1_4_3_release_blog_is_bilingual_and_verifiable( @pytest.mark.parametrize( - ('relative', 'href'), + ('relative', 'feature_href', 'history_href'), ( - ('blog/index.html', '/blog/funasr-v1-4-5-pypi-llama-cpp-release.html'), - ('en/blog/index.html', '/en/blog/funasr-v1-4-5-pypi-llama-cpp-release.html'), + ( + 'blog/index.html', + '/blog/funclip-v2-2-0-moss-speaker-clipping.html', + '/blog/funasr-v1-4-5-pypi-llama-cpp-release.html', + ), + ( + 'en/blog/index.html', + '/en/blog/funclip-v2-2-0-moss-speaker-clipping.html', + '/en/blog/funasr-v1-4-5-pypi-llama-cpp-release.html', + ), ), ) def test_blog_index_features_latest_release_and_preserves_history( - built_site, relative, href + built_site, relative, feature_href, history_href ): soup = read_soup(built_site / relative) - feature = soup.select_one(f'.launch-feature a[href="{href}"]') + feature = soup.select_one(f'.launch-feature a[href="{feature_href}"]') assert feature - assert 'v1.4.5' in feature.get_text(' ', strip=True) + assert 'FunClip v2.2.0' in feature.get_text(' ', strip=True) history = soup.select_one('.previous-release') assert history + assert history.select_one(f'a[href="{history_href}"]') history_text = history.get_text(' ', strip=True) + assert 'v1.4.5' in history_text assert 'v1.4.3' in history_text assert 'v1.4.0' in history_text + + +@pytest.mark.parametrize( + ('relative', 'peer', 'guide'), + ( + ( + 'blog/funclip-v2-2-0-moss-speaker-clipping.html', + '/en/blog/funclip-v2-2-0-moss-speaker-clipping.html', + '/deploy/moss-transcribe-diarize.html', + ), + ( + 'en/blog/funclip-v2-2-0-moss-speaker-clipping.html', + '/blog/funclip-v2-2-0-moss-speaker-clipping.html', + '/en/deploy/moss-transcribe-diarize.html', + ), + ), +) +def test_funclip_v220_blog_builds_with_real_media_and_product_routes( + built_site, relative, peer, guide +): + soup = read_soup(built_site / relative) + assert soup.select_one(f'link[rel="alternate"][href$="{peer}"]') + image = soup.select_one('article img[src]') + assert image + assert (built_site / image['src'].lstrip('/')).is_file() + assert soup.select_one(f'a[href="{guide}"]') + assert soup.select_one( + 'a[href="https://github.com/modelscope/FunClip/releases/tag/v2.2.0"]' + )