Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
27 commits
Select commit Hold shift + click to select a range
f325481
Add livekit-plugins-funasr: pyproject.toml
LauraGPT Jun 21, 2026
9b7295a
Add livekit-plugins-funasr: README.md
LauraGPT Jun 21, 2026
54c3ec6
Add livekit-plugins-funasr: __init__.py
LauraGPT Jun 21, 2026
8082fe4
Add livekit-plugins-funasr: stt.py
LauraGPT Jun 21, 2026
1e5063e
Add livekit-plugins-funasr: version.py
LauraGPT Jun 21, 2026
1b43298
Add livekit-plugins-funasr: log.py
LauraGPT Jun 21, 2026
dc90195
Register livekit-plugins-funasr workspace source
LauraGPT Jun 21, 2026
654b394
Fix CI: ruff format, type annotations, LanguageCode for SpeechData
LauraGPT Jun 21, 2026
5020f4b
Add py.typed marker so the package is type-checkable
LauraGPT Jun 21, 2026
3e70f06
Address review: real model name, drop nospeech as language, thread-sa…
LauraGPT Jun 21, 2026
348c46f
Address FunASR STT review feedback
LauraGPT Jun 30, 2026
8e5a871
Merge upstream main into add-funasr-stt-plugin
LauraGPT Jul 13, 2026
afab30a
fix(funasr): complete plugin packaging
LauraGPT Jul 13, 2026
3aba049
Merge remote-tracking branch 'upstream/main' into codex/refresh-livek…
LauraGPT Jul 16, 2026
666d540
Merge upstream/main into add-funasr-stt-plugin
LauraGPT Jul 17, 2026
cf327cc
Merge remote-tracking branch 'upstream/main' into codex/refresh-livek…
LauraGPT Jul 17, 2026
ef94cb8
Merge upstream main into FunASR plugin PR
LauraGPT Jul 18, 2026
bb30956
Align FunASR plugin version with agents extra
LauraGPT Jul 18, 2026
e07dd96
Merge remote-tracking branch 'upstream/main' into codex/refresh-livek…
LauraGPT Jul 18, 2026
cd6ec1d
Merge upstream main into FunASR plugin PR
LauraGPT Jul 22, 2026
07295ee
Merge remote FunASR plugin updates
LauraGPT Jul 22, 2026
9f995f7
Fix LiveKit CI dependency locks after upstream merge
LauraGPT Jul 22, 2026
8cc481c
Merge upstream main and refresh FunASR plugin version
LauraGPT Jul 26, 2026
553e0d7
Merge upstream main and refresh FunASR plugin
LauraGPT Aug 3, 2026
b181f02
Fix FunASR model loading lifecycle
LauraGPT Aug 3, 2026
38d2558
Keep FunASR inference serialized after cancellation
LauraGPT Aug 3, 2026
754b93c
Merge upstream main into add-funasr-stt-plugin
LauraGPT Aug 4, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions livekit-agents/pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -86,6 +86,7 @@ elevenlabs = ["livekit-plugins-elevenlabs>=1.6.8"]
fal = ["livekit-plugins-fal>=1.6.8"]
fishaudio = ["livekit-plugins-fishaudio>=1.6.8"]
fireworksai = ["livekit-plugins-fireworksai>=1.6.8"]
funasr = ["livekit-plugins-funasr>=1.6.8"]
gladia = ["livekit-plugins-gladia>=1.6.8"]
gnani = ["livekit-plugins-gnani>=1.6.8"]
google = ["livekit-plugins-google>=1.6.8"]
Expand Down
25 changes: 25 additions & 0 deletions livekit-plugins/livekit-plugins-funasr/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
# LiveKit Plugins FunASR

Agent Framework plugin for local speech-to-text with [FunASR](https://github.com/modelscope/FunASR) models such as [SenseVoice](https://github.com/FunAudioLLM/SenseVoice).

SenseVoice is an open-source, fully-local, non-autoregressive multilingual ASR model (Chinese, Cantonese, English, Japanese, Korean and more) with leading Chinese accuracy and fast inference. The model runs locally, so no API key is required.

## Installation

```bash
pip install livekit-plugins-funasr
```

## Usage

```python
from livekit.plugins import funasr

stt = funasr.STT(model="iic/SenseVoiceSmall")
```

The first run downloads the model from ModelScope/Hugging Face. Use `language=None` (default) for automatic language detection, or set e.g. `language="zh"`.

CPU inference works with the default installation. For GPU inference, install a
PyTorch and torchaudio build that matches your CUDA runtime, then pass
`device="cuda"`.
Comment thread
devin-ai-integration[bot] marked this conversation as resolved.
Original file line number Diff line number Diff line change
@@ -0,0 +1,44 @@
# Copyright 2024 LiveKit, Inc.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

"""FunASR plugin for LiveKit Agents.

Local, fully-offline multilingual speech-to-text using FunASR models such as
SenseVoice (Chinese, Cantonese, English, Japanese, Korean and more).
See https://github.com/modelscope/FunASR for more information.
"""

from .stt import _DEFAULT_MODEL, FunASRSTT, FunASRSTT as STT, _load_model
from .version import __version__

__all__ = ["FunASRSTT", "STT", "__version__"]

from livekit.agents import Plugin

from .log import logger


class FunASRPlugin(Plugin):
"""Register the FunASR integration and its model-download hook."""

def __init__(self) -> None:
"""Create the LiveKit plugin registration."""
super().__init__(__name__, __version__, __package__, logger)

def download_files(self) -> None:
"""Download the default SenseVoice model for offline agent startup."""
_load_model(_DEFAULT_MODEL, "cpu")


Plugin.register_plugin(FunASRPlugin())
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
import logging

logger = logging.getLogger("livekit.plugins.funasr")
Empty file.
183 changes: 183 additions & 0 deletions livekit-plugins/livekit-plugins-funasr/livekit/plugins/funasr/stt.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,183 @@
# Copyright 2024 LiveKit, Inc.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

from __future__ import annotations

import asyncio
import re
import threading
from dataclasses import dataclass
from typing import Any

import numpy as np

from livekit import rtc
from livekit.agents import APIConnectionError, APIConnectOptions, LanguageCode, stt
from livekit.agents.stt import SpeechEventType, STTCapabilities
from livekit.agents.types import NOT_GIVEN, NotGivenOr
from livekit.agents.utils import AudioBuffer, is_given

from .log import logger

try:
from funasr import AutoModel # type: ignore
from funasr.utils.postprocess_utils import rich_transcription_postprocess # type: ignore
except ImportError as e:
raise ImportError(
"funasr is required for the FunASR plugin. Install it with: pip install funasr"
) from e

# Languages natively supported by SenseVoice; anything else falls back to auto-detect.
_FUNASR_LANGUAGES = {"zh", "en", "ja", "ko", "yue", "nospeech"}
# Spoken-language codes reported as the detected language (excludes "nospeech",
# which is a SenseVoice classification label, not a language).
_DETECTED_LANGUAGES = {"zh", "en", "ja", "ko", "yue"}
_SAMPLE_RATE = 16000
_LANG_TAG_RE = re.compile(r"<\|([a-z]+)\|>")
_DEFAULT_MODEL = "iic/SenseVoiceSmall"


def _load_model(model: str, device: str) -> Any:
logger.info(f"loading FunASR model {model} on {device}...")
loaded_model = AutoModel(model=model, device=device, disable_update=True)
logger.info("FunASR model loaded")
return loaded_model


def _normalize_language(language: NotGivenOr[str]) -> str:
if not is_given(language) or not language:
return "auto"
code = str(language).split("-")[0].lower()
return code if code in _FUNASR_LANGUAGES else "auto"


@dataclass
class _STTOptions:
language: str = "auto"
use_itn: bool = True


class FunASRSTT(stt.STT):
"""Local speech-to-text using a FunASR model such as SenseVoice.

SenseVoice is an open-source, fully-local, non-autoregressive multilingual ASR
model (Chinese, Cantonese, English, Japanese, Korean and more) with strong
Chinese accuracy and fast inference. The model runs locally; no API key needed.
"""

def __init__(
self,
*,
model: str = _DEFAULT_MODEL,
device: str = "cpu",
language: NotGivenOr[str] = NOT_GIVEN,
use_itn: bool = True,
) -> None:
"""Create a FunASR STT instance.

Args:
model: FunASR model id on ModelScope/Hugging Face (default
``"iic/SenseVoiceSmall"``).
device: Inference device, ``"cpu"`` or ``"cuda"``.
language: Default language. When not given, the language is
auto-detected per utterance.
use_itn: Apply inverse text normalization (e.g. "nine" -> "9").
"""
super().__init__(capabilities=STTCapabilities(streaming=False, interim_results=False))
self._model_name = model
self._device = device
self._opts = _STTOptions(language=_normalize_language(language), use_itn=use_itn)
# FunASR's model.generate is not guaranteed thread-safe; serialize access
# and lazy initialization across calls that share this instance.
self._lock = threading.Lock()
self._model: Any | None = None

@property
def model(self) -> str:
"""Return the configured FunASR model identifier."""
return self._model_name

@property
def provider(self) -> str:
"""Return the speech-to-text provider name."""
return "FunASR"

def update_options(
self,
*,
language: NotGivenOr[str] = NOT_GIVEN,
use_itn: NotGivenOr[bool] = NOT_GIVEN,
) -> None:
"""Update recognition options used for subsequent requests.

Args:
language: Language code, or leave unset to keep the current value.
use_itn: Whether to apply inverse text normalization.
"""
if is_given(language):
self._opts.language = _normalize_language(language)
if is_given(use_itn):
self._opts.use_itn = use_itn
Comment thread
LauraGPT marked this conversation as resolved.

async def _recognize_impl(
self,
buffer: AudioBuffer,
*,
language: NotGivenOr[str] = NOT_GIVEN,
conn_options: APIConnectOptions,
) -> stt.SpeechEvent:
lang = _normalize_language(language) if is_given(language) else self._opts.language

combined = rtc.combine_audio_frames(buffer)
channels = combined.num_channels
if combined.sample_rate != _SAMPLE_RATE:
resampler = rtc.AudioResampler(
input_rate=combined.sample_rate,
output_rate=_SAMPLE_RATE,
num_channels=channels,
quality=rtc.AudioResamplerQuality.HIGH,
)
Comment thread
LauraGPT marked this conversation as resolved.
frames = list(resampler.push(combined)) + list(resampler.flush())
data = b"".join(bytes(f.data) for f in frames)
else:
data = bytes(combined.data)
samples = np.frombuffer(data, dtype=np.int16).astype(np.float32) / 32768.0
if channels > 1:
samples = samples.reshape(-1, channels).mean(axis=1)

def _run() -> str:
with self._lock:
if self._model is None:
self._model = _load_model(self._model_name, self._device)
result = self._model.generate(
input=samples,
cache={},
language=lang,
use_itn=self._opts.use_itn,
)
return result[0]["text"] if result else ""

try:
raw = await asyncio.to_thread(_run)
except Exception as e:
raise APIConnectionError("failed to run FunASR inference", retryable=False) from e
Comment thread
LauraGPT marked this conversation as resolved.

text = rich_transcription_postprocess(raw).strip()
m = _LANG_TAG_RE.match(raw)
detected = m.group(1) if m and m.group(1) in _DETECTED_LANGUAGES else ""

return stt.SpeechEvent(
type=SpeechEventType.FINAL_TRANSCRIPT,
alternatives=[stt.SpeechData(text=text, language=LanguageCode(detected))],
)
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
# Copyright 2024 LiveKit, Inc.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

__version__ = "1.6.8"
49 changes: 49 additions & 0 deletions livekit-plugins/livekit-plugins-funasr/pyproject.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,49 @@
[build-system]
requires = ["hatchling"]
build-backend = "hatchling.build"

[project]
name = "livekit-plugins-funasr"
dynamic = ["version"]
description = "FunASR (SenseVoice) local STT plugin for LiveKit Agents"
readme = "README.md"
license = "Apache-2.0"
requires-python = ">=3.10.0"
authors = [{ name = "LiveKit" }]
keywords = ["voice", "ai", "realtime", "audio", "video", "livekit", "webrtc"]
classifiers = [
"Intended Audience :: Developers",
"License :: OSI Approved :: Apache Software License",
"Topic :: Multimedia :: Sound/Audio",
"Topic :: Multimedia :: Video",
"Topic :: Scientific/Engineering :: Artificial Intelligence",
"Programming Language :: Python :: 3",
"Programming Language :: Python :: 3.10",
"Programming Language :: Python :: 3 :: Only",
]
dependencies = [
"livekit-agents>=1.6.8",
"funasr>=1.3.16",
"numba>=0.61.0",
"numpy",
"torch>=2.0",
"torchaudio>=2.0",
]

[project.urls]
Documentation = "https://docs.livekit.io"
Website = "https://livekit.io/"
Source = "https://github.com/livekit/agents"

[tool.hatch.version]
path = "livekit/plugins/funasr/version.py"

[tool.hatch.build.targets.wheel]
packages = ["livekit"]

[tool.hatch.build.targets.sdist]
include = ["/livekit"]

[tool.uv]
exclude-newer = "7 days"
exclude-newer-package = { livekit-agents = "0 days" }
1 change: 1 addition & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@ livekit-plugins-elevenlabs = { workspace = true }
livekit-plugins-fal = { workspace = true }
livekit-plugins-fireworksai = { workspace = true }
livekit-plugins-fishaudio = { workspace = true }
livekit-plugins-funasr = { workspace = true }
livekit-plugins-gladia = { workspace = true }
livekit-plugins-gnani = { workspace = true }
livekit-plugins-google = { workspace = true }
Expand Down
Loading
Loading