From 60f05cebd8e7968621a0ce015cf72987ec7577c7 Mon Sep 17 00:00:00 2001 From: "Somhairle H. Marisol" Date: Tue, 29 Sep 2026 10:50:12 +0800 Subject: feat(recorder): add offline voice clip pipeline MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit [变更性质] - 本提交新增本机离线的语音片段录音能力,不涉及网络服务或语音识别。 [新增功能] - 通过 Pulse 默认输入持续采集音频,以本地 Silero VAD 触发片段。 - 以私有目录和权限写入 24 kbps Ogg/Opus 录音,并提供 systemd 用户服务安装器。 [实现方案] - 使用有限前置缓冲、静音封段和最长段滚动状态机,避免静音落盘及内存无限增长。 - 增加真实 Opus 编解码、服务 dry-run 和纯逻辑状态转换测试;记录模型归属和部署前置条件。 [影响范围] - 新增 Python CLI、运行时模块、中文运维文档和自动测试。 - 未安装、启用或启动任何用户 systemd 服务;不删除或修改录音数据。 --- src/mic_clipper/__init__.py | 1 + src/mic_clipper/assets/__init__.py | 1 + src/mic_clipper/audio_input.py | 72 ++++++++++++++++++++++ src/mic_clipper/cli.py | 66 ++++++++++++++++++++ src/mic_clipper/runner.py | 92 ++++++++++++++++++++++++++++ src/mic_clipper/segmenter.py | 120 +++++++++++++++++++++++++++++++++++++ src/mic_clipper/service.py | 101 +++++++++++++++++++++++++++++++ src/mic_clipper/storage.py | 111 ++++++++++++++++++++++++++++++++++ src/mic_clipper/vad.py | 59 ++++++++++++++++++ 9 files changed, 623 insertions(+) create mode 100644 src/mic_clipper/__init__.py create mode 100644 src/mic_clipper/assets/__init__.py create mode 100644 src/mic_clipper/audio_input.py create mode 100644 src/mic_clipper/cli.py create mode 100644 src/mic_clipper/runner.py create mode 100644 src/mic_clipper/segmenter.py create mode 100644 src/mic_clipper/service.py create mode 100644 src/mic_clipper/storage.py create mode 100644 src/mic_clipper/vad.py (limited to 'src/mic_clipper') diff --git a/src/mic_clipper/__init__.py b/src/mic_clipper/__init__.py new file mode 100644 index 0000000..7d39855 --- /dev/null +++ b/src/mic_clipper/__init__.py @@ -0,0 +1 @@ +"""Offline, voice-activated microphone clip recording.""" diff --git a/src/mic_clipper/assets/__init__.py b/src/mic_clipper/assets/__init__.py new file mode 100644 index 0000000..480108b --- /dev/null +++ b/src/mic_clipper/assets/__init__.py @@ -0,0 +1 @@ +"""Bundled, offline model assets.""" diff --git a/src/mic_clipper/audio_input.py b/src/mic_clipper/audio_input.py new file mode 100644 index 0000000..1d77aa0 --- /dev/null +++ b/src/mic_clipper/audio_input.py @@ -0,0 +1,72 @@ +"""PipeWire/Pulse microphone capture through the local ffmpeg binary.""" + +from __future__ import annotations + +import subprocess +from collections.abc import Iterator + +import numpy as np + +from mic_clipper.vad import FRAME_SAMPLES + + +class AudioInputUnavailable(RuntimeError): + """The dynamic Pulse default source could not supply another frame.""" + + +class PulseAudioInput: + """Yield fixed 16 kHz mono float PCM frames from Pulse's `default` source.""" + + @staticmethod + def command() -> list[str]: + return [ + "ffmpeg", + "-nostdin", + "-hide_banner", + "-loglevel", + "error", + "-f", + "pulse", + "-sample_rate", + "16000", + "-channels", + "1", + "-i", + "default", + "-ac", + "1", + "-ar", + "16000", + "-f", + "f32le", + "pipe:1", + ] + + def frames(self) -> Iterator[np.ndarray]: + process = subprocess.Popen( + self.command(), + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + ) + assert process.stdout is not None + assert process.stderr is not None + try: + while True: + payload = process.stdout.read(FRAME_SAMPLES * np.dtype(" argparse.ArgumentParser: + parser = argparse.ArgumentParser(description="Offline voice-activated microphone clips") + commands = parser.add_subparsers(dest="command", required=True) + + run = commands.add_parser("run", help="listen and save speech clips") + run.add_argument("--output-dir", type=Path, default=DEFAULT_OUTPUT) + run.add_argument("--threshold", type=float, default=0.5) + + service_parser = commands.add_parser("service", help="manage the systemd user service") + service_commands = service_parser.add_subparsers(dest="service_command", required=True) + install = service_commands.add_parser("install", help="install and enable the user service") + install.add_argument("--dry-run", action="store_true") + install.add_argument("--unit-path", type=Path, default=DEFAULT_UNIT) + install.add_argument("--project-root", type=Path, default=PROJECT_ROOT) + install.add_argument("--python", type=Path, default=Path(sys.executable)) + uninstall = service_commands.add_parser("uninstall", help="remove only this tool's service") + uninstall.add_argument("--unit-path", type=Path, default=DEFAULT_UNIT) + service_commands.add_parser("status", help="show systemd user service status") + return parser + + +def main(arguments: list[str] | None = None) -> int: + args = build_parser().parse_args(arguments) + if args.command == "run": + if not 0 < args.threshold <= 1: + raise SystemExit("--threshold must be between 0 and 1") + logging.basicConfig(level=logging.INFO, format="%(levelname)s %(message)s") + run_forever(args.output_dir, threshold=args.threshold) + return 0 + if args.service_command == "install": + commands = service.install( + unit_path=args.unit_path, + python=args.python, + project_root=args.project_root, + dry_run=args.dry_run, + ) + if args.dry_run: + for command in commands: + print(" ".join(command)) + return 0 + if args.service_command == "uninstall": + service.uninstall(unit_path=args.unit_path) + return 0 + return service.status() + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/mic_clipper/runner.py b/src/mic_clipper/runner.py new file mode 100644 index 0000000..787bf58 --- /dev/null +++ b/src/mic_clipper/runner.py @@ -0,0 +1,92 @@ +"""Runtime wiring for capture, VAD, segmentation, and durable clips.""" + +from __future__ import annotations + +import logging +import time +from collections.abc import Callable, Iterable +from datetime import datetime +from pathlib import Path +from typing import Protocol + +import numpy as np + +from mic_clipper.audio_input import PulseAudioInput +from mic_clipper.segmenter import Audio, End, Segmenter, Start +from mic_clipper.storage import ClipWriter, OpenClip +from mic_clipper.vad import SileroVad + + +LOGGER = logging.getLogger(__name__) + + +class Vad(Protocol): + def probability(self, samples: np.ndarray) -> float: ... + + +class Writer(Protocol): + def open(self, started_at: datetime) -> OpenClip: ... + + +def record_frames( + frames: Iterable[np.ndarray], + *, + vad: Vad, + writer: Writer, + segmenter: Segmenter, + now: Callable[[], datetime], + threshold: float = 0.5, +) -> None: + """Record one finite or failing frame stream, closing an active encoder on exit.""" + clip: OpenClip | None = None + try: + for samples in frames: + events = segmenter.push(samples, vad.probability(samples) >= threshold, now()) + for event in events: + if isinstance(event, Start): + clip = writer.open(event.started_at) + try: + clip.write(event.samples) + except BaseException: + clip.abort() + clip = None + raise + elif isinstance(event, Audio): + if clip is None: + raise RuntimeError("segment audio arrived without an active encoder") + clip.write(event.samples) + elif isinstance(event, End): + if clip is None: + raise RuntimeError("segment end arrived without an active encoder") + clip.close() + clip = None + finally: + if clip is not None: + clip.close() + + +def run_forever(root: Path, *, threshold: float = 0.5) -> None: + """Recreate all live components after a transient audio-service failure.""" + retry_seconds = 1 + while True: + try: + record_frames( + PulseAudioInput().frames(), + vad=SileroVad(), + writer=ClipWriter(root), + segmenter=Segmenter(pre_roll_seconds=0.5, silence_seconds=1.5), + now=lambda: datetime.now().astimezone(), + threshold=threshold, + ) + raise RuntimeError("Pulse input ended unexpectedly") + except KeyboardInterrupt: + LOGGER.info("microphone clip recorder stopped") + return + except Exception as error: + LOGGER.warning( + "audio recorder unavailable (%s); retrying in %s seconds", + error, + retry_seconds, + ) + time.sleep(retry_seconds) + retry_seconds = min(retry_seconds * 2, 60) diff --git a/src/mic_clipper/segmenter.py b/src/mic_clipper/segmenter.py new file mode 100644 index 0000000..aa0e3b2 --- /dev/null +++ b/src/mic_clipper/segmenter.py @@ -0,0 +1,120 @@ +"""Pure streaming voice-clip state machine.""" + +from __future__ import annotations + +from collections import deque +from dataclasses import dataclass +from datetime import datetime, timedelta + +import numpy as np + + +SAMPLE_RATE = 16_000 + + +@dataclass(frozen=True) +class Start: + started_at: datetime + samples: np.ndarray + + +@dataclass(frozen=True) +class Audio: + samples: np.ndarray + + +@dataclass(frozen=True) +class End: + pass + + +class Segmenter: + """Emit stream events while retaining only the configured pre-roll in memory.""" + + def __init__( + self, + *, + pre_roll_seconds: float, + silence_seconds: float, + max_segment_seconds: float = 600, + sample_rate: int = SAMPLE_RATE, + ) -> None: + self.sample_rate = sample_rate + self.pre_roll_samples = round(pre_roll_seconds * sample_rate) + self.silence_samples_limit = round(silence_seconds * sample_rate) + self.max_segment_samples = round(max_segment_seconds * sample_rate) + self._pre_roll: deque[np.ndarray] = deque() + self._pre_roll_size = 0 + self._active = False + self._segment_samples = 0 + self._silence_samples = 0 + + @property + def active(self) -> bool: + return self._active + + def push( + self, samples: np.ndarray, is_speech: bool, captured_at: datetime + ) -> tuple[Start | Audio | End, ...]: + samples = np.asarray(samples, dtype=np.float32) + if samples.ndim != 1: + raise ValueError("audio must be a mono, one-dimensional array") + if not len(samples): + return () + + if not self._active: + if not is_speech: + self._append_pre_roll(samples) + return () + pre_roll = self._take_pre_roll() + started_at = captured_at - timedelta( + seconds=len(pre_roll) / self.sample_rate + ) + self._active = True + self._segment_samples = len(pre_roll) + len(samples) + self._silence_samples = 0 + self._append_pre_roll(samples) + return (Start(started_at, np.concatenate((pre_roll, samples))),) + + if self._segment_samples + len(samples) > self.max_segment_samples: + self._append_pre_roll(samples) + self._active = True + self._segment_samples = len(samples) + self._silence_samples = 0 + return (End(), Start(captured_at, samples)) + + self._segment_samples += len(samples) + if is_speech: + self._silence_samples = 0 + else: + self._silence_samples += len(samples) + self._append_pre_roll(samples) + + events: tuple[Start | Audio | End, ...] = (Audio(samples),) + if self._silence_samples >= self.silence_samples_limit: + self._active = False + self._segment_samples = 0 + self._silence_samples = 0 + events += (End(),) + return events + + def _append_pre_roll(self, samples: np.ndarray) -> None: + self._pre_roll.append(samples) + self._pre_roll_size += len(samples) + while self._pre_roll_size > self.pre_roll_samples: + excess = self._pre_roll_size - self.pre_roll_samples + oldest = self._pre_roll[0] + if len(oldest) <= excess: + self._pre_roll.popleft() + self._pre_roll_size -= len(oldest) + else: + self._pre_roll[0] = oldest[excess:] + self._pre_roll_size -= excess + + def _take_pre_roll(self) -> np.ndarray: + if not self._pre_roll: + return np.empty(0, dtype=np.float32) + samples = np.concatenate(tuple(self._pre_roll)) + self._pre_roll.clear() + self._pre_roll_size = 0 + return samples diff --git a/src/mic_clipper/service.py b/src/mic_clipper/service.py new file mode 100644 index 0000000..ecf5673 --- /dev/null +++ b/src/mic_clipper/service.py @@ -0,0 +1,101 @@ +"""Install and remove the systemd user service without touching recordings.""" + +from __future__ import annotations + +import os +import subprocess +import tempfile +from collections.abc import Callable +from pathlib import Path + + +UNIT_NAME = "mic-clipper.service" + + +def render_unit(python: Path, project_root: Path) -> str: + return f"""[Unit] +Description=Offline voice-activated microphone clip recorder +Wants=pipewire.service wireplumber.service pipewire-pulse.service +After=pipewire.service wireplumber.service pipewire-pulse.service + +[Service] +Type=simple +WorkingDirectory={project_root} +Environment=PYTHONPATH={project_root / 'src'} +ExecStart={python} -m mic_clipper run +Restart=on-failure +RestartSec=5s +RestartSteps=5 +RestartMaxDelaySec=60s +StartLimitIntervalSec=300 +StartLimitBurst=10 +NoNewPrivileges=yes +PrivateTmp=yes +UMask=0077 + +[Install] +WantedBy=default.target +""" + + +def install( + *, + unit_path: Path, + python: Path, + project_root: Path, + dry_run: bool = False, + run_command: Callable[[list[str]], object] | None = None, +) -> list[list[str]]: + commands = [ + ["systemctl", "--user", "daemon-reload"], + ["systemctl", "--user", "enable", "--now", UNIT_NAME], + ] + if dry_run: + return commands + unit_path.parent.mkdir(mode=0o700, parents=True, exist_ok=True) + _write_private(unit_path, render_unit(python, project_root)) + runner = run_command or _run_command + for command in commands: + runner(command) + return commands + + +def uninstall( + *, + unit_path: Path, + run_command: Callable[[list[str]], object] | None = None, +) -> list[list[str]]: + if not unit_path.exists(): + return [] + commands = [ + ["systemctl", "--user", "disable", "--now", UNIT_NAME], + ["systemctl", "--user", "daemon-reload"], + ] + runner = run_command or _run_command + for command in commands[:1]: + runner(command) + unit_path.unlink() + runner(commands[1]) + return commands + + +def status() -> int: + return subprocess.run( + ["systemctl", "--user", "status", UNIT_NAME], check=False + ).returncode + + +def _write_private(path: Path, content: str) -> None: + with tempfile.NamedTemporaryFile( + "w", encoding="utf-8", dir=path.parent, prefix=f".{path.name}.", delete=False + ) as temporary: + temporary.write(content) + temporary.flush() + os.fchmod(temporary.fileno(), 0o600) + temporary_path = Path(temporary.name) + os.replace(temporary_path, path) + os.chmod(path, 0o600) + + +def _run_command(command: list[str]) -> None: + subprocess.run(command, check=True) diff --git a/src/mic_clipper/storage.py b/src/mic_clipper/storage.py new file mode 100644 index 0000000..a7f00bd --- /dev/null +++ b/src/mic_clipper/storage.py @@ -0,0 +1,111 @@ +"""Private, streaming Opus clip storage.""" + +from __future__ import annotations + +import os +import subprocess +from dataclasses import dataclass +from datetime import datetime +from pathlib import Path + +import numpy as np + + +SAMPLE_RATE = 16_000 + + +@dataclass +class OpenClip: + path: Path + process: subprocess.Popen[bytes] + closed: bool = False + + def write(self, samples: np.ndarray) -> None: + if self.closed: + raise RuntimeError("cannot write a closed clip") + assert self.process.stdin is not None + audio = np.asarray(samples, dtype=" None: + if self.closed: + return + self.closed = True + assert self.process.stdin is not None + self.process.stdin.close() + assert self.process.stderr is not None + error_output = self.process.stderr.read().decode("utf-8", errors="replace").strip() + return_code = self.process.wait() + if return_code: + self.path.unlink(missing_ok=True) + raise RuntimeError(f"ffmpeg Opus encoding failed ({return_code}): {error_output}") + os.chmod(self.path, 0o600) + + def abort(self) -> None: + if self.closed: + return + self.closed = True + self.process.kill() + self.process.wait() + self.path.unlink(missing_ok=True) + + +class ClipWriter: + """Open one private Ogg/Opus file only after the segmenter emits a start.""" + + def __init__(self, root: Path) -> None: + self.root = Path(root).expanduser() + + def open(self, started_at: datetime) -> OpenClip: + self.root.mkdir(mode=0o700, parents=True, exist_ok=True) + os.chmod(self.root, 0o700) + destination = self.root / started_at.date().isoformat() + destination.mkdir(mode=0o700, exist_ok=True) + os.chmod(destination, 0o700) + path = self._reserve_path(destination, started_at) + process = subprocess.Popen( + [ + "ffmpeg", + "-nostdin", + "-v", + "error", + "-f", + "f32le", + "-ar", + str(SAMPLE_RATE), + "-ac", + "1", + "-i", + "pipe:0", + "-c:a", + "libopus", + "-b:a", + "24k", + "-application", + "voip", + "-vbr", + "constrained", + "-f", + "ogg", + "-y", + str(path), + ], + stdin=subprocess.PIPE, + stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE, + ) + return OpenClip(path, process) + + @staticmethod + def _reserve_path(destination: Path, started_at: datetime) -> Path: + base_name = started_at.strftime("%H-%M-%S") + for suffix in range(10_000): + name = f"{base_name}.opus" if suffix == 0 else f"{base_name}-{suffix:02d}.opus" + path = destination / name + try: + descriptor = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600) + except FileExistsError: + continue + os.close(descriptor) + return path + raise RuntimeError("unable to reserve a unique clip filename") diff --git a/src/mic_clipper/vad.py b/src/mic_clipper/vad.py new file mode 100644 index 0000000..705dfd8 --- /dev/null +++ b/src/mic_clipper/vad.py @@ -0,0 +1,59 @@ +"""Stateful, local Silero VAD ONNX inference.""" + +from __future__ import annotations + +from importlib import resources +from pathlib import Path + +import numpy as np +import onnxruntime + + +FRAME_SAMPLES = 512 +CONTEXT_SAMPLES = 64 + + +class SileroVad: + """Run the bundled Silero v6 model on exact 32 ms PCM frames.""" + + def __init__(self, model_path: Path | None = None) -> None: + self.model_path = Path(model_path) if model_path else self._default_model_path() + options = onnxruntime.SessionOptions() + options.inter_op_num_threads = 1 + options.intra_op_num_threads = 1 + options.enable_cpu_mem_arena = False + options.log_severity_level = 4 + self.session = onnxruntime.InferenceSession( + self.model_path, + providers=["CPUExecutionProvider"], + sess_options=options, + ) + self._context = np.zeros(CONTEXT_SAMPLES, dtype=np.float32) + self._h = np.zeros((1, 1, 128), dtype=np.float32) + self._c = np.zeros((1, 1, 128), dtype=np.float32) + + def probability(self, samples: np.ndarray) -> float: + samples = np.asarray(samples, dtype=np.float32) + if samples.ndim != 1 or len(samples) != FRAME_SAMPLES: + raise ValueError("Silero VAD requires exactly 512 mono samples at 16 kHz") + model_input = np.concatenate((self._context, samples)).reshape(1, -1) + output, self._h, self._c = self.session.run( + None, + {"input": model_input, "h": self._h, "c": self._c}, + ) + self._context = samples[-CONTEXT_SAMPLES:].copy() + return float(output[0]) + + @staticmethod + def _default_model_path() -> Path: + bundled = resources.files("mic_clipper.assets").joinpath("silero_vad_v6.onnx") + if bundled.is_file(): + return Path(bundled) + try: + import faster_whisper + except ImportError as error: + raise RuntimeError( + "Silero VAD model is missing; bundle silero_vad_v6.onnx or install the " + "pre-provisioned faster-whisper runtime asset" + ) from error + return Path(faster_whisper.__file__).parent / "assets" / "silero_vad_v6.onnx" -- cgit v1.2.3