From 60f05cebd8e7968621a0ce015cf72987ec7577c7 Mon Sep 17 00:00:00 2001 From: "Somhairle H. Marisol" Date: Tue, 29 Sep 2026 10:50:12 +0800 Subject: feat(recorder): add offline voice clip pipeline MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit [变更性质] - 本提交新增本机离线的语音片段录音能力,不涉及网络服务或语音识别。 [新增功能] - 通过 Pulse 默认输入持续采集音频,以本地 Silero VAD 触发片段。 - 以私有目录和权限写入 24 kbps Ogg/Opus 录音,并提供 systemd 用户服务安装器。 [实现方案] - 使用有限前置缓冲、静音封段和最长段滚动状态机,避免静音落盘及内存无限增长。 - 增加真实 Opus 编解码、服务 dry-run 和纯逻辑状态转换测试;记录模型归属和部署前置条件。 [影响范围] - 新增 Python CLI、运行时模块、中文运维文档和自动测试。 - 未安装、启用或启动任何用户 systemd 服务;不删除或修改录音数据。 --- .gitignore | 6 ++ README.md | 83 +++++++++++++++++++++++++ THIRD_PARTY_NOTICES.md | 13 ++++ pyproject.toml | 20 +++++++ src/mic_clipper/__init__.py | 1 + src/mic_clipper/assets/__init__.py | 1 + src/mic_clipper/audio_input.py | 72 ++++++++++++++++++++++ src/mic_clipper/cli.py | 66 ++++++++++++++++++++ src/mic_clipper/runner.py | 92 ++++++++++++++++++++++++++++ src/mic_clipper/segmenter.py | 120 +++++++++++++++++++++++++++++++++++++ src/mic_clipper/service.py | 101 +++++++++++++++++++++++++++++++ src/mic_clipper/storage.py | 111 ++++++++++++++++++++++++++++++++++ src/mic_clipper/vad.py | 59 ++++++++++++++++++ tests/test_cli.py | 25 ++++++++ tests/test_runtime.py | 82 +++++++++++++++++++++++++ tests/test_segmenter.py | 96 +++++++++++++++++++++++++++++ tests/test_service.py | 60 +++++++++++++++++++ tests/test_storage.py | 33 ++++++++++ tests/test_vad.py | 19 ++++++ 19 files changed, 1060 insertions(+) create mode 100644 .gitignore create mode 100644 README.md create mode 100644 THIRD_PARTY_NOTICES.md create mode 100644 pyproject.toml create mode 100644 src/mic_clipper/__init__.py create mode 100644 src/mic_clipper/assets/__init__.py create mode 100644 src/mic_clipper/audio_input.py create mode 100644 src/mic_clipper/cli.py create mode 100644 src/mic_clipper/runner.py create mode 100644 src/mic_clipper/segmenter.py create mode 100644 src/mic_clipper/service.py create mode 100644 src/mic_clipper/storage.py create mode 100644 src/mic_clipper/vad.py create mode 100644 tests/test_cli.py create mode 100644 tests/test_runtime.py create mode 100644 tests/test_segmenter.py create mode 100644 tests/test_service.py create mode 100644 tests/test_storage.py create mode 100644 tests/test_vad.py diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..0575620 --- /dev/null +++ b/.gitignore @@ -0,0 +1,6 @@ +__pycache__/ +*.py[cod] +.pytest_cache/ +build/ +dist/ +*.egg-info/ diff --git a/README.md b/README.md new file mode 100644 index 0000000..95a13f7 --- /dev/null +++ b/README.md @@ -0,0 +1,83 @@ +# Mic Clipper + +一个面向 Linux 桌面的本机常驻“说话片段录音器”。它持续从 PipeWire/Pulse 的动态默认麦克风输入读取音频,仅在本地 Silero VAD 判为语音时才创建文件。 + +## 行为与隐私 + +- 16 kHz 单声道输入;每 32 ms(512 个采样)运行一次本地 ONNX VAD。 +- 语音开始时带 0.5 秒内存前置音频;持续静音 1.5 秒后关闭片段。 +- 使用 Ogg/Opus、`libopus`、24 kbps、语音优化编码,保存为 `*.opus`。 +- 文件写入 `~/Documents/Mic Clips/YYYY-MM-DD/HH-MM-SS.opus`;跨午夜时目录按片段开始日期确定。 +- 静音不会创建文件。音频只在内存中保留有限的前置缓冲;不调用 ASR、不上传、不保存转写文本,也不自动删除旧录音。 +- 单段最长 10 分钟。连续说话会无缝滚动为下一段,避免内存无限增长;相邻段会在边界处保持音频连续。 +- 目录权限为 `0700`,录音文件为 `0600`。日志仅写设备/编码器故障与重试信息,不写音频或文本。 + +存储量约为 `24 kbit/s / 8 = 3 kB/s`,即每小时约 10.8 MB、每天连续录音约 259 MB。实际仅保存语音片段,通常会低于这个上限。 + +## 运行时依赖 + +目标解释器必须是: + +```text +/home/somhairle/.hermes/hermes-agent/venv/bin/python +``` + +该解释器需要已有 `numpy` 与 `onnxruntime`。系统需要带 `pulse` 输入和 `libopus` 编码器的 `ffmpeg`。本项目不执行 `pip install`、模型下载或任何联网操作。 + +部署前必须确认项目包内存在: + +```text +src/mic_clipper/assets/silero_vad_v6.onnx +``` + +该权重应从当前机器已验证的 `faster-whisper` 1.2.1 本地资源复制而来,归属和 MIT 许可说明见 [THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md)。当前开发代码为便于本机测试会回退读取该已预置的本地资源;它不会下载模型。该回退不是自包含部署方案,审阅/部署前必须把权重复制进项目包。 + +## 安装与服务 + +以下命令均从项目根目录执行。先用 dry-run 审阅将执行的 systemd 命令;它不会写入家目录。 + +```bash +RUNTIME=/home/somhairle/.hermes/hermes-agent/venv/bin/python +PROJECT=/home/somhairle/projects/mic-clipper +PYTHONPATH="$PROJECT/src" "$RUNTIME" -m mic_clipper.cli service install --dry-run +``` + +经独立审阅后,安装并立即启用用户服务: + +```bash +PYTHONPATH="$PROJECT/src" "$RUNTIME" -m mic_clipper.cli service install +``` + +安装器仅写 `~/.config/systemd/user/mic-clipper.service`,然后执行 `daemon-reload` 与 `enable --now`。服务 `Wants`/`After` PipeWire、WirePlumber 与 pipewire-pulse,应用本身会在默认麦克风暂不可用或设备切换时以 1、2、4...60 秒退避重试。`Restart=on-failure` 只处理异常退出,手动 `stop` 不会立即拉起服务。 + +## 日常命令 + +```bash +# 服务状态 +PYTHONPATH="$PROJECT/src" "$RUNTIME" -m mic_clipper.cli service status + +# 手动启动、停止、重启 +systemctl --user start mic-clipper.service +systemctl --user stop mic-clipper.service +systemctl --user restart mic-clipper.service + +# 前台运行(便于排障;Ctrl-C 正常收尾当前编码器) +PYTHONPATH="$PROJECT/src" "$RUNTIME" -m mic_clipper.cli run + +# 卸载 unit,不删除任何录音 +PYTHONPATH="$PROJECT/src" "$RUNTIME" -m mic_clipper.cli service uninstall +``` + +## 故障排查 + +```bash +# 查看不含音频内容的服务日志 +journalctl --user -u mic-clipper.service -f + +# 确认 PipeWire/Pulse 默认源与 ffmpeg 能力 +pactl get-default-source +ffmpeg -hide_banner -h demuxer=pulse +ffmpeg -hide_banner -h encoder=libopus +``` + +若日志显示 Pulse 输入退出,确认 `pipewire.service`、`wireplumber.service` 和 `pipewire-pulse.service` 正常运行;应用会自行重试。若 VAD 报模型缺失,先恢复上文列出的包内 ONNX 文件,不能用联网下载替代。 diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md new file mode 100644 index 0000000..e38fc8e --- /dev/null +++ b/THIRD_PARTY_NOTICES.md @@ -0,0 +1,13 @@ +# Third-Party Notices + +## Silero VAD ONNX weight + +The intended bundled file is `src/mic_clipper/assets/silero_vad_v6.onnx`. + +Its local source is the `assets/silero_vad_v6.onnx` file in the verified +`faster-whisper` 1.2.1 installation. That package's installed metadata declares +the MIT license and its VAD implementation attributes the model integration to +[`snakers4/silero-vad`](https://github.com/snakers4/silero-vad). + +The ONNX weight has no network retrieval path in this project. The asset must be +present before deployment; it is intentionally not downloaded at runtime. diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..af7d8bb --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,20 @@ +[build-system] +requires = ["setuptools>=68"] +build-backend = "setuptools.build_meta" + +[project] +name = "mic-clipper" +version = "0.1.0" +description = "Offline voice-activated microphone clip recorder" +requires-python = ">=3.11" +dependencies = ["numpy", "onnxruntime"] + +[project.scripts] +mic-clipper = "mic_clipper.cli:main" + +[tool.setuptools.package-data] +"mic_clipper.assets" = ["*.onnx"] + +[tool.pytest.ini_options] +pythonpath = ["src"] +testpaths = ["tests"] diff --git a/src/mic_clipper/__init__.py b/src/mic_clipper/__init__.py new file mode 100644 index 0000000..7d39855 --- /dev/null +++ b/src/mic_clipper/__init__.py @@ -0,0 +1 @@ +"""Offline, voice-activated microphone clip recording.""" diff --git a/src/mic_clipper/assets/__init__.py b/src/mic_clipper/assets/__init__.py new file mode 100644 index 0000000..480108b --- /dev/null +++ b/src/mic_clipper/assets/__init__.py @@ -0,0 +1 @@ +"""Bundled, offline model assets.""" diff --git a/src/mic_clipper/audio_input.py b/src/mic_clipper/audio_input.py new file mode 100644 index 0000000..1d77aa0 --- /dev/null +++ b/src/mic_clipper/audio_input.py @@ -0,0 +1,72 @@ +"""PipeWire/Pulse microphone capture through the local ffmpeg binary.""" + +from __future__ import annotations + +import subprocess +from collections.abc import Iterator + +import numpy as np + +from mic_clipper.vad import FRAME_SAMPLES + + +class AudioInputUnavailable(RuntimeError): + """The dynamic Pulse default source could not supply another frame.""" + + +class PulseAudioInput: + """Yield fixed 16 kHz mono float PCM frames from Pulse's `default` source.""" + + @staticmethod + def command() -> list[str]: + return [ + "ffmpeg", + "-nostdin", + "-hide_banner", + "-loglevel", + "error", + "-f", + "pulse", + "-sample_rate", + "16000", + "-channels", + "1", + "-i", + "default", + "-ac", + "1", + "-ar", + "16000", + "-f", + "f32le", + "pipe:1", + ] + + def frames(self) -> Iterator[np.ndarray]: + process = subprocess.Popen( + self.command(), + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + ) + assert process.stdout is not None + assert process.stderr is not None + try: + while True: + payload = process.stdout.read(FRAME_SAMPLES * np.dtype(" argparse.ArgumentParser: + parser = argparse.ArgumentParser(description="Offline voice-activated microphone clips") + commands = parser.add_subparsers(dest="command", required=True) + + run = commands.add_parser("run", help="listen and save speech clips") + run.add_argument("--output-dir", type=Path, default=DEFAULT_OUTPUT) + run.add_argument("--threshold", type=float, default=0.5) + + service_parser = commands.add_parser("service", help="manage the systemd user service") + service_commands = service_parser.add_subparsers(dest="service_command", required=True) + install = service_commands.add_parser("install", help="install and enable the user service") + install.add_argument("--dry-run", action="store_true") + install.add_argument("--unit-path", type=Path, default=DEFAULT_UNIT) + install.add_argument("--project-root", type=Path, default=PROJECT_ROOT) + install.add_argument("--python", type=Path, default=Path(sys.executable)) + uninstall = service_commands.add_parser("uninstall", help="remove only this tool's service") + uninstall.add_argument("--unit-path", type=Path, default=DEFAULT_UNIT) + service_commands.add_parser("status", help="show systemd user service status") + return parser + + +def main(arguments: list[str] | None = None) -> int: + args = build_parser().parse_args(arguments) + if args.command == "run": + if not 0 < args.threshold <= 1: + raise SystemExit("--threshold must be between 0 and 1") + logging.basicConfig(level=logging.INFO, format="%(levelname)s %(message)s") + run_forever(args.output_dir, threshold=args.threshold) + return 0 + if args.service_command == "install": + commands = service.install( + unit_path=args.unit_path, + python=args.python, + project_root=args.project_root, + dry_run=args.dry_run, + ) + if args.dry_run: + for command in commands: + print(" ".join(command)) + return 0 + if args.service_command == "uninstall": + service.uninstall(unit_path=args.unit_path) + return 0 + return service.status() + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/mic_clipper/runner.py b/src/mic_clipper/runner.py new file mode 100644 index 0000000..787bf58 --- /dev/null +++ b/src/mic_clipper/runner.py @@ -0,0 +1,92 @@ +"""Runtime wiring for capture, VAD, segmentation, and durable clips.""" + +from __future__ import annotations + +import logging +import time +from collections.abc import Callable, Iterable +from datetime import datetime +from pathlib import Path +from typing import Protocol + +import numpy as np + +from mic_clipper.audio_input import PulseAudioInput +from mic_clipper.segmenter import Audio, End, Segmenter, Start +from mic_clipper.storage import ClipWriter, OpenClip +from mic_clipper.vad import SileroVad + + +LOGGER = logging.getLogger(__name__) + + +class Vad(Protocol): + def probability(self, samples: np.ndarray) -> float: ... + + +class Writer(Protocol): + def open(self, started_at: datetime) -> OpenClip: ... + + +def record_frames( + frames: Iterable[np.ndarray], + *, + vad: Vad, + writer: Writer, + segmenter: Segmenter, + now: Callable[[], datetime], + threshold: float = 0.5, +) -> None: + """Record one finite or failing frame stream, closing an active encoder on exit.""" + clip: OpenClip | None = None + try: + for samples in frames: + events = segmenter.push(samples, vad.probability(samples) >= threshold, now()) + for event in events: + if isinstance(event, Start): + clip = writer.open(event.started_at) + try: + clip.write(event.samples) + except BaseException: + clip.abort() + clip = None + raise + elif isinstance(event, Audio): + if clip is None: + raise RuntimeError("segment audio arrived without an active encoder") + clip.write(event.samples) + elif isinstance(event, End): + if clip is None: + raise RuntimeError("segment end arrived without an active encoder") + clip.close() + clip = None + finally: + if clip is not None: + clip.close() + + +def run_forever(root: Path, *, threshold: float = 0.5) -> None: + """Recreate all live components after a transient audio-service failure.""" + retry_seconds = 1 + while True: + try: + record_frames( + PulseAudioInput().frames(), + vad=SileroVad(), + writer=ClipWriter(root), + segmenter=Segmenter(pre_roll_seconds=0.5, silence_seconds=1.5), + now=lambda: datetime.now().astimezone(), + threshold=threshold, + ) + raise RuntimeError("Pulse input ended unexpectedly") + except KeyboardInterrupt: + LOGGER.info("microphone clip recorder stopped") + return + except Exception as error: + LOGGER.warning( + "audio recorder unavailable (%s); retrying in %s seconds", + error, + retry_seconds, + ) + time.sleep(retry_seconds) + retry_seconds = min(retry_seconds * 2, 60) diff --git a/src/mic_clipper/segmenter.py b/src/mic_clipper/segmenter.py new file mode 100644 index 0000000..aa0e3b2 --- /dev/null +++ b/src/mic_clipper/segmenter.py @@ -0,0 +1,120 @@ +"""Pure streaming voice-clip state machine.""" + +from __future__ import annotations + +from collections import deque +from dataclasses import dataclass +from datetime import datetime, timedelta + +import numpy as np + + +SAMPLE_RATE = 16_000 + + +@dataclass(frozen=True) +class Start: + started_at: datetime + samples: np.ndarray + + +@dataclass(frozen=True) +class Audio: + samples: np.ndarray + + +@dataclass(frozen=True) +class End: + pass + + +class Segmenter: + """Emit stream events while retaining only the configured pre-roll in memory.""" + + def __init__( + self, + *, + pre_roll_seconds: float, + silence_seconds: float, + max_segment_seconds: float = 600, + sample_rate: int = SAMPLE_RATE, + ) -> None: + self.sample_rate = sample_rate + self.pre_roll_samples = round(pre_roll_seconds * sample_rate) + self.silence_samples_limit = round(silence_seconds * sample_rate) + self.max_segment_samples = round(max_segment_seconds * sample_rate) + self._pre_roll: deque[np.ndarray] = deque() + self._pre_roll_size = 0 + self._active = False + self._segment_samples = 0 + self._silence_samples = 0 + + @property + def active(self) -> bool: + return self._active + + def push( + self, samples: np.ndarray, is_speech: bool, captured_at: datetime + ) -> tuple[Start | Audio | End, ...]: + samples = np.asarray(samples, dtype=np.float32) + if samples.ndim != 1: + raise ValueError("audio must be a mono, one-dimensional array") + if not len(samples): + return () + + if not self._active: + if not is_speech: + self._append_pre_roll(samples) + return () + pre_roll = self._take_pre_roll() + started_at = captured_at - timedelta( + seconds=len(pre_roll) / self.sample_rate + ) + self._active = True + self._segment_samples = len(pre_roll) + len(samples) + self._silence_samples = 0 + self._append_pre_roll(samples) + return (Start(started_at, np.concatenate((pre_roll, samples))),) + + if self._segment_samples + len(samples) > self.max_segment_samples: + self._append_pre_roll(samples) + self._active = True + self._segment_samples = len(samples) + self._silence_samples = 0 + return (End(), Start(captured_at, samples)) + + self._segment_samples += len(samples) + if is_speech: + self._silence_samples = 0 + else: + self._silence_samples += len(samples) + self._append_pre_roll(samples) + + events: tuple[Start | Audio | End, ...] = (Audio(samples),) + if self._silence_samples >= self.silence_samples_limit: + self._active = False + self._segment_samples = 0 + self._silence_samples = 0 + events += (End(),) + return events + + def _append_pre_roll(self, samples: np.ndarray) -> None: + self._pre_roll.append(samples) + self._pre_roll_size += len(samples) + while self._pre_roll_size > self.pre_roll_samples: + excess = self._pre_roll_size - self.pre_roll_samples + oldest = self._pre_roll[0] + if len(oldest) <= excess: + self._pre_roll.popleft() + self._pre_roll_size -= len(oldest) + else: + self._pre_roll[0] = oldest[excess:] + self._pre_roll_size -= excess + + def _take_pre_roll(self) -> np.ndarray: + if not self._pre_roll: + return np.empty(0, dtype=np.float32) + samples = np.concatenate(tuple(self._pre_roll)) + self._pre_roll.clear() + self._pre_roll_size = 0 + return samples diff --git a/src/mic_clipper/service.py b/src/mic_clipper/service.py new file mode 100644 index 0000000..ecf5673 --- /dev/null +++ b/src/mic_clipper/service.py @@ -0,0 +1,101 @@ +"""Install and remove the systemd user service without touching recordings.""" + +from __future__ import annotations + +import os +import subprocess +import tempfile +from collections.abc import Callable +from pathlib import Path + + +UNIT_NAME = "mic-clipper.service" + + +def render_unit(python: Path, project_root: Path) -> str: + return f"""[Unit] +Description=Offline voice-activated microphone clip recorder +Wants=pipewire.service wireplumber.service pipewire-pulse.service +After=pipewire.service wireplumber.service pipewire-pulse.service + +[Service] +Type=simple +WorkingDirectory={project_root} +Environment=PYTHONPATH={project_root / 'src'} +ExecStart={python} -m mic_clipper run +Restart=on-failure +RestartSec=5s +RestartSteps=5 +RestartMaxDelaySec=60s +StartLimitIntervalSec=300 +StartLimitBurst=10 +NoNewPrivileges=yes +PrivateTmp=yes +UMask=0077 + +[Install] +WantedBy=default.target +""" + + +def install( + *, + unit_path: Path, + python: Path, + project_root: Path, + dry_run: bool = False, + run_command: Callable[[list[str]], object] | None = None, +) -> list[list[str]]: + commands = [ + ["systemctl", "--user", "daemon-reload"], + ["systemctl", "--user", "enable", "--now", UNIT_NAME], + ] + if dry_run: + return commands + unit_path.parent.mkdir(mode=0o700, parents=True, exist_ok=True) + _write_private(unit_path, render_unit(python, project_root)) + runner = run_command or _run_command + for command in commands: + runner(command) + return commands + + +def uninstall( + *, + unit_path: Path, + run_command: Callable[[list[str]], object] | None = None, +) -> list[list[str]]: + if not unit_path.exists(): + return [] + commands = [ + ["systemctl", "--user", "disable", "--now", UNIT_NAME], + ["systemctl", "--user", "daemon-reload"], + ] + runner = run_command or _run_command + for command in commands[:1]: + runner(command) + unit_path.unlink() + runner(commands[1]) + return commands + + +def status() -> int: + return subprocess.run( + ["systemctl", "--user", "status", UNIT_NAME], check=False + ).returncode + + +def _write_private(path: Path, content: str) -> None: + with tempfile.NamedTemporaryFile( + "w", encoding="utf-8", dir=path.parent, prefix=f".{path.name}.", delete=False + ) as temporary: + temporary.write(content) + temporary.flush() + os.fchmod(temporary.fileno(), 0o600) + temporary_path = Path(temporary.name) + os.replace(temporary_path, path) + os.chmod(path, 0o600) + + +def _run_command(command: list[str]) -> None: + subprocess.run(command, check=True) diff --git a/src/mic_clipper/storage.py b/src/mic_clipper/storage.py new file mode 100644 index 0000000..a7f00bd --- /dev/null +++ b/src/mic_clipper/storage.py @@ -0,0 +1,111 @@ +"""Private, streaming Opus clip storage.""" + +from __future__ import annotations + +import os +import subprocess +from dataclasses import dataclass +from datetime import datetime +from pathlib import Path + +import numpy as np + + +SAMPLE_RATE = 16_000 + + +@dataclass +class OpenClip: + path: Path + process: subprocess.Popen[bytes] + closed: bool = False + + def write(self, samples: np.ndarray) -> None: + if self.closed: + raise RuntimeError("cannot write a closed clip") + assert self.process.stdin is not None + audio = np.asarray(samples, dtype=" None: + if self.closed: + return + self.closed = True + assert self.process.stdin is not None + self.process.stdin.close() + assert self.process.stderr is not None + error_output = self.process.stderr.read().decode("utf-8", errors="replace").strip() + return_code = self.process.wait() + if return_code: + self.path.unlink(missing_ok=True) + raise RuntimeError(f"ffmpeg Opus encoding failed ({return_code}): {error_output}") + os.chmod(self.path, 0o600) + + def abort(self) -> None: + if self.closed: + return + self.closed = True + self.process.kill() + self.process.wait() + self.path.unlink(missing_ok=True) + + +class ClipWriter: + """Open one private Ogg/Opus file only after the segmenter emits a start.""" + + def __init__(self, root: Path) -> None: + self.root = Path(root).expanduser() + + def open(self, started_at: datetime) -> OpenClip: + self.root.mkdir(mode=0o700, parents=True, exist_ok=True) + os.chmod(self.root, 0o700) + destination = self.root / started_at.date().isoformat() + destination.mkdir(mode=0o700, exist_ok=True) + os.chmod(destination, 0o700) + path = self._reserve_path(destination, started_at) + process = subprocess.Popen( + [ + "ffmpeg", + "-nostdin", + "-v", + "error", + "-f", + "f32le", + "-ar", + str(SAMPLE_RATE), + "-ac", + "1", + "-i", + "pipe:0", + "-c:a", + "libopus", + "-b:a", + "24k", + "-application", + "voip", + "-vbr", + "constrained", + "-f", + "ogg", + "-y", + str(path), + ], + stdin=subprocess.PIPE, + stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE, + ) + return OpenClip(path, process) + + @staticmethod + def _reserve_path(destination: Path, started_at: datetime) -> Path: + base_name = started_at.strftime("%H-%M-%S") + for suffix in range(10_000): + name = f"{base_name}.opus" if suffix == 0 else f"{base_name}-{suffix:02d}.opus" + path = destination / name + try: + descriptor = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600) + except FileExistsError: + continue + os.close(descriptor) + return path + raise RuntimeError("unable to reserve a unique clip filename") diff --git a/src/mic_clipper/vad.py b/src/mic_clipper/vad.py new file mode 100644 index 0000000..705dfd8 --- /dev/null +++ b/src/mic_clipper/vad.py @@ -0,0 +1,59 @@ +"""Stateful, local Silero VAD ONNX inference.""" + +from __future__ import annotations + +from importlib import resources +from pathlib import Path + +import numpy as np +import onnxruntime + + +FRAME_SAMPLES = 512 +CONTEXT_SAMPLES = 64 + + +class SileroVad: + """Run the bundled Silero v6 model on exact 32 ms PCM frames.""" + + def __init__(self, model_path: Path | None = None) -> None: + self.model_path = Path(model_path) if model_path else self._default_model_path() + options = onnxruntime.SessionOptions() + options.inter_op_num_threads = 1 + options.intra_op_num_threads = 1 + options.enable_cpu_mem_arena = False + options.log_severity_level = 4 + self.session = onnxruntime.InferenceSession( + self.model_path, + providers=["CPUExecutionProvider"], + sess_options=options, + ) + self._context = np.zeros(CONTEXT_SAMPLES, dtype=np.float32) + self._h = np.zeros((1, 1, 128), dtype=np.float32) + self._c = np.zeros((1, 1, 128), dtype=np.float32) + + def probability(self, samples: np.ndarray) -> float: + samples = np.asarray(samples, dtype=np.float32) + if samples.ndim != 1 or len(samples) != FRAME_SAMPLES: + raise ValueError("Silero VAD requires exactly 512 mono samples at 16 kHz") + model_input = np.concatenate((self._context, samples)).reshape(1, -1) + output, self._h, self._c = self.session.run( + None, + {"input": model_input, "h": self._h, "c": self._c}, + ) + self._context = samples[-CONTEXT_SAMPLES:].copy() + return float(output[0]) + + @staticmethod + def _default_model_path() -> Path: + bundled = resources.files("mic_clipper.assets").joinpath("silero_vad_v6.onnx") + if bundled.is_file(): + return Path(bundled) + try: + import faster_whisper + except ImportError as error: + raise RuntimeError( + "Silero VAD model is missing; bundle silero_vad_v6.onnx or install the " + "pre-provisioned faster-whisper runtime asset" + ) from error + return Path(faster_whisper.__file__).parent / "assets" / "silero_vad_v6.onnx" diff --git a/tests/test_cli.py b/tests/test_cli.py new file mode 100644 index 0000000..09de22c --- /dev/null +++ b/tests/test_cli.py @@ -0,0 +1,25 @@ +from pathlib import Path + +from mic_clipper.cli import main + + +def test_service_install_dry_run_prints_commands_without_creating_unit(tmp_path, capsys): + unit_path = tmp_path / "mic-clipper.service" + + result = main( + [ + "service", + "install", + "--dry-run", + "--unit-path", + str(unit_path), + "--project-root", + str(tmp_path), + "--python", + "/runtime/python", + ] + ) + + assert result == 0 + assert not unit_path.exists() + assert "systemctl --user enable --now mic-clipper.service" in capsys.readouterr().out diff --git a/tests/test_runtime.py b/tests/test_runtime.py new file mode 100644 index 0000000..d0d45fb --- /dev/null +++ b/tests/test_runtime.py @@ -0,0 +1,82 @@ +from datetime import datetime, timezone + +import numpy as np + +from mic_clipper.audio_input import PulseAudioInput +from mic_clipper.runner import record_frames +from mic_clipper.segmenter import Segmenter + + +class FakeVad: + def __init__(self, probabilities: list[float]) -> None: + self.probabilities = iter(probabilities) + + def probability(self, _samples: np.ndarray) -> float: + return next(self.probabilities) + + +class FakeClip: + def __init__(self) -> None: + self.writes: list[np.ndarray] = [] + self.closed = False + + def write(self, samples: np.ndarray) -> None: + self.writes.append(samples) + + def close(self) -> None: + self.closed = True + + def abort(self) -> None: + self.closed = True + + +class FakeWriter: + def __init__(self) -> None: + self.opened: list[FakeClip] = [] + + def open(self, _started_at: datetime) -> FakeClip: + clip = FakeClip() + self.opened.append(clip) + return clip + + +def samples() -> np.ndarray: + return np.zeros(512, dtype=np.float32) + + +def test_silence_never_opens_an_output_clip(): + writer = FakeWriter() + + record_frames( + [samples(), samples()], + vad=FakeVad([0.1, 0.1]), + writer=writer, + segmenter=Segmenter(pre_roll_seconds=0.5, silence_seconds=1.5), + now=lambda: datetime(2026, 9, 29, tzinfo=timezone.utc), + ) + + assert writer.opened == [] + + +def test_end_event_closes_the_active_encoder(): + writer = FakeWriter() + + record_frames( + [samples()] + [samples()] * 47, + vad=FakeVad([0.9] + [0.1] * 47), + writer=writer, + segmenter=Segmenter(pre_roll_seconds=0.5, silence_seconds=1.5), + now=lambda: datetime(2026, 9, 29, tzinfo=timezone.utc), + ) + + assert len(writer.opened) == 1 + assert writer.opened[0].closed + + +def test_pulse_input_uses_dynamic_default_device_at_16khz_mono(): + command = PulseAudioInput.command() + + assert command[0] == "ffmpeg" + assert command[command.index("-f") + 1] == "pulse" + assert command[command.index("-i") + 1] == "default" + assert command[-2:] == ["f32le", "pipe:1"] diff --git a/tests/test_segmenter.py b/tests/test_segmenter.py new file mode 100644 index 0000000..d1f7ee5 --- /dev/null +++ b/tests/test_segmenter.py @@ -0,0 +1,96 @@ +from datetime import datetime, timedelta, timezone + +import numpy as np + +from mic_clipper.segmenter import Audio, End, Segmenter, Start + + +SAMPLE_RATE = 16_000 +FRAME_SAMPLES = 512 +FRAME_DURATION = timedelta(seconds=FRAME_SAMPLES / SAMPLE_RATE) +START = datetime(2026, 9, 29, 23, 59, 59, tzinfo=timezone.utc) + + +def frame(value: float = 0.0) -> np.ndarray: + return np.full(FRAME_SAMPLES, value, dtype=np.float32) + + +def push(segmenter: Segmenter, index: int, speech: bool, value: float = 0.0): + return segmenter.push(frame(value), speech, START + index * FRAME_DURATION) + + +def test_speech_starts_segment_with_half_second_preroll(): + segmenter = Segmenter(pre_roll_seconds=0.5, silence_seconds=1.5) + + for index in range(16): + assert push(segmenter, index, False) == () + + events = push(segmenter, 16, True, 1.0) + + assert len(events) == 1 + assert isinstance(events[0], Start) + assert events[0].started_at == START + 16 * FRAME_DURATION - timedelta(seconds=0.5) + assert events[0].samples.shape == (8_512,) + assert np.all(events[0].samples[:8_000] == 0.0) + assert np.all(events[0].samples[8_000:] == 1.0) + + +def test_short_silence_is_merged_into_active_segment(): + segmenter = Segmenter(pre_roll_seconds=0.5, silence_seconds=1.5) + + assert isinstance(push(segmenter, 0, True, 1.0)[0], Start) + assert isinstance(push(segmenter, 1, False)[0], Audio) + assert isinstance(push(segmenter, 2, True, 1.0)[0], Audio) + assert segmenter.active + + +def test_one_point_five_seconds_of_silence_closes_segment(): + segmenter = Segmenter(pre_roll_seconds=0.5, silence_seconds=1.5) + + push(segmenter, 0, True, 1.0) + events = () + for index in range(1, 48): + events = push(segmenter, index, False) + + assert isinstance(events[0], Audio) + assert isinstance(events[-1], End) + assert not segmenter.active + + +def test_maximum_segment_rolls_without_losing_audio(): + segmenter = Segmenter( + pre_roll_seconds=0.5, + silence_seconds=1.5, + max_segment_seconds=FRAME_SAMPLES * 2 / SAMPLE_RATE, + ) + + first = push(segmenter, 0, True, 1.0) + second = push(segmenter, 1, True, 2.0) + third = push(segmenter, 2, True, 3.0) + + assert isinstance(first[0], Start) + assert isinstance(second[0], Audio) + assert isinstance(third[0], End) + assert isinstance(third[1], Start) + assert np.all(third[1].samples == 3.0) + + +def test_segment_start_date_is_retained_across_midnight(): + segmenter = Segmenter(pre_roll_seconds=0.5, silence_seconds=1.5) + + events = push(segmenter, 16, True, 1.0) + + assert isinstance(events[0], Start) + assert events[0].started_at.date().isoformat() == "2026-09-29" + + +def test_speech_after_a_closed_segment_keeps_terminal_silence_as_preroll(): + segmenter = Segmenter(pre_roll_seconds=0.5, silence_seconds=1.5) + + push(segmenter, 0, True, 1.0) + for index in range(1, 48): + push(segmenter, index, False) + events = push(segmenter, 48, True, 1.0) + + assert isinstance(events[0], Start) + assert events[0].samples.shape == (8_512,) diff --git a/tests/test_service.py b/tests/test_service.py new file mode 100644 index 0000000..2bda2ab --- /dev/null +++ b/tests/test_service.py @@ -0,0 +1,60 @@ +from pathlib import Path + +from mic_clipper.service import install, render_unit, uninstall + + +def test_rendered_service_waits_for_audio_services_and_restarts_only_on_failure(): + unit = render_unit(Path("/runtime/python"), Path("/project")) + + assert "After=pipewire.service wireplumber.service pipewire-pulse.service" in unit + assert "Restart=on-failure" in unit + assert "RestartMaxDelaySec=60s" in unit + assert "ExecStart=/runtime/python -m mic_clipper run" in unit + assert "UMask=0077" in unit + + +def test_install_and_uninstall_only_manage_the_tool_unit(tmp_path): + unit_path = tmp_path / "systemd" / "mic-clipper.service" + clips = tmp_path / "Mic Clips" / "2026-09-29" / "keep.opus" + clips.parent.mkdir(parents=True) + clips.write_bytes(b"recording") + commands: list[list[str]] = [] + + install( + unit_path=unit_path, + python=Path("/runtime/python"), + project_root=Path("/project"), + run_command=commands.append, + ) + install( + unit_path=unit_path, + python=Path("/runtime/python"), + project_root=Path("/project"), + run_command=commands.append, + ) + uninstall(unit_path=unit_path, run_command=commands.append) + + assert not unit_path.exists() + assert clips.read_bytes() == b"recording" + assert commands == [ + ["systemctl", "--user", "daemon-reload"], + ["systemctl", "--user", "enable", "--now", "mic-clipper.service"], + ["systemctl", "--user", "daemon-reload"], + ["systemctl", "--user", "enable", "--now", "mic-clipper.service"], + ["systemctl", "--user", "disable", "--now", "mic-clipper.service"], + ["systemctl", "--user", "daemon-reload"], + ] + + +def test_dry_run_does_not_write_a_unit(tmp_path): + unit_path = tmp_path / "mic-clipper.service" + + commands = install( + unit_path=unit_path, + python=Path("/runtime/python"), + project_root=Path("/project"), + dry_run=True, + ) + + assert not unit_path.exists() + assert commands[-1] == ["systemctl", "--user", "enable", "--now", "mic-clipper.service"] diff --git a/tests/test_storage.py b/tests/test_storage.py new file mode 100644 index 0000000..389f199 --- /dev/null +++ b/tests/test_storage.py @@ -0,0 +1,33 @@ +import os +import subprocess +from datetime import datetime, timezone + +import numpy as np + +from mic_clipper.storage import ClipWriter + + +def test_writer_creates_private_opus_clip_in_segment_start_date(tmp_path): + started_at = datetime(2026, 9, 29, 23, 59, 59, tzinfo=timezone.utc) + samples = np.sin(np.linspace(0, 100, 16_000, dtype=np.float32)) + writer = ClipWriter(tmp_path) + + clip = writer.open(started_at) + clip.write(samples) + clip.close() + + destination = tmp_path / "2026-09-29" + files = list(destination.glob("*.opus")) + assert len(files) == 1 + assert files[0].name.startswith("23-59-59") + assert os.stat(destination).st_mode & 0o777 == 0o700 + assert os.stat(files[0]).st_mode & 0o777 == 0o600 + assert files[0].read_bytes()[:4] == b"OggS" + assert b"OpusHead" in files[0].read_bytes()[:128] + + decoded = subprocess.run( + ["ffmpeg", "-v", "error", "-i", str(files[0]), "-f", "f32le", "pipe:1"], + check=True, + capture_output=True, + ) + assert len(decoded.stdout) > 0 diff --git a/tests/test_vad.py b/tests/test_vad.py new file mode 100644 index 0000000..98826ae --- /dev/null +++ b/tests/test_vad.py @@ -0,0 +1,19 @@ +import numpy as np +import pytest + +from mic_clipper.vad import FRAME_SAMPLES, SileroVad + + +def test_bundled_silero_vad_accepts_512_samples_and_returns_probability(): + vad = SileroVad() + + probability = vad.probability(np.zeros(FRAME_SAMPLES, dtype=np.float32)) + + assert 0.0 <= probability <= 1.0 + + +def test_bundled_silero_vad_rejects_non_512_sample_frames(): + vad = SileroVad() + + with pytest.raises(ValueError, match="512"): + vad.probability(np.zeros(FRAME_SAMPLES - 1, dtype=np.float32)) -- cgit v1.2.3