summaryrefslogtreecommitdiff
path: root/src
diff options
context:
space:
mode:
authorSomhairle H. Marisol <[email protected]>2026-09-29 10:50:12 +0800
committerSomhairle H. Marisol <[email protected]>2026-09-29 10:50:12 +0800
commit60f05cebd8e7968621a0ce015cf72987ec7577c7 (patch)
tree7f3c572ae82015b17ceaff6eb901ec88ba809856 /src
downloadmic-clipper-60f05cebd8e7968621a0ce015cf72987ec7577c7.tar.gz
feat(recorder): add offline voice clip pipeline
[变更性质] - 本提交新增本机离线的语音片段录音能力,不涉及网络服务或语音识别。 [新增功能] - 通过 Pulse 默认输入持续采集音频,以本地 Silero VAD 触发片段。 - 以私有目录和权限写入 24 kbps Ogg/Opus 录音,并提供 systemd 用户服务安装器。 [实现方案] - 使用有限前置缓冲、静音封段和最长段滚动状态机,避免静音落盘及内存无限增长。 - 增加真实 Opus 编解码、服务 dry-run 和纯逻辑状态转换测试;记录模型归属和部署前置条件。 [影响范围] - 新增 Python CLI、运行时模块、中文运维文档和自动测试。 - 未安装、启用或启动任何用户 systemd 服务;不删除或修改录音数据。
Diffstat (limited to 'src')
-rw-r--r--src/mic_clipper/__init__.py1
-rw-r--r--src/mic_clipper/assets/__init__.py1
-rw-r--r--src/mic_clipper/audio_input.py72
-rw-r--r--src/mic_clipper/cli.py66
-rw-r--r--src/mic_clipper/runner.py92
-rw-r--r--src/mic_clipper/segmenter.py120
-rw-r--r--src/mic_clipper/service.py101
-rw-r--r--src/mic_clipper/storage.py111
-rw-r--r--src/mic_clipper/vad.py59
9 files changed, 623 insertions, 0 deletions
diff --git a/src/mic_clipper/__init__.py b/src/mic_clipper/__init__.py
new file mode 100644
index 0000000..7d39855
--- /dev/null
+++ b/src/mic_clipper/__init__.py
@@ -0,0 +1 @@
+"""Offline, voice-activated microphone clip recording."""
diff --git a/src/mic_clipper/assets/__init__.py b/src/mic_clipper/assets/__init__.py
new file mode 100644
index 0000000..480108b
--- /dev/null
+++ b/src/mic_clipper/assets/__init__.py
@@ -0,0 +1 @@
+"""Bundled, offline model assets."""
diff --git a/src/mic_clipper/audio_input.py b/src/mic_clipper/audio_input.py
new file mode 100644
index 0000000..1d77aa0
--- /dev/null
+++ b/src/mic_clipper/audio_input.py
@@ -0,0 +1,72 @@
+"""PipeWire/Pulse microphone capture through the local ffmpeg binary."""
+
+from __future__ import annotations
+
+import subprocess
+from collections.abc import Iterator
+
+import numpy as np
+
+from mic_clipper.vad import FRAME_SAMPLES
+
+
+class AudioInputUnavailable(RuntimeError):
+ """The dynamic Pulse default source could not supply another frame."""
+
+
+class PulseAudioInput:
+ """Yield fixed 16 kHz mono float PCM frames from Pulse's `default` source."""
+
+ @staticmethod
+ def command() -> list[str]:
+ return [
+ "ffmpeg",
+ "-nostdin",
+ "-hide_banner",
+ "-loglevel",
+ "error",
+ "-f",
+ "pulse",
+ "-sample_rate",
+ "16000",
+ "-channels",
+ "1",
+ "-i",
+ "default",
+ "-ac",
+ "1",
+ "-ar",
+ "16000",
+ "-f",
+ "f32le",
+ "pipe:1",
+ ]
+
+ def frames(self) -> Iterator[np.ndarray]:
+ process = subprocess.Popen(
+ self.command(),
+ stdin=subprocess.DEVNULL,
+ stdout=subprocess.PIPE,
+ stderr=subprocess.PIPE,
+ )
+ assert process.stdout is not None
+ assert process.stderr is not None
+ try:
+ while True:
+ payload = process.stdout.read(FRAME_SAMPLES * np.dtype("<f4").itemsize)
+ if len(payload) != FRAME_SAMPLES * np.dtype("<f4").itemsize:
+ error = process.stderr.read(4096).decode("utf-8", errors="replace").strip()
+ return_code = process.wait()
+ detail = f": {error}" if error else ""
+ raise AudioInputUnavailable(
+ f"ffmpeg Pulse input ended with exit code {return_code}{detail}"
+ )
+ yield np.frombuffer(payload, dtype="<f4").copy()
+ finally:
+ if process.poll() is None:
+ process.terminate()
+ try:
+ process.wait(timeout=5)
+ except subprocess.TimeoutExpired:
+ process.kill()
+ process.wait()
diff --git a/src/mic_clipper/cli.py b/src/mic_clipper/cli.py
new file mode 100644
index 0000000..cd62cef
--- /dev/null
+++ b/src/mic_clipper/cli.py
@@ -0,0 +1,66 @@
+"""Command-line entry points for recording and service administration."""
+
+from __future__ import annotations
+
+import argparse
+import logging
+import sys
+from pathlib import Path
+
+from mic_clipper import service
+from mic_clipper.runner import run_forever
+
+
+DEFAULT_OUTPUT = Path.home() / "Documents" / "Mic Clips"
+DEFAULT_UNIT = Path.home() / ".config" / "systemd" / "user" / service.UNIT_NAME
+PROJECT_ROOT = Path(__file__).resolve().parents[2]
+
+
+def build_parser() -> argparse.ArgumentParser:
+ parser = argparse.ArgumentParser(description="Offline voice-activated microphone clips")
+ commands = parser.add_subparsers(dest="command", required=True)
+
+ run = commands.add_parser("run", help="listen and save speech clips")
+ run.add_argument("--output-dir", type=Path, default=DEFAULT_OUTPUT)
+ run.add_argument("--threshold", type=float, default=0.5)
+
+ service_parser = commands.add_parser("service", help="manage the systemd user service")
+ service_commands = service_parser.add_subparsers(dest="service_command", required=True)
+ install = service_commands.add_parser("install", help="install and enable the user service")
+ install.add_argument("--dry-run", action="store_true")
+ install.add_argument("--unit-path", type=Path, default=DEFAULT_UNIT)
+ install.add_argument("--project-root", type=Path, default=PROJECT_ROOT)
+ install.add_argument("--python", type=Path, default=Path(sys.executable))
+ uninstall = service_commands.add_parser("uninstall", help="remove only this tool's service")
+ uninstall.add_argument("--unit-path", type=Path, default=DEFAULT_UNIT)
+ service_commands.add_parser("status", help="show systemd user service status")
+ return parser
+
+
+def main(arguments: list[str] | None = None) -> int:
+ args = build_parser().parse_args(arguments)
+ if args.command == "run":
+ if not 0 < args.threshold <= 1:
+ raise SystemExit("--threshold must be between 0 and 1")
+ logging.basicConfig(level=logging.INFO, format="%(levelname)s %(message)s")
+ run_forever(args.output_dir, threshold=args.threshold)
+ return 0
+ if args.service_command == "install":
+ commands = service.install(
+ unit_path=args.unit_path,
+ python=args.python,
+ project_root=args.project_root,
+ dry_run=args.dry_run,
+ )
+ if args.dry_run:
+ for command in commands:
+ print(" ".join(command))
+ return 0
+ if args.service_command == "uninstall":
+ service.uninstall(unit_path=args.unit_path)
+ return 0
+ return service.status()
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/src/mic_clipper/runner.py b/src/mic_clipper/runner.py
new file mode 100644
index 0000000..787bf58
--- /dev/null
+++ b/src/mic_clipper/runner.py
@@ -0,0 +1,92 @@
+"""Runtime wiring for capture, VAD, segmentation, and durable clips."""
+
+from __future__ import annotations
+
+import logging
+import time
+from collections.abc import Callable, Iterable
+from datetime import datetime
+from pathlib import Path
+from typing import Protocol
+
+import numpy as np
+
+from mic_clipper.audio_input import PulseAudioInput
+from mic_clipper.segmenter import Audio, End, Segmenter, Start
+from mic_clipper.storage import ClipWriter, OpenClip
+from mic_clipper.vad import SileroVad
+
+
+LOGGER = logging.getLogger(__name__)
+
+
+class Vad(Protocol):
+ def probability(self, samples: np.ndarray) -> float: ...
+
+
+class Writer(Protocol):
+ def open(self, started_at: datetime) -> OpenClip: ...
+
+
+def record_frames(
+ frames: Iterable[np.ndarray],
+ *,
+ vad: Vad,
+ writer: Writer,
+ segmenter: Segmenter,
+ now: Callable[[], datetime],
+ threshold: float = 0.5,
+) -> None:
+ """Record one finite or failing frame stream, closing an active encoder on exit."""
+ clip: OpenClip | None = None
+ try:
+ for samples in frames:
+ events = segmenter.push(samples, vad.probability(samples) >= threshold, now())
+ for event in events:
+ if isinstance(event, Start):
+ clip = writer.open(event.started_at)
+ try:
+ clip.write(event.samples)
+ except BaseException:
+ clip.abort()
+ clip = None
+ raise
+ elif isinstance(event, Audio):
+ if clip is None:
+ raise RuntimeError("segment audio arrived without an active encoder")
+ clip.write(event.samples)
+ elif isinstance(event, End):
+ if clip is None:
+ raise RuntimeError("segment end arrived without an active encoder")
+ clip.close()
+ clip = None
+ finally:
+ if clip is not None:
+ clip.close()
+
+
+def run_forever(root: Path, *, threshold: float = 0.5) -> None:
+ """Recreate all live components after a transient audio-service failure."""
+ retry_seconds = 1
+ while True:
+ try:
+ record_frames(
+ PulseAudioInput().frames(),
+ vad=SileroVad(),
+ writer=ClipWriter(root),
+ segmenter=Segmenter(pre_roll_seconds=0.5, silence_seconds=1.5),
+ now=lambda: datetime.now().astimezone(),
+ threshold=threshold,
+ )
+ raise RuntimeError("Pulse input ended unexpectedly")
+ except KeyboardInterrupt:
+ LOGGER.info("microphone clip recorder stopped")
+ return
+ except Exception as error:
+ LOGGER.warning(
+ "audio recorder unavailable (%s); retrying in %s seconds",
+ error,
+ retry_seconds,
+ )
+ time.sleep(retry_seconds)
+ retry_seconds = min(retry_seconds * 2, 60)
diff --git a/src/mic_clipper/segmenter.py b/src/mic_clipper/segmenter.py
new file mode 100644
index 0000000..aa0e3b2
--- /dev/null
+++ b/src/mic_clipper/segmenter.py
@@ -0,0 +1,120 @@
+"""Pure streaming voice-clip state machine."""
+
+from __future__ import annotations
+
+from collections import deque
+from dataclasses import dataclass
+from datetime import datetime, timedelta
+
+import numpy as np
+
+
+SAMPLE_RATE = 16_000
+
+
+@dataclass(frozen=True)
+class Start:
+ started_at: datetime
+ samples: np.ndarray
+
+
+@dataclass(frozen=True)
+class Audio:
+ samples: np.ndarray
+
+
+@dataclass(frozen=True)
+class End:
+ pass
+
+
+class Segmenter:
+ """Emit stream events while retaining only the configured pre-roll in memory."""
+
+ def __init__(
+ self,
+ *,
+ pre_roll_seconds: float,
+ silence_seconds: float,
+ max_segment_seconds: float = 600,
+ sample_rate: int = SAMPLE_RATE,
+ ) -> None:
+ self.sample_rate = sample_rate
+ self.pre_roll_samples = round(pre_roll_seconds * sample_rate)
+ self.silence_samples_limit = round(silence_seconds * sample_rate)
+ self.max_segment_samples = round(max_segment_seconds * sample_rate)
+ self._pre_roll: deque[np.ndarray] = deque()
+ self._pre_roll_size = 0
+ self._active = False
+ self._segment_samples = 0
+ self._silence_samples = 0
+
+ @property
+ def active(self) -> bool:
+ return self._active
+
+ def push(
+ self, samples: np.ndarray, is_speech: bool, captured_at: datetime
+ ) -> tuple[Start | Audio | End, ...]:
+ samples = np.asarray(samples, dtype=np.float32)
+ if samples.ndim != 1:
+ raise ValueError("audio must be a mono, one-dimensional array")
+ if not len(samples):
+ return ()
+
+ if not self._active:
+ if not is_speech:
+ self._append_pre_roll(samples)
+ return ()
+ pre_roll = self._take_pre_roll()
+ started_at = captured_at - timedelta(
+ seconds=len(pre_roll) / self.sample_rate
+ )
+ self._active = True
+ self._segment_samples = len(pre_roll) + len(samples)
+ self._silence_samples = 0
+ self._append_pre_roll(samples)
+ return (Start(started_at, np.concatenate((pre_roll, samples))),)
+
+ if self._segment_samples + len(samples) > self.max_segment_samples:
+ self._append_pre_roll(samples)
+ self._active = True
+ self._segment_samples = len(samples)
+ self._silence_samples = 0
+ return (End(), Start(captured_at, samples))
+
+ self._segment_samples += len(samples)
+ if is_speech:
+ self._silence_samples = 0
+ else:
+ self._silence_samples += len(samples)
+ self._append_pre_roll(samples)
+
+ events: tuple[Start | Audio | End, ...] = (Audio(samples),)
+ if self._silence_samples >= self.silence_samples_limit:
+ self._active = False
+ self._segment_samples = 0
+ self._silence_samples = 0
+ events += (End(),)
+ return events
+
+ def _append_pre_roll(self, samples: np.ndarray) -> None:
+ self._pre_roll.append(samples)
+ self._pre_roll_size += len(samples)
+ while self._pre_roll_size > self.pre_roll_samples:
+ excess = self._pre_roll_size - self.pre_roll_samples
+ oldest = self._pre_roll[0]
+ if len(oldest) <= excess:
+ self._pre_roll.popleft()
+ self._pre_roll_size -= len(oldest)
+ else:
+ self._pre_roll[0] = oldest[excess:]
+ self._pre_roll_size -= excess
+
+ def _take_pre_roll(self) -> np.ndarray:
+ if not self._pre_roll:
+ return np.empty(0, dtype=np.float32)
+ samples = np.concatenate(tuple(self._pre_roll))
+ self._pre_roll.clear()
+ self._pre_roll_size = 0
+ return samples
diff --git a/src/mic_clipper/service.py b/src/mic_clipper/service.py
new file mode 100644
index 0000000..ecf5673
--- /dev/null
+++ b/src/mic_clipper/service.py
@@ -0,0 +1,101 @@
+"""Install and remove the systemd user service without touching recordings."""
+
+from __future__ import annotations
+
+import os
+import subprocess
+import tempfile
+from collections.abc import Callable
+from pathlib import Path
+
+
+UNIT_NAME = "mic-clipper.service"
+
+
+def render_unit(python: Path, project_root: Path) -> str:
+ return f"""[Unit]
+Description=Offline voice-activated microphone clip recorder
+Wants=pipewire.service wireplumber.service pipewire-pulse.service
+After=pipewire.service wireplumber.service pipewire-pulse.service
+
+[Service]
+Type=simple
+WorkingDirectory={project_root}
+Environment=PYTHONPATH={project_root / 'src'}
+ExecStart={python} -m mic_clipper run
+Restart=on-failure
+RestartSec=5s
+RestartSteps=5
+RestartMaxDelaySec=60s
+StartLimitIntervalSec=300
+StartLimitBurst=10
+NoNewPrivileges=yes
+PrivateTmp=yes
+UMask=0077
+
+[Install]
+WantedBy=default.target
+"""
+
+
+def install(
+ *,
+ unit_path: Path,
+ python: Path,
+ project_root: Path,
+ dry_run: bool = False,
+ run_command: Callable[[list[str]], object] | None = None,
+) -> list[list[str]]:
+ commands = [
+ ["systemctl", "--user", "daemon-reload"],
+ ["systemctl", "--user", "enable", "--now", UNIT_NAME],
+ ]
+ if dry_run:
+ return commands
+ unit_path.parent.mkdir(mode=0o700, parents=True, exist_ok=True)
+ _write_private(unit_path, render_unit(python, project_root))
+ runner = run_command or _run_command
+ for command in commands:
+ runner(command)
+ return commands
+
+
+def uninstall(
+ *,
+ unit_path: Path,
+ run_command: Callable[[list[str]], object] | None = None,
+) -> list[list[str]]:
+ if not unit_path.exists():
+ return []
+ commands = [
+ ["systemctl", "--user", "disable", "--now", UNIT_NAME],
+ ["systemctl", "--user", "daemon-reload"],
+ ]
+ runner = run_command or _run_command
+ for command in commands[:1]:
+ runner(command)
+ unit_path.unlink()
+ runner(commands[1])
+ return commands
+
+
+def status() -> int:
+ return subprocess.run(
+ ["systemctl", "--user", "status", UNIT_NAME], check=False
+ ).returncode
+
+
+def _write_private(path: Path, content: str) -> None:
+ with tempfile.NamedTemporaryFile(
+ "w", encoding="utf-8", dir=path.parent, prefix=f".{path.name}.", delete=False
+ ) as temporary:
+ temporary.write(content)
+ temporary.flush()
+ os.fchmod(temporary.fileno(), 0o600)
+ temporary_path = Path(temporary.name)
+ os.replace(temporary_path, path)
+ os.chmod(path, 0o600)
+
+
+def _run_command(command: list[str]) -> None:
+ subprocess.run(command, check=True)
diff --git a/src/mic_clipper/storage.py b/src/mic_clipper/storage.py
new file mode 100644
index 0000000..a7f00bd
--- /dev/null
+++ b/src/mic_clipper/storage.py
@@ -0,0 +1,111 @@
+"""Private, streaming Opus clip storage."""
+
+from __future__ import annotations
+
+import os
+import subprocess
+from dataclasses import dataclass
+from datetime import datetime
+from pathlib import Path
+
+import numpy as np
+
+
+SAMPLE_RATE = 16_000
+
+
+@dataclass
+class OpenClip:
+ path: Path
+ process: subprocess.Popen[bytes]
+ closed: bool = False
+
+ def write(self, samples: np.ndarray) -> None:
+ if self.closed:
+ raise RuntimeError("cannot write a closed clip")
+ assert self.process.stdin is not None
+ audio = np.asarray(samples, dtype="<f4")
+ self.process.stdin.write(audio.tobytes())
+
+ def close(self) -> None:
+ if self.closed:
+ return
+ self.closed = True
+ assert self.process.stdin is not None
+ self.process.stdin.close()
+ assert self.process.stderr is not None
+ error_output = self.process.stderr.read().decode("utf-8", errors="replace").strip()
+ return_code = self.process.wait()
+ if return_code:
+ self.path.unlink(missing_ok=True)
+ raise RuntimeError(f"ffmpeg Opus encoding failed ({return_code}): {error_output}")
+ os.chmod(self.path, 0o600)
+
+ def abort(self) -> None:
+ if self.closed:
+ return
+ self.closed = True
+ self.process.kill()
+ self.process.wait()
+ self.path.unlink(missing_ok=True)
+
+
+class ClipWriter:
+ """Open one private Ogg/Opus file only after the segmenter emits a start."""
+
+ def __init__(self, root: Path) -> None:
+ self.root = Path(root).expanduser()
+
+ def open(self, started_at: datetime) -> OpenClip:
+ self.root.mkdir(mode=0o700, parents=True, exist_ok=True)
+ os.chmod(self.root, 0o700)
+ destination = self.root / started_at.date().isoformat()
+ destination.mkdir(mode=0o700, exist_ok=True)
+ os.chmod(destination, 0o700)
+ path = self._reserve_path(destination, started_at)
+ process = subprocess.Popen(
+ [
+ "ffmpeg",
+ "-nostdin",
+ "-v",
+ "error",
+ "-f",
+ "f32le",
+ "-ar",
+ str(SAMPLE_RATE),
+ "-ac",
+ "1",
+ "-i",
+ "pipe:0",
+ "-c:a",
+ "libopus",
+ "-b:a",
+ "24k",
+ "-application",
+ "voip",
+ "-vbr",
+ "constrained",
+ "-f",
+ "ogg",
+ "-y",
+ str(path),
+ ],
+ stdin=subprocess.PIPE,
+ stdout=subprocess.DEVNULL,
+ stderr=subprocess.PIPE,
+ )
+ return OpenClip(path, process)
+
+ @staticmethod
+ def _reserve_path(destination: Path, started_at: datetime) -> Path:
+ base_name = started_at.strftime("%H-%M-%S")
+ for suffix in range(10_000):
+ name = f"{base_name}.opus" if suffix == 0 else f"{base_name}-{suffix:02d}.opus"
+ path = destination / name
+ try:
+ descriptor = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
+ except FileExistsError:
+ continue
+ os.close(descriptor)
+ return path
+ raise RuntimeError("unable to reserve a unique clip filename")
diff --git a/src/mic_clipper/vad.py b/src/mic_clipper/vad.py
new file mode 100644
index 0000000..705dfd8
--- /dev/null
+++ b/src/mic_clipper/vad.py
@@ -0,0 +1,59 @@
+"""Stateful, local Silero VAD ONNX inference."""
+
+from __future__ import annotations
+
+from importlib import resources
+from pathlib import Path
+
+import numpy as np
+import onnxruntime
+
+
+FRAME_SAMPLES = 512
+CONTEXT_SAMPLES = 64
+
+
+class SileroVad:
+ """Run the bundled Silero v6 model on exact 32 ms PCM frames."""
+
+ def __init__(self, model_path: Path | None = None) -> None:
+ self.model_path = Path(model_path) if model_path else self._default_model_path()
+ options = onnxruntime.SessionOptions()
+ options.inter_op_num_threads = 1
+ options.intra_op_num_threads = 1
+ options.enable_cpu_mem_arena = False
+ options.log_severity_level = 4
+ self.session = onnxruntime.InferenceSession(
+ self.model_path,
+ providers=["CPUExecutionProvider"],
+ sess_options=options,
+ )
+ self._context = np.zeros(CONTEXT_SAMPLES, dtype=np.float32)
+ self._h = np.zeros((1, 1, 128), dtype=np.float32)
+ self._c = np.zeros((1, 1, 128), dtype=np.float32)
+
+ def probability(self, samples: np.ndarray) -> float:
+ samples = np.asarray(samples, dtype=np.float32)
+ if samples.ndim != 1 or len(samples) != FRAME_SAMPLES:
+ raise ValueError("Silero VAD requires exactly 512 mono samples at 16 kHz")
+ model_input = np.concatenate((self._context, samples)).reshape(1, -1)
+ output, self._h, self._c = self.session.run(
+ None,
+ {"input": model_input, "h": self._h, "c": self._c},
+ )
+ self._context = samples[-CONTEXT_SAMPLES:].copy()
+ return float(output[0])
+
+ @staticmethod
+ def _default_model_path() -> Path:
+ bundled = resources.files("mic_clipper.assets").joinpath("silero_vad_v6.onnx")
+ if bundled.is_file():
+ return Path(bundled)
+ try:
+ import faster_whisper
+ except ImportError as error:
+ raise RuntimeError(
+ "Silero VAD model is missing; bundle silero_vad_v6.onnx or install the "
+ "pre-provisioned faster-whisper runtime asset"
+ ) from error
+ return Path(faster_whisper.__file__).parent / "assets" / "silero_vad_v6.onnx"