summaryrefslogtreecommitdiff
path: root/src/mic_clipper/audio_input.py
diff options
context:
space:
mode:
authorSomhairle H. Marisol <[email protected]>2026-09-29 10:50:12 +0800
committerSomhairle H. Marisol <[email protected]>2026-09-29 10:50:12 +0800
commit60f05cebd8e7968621a0ce015cf72987ec7577c7 (patch)
tree7f3c572ae82015b17ceaff6eb901ec88ba809856 /src/mic_clipper/audio_input.py
downloadmic-clipper-60f05cebd8e7968621a0ce015cf72987ec7577c7.tar.gz
feat(recorder): add offline voice clip pipeline
[变更性质] - 本提交新增本机离线的语音片段录音能力,不涉及网络服务或语音识别。 [新增功能] - 通过 Pulse 默认输入持续采集音频,以本地 Silero VAD 触发片段。 - 以私有目录和权限写入 24 kbps Ogg/Opus 录音,并提供 systemd 用户服务安装器。 [实现方案] - 使用有限前置缓冲、静音封段和最长段滚动状态机,避免静音落盘及内存无限增长。 - 增加真实 Opus 编解码、服务 dry-run 和纯逻辑状态转换测试;记录模型归属和部署前置条件。 [影响范围] - 新增 Python CLI、运行时模块、中文运维文档和自动测试。 - 未安装、启用或启动任何用户 systemd 服务;不删除或修改录音数据。
Diffstat (limited to 'src/mic_clipper/audio_input.py')
-rw-r--r--src/mic_clipper/audio_input.py72
1 files changed, 72 insertions, 0 deletions
diff --git a/src/mic_clipper/audio_input.py b/src/mic_clipper/audio_input.py
new file mode 100644
index 0000000..1d77aa0
--- /dev/null
+++ b/src/mic_clipper/audio_input.py
@@ -0,0 +1,72 @@
+"""PipeWire/Pulse microphone capture through the local ffmpeg binary."""
+
+from __future__ import annotations
+
+import subprocess
+from collections.abc import Iterator
+
+import numpy as np
+
+from mic_clipper.vad import FRAME_SAMPLES
+
+
+class AudioInputUnavailable(RuntimeError):
+ """The dynamic Pulse default source could not supply another frame."""
+
+
+class PulseAudioInput:
+ """Yield fixed 16 kHz mono float PCM frames from Pulse's `default` source."""
+
+ @staticmethod
+ def command() -> list[str]:
+ return [
+ "ffmpeg",
+ "-nostdin",
+ "-hide_banner",
+ "-loglevel",
+ "error",
+ "-f",
+ "pulse",
+ "-sample_rate",
+ "16000",
+ "-channels",
+ "1",
+ "-i",
+ "default",
+ "-ac",
+ "1",
+ "-ar",
+ "16000",
+ "-f",
+ "f32le",
+ "pipe:1",
+ ]
+
+ def frames(self) -> Iterator[np.ndarray]:
+ process = subprocess.Popen(
+ self.command(),
+ stdin=subprocess.DEVNULL,
+ stdout=subprocess.PIPE,
+ stderr=subprocess.PIPE,
+ )
+ assert process.stdout is not None
+ assert process.stderr is not None
+ try:
+ while True:
+ payload = process.stdout.read(FRAME_SAMPLES * np.dtype("<f4").itemsize)
+ if len(payload) != FRAME_SAMPLES * np.dtype("<f4").itemsize:
+ error = process.stderr.read(4096).decode("utf-8", errors="replace").strip()
+ return_code = process.wait()
+ detail = f": {error}" if error else ""
+ raise AudioInputUnavailable(
+ f"ffmpeg Pulse input ended with exit code {return_code}{detail}"
+ )
+ yield np.frombuffer(payload, dtype="<f4").copy()
+ finally:
+ if process.poll() is None:
+ process.terminate()
+ try:
+ process.wait(timeout=5)
+ except subprocess.TimeoutExpired:
+ process.kill()
+ process.wait()