diff options
| author | Somhairle H. Marisol <[email protected]> | 2026-09-29 10:50:12 +0800 |
|---|---|---|
| committer | Somhairle H. Marisol <[email protected]> | 2026-09-29 10:50:12 +0800 |
| commit | 60f05cebd8e7968621a0ce015cf72987ec7577c7 (patch) | |
| tree | 7f3c572ae82015b17ceaff6eb901ec88ba809856 /src/mic_clipper/audio_input.py | |
| download | mic-clipper-60f05cebd8e7968621a0ce015cf72987ec7577c7.tar.gz | |
feat(recorder): add offline voice clip pipeline
[变更性质]
- 本提交新增本机离线的语音片段录音能力,不涉及网络服务或语音识别。
[新增功能]
- 通过 Pulse 默认输入持续采集音频,以本地 Silero VAD 触发片段。
- 以私有目录和权限写入 24 kbps Ogg/Opus 录音,并提供 systemd 用户服务安装器。
[实现方案]
- 使用有限前置缓冲、静音封段和最长段滚动状态机,避免静音落盘及内存无限增长。
- 增加真实 Opus 编解码、服务 dry-run 和纯逻辑状态转换测试;记录模型归属和部署前置条件。
[影响范围]
- 新增 Python CLI、运行时模块、中文运维文档和自动测试。
- 未安装、启用或启动任何用户 systemd 服务;不删除或修改录音数据。
Diffstat (limited to 'src/mic_clipper/audio_input.py')
| -rw-r--r-- | src/mic_clipper/audio_input.py | 72 |
1 files changed, 72 insertions, 0 deletions
diff --git a/src/mic_clipper/audio_input.py b/src/mic_clipper/audio_input.py new file mode 100644 index 0000000..1d77aa0 --- /dev/null +++ b/src/mic_clipper/audio_input.py @@ -0,0 +1,72 @@ +"""PipeWire/Pulse microphone capture through the local ffmpeg binary.""" + +from __future__ import annotations + +import subprocess +from collections.abc import Iterator + +import numpy as np + +from mic_clipper.vad import FRAME_SAMPLES + + +class AudioInputUnavailable(RuntimeError): + """The dynamic Pulse default source could not supply another frame.""" + + +class PulseAudioInput: + """Yield fixed 16 kHz mono float PCM frames from Pulse's `default` source.""" + + @staticmethod + def command() -> list[str]: + return [ + "ffmpeg", + "-nostdin", + "-hide_banner", + "-loglevel", + "error", + "-f", + "pulse", + "-sample_rate", + "16000", + "-channels", + "1", + "-i", + "default", + "-ac", + "1", + "-ar", + "16000", + "-f", + "f32le", + "pipe:1", + ] + + def frames(self) -> Iterator[np.ndarray]: + process = subprocess.Popen( + self.command(), + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + ) + assert process.stdout is not None + assert process.stderr is not None + try: + while True: + payload = process.stdout.read(FRAME_SAMPLES * np.dtype("<f4").itemsize) + if len(payload) != FRAME_SAMPLES * np.dtype("<f4").itemsize: + error = process.stderr.read(4096).decode("utf-8", errors="replace").strip() + return_code = process.wait() + detail = f": {error}" if error else "" + raise AudioInputUnavailable( + f"ffmpeg Pulse input ended with exit code {return_code}{detail}" + ) + yield np.frombuffer(payload, dtype="<f4").copy() + finally: + if process.poll() is None: + process.terminate() + try: + process.wait(timeout=5) + except subprocess.TimeoutExpired: + process.kill() + process.wait() |
