From e14a6bfe9fc630ad2e8763fe396a1974b45be268 Mon Sep 17 00:00:00 2001 From: Florian Egger Date: Mon, 7 Sep 2026 11:04:53 +0200 Subject: [PATCH] Record at the input device's native sample rate - Recorder queries default_samplerate and streams/WAVs at that rate instead of forcing 16 kHz (helps devices that don't support it natively and keeps recordings at full fidelity) - new resample_16k() linear resampler: the live rolling window is converted to Whisper's 16 kHz before being handed to the engines - snapshot() now returns (16 kHz audio, absolute start time in s) - final pass unchanged: faster-whisper (PyAV) and whisper.cpp (miniaudio) resample native-rate WAVs themselves - GUI status shows the active rate; README updated Validated with real 44.1 kHz and 48 kHz inputs: WAV headers match the device rate, live-window resampling is length- and spectrum-correct. --- README.md | 4 +++- meetrec.py | 45 ++++++++++++++++++++++++++++++++++----------- meetrec_gui.py | 4 +++- 3 files changed, 40 insertions(+), 13 deletions(-) diff --git a/README.md b/README.md index 8d5bc0b..6b4a1cb 100644 --- a/README.md +++ b/README.md @@ -16,12 +16,14 @@ With output stem `-o meeting`: | File | Contents | | ------------------ | ----------------------------------------------- | -| `meeting.wav` | raw recording (16 kHz mono WAV) | +| `meeting.wav` | raw recording (mono WAV at the input's native rate) | | `meeting.live.srt` | live rolling transcript (only with `--live > 0`) | | `meeting.txt` | final transcript | | `meeting.srt` | final transcript with timestamps | | `meeting.json` | final transcript, structured (language, segments) | +Recording always uses the input device's native sample rate; audio is resampled to Whisper's 16 kHz automatically (both for the live window and the final pass). + ## Requirements - Linux with an audio backend: **PipeWire** (or PulseAudio/ALSA). On Arch: `sudo pacman -S pipewire pipewire-pulse` diff --git a/meetrec.py b/meetrec.py index d9a5780..c368022 100644 --- a/meetrec.py +++ b/meetrec.py @@ -8,12 +8,15 @@ printing a live rolling transcript. On Ctrl+C it stops cleanly and runs a final, higher-quality full transcription. Outputs (with -o meeting): - meeting.wav raw recording, 16 kHz mono WAV + meeting.wav raw recording, mono WAV at the input's native rate meeting.live.srt live rolling transcript (only when --live > 0) meeting.txt final transcript meeting.srt final transcript with timestamps meeting.json final transcript, structured +Audio is recorded at the input device's native sample rate; both the live +window and the engines get it resampled to Whisper's 16 kHz internally. + Setup on Arch: sudo pacman -S python pipewire pipewire-pulse python -m venv .venv && source .venv/bin/activate @@ -75,6 +78,22 @@ def write_srt(path: str, items) -> None: f.write(f"{i}\n{fmt_srt(a)} --> {fmt_srt(b)}\n{text}\n\n") +def resample_16k(audio: np.ndarray, rate: int) -> np.ndarray: + """Linearly resample mono float32 audio to 16 kHz (Whisper's rate). + + Only needed for the live rolling buffer, which is fed to the engines as + a raw array (they assume 16 kHz there). WAV files are resampled by the + engines themselves. + """ + audio = np.asarray(audio, dtype=np.float32) + if rate == SR or audio.size == 0: + return audio + n_out = max(1, int(round(audio.size * SR / rate))) + t_out = np.arange(n_out, dtype=np.float64) * (rate / SR) + return np.interp(t_out, np.arange(audio.size, dtype=np.float64), + audio).astype(np.float32) + + def list_sources() -> None: print("PortAudio (PipeWire/Pulse/ALSA) input devices:\n") for i, d in enumerate(sd.query_devices()): @@ -121,11 +140,15 @@ def resolve_source(name: str): # -------------------------------------------------------------------------- class Recorder: - """Records mono 16 kHz float32 to a WAV file and keeps a rolling buffer.""" + """Records mono float32 at the input's native rate to a WAV file and + keeps a rolling buffer for live transcription.""" def __init__(self, device, wav_path: str): self.device = device self.wav_path = wav_path + dev = sd.query_devices(kind="input") if device is None \ + else sd.query_devices(device) + self.rate = int(dev["default_samplerate"]) # input's native rate self.q: "queue.Queue" = queue.Queue() self.lock = threading.Lock() self.buf = np.zeros(0, dtype=np.float32) # rolling window @@ -146,7 +169,7 @@ class Recorder: block = indata[:, 0].astype(np.float32, copy=False) self.buf = np.append(self.buf, block) self.total += len(block) - limit = LIVE_WINDOW * SR + limit = LIVE_WINDOW * self.rate if len(self.buf) > limit: drop = len(self.buf) - limit self.buf = self.buf[drop:] @@ -154,13 +177,13 @@ class Recorder: def duration(self) -> float: with self.lock: - return self.total / SR + return self.total / self.rate def _write(self): with wave.open(self.wav_path, "wb") as w: w.setnchannels(1) w.setsampwidth(2) - w.setframerate(SR) + w.setframerate(self.rate) while True: block = self.q.get() if block is None: @@ -169,13 +192,13 @@ class Recorder: w.writeframes(pcm.tobytes()) def snapshot(self): - """Return (rolling audio, sample offset of its start).""" + """Return (rolling window resampled to 16 kHz, its start time in s).""" with self.lock: - return self.buf.copy(), self.base + return resample_16k(self.buf, self.rate), self.base / self.rate def start(self): self.stream = sd.InputStream( - samplerate=SR, channels=1, dtype="float32", + samplerate=self.rate, channels=1, dtype="float32", device=self.device, callback=self._callback) self.stream.start() @@ -224,7 +247,7 @@ def live_worker(model, rec: Recorder, args, live_srt_path: str, stop.wait(args.live) if stop.is_set(): break - buf, base = rec.snapshot() + buf, base_s = rec.snapshot() # 16 kHz audio, absolute start time if len(buf) < 2 * SR: # need at least 2 s of audio continue try: @@ -234,7 +257,6 @@ def live_worker(model, rec: Recorder, args, live_srt_path: str, print(f"[live] transcription error: {e}", file=sys.stderr) continue - base_s = base / SR added = False for s in segs: a, b = base_s + s.start, base_s + s.end @@ -538,7 +560,8 @@ def main(): rec = Recorder(device, args.output + ".wav") rec.start() - print(f"recording to {args.output}.wav (Ctrl+C to stop)", flush=True) + print(f"recording to {args.output}.wav at {rec.rate} Hz " + f"(Ctrl+C to stop)", flush=True) stop = threading.Event() worker = None diff --git a/meetrec_gui.py b/meetrec_gui.py index 03aa948..7021648 100644 --- a/meetrec_gui.py +++ b/meetrec_gui.py @@ -400,7 +400,9 @@ class MeetRecWindow(QWidget): name = sd.query_devices(idx)["name"] except Exception: name = self._args.source - self.status.setText(f"Recording from: {name} → {self._args.output}.wav") + self.status.setText( + f"Recording from: {name} @ {self.rec.rate} Hz → " + f"{self._args.output}.wav") if self._args.live: self.live_thread = threading.Thread(