Files
meetrec/meetrec.py
T
fegger c4b86213d4 Desktop: M3 feature parity with the Android app
- final_transcribe gains optional speaker labeling and library upload:
  diarize_url runs a second tinydiarize pass and merges the turn times
  onto the quality transcript as 'Sprecher 1/2:' labels (diarize_merge,
  port of the Android Diarization object; failures keep the transcript
  unlabeled); library_url uploads the WAV + txt/srt/json bundle with the
  agenda — the server then generates summary + agenda coverage
- WhisperServerEngine supports diarize mode (tinydiarize form field +
  speaker_turn_next parsing); multipart builder handles multiple files
- CLI: --agenda-file, --diarize-url, --library-url (env vars
  MEETREC_DIARIZE_URL / MEETREC_LIBRARY_URL)
- GUI: Library URL + Diarize URL fields and an agenda editor; upload
  progress lands in the status bar
- validated end-to-end against the live servers: recording -> large-v3
  transcription -> txt/srt/json -> library upload (all four files,
  device=desktop) -> server summary/agenda pipeline ran (honest error
  for a noise-only test recording)
2026-09-08 21:04:53 +02:00

823 lines
33 KiB
Python

#!/usr/bin/env python3
"""
meetrec.py — record an in-person meeting and transcribe it with Whisper.
Records from any PipeWire/ALSA input (microphone, USB conference mic,
USB audio interface, or a Pulse/PipeWire sink "monitor") while optionally
printing a live rolling transcript. On Ctrl+C it stops cleanly and runs a
final, higher-quality full transcription.
Outputs (with -o meeting):
meeting.wav raw recording, mono WAV at the input's native rate
meeting.live.srt live rolling transcript (only when --live > 0)
meeting.txt final transcript
meeting.srt final transcript with timestamps
meeting.json final transcript, structured
Audio is recorded at the input device's native sample rate; both the live
window and the engines get it resampled to Whisper's 16 kHz internally.
Setup on Arch:
sudo pacman -S python pipewire pipewire-pulse
python -m venv .venv && source .venv/bin/activate
pip install numpy sounddevice faster-whisper
Usage:
python meetrec.py --list-sources
python meetrec.py -s "Conference Mic" --model small --live 8 -o meeting
python meetrec.py -s monitor --model base --language en # record system audio
python meetrec.py --engine whisper-cpp -o meeting # GPU via Vulkan (AMD/NVIDIA)
Ctrl+C to stop -> final transcripts are written.
"""
import argparse
import json
import os
import queue
import shutil
import subprocess
import sys
import tempfile
import threading
import time
import urllib.error
import urllib.request
import wave
from pathlib import Path
from types import SimpleNamespace
import numpy as np
import sounddevice as sd
SR = 16000 # sample rate — what Whisper wants
LIVE_WINDOW = 60 # seconds of rolling audio kept for live transcription
# --------------------------------------------------------------------------
# Helpers
# --------------------------------------------------------------------------
def fmt_hms(t: float) -> str:
h, rem = divmod(int(t), 3600)
m, s = divmod(rem, 60)
return f"{h:02d}:{m:02d}:{s:02d}"
def fmt_srt(t: float) -> str:
ms = int(round((t - int(t)) * 1000))
h, rem = divmod(int(t) + (1 if ms >= 1000 else 0), 3600)
m, s = divmod(rem, 60)
if ms >= 1000:
ms -= 1000
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
def write_srt(path: str, items) -> None:
with open(path, "w") as f:
for i, (a, b, text) in enumerate(items, 1):
f.write(f"{i}\n{fmt_srt(a)} --> {fmt_srt(b)}\n{text}\n\n")
def resample_16k(audio: np.ndarray, rate: int) -> np.ndarray:
"""Linearly resample mono float32 audio to 16 kHz (Whisper's rate).
Only needed for the live rolling buffer, which is fed to the engines as
a raw array (they assume 16 kHz there). WAV files are resampled by the
engines themselves.
"""
audio = np.asarray(audio, dtype=np.float32)
if rate == SR or audio.size == 0:
return audio
n_out = max(1, int(round(audio.size * SR / rate)))
t_out = np.arange(n_out, dtype=np.float64) * (rate / SR)
return np.interp(t_out, np.arange(audio.size, dtype=np.float64),
audio).astype(np.float32)
def list_sources() -> None:
print("PortAudio (PipeWire/Pulse/ALSA) input devices:\n")
for i, d in enumerate(sd.query_devices()):
if d["max_input_channels"] > 0:
print(f" {i:>3} {d['name']}\n"
f" api={d['hostapi']} in={d['max_input_channels']}ch "
f"rate={d['default_samplerate']:.0f}")
pactl = shutil.which("pactl")
if pactl:
print("\nPulseAudio/PipeWire sources (pactl):\n")
try:
out = subprocess.run(
[pactl, "list", "short", "sources"],
capture_output=True, text=True, timeout=5).stdout.strip()
for line in out.splitlines():
print(" " + line)
except Exception:
pass
print("\nTip: sink monitors (record what the computer plays) appear as "
"\"<sink name>.monitor\". Match them with -s <part of name>.")
def resolve_source(name: str):
"""Return a sounddevice device index, or None for the default device."""
if name in ("default", None, ""):
return None
inputs = [(i, d) for i, d in enumerate(sd.query_devices())
if d["max_input_channels"] > 0]
matches = [(i, d) for i, d in inputs if name.lower() in d["name"].lower()]
if not matches:
print(f"error: no input device matching {name!r}. "
f"Run with --list-sources.", file=sys.stderr)
sys.exit(1)
if len(matches) > 1:
print(f"note: multiple devices match {name!r}, using: "
f"{matches[0][1]['name']}", file=sys.stderr)
idx, dev = matches[0]
print(f"recording from: [{idx}] {dev['name']} ({dev['hostapi']})")
return idx
# --------------------------------------------------------------------------
# Recording
# --------------------------------------------------------------------------
class Recorder:
"""Records mono float32 at the input's native rate to a WAV file and
keeps a rolling buffer for live transcription."""
def __init__(self, device, wav_path: str):
self.device = device
self.wav_path = wav_path
dev = sd.query_devices(kind="input") if device is None \
else sd.query_devices(device)
self.rate = int(dev["default_samplerate"]) # input's native rate
self.q: "queue.Queue" = queue.Queue()
self.lock = threading.Lock()
self.buf = np.zeros(0, dtype=np.float32) # rolling window
self.base = 0 # samples recorded before buf
self.total = 0 # total samples recorded
self.peak = 0.0 # peak amplitude of last block
self.stop_event = threading.Event()
self.stream = None
self._writer = threading.Thread(target=self._write, daemon=True)
self._writer.start()
def _callback(self, indata, frames, time_info, status):
if status:
print(f"[rec] status: {status}", file=sys.stderr)
self.peak = float(np.max(np.abs(indata[:, 0]))) if frames else 0.0
self.q.put(indata.copy())
with self.lock:
block = indata[:, 0].astype(np.float32, copy=False)
self.buf = np.append(self.buf, block)
self.total += len(block)
limit = LIVE_WINDOW * self.rate
if len(self.buf) > limit:
drop = len(self.buf) - limit
self.buf = self.buf[drop:]
self.base += drop
def duration(self) -> float:
with self.lock:
return self.total / self.rate
def _write(self):
with wave.open(self.wav_path, "wb") as w:
w.setnchannels(1)
w.setsampwidth(2)
w.setframerate(self.rate)
while True:
block = self.q.get()
if block is None:
break
pcm = (np.clip(block[:, 0], -1.0, 1.0) * 32767).astype(np.int16)
w.writeframes(pcm.tobytes())
def snapshot(self):
"""Return (rolling window resampled to 16 kHz, its start time in s)."""
with self.lock:
return resample_16k(self.buf, self.rate), self.base / self.rate
def start(self):
self.stream = sd.InputStream(
samplerate=self.rate, channels=1, dtype="float32",
device=self.device, callback=self._callback)
self.stream.start()
def stop(self):
if self.stop_event.is_set():
return # idempotent: the GUI close path may call this twice
self.stop_event.set()
self.q.put(None)
if self.stream:
self.stream.stop()
self.stream.close()
# The writer must drain the queue and close the WAV before we return,
# otherwise the file can be truncated. Join in slices instead of one
# long timeout so a slow disk still gets the full flush; only warn
# after a total of 60 s (effectively never in practice).
waited = 0.0
while self._writer.is_alive() and waited < 60:
self._writer.join(timeout=2.0)
waited += 2.0
if self._writer.is_alive():
print("warning: WAV writer still running after 60 s; "
"the file may be truncated", file=sys.stderr)
# --------------------------------------------------------------------------
# Transcription
# --------------------------------------------------------------------------
def live_worker(model, rec: Recorder, args, live_srt_path: str,
stop: threading.Event, on_line=None):
"""Transcribes the rolling window every `args.live` seconds and appends
newly-finalized segments to the live .srt. If `on_line` is given, new
lines are delivered to it (a, text) instead of being printed."""
def emit(a, text):
if on_line is not None:
on_line(a, text)
else:
print(f"[{fmt_hms(a)}] {text}", flush=True)
items = [] # (abs_start, abs_end, text) for the live .srt
last_end = 0.0
if on_line is None:
print("live transcription on (Ctrl+C to stop)", flush=True)
while not stop.is_set():
stop.wait(args.live)
if stop.is_set():
break
buf, base_s = rec.snapshot() # 16 kHz audio, absolute start time
if len(buf) < 2 * SR: # need at least 2 s of audio
continue
try:
segs, _info = model.transcribe(
buf, beam_size=1, vad_filter=True, language=args.language)
except Exception as e:
print(f"[live] transcription error: {e}", file=sys.stderr)
continue
added = False
for s in segs:
a, b = base_s + s.start, base_s + s.end
text = s.text.strip()
if not text or b < last_end:
continue
items.append((a, b, text))
last_end = max(last_end, b)
emit(a, text)
added = True
if added:
write_srt(live_srt_path, items)
def final_transcribe(model, wav_path: str, out_stem: str, language,
on_info=None, on_segment=None, on_done=None,
diarize_url=None, agenda=None, library_url=None,
device_name="desktop"):
"""Full-pass transcription of the whole recording, with optional
speaker labeling (tinydiarize server) and library upload.
on_info(msg), on_segment(start, end, text) and on_done() are optional
callbacks; when absent the same progress is printed to stdout.
Returns (items, info): items are (start, end, text) tuples of the
final — possibly speaker-labeled — transcript."""
def info(msg):
if on_info is not None:
on_info(msg)
else:
print(msg, flush=True)
def emit(a, b, text):
if on_segment is not None:
on_segment(a, b, text)
else:
print(f"[{fmt_hms(a)}] {text}", flush=True)
info("\nRunning final transcription (this can take a while)...")
segments, info_ = model.transcribe(
wav_path, beam_size=5, vad_filter=True, language=language,
vad_parameters=dict(min_silence_duration_ms=500))
info(f"language: {info_.language} (p={info_.language_probability:.2f}) "
f"duration: {fmt_hms(info_.duration)}")
items = []
if diarize_url:
for s in segments: # collect first: labels need the whole pass
items.append((s.start, s.end, s.text.strip()))
info("Running speaker-diarization pass (tinydiarize server)...")
try:
tdrz = WhisperServerEngine(diarize_url, diarize=True)
turns = tdrz.transcribe(wav_path, language=language)
labeled = diarize_merge(
[SimpleNamespace(start=a, end=b, text=t) for a, b, t in items],
turns)
items = [(s.start, s.end, s.text) for s in labeled]
except Exception as e:
info(f"diarize pass failed (continuing unlabeled): {e}")
for a, b, text in items:
emit(a, b, text)
else:
for s in segments: # generator: streams as it decodes
text = s.text.strip()
items.append((s.start, s.end, text))
emit(s.start, s.end, text)
full = " ".join(t for _, _, t in items)
with open(out_stem + ".txt", "w") as f:
f.write(full + "\n")
write_srt(out_stem + ".srt", items)
with open(out_stem + ".json", "w") as f:
json.dump({
"language": info_.language,
"duration": round(info_.duration, 2),
"segments": [{"start": round(a, 2), "end": round(b, 2), "text": t}
for a, b, t in items],
}, f, ensure_ascii=False, indent=2)
info(f"\nwrote: {out_stem}.wav {out_stem}.txt {out_stem}.srt {out_stem}.json")
if library_url:
info("Uploading recording to the library...")
try:
rid = upload_recording(
library_url, wav_path, out_stem, info_.duration,
getattr(info_, "language", None), device_name, agenda or [])
info(f"Uploaded to library as {rid} — summary and agenda check "
"follow automatically on the server.")
except Exception as e:
info(f"upload failed: {e}")
if on_done is not None:
on_done()
return items, info_
# --------------------------------------------------------------------------
# Engines
# --------------------------------------------------------------------------
#
# Two interchangeable transcription backends:
# faster-whisper CTranslate2 — NVIDIA CUDA or CPU (default)
# whisper-cpp whisper.cpp CLI — GPU via Vulkan (AMD, Intel, NVIDIA)
# or CPU; preferred on AMD hardware (e.g. Ryzen AI laptops)
# whisper-server remote whisper.cpp HTTP server (server/whisper-server) —
# the model lives on the server; ideal for phones/laptops
#
# Both expose: transcribe(audio, beam_size, vad_filter, language,
# vad_parameters) -> (segments, info)
# where `audio` is a 1-D float32 @16 kHz array or a WAV path, segments have
# .start/.end/.text and info has .language/.language_probability/.duration.
WHISPER_CPP_MODEL_FILES = {
"tiny": "ggml-tiny.bin",
"base": "ggml-base.bin",
"small": "ggml-small.bin",
"medium": "ggml-medium.bin",
"large-v3": "ggml-large-v3.bin",
}
WHISPER_CPP_MODEL_URL = ("https://huggingface.co/ggerganov/whisper.cpp/"
"resolve/main/{name}")
WHISPER_CPP_VAD_FILE = "ggml-silero-v6.2.0.bin"
WHISPER_CPP_VAD_URL = ("https://huggingface.co/ggml-org/whisper-vad/"
"resolve/main/" + WHISPER_CPP_VAD_FILE)
def _download_file(url: str, dest: Path) -> None:
tmp = dest.with_suffix(dest.suffix + ".part")
req = urllib.request.Request(url, headers={"User-Agent": "meetrec"})
with urllib.request.urlopen(req, timeout=60) as r:
total = int(r.headers.get("Content-Length") or 0)
done = 0
with open(tmp, "wb") as f:
while True:
chunk = r.read(1024 * 1024)
if not chunk:
break
f.write(chunk)
done += len(chunk)
if total:
print(f"\r {done/1e6:.0f}/{total/1e6:.0f} MB", end="",
flush=True)
if total:
print(flush=True)
os.replace(tmp, dest)
def find_whisper_cpp_bin() -> str:
"""Locate the whisper.cpp 'whisper-cli' binary."""
candidates = [
os.environ.get("WHISPER_CPP_BIN"),
shutil.which("whisper-cli"),
str(Path.home() / ".local/share/meetrec/whisper-cpp/bin/whisper-cli"),
]
for c in candidates:
if c and os.path.isfile(c) and os.access(c, os.X_OK):
return c
raise RuntimeError(
"whisper-cli not found. Build whisper.cpp (e.g. './install.sh "
"--whisper-cpp', or cmake -DWHISPER_VULKAN=ON) or set WHISPER_CPP_BIN "
"to the binary.")
def _whisper_cpp_cache() -> Path:
cache = Path(os.environ.get(
"MEETREC_CACHE", str(Path.home() / ".cache/meetrec")))
d = cache / "whisper-cpp"
d.mkdir(parents=True, exist_ok=True)
return d
def _cached_ggml(fname: str, url: str) -> str:
"""Return the path of a cached GGML file, downloading it if missing."""
p = _whisper_cpp_cache() / fname
if p.exists() and p.stat().st_size > 0:
return str(p)
print(f"downloading {fname} to {p} ...", flush=True)
_download_file(url, p)
return str(p)
def whisper_cpp_model_path(model_name: str) -> str:
"""Return the local path of a whisper.cpp GGML model, downloading it."""
try:
fname = WHISPER_CPP_MODEL_FILES[model_name]
except KeyError:
raise RuntimeError(
f"unknown model {model_name!r}; use one of: "
f"{', '.join(WHISPER_CPP_MODEL_FILES)}")
return _cached_ggml(fname, WHISPER_CPP_MODEL_URL.format(name=fname))
def write_wav_float(path: str, audio: np.ndarray) -> None:
"""Write a 1-D float32 mono @16 kHz array as 16-bit PCM WAV."""
audio = np.asarray(audio, dtype=np.float32).ravel()
pcm = (np.clip(audio, -1.0, 1.0) * 32767).astype(np.int16)
with wave.open(path, "wb") as w:
w.setnchannels(1)
w.setsampwidth(2)
w.setframerate(SR)
w.writeframes(pcm.tobytes())
def _parse_ts(s: str) -> float:
"""whisper.cpp timestamp 'HH:MM:SS,mmm' -> seconds."""
h, m, rest = s.split(":")
sec, ms = rest.split(",")
return int(h) * 3600 + int(m) * 60 + int(sec) + int(ms) / 1000.0
class WhisperCppEngine:
"""Transcription via the whisper.cpp 'whisper-cli' binary.
Build whisper.cpp with -DWHISPER_VULKAN=ON to run on a GPU (Radeon,
Intel, or NVIDIA); it falls back to CPU otherwise. GPU vs. CPU is decided
by the build, so --device/--compute-type do not apply. vad_filter=True
enables whisper.cpp's Silero VAD (auto-downloaded); vad_parameters are
accepted but not mapped.
"""
def __init__(self, model_name: str, cli: str = None):
self.model_name = model_name
self.cli = cli or find_whisper_cpp_bin()
self.model_path = whisper_cpp_model_path(model_name)
def transcribe(self, audio, beam_size=5, vad_filter=False,
language=None, vad_parameters=None):
# vad_parameters accepted for interface compatibility; Silero VAD uses
# whisper.cpp's own defaults
with tempfile.TemporaryDirectory(prefix="meetrec-") as td:
if isinstance(audio, (str, os.PathLike)):
wav = str(audio)
with wave.open(wav, "rb") as w:
duration = w.getnframes() / w.getframerate()
else:
wav = os.path.join(td, "in.wav")
write_wav_float(wav, audio)
duration = len(audio) / SR
cmd = [self.cli, "-m", self.model_path, "-f", wav,
"-oj", "-of", os.path.join(td, "out"),
"-bs", str(max(1, int(beam_size))), "-np",
"-l", language or "auto"]
if vad_filter:
cmd += ["--vad", "-vm", _cached_ggml(
WHISPER_CPP_VAD_FILE, WHISPER_CPP_VAD_URL)]
proc = subprocess.run(cmd, capture_output=True, text=True)
if proc.returncode != 0:
raise RuntimeError(
(proc.stderr or proc.stdout).strip()
or f"whisper-cli exited {proc.returncode}")
with open(os.path.join(td, "out.json")) as f:
data = json.load(f)
result = data.get("result", {})
segs = []
for item in data.get("transcription", []):
off = item.get("offsets", {})
if "from" in off and "to" in off: # milliseconds (current)
a, b = off["from"] / 1000.0, off["to"] / 1000.0
else: # 'HH:MM:SS,mmm' (older)
ts = item.get("timestamps", {})
a = _parse_ts(ts.get("from", "00:00:00,000"))
b = _parse_ts(ts.get("to", "00:00:00,000"))
text = item.get("text", "").strip()
if text:
segs.append(SimpleNamespace(start=a, end=b, text=text))
info = SimpleNamespace(
language=result.get("language") or language or "unknown",
language_probability=1.0, # not reported by whisper.cpp
duration=duration,
)
return segs, info
MULTIPART_BOUNDARY = "meetrec-9f3e1c7a5b2d"
def _multipart_body(fields: dict, files) -> bytes:
"""Build a multipart/form-data body with stdlib only.
files: list of (field_name, filename, file_obj) tuples."""
parts = []
for name, value in fields.items():
parts.append(
f"--{MULTIPART_BOUNDARY}\r\n"
f'Content-Disposition: form-data; name="{name}"\r\n\r\n'
f"{value}\r\n".encode())
for name, fname, fh in files:
parts.append(
f"--{MULTIPART_BOUNDARY}\r\n"
f'Content-Disposition: form-data; name="{name}"; '
f'filename="{fname}"\r\n'
f"Content-Type: application/octet-stream\r\n\r\n".encode())
parts.append(fh.read())
parts.append(b"\r\n")
parts.append(f"--{MULTIPART_BOUNDARY}--\r\n".encode())
return b"".join(parts)
class WhisperServerEngine:
"""Transcription via a remote whisper.cpp server (POST /inference).
Use the Docker Compose stack in server/whisper-server/ to run one
(Vulkan GPU on AMD, or CPU). The model lives on the server, so
--model / --device / --compute-type do not apply, and beam size is a
server-start setting in whisper.cpp v1.9.3 (no per-request override) —
beam_size is accepted for interface compatibility but ignored.
vad_filter/vad_parameters are likewise accepted but unused; the server
runs its own pipeline. Note: verbose_json reports language names like
"german" rather than ISO codes.
With diarize=True (a tinydiarize server, see whisper-server-tdrz)
segments carry .speaker_turn_next and their text is NOT used for
labeling — use diarize_merge() for that.
"""
def __init__(self, server_url: str, diarize: bool = False):
self.server_url = server_url.rstrip("/")
self.diarize = diarize
self.model_name = None # the server owns the model
def transcribe(self, audio, beam_size=5, vad_filter=False,
language=None, vad_parameters=None, timeout=600):
with tempfile.TemporaryDirectory(prefix="meetrec-") as td:
if isinstance(audio, (str, os.PathLike)):
wav = str(audio) # server resamples to 16 kHz itself
else:
wav = os.path.join(td, "in.wav")
write_wav_float(wav, audio)
files = [("file", "audio.wav", open(wav, "rb"))]
fields = {"response_format": "verbose_json",
"language": language or "auto"}
if self.diarize:
fields["tinydiarize"] = "true"
try:
body = _multipart_body(fields, files)
finally:
files[0][2].close()
req = urllib.request.Request(
self.server_url + "/inference", data=body,
headers={"Content-Type":
"multipart/form-data; boundary=" + MULTIPART_BOUNDARY})
try:
with urllib.request.urlopen(req, timeout=timeout) as r:
data = json.load(r)
except urllib.error.HTTPError as e:
raise RuntimeError(f"server returned {e.code}: "
f"{e.read()[:300]!r}") from None
except urllib.error.URLError as e:
raise RuntimeError(f"cannot reach whisper server at "
f"{self.server_url}: {e.reason}") from None
segs = [SimpleNamespace(start=s["start"], end=s["end"],
text=s["text"].strip(),
speaker_turn_next=bool(s.get("speaker_turn_next", False)))
for s in data.get("segments", [])
if s.get("text", "").strip()]
info = SimpleNamespace(
language=data.get("language") or language or "unknown",
language_probability=1.0, # not reported by verbose_json
duration=data.get("duration", 0.0),
)
return segs, info
def diarize_merge(segments, turn_segments):
"""Label segments with alternating "Sprecher 1/2:" using the TURN
TIMES of a tinydiarize pass; segments in which a turn falls are split.
Input is returned unchanged when no turns were detected (never
mislabels). Mirrors the Android app's Diarization object."""
turns = sorted(t.end for t in turn_segments
if getattr(t, "speaker_turn_next", False))
if not turns:
return list(segments)
def seg(a, b, text):
return SimpleNamespace(start=a, end=b, text=text)
speaker, ti = 1, 0
out = []
for s in segments:
while ti < len(turns) and turns[ti] <= s.start:
speaker = 3 - speaker
ti += 1
start, text = s.start, s.text
while ti < len(turns) and turns[ti] < s.end:
t = turns[ti]
frac = (t - s.start) / max(s.end - s.start, 1e-9)
cut = int(len(text) * frac)
out.append(seg(start, t, f"Sprecher {speaker}: {text[:cut].strip()}"))
speaker = 3 - speaker
ti += 1
start, text = t, text[cut:]
out.append(seg(start, s.end, f"Sprecher {speaker}: {text.strip()}"))
return out
def upload_recording(library_url: str, wav_path: str, out_stem: str,
duration: float, language, device: str, agenda) -> str:
"""Upload a recording bundle to the meetrec-server library;
returns the new recording id. The server then generates the summary
and agenda coverage automatically (Ollama)."""
files = []
for ext in ("wav", "txt", "srt", "json"):
p = wav_path if ext == "wav" else out_stem + "." + ext
if os.path.isfile(p):
files.append((ext, "meeting." + ext, open(p, "rb")))
started = time.strftime("%Y%m%d-%H%M%S",
time.localtime(os.path.getmtime(wav_path)))
fields = {
"started_at": started,
"duration_ms": int(duration * 1000),
"language": language or "",
"device": device,
"agenda": json.dumps(list(agenda or []), ensure_ascii=False),
}
try:
body = _multipart_body(fields, files)
finally:
for _, _, fh in files:
fh.close()
req = urllib.request.Request(
library_url.rstrip("/") + "/api/recordings", data=body, method="POST",
headers={"Content-Type": "multipart/form-data; boundary=" + MULTIPART_BOUNDARY})
try:
with urllib.request.urlopen(req, timeout=300) as r:
return json.load(r)["id"]
except urllib.error.HTTPError as e:
raise RuntimeError(f"library upload failed: {e.code} "
f"{e.read()[:200]!r}") from None
except urllib.error.URLError as e:
raise RuntimeError(f"cannot reach library at {library_url}: "
f"{e.reason}") from None
def load_engine(model_name, engine="faster-whisper",
device="auto", compute_type="int8", server_url=None):
"""Create a transcription engine. See the Engines section above."""
if engine == "whisper-cpp":
return WhisperCppEngine(model_name)
if engine == "whisper-server":
url = server_url or os.environ.get("MEETREC_SERVER_URL")
if not url:
raise RuntimeError("no server URL: pass --server-url or set "
"MEETREC_SERVER_URL (e.g. http://100.103.83.12:8085)")
return WhisperServerEngine(url)
if engine == "faster-whisper":
from faster_whisper import WhisperModel
return WhisperModel(model_name, device=device,
compute_type=compute_type)
raise RuntimeError(f"unknown engine {engine!r}; use 'faster-whisper' "
"or 'whisper-cpp'")
# --------------------------------------------------------------------------
# Main
# --------------------------------------------------------------------------
def main():
ap = argparse.ArgumentParser(
description="Record a meeting and transcribe it with Whisper "
"(PipeWire / ALSA via PortAudio).")
ap.add_argument("-s", "--source", default="default",
help="input to record: 'default' or a substring of a "
"device name (see --list-sources)")
ap.add_argument("-m", "--model", default="small",
help="Whisper model: tiny/base/small/medium/large-v3 "
"(default: small)")
ap.add_argument("-o", "--output", default="meeting",
help="output file stem (default: meeting)")
ap.add_argument("--live", type=int, default=8, metavar="SEC",
help="live transcription interval in seconds; 0 disables "
"(default: 8)")
ap.add_argument("--language", default=None,
help="force language code, e.g. en, de (default: autodetect)")
ap.add_argument("--engine", default="faster-whisper",
choices=["faster-whisper", "whisper-cpp", "whisper-server"],
help="transcription backend (default: faster-whisper). "
"whisper-cpp can use a GPU via Vulkan; whisper-server "
"transcribes on a remote whisper.cpp server")
ap.add_argument("--server-url", default=None,
help="whisper-server base URL, e.g. http://100.103.83.12:8085 "
"(or set MEETREC_SERVER_URL); whisper-server engine only")
ap.add_argument("--device", default="auto",
help="compute device: auto / cpu / cuda (default: auto); "
"faster-whisper only")
ap.add_argument("--compute-type", default="int8",
help="int8 / int8_float16 / float16 / float32 "
"(default: int8); faster-whisper only")
ap.add_argument("--list-sources", action="store_true",
help="list available input devices and exit")
ap.add_argument("--agenda-file", default=None, metavar="PATH",
help="text file with agenda items, one per line — "
"uploaded with the recording for the coverage "
"check (see --library-url)")
ap.add_argument("--diarize-url", default=os.environ.get("MEETREC_DIARIZE_URL"),
help="tinydiarize whisper-server URL for speaker labels "
"(Sprecher 1/2), e.g. http://100.103.83.12:8086 "
"(or set MEETREC_DIARIZE_URL)")
ap.add_argument("--library-url", default=os.environ.get("MEETREC_LIBRARY_URL"),
help="meetrec-server URL; uploads the recording after "
"transcription and triggers summary + agenda check "
"on the server (or set MEETREC_LIBRARY_URL)")
args = ap.parse_args()
if args.list_sources:
list_sources()
return
device = resolve_source(args.source)
if args.engine == "whisper-server":
print(f"using whisper server at "
f"{args.server_url or os.environ.get('MEETREC_SERVER_URL')} "
f"(model lives on the server)")
else:
print(f"loading Whisper model {args.model!r} via {args.engine!r} "
f"(first run downloads it to ~/.cache)...")
try:
model = load_engine(args.model, engine=args.engine,
device=args.device,
compute_type=args.compute_type,
server_url=args.server_url)
except Exception as e:
print(f"error: {e}", file=sys.stderr)
sys.exit(1)
rec = Recorder(device, args.output + ".wav")
rec.start()
print(f"recording to {args.output}.wav at {rec.rate} Hz "
f"(Ctrl+C to stop)", flush=True)
stop = threading.Event()
worker = None
if args.live > 0:
worker = threading.Thread(
target=live_worker, args=(model, rec, args,
args.output + ".live.srt", stop),
daemon=True)
worker.start()
try:
while not stop.is_set():
stop.wait(0.5)
except KeyboardInterrupt:
print("\nstopping...", flush=True)
finally:
stop.set()
rec.stop()
if worker:
worker.join(timeout=60)
agenda = []
if args.agenda_file:
with open(args.agenda_file) as f:
agenda = [ln.strip() for ln in f if ln.strip()]
final_transcribe(model, args.output + ".wav", args.output, args.language,
diarize_url=args.diarize_url, agenda=agenda,
library_url=args.library_url)
if __name__ == "__main__":
main()