Merge master into the Windows port

Three of the four collisions were the same one: master moved the directory
rule into paths.py while this branch was adding a Windows case to the copy in
config.py and the second copy in ggml.py. The case moves to paths.py with the
rest of it, and the directories test moves to tests/test_paths.py where master
put its neighbours.

The fourth is MeetingRecorder, which now starts a process per capture device.
Windows keeps its two lines there: no console window for either process, and
a stop that terminates rather than sending a signal the platform does not have.
This commit is contained in:
2026-08-16 10:14:37 +03:00
26 changed files with 1691 additions and 220 deletions
+278 -67
View File
@@ -2,9 +2,11 @@
Dictation records one source. A meeting records two of them at once, the
microphone and what comes out of the speakers, and for that it goes through
ffmpeg: one process reading both devices and merging them into the two channels
of a single stream, which is the only way the two stay aligned with each other
over an hour.
ffmpeg. PulseAudio hands both devices to a single process, which merges them
into the two channels of one stream and keeps them aligned itself. AVFoundation
cannot be asked the same: two of its sessions inside one process starve each
other, so a Mac captures each device on its own and the two mono streams are
interleaved here as they arrive.
Which programs do the capturing is a property of the machine, not of the code
above: PulseAudio or PipeWire on Linux, AVFoundation through ffmpeg on macOS.
@@ -17,6 +19,7 @@ import collections
import json
import math
import os
import queue
import re
import shutil
import signal
@@ -42,6 +45,21 @@ CHUNK_BYTES = CHUNK_FRAMES * SAMPLE_WIDTH * CHANNELS
CHUNK_LATENCY_MS = round(CHUNK_FRAMES / RATE * 1000)
MIN_FRAMES = int(RATE * 0.25)
# A capture process hands over a block every CHUNK_LATENCY_MS. One that has said
# nothing for this long has stopped rather than fallen behind, and the meeting
# ends and says so instead of sitting on a read that will never return.
STALL_SECONDS = 5.0
# Room for a whole stall of the other stream, so the side still delivering is
# never the one left waiting.
QUEUE_BLOCKS = int(STALL_SECONDS * RATE / CHUNK_FRAMES) + 8
# Exact zeroes are not quiet, they are nothing: a microphone that is really in
# the room has a noise floor. This much of a recording that long means it handed
# nothing over, which is worth saying once the meeting is over and nothing can
# be done about it any more.
QUIET_MIC_SECONDS = 10
QUIET_MIC_SHARE = 0.5
def _interrupt(proc):
"""Ask a recorder process to end.
@@ -80,7 +98,11 @@ class Recorder(QObject):
def start(self, target="", max_seconds=300):
if self.active:
return
cmd = recording_command(target)
try:
cmd = recording_command(target)
except AudioDeviceError as exc:
self.failed.emit(str(exc))
return
if not cmd:
self.failed.emit(t(sound().missing))
return
@@ -203,11 +225,21 @@ def recording_command(target=""):
return sound().record(target)
def meeting_command(mic_target, system_target):
"""One ffmpeg reading both devices and merging them into two channels."""
def meeting_commands(mic_target, system_target):
"""The capture processes that produce one stereo meeting stream.
One of them on PulseAudio, which merges both inputs itself; one per device
on a Mac, because two AVFoundation sessions in a process starve each other.
Which of the two it is stays in the table with everything else the sound
system decides, and MeetingRecorder reads the count rather than the machine.
"""
return sound().meeting(mic_target, system_target)
class AudioDeviceError(RuntimeError):
"""A saved capture device can no longer be selected safely."""
class MeetingRecorder(QObject):
"""Microphone and speaker output into one stereo file: left is you, right is
everyone else.
@@ -221,16 +253,19 @@ class MeetingRecorder(QObject):
levels = pyqtSignal(float, float) # mine, theirs
stopped = pyqtSignal(str, float) # wav path, duration (s)
died = pyqtSignal() # ffmpeg quit on its own
warned = pyqtSignal(str) # recorded, but something was wrong
failed = pyqtSignal(str)
def __init__(self, parent=None):
super().__init__(parent)
self._proc = None
self._procs = []
self._thread = None
self._wav = None
self._log = None
self._logs = []
self._path = ""
self._frames = 0
self._mic_zero_frames = 0
self._split_inputs = False
self._cancelled = False
self._stopping = False
self._lock = threading.Lock()
@@ -252,7 +287,11 @@ class MeetingRecorder(QObject):
"Pick one in Settings → Meeting."))
return
cmd = meeting_command(mic_target, system_target)
try:
commands = meeting_commands(mic_target, system_target)
except AudioDeviceError as exc:
self.failed.emit(str(exc))
return
try:
os.makedirs(os.path.dirname(path), exist_ok=True)
@@ -262,12 +301,18 @@ class MeetingRecorder(QObject):
self._wav.setframerate(RATE)
# ffmpeg keeps talking to stderr for as long as it runs; a pipe
# nobody drains would eventually block it, so it writes to a file.
self._log = tempfile.TemporaryFile()
self._proc = subprocess.Popen(
cmd, stdout=subprocess.PIPE, stderr=self._log, bufsize=0,
creationflags=NO_WINDOW,
)
self._logs = [tempfile.TemporaryFile() for _ in commands]
self._procs = []
for command, log in zip(commands, self._logs):
self._procs.append(subprocess.Popen(
command, stdout=subprocess.PIPE, stderr=log, bufsize=0,
creationflags=NO_WINDOW,
))
except (OSError, wave.Error) as exc:
# One of two capture processes may already be running, and a Mac
# left holding an open AVFoundation session records nothing else.
self._terminate_processes()
self._procs = []
self._close_file()
self._drop_log()
try:
@@ -279,6 +324,8 @@ class MeetingRecorder(QObject):
self._path = path
self._frames = 0
self._mic_zero_frames = 0
self._split_inputs = len(self._procs) == 2
self._cancelled = False
self._stopping = False
self._max_frames = int(max_seconds * RATE)
@@ -286,38 +333,83 @@ class MeetingRecorder(QObject):
self._thread.start()
def _pump(self):
stdout = self._proc.stdout
block = CHUNK_FRAMES * SAMPLE_WIDTH * 2
try:
while True:
chunk = stdout.read(block)
if not chunk:
break
mine, theirs = stereo_levels(chunk)
with self._lock:
if self._wav is None:
break
self._wav.writeframes(chunk)
self._frames += len(chunk) // (SAMPLE_WIDTH * 2)
too_long = self._frames >= self._max_frames
self.levels.emit(mine, theirs)
if too_long:
self._terminate()
break
except (OSError, ValueError, wave.Error):
pass
if self._split_inputs:
self._pump_split()
else:
self._pump_merged()
# Nobody asked it to end: the sound device went away, or ffmpeg fell
# over. An hour into a meeting that has to be said out loud rather than
# discovered afterwards.
if not self._stopping:
self.died.emit()
def _pump_merged(self):
stdout = self._procs[0].stdout
block = CHUNK_FRAMES * SAMPLE_WIDTH * 2
try:
while True:
chunk = stdout.read(block)
if not chunk:
break
if not self._write_chunk(chunk):
break
except (OSError, ValueError, wave.Error):
pass
def _pump_split(self):
# A reader thread per process. Taking turns on the two pipes from one
# thread would let a starved microphone hold up the far side too: its
# blocks would sit unread until the pipe filled and its ffmpeg stopped
# writing, and an hour of meeting would freeze with nothing said. Each
# stream is read as fast as it arrives, and a side that goes quiet for
# STALL_SECONDS ends the recording rather than hanging it.
block = CHUNK_FRAMES * SAMPLE_WIDTH
streams = [queue.Queue(maxsize=QUEUE_BLOCKS) for _ in self._procs]
for proc, blocks in zip(self._procs, streams):
threading.Thread(target=_read_blocks, daemon=True,
args=(proc.stdout, blocks, block)).start()
try:
while True:
mine = _next_block(streams[0])
theirs = _next_block(streams[1])
if not mine or not theirs:
break
frames = min(len(mine), len(theirs)) // SAMPLE_WIDTH
mine = mine[:frames * SAMPLE_WIDTH]
theirs = theirs[:frames * SAMPLE_WIDTH]
self._mic_zero_frames += _zero_samples(mine)
if not self._write_chunk(interleave_mono(mine, theirs)):
break
except (OSError, ValueError, wave.Error):
pass
def _write_chunk(self, chunk):
mine, theirs = stereo_levels(chunk)
with self._lock:
if self._wav is None:
return False
self._wav.writeframes(chunk)
self._frames += len(chunk) // (SAMPLE_WIDTH * 2)
too_long = self._frames >= self._max_frames
self.levels.emit(mine, theirs)
if too_long:
self._terminate()
return False
return True
def _terminate(self):
self._stopping = True
proc = self._proc
if proc and proc.poll() is None:
self._terminate_processes()
def _terminate_processes(self):
running = [proc for proc in self._procs if proc.poll() is None]
for proc in running:
try:
_interrupt(proc)
except OSError:
pass
for proc in running:
try:
proc.wait(timeout=2)
except (subprocess.TimeoutExpired, OSError):
try:
@@ -335,23 +427,26 @@ class MeetingRecorder(QObject):
pass
def _error_tail(self):
if self._log is None:
return ""
try:
self._log.seek(0)
text = self._log.read().decode("utf-8", "replace").strip()
except OSError:
return ""
lines = [line for line in text.splitlines() if line.strip()]
return lines[-1] if lines else ""
tails = []
for log in self._logs:
try:
log.seek(0)
text = log.read().decode("utf-8", "replace").strip()
except OSError:
continue
lines = [line for line in text.splitlines() if line.strip()]
if lines:
tails.append(lines[-1])
return " | ".join(tails)
def _finish_process(self):
self._terminate()
if self._thread:
self._thread.join(timeout=3)
self._thread = None
code = self._proc.poll() if self._proc else 0
self._proc = None
codes = [proc.poll() for proc in self._procs]
code = next((value for value in codes if value), 0)
self._procs = []
self._close_file()
return code
@@ -365,7 +460,7 @@ class MeetingRecorder(QObject):
pass
def stop(self):
if not self._proc:
if not self._procs:
return
# The count is read after the join: the pump thread is still appending
# the last blocks up to the moment it ends.
@@ -390,15 +485,27 @@ class MeetingRecorder(QObject):
)
return
self._drop_log()
# A microphone that handed nothing over costs the left channel, and the
# recording is kept anyway: the right one is everyone else, and an hour
# of them is worth more than an empty channel costs. Only the split
# capture can starve a device this way; one ffmpeg reading both cannot.
if (self._split_inputs and frames >= RATE * QUIET_MIC_SECONDS
and self._mic_zero_frames / frames > QUIET_MIC_SHARE):
self.warned.emit(t(
"The microphone handed over almost nothing ({percent}% of the "
"recording was empty), so your own side of the meeting will be "
"mostly missing. Check the device before the next one.",
percent=round(self._mic_zero_frames / frames * 100),
))
self.stopped.emit(self._path, frames / RATE)
def _drop_log(self):
if self._log is not None:
for log in self._logs:
try:
self._log.close()
log.close()
except OSError:
pass
self._log = None
self._logs = []
def chunk_levels(chunk):
@@ -424,6 +531,62 @@ def stereo_levels(chunk):
return _peak(left), _peak(right)
def interleave_mono(left, right):
"""Two mono-s16 buffers into one stereo-s16 buffer, the shorter one setting
the length."""
left_samples, right_samples = _samples(left), _samples(right)
frames = min(len(left_samples), len(right_samples))
stereo_samples = array.array("h", bytes(frames * 2 * SAMPLE_WIDTH))
stereo_samples[0::2] = left_samples[:frames]
stereo_samples[1::2] = right_samples[:frames]
return stereo_samples.tobytes()
def _read_blocks(stream, blocks, size):
"""One stream's blocks onto its queue, ending with the empty one.
A queue that stays full is the pump having given up on this recording, and
then there is nobody left to hand anything to.
"""
try:
while True:
block = _read_exact(stream, size)
blocks.put(block, timeout=STALL_SECONDS)
if not block:
return
except (OSError, ValueError, queue.Full):
pass
def _next_block(blocks):
"""The next block of a stream, empty once it ends or falls silent."""
try:
return blocks.get(timeout=STALL_SECONDS)
except queue.Empty:
return b""
def _read_exact(stream, size):
"""Read one meter-sized block, tolerating short unbuffered pipe reads."""
out = bytearray()
while len(out) < size:
chunk = stream.read(size - len(out))
if not chunk:
break
out.extend(chunk)
return bytes(out)
def _zero_samples(chunk):
return _samples(chunk).count(0)
def _samples(chunk):
samples = array.array("h")
samples.frombytes(chunk[:len(chunk) - len(chunk) % SAMPLE_WIDTH])
return samples
def _peak(samples):
if not samples:
return 0.0
@@ -432,8 +595,9 @@ def _peak(samples):
# --- the sound system, one group per machine -------------------------------
# Both meeting commands merge the same way: each input down to mono at our own
# rate, then the two of them into the left and right of one stream.
# How the one PulseAudio process merges: each input down to mono at our own
# rate, then the two of them into the left and right of one stream. A Mac does
# the first half per process and the second half itself, in interleave_mono().
MERGE_FILTER = (
f"[0:a]aresample={RATE}:async=1,aformat=sample_fmts=s16:channel_layouts=mono[m];"
f"[1:a]aresample={RATE}:async=1,aformat=sample_fmts=s16:channel_layouts=mono[s];"
@@ -497,13 +661,14 @@ def _pw_record_raw_option():
def _pulse_meeting(mic_target, system_target):
return [
"""One process for both devices: PulseAudio keeps them aligned itself."""
return [[
"ffmpeg", "-hide_banner", "-nostdin", "-loglevel", "error",
"-f", "pulse", "-thread_queue_size", "4096", "-i", mic_target or "default",
"-f", "pulse", "-thread_queue_size", "4096", "-i", system_target,
"-filter_complex", MERGE_FILTER, "-map", "[out]",
"-f", "s16le", "-ar", str(RATE), "-",
]
]]
def _pactl_sources():
@@ -561,6 +726,7 @@ LOOPBACK_DEVICES = ("blackhole", "loopback", "soundflower")
def _avfoundation_record(target):
if not shutil.which("ffmpeg"):
return []
target = _resolve_avfoundation_target(target)
return [
"ffmpeg", "-hide_banner", "-nostdin", "-loglevel", "error",
# AVFoundation names an input "video:audio", so the empty half in front
@@ -571,14 +737,26 @@ def _avfoundation_record(target):
def _avfoundation_meeting(mic_target, system_target):
"""A process per device, both names read off the same device listing.
Asking ffmpeg what is plugged in costs a process of its own, and a listing
taken twice could renumber in between: the two targets have to be resolved
against the same one to name the same machine the user picked from.
"""
inputs = _avfoundation_inputs()
return [_avfoundation_meeting_capture(mic_target, inputs),
_avfoundation_meeting_capture(system_target, inputs)]
def _avfoundation_meeting_capture(target, inputs=None):
target = _resolve_avfoundation_target(target, inputs)
return [
"ffmpeg", "-hide_banner", "-nostdin", "-loglevel", "error",
"-thread_queue_size", "4096",
"-f", "avfoundation", "-i", f":{mic_target or 'default'}",
"-thread_queue_size", "4096",
"-f", "avfoundation", "-i", f":{system_target}",
"-filter_complex", MERGE_FILTER, "-map", "[out]",
"-f", "s16le", "-ar", str(RATE), "-",
"-f", "avfoundation", "-i", f":{target or 'default'}",
"-af", (f"aresample={RATE}:async=1:first_pts=0,"
"aformat=sample_fmts=s16:channel_layouts=mono"),
"-f", "s16le", "-ar", str(RATE), "-ac", "1", "-",
]
@@ -615,10 +793,42 @@ def _avfoundation_inputs():
return devices
def _avfoundation_named_inputs():
"""Stable settings values: the name is saved, never the moving index."""
return [(description, description)
for _index, description in _avfoundation_inputs()]
def _resolve_avfoundation_target(target, inputs=None):
"""Resolve a stored device name to its current, positional ffmpeg index."""
if not target or target == "default":
return "default"
if str(target).isdigit():
raise AudioDeviceError(t(
"The saved macOS audio device uses an old numeric index. Open "
"Settings and select the device again before recording."
))
if inputs is None:
inputs = _avfoundation_inputs()
matches = [index for index, description in inputs
if description == target]
if not matches:
raise AudioDeviceError(t(
"The saved macOS audio device is no longer connected: {device}. "
"Open Settings and select another device.", device=target,
))
if len(matches) > 1:
raise AudioDeviceError(t(
"More than one macOS audio device is named {device}. Disconnect the "
"duplicate or choose a different device.", device=target,
))
return matches[0]
def _avfoundation_default_output():
for name, description in _avfoundation_inputs():
for _index, description in _avfoundation_inputs():
if any(word in description.lower() for word in LOOPBACK_DEVICES):
return name
return description
return ""
@@ -689,9 +899,10 @@ def _dshow_no_default_output():
Sound = collections.namedtuple(
"Sound",
# How to capture one source and two at once, the two device lists, which
# device a meeting records the far side from, and what to say when the
# programs for any of it are not installed.
# How to capture one source and how to capture two at once, that one as the
# list of processes it takes, the two device lists, which device a meeting
# records the far side from, and what to say when the programs for any of
# it are not installed.
"record meeting inputs outputs default_output missing",
)
@@ -707,11 +918,11 @@ PULSE = Sound(
COREAUDIO = Sound(
record=_avfoundation_record,
meeting=_avfoundation_meeting,
inputs=_avfoundation_inputs,
inputs=_avfoundation_named_inputs,
# Every macOS capture device is offered as the far side of a meeting, the
# loopback driver among them: there is no way to tell them apart, and an
# empty list would leave nothing to pick.
outputs=_avfoundation_inputs,
outputs=_avfoundation_named_inputs,
default_output=_avfoundation_default_output,
missing="ffmpeg not found. Install it with: brew install ffmpeg",
)