Merge pull request #26 from benfirad/codex/macos-audio-device-safety

Harden macOS meeting audio capture
This commit is contained in:
Yusuf İpek
2026-08-16 10:20:23 +03:00
committed by yusufipek
4 changed files with 544 additions and 93 deletions
+277 -66
View File
@@ -2,9 +2,11 @@
Dictation records one source. A meeting records two of them at once, the Dictation records one source. A meeting records two of them at once, the
microphone and what comes out of the speakers, and for that it goes through microphone and what comes out of the speakers, and for that it goes through
ffmpeg: one process reading both devices and merging them into the two channels ffmpeg. PulseAudio hands both devices to a single process, which merges them
of a single stream, which is the only way the two stay aligned with each other into the two channels of one stream and keeps them aligned itself. AVFoundation
over an hour. cannot be asked the same: two of its sessions inside one process starve each
other, so a Mac captures each device on its own and the two mono streams are
interleaved here as they arrive.
Which programs do the capturing is a property of the machine, not of the code Which programs do the capturing is a property of the machine, not of the code
above: PulseAudio or PipeWire on Linux, AVFoundation through ffmpeg on macOS. above: PulseAudio or PipeWire on Linux, AVFoundation through ffmpeg on macOS.
@@ -17,6 +19,7 @@ import collections
import json import json
import math import math
import os import os
import queue
import re import re
import shutil import shutil
import signal import signal
@@ -38,6 +41,21 @@ CHUNK_BYTES = CHUNK_FRAMES * SAMPLE_WIDTH * CHANNELS
CHUNK_LATENCY_MS = round(CHUNK_FRAMES / RATE * 1000) CHUNK_LATENCY_MS = round(CHUNK_FRAMES / RATE * 1000)
MIN_FRAMES = int(RATE * 0.25) MIN_FRAMES = int(RATE * 0.25)
# A capture process hands over a block every CHUNK_LATENCY_MS. One that has said
# nothing for this long has stopped rather than fallen behind, and the meeting
# ends and says so instead of sitting on a read that will never return.
STALL_SECONDS = 5.0
# Room for a whole stall of the other stream, so the side still delivering is
# never the one left waiting.
QUEUE_BLOCKS = int(STALL_SECONDS * RATE / CHUNK_FRAMES) + 8
# Exact zeroes are not quiet, they are nothing: a microphone that is really in
# the room has a noise floor. This much of a recording that long means it handed
# nothing over, which is worth saying once the meeting is over and nothing can
# be done about it any more.
QUIET_MIC_SECONDS = 10
QUIET_MIC_SHARE = 0.5
class Recorder(QObject): class Recorder(QObject):
"""Runs the available sound-server recorder and reads raw PCM from stdout.""" """Runs the available sound-server recorder and reads raw PCM from stdout."""
@@ -63,7 +81,11 @@ class Recorder(QObject):
def start(self, target="", max_seconds=300): def start(self, target="", max_seconds=300):
if self.active: if self.active:
return return
cmd = recording_command(target) try:
cmd = recording_command(target)
except AudioDeviceError as exc:
self.failed.emit(str(exc))
return
if not cmd: if not cmd:
self.failed.emit(t(sound().missing)) self.failed.emit(t(sound().missing))
return return
@@ -185,11 +207,21 @@ def recording_command(target=""):
return sound().record(target) return sound().record(target)
def meeting_command(mic_target, system_target): def meeting_commands(mic_target, system_target):
"""One ffmpeg reading both devices and merging them into two channels.""" """The capture processes that produce one stereo meeting stream.
One of them on PulseAudio, which merges both inputs itself; one per device
on a Mac, because two AVFoundation sessions in a process starve each other.
Which of the two it is stays in the table with everything else the sound
system decides, and MeetingRecorder reads the count rather than the machine.
"""
return sound().meeting(mic_target, system_target) return sound().meeting(mic_target, system_target)
class AudioDeviceError(RuntimeError):
"""A saved capture device can no longer be selected safely."""
class MeetingRecorder(QObject): class MeetingRecorder(QObject):
"""Microphone and speaker output into one stereo file: left is you, right is """Microphone and speaker output into one stereo file: left is you, right is
everyone else. everyone else.
@@ -203,16 +235,19 @@ class MeetingRecorder(QObject):
levels = pyqtSignal(float, float) # mine, theirs levels = pyqtSignal(float, float) # mine, theirs
stopped = pyqtSignal(str, float) # wav path, duration (s) stopped = pyqtSignal(str, float) # wav path, duration (s)
died = pyqtSignal() # ffmpeg quit on its own died = pyqtSignal() # ffmpeg quit on its own
warned = pyqtSignal(str) # recorded, but something was wrong
failed = pyqtSignal(str) failed = pyqtSignal(str)
def __init__(self, parent=None): def __init__(self, parent=None):
super().__init__(parent) super().__init__(parent)
self._proc = None self._procs = []
self._thread = None self._thread = None
self._wav = None self._wav = None
self._log = None self._logs = []
self._path = "" self._path = ""
self._frames = 0 self._frames = 0
self._mic_zero_frames = 0
self._split_inputs = False
self._cancelled = False self._cancelled = False
self._stopping = False self._stopping = False
self._lock = threading.Lock() self._lock = threading.Lock()
@@ -234,7 +269,11 @@ class MeetingRecorder(QObject):
"Pick one in Settings → Meeting.")) "Pick one in Settings → Meeting."))
return return
cmd = meeting_command(mic_target, system_target) try:
commands = meeting_commands(mic_target, system_target)
except AudioDeviceError as exc:
self.failed.emit(str(exc))
return
try: try:
os.makedirs(os.path.dirname(path), exist_ok=True) os.makedirs(os.path.dirname(path), exist_ok=True)
@@ -244,11 +283,17 @@ class MeetingRecorder(QObject):
self._wav.setframerate(RATE) self._wav.setframerate(RATE)
# ffmpeg keeps talking to stderr for as long as it runs; a pipe # ffmpeg keeps talking to stderr for as long as it runs; a pipe
# nobody drains would eventually block it, so it writes to a file. # nobody drains would eventually block it, so it writes to a file.
self._log = tempfile.TemporaryFile() self._logs = [tempfile.TemporaryFile() for _ in commands]
self._proc = subprocess.Popen( self._procs = []
cmd, stdout=subprocess.PIPE, stderr=self._log, bufsize=0 for command, log in zip(commands, self._logs):
) self._procs.append(subprocess.Popen(
command, stdout=subprocess.PIPE, stderr=log, bufsize=0
))
except (OSError, wave.Error) as exc: except (OSError, wave.Error) as exc:
# One of two capture processes may already be running, and a Mac
# left holding an open AVFoundation session records nothing else.
self._terminate_processes()
self._procs = []
self._close_file() self._close_file()
self._drop_log() self._drop_log()
try: try:
@@ -260,6 +305,8 @@ class MeetingRecorder(QObject):
self._path = path self._path = path
self._frames = 0 self._frames = 0
self._mic_zero_frames = 0
self._split_inputs = len(self._procs) == 2
self._cancelled = False self._cancelled = False
self._stopping = False self._stopping = False
self._max_frames = int(max_seconds * RATE) self._max_frames = int(max_seconds * RATE)
@@ -267,38 +314,83 @@ class MeetingRecorder(QObject):
self._thread.start() self._thread.start()
def _pump(self): def _pump(self):
stdout = self._proc.stdout if self._split_inputs:
block = CHUNK_FRAMES * SAMPLE_WIDTH * 2 self._pump_split()
try: else:
while True: self._pump_merged()
chunk = stdout.read(block)
if not chunk:
break
mine, theirs = stereo_levels(chunk)
with self._lock:
if self._wav is None:
break
self._wav.writeframes(chunk)
self._frames += len(chunk) // (SAMPLE_WIDTH * 2)
too_long = self._frames >= self._max_frames
self.levels.emit(mine, theirs)
if too_long:
self._terminate()
break
except (OSError, ValueError, wave.Error):
pass
# Nobody asked it to end: the sound device went away, or ffmpeg fell # Nobody asked it to end: the sound device went away, or ffmpeg fell
# over. An hour into a meeting that has to be said out loud rather than # over. An hour into a meeting that has to be said out loud rather than
# discovered afterwards. # discovered afterwards.
if not self._stopping: if not self._stopping:
self.died.emit() self.died.emit()
def _pump_merged(self):
stdout = self._procs[0].stdout
block = CHUNK_FRAMES * SAMPLE_WIDTH * 2
try:
while True:
chunk = stdout.read(block)
if not chunk:
break
if not self._write_chunk(chunk):
break
except (OSError, ValueError, wave.Error):
pass
def _pump_split(self):
# A reader thread per process. Taking turns on the two pipes from one
# thread would let a starved microphone hold up the far side too: its
# blocks would sit unread until the pipe filled and its ffmpeg stopped
# writing, and an hour of meeting would freeze with nothing said. Each
# stream is read as fast as it arrives, and a side that goes quiet for
# STALL_SECONDS ends the recording rather than hanging it.
block = CHUNK_FRAMES * SAMPLE_WIDTH
streams = [queue.Queue(maxsize=QUEUE_BLOCKS) for _ in self._procs]
for proc, blocks in zip(self._procs, streams):
threading.Thread(target=_read_blocks, daemon=True,
args=(proc.stdout, blocks, block)).start()
try:
while True:
mine = _next_block(streams[0])
theirs = _next_block(streams[1])
if not mine or not theirs:
break
frames = min(len(mine), len(theirs)) // SAMPLE_WIDTH
mine = mine[:frames * SAMPLE_WIDTH]
theirs = theirs[:frames * SAMPLE_WIDTH]
self._mic_zero_frames += _zero_samples(mine)
if not self._write_chunk(interleave_mono(mine, theirs)):
break
except (OSError, ValueError, wave.Error):
pass
def _write_chunk(self, chunk):
mine, theirs = stereo_levels(chunk)
with self._lock:
if self._wav is None:
return False
self._wav.writeframes(chunk)
self._frames += len(chunk) // (SAMPLE_WIDTH * 2)
too_long = self._frames >= self._max_frames
self.levels.emit(mine, theirs)
if too_long:
self._terminate()
return False
return True
def _terminate(self): def _terminate(self):
self._stopping = True self._stopping = True
proc = self._proc self._terminate_processes()
if proc and proc.poll() is None:
def _terminate_processes(self):
running = [proc for proc in self._procs if proc.poll() is None]
for proc in running:
try: try:
proc.send_signal(signal.SIGINT) proc.send_signal(signal.SIGINT)
except OSError:
pass
for proc in running:
try:
proc.wait(timeout=2) proc.wait(timeout=2)
except (subprocess.TimeoutExpired, OSError): except (subprocess.TimeoutExpired, OSError):
try: try:
@@ -316,23 +408,26 @@ class MeetingRecorder(QObject):
pass pass
def _error_tail(self): def _error_tail(self):
if self._log is None: tails = []
return "" for log in self._logs:
try: try:
self._log.seek(0) log.seek(0)
text = self._log.read().decode("utf-8", "replace").strip() text = log.read().decode("utf-8", "replace").strip()
except OSError: except OSError:
return "" continue
lines = [line for line in text.splitlines() if line.strip()] lines = [line for line in text.splitlines() if line.strip()]
return lines[-1] if lines else "" if lines:
tails.append(lines[-1])
return " | ".join(tails)
def _finish_process(self): def _finish_process(self):
self._terminate() self._terminate()
if self._thread: if self._thread:
self._thread.join(timeout=3) self._thread.join(timeout=3)
self._thread = None self._thread = None
code = self._proc.poll() if self._proc else 0 codes = [proc.poll() for proc in self._procs]
self._proc = None code = next((value for value in codes if value), 0)
self._procs = []
self._close_file() self._close_file()
return code return code
@@ -346,7 +441,7 @@ class MeetingRecorder(QObject):
pass pass
def stop(self): def stop(self):
if not self._proc: if not self._procs:
return return
# The count is read after the join: the pump thread is still appending # The count is read after the join: the pump thread is still appending
# the last blocks up to the moment it ends. # the last blocks up to the moment it ends.
@@ -371,15 +466,27 @@ class MeetingRecorder(QObject):
) )
return return
self._drop_log() self._drop_log()
# A microphone that handed nothing over costs the left channel, and the
# recording is kept anyway: the right one is everyone else, and an hour
# of them is worth more than an empty channel costs. Only the split
# capture can starve a device this way; one ffmpeg reading both cannot.
if (self._split_inputs and frames >= RATE * QUIET_MIC_SECONDS
and self._mic_zero_frames / frames > QUIET_MIC_SHARE):
self.warned.emit(t(
"The microphone handed over almost nothing ({percent}% of the "
"recording was empty), so your own side of the meeting will be "
"mostly missing. Check the device before the next one.",
percent=round(self._mic_zero_frames / frames * 100),
))
self.stopped.emit(self._path, frames / RATE) self.stopped.emit(self._path, frames / RATE)
def _drop_log(self): def _drop_log(self):
if self._log is not None: for log in self._logs:
try: try:
self._log.close() log.close()
except OSError: except OSError:
pass pass
self._log = None self._logs = []
def chunk_levels(chunk): def chunk_levels(chunk):
@@ -405,6 +512,62 @@ def stereo_levels(chunk):
return _peak(left), _peak(right) return _peak(left), _peak(right)
def interleave_mono(left, right):
"""Two mono-s16 buffers into one stereo-s16 buffer, the shorter one setting
the length."""
left_samples, right_samples = _samples(left), _samples(right)
frames = min(len(left_samples), len(right_samples))
stereo_samples = array.array("h", bytes(frames * 2 * SAMPLE_WIDTH))
stereo_samples[0::2] = left_samples[:frames]
stereo_samples[1::2] = right_samples[:frames]
return stereo_samples.tobytes()
def _read_blocks(stream, blocks, size):
"""One stream's blocks onto its queue, ending with the empty one.
A queue that stays full is the pump having given up on this recording, and
then there is nobody left to hand anything to.
"""
try:
while True:
block = _read_exact(stream, size)
blocks.put(block, timeout=STALL_SECONDS)
if not block:
return
except (OSError, ValueError, queue.Full):
pass
def _next_block(blocks):
"""The next block of a stream, empty once it ends or falls silent."""
try:
return blocks.get(timeout=STALL_SECONDS)
except queue.Empty:
return b""
def _read_exact(stream, size):
"""Read one meter-sized block, tolerating short unbuffered pipe reads."""
out = bytearray()
while len(out) < size:
chunk = stream.read(size - len(out))
if not chunk:
break
out.extend(chunk)
return bytes(out)
def _zero_samples(chunk):
return _samples(chunk).count(0)
def _samples(chunk):
samples = array.array("h")
samples.frombytes(chunk[:len(chunk) - len(chunk) % SAMPLE_WIDTH])
return samples
def _peak(samples): def _peak(samples):
if not samples: if not samples:
return 0.0 return 0.0
@@ -413,8 +576,9 @@ def _peak(samples):
# --- the sound system, one group per machine ------------------------------- # --- the sound system, one group per machine -------------------------------
# Both meeting commands merge the same way: each input down to mono at our own # How the one PulseAudio process merges: each input down to mono at our own
# rate, then the two of them into the left and right of one stream. # rate, then the two of them into the left and right of one stream. A Mac does
# the first half per process and the second half itself, in interleave_mono().
MERGE_FILTER = ( MERGE_FILTER = (
f"[0:a]aresample={RATE}:async=1,aformat=sample_fmts=s16:channel_layouts=mono[m];" f"[0:a]aresample={RATE}:async=1,aformat=sample_fmts=s16:channel_layouts=mono[m];"
f"[1:a]aresample={RATE}:async=1,aformat=sample_fmts=s16:channel_layouts=mono[s];" f"[1:a]aresample={RATE}:async=1,aformat=sample_fmts=s16:channel_layouts=mono[s];"
@@ -478,13 +642,14 @@ def _pw_record_raw_option():
def _pulse_meeting(mic_target, system_target): def _pulse_meeting(mic_target, system_target):
return [ """One process for both devices: PulseAudio keeps them aligned itself."""
return [[
"ffmpeg", "-hide_banner", "-nostdin", "-loglevel", "error", "ffmpeg", "-hide_banner", "-nostdin", "-loglevel", "error",
"-f", "pulse", "-thread_queue_size", "4096", "-i", mic_target or "default", "-f", "pulse", "-thread_queue_size", "4096", "-i", mic_target or "default",
"-f", "pulse", "-thread_queue_size", "4096", "-i", system_target, "-f", "pulse", "-thread_queue_size", "4096", "-i", system_target,
"-filter_complex", MERGE_FILTER, "-map", "[out]", "-filter_complex", MERGE_FILTER, "-map", "[out]",
"-f", "s16le", "-ar", str(RATE), "-", "-f", "s16le", "-ar", str(RATE), "-",
] ]]
def _pactl_sources(): def _pactl_sources():
@@ -542,6 +707,7 @@ LOOPBACK_DEVICES = ("blackhole", "loopback", "soundflower")
def _avfoundation_record(target): def _avfoundation_record(target):
if not shutil.which("ffmpeg"): if not shutil.which("ffmpeg"):
return [] return []
target = _resolve_avfoundation_target(target)
return [ return [
"ffmpeg", "-hide_banner", "-nostdin", "-loglevel", "error", "ffmpeg", "-hide_banner", "-nostdin", "-loglevel", "error",
# AVFoundation names an input "video:audio", so the empty half in front # AVFoundation names an input "video:audio", so the empty half in front
@@ -552,14 +718,26 @@ def _avfoundation_record(target):
def _avfoundation_meeting(mic_target, system_target): def _avfoundation_meeting(mic_target, system_target):
"""A process per device, both names read off the same device listing.
Asking ffmpeg what is plugged in costs a process of its own, and a listing
taken twice could renumber in between: the two targets have to be resolved
against the same one to name the same machine the user picked from.
"""
inputs = _avfoundation_inputs()
return [_avfoundation_meeting_capture(mic_target, inputs),
_avfoundation_meeting_capture(system_target, inputs)]
def _avfoundation_meeting_capture(target, inputs=None):
target = _resolve_avfoundation_target(target, inputs)
return [ return [
"ffmpeg", "-hide_banner", "-nostdin", "-loglevel", "error", "ffmpeg", "-hide_banner", "-nostdin", "-loglevel", "error",
"-thread_queue_size", "4096", "-thread_queue_size", "4096",
"-f", "avfoundation", "-i", f":{mic_target or 'default'}", "-f", "avfoundation", "-i", f":{target or 'default'}",
"-thread_queue_size", "4096", "-af", (f"aresample={RATE}:async=1:first_pts=0,"
"-f", "avfoundation", "-i", f":{system_target}", "aformat=sample_fmts=s16:channel_layouts=mono"),
"-filter_complex", MERGE_FILTER, "-map", "[out]", "-f", "s16le", "-ar", str(RATE), "-ac", "1", "-",
"-f", "s16le", "-ar", str(RATE), "-",
] ]
@@ -596,18 +774,51 @@ def _avfoundation_inputs():
return devices return devices
def _avfoundation_named_inputs():
"""Stable settings values: the name is saved, never the moving index."""
return [(description, description)
for _index, description in _avfoundation_inputs()]
def _resolve_avfoundation_target(target, inputs=None):
"""Resolve a stored device name to its current, positional ffmpeg index."""
if not target or target == "default":
return "default"
if str(target).isdigit():
raise AudioDeviceError(t(
"The saved macOS audio device uses an old numeric index. Open "
"Settings and select the device again before recording."
))
if inputs is None:
inputs = _avfoundation_inputs()
matches = [index for index, description in inputs
if description == target]
if not matches:
raise AudioDeviceError(t(
"The saved macOS audio device is no longer connected: {device}. "
"Open Settings and select another device.", device=target,
))
if len(matches) > 1:
raise AudioDeviceError(t(
"More than one macOS audio device is named {device}. Disconnect the "
"duplicate or choose a different device.", device=target,
))
return matches[0]
def _avfoundation_default_output(): def _avfoundation_default_output():
for name, description in _avfoundation_inputs(): for _index, description in _avfoundation_inputs():
if any(word in description.lower() for word in LOOPBACK_DEVICES): if any(word in description.lower() for word in LOOPBACK_DEVICES):
return name return description
return "" return ""
Sound = collections.namedtuple( Sound = collections.namedtuple(
"Sound", "Sound",
# How to capture one source and two at once, the two device lists, which # How to capture one source and how to capture two at once, that one as the
# device a meeting records the far side from, and what to say when the # list of processes it takes, the two device lists, which device a meeting
# programs for any of it are not installed. # records the far side from, and what to say when the programs for any of
# it are not installed.
"record meeting inputs outputs default_output missing", "record meeting inputs outputs default_output missing",
) )
@@ -623,11 +834,11 @@ PULSE = Sound(
COREAUDIO = Sound( COREAUDIO = Sound(
record=_avfoundation_record, record=_avfoundation_record,
meeting=_avfoundation_meeting, meeting=_avfoundation_meeting,
inputs=_avfoundation_inputs, inputs=_avfoundation_named_inputs,
# Every macOS capture device is offered as the far side of a meeting, the # Every macOS capture device is offered as the far side of a meeting, the
# loopback driver among them: there is no way to tell them apart, and an # loopback driver among them: there is no way to tell them apart, and an
# empty list would leave nothing to pick. # empty list would leave nothing to pick.
outputs=_avfoundation_inputs, outputs=_avfoundation_named_inputs,
default_output=_avfoundation_default_output, default_output=_avfoundation_default_output,
missing="ffmpeg not found. Install it with: brew install ffmpeg", missing="ffmpeg not found. Install it with: brew install ffmpeg",
) )
+6
View File
@@ -126,6 +126,7 @@ class Dikte:
self.meeting_recorder.levels.connect(self._on_meeting_levels) self.meeting_recorder.levels.connect(self._on_meeting_levels)
self.meeting_recorder.stopped.connect(self._on_meeting_recorded) self.meeting_recorder.stopped.connect(self._on_meeting_recorded)
self.meeting_recorder.died.connect(self._on_meeting_died) self.meeting_recorder.died.connect(self._on_meeting_died)
self.meeting_recorder.warned.connect(self._on_meeting_warning)
self.meeting_recorder.failed.connect(self._on_meeting_error) self.meeting_recorder.failed.connect(self._on_meeting_error)
self.meetings.progress.connect(self._on_meeting_progress) self.meetings.progress.connect(self._on_meeting_progress)
self.meetings.finished.connect(self._on_meeting_finished) self.meetings.finished.connect(self._on_meeting_finished)
@@ -735,6 +736,11 @@ class Dikte:
self._settle(MEETING, {"ok": False, "error": message}) self._settle(MEETING, {"ok": False, "error": message})
self._on_error(message) self._on_error(message)
def _on_meeting_warning(self, message):
"""It was recorded and it is being written up, but read this first."""
self.tray.showMessage("Dikte", message,
QSystemTrayIcon.MessageIcon.Warning, 12000)
def _on_meeting_died(self): def _on_meeting_died(self):
if self.meeting_state != M_RECORDING: if self.meeting_state != M_RECORDING:
return return
+18
View File
@@ -561,6 +561,24 @@ TR = {
"Hangi ses çıkışının kaydedileceği anlaşılamadı. Ayarlar → Toplantı " "Hangi ses çıkışının kaydedileceği anlaşılamadı. Ayarlar → Toplantı "
"sekmesinden seç.", "sekmesinden seç.",
"Nothing was recorded: {error}": "Hiçbir şey kaydedilmedi: {error}", "Nothing was recorded: {error}": "Hiçbir şey kaydedilmedi: {error}",
"The saved macOS audio device uses an old numeric index. Open Settings and "
"select the device again before recording.":
"Kayıtlı macOS ses aygıtı eski bir sayısal indeks kullanıyor. Kayıttan "
"önce Ayarlar'ı açıp aygıtı yeniden seç.",
"The saved macOS audio device is no longer connected: {device}. Open "
"Settings and select another device.":
"Kayıtlı macOS ses aygıtı artık bağlı değil: {device}. Ayarlar'ı açıp "
"başka bir aygıt seç.",
"More than one macOS audio device is named {device}. Disconnect the "
"duplicate or choose a different device.":
"Birden fazla macOS ses aygıtının adı {device}. Aynı adlı aygıtlardan "
"birini çıkar ya da başka bir aygıt seç.",
"The microphone handed over almost nothing ({percent}% of the recording was "
"empty), so your own side of the meeting will be mostly missing. Check the "
"device before the next one.":
"Mikrofon neredeyse hiçbir şey iletmedi (kaydın %{percent} kadarı boştu), "
"toplantının senin tarafın büyük ölçüde eksik olacak. Bir sonrakinden "
"önce aygıtı kontrol et.",
"Transcribing {side}: {index}/{count}": "Transcribing {side}: {index}/{count}":
"{side} yazıya çevriliyor: {index}/{count}", "{side} yazıya çevriliyor: {index}/{count}",
"you": "sen", "you": "sen",
+243 -27
View File
@@ -16,6 +16,7 @@ import json
import os import os
import subprocess import subprocess
import sys import sys
import threading
import unittest import unittest
import wave import wave
from unittest import mock from unittest import mock
@@ -100,6 +101,22 @@ class StereoLevels(unittest.TestCase):
self.assertAlmostEqual(left, 0.25, places=3) self.assertAlmostEqual(left, 0.25, places=3)
self.assertAlmostEqual(right, 0.5, places=3) self.assertAlmostEqual(right, 0.5, places=3)
def test_two_mono_streams_are_interleaved_left_then_right(self):
self.assertEqual(
list(array.array("h", audio.interleave_mono(
pcm([100, 200, 300]), pcm([-100, -200, -300])
))),
[100, -100, 200, -200, 300, -300],
)
def test_interleaving_stops_at_the_shorter_stream(self):
self.assertEqual(
list(array.array("h", audio.interleave_mono(
pcm([100, 200]), pcm([-100])
))),
[100, -100],
)
class WriteWav(DikteTest): class WriteWav(DikteTest):
def test_the_header_says_what_the_recorder_captured(self): def test_the_header_says_what_the_recorder_captured(self):
@@ -235,6 +252,34 @@ class FakeProcess:
self._alive = False self._alive = False
class StalledProcess(FakeProcess):
"""A capture that hands over a buffer and then stops answering at all.
Not the same thing as one that ends: the device is still there and the pipe
is still open, and a read of it never comes back.
"""
def __init__(self, data):
super().__init__(data)
self.stdout = _StalledStream(data)
class _StalledStream:
def __init__(self, data):
self._data = io.BytesIO(data)
self._released = threading.Event()
def read(self, size):
chunk = self._data.read(size)
if chunk:
return chunk
self._released.wait()
return b""
def release(self):
self._released.set()
class RecordingCommand(OnLinux, DikteTest): class RecordingCommand(OnLinux, DikteTest):
"""Which program captures the microphone, and how it is asked to.""" """Which program captures the microphone, and how it is asked to."""
@@ -465,42 +510,178 @@ class RecorderChain(OnLinux, DikteTest):
self.assertFalse(recorder.active) self.assertFalse(recorder.active)
class MeetingCommand(unittest.TestCase): class MeetingCommands(unittest.TestCase):
"""One process reading both devices, because two would drift apart.""" """Pulse can share a process; AVFoundation sessions cannot."""
def command(self, platform, mic="", system="them"): def commands(self, platform, mic="", system="them"):
with mock.patch.object(sys, "platform", platform): with mock.patch.object(sys, "platform", platform), \
return audio.meeting_command(mic, system) mock.patch.object(audio, "_avfoundation_inputs", return_value=[]), \
mock.patch.object(
audio, "_resolve_avfoundation_target",
side_effect=lambda target, inputs=None: target or "default"):
return audio.meeting_commands(mic, system)
def test_linux_reads_both_through_pulse(self): def test_linux_reads_both_through_pulse(self):
cmd = self.command("linux", mic="mine") commands = self.commands("linux", mic="mine")
self.assertEqual(len(commands), 1)
cmd = commands[0]
self.assertEqual(cmd.count("pulse"), 2) self.assertEqual(cmd.count("pulse"), 2)
self.assertEqual(cmd[cmd.index("mine") - 1], "-i") self.assertEqual(cmd[cmd.index("mine") - 1], "-i")
self.assertEqual(cmd[cmd.index("them") - 1], "-i") self.assertEqual(cmd[cmd.index("them") - 1], "-i")
def test_a_mac_reads_both_through_avfoundation(self): def test_a_mac_gives_each_avfoundation_device_its_own_process(self):
cmd = self.command("darwin", mic="1") commands = self.commands("darwin", mic="mine")
self.assertEqual(cmd.count("avfoundation"), 2) self.assertEqual(len(commands), 2)
self.assertIn(":1", cmd) self.assertTrue(all(command.count("avfoundation") == 1
self.assertIn(":them", cmd) for command in commands))
self.assertIn(":mine", commands[0])
self.assertIn(":them", commands[1])
def test_no_microphone_named_means_the_default_one(self): def test_no_microphone_named_means_the_default_one(self):
self.assertIn("default", self.command("linux")) self.assertIn("default", self.commands("linux")[0])
self.assertIn(":default", self.command("darwin")) self.assertIn(":default", self.commands("darwin")[0])
def test_both_merge_the_two_into_one_stereo_stream(self): def test_pulse_merges_the_two_into_one_stereo_stream(self):
for platform in ("linux", "darwin"): cmd = self.commands("linux")[0]
with self.subTest(platform=platform): self.assertIn(audio.MERGE_FILTER, cmd)
cmd = self.command(platform) self.assertEqual(cmd[cmd.index("-map") + 1], "[out]")
self.assertIn(audio.MERGE_FILTER, cmd) self.assertEqual(cmd[cmd.index("-f", cmd.index("-map")) + 1], "s16le")
self.assertEqual(cmd[cmd.index("-map") + 1], "[out]")
self.assertEqual(cmd[cmd.index("-f", cmd.index("-map")) + 1], "s16le") def test_each_mac_process_produces_clock_corrected_mono_pcm(self):
for cmd in self.commands("darwin"):
self.assertIn("first_pts=0", cmd[cmd.index("-af") + 1])
self.assertEqual(cmd[cmd.index("-ac") + 1], "1")
self.assertEqual(cmd[-2:], ["1", "-"])
def test_neither_lets_ffmpeg_read_the_terminal(self): def test_neither_lets_ffmpeg_read_the_terminal(self):
"""It shares stdin with Dikte, and would eat a keypress meant for it.""" """It shares stdin with Dikte, and would eat a keypress meant for it."""
for platform in ("linux", "darwin"): for platform in ("linux", "darwin"):
with self.subTest(platform=platform): with self.subTest(platform=platform):
self.assertIn("-nostdin", self.command(platform)) for command in self.commands(platform):
self.assertIn("-nostdin", command)
def test_both_mac_devices_are_read_off_one_listing(self):
"""Asking twice costs an ffmpeg run, and the second answer could have
renumbered between the two."""
with mock.patch.object(sys, "platform", "darwin"), \
mock.patch.object(audio, "_avfoundation_inputs",
return_value=[("0", "mine"),
("1", "them")]) as inputs:
audio.meeting_commands("mine", "them")
inputs.assert_called_once_with()
class MacMeetingRecorder(OnMacOS, DikteTest):
# Whichever way the machine has them ordered, a name is what is saved and
# the index it happens to hold now is what ffmpeg is given.
DEVICES = [("0", "External Headset"), ("1", "BlackHole 2ch"),
("2", "MacBook Pro Microphone")]
def devices(self):
return mock.patch.object(audio, "_avfoundation_inputs",
return_value=self.DEVICES)
def record(self, mine, theirs):
path = str(self.path("meeting.wav"))
recorder = audio.MeetingRecorder()
stopped, failed, warnings = [], [], []
recorder.stopped.connect(lambda *args: stopped.append(args))
recorder.failed.connect(failed.append)
recorder.warned.connect(warnings.append)
processes = [FakeProcess(mine), FakeProcess(theirs)]
with only_these_tools("ffmpeg"), self.devices(), \
mock.patch.object(subprocess, "Popen", side_effect=processes) as popen:
recorder.start(path, "MacBook Pro Microphone", "BlackHole 2ch")
recorder._thread.join(timeout=5)
recorder.stop()
return path, warnings, stopped, failed, processes, popen
def test_the_two_capture_processes_become_one_stereo_file(self):
path, _, stopped, failed, _, _ = self.record(
tone(1.0, freq=440), tone(1.0, freq=880)
)
self.assertEqual(failed, [])
self.assertEqual(len(stopped), 1)
with contextlib.closing(wave.open(path, "rb")) as wav:
self.assertEqual(wav.getnchannels(), 2)
self.assertEqual(wav.getframerate(), audio.RATE)
self.assertEqual(wav.getnframes(), audio.RATE)
def test_each_avfoundation_device_is_opened_by_a_different_process(self):
_, _, _, _, _, popen = self.record(tone(0.5), tone(0.5))
commands = [call.args[0] for call in popen.call_args_list]
self.assertEqual(len(commands), 2)
self.assertTrue(all(command.count("avfoundation") == 1
for command in commands))
self.assertIn(":2", commands[0])
self.assertIn(":1", commands[1])
def test_a_mostly_empty_microphone_is_said_out_loud_and_still_kept(self):
"""Half the file is everyone else, and an hour of them is worth more
than the empty channel costs."""
path, warnings, stopped, failed, _, _ = self.record(
silence(11.0), tone(11.0)
)
self.assertEqual(failed, [])
self.assertEqual(len(stopped), 1)
self.assertTrue(os.path.exists(path))
self.assertIn("empty", warnings[0])
def test_a_microphone_that_was_merely_quiet_is_not_complained_about(self):
_, warnings, stopped, _, _, _ = self.record(tone(11.0), tone(11.0))
self.assertEqual(warnings, [])
self.assertEqual(len(stopped), 1)
def test_a_capture_that_falls_silent_ends_the_meeting_rather_than_hanging(self):
"""One thread taking turns on both pipes would sit on the dead read
until somebody noticed, an hour later."""
path = str(self.path("meeting.wav"))
recorder = audio.MeetingRecorder()
stopped = []
recorder.stopped.connect(lambda *args: stopped.append(args))
mine, theirs = StalledProcess(tone(0.512)), FakeProcess(tone(30.0))
with only_these_tools("ffmpeg"), self.devices(), \
mock.patch.object(audio, "STALL_SECONDS", 0.2), \
mock.patch.object(subprocess, "Popen", side_effect=(mine, theirs)):
try:
recorder.start(path, "MacBook Pro Microphone", "BlackHole 2ch")
recorder._thread.join(timeout=2)
self.assertFalse(recorder.active)
recorder.stop()
finally:
mine.stdout.release()
self.assertAlmostEqual(stopped[0][1], 0.512, places=3)
self.assertTrue(os.path.exists(path))
def test_stopping_ends_both_capture_processes(self):
_, _, _, _, processes, _ = self.record(tone(0.5), tone(0.5))
self.assertTrue(all(process.signals for process in processes))
def test_a_legacy_numeric_target_fails_before_recording(self):
recorder = audio.MeetingRecorder()
failed = []
recorder.failed.connect(failed.append)
with only_these_tools("ffmpeg"), self.devices(), \
mock.patch.object(subprocess, "Popen") as popen:
recorder.start(str(self.path("meeting.wav")), "2", "1")
popen.assert_not_called()
self.assertIn("old numeric index", failed[0])
def test_a_second_capture_process_that_cannot_start_cleans_up_the_first(self):
"""A Mac left holding an open AVFoundation session records nothing
else until it is let go."""
path = str(self.path("meeting.wav"))
recorder = audio.MeetingRecorder()
failed = []
recorder.failed.connect(failed.append)
first = FakeProcess(tone(1.0))
with only_these_tools("ffmpeg"), self.devices(), \
mock.patch.object(subprocess, "Popen",
side_effect=(first, OSError("refused"))):
recorder.start(path, "MacBook Pro Microphone", "BlackHole 2ch")
self.assertTrue(first.signals)
self.assertIn("refused", failed[0])
self.assertFalse(os.path.exists(path))
class MacDevices(OnMacOS, DikteTest): class MacDevices(OnMacOS, DikteTest):
@@ -527,12 +708,13 @@ class MacDevices(OnMacOS, DikteTest):
def test_the_audio_half_of_the_listing_is_the_only_half_read(self): def test_the_audio_half_of_the_listing_is_the_only_half_read(self):
with self.listing(): with self.listing():
self.assertEqual(audio.list_sources(), self.assertEqual(audio.list_sources(),
[("0", "MacBook Pro Microphone"), ("1", "BlackHole 2ch")]) [("MacBook Pro Microphone", "MacBook Pro Microphone"),
("BlackHole 2ch", "BlackHole 2ch")])
def test_the_index_is_what_ffmpeg_is_given_and_the_name_what_is_shown(self): def test_the_name_is_both_saved_and_shown(self):
with self.listing(): with self.listing():
name, description = audio.list_sources()[1] name, description = audio.list_sources()[1]
self.assertEqual(name, "1") self.assertEqual(name, "BlackHole 2ch")
self.assertIn("BlackHole", description) self.assertIn("BlackHole", description)
def test_no_ffmpeg_installed(self): def test_no_ffmpeg_installed(self):
@@ -557,7 +739,7 @@ class MacDevices(OnMacOS, DikteTest):
def test_the_loopback_driver_is_picked_out_by_name(self): def test_the_loopback_driver_is_picked_out_by_name(self):
with self.listing(): with self.listing():
self.assertEqual(audio.default_monitor(), "1") self.assertEqual(audio.default_monitor(), "BlackHole 2ch")
def test_the_other_two_drivers_people_install(self): def test_the_other_two_drivers_people_install(self):
for name in ("Loopback Audio", "Soundflower (2ch)"): for name in ("Loopback Audio", "Soundflower (2ch)"):
@@ -565,13 +747,43 @@ class MacDevices(OnMacOS, DikteTest):
listing = ("AVFoundation audio devices:\n" listing = ("AVFoundation audio devices:\n"
f"[0] Built-in Microphone\n[1] {name}\n") f"[0] Built-in Microphone\n[1] {name}\n")
with self.listing(stderr=listing): with self.listing(stderr=listing):
self.assertEqual(audio.default_monitor(), "1") self.assertEqual(audio.default_monitor(), name)
def test_a_mac_with_nothing_to_record_the_far_side_from(self): def test_a_mac_with_nothing_to_record_the_far_side_from(self):
listing = "AVFoundation audio devices:\n[0] MacBook Pro Microphone\n" listing = "AVFoundation audio devices:\n[0] MacBook Pro Microphone\n"
with self.listing(stderr=listing): with self.listing(stderr=listing):
self.assertEqual(audio.default_monitor(), "") self.assertEqual(audio.default_monitor(), "")
def test_a_saved_name_is_resolved_against_the_current_index(self):
with self.listing():
self.assertEqual(audio._resolve_avfoundation_target("BlackHole 2ch"), "1")
def test_a_saved_name_follows_the_device_when_an_earlier_one_disappears(self):
listing = ("AVFoundation audio devices:\n"
"[0] BlackHole 2ch\n[1] MacBook Pro Microphone\n")
with self.listing(stderr=listing):
self.assertEqual(
audio._resolve_avfoundation_target("MacBook Pro Microphone"), "1"
)
def test_an_old_numeric_setting_is_not_silently_reused(self):
with self.assertRaises(audio.AudioDeviceError) as caught:
audio._resolve_avfoundation_target("1")
self.assertIn("old numeric index", str(caught.exception))
def test_a_device_that_went_away_is_said_out_loud(self):
with self.listing(), self.assertRaises(audio.AudioDeviceError) as caught:
audio._resolve_avfoundation_target("USB Microphone")
self.assertIn("no longer connected", str(caught.exception))
def test_duplicate_names_are_not_guessed_between(self):
listing = ("AVFoundation audio devices:\n"
"[0] USB Microphone\n[1] USB Microphone\n")
with self.listing(stderr=listing), \
self.assertRaises(audio.AudioDeviceError) as caught:
audio._resolve_avfoundation_target("USB Microphone")
self.assertIn("More than one", str(caught.exception))
class MacRecordingCommand(OnMacOS, DikteTest): class MacRecordingCommand(OnMacOS, DikteTest):
def test_the_microphone_is_read_through_avfoundation(self): def test_the_microphone_is_read_through_avfoundation(self):
@@ -583,7 +795,11 @@ class MacRecordingCommand(OnMacOS, DikteTest):
def test_the_empty_half_in_front_of_the_colon_is_the_missing_picture(self): def test_the_empty_half_in_front_of_the_colon_is_the_missing_picture(self):
with only_these_tools("ffmpeg"): with only_these_tools("ffmpeg"):
self.assertIn(":default", audio.recording_command()) self.assertIn(":default", audio.recording_command())
self.assertIn(":2", audio.recording_command("2")) listing = "AVFoundation audio devices:\n[2] USB Microphone\n"
completed = FakeCompleted(returncode=1, stderr=listing)
with only_these_tools("ffmpeg"), \
mock.patch.object(subprocess, "run", return_value=completed):
self.assertIn(":2", audio.recording_command("USB Microphone"))
def test_it_captures_the_format_the_rest_of_the_code_expects(self): def test_it_captures_the_format_the_rest_of_the_code_expects(self):
with only_these_tools("ffmpeg"): with only_these_tools("ffmpeg"):