Compare commits

..
Author SHA1 Message Date
yusufipek be09cc3ac6 Let the next dictation start while the last one is still working
Pressing the shortcut while a transcript was being transcribed or cleaned up
did nothing, and the thought you had while waiting was lost. The microphone is
free the moment a recording stops, so the next dictation can now be spoken at
once; the pipeline queues it and each one is finished, pasted and reported in
the order it was spoken. The corner indicator stays with the recording under
way rather than being wiped by the previous run's progress, and a stop that
lands behind an unfinished run says it is waiting its turn.
2026-08-25 14:52:17 +03:00
10 changed files with 146 additions and 179 deletions
+2 -1
View File
@@ -139,7 +139,8 @@ set next to it.
An indicator in the screen corner shows a red dot, a live waveform and the
elapsed time, then the stage it is on. It never takes focus. Pressing
`Ctrl+Space` again while Dikte is still working does nothing; nothing queues up.
`Ctrl+Space` again while the last dictation is still being cleaned up starts
the next one; it is transcribed and pasted in turn, behind the one still going.
A dictation and a command to the agent do wait on each other for the microphone,
which is one device, but for nothing else: each has its own indicator, and the
second one stacks above the first while both are up.
+3 -2
View File
@@ -134,8 +134,9 @@ yanındaki kutudan düşünme seviyesini de seçebilirsin.
| Çık | Tepsi menüsü → *Çık*, ya da `dikte quit` |
Ekranın köşesindeki gösterge kırmızı kayıt noktasını, canlı ses dalgasını ve
süreyi, ardından hangi aşamada olduğunu gösterir. Odak almaz. Dikte çalışırken
`Ctrl+Space`'e tekrar basmak bir şey yapmaz, sıraya da girmez. Dikte ile ajana
süreyi, ardından hangi aşamada olduğunu gösterir. Odak almaz. Önceki dikte daha
temizlenirken `Ctrl+Space`'e tekrar basmak yenisini başlatır; o da sırası
gelince, öndekinin ardından yazılıp yapıştırılır. Dikte ile ajana
verilen komut yalnızca mikrofon için birbirini bekler, o da tek aygıt olduğu
için; başka hiçbir şeyde beklemezler. Her birinin kendi göstergesi var, ikisi
birden ekrandayken ikincisi birincinin üstüne yerleşir.
+72 -18
View File
@@ -145,6 +145,11 @@ class Dikte:
# Which recording is the current one, so a timer set for the run that
# started it cannot stop the one that came after.
self._run_id = 0
# Dictations handed to the pipeline and not yet out of it. More than
# one is normal: the microphone is free while a transcript is being
# cleaned up, so the next dictation can already be spoken, and it then
# queues up behind the one still going.
self._transcripts_pending = 0
self.overlay = Overlay(self.conf["overlay_corner"])
# The agent's indicator sits on top of the dictation one when both are
@@ -164,9 +169,9 @@ class Dikte:
self.recorder.level.connect(self._on_level)
self.recorder.stopped.connect(self._on_recorded)
self.recorder.failed.connect(self._on_recorder_error)
self.pipeline.stage.connect(self.overlay.show_busy)
self.pipeline.stage.connect(self._on_stage)
self.pipeline.finished.connect(self._on_finished)
self.pipeline.failed.connect(self._on_error)
self.pipeline.failed.connect(self._on_pipeline_failed)
self.ask_pipeline.stage.connect(self.ask_overlay.show_busy)
self.ask_pipeline.finished.connect(self._on_ask_finished)
self.ask_pipeline.failed.connect(self._on_ask_error)
@@ -336,13 +341,18 @@ class Dikte:
BUSY: ("Working…", "view-refresh", "Dikte: working"),
}
label, icon, tip = labels[self.state]
if self.state == BUSY:
# Still working, but the microphone is free again: the menu offers
# the next dictation rather than a wait.
label = "Start recording"
agent = assistant.display_name(self.conf)
self.toggle_action.setText(t(label))
# Free while the other one is thinking, blocked only while it is holding
# the microphone.
# Blocked only while something is holding the microphone: a transcript
# still being cleaned up queues the next dictation behind it, and the
# agent thinking never blocked it at all.
self.toggle_action.setEnabled(
self.state == RECORDING or (self.state == IDLE and not self.recording)
self.state == RECORDING or not self.recording
)
asked = i18n.name(agent, "dative")
self.ask_action.setText(
@@ -587,9 +597,11 @@ class Dikte:
return
if self.state == RECORDING:
self.stop()
elif self.state == IDLE:
else:
# BUSY does not block: the microphone is free while the last
# dictation is being cleaned up, and the next one starts now and
# waits its turn in the pipeline.
self.start()
# a request during its own BUSY is ignored; nothing queues up
def _toggle_ask(self):
if self._repeated():
@@ -606,7 +618,10 @@ class Dikte:
return False
def start(self):
if self.state != IDLE or self.recording:
# Only a held microphone blocks: a previous dictation still being
# transcribed or cleaned up is the pipeline's business, not the
# recorder's.
if self.state == RECORDING or self.recording:
return
self.overlay.show_recording()
self._begin_recording(DICTATION)
@@ -634,7 +649,9 @@ class Dikte:
self.ticker.stop()
self._clear_pause()
self._set_state(BUSY)
self.overlay.show_busy(t("Transcribing"))
self.overlay.show_busy(t("Waiting for the one before it")
if self._transcripts_pending
else t("Transcribing…"))
self.recorder.stop()
def stop_ask(self):
@@ -697,7 +714,9 @@ class Dikte:
self._settle(ASK, dropped)
else:
self.overlay.dismiss()
self._set_state(IDLE)
# An earlier dictation may still be in the pipeline; only the
# recording was thrown away.
self._set_state(BUSY if self._transcripts_pending else IDLE)
self._settle(DICTATION, dropped)
def cancel_ask(self):
@@ -880,27 +899,50 @@ class Dikte:
self.ask_pipeline.run(wav_path, duration, rms_values, ask=True,
paste=wants_paste)
else:
self._transcripts_pending += 1
self.pipeline.run(wav_path, duration, rms_values, paste=wants_paste)
def _on_stage(self, message):
# The corner belongs to the recording when one is on: the previous
# run's progress must not wipe the waveform mid-sentence.
if self.state != RECORDING:
self.overlay.show_busy(message)
def _transcript_settled(self, payload):
"""One run out of the pipeline; where dictation stands now.
A request that asked to wait is answered once the queue is empty: with
runs finishing in the order they were spoken, the one it stopped is the
last of them, and an earlier run's result would be the wrong answer.
"""
self._transcripts_pending -= 1
if self.state != RECORDING:
self._set_state(BUSY if self._transcripts_pending else IDLE)
if not self._transcripts_pending:
self._settle(DICTATION, payload)
def _on_finished(self, _raw, text, warning):
if warning:
# The text was still pasted, but cleanup did not run. Say so loudly:
# a rejected key otherwise looks exactly like working dictation.
self.overlay.show_warning(
t("Pasted raw, cleanup failed: {error}", error=warning.splitlines()[0])
)
if self.state != RECORDING:
self.overlay.show_warning(
t("Pasted raw, cleanup failed: {error}",
error=warning.splitlines()[0])
)
self.tray.showMessage(
t("Dikte: cleanup failed"), warning,
QSystemTrayIcon.MessageIcon.Warning, 10000,
)
else:
elif self.state != RECORDING:
# While a new recording is on, the flash is skipped: the text
# arriving where the cursor is says everything it would have.
action = t("Pasted") if self.conf["auto_paste"] else t("Copied")
self.overlay.show_done(
t("{action}: {preview}", action=action, preview=_preview(text))
)
self._set_state(IDLE)
self._settle(DICTATION, {"ok": True, "text": text, "raw": _raw,
"warning": warning})
self._transcript_settled({"ok": True, "text": text, "raw": _raw,
"warning": warning})
def _on_ask_finished(self, _raw, text, warning):
agent = assistant.display_name(self.conf)
@@ -936,9 +978,21 @@ class Dikte:
self.ticker.stop()
(self._on_ask_error if owner == ASK else self._on_error)(message)
def _on_pipeline_failed(self, message):
"""A run the pipeline gave up on; whatever queued behind it still runs."""
if self.state == RECORDING:
# The corner belongs to the new recording; the failure still has to
# be seen somewhere.
self.tray.showMessage("Dikte", message,
QSystemTrayIcon.MessageIcon.Warning, 8000)
else:
self._report(message, self.overlay)
self._transcript_settled({"ok": False, "error": message})
def _on_error(self, message):
"""The recorder or the key listener failed; no run reached the pipeline."""
self._report(message, self.overlay)
self._set_state(IDLE)
self._set_state(BUSY if self._transcripts_pending else IDLE)
self._settle(DICTATION, {"ok": False, "error": message})
def _on_ask_error(self, message):
-27
View File
@@ -338,33 +338,6 @@ def _codex_label(item):
return t("Using {name}", name=item_type or "a tool")
def codex_models():
"""The models Codex itself would offer right now, best first.
`codex debug models` prints the catalog the CLI's own model picker reads,
fetched from OpenAI and cached beside Codex's config, so the list is as
current as the installed Codex and there is no second list to keep up to
date here. Entries the picker hides are internal and stay hidden. A machine
without Codex, or one too old to have the command, answers with nothing and
the caller keeps its built-in list.
"""
if not shutil.which("codex"):
return []
try:
proc = subprocess.run(["codex", "debug", "models"],
capture_output=True, text=True, timeout=30)
catalog = json.loads(proc.stdout or "null")
except (OSError, subprocess.SubprocessError, ValueError):
return []
if not isinstance(catalog, dict):
return []
rows = [row for row in catalog.get("models") or []
if isinstance(row, dict) and row.get("slug")
and row.get("visibility") != "hide"]
rows.sort(key=lambda row: row.get("priority") or 0)
return [row["slug"] for row in rows]
# --- OpenRouter -----------------------------------------------------------
def _ask_openrouter(prompt, conf, on_stage):
+1 -4
View File
@@ -69,6 +69,7 @@ TR = {
# --- overlay / pipeline -------------------------------------------
"Transcribing…": "Yazıya çevriliyor…",
"Waiting for the one before it…": "Öncekinin bitmesi bekleniyor…",
"Cleaning up…": "Temizleniyor…",
"Pasting…": "Yapıştırılıyor…",
"Pasted": "Yapıştırıldı",
@@ -556,10 +557,6 @@ TR = {
"“sonnet” gibi bir ad her zaman o serinin en yenisini seçer. Opus daha "
"çok düşünür ve daha geç cevaplar; bu da en çok burada hissedilir, "
"çünkü ekranın başında bekliyorsun.",
"The list is a starting point, not a fence: any model name {name} accepts "
"can be typed straight in.":
"Liste başlangıç için, sınır değil: {name} hangi model adını kabul "
"ediyorsa buraya elle yazılabilir.",
"Permissions": "İzinler",
"Decide on its own, with the safety checks on":
"Kendi karar versin, güvenlik denetimleri açık",
+2 -45
View File
@@ -79,11 +79,7 @@ ASSISTANT_PROVIDERS = [
# Aliases resolve to the newest model of that name, so they age better than an
# id does; a full id can be typed in when a particular one is wanted.
ASSISTANT_MODELS = ["sonnet", "opus", "haiku", "fable"]
# What the Codex boxes offer before Codex itself has answered, and everything
# they offer when it cannot: the real list comes from `codex debug models` when
# the window opens, so this only has to be roughly right.
CODEX_MODELS = ["gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna",
"gpt-5.5", "gpt-5.4", "gpt-5.4-mini"]
CODEX_MODELS = ["gpt-5.4-codex", "gpt-5.4", "o4-mini"]
# Starting points only; the box is editable and OpenRouter has hundreds.
ASSISTANT_OR_MODELS = [
"google/gemini-3.5-flash", "anthropic/claude-sonnet-5", "openai/gpt-5.4",
@@ -96,12 +92,6 @@ PERMISSION_MODES = [
("Allow everything", "bypassPermissions"),
("Only what needs no permission", "manual"),
]
def _typed_model_note(name):
"""Every model box takes a typed name too; the tooltip that says so."""
return t("The list is a starting point, not a fence: any model name "
"{name} accepts can be typed straight in.", name=name)
# Codex confines the commands it runs instead of asking about them.
CODEX_SANDBOXES = [
("Read anything, write in the working directory", "workspace-write"),
@@ -531,7 +521,6 @@ class SettingsWindow(QDialog):
_models_loaded = pyqtSignal(list, str)
_transcribe_models_loaded = pyqtSignal(list, str)
_codex_models_loaded = pyqtSignal(list)
# Which key was tested, whether it worked, and what to write under it.
_test_done = pyqtSignal(str, bool, str)
# The release that was found, or None, and what went wrong instead.
@@ -588,7 +577,6 @@ class SettingsWindow(QDialog):
self._models_loaded.connect(self._on_models_loaded)
self._transcribe_models_loaded.connect(self._on_transcribe_models_loaded)
self._codex_models_loaded.connect(self._on_codex_models_loaded)
self._test_done.connect(self._on_test_done)
self._update_checked.connect(self._on_update_checked)
self.transcriber.progress.connect(self._on_file_progress)
@@ -599,7 +587,6 @@ class SettingsWindow(QDialog):
self.meetings.finished.connect(self._on_minutes_finished)
self.meetings.failed.connect(self._on_minutes_failed)
self._load()
self._load_codex_models()
# Connected after the load, so that filling the boxes in is not taken
# for the user ticking them.
self.file_timestamps.toggled.connect(self._remember_file_choices)
@@ -854,13 +841,11 @@ class SettingsWindow(QDialog):
self.cleanup_claude_model = QComboBox()
self.cleanup_claude_model.setEditable(True)
self.cleanup_claude_model.addItems(CLEANUP_CLAUDE_MODELS)
self.cleanup_claude_model.setToolTip(_typed_model_note("Claude Code"))
orr_form.addRow(t("Model"), self.cleanup_claude_model)
self.cleanup_codex_model = QComboBox()
self.cleanup_codex_model.setEditable(True)
self.cleanup_codex_model.addItems([t("Codex's own default")] + CODEX_MODELS)
self.cleanup_codex_model.setToolTip(_typed_model_note("Codex"))
orr_form.addRow(t("Model"), self.cleanup_codex_model)
self.cleanup_reasoning = QComboBox()
@@ -1020,7 +1005,7 @@ class SettingsWindow(QDialog):
"A name like “sonnet” always means the newest model of that line. "
"Opus thinks harder and answers slower, which is felt here more "
"than anywhere else: you are standing in front of the screen."
) + " " + _typed_model_note("Claude Code"))
))
claude_form.addRow(t("Model"), self.assistant_model)
self.assistant_permission = QComboBox()
for label, value in PERMISSION_MODES:
@@ -1035,7 +1020,6 @@ class SettingsWindow(QDialog):
self.assistant_codex_model.addItem(t("Codex's own default"), "")
for name in CODEX_MODELS:
self.assistant_codex_model.addItem(name, name)
self.assistant_codex_model.setToolTip(_typed_model_note("Codex"))
codex_form.addRow(t("Model"), self.assistant_codex_model)
self.assistant_codex_sandbox = QComboBox()
for label, value in CODEX_SANDBOXES:
@@ -1892,33 +1876,6 @@ class SettingsWindow(QDialog):
combo.setCurrentText(current)
self.models_label.setText(t("{count} models loaded.", count=len(models)))
def _load_codex_models(self):
"""Ask Codex which models it offers, off the interface thread.
No button and no network of ours: the CLI answers from its own cache in
well under a second. Skipped when Codex is not installed, which is also
when the built-in list stays on screen and nobody is running Codex
anyway.
"""
if not shutil.which("codex"):
return
def work():
found = assistant.codex_models()
if found:
self._codex_models_loaded.emit(found)
threading.Thread(target=work, daemon=True).start()
def _on_codex_models_loaded(self, models):
for combo in (self.cleanup_codex_model, self.assistant_codex_model):
current = combo.currentText()
combo.clear()
combo.addItem(t("Codex's own default"), "")
for name in models:
combo.addItem(name, name)
combo.setCurrentText(current)
def _test_openai(self):
key, base = self._typed_key("openai")
self._test_key("openai", lambda: t(
+29 -8
View File
@@ -6,6 +6,7 @@ whatever came of it: an answer to a question, or a sentence saying what was
done.
"""
import collections
import os
import shutil
import sys
@@ -45,6 +46,13 @@ class Pipeline(QObject):
self.conf = conf
self._thread = None
self._stop = threading.Event()
# Recordings waiting their turn, and whether a thread is working them
# off. The flag rather than the thread's own liveness, because a thread
# stays alive for a moment after deciding it is done, and a job arriving
# in that moment would be left in the queue with nobody coming back.
self._jobs = collections.deque()
self._draining = False
self._jobs_lock = threading.Lock()
@property
def busy(self):
@@ -53,17 +61,30 @@ class Pipeline(QObject):
def run(self, wav_path, duration, rms_values=(), ask=False, paste=None):
"""`paste` overrides the setting for this one run, which is what a
dictation asked for from a terminal wants: the text comes back down the
socket, and pasting it into whatever had focus is nobody's intention."""
if self.busy:
return
socket, and pasting it into whatever had focus is nobody's intention.
A run started while one is going waits its turn rather than being
dropped: the next dictation can be spoken while the last one is still
being cleaned up, and each one is finished, pasted and reported in the
order it was spoken."""
with self._jobs_lock:
self._jobs.append((wav_path, duration, list(rms_values), ask, paste))
if self._draining:
return
self._draining = True
self._stop.clear()
self._thread = threading.Thread(
target=self._work,
args=(wav_path, duration, list(rms_values), ask, paste),
daemon=True,
)
self._thread = threading.Thread(target=self._drain, daemon=True)
self._thread.start()
def _drain(self):
while True:
with self._jobs_lock:
if not self._jobs:
self._draining = False
return
job = self._jobs.popleft()
self._work(*job)
def cancel(self):
"""Give up on a job already under way.
+1 -47
View File
@@ -15,8 +15,7 @@ import unittest
from unittest import mock
from dikte import assistant
from tests.support import (DikteTest, FakeCompleted, fake_urlopen,
only_these_tools)
from tests.support import DikteTest, fake_urlopen, only_these_tools
class FakeCli:
@@ -535,50 +534,5 @@ class Ask(DikteTest):
self.assertEqual(assistant.stored_provider(), "")
class CodexModels(DikteTest):
"""The model list read off `codex debug models`."""
CATALOG = {"models": [
{"slug": "gpt-6-mini", "visibility": "list", "priority": 9},
{"slug": "gpt-6", "visibility": "list", "priority": 1},
{"slug": "codex-auto-review", "visibility": "hide", "priority": 3},
]}
def models(self, reply, code=0):
with only_these_tools("codex"), \
mock.patch.object(subprocess, "run",
return_value=FakeCompleted(
returncode=code, stdout=reply)) as run:
found = assistant.codex_models()
self.run_call = run
return found
def test_the_catalog_arrives_best_first_without_the_hidden_ones(self):
found = self.models(json.dumps(self.CATALOG))
self.assertEqual(found, ["gpt-6", "gpt-6-mini"])
self.assertEqual(self.run_call.call_args.args[0],
["codex", "debug", "models"])
def test_a_codex_that_is_not_installed_is_not_run(self):
with only_these_tools(), \
mock.patch.object(subprocess, "run") as run:
self.assertEqual(assistant.codex_models(), [])
run.assert_not_called()
def test_a_codex_too_old_to_have_the_command(self):
self.assertEqual(self.models("error: unknown subcommand", code=2), [])
def test_a_catalog_that_is_not_what_was_expected(self):
self.assertEqual(self.models(json.dumps(["gpt-6"])), [])
self.assertEqual(self.models(""), [])
def test_a_codex_that_hangs_is_given_up_on(self):
with only_these_tools("codex"), \
mock.patch.object(subprocess, "run",
side_effect=subprocess.TimeoutExpired(
["codex"], 30)):
self.assertEqual(assistant.codex_models(), [])
if __name__ == "__main__":
unittest.main()
+1 -21
View File
@@ -133,8 +133,6 @@ class Settings(DikteTest):
"_load_models"))
self.enterContext(mock.patch.object(settings_ui.SettingsWindow,
"_load_transcribe_models"))
self.enterContext(mock.patch.object(settings_ui.SettingsWindow,
"_load_codex_models"))
# The local model boxes fetch their own list the moment they are shown,
# from a thread, which is nobody's test failing but a real request.
self.enterContext(mock.patch.object(settings_ui.LocalModelBox,
@@ -256,19 +254,6 @@ class Settings(DikteTest):
self.assertEqual(shown, [provider])
self.assertFalse(box.isHidden())
def test_codex_answering_refills_both_of_its_boxes(self):
"""The list Codex gave replaces the built-in one, in both places, and
neither loses what was already picked."""
conf = self.config(cleanup_codex_model="my-own-model")
window = self.window(conf)
window._on_codex_models_loaded(["gpt-6", "gpt-6-mini"])
for combo in (window.cleanup_codex_model, window.assistant_codex_model):
with self.subTest(combo=combo.objectName() or "combo"):
offered = [combo.itemText(i) for i in range(combo.count())]
self.assertEqual(offered[1:], ["gpt-6", "gpt-6-mini"])
self.assertEqual(window.cleanup_codex_model.currentText(),
"my-own-model")
def test_the_update_line_names_the_version_that_is_running(self):
window = self.window(cfg.Config())
self.assertIn(settings_ui.__version__, window.update_status.text())
@@ -669,9 +654,7 @@ class MeetingSources(DikteTest):
only_these_tools(), \
mock.patch.object(settings_ui.SettingsWindow, "_load_models"), \
mock.patch.object(settings_ui.SettingsWindow,
"_load_transcribe_models"), \
mock.patch.object(settings_ui.SettingsWindow,
"_load_codex_models"):
"_load_transcribe_models"):
window = settings_ui.SettingsWindow(cfg.Config())
self.addCleanup(window.deleteLater)
self.addCleanup(window.close)
@@ -700,9 +683,6 @@ class LocalModels(DikteTest):
# "nothing can transcribe" question from its real binary and model.
self.patch_attr(ggml, "BIN_DIR", self.path("bin"))
self.patch_attr(ggml, "MODELS_DIR", self.path("models"))
# And one with Codex on it would ask it for its model list.
self.enterContext(mock.patch.object(settings_ui.SettingsWindow,
"_load_codex_models"))
def window(self, conf):
window = settings_ui.SettingsWindow(conf)
+35 -6
View File
@@ -8,6 +8,7 @@ afterwards. A pull request that reorders any of it shows up here.
import contextlib
import io
import os
import threading
import unittest
from unittest import mock
@@ -278,13 +279,41 @@ class Chain(DikteTest):
class Busy(DikteTest):
def test_a_second_run_while_one_is_going_is_ignored(self):
def test_a_second_run_while_one_is_going_waits_its_turn(self):
"""The microphone is free while a transcript is being cleaned up, so
the next dictation can already have been spoken by then. It has to run
once the first is done, in the order they were spoken, on one thread."""
pipeline = worker.Pipeline(self.config())
pipeline._thread = mock.Mock(is_alive=lambda: True)
self.assertTrue(pipeline.busy)
with mock.patch.object(worker.threading, "Thread") as thread:
pipeline.run("/tmp/nope.wav", 1.0)
thread.assert_not_called()
order = []
started, gate = threading.Event(), threading.Event()
def work(wav_path, *_rest):
order.append(wav_path)
started.set()
if wav_path == "first.wav":
gate.wait(5)
with mock.patch.object(pipeline, "_work", side_effect=work):
pipeline.run("first.wav", 1.0)
self.assertTrue(started.wait(5))
pipeline.run("second.wav", 1.0)
# Held, not dropped and not running beside the first.
self.assertEqual(order, ["first.wav"])
gate.set()
pipeline._thread.join(5)
self.assertEqual(order, ["first.wav", "second.wav"])
def test_a_run_arriving_after_the_queue_drained(self):
"""The worker thread ends with the queue; the next run brings one."""
pipeline = worker.Pipeline(self.config())
order = []
with mock.patch.object(pipeline, "_work",
side_effect=lambda wav, *rest: order.append(wav)):
pipeline.run("first.wav", 1.0)
pipeline._thread.join(5)
pipeline.run("second.wav", 1.0)
pipeline._thread.join(5)
self.assertEqual(order, ["first.wav", "second.wav"])
def test_the_chunk_length_matches_the_level_meter(self):
"""The silence thresholds are read in seconds, so the two must agree."""