Files
dikte/worker.py
T
yusufipk 2cfbbb2d99 Transcribe and clean up on this machine, without installing anything first
whisper-server is started on --inference-path /v1/audio/transcriptions,
which is exactly the path api.py already builds for the hosted providers,
and llama-server answers /chat/completions the way OpenRouter does. So the
local half is one more base URL rather than a second code path: worker.py,
filetranscribe.py and meeting.py are untouched, and dictation, subtitles
and meetings all work here on the first try.

Three findings worth naming, none of them in the new code:

whisper.cpp cuts segments on tokens, which in Turkish lands inside a word
about as often as between two. Pasted raw that gives "akraba değ\niller.";
in a subtitle it gives a cue reading "değ". Whisper marks the start of a
word with a leading space, so a piece that does not begin with one
continues the word above it.

A small model will repeat the transcript until the context is full, and
every one of those tokens is a second of somebody waiting: measured at 206
seconds, and 25 with a ceiling on the reply. Hosted models are left alone,
where the same runaway is rare and a ceiling would cut the minutes short.

A server outlives SIGTERM and SIGKILL holding its model in memory. Signals
are now turned into an event Qt delivers, since Qt blocks in C where a
Python handler never runs, and a pid file lets the next start sweep up
what a SIGKILL left behind.

The minutes keep their own provider rather than following cleanup's. The
two jobs are not the same size: a 4B model here will strip the filler words
out of a dictation and will not write up an hour long meeting.

The suite runs offline now: a test that reaches the network says so instead
of quietly going there.
2026-08-01 20:00:35 +03:00

195 lines
7.2 KiB
Python

"""The dictation chain: transcribe → clean up → clipboard → paste.
The same chain also carries the other thing a dictation can be. Asked to, it
hands the transcript to Claude Code instead of pasting it, and pastes back
whatever came of it: an answer to a question, or a sentence saying what was
done.
"""
import os
import shutil
import sys
import threading
import time
import traceback
from PyQt6.QtCore import QObject, pyqtSignal
import api
import assistant
import audio
import config as cfg
import i18n
import paste
import vad
from i18n import t
CHUNK_SECONDS = audio.CHUNK_FRAMES / audio.RATE
# A dictation and a command to the agent run side by side and can finish at the
# same moment. Pasting is not one step but three that must not interleave: read
# what is on the clipboard, put ours there, press the key. Two runs doing that
# at once would paste one answer and restore the other's clipboard over it.
_paste_lock = threading.Lock()
class Pipeline(QObject):
stage = pyqtSignal(str) # human-readable progress line
finished = pyqtSignal(str, str, str) # raw transcript, final text, warning
failed = pyqtSignal(str)
cancelled = pyqtSignal()
def __init__(self, conf, parent=None):
super().__init__(parent)
self.conf = conf
self._thread = None
self._stop = threading.Event()
@property
def busy(self):
return self._thread is not None and self._thread.is_alive()
def run(self, wav_path, duration, rms_values=(), ask=False, paste=None):
"""`paste` overrides the setting for this one run, which is what a
dictation asked for from a terminal wants: the text comes back down the
socket, and pasting it into whatever had focus is nobody's intention."""
if self.busy:
return
self._stop.clear()
self._thread = threading.Thread(
target=self._work,
args=(wav_path, duration, list(rms_values), ask, paste),
daemon=True,
)
self._thread.start()
def cancel(self):
"""Give up on a job already under way.
Only the Claude call can honour this, and it is the only one long enough
to be worth interrupting: a transcription is over in seconds, a command
that went looking through the web is not.
"""
self._stop.set()
def _work(self, wav_path, duration, rms_values, ask, paste_override=None):
conf = self.conf
started = time.monotonic()
raw = ""
# Room tone only: don't spend an API call, and don't invite a
# hallucinated sentence back.
if conf["skip_silent"]:
stats = vad.analyse(rms_values, CHUNK_SECONDS, conf["speech_margin_db"])
if vad.is_silent(stats, conf["silence_db"], conf["speech_margin_db"],
conf["min_voiced_seconds"]):
self._discard(wav_path)
self.failed.emit(
t("No speech detected ({level} dB)", level=round(stats["speech_db"]))
)
return
try:
self.stage.emit(t("Transcribing…"))
target = conf.transcribe_target()
raw = api.transcribe(
target,
wav_path,
language=conf["language"],
prompt=conf["transcribe_prompt"],
)
if conf["filter_hallucinations"] and vad.looks_like_hallucination(raw, duration):
self._discard(wav_path)
self.failed.emit(t("Discarded a stock phrase: “{text}”", text=raw[:60]))
return
text = raw
warning = ""
cleaner = conf.cleanup_target()
# Claude reads through “eee” and “hani” without help, so a dictation
# on its way there is normally sent as it was heard, one API call and
# a second or two lighter.
if (conf["assistant_cleanup"] if ask else conf["cleanup_enabled"]):
self.stage.emit(t("Cleaning up…"))
try:
text = api.cleanup(cleaner, raw, conf.cleanup_prompt())
except api.ApiError as exc:
# Keep the transcript, but never let the failure pass unseen:
# a rejected key would otherwise look like working dictation.
text = raw
warning = str(exc)
print(f"dikte: cleanup failed: {exc}", file=sys.stderr)
question = ""
if ask:
question = text
self.stage.emit(t("Asking {name}…", name=i18n.name(
assistant.display_name(conf), "dative")))
text, denied = assistant.ask(
question, conf,
on_stage=self.stage.emit,
should_stop=self._stop.is_set,
)
warning = "\n".join(x for x in (warning, denied) if x)
wants_paste = (conf["assistant_paste"] if ask else conf["auto_paste"])
if paste_override is not None:
wants_paste = paste_override
with _paste_lock:
previous = paste.read_clipboard() if conf["restore_clipboard"] else None
paste.copy(text)
if wants_paste:
self.stage.emit(t("Pasting…"))
paste.press(conf["paste_shortcut"])
if previous is not None:
time.sleep(0.35)
paste.copy_bytes(previous)
cfg.append_history({
"ts": time.strftime("%Y-%m-%d %H:%M:%S"),
"duration": round(duration, 1),
"elapsed": round(time.monotonic() - started, 1),
"model": target.model,
"cleanup_model": cleaner.model if conf["cleanup_enabled"] else "",
"cleanup_error": warning,
"mode": "ask" if ask else "",
"question": question,
"assistant_model": conf["assistant_model"] if ask else "",
"raw": raw,
"text": text,
})
try:
cfg.trim_history(conf["history_limit"])
except OSError as exc:
print(f"dikte: could not trim the history: {exc}", file=sys.stderr)
self.finished.emit(raw, text, warning)
except assistant.Cancelled:
self.cancelled.emit()
except (api.ApiError, paste.PasteError, assistant.AssistantError) as exc:
print(f"dikte: {exc}", file=sys.stderr)
self.failed.emit(str(exc))
except Exception as exc: # never fail silently
traceback.print_exc()
self.failed.emit(t("Unexpected error: {error}", error=exc))
finally:
self._discard(wav_path)
def _discard(self, wav_path):
if not os.path.exists(wav_path):
return
if self.conf["keep_audio"]:
try:
cfg.RECORDINGS_DIR.mkdir(parents=True, exist_ok=True)
shutil.move(wav_path, cfg.RECORDINGS_DIR / (time.strftime("%Y%m%d-%H%M%S") + ".wav"))
return
except OSError:
pass
try:
os.unlink(wav_path)
except OSError:
pass