mirror of
https://github.com/yusufipk/dikte.git
synced 2026-09-11 10:56:10 +00:00
whisper-server is started on --inference-path /v1/audio/transcriptions, which is exactly the path api.py already builds for the hosted providers, and llama-server answers /chat/completions the way OpenRouter does. So the local half is one more base URL rather than a second code path: worker.py, filetranscribe.py and meeting.py are untouched, and dictation, subtitles and meetings all work here on the first try. Three findings worth naming, none of them in the new code: whisper.cpp cuts segments on tokens, which in Turkish lands inside a word about as often as between two. Pasted raw that gives "akraba değ\niller."; in a subtitle it gives a cue reading "değ". Whisper marks the start of a word with a leading space, so a piece that does not begin with one continues the word above it. A small model will repeat the transcript until the context is full, and every one of those tokens is a second of somebody waiting: measured at 206 seconds, and 25 with a ceiling on the reply. Hosted models are left alone, where the same runaway is rare and a ceiling would cut the minutes short. A server outlives SIGTERM and SIGKILL holding its model in memory. Signals are now turned into an event Qt delivers, since Qt blocks in C where a Python handler never runs, and a pid file lets the next start sweep up what a SIGKILL left behind. The minutes keep their own provider rather than following cleanup's. The two jobs are not the same size: a 4B model here will strip the filler words out of a dictation and will not write up an hour long meeting. The suite runs offline now: a test that reaches the network says so instead of quietly going there.
195 lines
7.2 KiB
Python
195 lines
7.2 KiB
Python
"""The dictation chain: transcribe → clean up → clipboard → paste.
|
|
|
|
The same chain also carries the other thing a dictation can be. Asked to, it
|
|
hands the transcript to Claude Code instead of pasting it, and pastes back
|
|
whatever came of it: an answer to a question, or a sentence saying what was
|
|
done.
|
|
"""
|
|
|
|
import os
|
|
import shutil
|
|
import sys
|
|
import threading
|
|
import time
|
|
import traceback
|
|
|
|
from PyQt6.QtCore import QObject, pyqtSignal
|
|
|
|
import api
|
|
import assistant
|
|
import audio
|
|
import config as cfg
|
|
import i18n
|
|
import paste
|
|
import vad
|
|
from i18n import t
|
|
|
|
CHUNK_SECONDS = audio.CHUNK_FRAMES / audio.RATE
|
|
|
|
# A dictation and a command to the agent run side by side and can finish at the
|
|
# same moment. Pasting is not one step but three that must not interleave: read
|
|
# what is on the clipboard, put ours there, press the key. Two runs doing that
|
|
# at once would paste one answer and restore the other's clipboard over it.
|
|
_paste_lock = threading.Lock()
|
|
|
|
|
|
class Pipeline(QObject):
|
|
stage = pyqtSignal(str) # human-readable progress line
|
|
finished = pyqtSignal(str, str, str) # raw transcript, final text, warning
|
|
failed = pyqtSignal(str)
|
|
cancelled = pyqtSignal()
|
|
|
|
def __init__(self, conf, parent=None):
|
|
super().__init__(parent)
|
|
self.conf = conf
|
|
self._thread = None
|
|
self._stop = threading.Event()
|
|
|
|
@property
|
|
def busy(self):
|
|
return self._thread is not None and self._thread.is_alive()
|
|
|
|
def run(self, wav_path, duration, rms_values=(), ask=False, paste=None):
|
|
"""`paste` overrides the setting for this one run, which is what a
|
|
dictation asked for from a terminal wants: the text comes back down the
|
|
socket, and pasting it into whatever had focus is nobody's intention."""
|
|
if self.busy:
|
|
return
|
|
self._stop.clear()
|
|
self._thread = threading.Thread(
|
|
target=self._work,
|
|
args=(wav_path, duration, list(rms_values), ask, paste),
|
|
daemon=True,
|
|
)
|
|
self._thread.start()
|
|
|
|
def cancel(self):
|
|
"""Give up on a job already under way.
|
|
|
|
Only the Claude call can honour this, and it is the only one long enough
|
|
to be worth interrupting: a transcription is over in seconds, a command
|
|
that went looking through the web is not.
|
|
"""
|
|
self._stop.set()
|
|
|
|
def _work(self, wav_path, duration, rms_values, ask, paste_override=None):
|
|
conf = self.conf
|
|
started = time.monotonic()
|
|
raw = ""
|
|
|
|
# Room tone only: don't spend an API call, and don't invite a
|
|
# hallucinated sentence back.
|
|
if conf["skip_silent"]:
|
|
stats = vad.analyse(rms_values, CHUNK_SECONDS, conf["speech_margin_db"])
|
|
if vad.is_silent(stats, conf["silence_db"], conf["speech_margin_db"],
|
|
conf["min_voiced_seconds"]):
|
|
self._discard(wav_path)
|
|
self.failed.emit(
|
|
t("No speech detected ({level} dB)", level=round(stats["speech_db"]))
|
|
)
|
|
return
|
|
|
|
try:
|
|
self.stage.emit(t("Transcribing…"))
|
|
target = conf.transcribe_target()
|
|
raw = api.transcribe(
|
|
target,
|
|
wav_path,
|
|
language=conf["language"],
|
|
prompt=conf["transcribe_prompt"],
|
|
)
|
|
|
|
if conf["filter_hallucinations"] and vad.looks_like_hallucination(raw, duration):
|
|
self._discard(wav_path)
|
|
self.failed.emit(t("Discarded a stock phrase: “{text}”", text=raw[:60]))
|
|
return
|
|
|
|
text = raw
|
|
warning = ""
|
|
cleaner = conf.cleanup_target()
|
|
# Claude reads through “eee” and “hani” without help, so a dictation
|
|
# on its way there is normally sent as it was heard, one API call and
|
|
# a second or two lighter.
|
|
if (conf["assistant_cleanup"] if ask else conf["cleanup_enabled"]):
|
|
self.stage.emit(t("Cleaning up…"))
|
|
try:
|
|
text = api.cleanup(cleaner, raw, conf.cleanup_prompt())
|
|
except api.ApiError as exc:
|
|
# Keep the transcript, but never let the failure pass unseen:
|
|
# a rejected key would otherwise look like working dictation.
|
|
text = raw
|
|
warning = str(exc)
|
|
print(f"dikte: cleanup failed: {exc}", file=sys.stderr)
|
|
|
|
question = ""
|
|
if ask:
|
|
question = text
|
|
self.stage.emit(t("Asking {name}…", name=i18n.name(
|
|
assistant.display_name(conf), "dative")))
|
|
text, denied = assistant.ask(
|
|
question, conf,
|
|
on_stage=self.stage.emit,
|
|
should_stop=self._stop.is_set,
|
|
)
|
|
warning = "\n".join(x for x in (warning, denied) if x)
|
|
|
|
wants_paste = (conf["assistant_paste"] if ask else conf["auto_paste"])
|
|
if paste_override is not None:
|
|
wants_paste = paste_override
|
|
|
|
with _paste_lock:
|
|
previous = paste.read_clipboard() if conf["restore_clipboard"] else None
|
|
paste.copy(text)
|
|
|
|
if wants_paste:
|
|
self.stage.emit(t("Pasting…"))
|
|
paste.press(conf["paste_shortcut"])
|
|
if previous is not None:
|
|
time.sleep(0.35)
|
|
paste.copy_bytes(previous)
|
|
|
|
cfg.append_history({
|
|
"ts": time.strftime("%Y-%m-%d %H:%M:%S"),
|
|
"duration": round(duration, 1),
|
|
"elapsed": round(time.monotonic() - started, 1),
|
|
"model": target.model,
|
|
"cleanup_model": cleaner.model if conf["cleanup_enabled"] else "",
|
|
"cleanup_error": warning,
|
|
"mode": "ask" if ask else "",
|
|
"question": question,
|
|
"assistant_model": conf["assistant_model"] if ask else "",
|
|
"raw": raw,
|
|
"text": text,
|
|
})
|
|
try:
|
|
cfg.trim_history(conf["history_limit"])
|
|
except OSError as exc:
|
|
print(f"dikte: could not trim the history: {exc}", file=sys.stderr)
|
|
self.finished.emit(raw, text, warning)
|
|
|
|
except assistant.Cancelled:
|
|
self.cancelled.emit()
|
|
except (api.ApiError, paste.PasteError, assistant.AssistantError) as exc:
|
|
print(f"dikte: {exc}", file=sys.stderr)
|
|
self.failed.emit(str(exc))
|
|
except Exception as exc: # never fail silently
|
|
traceback.print_exc()
|
|
self.failed.emit(t("Unexpected error: {error}", error=exc))
|
|
finally:
|
|
self._discard(wav_path)
|
|
|
|
def _discard(self, wav_path):
|
|
if not os.path.exists(wav_path):
|
|
return
|
|
if self.conf["keep_audio"]:
|
|
try:
|
|
cfg.RECORDINGS_DIR.mkdir(parents=True, exist_ok=True)
|
|
shutil.move(wav_path, cfg.RECORDINGS_DIR / (time.strftime("%Y%m%d-%H%M%S") + ".wav"))
|
|
return
|
|
except OSError:
|
|
pass
|
|
try:
|
|
os.unlink(wav_path)
|
|
except OSError:
|
|
pass
|