Voice dictation for KDE Wayland: record, transcribe, clean up, paste

Ctrl+Space starts and stops a recording. The audio goes to OpenAI for
transcription, a model on OpenRouter strips the fillers and restores
punctuation, and the result is copied and pasted into the focused window.

Only the Python standard library and PyQt6 — HTTP, multipart uploads and
WAV writing are all hand-rolled.

- pw-record captures raw 16 kHz mono PCM with a live level meter
- the corner indicator is drawn through XWayland, since a Wayland client
  cannot position its own window
- silence is caught before it costs an API call, relative to each
  recording's own noise floor, plus a filter for the stock phrases models
  invent when handed silence
- audio and video files can be transcribed too, optionally with [mm:ss]
  timestamps, chunked through ffmpeg for long inputs
- global shortcut installs as a KDE custom shortcut, with an evdev
  listener as a fallback until the session is restarted
- Turkish and English interface, following the system locale by default
This commit is contained in:
yusufipk
2026-07-25 19:24:46 +07:00
commit efa8687b23
22 changed files with 3822 additions and 0 deletions
+137
View File
@@ -0,0 +1,137 @@
"""The dictation chain: transcribe → clean up → clipboard → paste."""
import os
import shutil
import sys
import threading
import time
import traceback
from PyQt6.QtCore import QObject, pyqtSignal
import api
import audio
import config as cfg
import paste
import vad
from i18n import t
CHUNK_SECONDS = audio.CHUNK_FRAMES / audio.RATE
class Pipeline(QObject):
stage = pyqtSignal(str) # human-readable progress line
finished = pyqtSignal(str, str) # raw transcript, final text
failed = pyqtSignal(str)
def __init__(self, conf, parent=None):
super().__init__(parent)
self.conf = conf
self._thread = None
@property
def busy(self):
return self._thread is not None and self._thread.is_alive()
def run(self, wav_path, duration, rms_values=()):
if self.busy:
return
self._thread = threading.Thread(
target=self._work, args=(wav_path, duration, list(rms_values)), daemon=True
)
self._thread.start()
def _work(self, wav_path, duration, rms_values):
conf = self.conf
started = time.monotonic()
raw = ""
# Room tone only: don't spend an API call, and don't invite a
# hallucinated sentence back.
if conf["skip_silent"]:
stats = vad.analyse(rms_values, CHUNK_SECONDS, conf["speech_margin_db"])
if vad.is_silent(stats, conf["silence_db"], conf["speech_margin_db"],
conf["min_voiced_seconds"]):
self._discard(wav_path)
self.failed.emit(
t("No speech detected ({level} dB)", level=round(stats["speech_db"]))
)
return
try:
self.stage.emit(t("Transcribing…"))
raw = api.transcribe(
wav_path,
conf.openai_key(),
model=conf["transcribe_model"],
language=conf["language"],
prompt=conf["transcribe_prompt"],
base_url=conf["openai_base_url"],
)
if conf["filter_hallucinations"] and vad.looks_like_hallucination(raw, duration):
self._discard(wav_path)
self.failed.emit(t("Discarded a stock phrase: “{text}", text=raw[:60]))
return
text = raw
if conf["cleanup_enabled"]:
self.stage.emit(t("Cleaning up…"))
try:
text = api.cleanup(
raw,
conf.openrouter_key(),
conf["cleanup_model"],
conf.cleanup_prompt(),
base_url=conf["openrouter_base_url"],
)
except api.ApiError as exc:
# A failed cleanup must not cost us the transcript.
text = raw
self.stage.emit(t("Cleanup skipped: {error}", error=exc))
previous = paste.read_clipboard() if conf["restore_clipboard"] else None
paste.copy(text)
if conf["auto_paste"]:
self.stage.emit(t("Pasting…"))
paste.press(conf["paste_shortcut"])
if previous is not None:
time.sleep(0.35)
paste.copy_bytes(previous)
cfg.append_history({
"ts": time.strftime("%Y-%m-%d %H:%M:%S"),
"duration": round(duration, 1),
"elapsed": round(time.monotonic() - started, 1),
"model": conf["transcribe_model"],
"cleanup_model": conf["cleanup_model"] if conf["cleanup_enabled"] else "",
"raw": raw,
"text": text,
})
cfg.trim_history(conf["history_limit"])
self.finished.emit(raw, text)
except (api.ApiError, paste.PasteError) as exc:
print(f"dikte: {exc}", file=sys.stderr)
self.failed.emit(str(exc))
except Exception as exc: # never fail silently
traceback.print_exc()
self.failed.emit(t("Unexpected error: {error}", error=exc))
finally:
self._discard(wav_path)
def _discard(self, wav_path):
if not os.path.exists(wav_path):
return
if self.conf["keep_audio"]:
try:
cfg.RECORDINGS_DIR.mkdir(parents=True, exist_ok=True)
shutil.move(wav_path, cfg.RECORDINGS_DIR / (time.strftime("%Y%m%d-%H%M%S") + ".wav"))
return
except OSError:
pass
try:
os.unlink(wav_path)
except OSError:
pass