Make auto the default speech language and carry what was detected through

New installs start detecting instead of being locked to one language; a stored
value from before this default still wins. The dictation chain asks
transcribe_detected() in auto mode, records the detected code in history as
speech_language, hands it to the cleanup prompt (a detected Turkish recording
gets the Turkish prompt and glossary rule), and reports it on the socket reply.
The stale comment claiming whisper.cpp's -l auto does not detect is corrected.
This commit is contained in:
sudoeren
2026-08-27 21:40:09 +03:00
parent 1bb5c9ebbc
commit 7da871c567
6 changed files with 115 additions and 30 deletions
+6 -4
View File
@@ -1058,7 +1058,7 @@ class Dikte:
if not self._transcripts_pending:
self._settle(DICTATION, payload)
def _on_finished(self, _raw, text, warning):
def _on_finished(self, _raw, text, warning, speech_language):
if warning:
# The text was still pasted, but cleanup did not run. Say so loudly:
# a rejected key otherwise looks exactly like working dictation.
@@ -1079,9 +1079,10 @@ class Dikte:
t("{action}: {preview}", action=action, preview=_preview(text))
)
self._transcript_settled({"ok": True, "text": text, "raw": _raw,
"warning": warning})
"warning": warning,
"speech_language": speech_language})
def _on_ask_finished(self, _raw, text, warning):
def _on_ask_finished(self, _raw, text, warning, speech_language):
agent = assistant.display_name(self.conf)
if warning:
# A tool the agent was not allowed to touch otherwise looks exactly
@@ -1102,7 +1103,8 @@ class Dikte:
)
self._set_ask_state(IDLE)
self._settle(ASK, {"ok": True, "answer": text, "question": _raw,
"warning": warning, "agent": agent})
"warning": warning, "agent": agent,
"speech_language": speech_language})
def _on_ask_cancelled(self):
self.ask_overlay.show_done(t("Stopped."), 2000)
+14 -3
View File
@@ -395,7 +395,10 @@ DEFAULTS = {
"transcribe_model": "gpt-4o-transcribe", # used when provider is openai
"groq_transcribe_model": "whisper-large-v3-turbo",
"openrouter_transcribe_model": "openai/gpt-4o-transcribe",
"language": "tr",
# Detect on the machine by default, so a new install needs no language to
# be told. whisper.cpp detects; the hosted providers detect when handed no
# language; a stored value from before this default overrides it.
"language": "auto",
"transcribe_prompt": "",
# --- whisper.cpp, on this machine ---------------------------------------
@@ -702,8 +705,16 @@ class Config:
return self["cleanup_provider"] == "local"
def cleanup_prompt(self, with_timestamps=False, with_speakers=False,
subtitles=False):
turkish = i18n.language() == "tr"
subtitles=False, speech=""):
"""`speech` is the two-letter code of the language that was heard, when
the transcription model reported one. The default prompts and the
glossary rule only exist in Turkish and English, so a detected Turkish
recording gets the Turkish prompt and any other detected language — or
none at all — the English one, which is written not to care what
language the transcript is in. Nothing else calls this with it, so the
interface language keeps deciding everywhere the speech was not asked
about."""
turkish = (speech == "tr") if speech else i18n.language() == "tr"
if subtitles:
prompt = (self["file_cleanup_prompt"].strip()
or default_file_cleanup_prompt())
+4 -3
View File
@@ -967,12 +967,13 @@ def _whisper_args(settings):
binary, "-m", str(model),
"--inference-path", INFERENCE_PATH,
# Whatever language the request does not name. api.py leaves the field
# out when the language is "auto", and the server's own default is
# English rather than detection.
# out when the language is "auto", and the server's own language is
# set here: "auto" makes whisper.cpp detect what it hears.
"-l", "auto",
# Stock phrases invented for near-silence come from non-speech tokens,
# and verbose_json otherwise pays for a language probability sweep
# nothing here reads.
# nobody asked for. A request that wants the detected language switches
# that back on per request.
"-sns", "-nlp",
]
if int(settings["threads"]) > 0:
+24 -9
View File
@@ -37,7 +37,7 @@ _paste_lock = threading.Lock()
class Pipeline(QObject):
stage = pyqtSignal(str) # human-readable progress line
finished = pyqtSignal(str, str, str) # raw transcript, final text, warning
finished = pyqtSignal(str, str, str, str) # raw, final text, warning, language
failed = pyqtSignal(str)
cancelled = pyqtSignal()
@@ -120,12 +120,24 @@ class Pipeline(QObject):
try:
self.stage.emit(t("Transcribing…"))
target = conf.transcribe_target()
raw = api.transcribe(
target,
wav_path,
language=conf["language"],
prompt=conf["transcribe_prompt"],
)
# The spoken language is only knowable after the fact, and only the
# local server says what it heard: auto mode asks it there, and
# every other run (a fixed language, or a hosted provider that
# detects but stays silent) transcribes as before.
auto = conf["language"] == "auto"
if auto:
raw, detected = api.transcribe_detected(
target, wav_path, language=conf["language"],
prompt=conf["transcribe_prompt"],
)
else:
raw = api.transcribe(
target,
wav_path,
language=conf["language"],
prompt=conf["transcribe_prompt"],
)
detected = ""
if conf["filter_hallucinations"] and vad.looks_like_hallucination(raw, duration):
self._discard(wav_path)
@@ -145,7 +157,7 @@ class Pipeline(QObject):
self.stage.emit(t("Cleaning up…"))
cleaned = True
try:
text = cleanup.run(raw, conf, conf.cleanup_prompt())
text = cleanup.run(raw, conf, conf.cleanup_prompt(speech=detected))
except api.ApiError as exc:
# Keep the transcript, but never let the failure pass unseen:
# a rejected key would otherwise look like working dictation.
@@ -183,6 +195,9 @@ class Pipeline(QObject):
"question": question,
"assistant": assistant.provider(conf) if ask else "",
"assistant_model": assistant.model(conf) if ask else "",
# The language the run actually spoke: the detected code, or the
# configured one when nothing was detected to replace it.
"speech_language": detected or conf["language"],
"raw": raw,
"text": text,
}
@@ -221,7 +236,7 @@ class Pipeline(QObject):
time.sleep(0.35)
paste.copy_bytes(previous)
self.finished.emit(raw, text, warning)
self.finished.emit(raw, text, warning, detected)
except assistant.Cancelled:
self.cancelled.emit()