mirror of
https://github.com/yusufipk/dikte.git
synced 2026-09-12 03:16:19 +00:00
Make auto the default speech language and carry what was detected through
New installs start detecting instead of being locked to one language; a stored value from before this default still wins. The dictation chain asks transcribe_detected() in auto mode, records the detected code in history as speech_language, hands it to the cleanup prompt (a detected Turkish recording gets the Turkish prompt and glossary rule), and reports it on the socket reply. The stale comment claiming whisper.cpp's -l auto does not detect is corrected.
This commit is contained in:
+6
-4
@@ -1058,7 +1058,7 @@ class Dikte:
|
||||
if not self._transcripts_pending:
|
||||
self._settle(DICTATION, payload)
|
||||
|
||||
def _on_finished(self, _raw, text, warning):
|
||||
def _on_finished(self, _raw, text, warning, speech_language):
|
||||
if warning:
|
||||
# The text was still pasted, but cleanup did not run. Say so loudly:
|
||||
# a rejected key otherwise looks exactly like working dictation.
|
||||
@@ -1079,9 +1079,10 @@ class Dikte:
|
||||
t("{action}: {preview}", action=action, preview=_preview(text))
|
||||
)
|
||||
self._transcript_settled({"ok": True, "text": text, "raw": _raw,
|
||||
"warning": warning})
|
||||
"warning": warning,
|
||||
"speech_language": speech_language})
|
||||
|
||||
def _on_ask_finished(self, _raw, text, warning):
|
||||
def _on_ask_finished(self, _raw, text, warning, speech_language):
|
||||
agent = assistant.display_name(self.conf)
|
||||
if warning:
|
||||
# A tool the agent was not allowed to touch otherwise looks exactly
|
||||
@@ -1102,7 +1103,8 @@ class Dikte:
|
||||
)
|
||||
self._set_ask_state(IDLE)
|
||||
self._settle(ASK, {"ok": True, "answer": text, "question": _raw,
|
||||
"warning": warning, "agent": agent})
|
||||
"warning": warning, "agent": agent,
|
||||
"speech_language": speech_language})
|
||||
|
||||
def _on_ask_cancelled(self):
|
||||
self.ask_overlay.show_done(t("Stopped."), 2000)
|
||||
|
||||
+14
-3
@@ -395,7 +395,10 @@ DEFAULTS = {
|
||||
"transcribe_model": "gpt-4o-transcribe", # used when provider is openai
|
||||
"groq_transcribe_model": "whisper-large-v3-turbo",
|
||||
"openrouter_transcribe_model": "openai/gpt-4o-transcribe",
|
||||
"language": "tr",
|
||||
# Detect on the machine by default, so a new install needs no language to
|
||||
# be told. whisper.cpp detects; the hosted providers detect when handed no
|
||||
# language; a stored value from before this default overrides it.
|
||||
"language": "auto",
|
||||
"transcribe_prompt": "",
|
||||
|
||||
# --- whisper.cpp, on this machine ---------------------------------------
|
||||
@@ -702,8 +705,16 @@ class Config:
|
||||
return self["cleanup_provider"] == "local"
|
||||
|
||||
def cleanup_prompt(self, with_timestamps=False, with_speakers=False,
|
||||
subtitles=False):
|
||||
turkish = i18n.language() == "tr"
|
||||
subtitles=False, speech=""):
|
||||
"""`speech` is the two-letter code of the language that was heard, when
|
||||
the transcription model reported one. The default prompts and the
|
||||
glossary rule only exist in Turkish and English, so a detected Turkish
|
||||
recording gets the Turkish prompt and any other detected language — or
|
||||
none at all — the English one, which is written not to care what
|
||||
language the transcript is in. Nothing else calls this with it, so the
|
||||
interface language keeps deciding everywhere the speech was not asked
|
||||
about."""
|
||||
turkish = (speech == "tr") if speech else i18n.language() == "tr"
|
||||
if subtitles:
|
||||
prompt = (self["file_cleanup_prompt"].strip()
|
||||
or default_file_cleanup_prompt())
|
||||
|
||||
+4
-3
@@ -967,12 +967,13 @@ def _whisper_args(settings):
|
||||
binary, "-m", str(model),
|
||||
"--inference-path", INFERENCE_PATH,
|
||||
# Whatever language the request does not name. api.py leaves the field
|
||||
# out when the language is "auto", and the server's own default is
|
||||
# English rather than detection.
|
||||
# out when the language is "auto", and the server's own language is
|
||||
# set here: "auto" makes whisper.cpp detect what it hears.
|
||||
"-l", "auto",
|
||||
# Stock phrases invented for near-silence come from non-speech tokens,
|
||||
# and verbose_json otherwise pays for a language probability sweep
|
||||
# nothing here reads.
|
||||
# nobody asked for. A request that wants the detected language switches
|
||||
# that back on per request.
|
||||
"-sns", "-nlp",
|
||||
]
|
||||
if int(settings["threads"]) > 0:
|
||||
|
||||
+24
-9
@@ -37,7 +37,7 @@ _paste_lock = threading.Lock()
|
||||
|
||||
class Pipeline(QObject):
|
||||
stage = pyqtSignal(str) # human-readable progress line
|
||||
finished = pyqtSignal(str, str, str) # raw transcript, final text, warning
|
||||
finished = pyqtSignal(str, str, str, str) # raw, final text, warning, language
|
||||
failed = pyqtSignal(str)
|
||||
cancelled = pyqtSignal()
|
||||
|
||||
@@ -120,12 +120,24 @@ class Pipeline(QObject):
|
||||
try:
|
||||
self.stage.emit(t("Transcribing…"))
|
||||
target = conf.transcribe_target()
|
||||
raw = api.transcribe(
|
||||
target,
|
||||
wav_path,
|
||||
language=conf["language"],
|
||||
prompt=conf["transcribe_prompt"],
|
||||
)
|
||||
# The spoken language is only knowable after the fact, and only the
|
||||
# local server says what it heard: auto mode asks it there, and
|
||||
# every other run (a fixed language, or a hosted provider that
|
||||
# detects but stays silent) transcribes as before.
|
||||
auto = conf["language"] == "auto"
|
||||
if auto:
|
||||
raw, detected = api.transcribe_detected(
|
||||
target, wav_path, language=conf["language"],
|
||||
prompt=conf["transcribe_prompt"],
|
||||
)
|
||||
else:
|
||||
raw = api.transcribe(
|
||||
target,
|
||||
wav_path,
|
||||
language=conf["language"],
|
||||
prompt=conf["transcribe_prompt"],
|
||||
)
|
||||
detected = ""
|
||||
|
||||
if conf["filter_hallucinations"] and vad.looks_like_hallucination(raw, duration):
|
||||
self._discard(wav_path)
|
||||
@@ -145,7 +157,7 @@ class Pipeline(QObject):
|
||||
self.stage.emit(t("Cleaning up…"))
|
||||
cleaned = True
|
||||
try:
|
||||
text = cleanup.run(raw, conf, conf.cleanup_prompt())
|
||||
text = cleanup.run(raw, conf, conf.cleanup_prompt(speech=detected))
|
||||
except api.ApiError as exc:
|
||||
# Keep the transcript, but never let the failure pass unseen:
|
||||
# a rejected key would otherwise look like working dictation.
|
||||
@@ -183,6 +195,9 @@ class Pipeline(QObject):
|
||||
"question": question,
|
||||
"assistant": assistant.provider(conf) if ask else "",
|
||||
"assistant_model": assistant.model(conf) if ask else "",
|
||||
# The language the run actually spoke: the detected code, or the
|
||||
# configured one when nothing was detected to replace it.
|
||||
"speech_language": detected or conf["language"],
|
||||
"raw": raw,
|
||||
"text": text,
|
||||
}
|
||||
@@ -221,7 +236,7 @@ class Pipeline(QObject):
|
||||
time.sleep(0.35)
|
||||
paste.copy_bytes(previous)
|
||||
|
||||
self.finished.emit(raw, text, warning)
|
||||
self.finished.emit(raw, text, warning, detected)
|
||||
|
||||
except assistant.Cancelled:
|
||||
self.cancelled.emit()
|
||||
|
||||
Reference in New Issue
Block a user