mirror of
https://github.com/yusufipk/dikte.git
synced 2026-09-11 19:06:11 +00:00
Save the transcript as subtitles too, not just as text
whisper-1's verbose response carries a start and an end for every segment, and the file tab was reading only the start, to build the [mm:ss] prefix. Keeping the end as well is all an SRT needs. The text stays the authority on wording and the segments on timing; they meet at that prefix, which the cleanup model is already told to leave alone. So a transcript that went through cleanup still turns into properly timed subtitles. A line whose stamp matches no segment runs until the next line starts, a line with no stamp at all joins the cue above it, and an end that would run into the next cue is trimmed back. The button is dead until a timestamped run finishes, because without timestamps there are no segments to time anything with.
This commit is contained in:
@@ -80,7 +80,8 @@ elapsed time, then the stage it is on. It never takes focus. Pressing
|
|||||||
dictation is not lost, but the indicator turns amber with the reason instead of
|
dictation is not lost, but the indicator turns amber with the reason instead of
|
||||||
looking like a normal run.
|
looking like a normal run.
|
||||||
- **Audio and video files** run through the same models under Settings → Audio
|
- **Audio and video files** run through the same models under Settings → Audio
|
||||||
file, optionally with `[mm:ss]` timestamps, chunked through ffmpeg when long.
|
file, optionally with `[mm:ss]` timestamps, chunked through ffmpeg when long,
|
||||||
|
and saved as `.txt` or as `.srt` subtitles.
|
||||||
- **History** of every dictation under Settings → History, with a size limit and
|
- **History** of every dictation under Settings → History, with a size limit and
|
||||||
right-click to delete.
|
right-click to delete.
|
||||||
- **Turkish and English interface**, following the system locale by default.
|
- **Turkish and English interface**, following the system locale by default.
|
||||||
|
|||||||
+1
-1
@@ -81,7 +81,7 @@ süreyi, ardından hangi aşamada olduğunu gösterir. Odak almaz. Dikte çalı
|
|||||||
normal bir çalışma gibi görünmez.
|
normal bir çalışma gibi görünmez.
|
||||||
- **Ses ve video dosyaları** Ayarlar → Ses dosyası sekmesinde aynı modellerden
|
- **Ses ve video dosyaları** Ayarlar → Ses dosyası sekmesinde aynı modellerden
|
||||||
geçer; istersen `[dd:ss]` zaman damgalarıyla, uzun dosyalar ffmpeg ile
|
geçer; istersen `[dd:ss]` zaman damgalarıyla, uzun dosyalar ffmpeg ile
|
||||||
parçalanarak.
|
parçalanarak, sonuç `.txt` ya da `.srt` altyazı olarak kaydedilerek.
|
||||||
- **Geçmiş** Ayarlar → Geçmiş sekmesinde; boyut sınırı var, sağ tıklayıp
|
- **Geçmiş** Ayarlar → Geçmiş sekmesinde; boyut sınırı var, sağ tıklayıp
|
||||||
silebilirsin.
|
silebilirsin.
|
||||||
- **Türkçe ve İngilizce arayüz**, varsayılan olarak sistem dilini izler.
|
- **Türkçe ve İngilizce arayüz**, varsayılan olarak sistem dilini izler.
|
||||||
|
|||||||
@@ -151,7 +151,7 @@ def transcribe(target, wav_path, language="", prompt="", timeout=300):
|
|||||||
|
|
||||||
|
|
||||||
def transcribe_segments(target, wav_path, language="", prompt="", timeout=300):
|
def transcribe_segments(target, wav_path, language="", prompt="", timeout=300):
|
||||||
"""[(start_seconds, text)] using whisper-1's verbose response."""
|
"""[(start_seconds, end_seconds, text)] using whisper-1's verbose response."""
|
||||||
data = _transcribe_request(
|
data = _transcribe_request(
|
||||||
target._replace(model=timestamp_model(target.provider)),
|
target._replace(model=timestamp_model(target.provider)),
|
||||||
wav_path, language, prompt, "verbose_json",
|
wav_path, language, prompt, "verbose_json",
|
||||||
@@ -162,12 +162,14 @@ def transcribe_segments(target, wav_path, language="", prompt="", timeout=300):
|
|||||||
for seg in segments:
|
for seg in segments:
|
||||||
text = (seg.get("text") or "").strip()
|
text = (seg.get("text") or "").strip()
|
||||||
if text:
|
if text:
|
||||||
out.append((float(seg.get("start") or 0.0), text))
|
start = float(seg.get("start") or 0.0)
|
||||||
|
end = float(seg.get("end") or 0.0)
|
||||||
|
out.append((start, max(end, start), text))
|
||||||
if not out:
|
if not out:
|
||||||
text = (data.get("text") or "").strip()
|
text = (data.get("text") or "").strip()
|
||||||
if not text:
|
if not text:
|
||||||
raise ApiError(t("Transcript came back empty."))
|
raise ApiError(t("Transcript came back empty."))
|
||||||
out = [(0.0, text)]
|
out = [(0.0, 0.0, text)]
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+70
-11
@@ -7,6 +7,7 @@ their timestamps shifted into place.
|
|||||||
|
|
||||||
import contextlib
|
import contextlib
|
||||||
import os
|
import os
|
||||||
|
import re
|
||||||
import shutil
|
import shutil
|
||||||
import subprocess
|
import subprocess
|
||||||
import tempfile
|
import tempfile
|
||||||
@@ -21,6 +22,10 @@ from i18n import t
|
|||||||
CHUNK_SECONDS = 600 # 10 min ≈ 19 MB at 16 kHz mono s16
|
CHUNK_SECONDS = 600 # 10 min ≈ 19 MB at 16 kHz mono s16
|
||||||
CLEANUP_CHUNK_CHARS = 12000 # keep each cleanup call comfortably small
|
CLEANUP_CHUNK_CHARS = 12000 # keep each cleanup call comfortably small
|
||||||
RATE = 16000
|
RATE = 16000
|
||||||
|
MIN_SUBTITLE_SECONDS = 1.5 # how long a cue with no end time of its own stays up
|
||||||
|
|
||||||
|
# The [mm:ss] or [h:mm:ss] prefix a timestamped line starts with.
|
||||||
|
STAMP_RE = re.compile(r"^\[(?:(\d+):)?(\d{1,2}):(\d{2})\]\s*")
|
||||||
|
|
||||||
|
|
||||||
class Cancelled(Exception):
|
class Cancelled(Exception):
|
||||||
@@ -29,7 +34,7 @@ class Cancelled(Exception):
|
|||||||
|
|
||||||
class FileTranscriber(QObject):
|
class FileTranscriber(QObject):
|
||||||
progress = pyqtSignal(str)
|
progress = pyqtSignal(str)
|
||||||
finished = pyqtSignal(str)
|
finished = pyqtSignal(str, list) # text, [(start, end, text)] when timestamped
|
||||||
failed = pyqtSignal(str)
|
failed = pyqtSignal(str)
|
||||||
|
|
||||||
def __init__(self, conf, parent=None):
|
def __init__(self, conf, parent=None):
|
||||||
@@ -76,22 +81,24 @@ class FileTranscriber(QObject):
|
|||||||
|
|
||||||
target = conf.transcribe_target()
|
target = conf.transcribe_target()
|
||||||
pieces = []
|
pieces = []
|
||||||
|
segments = []
|
||||||
for index, (chunk_path, offset) in enumerate(chunks, start=1):
|
for index, (chunk_path, offset) in enumerate(chunks, start=1):
|
||||||
self._check()
|
self._check()
|
||||||
self.progress.emit(
|
self.progress.emit(
|
||||||
t("Transcribing chunk {index}/{count}…", index=index, count=len(chunks))
|
t("Transcribing chunk {index}/{count}…", index=index, count=len(chunks))
|
||||||
)
|
)
|
||||||
if timestamps:
|
if timestamps:
|
||||||
segments = api.transcribe_segments(
|
segments.extend(
|
||||||
target,
|
(start + offset, end + offset, line)
|
||||||
chunk_path,
|
for start, end, line in api.transcribe_segments(
|
||||||
language=conf["language"],
|
target,
|
||||||
prompt=conf["transcribe_prompt"],
|
chunk_path,
|
||||||
)
|
language=conf["language"],
|
||||||
pieces.extend(
|
prompt=conf["transcribe_prompt"],
|
||||||
f"[{format_timestamp(start + offset)}] {text}"
|
)
|
||||||
for start, text in segments
|
|
||||||
)
|
)
|
||||||
|
pieces = [f"[{format_timestamp(start)}] {line}"
|
||||||
|
for start, _, line in segments]
|
||||||
else:
|
else:
|
||||||
pieces.append(api.transcribe(
|
pieces.append(api.transcribe(
|
||||||
target,
|
target,
|
||||||
@@ -107,7 +114,7 @@ class FileTranscriber(QObject):
|
|||||||
self.progress.emit(t("Cleaning up…"))
|
self.progress.emit(t("Cleaning up…"))
|
||||||
text = self._cleanup(text, timestamps)
|
text = self._cleanup(text, timestamps)
|
||||||
|
|
||||||
self.finished.emit(text)
|
self.finished.emit(text, segments)
|
||||||
|
|
||||||
except Cancelled:
|
except Cancelled:
|
||||||
self.progress.emit(t("Stopped."))
|
self.progress.emit(t("Stopped."))
|
||||||
@@ -140,6 +147,58 @@ def format_timestamp(seconds):
|
|||||||
return f"{hours}:{minutes:02d}:{secs:02d}" if hours else f"{minutes:02d}:{secs:02d}"
|
return f"{hours}:{minutes:02d}:{secs:02d}" if hours else f"{minutes:02d}:{secs:02d}"
|
||||||
|
|
||||||
|
|
||||||
|
def srt_timestamp(seconds):
|
||||||
|
millis = int(round(max(seconds, 0.0) * 1000))
|
||||||
|
hours, rest = divmod(millis, 3600000)
|
||||||
|
minutes, rest = divmod(rest, 60000)
|
||||||
|
secs, millis = divmod(rest, 1000)
|
||||||
|
return f"{hours:02d}:{minutes:02d}:{secs:02d},{millis:03d}"
|
||||||
|
|
||||||
|
|
||||||
|
def to_srt(text, segments):
|
||||||
|
"""Turn the timestamped transcript into SRT cues.
|
||||||
|
|
||||||
|
The text is the authority on wording, so cleanup edits survive; the segments
|
||||||
|
are the authority on timing. They meet at the [mm:ss] prefix, which cleanup
|
||||||
|
is told to leave alone: a line's whole-second stamp finds the segment it came
|
||||||
|
from, and with it the fractional start and the end time whisper reported. A
|
||||||
|
line whose stamp finds nothing runs until the next line starts.
|
||||||
|
"""
|
||||||
|
cues = []
|
||||||
|
for line in text.splitlines():
|
||||||
|
line = line.strip()
|
||||||
|
if not line:
|
||||||
|
continue
|
||||||
|
match = STAMP_RE.match(line)
|
||||||
|
body = line[match.end():].strip() if match else line
|
||||||
|
if not match:
|
||||||
|
if cues and body: # a wrapped line belongs to the cue above it
|
||||||
|
cues[-1][2] += " " + body
|
||||||
|
continue
|
||||||
|
if not body:
|
||||||
|
continue
|
||||||
|
hours, minutes, secs = (int(g or 0) for g in match.groups())
|
||||||
|
cues.append([hours * 3600 + minutes * 60 + secs, None, body])
|
||||||
|
|
||||||
|
timing = {}
|
||||||
|
for start, end, _ in segments:
|
||||||
|
timing.setdefault(int(start), (start, end))
|
||||||
|
for cue in cues:
|
||||||
|
cue[0], cue[1] = timing.get(cue[0], (float(cue[0]), 0.0))
|
||||||
|
for index, cue in enumerate(cues):
|
||||||
|
following = cues[index + 1][0] if index + 1 < len(cues) else 0.0
|
||||||
|
if following > cue[0]:
|
||||||
|
cue[1] = min(cue[1], following) if cue[1] > cue[0] else following
|
||||||
|
elif cue[1] <= cue[0]:
|
||||||
|
cue[1] = cue[0] + MIN_SUBTITLE_SECONDS
|
||||||
|
|
||||||
|
blocks = [
|
||||||
|
f"{number}\n{srt_timestamp(start)} --> {srt_timestamp(end)}\n{body}"
|
||||||
|
for number, (start, end, body) in enumerate(cues, start=1)
|
||||||
|
]
|
||||||
|
return "\n\n".join(blocks) + "\n" if blocks else ""
|
||||||
|
|
||||||
|
|
||||||
def _to_wav(path, workdir):
|
def _to_wav(path, workdir):
|
||||||
out = os.path.join(workdir, "audio.wav")
|
out = os.path.join(workdir, "audio.wav")
|
||||||
res = subprocess.run(
|
res = subprocess.run(
|
||||||
|
|||||||
@@ -197,8 +197,15 @@ TR = {
|
|||||||
"Stop": "Durdur",
|
"Stop": "Durdur",
|
||||||
"Copy": "Panoya kopyala",
|
"Copy": "Panoya kopyala",
|
||||||
"Save as .txt": "'.txt' olarak kaydet",
|
"Save as .txt": "'.txt' olarak kaydet",
|
||||||
|
"Save as .srt": "'.srt' olarak kaydet",
|
||||||
|
"Subtitles, timed from the segments. Needs the timestamps option.":
|
||||||
|
"Altyazı; zamanlaması bölüm damgalarından gelir. Zaman damgası seçeneği "
|
||||||
|
"işaretliyken çalışır.",
|
||||||
|
"No timestamped lines to turn into subtitles.":
|
||||||
|
"Altyazıya çevrilecek zaman damgalı satır yok.",
|
||||||
"Save transcript": "Transkripti kaydet",
|
"Save transcript": "Transkripti kaydet",
|
||||||
"Text files": "Metin dosyaları",
|
"Text files": "Metin dosyaları",
|
||||||
|
"Subtitle files": "Altyazı dosyaları",
|
||||||
"Converting audio…": "Ses dönüştürülüyor…",
|
"Converting audio…": "Ses dönüştürülüyor…",
|
||||||
"Splitting into {count} chunks…": "{count} parçaya bölünüyor…",
|
"Splitting into {count} chunks…": "{count} parçaya bölünüyor…",
|
||||||
"Transcribing chunk {index}/{count}…": "{index}/{count} parça yazıya çevriliyor…",
|
"Transcribing chunk {index}/{count}…": "{index}/{count} parça yazıya çevriliyor…",
|
||||||
|
|||||||
+29
-4
@@ -15,6 +15,7 @@ from PyQt6.QtWidgets import (
|
|||||||
import api
|
import api
|
||||||
import audio
|
import audio
|
||||||
import config as cfg
|
import config as cfg
|
||||||
|
import filetranscribe
|
||||||
import hotkey
|
import hotkey
|
||||||
from filetranscribe import FileTranscriber
|
from filetranscribe import FileTranscriber
|
||||||
from i18n import t
|
from i18n import t
|
||||||
@@ -315,9 +316,16 @@ class SettingsWindow(QDialog):
|
|||||||
)
|
)
|
||||||
save = QPushButton(t("Save as .txt"))
|
save = QPushButton(t("Save as .txt"))
|
||||||
save.clicked.connect(self._save_transcript)
|
save.clicked.connect(self._save_transcript)
|
||||||
|
self.file_save_srt = QPushButton(t("Save as .srt"))
|
||||||
|
self.file_save_srt.setToolTip(
|
||||||
|
t("Subtitles, timed from the segments. Needs the timestamps option.")
|
||||||
|
)
|
||||||
|
self.file_save_srt.setEnabled(False)
|
||||||
|
self.file_save_srt.clicked.connect(self._save_subtitles)
|
||||||
out_row = QHBoxLayout()
|
out_row = QHBoxLayout()
|
||||||
out_row.addWidget(copy)
|
out_row.addWidget(copy)
|
||||||
out_row.addWidget(save)
|
out_row.addWidget(save)
|
||||||
|
out_row.addWidget(self.file_save_srt)
|
||||||
out_row.addStretch(1)
|
out_row.addStretch(1)
|
||||||
layout.addLayout(out_row)
|
layout.addLayout(out_row)
|
||||||
return page
|
return page
|
||||||
@@ -642,6 +650,8 @@ class SettingsWindow(QDialog):
|
|||||||
if not getattr(self, "file_path", "") or self.transcriber.busy:
|
if not getattr(self, "file_path", "") or self.transcriber.busy:
|
||||||
return
|
return
|
||||||
self.file_output.clear()
|
self.file_output.clear()
|
||||||
|
self.file_segments = []
|
||||||
|
self.file_save_srt.setEnabled(False)
|
||||||
self.file_run.setEnabled(False)
|
self.file_run.setEnabled(False)
|
||||||
self.file_stop.setEnabled(True)
|
self.file_stop.setEnabled(True)
|
||||||
self.transcriber.start(
|
self.transcriber.start(
|
||||||
@@ -655,8 +665,10 @@ class SettingsWindow(QDialog):
|
|||||||
if message == t("Stopped."):
|
if message == t("Stopped."):
|
||||||
self._file_idle()
|
self._file_idle()
|
||||||
|
|
||||||
def _on_file_finished(self, text):
|
def _on_file_finished(self, text, segments):
|
||||||
self.file_output.setPlainText(text)
|
self.file_output.setPlainText(text)
|
||||||
|
self.file_segments = segments
|
||||||
|
self.file_save_srt.setEnabled(bool(segments))
|
||||||
self.file_status.setText(t("Done: {chars} characters.", chars=len(text)))
|
self.file_status.setText(t("Done: {chars} characters.", chars=len(text)))
|
||||||
self._file_idle()
|
self._file_idle()
|
||||||
|
|
||||||
@@ -669,17 +681,30 @@ class SettingsWindow(QDialog):
|
|||||||
self.file_stop.setEnabled(False)
|
self.file_stop.setEnabled(False)
|
||||||
|
|
||||||
def _save_transcript(self):
|
def _save_transcript(self):
|
||||||
text = self.file_output.toPlainText()
|
self._write_transcript(self.file_output.toPlainText(), ".txt",
|
||||||
|
f"{t('Text files')} (*.txt)")
|
||||||
|
|
||||||
|
def _save_subtitles(self):
|
||||||
|
srt = filetranscribe.to_srt(self.file_output.toPlainText(),
|
||||||
|
getattr(self, "file_segments", []))
|
||||||
|
if not srt:
|
||||||
|
self.file_status.setText(t("No timestamped lines to turn into subtitles."))
|
||||||
|
return
|
||||||
|
self._write_transcript(srt, ".srt", f"{t('Subtitle files')} (*.srt)")
|
||||||
|
|
||||||
|
def _write_transcript(self, text, suffix, file_filter):
|
||||||
if not text:
|
if not text:
|
||||||
return
|
return
|
||||||
base = os.path.splitext(os.path.basename(getattr(self, "file_path", "")))[0]
|
base = os.path.splitext(os.path.basename(getattr(self, "file_path", "")))[0]
|
||||||
start = os.path.join(self.conf["file_last_dir"] or os.path.expanduser("~"),
|
start = os.path.join(self.conf["file_last_dir"] or os.path.expanduser("~"),
|
||||||
f"{base or 'transcript'}.txt")
|
f"{base or 'transcript'}{suffix}")
|
||||||
path, _ = QFileDialog.getSaveFileName(
|
path, _ = QFileDialog.getSaveFileName(
|
||||||
self, t("Save transcript"), start, f"{t('Text files')} (*.txt)"
|
self, t("Save transcript"), start, file_filter
|
||||||
)
|
)
|
||||||
if not path:
|
if not path:
|
||||||
return
|
return
|
||||||
|
if not path.lower().endswith(suffix):
|
||||||
|
path += suffix
|
||||||
try:
|
try:
|
||||||
with open(path, "w", encoding="utf-8") as fh:
|
with open(path, "w", encoding="utf-8") as fh:
|
||||||
fh.write(text)
|
fh.write(text)
|
||||||
|
|||||||
Reference in New Issue
Block a user