From 6a57d984c5457754080888088fe4ef3298a187f9 Mon Sep 17 00:00:00 2001 From: debpalash <4178343+debpalash@users.noreply.github.com> Date: Sat, 3 Oct 2026 03:21:49 +0530 Subject: [PATCH 01/10] fix: resolve open audio and desktop regressions Preserve job cancellation and retry ownership, snapshot voice references, score the selected dub track safely, and repair audio previews, exports, model setup and saved metadata. Keep design saves from starting cold engine loads. Include regression tests, recovery translations and matching documentation. Co-authored-by: rudycelekli <47457359+rudycelekli@users.noreply.github.com> --- CHANGELOG.md | 29 ++ backend/api/routers/audiobook.py | 107 +++-- backend/api/routers/batch.py | 51 ++- backend/api/routers/dub_export.py | 148 ++++++- backend/api/routers/dub_generate.py | 76 +++- backend/api/routers/dub_translate.py | 5 +- backend/api/routers/profiles.py | 126 ++++-- backend/api/routers/pronunciation.py | 20 +- backend/core/user_env.py | 14 +- backend/core/voice_reference_snapshots.py | 32 ++ backend/services/asr_backend.py | 5 +- backend/services/longform_render.py | 3 +- backend/services/mcp_bindings.py | 38 +- backend/services/pronunciation.py | 85 ++-- backend/services/segmentation.py | 23 +- backend/services/storage_report.py | 23 +- backend/services/translator.py | 10 +- backend/services/video_context.py | 47 +-- docs/audio-quality.md | 6 +- docs/desktop-build.md | 2 + docs/electron-batch.md | 2 + docs/electron-connections.md | 13 +- docs/electron-dubbing.md | 44 +- docs/electron-gallery.md | 2 + docs/electron-longform.md | 6 +- docs/electron-storage.md | 10 +- docs/electron-transcriptions.md | 9 +- docs/electron-workflows.md | 2 + docs/expressive-speech.md | 4 + docs/mcp.md | 2 + docs/voice-design.md | 5 + electron/src/main/atomic-export.test.ts | 229 ++++++++++ electron/src/main/atomic-export.ts | 69 +++ electron/src/main/data-relocation.test.ts | 48 ++- electron/src/main/data-relocation.ts | 3 +- electron/src/main/ipc.ts | 6 +- electron/src/main/remote-backend.test.ts | 124 ++++++ electron/src/main/remote-backend.ts | 22 +- electron/src/main/setup-progress.test.ts | 44 ++ electron/src/main/setup-progress.ts | 4 +- .../src/features/projects/projects-page.tsx | 8 +- .../projects/projects-take-reuse.test.tsx | 47 +++ .../features/tools/compare-voices.test.tsx | 49 ++- .../src/features/tools/compare-voices.tsx | 11 + .../features/transcriptions/history.test.ts | 69 ++- .../transcriptions/live-dictation.test.ts | 55 +++ .../features/transcriptions/live-dictation.ts | 1 + .../src/renderer/src/i18n/locales/ar.json | 2 + .../src/renderer/src/i18n/locales/de.json | 2 + .../src/renderer/src/i18n/locales/en.json | 2 + .../src/renderer/src/i18n/locales/es.json | 2 + .../src/renderer/src/i18n/locales/fr.json | 2 + .../src/renderer/src/i18n/locales/hi.json | 2 + .../src/renderer/src/i18n/locales/id.json | 2 + .../src/renderer/src/i18n/locales/it.json | 2 + .../src/renderer/src/i18n/locales/ja.json | 2 + .../src/renderer/src/i18n/locales/ko.json | 2 + .../src/renderer/src/i18n/locales/nl.json | 2 + .../src/renderer/src/i18n/locales/pl.json | 2 + .../src/renderer/src/i18n/locales/pt.json | 2 + .../src/renderer/src/i18n/locales/ru.json | 2 + .../src/renderer/src/i18n/locales/sv.json | 2 + .../src/renderer/src/i18n/locales/th.json | 2 + .../src/renderer/src/i18n/locales/tr.json | 2 + .../src/renderer/src/i18n/locales/uk.json | 2 + .../src/renderer/src/i18n/locales/vi.json | 2 + .../src/renderer/src/i18n/locales/zh-CN.json | 2 + .../src/renderer/src/i18n/locales/zh-TW.json | 2 + .../src/renderer/src/lib/api/client.test.ts | 12 + electron/src/renderer/src/lib/api/client.ts | 9 + .../src/lib/audio/streaming-preview.test.ts | 109 +++++ .../src/lib/audio/streaming-preview.ts | 18 +- .../src/renderer/src/lib/store/takes.test.ts | 43 +- electron/src/renderer/src/lib/store/takes.ts | 4 + electron/src/shared/i18n/locales/ar.json | 2 + electron/src/shared/i18n/locales/de.json | 2 + electron/src/shared/i18n/locales/en.json | 2 + electron/src/shared/i18n/locales/es.json | 2 + electron/src/shared/i18n/locales/fr.json | 2 + electron/src/shared/i18n/locales/hi.json | 2 + electron/src/shared/i18n/locales/id.json | 2 + electron/src/shared/i18n/locales/it.json | 2 + electron/src/shared/i18n/locales/ja.json | 2 + electron/src/shared/i18n/locales/ko.json | 2 + electron/src/shared/i18n/locales/nl.json | 2 + electron/src/shared/i18n/locales/pl.json | 2 + electron/src/shared/i18n/locales/pt.json | 2 + electron/src/shared/i18n/locales/ru.json | 2 + electron/src/shared/i18n/locales/sv.json | 2 + electron/src/shared/i18n/locales/th.json | 2 + electron/src/shared/i18n/locales/tr.json | 2 + electron/src/shared/i18n/locales/uk.json | 2 + electron/src/shared/i18n/locales/vi.json | 2 + electron/src/shared/i18n/locales/zh-CN.json | 2 + electron/src/shared/i18n/locales/zh-TW.json | 2 + electron/src/shared/utils/audioTrim.js | 41 +- .../src/shared/utils/audioTrimDecode.test.ts | 43 ++ .../src/shared/utils/transcriptionsStore.js | 8 +- tests/test_asr_device_aware_autodetect.py | 26 ++ tests/test_audiobook_cancel.py | 286 +++++++++++++ tests/test_batch_retry_custody.py | 231 ++++++++++ tests/test_delete_after_commit_regression.py | 11 +- tests/test_dub_complete_audio.py | 68 +++ tests/test_dub_qc_concurrency.py | 76 ++++ tests/test_dub_qc_track_language.py | 399 ++++++++++++++++++ tests/test_dub_translate.py | 12 + tests/test_longform_render.py | 96 +++++ tests/test_longform_segment_cache.py | 35 ++ tests/test_mcp_binding_concurrent_updates.py | 103 +++++ tests/test_models_dir_setting.py | 24 +- tests/test_profile_design_save_decouple.py | 54 +++ tests/test_profile_relock_reference.py | 364 ++++++++++++++++ tests/test_profile_unification.py | 2 + tests/test_pronunciation_backup_order.py | 83 ++++ tests/test_pronunciation_language_scopes.py | 238 +++++++++++ tests/test_segmentation.py | 29 ++ tests/test_storage_wide_deadline.py | 112 +++++ tests/test_translator.py | 12 + tests/test_video_context_temp_custody.py | 93 ++++ 119 files changed, 4163 insertions(+), 300 deletions(-) create mode 100644 backend/core/voice_reference_snapshots.py create mode 100644 electron/src/main/atomic-export.test.ts create mode 100644 electron/src/main/atomic-export.ts create mode 100644 electron/src/renderer/src/features/projects/projects-take-reuse.test.tsx create mode 100644 electron/src/renderer/src/lib/audio/streaming-preview.test.ts create mode 100644 electron/src/shared/utils/audioTrimDecode.test.ts create mode 100644 tests/test_batch_retry_custody.py create mode 100644 tests/test_dub_qc_concurrency.py create mode 100644 tests/test_dub_qc_track_language.py create mode 100644 tests/test_mcp_binding_concurrent_updates.py create mode 100644 tests/test_profile_relock_reference.py create mode 100644 tests/test_pronunciation_backup_order.py create mode 100644 tests/test_pronunciation_language_scopes.py create mode 100644 tests/test_storage_wide_deadline.py create mode 100644 tests/test_video_context_temp_custody.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 362b82f7f..96553aa9a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -87,6 +87,35 @@ metadata and the backend fallback mirror it. - The Twilio guide and integration directory describe the guided setup and in-app integration pages (#2304) ### Fixed + +- Saving a voice design skips cold engine loading and downloads (#2583) — thanks @simoncheese! + +- Electron streaming previews drain the final PCM chunk and crossfade recovered chunks only while audio overlaps (#2518) — thanks @rudycelekli! +- Allow application-data relocation into existing empty folders without removing files added during copying (#2521) — thanks @rudycelekli! +- Preserve models folder names containing comment characters across restart (#2519) — thanks @rudycelekli! +- Re-render cached longform audio when its synthesis language changes (#2524) — thanks @rudycelekli! +- Reusing Clone takes restores their saved WAV precision and mastering controls (#2526) — thanks @rudycelekli! +- Keep Electron remote connection and WebSocket-ticket deadlines active while reading response bodies (#2527) — thanks @rudycelekli! +- Preserve CRLF and CR metadata paragraphs in longform audio exports (#2528) — thanks @rudycelekli! +- Cancelled dictation starts cannot replace the next session after a delayed connection ticket arrives (#2533) — thanks @rudycelekli! +- Electron remote WebSockets retain the selected backend path prefix and path-bound tickets (#2537) — thanks @rudycelekli! +- Refresh longform audio after locking a voice to another take (#2535) — thanks @rudycelekli! +- Retire cancelled longform streams and failed setup jobs while preserving resume checkpoints (#2536) — thanks @rudycelekli! +- Buffered dictation utterances receive distinct saved-history IDs so deleting one preserves the others (#2538) — thanks @rudycelekli! +- Compare voices warns when a generated preview omits speech, while keeping the surviving audio playable (#2548) — thanks @rudycelekli! +- Pronunciation scopes match language picker names and ISO codes without confusing Spanish and Estonian (#2542) — thanks @rudycelekli! +- Keep batch retries and deletion from racing over active job files (#2547) — thanks @rudycelekli! +- Dictionary backups preserve duplicate-entry pronunciation order across preview, synthesis and restore (#2552) — thanks @rudycelekli! +- Gallery trimming no longer stalls waiting for audio metadata before decoding (#2558) — thanks @rudycelekli! +- Native exports preserve existing files when a replacement write fails (#2560) — thanks @rudycelekli! +- Runtime setup keeps concurrent package download progress separate (#2562) — thanks @rudycelekli! +- Honor storage scan budgets in large flat directories (#2564) — thanks @rudycelekli! +- Clean visual-context frame directories after worker completion (#2566) — thanks @rudycelekli! +- Preserve concurrent partial MCP binding edits (#2568) — thanks @rudycelekli! +- Preserve CPU forced-alignment fallback when loading the MPS aligner fails (#2570) — thanks @rudycelekli! +- Keep untimed transcript segments alongside precise word-timed speech (#2572) — thanks @rudycelekli! +- Score dub quality against the selected track’s saved language text, reject stale checks, and preserve audio when segment identities conflict (#2574) — thanks @rudycelekli! +- Accept Japanese Han letters in translation and refinement script checks (#2576) — thanks @rudycelekli! - Keep audiobook chapter boundaries when importing CR-only manuscripts (#2508) — thanks @rudycelekli! - Preserve busy sidecars during engine-level unload instead of terminating their active operation (#2507) — thanks @Anuj04432 and @rudycelekli! - Exclude downloaded caption comments while preserving spoken metadata words (#2510) — thanks @rudycelekli! diff --git a/backend/api/routers/audiobook.py b/backend/api/routers/audiobook.py index 22e270e40..dff17b4b5 100644 --- a/backend/api/routers/audiobook.py +++ b/backend/api/routers/audiobook.py @@ -21,6 +21,7 @@ """ import asyncio +import anyio import json import logging import os @@ -587,8 +588,11 @@ def _build_synth( from services.tts_backend import OmniVoiceBackend, active_backend_id, get_backend_class opts = opts or ExpressiveOptions() + from core.voice_reference_snapshots import VoiceReferenceSnapshot, voice_file_lock + cache: dict = {} token_cache: dict = {} + references = VoiceReferenceSnapshot() def resolve(voice_id): # Translate the span token ([voice:NAME] / exact id / None) to a profile @@ -598,9 +602,13 @@ def resolve(voice_id): if voice_id not in token_cache: token_cache[voice_id] = _map_span_voice(voice_id, default_voice, voice_map) key = token_cache[voice_id] - if key not in cache: - cache[key] = _resolve_voice(key) - return cache[key] + # Resolve and claim custody atomically with profile file deletion. + # The closure keeps the snapshot alive for any pending render worker. + with voice_file_lock: + if key not in cache: + cache[key] = _resolve_voice(key) + references.retain(cache[key]["ref_audio"]) + return cache[key] engine_id = active_backend_id() cls = get_backend_class(engine_id) @@ -778,6 +786,12 @@ def _render_chapter_cached(chapter, synth, sr, engine_id, resolve, cache_dir, le seg_extra_sig = f"{lex_sig}\x00{expr_sig}" if expr_sig else lex_sig if vmap_sig: seg_extra_sig = f"{seg_extra_sig}\x00{vmap_sig}" + # Language reaches the engine even when normalization leaves text unchanged. + # Partition both layers; unknown-language legacy audio cannot satisfy an + # explicit language. Autodetect keeps its released cache derivation. + if language: + sig["\x00language"] = language + seg_extra_sig = f"{seg_extra_sig}\x00language={json.dumps(language)}" marking = will_mark() if marking: # Provenance-marked chapters cache under their own key (#1169): a @@ -808,7 +822,7 @@ def _render_chapter_cached(chapter, synth, sr, engine_id, resolve, cache_dir, le inputs: dict = { "sample rate": sr, "engine": engine_id, "normalized text": spans_tuples, "pronunciation lexicon": lex_sig, "expressive settings": expr_sig, - "voice map": vmap_sig, "watermark": marking, + "voice map": vmap_sig, "watermark": marking, "language": language, } for k, v in resolved.items(): label = f"voice {re.sub(r'[^A-Za-z0-9_-]', '', k)[:40] or '(default)'}" @@ -1116,35 +1130,41 @@ async def _render_longform_sse( def _emit(payload: dict) -> str: if job_store is not None: + if payload.get("type") == "error": + try: + if (job_store.get(job_id) or {}).get("status") in ("pending", "running"): + job_store.mark_failed(job_id, payload.get("error") or "render failed") + except Exception: + pass # terminal setup errors must still reach the client try: job_store.append_event(job_id, json.dumps(payload)) except Exception: pass # best-effort job history; never block the stream return f"data: {json.dumps(payload)}\n\n" - if not plan.chapters: - yield _emit({"type": "error", "error": "nothing to render (no chapters)"}) - return - ffmpeg = find_ffmpeg() - if not ffmpeg: - yield _emit({"type": "error", "error": "ffmpeg not available; the output needs it"}) - return - - # Confined work dir (job_id is already token-sanitized above; work_dir adds - # the basename + realpath barrier so CodeQL sees a clean path). - work = longform_resume.work_dir(job_type, job_id) - if work is None: - yield _emit({"type": "error", "error": "invalid job id"}) - return - os.makedirs(work, exist_ok=True) - # Chapter WAVs are content-addressed in a shared cache so a re-run (after a - # failure or interruption) reuses what already rendered — only the - # missing/changed chapters synthesize again (resume). Shared across both - # front doors: an identical chapter renders once. - cache_dir = os.path.join(OUTPUTS_DIR, LONGFORM_CACHE_SUBDIR) - os.makedirs(cache_dir, exist_ok=True) - prune_cache_dir(cache_dir) # bound disk before this job adds its chapters try: + if not plan.chapters: + yield _emit({"type": "error", "error": "nothing to render (no chapters)"}) + return + ffmpeg = find_ffmpeg() + if not ffmpeg: + yield _emit({"type": "error", "error": "ffmpeg not available; the output needs it"}) + return + + # Confined work dir (job_id is already token-sanitized above; work_dir adds + # the basename + realpath barrier so CodeQL sees a clean path). + work = longform_resume.work_dir(job_type, job_id) + if work is None: + yield _emit({"type": "error", "error": "invalid job id"}) + return + os.makedirs(work, exist_ok=True) + # Chapter WAVs are content-addressed in a shared cache so a re-run (after a + # failure or interruption) reuses what already rendered — only the + # missing/changed chapters synthesize again (resume). Shared across both + # front doors: an identical chapter renders once. + cache_dir = os.path.join(OUTPUTS_DIR, LONGFORM_CACHE_SUBDIR) + os.makedirs(cache_dir, exist_ok=True) + prune_cache_dir(cache_dir) # bound disk before this job adds its chapters resolved_lang = _resolve_default_language(language, default_voice) operation = "audiobook" if job_type == "audiobook" else "longform" decision = gpu_gateway.decide(operation) @@ -1285,7 +1305,7 @@ def _emit(payload: dict) -> str: yield _emit({"type": "assembling"}) meta_path = os.path.join(work, "chapters.ffmeta") - with open(meta_path, "w", encoding="utf-8") as f: + with open(meta_path, "w", encoding="utf-8", newline="") as f: f.write(build_ffmetadata(chapters_meta, global_meta=metadata)) concat_path = os.path.join(work, "concat.txt") with open(concat_path, "w", encoding="utf-8") as f: @@ -1356,6 +1376,16 @@ def _emit(payload: dict) -> str: "measured_i": measured.input_i if measured else None, } yield _emit(done) + except (asyncio.CancelledError, GeneratorExit): + # Transport cancellation/iterator closure bypass Exception; keep the + # checkpoint, but do not leave this finished response recorded as live. + if job_store is not None: + try: + if (job_store.get(job_id) or {}).get("status") in ("pending", "running"): + job_store.mark_cancelled(job_id) + except Exception: + pass # job history is best-effort, including during shutdown + raise except Exception as e: # surface, don't 500 the stream logger.exception("[%s] longform render failed", job_id) if job_store is not None: @@ -1369,8 +1399,10 @@ def _emit(payload: dict) -> str: async def _public_longform_stream(plan, **render_kwargs): """Keep generator diagnostics local if setup fails before its own guard.""" + stream = None try: - async for event in _render_longform_sse(plan, **render_kwargs): + stream = _render_longform_sse(plan, **render_kwargs) + async for event in stream: yield event except asyncio.CancelledError: raise @@ -1385,6 +1417,19 @@ async def _public_longform_stream(plan, **render_kwargs): ) yield f"data: {json.dumps({'type': 'error', 'error': error})}\n\n" + finally: + if stream is not None: + await stream.aclose() + +class _ClosingLongformResponse(StreamingResponse): + async def __call__(self, scope, receive, send): + try: + await super().__call__(scope, receive, send) + finally: + # Older ASGI disconnects and outer AnyIO scopes can cancel cleanup. + with anyio.CancelScope(shield=True): + await self.body_iterator.aclose() + @router.post("/audiobook") async def audiobook_synthesize(req: AudiobookRequest, request: Request = None): @@ -1393,7 +1438,7 @@ async def audiobook_synthesize(req: AudiobookRequest, request: Request = None): # `request` is injected by FastAPI on the HTTP path (the default only applies # to a direct in-process call, e.g. a unit test); its disconnect poll is what # lets Stop cancel the render mid-book (#1216). - return StreamingResponse( + return _ClosingLongformResponse( _public_longform_stream( plan, default_voice=req.default_voice, language=req.language, fmt=req.format, bitrate=req.bitrate, @@ -1458,7 +1503,7 @@ async def longform_render(req: LongformRenderRequest, request: Request = None): if spans: chapters.append(Chapter(title=c.title or f"Chapter {i + 1}", spans=spans)) plan = AudiobookPlan(chapters=chapters) - return StreamingResponse( + return _ClosingLongformResponse( _public_longform_stream( plan, default_voice=req.default_voice, language=req.language, fmt=req.format, bitrate=req.bitrate, @@ -1553,7 +1598,7 @@ def retire_checkpoint(): # job id), so the already-rendered chapters still hit instantly — only the # unrendered ones synthesize. Using a fresh id means the request's job_id # never names a work dir / output file (defence-in-depth path-injection). - return StreamingResponse( + return _ClosingLongformResponse( _public_longform_stream( plan, default_voice=p.get("default_voice"), language=p.get("language"), fmt=p.get("fmt", "m4b"), bitrate=p.get("bitrate", "128k"), diff --git a/backend/api/routers/batch.py b/backend/api/routers/batch.py index f55381cc1..0791c815a 100644 --- a/backend/api/routers/batch.py +++ b/backend/api/routers/batch.py @@ -14,6 +14,8 @@ import uuid import time import asyncio +import contextlib +import threading import logging from typing import Optional, List @@ -55,9 +57,25 @@ _queue: asyncio.Queue = None # Lazily initialised _worker_task: asyncio.Task = None # Background consumer _processing_job_ids: set[str] = set() +_mutating_job_ids: set[str] = set() +_job_mutation_lock = threading.Lock() _jobs: dict = {} # job_id → status dict +@contextlib.contextmanager +def _job_mutation(job_id: str): + """Reserve retry/deletion across awaits and sync endpoint threads.""" + with _job_mutation_lock: + if job_id in _mutating_job_ids: + raise HTTPException(409, "A batch job action is already in progress") + _mutating_job_ids.add(job_id) + try: + yield + finally: + with _job_mutation_lock: + _mutating_job_ids.discard(job_id) + + class BatchJobStatus(BaseModel): id: str status: str # "queued" | "running" | "done" | "failed" | "cancelled" @@ -1031,6 +1049,12 @@ def cancel_batch_job(job_id: str): @router.post("/batch/jobs/{job_id}/retry") async def retry_batch_job(job_id: str): + """Retry a terminal job once while protecting its input and output files.""" + with _job_mutation(job_id): + return await _retry_batch_job(job_id) + + +async def _retry_batch_job(job_id: str): """Retry a terminal job using its original app-owned upload and settings.""" job = _jobs.get(job_id) if not job: @@ -1081,7 +1105,24 @@ async def retry_batch_job(job_id: str): raise HTTPException(status_code=400, detail="Invalid batch job path") try: if os.path.isdir(output_dir): - await asyncio.to_thread(shutil.rmtree, output_dir) + # Cancelling the request cannot stop a filesystem worker. Keep + # custody until it settles so a second action cannot remove or + # recreate the directory while the first cleanup is still running. + cleanup = asyncio.get_running_loop().run_in_executor(None, shutil.rmtree, output_dir) + cancellation = None + while not cleanup.done(): + try: + await asyncio.shield(cleanup) + except asyncio.CancelledError as exc: + cancellation = exc + except OSError: + if cancellation is None: + raise + if cancellation is not None: + with contextlib.suppress(OSError): + cleanup.result() + raise cancellation + cleanup.result() except OSError as exc: raise HTTPException( status_code=500, @@ -1114,10 +1155,18 @@ async def retry_batch_job(job_id: str): @router.delete("/batch/jobs/{job_id}") def delete_batch_job(job_id: str): + """Delete a settled job without racing retry admission or an active worker.""" + with _job_mutation(job_id): + return _delete_batch_job(job_id) + + +def _delete_batch_job(job_id: str): """Delete a batch job record and every app-owned input/output file.""" job = _jobs.get(job_id) if not job: raise HTTPException(404, "Job not found") + if job.get("status") in ("queued", "running") or job_id in _processing_job_ids: + raise HTTPException(409, "Cancel the batch job and wait for it to stop before deleting") if job.get("video_path"): try: unlink_if_present(job["video_path"]) diff --git a/backend/api/routers/dub_export.py b/backend/api/routers/dub_export.py index 7b633a719..e743c2d49 100644 --- a/backend/api/routers/dub_export.py +++ b/backend/api/routers/dub_export.py @@ -1,4 +1,5 @@ import asyncio +import copy import io import json import logging @@ -1624,6 +1625,60 @@ async def dub_preview_segment(job_id: str, segment_index: int, lang: str = Query # ── Second-pass ASR QC (Wave 3.3 / Spec 5) ─────────────────────────────────── +def _qc_snapshot(job: dict, lang: str) -> dict: + """Freeze the selected track's scoring inputs across an ASR await.""" + return copy.deepcopy({ + "segments": job.get("segments"), + "seg_order": job.get("seg_order"), + "timing_strategy": job.get("timing_strategy"), + **{key: {lang: (job.get(key) or {}).get(lang)} for key in ( + "dubbed_tracks", "segments_i18n", "segments_i18n_cue_sources", + "fit_plans", "video_stretch_plans", + )}, + }) + + +def _qc_audio_revision(path: str) -> tuple | None: + # Regeneration writes the same WAV path before publishing new metadata. + try: + stat = os.stat(path) + return stat.st_dev, stat.st_ino, stat.st_size, stat.st_mtime_ns, stat.st_ctime_ns + except OSError: + return None + + +def _qc_source_segments(job: dict, lang: str, segments: list[dict]) -> list[dict] | None: + """Resolve source ownership before ASR; legacy ordinal times can be ambiguous.""" + track = job["dubbed_tracks"][lang] + source = track.get("source_segments") + if not source: + return source + segment_ids = [str(seg["id"]) for seg in segments if seg.get("id") is not None] + source_ids = [str(row["id"]) for row in source if row.get("id") is not None] + if (len(segment_ids) == len(set(segment_ids)) == len(segments) + and len(source_ids) == len(set(source_ids)) == len(source) == len(segments) + and set(segment_ids) == set(source_ids)): + return source + texts = (job.get("segments_i18n") or {}).get(lang) or {} + if not source_ids and len(source) == len(segments) == len(texts) == 1: + sid = segments[0].get("id") + if sid is not None and str(sid) in texts: + return [dict(source[0], id=sid)] + # A complete Smart Fit cue set proves the final timeline by stable identity + # even when the original source snapshot predates identity persistence. + if track.get("timing_strategy") == "smart_fit": + fitted = ((job.get("fit_plans") or {}).get(lang) or {}).get("fitted_segments") or [] + fitted_ids = [str(cue["id"]) for cue in fitted if cue.get("id") is not None] + if (len(segment_ids) == len(set(segment_ids)) == len(segments) + and len(fitted_ids) == len(set(fitted_ids)) == len(fitted) == len(segments) + and set(segment_ids) == set(fitted_ids)): + return None + raise HTTPException(status_code=409, detail={ + "code": "dub_qc_timing_identity_missing", + "message": "Regenerate the selected track with explicit unique segment IDs before QC: saved source timing identities are ambiguous.", + }) + + @router.post("/dub/qc/{job_id}") async def dub_qc_pass(job_id: str, lang: str = Query(None), drift_threshold: float = Query(0.5)): """Re-recognize the dubbed audio and flag lines whose recognized text @@ -1633,7 +1688,7 @@ async def dub_qc_pass(job_id: str, lang: str = Query(None), drift_threshold: flo re-dub. The generated text stays authoritative (design delta from pyvideotrans, which overwrites subtitles).""" from services import dub_qc - from services.dub_pipeline import put_job, save_job + from services.dub_pipeline import _dub_jobs_lock, put_and_save_job _job_dir_or_400(job_id) lang = _safe_lang_or_400(lang) @@ -1641,15 +1696,21 @@ async def dub_qc_pass(job_id: str, lang: str = Query(None), drift_threshold: flo if not job: raise HTTPException(status_code=404, detail="Job not found") tracks = job.get("dubbed_tracks", {}) - if lang and lang in tracks: - wav_path = _dub_artifact(tracks[lang].get("path"), job_id, missing_detail="Dubbed audio file not found") - elif tracks: - wav_path = _dub_artifact(list(tracks.values())[0].get("path"), job_id, missing_detail="Dubbed audio file not found") - else: + if not tracks: raise HTTPException(status_code=400, detail="No dubbed audio track generated yet") + # Resolve the text against the same track chosen for recognition, including + # the legacy first-track fallback when no matching language is requested. + selected_lang = lang if lang and lang in tracks else next(iter(tracks)) + wav_path = _dub_artifact(tracks[selected_lang].get("path"), job_id, missing_detail="Dubbed audio file not found") + live_job = job + with _dub_jobs_lock: + job = _qc_snapshot(live_job, selected_lang) + audio_revision = _qc_audio_revision(wav_path) + tracks = job["dubbed_tracks"] segments = job.get("segments") or [] if not segments: raise HTTPException(status_code=400, detail="Job has no segments") + source = _qc_source_segments(job, selected_lang, segments) # TTS-only install: no ASR model on disk → typed 409 with a download CTA, # BEFORE any backend load could silently auto-download whisper weights. @@ -1696,23 +1757,64 @@ def _recognize(): raise HTTPException(status_code=500, detail=f"QC transcription failed: {e}") seg_ids = job.get("seg_order") or [s.get("id", i) for i, s in enumerate(segments)] - scored = dub_qc.score_dub(segments, recognized, drift_threshold=drift_threshold, seg_ids=seg_ids) - - # Annotate each segment (non-destructive — content text untouched). - by_id = {q.seg_id: q for q in scored} - for i, s in enumerate(segments): - sid = str(seg_ids[i]) if i < len(seg_ids) else str(s.get("id", i)) - q = by_id.get(sid) - if q is None: - continue - s["qc_drift"] = q.drift - s["qc_flagged"] = q.flagged - s["qc_recognized"] = q.recognized_text - if q.new_start is not None: - s["qc_measured_start"] = q.new_start - s["qc_measured_end"] = q.new_end - put_job(job_id, job) - save_job(job_id, job) + qc_segments = _segments_for_lang(job, selected_lang) + track = tracks[selected_lang] + # A later generate can replace both the job's text and source timings. + # The selected track snapshots its own source timeline before fitting. + # Legacy snapshots have no identities: positional pairing can silently + # assign another line's times after a later render reorders the job. + if source and any(row.get("id") is not None for row in source): + qc_segments = _apply_fitted_times(qc_segments, source) + strategy = track.get("timing_strategy") or job.get("timing_strategy") + if strategy == "smart_fit": + entry = (job.get("fit_plans") or {}).get(selected_lang) or {} + fitted = entry.get("fitted_segments") + if fitted: + qc_segments = _apply_fitted_times(qc_segments, fitted) + elif strategy == "stretch_video": + entry = (job.get("video_stretch_plans") or {}).get(selected_lang) or {} + plan = entry.get("plan") + if plan: + from services.fitted_subtitles import map_time_to_fitted + # QC follows stable ids, not chronological list order; the subtitle + # helper's monotonic-list guard would corrupt reordered lines. + qc_segments = [dict(seg, start=map_time_to_fitted(seg["start"], plan), + end=map_time_to_fitted(seg["end"], plan)) + for seg in qc_segments] + scored = dub_qc.score_dub(qc_segments, recognized, + drift_threshold=drift_threshold, seg_ids=seg_ids) + + # Scoring uses the frozen inputs, but a render/edit/deletion can complete + # while ASR runs. Publish only into that same revision, under the same lock + # as history deletion so QC cannot resurrect a withdrawn job. + with _dub_jobs_lock: + current = _get_job(job_id) + if (current is not live_job + or _qc_snapshot(current, selected_lang) != job + or _qc_audio_revision(wav_path) != audio_revision): + raise HTTPException(status_code=409, detail={ + "code": "dub_qc_track_changed", + "message": "The dub changed during QC. Run QC again on the current track.", + }) + # Annotate each segment (non-destructive — content text untouched). + by_id = {q.seg_id: q for q in scored} + for i, s in enumerate(segments): + sid = str(seg_ids[i]) if i < len(seg_ids) else str(s.get("id", i)) + q = by_id.get(sid) + if q is None: + continue + s["qc_drift"] = q.drift + s["qc_flagged"] = q.flagged + s["qc_recognized"] = q.recognized_text + if q.new_start is not None: + s["qc_measured_start"] = q.new_start + s["qc_measured_end"] = q.new_end + current["segments"] = segments + if not put_and_save_job(job_id, current): + raise HTTPException(status_code=409, detail={ + "code": "dub_qc_track_changed", + "message": "The dub changed during QC. Run QC again on the current track.", + }) flagged = [q for q in scored if q.flagged] payload = json.dumps({"event": "qc_done", "engine": engine_id, diff --git a/backend/api/routers/dub_generate.py b/backend/api/routers/dub_generate.py index 5824f8a24..7a93442ea 100644 --- a/backend/api/routers/dub_generate.py +++ b/backend/api/routers/dub_generate.py @@ -170,6 +170,12 @@ def _underrun_min_rate() -> float: GAP_OVERFLOW_BUFFER_S = 0.05 +def _track_source_segments(job: dict) -> list[dict]: + """Snapshot source times by the identities persisted for this render.""" + return [{"id": seg.get("id"), "start": seg["start"], "end": seg["end"]} + for seg in job["segments"]] + + def _sync_job_segments(job: dict, req: DubRequest) -> None: """Persist the segments this dub was actually generated from back onto the job. @@ -187,20 +193,50 @@ def _sync_job_segments(job: dict, req: DubRequest) -> None: if not req.segments: return existing = [s for s in (job.get("segments") or []) if isinstance(s, dict)] - by_id = {str(s["id"]): s for s in existing if s.get("id") is not None} + by_id = {str(s["id"]): i for i, s in enumerate(existing) if s.get("id") is not None} + used_existing: set[int] = set() seg_ids = req.segment_ids or [] + # An unmatched earlier row must not consume metadata explicitly requested + # by a later row. Those stable-ID matches take priority over index fallback. + reserved_existing = {by_id[str(sid)] for sid in seg_ids + if sid is not None and str(sid) in by_id} + # The current render seeds exactly this manifest before synthesis. Only a + # complete matching vector can supply an otherwise missing row identity. + render_order = job.get("seg_order") + expected_order = [seg_ids[i] if i < len(seg_ids) else f"seg_{i}" + for i in range(len(req.segments))] + if not (isinstance(render_order, list) and render_order == expected_order + and all(isinstance(sid, str) and sid for sid in render_order) + and len(set(render_order)) == len(render_order)): + render_order = [] + # Reserve final explicit/prior identities before considering any fallback, + # including identities retained by a later row in the current request. + reserved_ids = {str(sid) for sid in seg_ids if sid is not None} + reserved_ids.update(str(row["id"]) for i, row in enumerate(existing) + if i < len(req.segments) and row.get("id") is not None + and (i >= len(seg_ids) or seg_ids[i] is None)) merged: list[dict] = [] vouched: list[str | None] = [] for i, seg in enumerate(req.segments): seg_id = seg_ids[i] if i < len(seg_ids) else None - prev = by_id.get(str(seg_id)) if seg_id is not None else None - if prev is None and i < len(existing): - prev = existing[i] + prev_index = by_id.get(str(seg_id)) if seg_id is not None else None + if prev_index in used_existing: + prev_index = None + if (prev_index is None and i < len(existing) + and i not in used_existing and i not in reserved_existing): + prev_index = i + prev = existing[prev_index] if prev_index is not None else None + if prev_index is not None: + used_existing.add(prev_index) row = dict(prev) if prev else {} if seg_id is not None: # The request id is authoritative — seg_order and the per-segment # WAV manifest are keyed by it. row["id"] = seg_id + elif (row.get("id") is None and i < len(render_order) + and str(render_order[i]) not in reserved_ids): + row["id"] = render_order[i] + reserved_ids.add(str(render_order[i])) # Source-language text survives the overwrite so dual-subtitle export # keeps working; never let the translation clobber it. row["text_original"] = row.get("text_original") or row.get("text") or "" @@ -211,6 +247,13 @@ def _sync_job_segments(job: dict, req: DubRequest) -> None: # still that import's text, never because the text happens to match. vouched.append(vouch_cue_source(row, seg.text, seg.cue_source_id)) merged.append(row) + text_keys = [str(row["id"]) if row.get("id") is not None else str(i) + for i, row in enumerate(merged)] + if len(set(text_keys)) != len(text_keys): + raise HTTPException(status_code=409, detail={ + "code": "dub_segment_identity_conflict", + "message": "Provide explicit unique segment IDs before regenerating: segment identities would overwrite language text.", + }) job["segments"] = merged # P1.2 — per-language text, additively. `job["segments"]` stays the flat @@ -523,6 +566,23 @@ async def dub_generate(job_id: str, req: DubRequest): detail="This dub session has expired or was never created. Re-upload the video to start a new one.", ) + # Validate identities against the manifest this render will seed, before + # loading a backend or admitting work that can replace a saved track. The + # projection shares only read-only source rows; synchronization rebuilds + # its own rows and language maps without publishing any job metadata. + seg_ids = req.segment_ids or [] + expected_order = [seg_ids[i] if i < len(seg_ids) else f"seg_{i}" + for i in range(len(req.segments))] + if len(set(expected_order)) != len(expected_order): + raise HTTPException(status_code=409, detail={ + "code": "dub_segment_identity_conflict", + "message": "Provide explicit unique segment IDs before regenerating: render identities would overwrite segment audio.", + }) + _sync_job_segments({ + "segments": job.get("segments"), + "seg_order": expected_order, + }, req) + # ── Engine resolution (issue #312 class) ──────────────────────────────── # Every rendered segment clones either source speech or a saved profile, so # local execution still requires a cloning-capable engine. Remote execution @@ -754,7 +814,7 @@ def _write_memmap_wav_atomic(target_path: str, samples, sample_rate: int) -> Non # Manifest: stable segment id per current index. Per-segment WAVs are # named by stable id (dub_seg_path) so regen reuses the right audio after # reorder; index-keyed readers (preview/export) resolve via this manifest. - job["seg_order"] = [seg_ids[k] if k < len(seg_ids) else f"seg_{k}" for k in range(len(req.segments))] + job["seg_order"] = list(expected_order) # Per-segment metadata to persist after the hot loop. Audio itself is # written immediately and only file paths are kept, so long videos don't @@ -1981,13 +2041,15 @@ def _render_batch() -> list[torch.Tensor]: # mux step needs this to know whether to use the original video as-is # or stretch it per the plan. track_dur = total_samples / sr if total_samples > 0 else 0.0 + # Validate synchronized identity/text keys before installing the new + # track metadata; a collision must not publish a misleading snapshot. + _sync_job_segments(job, req) job["dubbed_tracks"][lang_code] = { "path": track_path, "language": req.language, "language_code": lang_code, "duration": round(track_dur, 4), "timing_strategy": strategy, - "source_segments": [{"start": seg.start, "end": seg.end} for seg in req.segments], } # Persist the timing strategy + (for Mode B) the per-segment stretch @@ -1999,7 +2061,7 @@ def _render_batch() -> list[torch.Tensor]: job["timing_strategy"] = strategy # Keep job segments in lock-step with what was just rendered so # subtitle export / burn-in use the translated text (#309). - _sync_job_segments(job, req) + job["dubbed_tracks"][lang_code]["source_segments"] = _track_source_segments(job) if strategy == "stretch_video": stretch_plans = job.setdefault("video_stretch_plans", {}) stretch_plans[lang_code] = { diff --git a/backend/api/routers/dub_translate.py b/backend/api/routers/dub_translate.py index 3b8e16589..0acddfd82 100644 --- a/backend/api/routers/dub_translate.py +++ b/backend/api/routers/dub_translate.py @@ -10,7 +10,7 @@ from schemas.requests import AgentFitRequest, TranslateRequest from services.model_manager import _cpu_pool, _gpu_pool from services.hf_revisions import revision_for -from services.translator import cinematic_available, cinematic_refine_many, _cinematic_budget +from services.translator import cinematic_available, cinematic_refine_many, _cinematic_budget, _is_japanese_han from api.routers.dub_core import _get_job, _save_job router = APIRouter() @@ -178,7 +178,8 @@ def _script_ratio(text: str, code: str) -> float: letters = [c for c in text if c.isalpha()] if not letters: return 1.0 - inside = sum(1 for c in letters if lo <= ord(c) <= hi) + inside = sum(1 for c in letters if lo <= ord(c) <= hi + or (code == "ja" and _is_japanese_han(ord(c)))) return inside / len(letters) diff --git a/backend/api/routers/profiles.py b/backend/api/routers/profiles.py index 114959da2..25def9b26 100644 --- a/backend/api/routers/profiles.py +++ b/backend/api/routers/profiles.py @@ -7,7 +7,6 @@ import weakref import time import shutil -import threading from typing import Optional from fastapi import APIRouter, File, Form, UploadFile, HTTPException from fastapi.responses import FileResponse, Response @@ -21,6 +20,7 @@ from omnivoice.utils.voice_design import heal_design_instruct, sanitize_instruct from core.path_security import UnsafePath, resolve_within from core.profile_images import MAX_IMAGE_BYTES, normalize_portrait +from core.voice_reference_snapshots import voice_file_lock as _voice_file_lock, references_in_use from starlette.datastructures import UploadFile as StarletteUploadFile router = APIRouter() @@ -187,41 +187,35 @@ async def create_profile( ref_text = await _auto_transcribe_reference(audio_path) used_seed = seed else: - # Saving a design profile is a pure persistence operation — it must not - # depend on a loaded TTS model (issue #476: on a fresh model-less Docker - # image the render forced a full model load + inference that 503'd, so - # the save failed). We try the deterministic identity sample opportunist- - # ically through the one shared TTS path (archetypes' renderer, never a - # second inference code path); if the engine isn't ready it's rendered - # lazily on first preview/use. The row carries vd_states + instruct, so - # the voice is fully usable without the sample (synthesis falls back to - # instruct-only conditioning — see generation.py's design path). + # Saving must not start a cold model load or download (#2583). + # Preserve the identity sample for an already resident engine; otherwise + # use the existing pending-sample path, rendered on explicit preview. from pathlib import Path from api.routers.archetypes import _render_archetype_wav - audio_filename = f"{profile_id}.wav" - audio_path = os.path.join(VOICES_DIR, audio_filename) - try: - await _render_archetype_wav( - { - "language": language, - "sample_script": ref_text, # optional custom sample line - "instruct": instruct, - }, - Path(audio_path), - ) - except Exception: - # Engine unavailable / OOM / inference failure — defer the sample. - # Store the row with no ref_audio_path; the identity sample is - # rendered on first preview or use. Never let this block the save. - import logging - logging.getLogger("omnivoice.profiles").info( - "Design profile %s saved with sample pending — " - "voice engine not ready; will render on first use", profile_id, - ) - if os.path.exists(audio_path): # partial/blank render: don't keep it - with __import__("contextlib").suppress(OSError): - os.remove(audio_path) - audio_filename = None + from services.model_manager import get_model_status + audio_path = os.path.join(VOICES_DIR, f"{profile_id}.wav") + audio_filename = None + if get_model_status()["loaded"]: + try: + await _render_archetype_wav( + { + "language": language, + "sample_script": ref_text, # optional custom sample line + "instruct": instruct, + }, + Path(audio_path), + ) + audio_filename = f"{profile_id}.wav" + except Exception: + # OOM / inference failure — defer the sample, clearing partials. + import logging + logging.getLogger("omnivoice.profiles").info( + "Design profile %s saved with sample pending — " + "voice engine not ready; will render on preview", profile_id, + ) + if os.path.exists(audio_path): + with contextlib.suppress(OSError): + os.remove(audio_path) used_seed = seed if seed is not None else _DESIGN_SEED try: @@ -770,10 +764,6 @@ async def _materialize_design_sample(profile_id: str, row) -> Optional[str]: ) return audio_filename -# Serializes the lock/unlock/consent file swaps so one request's cleanup can -# never unlink audio another request just installed. -_voice_file_lock = threading.RLock() - def _install_staged(staged: str, target: str): """Move ``staged`` onto ``target``, keeping any previous ``target`` as a @@ -834,7 +824,8 @@ async def lock_profile( if not src_path.is_file(): raise HTTPException(status_code=404, detail="Audio file not found on disk") - locked_filename = f"{profile_id}_locked.wav" + # A new take must change reference identity even if text and seed match. + locked_filename = f"{profile_id}_locked_{uuid.uuid4().hex}.wav" locked_path = _voices_path(locked_filename) if locked_path is None: raise HTTPException(status_code=400, detail="Invalid profile id") @@ -862,6 +853,9 @@ async def lock_profile( with contextlib.suppress(OSError): os.remove(staged_path) finalize() + # Existing renders may have cached the previous filename before their + # engine reads it. Keep immutable locked versions until profile deletion; + # a database reference check cannot identify those in-flight readers. event_bus.emit("profiles", {"action": "locked", "id": profile_id}) return {"locked": True, "profile_id": profile_id, "locked_audio_path": locked_filename} @@ -883,12 +877,12 @@ async def unlock_profile(profile_id: str): "UPDATE voice_profiles SET locked_audio_path='', seed=NULL, is_locked=0 WHERE id=?", (profile_id,) ) - # Unlink only after the row change committed (a rolled-back unlock must - # keep its locked take). Holding the lock stops a concurrent re-lock - # from installing a take that this unlink would then remove. - if locked_path: - with contextlib.suppress(OSError): - os.remove(locked_path) + # Unlink only after the row change committed. An admitted render may + # still hold this immutable version; profile deletion reclaims it once + # all readers finish. Shared references (including this profile's other + # audio fields) must also remain usable after unlocking. + if locked_path and not references_in_use([locked_path]): + _remove_voice_file(profile["locked_audio_path"], keep="") event_bus.emit("profiles", {"action": "unlocked", "id": profile_id}) return {"unlocked": True, "profile_id": profile_id} @@ -1018,6 +1012,11 @@ def revoke_consent(profile_id: str): @router.delete("/profiles/{profile_id}") def delete_profile(profile_id: str): + with _voice_file_lock: + return _delete_profile(profile_id) + + +def _delete_profile(profile_id: str): paths = [] with db_conn() as conn: row = conn.execute("SELECT ref_audio_path, locked_audio_path, consent_audio_path FROM voice_profiles WHERE id=?", (profile_id,)).fetchone() @@ -1027,15 +1026,50 @@ def delete_profile(profile_id: str): path = _voices_path(row[col]) if path: paths.append(path) + if row: + # Relocking retains immutable versions for already-admitted renders. + # Explicit deletion reclaims only this profile's generated names. + versions = re.compile(re.escape(profile_id) + r"_locked(?:_[0-9a-f]{32})?\.wav") + if os.path.isdir(VOICES_DIR): + for filename in os.listdir(VOICES_DIR): + if versions.fullmatch(filename): + path = _voices_path(filename) + if path: + paths.append(path) portrait_path = _voices_path(f"{profile_id}.portrait.jpg") if portrait_path and os.path.isfile(portrait_path): paths.append(portrait_path) + # A live longform resolver may still read an earlier immutable version. + # Reject deletion before changing either the record or its assets. Files + # shared by another profile will not be removed and need no such guard. + exclusive_paths = [] + for path in dict.fromkeys(paths): + filename = os.path.basename(path) + shared = conn.execute( + "SELECT 1 FROM voice_profiles WHERE id<>? AND " + "(ref_audio_path=? OR locked_audio_path=? OR consent_audio_path=?) LIMIT 1", + (profile_id, filename, filename, filename), + ).fetchone() + if not shared: + exclusive_paths.append(path) + if references_in_use(exclusive_paths): + raise HTTPException(409, "Wait for renders using this voice to finish before deleting the profile") # Commit the database change before removing assets: a failed write or # commit must leave the rolled-back profile's files usable. conn.execute("UPDATE generation_history SET profile_id = NULL WHERE profile_id=?", (profile_id,)) conn.execute("DELETE FROM voice_profiles WHERE id=?", (profile_id,)) failed_assets = [] - for path in dict.fromkeys(paths): + # A shared path excluded from the guarded decision must not become a new + # deletion candidate if the other profile changes after our commit. + for path in exclusive_paths: + filename = os.path.basename(path) + with db_conn() as conn: + shared = conn.execute( + "SELECT 1 FROM voice_profiles WHERE ref_audio_path=? OR locked_audio_path=? " + "OR consent_audio_path=? LIMIT 1", (filename, filename, filename), + ).fetchone() + if shared: + continue try: os.remove(path) except FileNotFoundError: diff --git a/backend/api/routers/pronunciation.py b/backend/api/routers/pronunciation.py index af2f36883..361c80c8d 100644 --- a/backend/api/routers/pronunciation.py +++ b/backend/api/routers/pronunciation.py @@ -16,8 +16,8 @@ GET /pronunciation/export → all entries as JSON (round-trips import) POST /pronunciation/import → bulk add entries from JSON -Scope: ``language='*'`` is global (applies to every request); a 2-letter code -(``'en'``, ``'de'``) applies only when the request language matches. +Scope: ``language='*'`` is global (applies to every request); a canonical language +ID (``'en'``, ``'de'``, ``'kbt'``) applies only when the request language matches. """ from __future__ import annotations @@ -36,6 +36,7 @@ apply_pronunciation, entries_for_language, inert_entries_for_language, + normalize_language_scope, ) logger = logging.getLogger("omnivoice.pronunciation") @@ -88,13 +89,8 @@ def _validate_type_replacement(etype: str, replacement: str) -> None: def _norm_language(language: Optional[str]) -> str: - """Normalize a scope to '*' (global) or a lowercase 2-letter code.""" - if not language: - return _ALL_LANG - s = str(language).strip() - if not s or s == _ALL_LANG or s.lower() == "auto": - return _ALL_LANG - return s.lower()[:2] + """Normalize a saved scope using the same resolver as synthesis requests.""" + return normalize_language_scope(language) or _ALL_LANG def _row_to_dict(r) -> dict: @@ -142,7 +138,7 @@ def list_entries(): with db_conn() as conn: rows = conn.execute( "SELECT id, term, replacement, type, language, enabled, created_at " - "FROM pronunciation_entries ORDER BY created_at ASC, id ASC" + "FROM pronunciation_entries ORDER BY created_at ASC, rowid ASC" ).fetchall() return [_row_to_dict(r) for r in rows] @@ -250,7 +246,7 @@ def test_substitution(req: PronTestRequest): with db_conn() as conn: rows = conn.execute( "SELECT id, term, replacement, type, language, enabled, created_at " - "FROM pronunciation_entries" + "FROM pronunciation_entries ORDER BY created_at ASC, rowid ASC" ).fetchall() substituted = apply_pronunciation(req.text, rows, req.language) applied = entries_for_language(rows, req.language) @@ -275,7 +271,7 @@ def export_entries(): with db_conn() as conn: rows = conn.execute( "SELECT term, replacement, type, language, enabled " - "FROM pronunciation_entries ORDER BY created_at ASC, id ASC" + "FROM pronunciation_entries ORDER BY created_at ASC, rowid ASC" ).fetchall() return {"entries": [ {"term": r["term"], "replacement": r["replacement"], "type": r["type"], diff --git a/backend/core/user_env.py b/backend/core/user_env.py index deb5ac1b1..fa43aa5b0 100644 --- a/backend/core/user_env.py +++ b/backend/core/user_env.py @@ -6,12 +6,15 @@ ``OMNIVOICE_CACHE_DIR`` here, which main.py then maps to ``HF_HOME`` / ``HF_HUB_CACHE`` / ``TORCH_HOME``. -Format is dotenv-style ``KEY=value`` lines. Upsert preserves other keys (e.g. a -persisted ``HF_TOKEN``) and writes the file ``0600`` (it can hold secrets). +Format is dotenv-style ``KEY=value`` lines, with single-quoted escaping for +values containing comment, quote, whitespace or backslash characters. This +matches Electron's durable data-directory format. Upsert preserves other keys +(e.g. a persisted ``HF_TOKEN``) and writes the file ``0600`` (it can hold secrets). """ from __future__ import annotations import os +import re from typing import Optional USER_ENV_PATH = os.path.expanduser("~/.config/omnivoice/env") @@ -56,7 +59,10 @@ def get_user_env(key: str, path: Optional[str] = None) -> Optional[str]: prefix = f"{key}=" for line in _read_lines(path): if line.startswith(prefix): - return line[len(prefix):] + value = line[len(prefix):] + if len(value) >= 2 and value.startswith("'") and value.endswith("'"): + return re.sub(r"\\(['\\])", r"\1", value[1:-1]) + return value return None @@ -64,6 +70,8 @@ def set_user_env(key: str, value: str, path: Optional[str] = None) -> None: """Upsert ``KEY=value``, preserving all other lines.""" path = path or os.environ.get("OMNIVOICE_ENV_FILE") or USER_ENV_PATH prefix = f"{key}=" + if any(char.isspace() or char in "#'\"\\" for char in value): + value = "'" + value.replace("\\", "\\\\").replace("'", "\\'") + "'" lines = _read_lines(path) replaced = False for i, line in enumerate(lines): diff --git a/backend/core/voice_reference_snapshots.py b/backend/core/voice_reference_snapshots.py new file mode 100644 index 000000000..e6d3d2563 --- /dev/null +++ b/backend/core/voice_reference_snapshots.py @@ -0,0 +1,32 @@ +"""Track reference paths while longform resolvers can still read them.""" +import os +import threading +import weakref + + +# Profile file swaps/deletion and resolution share this short synchronous lock. +voice_file_lock = threading.RLock() +_snapshots = weakref.WeakSet() + + +class VoiceReferenceSnapshot: + """A cached resolver owns this object; registry membership is weak. + + Workers keep their resolver alive until they finish, even if their HTTP + request was cancelled. No timers or job-status guesses govern file custody. + """ + + def __init__(self): + self.paths = set() + with voice_file_lock: + _snapshots.add(self) + + def retain(self, path): + if path: + self.paths.add(os.path.realpath(path)) + + +def references_in_use(paths): + """Called under voice_file_lock before committing a profile deletion.""" + targets = {os.path.realpath(path) for path in paths} + return any(targets.intersection(snapshot.paths) for snapshot in _snapshots) diff --git a/backend/services/asr_backend.py b/backend/services/asr_backend.py index d4ed858ad..322431799 100644 --- a/backend/services/asr_backend.py +++ b/backend/services/asr_backend.py @@ -626,7 +626,10 @@ def forced_align(segments: list, audio, language_code: str, device: str | None = for i, dev in enumerate(devices): align = load_align_model(language_code, dev) if align is None: - return segments # no aligner for this language — not a device problem + # Loading can fail during device transfer as well as for an + # unsupported language. Try the remaining CPU fallback before + # giving up on alignment; load failures are cached per device. + continue model_a, metadata = align try: import whisperx diff --git a/backend/services/longform_render.py b/backend/services/longform_render.py index eefabfba9..3e0dfab19 100644 --- a/backend/services/longform_render.py +++ b/backend/services/longform_render.py @@ -66,7 +66,8 @@ def _escape_meta(value: str) -> str: """Escape an FFMETADATA value (``=``, ``;``, ``#``, ``\\``, newline).""" - return re.sub(r"([=;#\\\n])", r"\\\1", value or "") + value = (value or "").replace("\r\n", "\n").replace("\r", "\n") + return re.sub(r"([=;#\\\n])", r"\\\1", value) def prune_cache_dir(cache_dir: str, max_bytes: int = _CACHE_MAX_BYTES) -> tuple[int, int]: diff --git a/backend/services/mcp_bindings.py b/backend/services/mcp_bindings.py index 6b7285795..0826ab9af 100644 --- a/backend/services/mcp_bindings.py +++ b/backend/services/mcp_bindings.py @@ -46,28 +46,24 @@ def upsert_binding( if not client_id or not client_id.strip(): raise ValueError("client_id must be non-empty") cid = client_id.strip() - existing = get_binding(cid) now = time.time() - if existing: - merged = { - "label": existing["label"] if label is None else label, - "profile_id": existing["profile_id"] if profile_id is None else (profile_id or None), - "default_engine": existing["default_engine"] if default_engine is None else (default_engine or None), - } - with db_conn() as conn: - conn.execute( - "UPDATE mcp_client_bindings SET label=?, profile_id=?, default_engine=? WHERE client_id=?", - (merged["label"], merged["profile_id"], merged["default_engine"], cid), - ) - else: - with db_conn() as conn: - conn.execute( - "INSERT INTO mcp_client_bindings " - "(client_id, label, profile_id, default_engine, last_seen_at, created_at) " - "VALUES (?, ?, ?, ?, NULL, ?)", - (cid, label or "", profile_id or None, default_engine or None, now), - ) - return get_binding(cid) + with db_conn() as conn: + # Merge partial edits inside SQLite's write, rather than against a + # snapshot from another connection. Concurrent creation is an upsert, + # and omitted fields preserve the latest committed values. + row = conn.execute( + "INSERT INTO mcp_client_bindings " + "(client_id, label, profile_id, default_engine, last_seen_at, created_at) " + "VALUES (?, ?, ?, ?, NULL, ?) " + "ON CONFLICT(client_id) DO UPDATE SET " + "label=CASE WHEN ? IS NULL THEN mcp_client_bindings.label ELSE excluded.label END, " + "profile_id=CASE WHEN ? IS NULL THEN mcp_client_bindings.profile_id ELSE excluded.profile_id END, " + "default_engine=CASE WHEN ? IS NULL THEN mcp_client_bindings.default_engine ELSE excluded.default_engine END " + "RETURNING *", + (cid, label or "", profile_id or None, default_engine or None, now, + label, profile_id, default_engine), + ).fetchone() + return dict(row) def delete_binding(client_id: str) -> bool: diff --git a/backend/services/pronunciation.py b/backend/services/pronunciation.py index dccdaa700..d0a78cbae 100644 --- a/backend/services/pronunciation.py +++ b/backend/services/pronunciation.py @@ -31,6 +31,7 @@ from pathlib import Path from typing import Optional +from omnivoice.utils.lang_map import LANG_NAME_TO_ID from services.dub_qc import _NO_SPACE_SCRIPT # A "word" character for boundary purposes. We treat the standard regex word @@ -170,49 +171,67 @@ def save_lexicon(path, lexicon: Optional[dict]) -> dict[str, str]: # The JSON ``load_lexicon``/``save_lexicon`` above stay the per-project audiobook # override. THIS layer is the user-editable, DB-persisted, per-language default # dictionary surfaced in Settings → Pronunciation. Rows scoped ``language="*"`` -# apply to every request; a 2-letter language row applies only when the request -# language's prefix matches (case-insensitive), so a German entry never fires on +# apply to every request; a language row applies only when its canonical ID +# matches the request language, so a German entry never fires on # an English render. Both layers are pure text substitution — they ride the same # ReDoS-safe ``apply_lexicon`` matcher, so every engine honors them. _ALL_LANG = "*" +_CHINESE_SCRIPT_SCOPES = {"cmn-hans", "cmn-hant", "zho-hans", "zho-hant"} -def _lang_prefix(language: Optional[str]) -> Optional[str]: - """Normalize a request language to a lowercase 2-letter prefix. +def normalize_language_scope(language: Optional[str]) -> Optional[str]: + """Resolve picker names and ISO region tags to a dictionary language ID. - ``"Auto"``/``None``/``""`` → ``None`` (means "no language pin": only global - ``*`` rows apply, language-tagged rows are skipped, mirroring how the engines - treat an unset language). A value like ``"en-US"`` / ``"English"`` → - ``"en"`` (first two letters); matching against entries is on this prefix. + Auto/unset/global requests have no language pin. Unknown values remain + literal: old truncated codes cannot be unambiguously assigned a language. """ if not language: return None - s = str(language).strip().lower() - if not s or s == "auto": + value = str(language).strip().lower() + if not value or value in ("auto", _ALL_LANG): return None - return s[:2] + aliases = {"mandarin": "zh", "arabic": "ar", "tagalog": "tl"} + if value in aliases: + return aliases[value] + if value in LANG_NAME_TO_ID: + return LANG_NAME_TO_ID[value] + tag = value.replace("_", "-") + if tag in _CHINESE_SCRIPT_SCOPES: + return tag + head = tag.split("-", 1)[0] + if head in LANG_NAME_TO_ID.values(): + return head + return value + + +def _language_scope_chain(language: Optional[str]) -> tuple[str, ...]: + """Matching scopes from base fallback to exact supported script scope.""" + scope = normalize_language_scope(language) + if scope is None: + return () + if scope in _CHINESE_SCRIPT_SCOPES: + return (scope.split("-", 1)[0], scope) + return (scope,) def entries_for_language(entries, language: Optional[str]) -> dict[str, str]: """Collapse DB rows into a ``{term: replacement}`` map for ``apply_lexicon``. Filters to ``enabled`` rows whose scope is global (``*``) OR whose language - prefix matches the request language. Only the **respelling** path produces a + ID matches the request language. Only the **respelling** path produces a plain substitution here (Phase 1); IPA/CMU rows that carry no respelling are skipped at this layer (they're handled — or honestly degraded — by the engine-markup path, never silently mangling text). A language-specific row - overrides a global row with the same (case-folded) term, so a per-language - pronunciation can refine the global default. + overrides a global row for the same literal match. For supported Chinese + script tags, the exact script overrides its base-language fallback. ``entries`` is any iterable of mappings/rows with ``term``, ``replacement``, ``type``, ``language``, ``enabled`` keys (a ``sqlite3.Row`` works directly). """ - req_prefix = _lang_prefix(language) - # Two passes so language rows win over global rows on the same term: collect - # global first, then overlay matching-language rows. + # Exact script scopes override base fallbacks, which override global rows. + scoped = {scope: {} for scope in _language_scope_chain(language)} glob: dict[str, str] = {} - lang: dict[str, str] = {} for e in entries: try: if not int(e["enabled"]): @@ -230,13 +249,20 @@ def entries_for_language(entries, language: Optional[str]) -> dict[str, str]: if etype != "respelling": continue scope = (e["language"] or _ALL_LANG).strip() or _ALL_LANG - if scope == _ALL_LANG: - glob[term] = str(replacement) - else: - if req_prefix is not None and scope[:2].lower() == req_prefix: - lang[term] = str(replacement) - merged = dict(glob) - merged.update(lang) # language rows override global on the same term + layer = ( + glob if scope == _ALL_LANG + else scoped.get(normalize_language_scope(scope)) + ) + if layer is not None: + layer.pop(term, None) + layer[term] = str(replacement) + merged: dict[str, str] = {} + for layer in (glob, *scoped.values()): + for term, replacement in layer.items(): + # Move overrides after weaker rows, including case-variant keys, + # so the existing literal matcher honors scope precedence. + merged.pop(term, None) + merged[term] = replacement return merged @@ -259,7 +285,7 @@ def inert_entries_for_language(entries, language: str | None) -> list[dict]: caller can now say WHY nothing happened instead of implying nothing matched. """ - req_prefix = _lang_prefix(language) + matching_scopes = _language_scope_chain(language) out: list[dict] = [] for e in entries or []: try: @@ -274,7 +300,10 @@ def inert_entries_for_language(entries, language: str | None) -> list[dict]: if etype == "respelling": continue scope = (e["language"] or _ALL_LANG).strip() or _ALL_LANG - if scope != _ALL_LANG and (req_prefix is None or scope[:2].lower() != req_prefix): + if ( + scope != _ALL_LANG + and normalize_language_scope(scope) not in matching_scopes + ): continue out.append({"term": term, "type": etype}) return out @@ -364,7 +393,7 @@ def load_entries_from_db() -> list[dict]: with db_conn() as conn: rows = conn.execute( "SELECT id, term, replacement, type, language, enabled, created_at " - "FROM pronunciation_entries ORDER BY created_at ASC, id ASC" + "FROM pronunciation_entries ORDER BY created_at ASC, rowid ASC" ).fetchall() return [dict(r) for r in rows] diff --git a/backend/services/segmentation.py b/backend/services/segmentation.py index 9a3fb8b4d..4fdc02e1d 100644 --- a/backend/services/segmentation.py +++ b/backend/services/segmentation.py @@ -208,7 +208,9 @@ def _words_from_whisper(result: dict) -> List[Word]: words: List[Word] = [] segs = result.get("segments") if isinstance(result, dict) else None if segs: + has_precise_words = False for seg in segs: + segment_words: List[Word] = [] for w in seg.get("words", []) or []: wt = (w.get("word") or w.get("text") or "").strip() if not wt: @@ -217,9 +219,26 @@ def _words_from_whisper(result: dict) -> List[Word]: we = float(w.get("end", seg.get("end", ws + 0.1))) if we <= ws: we = ws + 0.05 - words.append(Word(start=ws, end=we, text=wt)) - if words: + segment_words.append(Word(start=ws, end=we, text=wt)) + if segment_words: + has_precise_words = True + words.extend(segment_words) + else: + # Forced alignment can leave individual segments without word + # timings. Preserve their speech using their own chunk span, + # without replacing precise timings on neighboring segments. + text = _clean(seg.get("text", "")) + start = float(seg.get("start") or 0.0) + end = float(seg.get("end") or start + 0.1) + if text and end > start: + tokens = text.split(" ") + duration = (end - start) / len(tokens) + words.extend(Word(start=start + i * duration, + end=start + (i + 1) * duration, text=token) + for i, token in enumerate(tokens)) + if has_precise_words: return words + words.clear() # Keep the legacy chunks fallback when no words are timed. # Fallback: chunk-level timings (no per-word granularity) for chunk in result.get("chunks", []) or []: diff --git a/backend/services/storage_report.py b/backend/services/storage_report.py index 25a71cf2c..0b373557b 100644 --- a/backend/services/storage_report.py +++ b/backend/services/storage_report.py @@ -131,6 +131,8 @@ def _onerror(e: OSError) -> None: try: if not os.path.exists(path): return 0, True, None + if time.monotonic() >= deadline: + return 0, False, None if not os.path.isdir(path): return os.lstat(path).st_size, True, None except OSError: @@ -139,10 +141,14 @@ def _onerror(e: OSError) -> None: total = 0 complete = True for root, _dirs, files in os.walk(path, onerror=_onerror): - if time.monotonic() > deadline: + if time.monotonic() >= deadline: complete = False break for name in files: + # A large flat directory yields only one walk root. Checking just + # between roots can spend minutes stat'ing that root's files. + if time.monotonic() >= deadline: + return total, False, err_path fp = os.path.join(root, name) try: total += os.lstat(fp).st_size @@ -343,39 +349,52 @@ def _finish(category_id: str, cat: dict, complete: bool, err_path: str | None) - }) other_bytes = 0 + # Unlisted or unclassifiable managed engines may hide bytes owned by Other. + other_complete = not (engines_child and engine_err is not None) try: with os.scandir(data_dir) as it: for e in it: if e.name in claimed: continue + if time.monotonic() >= deadline: + data_complete = other_complete = False + break if e.is_dir(follow_symlinks=False): size, ok, err = _dir_size(e.path, deadline) other_bytes += size data_complete = data_complete and ok + other_complete = other_complete and ok and err is None data_err = data_err or err else: try: other_bytes += e.stat(follow_symlinks=False).st_size except OSError: data_err = data_err or e.path + other_complete = False except OSError: if os.path.exists(data_dir): data_err = data_err or data_dir + other_complete = False if engines_child: for e in engine_entries: if e.path in engine_dirs or e.path in unclassified_engines: continue + if time.monotonic() >= deadline: + data_complete = other_complete = False + break try: if e.is_dir(follow_symlinks=False): size, ok, err = _dir_size(e.path, deadline) other_bytes += size data_complete = data_complete and ok + other_complete = other_complete and ok and err is None data_err = data_err or err else: other_bytes += e.stat(follow_symlinks=False).st_size except OSError: data_err = data_err or e.path - children.append({"id": "other", "path": data_dir, "bytes": other_bytes, "complete": True}) + other_complete = False + children.append({"id": "other", "path": data_dir, "bytes": other_bytes, "complete": other_complete}) data_cat = { "id": "data", diff --git a/backend/services/translator.py b/backend/services/translator.py index 41721dcf7..7f54235f7 100644 --- a/backend/services/translator.py +++ b/backend/services/translator.py @@ -84,6 +84,13 @@ } +def _is_japanese_han(codepoint: int) -> bool: + """Japanese also uses Han letters and iteration marks, not just kana.""" + return (0x3005 <= codepoint <= 0x3007 or 0x3400 <= codepoint <= 0x4DBF + or 0x4E00 <= codepoint <= 0x9FFF or 0xF900 <= codepoint <= 0xFAFF + or 0x20000 <= codepoint <= 0x323AF) + + def _looks_like_target_script(text: str, code: str, threshold: float = 0.5) -> bool: rng = _SCRIPT_RANGES.get(code) if not rng: @@ -92,7 +99,8 @@ def _looks_like_target_script(text: str, code: str, threshold: float = 0.5) -> b letters = [c for c in text if c.isalpha()] if not letters: return True - inside = sum(1 for c in letters if lo <= ord(c) <= hi) + inside = sum(1 for c in letters if lo <= ord(c) <= hi + or (code == "ja" and _is_japanese_han(ord(c)))) return (inside / len(letters)) >= threshold diff --git a/backend/services/video_context.py b/backend/services/video_context.py index f57521cd8..ad589d619 100644 --- a/backend/services/video_context.py +++ b/backend/services/video_context.py @@ -39,6 +39,7 @@ def _extract_keyframes( video_path: str, timestamps: list[float], max_frames: int = 30, + output_dir: str | None = None, ) -> list[tuple[float, str]]: """Extract frames at specified timestamps using ffmpeg. @@ -57,7 +58,7 @@ def _extract_keyframes( logger.warning("ffmpeg not found, skipping frame extraction") return [] - tmp_dir = tempfile.mkdtemp(prefix="omnivoice_frames_") + tmp_dir = output_dir or tempfile.mkdtemp(prefix="omnivoice_frames_") frames = [] # Subsample if too many timestamps @@ -240,41 +241,21 @@ async def analyse_video( Returns: VideoContext with per-segment and global visual analysis. """ - loop = asyncio.get_running_loop() - ctx = VideoContext() - - # Extract timestamps at segment midpoints - timestamps = [ - (seg.get("start", 0) + seg.get("end", 0)) / 2 - for seg in segments - ] - - # Extract frames (CPU-bound, run in pool) - frames = await loop.run_in_executor( - _analysis_pool, - _extract_keyframes, - video_path, timestamps, max_frames, + # The executor may outlive a cancelled request. Keep extraction, analysis + # and cleanup in the same worker so its files retain one lifetime owner. + return await asyncio.get_running_loop().run_in_executor( + _analysis_pool, _analyse_video_worker, video_path, segments, max_frames, ) - # Analyse each frame - for ts, frame_path in frames: - analysis = await loop.run_in_executor( - _analysis_pool, - _analyse_frame_basic, - frame_path, - ) - ctx.frame_analyses[ts] = analysis - - # Build segment-level context - ctx = _build_segment_context(ctx, segments) - - # Cleanup temp frames - for _, frame_path in frames: - try: - os.remove(frame_path) - except Exception: - pass +def _analyse_video_worker(video_path: str, segments: list[dict], max_frames: int) -> VideoContext: + ctx = VideoContext() + timestamps = [(seg.get("start", 0) + seg.get("end", 0)) / 2 for seg in segments] + with tempfile.TemporaryDirectory(prefix="omnivoice_frames_") as directory: + frames = _extract_keyframes(video_path, timestamps, max_frames, directory) + for ts, frame_path in frames: + ctx.frame_analyses[ts] = _analyse_frame_basic(frame_path) + ctx = _build_segment_context(ctx, segments) logger.info( "Video analysis complete: %d frames, global_mood=%s, global_brightness=%s", len(frames), ctx.global_mood, ctx.global_brightness, diff --git a/docs/audio-quality.md b/docs/audio-quality.md index 6e350927a..33398137b 100644 --- a/docs/audio-quality.md +++ b/docs/audio-quality.md @@ -4,6 +4,10 @@ The **Audio quality** controls apply to the next generated take. Existing files stay unchanged. Normal defaults remain 16-bit WAV, 16 sampling steps and broadcast mastering; model-specific limits still apply. +When reusing a Clone take with saved local generation settings, its WAV precision +and mastering choice are restored with its other controls. Older takes without +these settings retain the current quality selection. + The compact slider sits below the script in Clone and Design. **More options** reveals voice refinement (sampling steps), volume balancing and format guidance. The three plain-language choices are **Standard**, **For editing**, and @@ -51,7 +55,7 @@ the existing provenance watermark setting remain in effect. An engine that performs its own mastering continues to skip the app's mastering pre-stage. The final playback response uses the saved WAV itself. Streaming previews still -use PCM16; their completed take uses the selected precision. Updated remote +use PCM16; their completed take uses the selected precision. The Electron renderer schedules a short lead before playback and drains its final PCM buffer before ending the preview. After a pause in chunk delivery, a crossfade applies only while the previous chunk is still playing. Updated remote workers honor the requested precision before returning audio. Older workers or engines that only deliver PCM16 cannot recover extra detail through a larger export. OmniVoice and VoxCPM2 subprocesses negotiate float32 transport; legacy PCM16 diff --git a/docs/desktop-build.md b/docs/desktop-build.md index 9882eb19e..e28c50f64 100644 --- a/docs/desktop-build.md +++ b/docs/desktop-build.md @@ -216,3 +216,5 @@ One session per platform. Ordered by payoff: tqdm monkey-patch + `/setup/status` + SSE stream + `/setup/warmup` live and smoke-tested. Phase F skeleton committed (CI workflow, primary target only — non-arm64 rows parked). Phases A, B, D, E pending. + +Runtime download byte progress matches complete package identifiers, so concurrent downloads such as `torch` and `torchvision` retain separate received bytes and totals. Unknown package progress does not change another package’s planned size. diff --git a/docs/electron-batch.md b/docs/electron-batch.md index bf3ca6dc5..a2913a2a2 100644 --- a/docs/electron-batch.md +++ b/docs/electron-batch.md @@ -25,3 +25,5 @@ and confirmed deletion. The enqueue unit regression covers partial failure and language/voice fields. The native helper smoke verifies confined streaming, authorization replacement, rename detection and revocation. No test yet establishes real model-backed batch completion or a separate remote-machine transfer. + +Retry and deletion reserve the same job while admission or file cleanup is in progress. Concurrent actions return 409 instead of queuing a second attempt or removing its upload. Active jobs must be cancelled and finish stopping before their records and files can be deleted; settled terminal jobs retain normal deletion. diff --git a/docs/electron-connections.md b/docs/electron-connections.md index 0dd3fa802..830fa7815 100644 --- a/docs/electron-connections.md +++ b/docs/electron-connections.md @@ -45,6 +45,17 @@ device. Sharing exposes the backend only after an inline confirmation. It shows the active PIN and LAN addresses, generates QR links locally, supports a configurable share port, and controls the backend's Tailscale serve integration. These actions remain explicit and do not run during settings reads. -Remote backend configuration is owned by Electron main. Connection tests validate a VoiceStudio health response and exchange an optional server master key once for a scoped, expiring session. The renderer clears the key immediately; only the URL is persisted. Main injects the session into production and development HTTP proxy traffic, native watch-folder uploads, and path-bound dictation WebSocket tickets. Switching back to the local backend is always available. +Compare voices reports missing speech when a generated preview contains only +part of the requested phrase. Its warning includes the omitted text when +available, or the number of omitted parts. Surviving audio remains playable, +and comparison continues with the other voice. Healthy previews stay quiet; +leaving the comparison suppresses notices from cancelled results. + +Remote backend configuration is owned by Electron main. Connection tests validate a VoiceStudio health response and exchange an optional server master key once for a scoped, expiring session. The renderer clears the key immediately; only the URL is persisted. Main injects the session into production and development HTTP proxy traffic, native watch-folder uploads, and path-bound dictation WebSocket tickets. Connection probes, session exchanges and WebSocket-ticket requests keep their per-request deadline active through successful JSON-body reads, so receiving headers cannot leave the action waiting indefinitely. HTTP and authentication failures retain their existing classifications. Switching back to the local backend is always available. Remote-worker routing was exercised against an Ubuntu 26.04 WSL worker with an RTX 4090. A real profile-backed TTS request returned a WAV with `X-OmniVoice-Routing: remote`; stopping the worker changed the same selected target to an explicit local fallback, and a second request returned `X-OmniVoice-Routing: local_fallback`. Restarting the worker restored remote readiness without re-enrollment. Dubbing and Batch also completed real multi-segment worker tasks: the Batch proof returned two exact indexed, non-silent mono WAVs at 24 kHz in one committed bundle. The selected target was returned to Local after verification. + +Remote backend URLs may include a reverse-proxy path prefix. HTTP connection +checks and WebSocket handshakes both retain that prefix, including nested paths +and trailing-slash normalization. Ticket requests bind the backend's canonical +WebSocket route; only the opaque ticket is added to the connection URL. diff --git a/docs/electron-dubbing.md b/docs/electron-dubbing.md index c82d6a352..c37dcdd64 100644 --- a/docs/electron-dubbing.md +++ b/docs/electron-dubbing.md @@ -1,5 +1,7 @@ # Electron dubbing workspace +Visual-context analysis owns its extracted frames in one background worker. Completed or failed analysis removes its entire temporary frame directory. If the request is cancelled while native work is running, cleanup stays with that worker and occurs when it finishes, so cancellation neither deletes active frames nor leaves them behind after completion. + The idle workspace includes an original/dubbed demo comparison with compact player controls. Sync playheads aligns positions without starting both videos. Sample transcript edits are retained per language while the demo is mounted; @@ -7,7 +9,10 @@ they do not regenerate the prerecorded audio. Edit on the dubbed card imports that sample video into the normal upload/transcription and editing workflow. Open Dub from the cloning sidebar or command search. Upload or drop audio/video, or explicitly submit a video URL; -preparation completes before transcription starts. The editor shows source text, +preparation completes before transcription starts. +On Apple Silicon, forced alignment retries on CPU if its alignment model cannot +load on MPS. If neither device can load the language aligner, transcription keeps +the existing segment timings. The editor shows source text, editable translated text, and per-segment voice/timing controls. Translation uses the selected Settings > Models > Translation provider. Choose a target language, translate, review the text, then generate. Completed tracks can be previewed and @@ -55,7 +60,11 @@ Interrupted preparation/generation offers Resume, which reads the existing task and replays its stream; generation is never resubmitted just because the UI reloaded. Interrupted transcription offers an explicit Retry against the existing prepared media, without uploading or preparing the source again. ASR restarts from the -beginning because its backend stream is request-scoped, not a replayable task. Batch language runs and advanced QC controls remain in `electron/PARITY.md`. +beginning because its backend stream is request-scoped, not a replayable task. +When only some ASR segments have word timings, dubbing retains text from the +remaining segments by estimating word spans within their segment bounds. +Existing precise word timings remain unchanged; fully untimed transcripts keep +the chunk-based fallback. Batch language runs and advanced QC controls remain in `electron/PARITY.md`. Verification: `node electron/tests/dub-smoke.mjs` against the development renderer. The test mocks backend jobs and never uploads or generates user media. Optionally @@ -87,6 +96,10 @@ URL import runs only after clicking Ingest; it uses the backend's existing yt-dl pipeline. Explicit cookies.txt selection is available under URL sign-in options; optional caption downloads are available. Translation quality uses the existing backend Fast, Autofit and Cinematic modes. +Japanese script checks accept kana and Han letters, including kanji-heavy place +names and supplementary Han characters, in both ordinary translation and the +Autofit/Cinematic quality pass. Latin-only responses still fail the Japanese +script check. The choice persists in the working draft and saved project (`translateQuality`), including legacy project imports. New media preserves the user's quality choice. If the backend reports that no LLM is configured, the UI selects Fast and shows @@ -109,7 +122,27 @@ Export options expand inside the existing sidebar. Users can select included vid tracks and the default track, background mixing, burned subtitles, dual layout and karaoke (disabled with dual layout). Audio supports WAV or MP3 with bitrate choice; SRT/VTT/ASS sidecars and per-language stem/segment ZIPs use the existing backend. -Each download is explicit and targets the selected language. Export errors retain +Each download is explicit and targets the selected language. +Advanced QC compares recognized audio with the translation text for the selected +track language, including when another language was generated more recently. +Newly generated tracks save source timing by stable segment identity, so QC +matches the correct line even after another language reorders the job. Smart Fit +cues and stretch-video plans are applied for the selected track when available. +Older source snapshots without segment identities cannot prove line ownership +for multiple lines. QC stops before recognition with a regeneration request when +those times are ambiguous, instead of scoring against another track's windows. +Complete Smart Fit cues with matching unique IDs still establish the final +timeline; a single line with one matching selected-track text ID also establishes +ownership. Jobs with no saved source snapshot retain their existing fallback. +Regenerate an ambiguous track to save identity-aware source timing. Regeneration +fills missing segment IDs from the validated current-render manifest when the +request omits IDs, preserving explicit request IDs and existing stable IDs. +A fallback ID is never assigned when it would collide with a retained or explicit +ID. Mixed-ID requests that remain ambiguous need explicit unique IDs; QC stops +before recognition for partial or duplicate source identities. Synchronization +rejects duplicate language-text keys before loading a backend or starting regeneration, preserving the previous WAV and saved metadata. Each existing segment is matched at most once, so a partial-ID request can use a current-render fallback ID without reusing another line’s source text or speaker. +QC annotations +preserve saved source times, text, and track settings. Export errors retain all choices for retry. Native save filters match the encoded file format. Browser fixtures verify MP3/SRT downloads, query options and failed-export retry; unit tests cover video/package parameters. Real rendered exports, batch presets @@ -371,3 +404,8 @@ audio if a later import fails validation or runs out of space. Cancelling an upload waits for its copy worker to stop before closing the input and clearing the reserved job, so the same upload can be retried safely. + +Advanced QC keeps the selected track's text and timing fixed while recognition +runs. If the track, transcript, timing, or audio changes during that pass, QC +asks you to run it again instead of publishing stale scores. Deleting the job +during QC also discards the result and keeps it out of history. diff --git a/docs/electron-gallery.md b/docs/electron-gallery.md index 820d26cb0..76a88002e 100644 --- a/docs/electron-gallery.md +++ b/docs/electron-gallery.md @@ -28,3 +28,5 @@ Separate real smokes cover upload, trim, persona import/export, and profile materialization without retaining disposable profiles. The packaged Ubuntu app also completed profile creation, native `.ovsvoice` Save As, bundle inspection and cleanup against a reused installed runtime without downloading anything. + +Inline trimming decodes the source directly with Web Audio at 22,050 Hz. It does not wait for a separate media-element duration probe; files still need to be supported by the desktop audio decoder. diff --git a/docs/electron-longform.md b/docs/electron-longform.md index dc9f18b36..fbcdf19e1 100644 --- a/docs/electron-longform.md +++ b/docs/electron-longform.md @@ -4,11 +4,11 @@ Stories and Audiobook are available from the sidebar and command search. Each ke Plain-text audiobook imports recognize chapter-title lines with LF, CRLF or CR line endings. Manuscripts with existing Markdown H1 headings retain their explicit structure and original text. -Both editors use the shared longform backend, real chapter progress, assembly status, and an explicit Stop action. Stop aborts the HTTP stream so the backend stops scheduling chapters. A truncated stream cannot replace the last successful output. Interrupted server manifests can be resumed explicitly without sending the edited manuscript again. Navigation leaves a render running; a renderer restart disconnects it and recovery uses the server manifest inventory. +Both editors use the shared longform backend, real chapter progress, assembly status, and an explicit Stop action. Stop aborts the HTTP stream so the backend stops scheduling chapters. Transport cancellation closes the active job as cancelled while retaining its resume checkpoint. Finite setup errors close the job as failed before the error response finishes. A truncated stream cannot replace the last successful output. Interrupted server manifests can be resumed explicitly without sending the edited manuscript again. Navigation leaves a render running; a renderer restart disconnects it and recovery uses the server manifest inventory. Playback and export remain independent of generation. Vidstack plays the output; long-form audio uses a seek bar without decoding an entire book into a waveform. Play waits until the native media provider is ready. MP3 and M4B are supported by the existing backend. The native export dialog saves a local copy. A finished render also offers a chapter cue sheet — a `.txt` of `HH:MM:SSTitle`, one line per chapter — for show notes, podcast platforms and players that do not read the chapters M4B embeds and MP3 cannot carry. Its timestamps are summed from the exact per-chapter milliseconds the backend writes into the M4B chapters, over only the chapters that rendered, so they match the embedded chapter starts (shown in whole seconds, hours unbounded); a chapter that failed contributes no line and no elapsed time. The file is saved through the native save dialog. -Draft writes reuse the coalesced persistence helper and flush on lifecycle events. Working drafts currently use browser storage. Named Stories and Audiobook projects use the shared native IndexedDB adapter in a separate Electron database. Saves await the committed transaction, concurrent writes are serialized, and opening or deleting a project requires confirmation. Projects include script/lines, voices, settings and the last output reference. Storage failures are surfaced. Book details include author, narrator, year, genre, description and an uploaded JPEG/PNG cover. Loudness offers off, ACX and podcast presets. Audiobook pronunciation rows reach the existing lexicon engine; duplicate words are flagged before rendering. These settings persist with the draft. Full cast management and other advanced controls remain tracked in `electron/PARITY.md`. +Draft writes reuse the coalesced persistence helper and flush on lifecycle events. Working drafts currently use browser storage. Named Stories and Audiobook projects use the shared native IndexedDB adapter in a separate Electron database. Saves await the committed transaction, concurrent writes are serialized, and opening or deleting a project requires confirmation. Projects include script/lines, voices, settings and the last output reference. Storage failures are surfaced. Book details include author, narrator, year, genre, description and an uploaded JPEG/PNG cover. Exported M4B and MP3 descriptions preserve paragraphs from LF, CRLF or CR line breaks. Loudness offers off, ACX and podcast presets. Audiobook pronunciation rows reach the existing lexicon engine; duplicate words are flagged before rendering. These settings persist with the draft. Full cast management and other advanced controls remain tracked in `electron/PARITY.md`. `electron/tests/longform-smoke.mjs` verifies the main flows with mocked generation and real audio playback. Runtime tests cover stop, duplicate prevention, truncated responses, explicit resume and shared Stories compilation. Real installed-model verification passed for one English chapter: MP3 rendering, cached WAV chapter audition, and Electron-driven M4B rendering/playback. Multi-voice, multi-chapter, cancellation/recovery and other engines still require real runtime coverage. @@ -16,6 +16,8 @@ Script voice tags now expose saved-profile assignments in the Cast section. Only Audiobook Preview plan uses the backend parser and exposes per-chapter auditions. Auditions send the same voice, cast, language and pronunciation inputs as full renders, warming the shared cache. Editing those inputs clears stale auditions. Preview requests are aborted on navigation, never replace a full render, and expose their own Vidstack playback. +Local chapter and segment caches include an explicitly resolved synthesis language, so changing it renders new audio even when the normalized text stays identical. Autodetected-language caches keep their existing keys. + Production overrides now expose synthesis steps, guidance, sampling temperatures, postprocessing, seed and repeat variation. Emotion controls appear when the active engine advertises support. Reset restores the shared Tauri defaults; untouched fields are omitted from requests. The extracted `longformOverrides` helper is used by both apps, and chapter auditions carry the same overrides as full renders. Seamless-join controls (gap between lines, gap between paragraphs, trim engine silence) live in the same panel; untouched or reset controls show the preserved server defaults (zero gaps, trimming off). Set gaps and enable trimming explicitly for seamless joins; see `docs/expressive-speech.md`. Run `electron/tests/longform-live.mjs` with `VOICESTUDIO_LIVE_PROFILE` set to a local saved profile for an explicit real-render check. It verifies the active model is already installed before synthesis. The real run exposed missing JSON request headers in the shared client and array-shaped failed-chapter results; regression tests now cover both. diff --git a/docs/electron-storage.md b/docs/electron-storage.md index ca83b76a5..b8b5d2c52 100644 --- a/docs/electron-storage.md +++ b/docs/electron-storage.md @@ -2,15 +2,21 @@ Settings > Storage reads the existing cached disk report, shows volume use/free space, model cache, application data, engine environments and temporary files, and marks incomplete scans explicitly. Largest models and data subtotals expand inline. Warning formatting and byte formatting are shared with Tauri. +Storage scan budgets are checked between files in large flat directories and between loose application-data entries. A scan that exhausts its budget reports partial bytes and a timeout warning, including an incomplete Other subtotal. A single operating-system filesystem call can still take longer than the budget. + Open folder uses Electron's native reveal bridge, with the existing backend reveal route for browser development. Model and log links open their existing management views. Temporary-file cleanup requires explicit confirmation with the running-job warning. A partial deletion reports failure instead of claiming all files were cleared, and refreshes usage. Opening the page never deletes anything. Database backup status displays the latest pre-migration snapshot and date. This is database backup information, not a claim that source media and generated files are backed up. History retention reuses the confirmed cap editor from Privacy. +Locking a profile to another take stores a new reference filename, so longform caches recognize the changed voice even when the take text and seed are unchanged. The replacement commits before the response returns; prior versions are retained as described below, and a failed update keeps the previous take usable. + The model cache location uses Electron's native directory picker. Main verifies the directory is writable, stores a one-shot `models_dir` capability in the backend data directory, and returns only its token to the renderer. The backend consumes that token when persisting `OMNIVOICE_CACHE_DIR`; raw host paths never cross the HTTP boundary. Reset uses the same capability flow with an empty path, and either change takes effect after restart. +Model folder names containing a hash after a space, apostrophes or backslashes are quoted and escaped in the durable environment so the startup loader retains the chosen directory. Existing plain values remain supported. + Reset & remove provides four common presets and an advanced per-scope list with measured disk sizes. The renderer owns UI preferences, history, drafts and IndexedDB projects. Main owns settings, generated content, engines, tools, models, caches and logs; it accepts only known scopes from the trusted main frame, rejects remote-backend resets and unsafe roots, stops the local backend before deletion and restarts it afterward. Removing voices/projects/audio requires typing the localized confirmation word. Shared Hugging Face caches carry a separate warning. Pending draft writes are suspended before reload so deleted work cannot be recreated by pagehide persistence. -The application data location uses Electron's native folder picker and a main-process one-shot authorization. Relocation stops the managed local backend, copies and verifies every file into an empty destination, atomically persists `OMNIVOICE_DATA_DIR` in the shared durable environment, restarts the backend and confirms `/system/info` advertises the new path before removing the old copy. A failed copy or activation restores the previous setting, deletes only the verified destination and restarts from the original folder. If final old-folder cleanup fails, the new location stays active and the UI tells the user that the old copy can be removed manually. Remote and separately started backends are rejected because Electron cannot freeze their writes safely. +The application data location uses Electron's native folder picker and a main-process one-shot authorization. Relocation stops the managed local backend, copies and verifies every file into an empty destination (including an existing empty folder selected in the picker), atomically persists `OMNIVOICE_DATA_DIR` in the shared durable environment, restarts the backend and confirms `/system/info` advertises the new path before removing the old copy. A destination that gains files during copying is refused without deleting those files. A failed copy or activation restores the previous setting, deletes only the verified destination and restarts from the original folder. If final old-folder cleanup fails, the new location stays active and the UI tells the user that the old copy can be removed manually. Remote and separately started backends are rejected because Electron cannot freeze their writes safely. Remove all data scans the backend data root, Electron runtime/configuration, logs, durable environment and model cache with real sizes. Shared Hugging Face caches remain an explicit opt-in. After typed confirmation, main rescans and validates every root, stops the backend and hands the exact plan to the signed desktop helper. The helper canonicalizes every path, waits for Electron to exit, then removes the locked Chromium/runtime tree without following a path alias outside VoiceStudio-owned data. A failed helper launch restores the backend and keeps the app open; a successful handoff quits immediately. Removing the installed application binary remains the operating system's normal uninstall step. @@ -19,3 +25,5 @@ Remove all data scans the backend data root, Electron runtime/configuration, log Interrupted sidecar installs without an environment remain included in application data. Completed sidecar environments, checkouts and weights are counted once in the engine category. An unreadable engine directory or entry produces an incomplete report and a warning; unavailable bytes are not presented as a complete empty footprint. + +Re-locking keeps prior immutable locked-reference clips for renders that already captured their filenames. These clips remain in the voices folder for the lifetime of the profile; explicit profile deletion reclaims its generated versions after committing the record deletion, while preserving versions still referenced by another profile. Deletion returns a conflict while a longform render still holds a cached reference that would be removed; the profile record and files remain intact, and deletion can be retried once all those render workers finish. Unlocking still succeeds while a render holds the current locked clip, but retains that immutable clip until explicit profile deletion can safely reclaim it. Unlocking also preserves clips referenced by any profile, including the unlocked profile’s other audio fields; an unused, unshared current clip is removed after the unlock commits. Repeated locks can therefore use additional local storage until the profile is deleted. diff --git a/docs/electron-transcriptions.md b/docs/electron-transcriptions.md index 402a82f9e..dddd9c998 100644 --- a/docs/electron-transcriptions.md +++ b/docs/electron-transcriptions.md @@ -59,6 +59,11 @@ Live Dictation now streams mono 16 kHz PCM through the shared Tauri AudioWorklet and anti-alias capture graph. Record remains the separate clip-preview workflow. Dictation checks the selected model is enabled and installed before requesting a microphone; pause, stop and cancel preserve explicit session boundaries. +New history rows keep distinct numeric identities even when utterances arrive +together or the wall clock moves backwards. Deleting one newly saved row preserves the +other saved utterances; existing history records and the 200-entry limit remain +unchanged. + Live utterances enter history once, including repeated speech, and EOF summaries avoid duplicate entries. Legacy fallback finals also finish normally. Failed history writes retain visible text for copying. Leaving the page releases capture. @@ -77,7 +82,9 @@ by a dedicated, non-activating recorder window. It captures the output target before revealing the recorder, queues startup/stop events until registration, and uses the shared native delivery helper. The main app frame cannot invoke recorder-only output IPC. Cancellation/navigation/crash invalidate pending work; -sequence numbers reject duplicate deliveries. Clipboard fallback retains the +sequence numbers reject duplicate deliveries. A late WebSocket URL or remote +ticket from a cancelled start cannot reconnect or replace the next dictation +session. Clipboard fallback retains the complete transcript and is labeled as copied, not inserted. Native Windows smoke checks cover helper acceptance, pause/resume, no-speech, diff --git a/docs/electron-workflows.md b/docs/electron-workflows.md index 98825460c..22c8d0a15 100644 --- a/docs/electron-workflows.md +++ b/docs/electron-workflows.md @@ -49,3 +49,5 @@ downloaded automatically. A branch that goes directly from a Condition to End keeps the incoming text as its exportable result. Conditions have one phrase editor; the generic Instructions field is hidden for those steps. + +Native saves stage the complete replacement in the selected destination directory before replacing an existing export. A write or replacement failure leaves the previous file intact. Working symlink targets are resolved before a non-truncating write-authorization open; permission bits are read from that opened file descriptor. A changed destination inode or newly appeared file observed before replacement is refused. This identity check is not an atomic compare-and-swap and does not protect against all concurrent changes in a hostile directory. Existing POSIX permission bits and working symbolic links are preserved; a dangling symbolic link is left unchanged and reports a filesystem error. Atomic replacement requires write permission on the parent directory, even when the existing file is writable. It replaces the original file inode: existing ACLs, ownership, and relationships to other hard links are not preserved. This does not provide a power-loss recovery guarantee. diff --git a/docs/expressive-speech.md b/docs/expressive-speech.md index 7666f56ac..c44608824 100644 --- a/docs/expressive-speech.md +++ b/docs/expressive-speech.md @@ -64,6 +64,10 @@ for your engine below. Pronunciation dictionary matching uses Unicode case-insensitive literal matches. Each matched term uses its own respelling; distinct terms such as Straße and STRASSE can have different respellings. Longer terms win overlaps, and later equal-length case variants retain precedence. +Pronunciation scopes accept picker names such as Spanish, their bundled ISO IDs such as `es` or `kbt`, and regional forms such as `es-MX`; names resolve to the existing picker IDs. Explicit Chinese script scopes (`cmn-Hans`/`cmn-Hant` and `zho-Hans`/`zho-Hant`) remain distinct when saved or exported. A script-tagged request uses its own script scope before its base `cmn` or `zho` fallback, then the global scope; underscore spellings of these known script tags are equivalent. Unknown suffixes on `cmn`/`zho` remain literal scopes. Other alternate codes absent from the bundled engine map remain literal scopes (for example `spa` is not remapped to `es`). Global scopes still apply with Auto. Existing ambiguous truncated codes (for example `po`) keep their literal meaning: edit them to the intended name or ISO code rather than relying on an automatic migration. + +Dictionary lists, previews, synthesis and exports use creation time, then insertion order for tied timestamps. A bulk import therefore keeps its authored entry order, and exporting/restoring the dictionary preserves duplicate and case-variant precedence. + ### Default engine (VoiceStudio) **Non-verbal tags.** The bundled model natively tokenizes 13 reaction tags diff --git a/docs/mcp.md b/docs/mcp.md index 55d8026c7..5d37b3cac 100644 --- a/docs/mcp.md +++ b/docs/mcp.md @@ -198,6 +198,8 @@ in Scarlett". Voice resolution precedence on every `generate_speech` call: Manage bindings over the loopback REST API (the Settings UI uses these): +Partial binding updates preserve omitted fields, including concurrent edits from different Settings clients. An empty profile or engine clears that field; an omitted value preserves it. Concurrent creation of the same client binding merges the supplied fields atomically. + ```bash # list curl localhost:3900/api/mcp/bindings diff --git a/docs/voice-design.md b/docs/voice-design.md index e960cb77b..eb2a11361 100644 --- a/docs/voice-design.md +++ b/docs/voice-design.md @@ -3,6 +3,11 @@ Voice Design mode lets you describe the desired speaker through speaker attributes (`instruct` parameter) — no reference audio needed. The model generates a matching voice on the fly. +Saving a design does not load or download a voice engine. If the engine is +already loaded, Save also renders its identity sample. Otherwise the design +is saved with its attributes and the sample is generated when you preview it. +Until that preview exists, synthesis uses the saved attributes directly. + ## Quick Example ```python diff --git a/electron/src/main/atomic-export.test.ts b/electron/src/main/atomic-export.test.ts new file mode 100644 index 000000000..0b899fbb1 --- /dev/null +++ b/electron/src/main/atomic-export.test.ts @@ -0,0 +1,229 @@ +// @vitest-environment node +import { constants } from 'node:fs'; +import { chmod, lstat, mkdtemp, readFile, readlink, readdir, rm, stat, symlink, writeFile } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { afterEach, beforeEach, expect, it, vi } from 'vitest'; + +const controls = vi.hoisted(() => ({ + handlers: new Map Promise>(), + destination: '', failWrite: false, failRename: false, swapAfterProbeOpen: false, swapWithSymlink: false, appearWhileStaging: false, +})); +vi.mock('electron', () => ({ + app: {}, + BrowserWindow: { fromWebContents: () => owner }, + dialog: { showSaveDialog: async () => ({ canceled: false, filePath: controls.destination }) }, + ipcMain: { handle: (name: string, handler: (event: unknown, request: unknown) => Promise) => controls.handlers.set(name, handler), on: () => {} }, + net: { fetch: async () => new Response(new Uint8Array([9, 8, 7, 6])) }, + shell: {}, systemPreferences: {}, +})); +vi.mock('./uninstall-cleanup', () => ({ scheduleUninstallCleanup: () => {} })); +vi.mock('./site-browser', () => ({ registerSiteBrowser: () => {} })); +vi.mock('./pro-license', () => ({ activateProLicense: () => {}, deactivateProLicense: () => {}, proLicenseStatus: () => {} })); +vi.mock('node:fs/promises', async () => { + const real = await vi.importActual('node:fs/promises'); + return { + ...real, + writeFile: async (...args: Parameters) => { + if (controls.failWrite) { + await real.writeFile(args[0], new Uint8Array([9]), args[2]); + throw Object.assign(new Error('injected partial write failure'), { code: 'EIO' }); + } + return real.writeFile(...args); + }, + open: async (...args: Parameters) => { + const file = await real.open(...args); + if (controls.swapAfterProbeOpen && args[1] === constants.O_WRONLY && (args[0] === controls.destination || args[0] === await real.realpath(controls.destination))) { + await real.rename(controls.destination, join(directory, 'authorized-original.wav')); + if (controls.swapWithSymlink) { + const unrelated = join(directory, 'unrelated.wav'); + await real.writeFile(unrelated, 'unrelated target', { mode: 0o444 }); + await real.chmod(unrelated, 0o444); + await real.symlink('unrelated.wav', controls.destination); + } else { + await real.writeFile(controls.destination, 'unrelated replacement', { mode: 0o644 }); + await real.chmod(controls.destination, 0o644); + } + } + const write = file.writeFile.bind(file); + vi.spyOn(file, 'writeFile').mockImplementation(async (...data) => { + if (controls.failWrite) { + await write(new Uint8Array([9])); + throw Object.assign(new Error('injected partial write failure'), { code: 'EIO' }); + } + const result = await write(...data); + if (controls.appearWhileStaging) { + await real.writeFile(controls.destination, 'concurrent new export', { mode: 0o600 }); + controls.appearWhileStaging = false; + } + return result; + }); + return file; + }, + rename: async (...args: Parameters) => { + if (controls.failRename) throw Object.assign(new Error('injected rename failure'), { code: 'EACCES' }); + return real.rename(...args); + }, + }; +}); +import { CHANNELS, registerIpc } from './ipc'; + +const frame = { url: 'app://voicestudio/index.html' }; +const owner = { webContents: { mainFrame: frame } }; +const event = { sender: owner.webContents, senderFrame: frame }; +let directory: string; +beforeEach(async () => { + controls.failWrite = false; controls.failRename = false; controls.swapAfterProbeOpen = false; controls.swapWithSymlink = false; controls.appearWhileStaging = false; controls.handlers.clear(); + directory = await mkdtemp(join(tmpdir(), 'voicestudio-export-')); + controls.destination = join(directory, 'saved.wav'); + await writeFile(controls.destination, 'previous complete export', { mode: 0o600 }); + registerIpc({ baseUrl: 'http://127.0.0.1:8000', requestHeaders: () => ({}), subscribe: () => {} } as never, () => owner as never); +}); +afterEach(async () => { controls.failWrite = false; controls.failRename = false; await rm(directory, { recursive: true, force: true }); vi.restoreAllMocks(); }); +const requests = [ + [CHANNELS.filesSaveData, { suggestedName: 'saved.wav', data: new Uint8Array([9, 8, 7, 6]) }], + [CHANNELS.filesSaveAudio, { suggestedName: 'saved.wav', url: 'http://127.0.0.1:8000/audio/source.wav' }], +] as const; + +it.each(requests)('preserves the old export after a partial write in %s', async (channel, request) => { + controls.failWrite = true; + await expect(controls.handlers.get(channel)!(event, request)).rejects.toThrow('partial write'); + expect(await readFile(controls.destination, 'utf8')).toBe('previous complete export'); + expect(await readdir(directory)).toEqual(['saved.wav']); +}); + +it.each(requests)('preserves the old export after a failed replacement in %s', async (channel, request) => { + controls.failRename = true; + await expect(controls.handlers.get(channel)!(event, request)).rejects.toThrow('rename failure'); + expect(await readFile(controls.destination, 'utf8')).toBe('previous complete export'); + expect(await readdir(directory)).toEqual(['saved.wav']); +}); + +it.each(requests)('replaces the complete export and preserves its permissions in %s', async (channel, request) => { + await expect(controls.handlers.get(channel)!(event, request)).resolves.toEqual({ canceled: false, path: controls.destination }); + expect([...await readFile(controls.destination)]).toEqual([9, 8, 7, 6]); + if (process.platform !== 'win32') expect((await stat(controls.destination)).mode & 0o777).toBe(0o600); + expect(await readdir(directory)).toEqual(['saved.wav']); +}); + +async function supportsSymlink(target: string, link: string, context: { skip(): void }): Promise { + try { await symlink(target, link); return true; } + catch (error) { + if (process.platform === 'win32' && ['EPERM', 'EACCES'].includes((error as NodeJS.ErrnoException).code || '')) { + context.skip(); + return false; + } + throw error; + } +} + +it('preserves an existing symbolic link while replacing its target', async (context) => { + const target = controls.destination; + const link = join(directory, 'linked.wav'); + if (!await supportsSymlink('saved.wav', link, context)) return; + controls.destination = link; + await controls.handlers.get(CHANNELS.filesSaveData)!(event, requests[0][1]); + expect([...await readFile(target)]).toEqual([9, 8, 7, 6]); + expect([...await readFile(link)]).toEqual([9, 8, 7, 6]); + expect(await readdir(directory)).toEqual(['linked.wav', 'saved.wav']); +}); + + +it.each(requests)('creates a new export without leaving staging files in %s', async (channel, request) => { + await rm(controls.destination); + await controls.handlers.get(channel)!(event, request); + expect([...await readFile(controls.destination)]).toEqual([9, 8, 7, 6]); + expect(await readdir(directory)).toEqual(['saved.wav']); +}); + + +it.each(requests)('preserves an existing mode despite a restrictive umask in %s', async (channel, request) => { + if (process.platform === 'win32') return; // Windows does not implement POSIX mode bits. + await chmod(controls.destination, 0o644); + const previous = process.umask(0o077); + try { + await controls.handlers.get(channel)!(event, request); + expect((await stat(controls.destination)).mode & 0o777).toBe(0o644); + } finally { process.umask(previous); } +}); + +it('preserves a dangling symbolic link when its destination cannot be resolved', async (context) => { + const link = join(directory, 'dangling.wav'); + if (!await supportsSymlink('missing.wav', link, context)) return; + controls.destination = link; + await expect(controls.handlers.get(CHANNELS.filesSaveData)!(event, requests[0][1])).rejects.toMatchObject({code: 'ENOENT'}); + expect((await lstat(link)).isSymbolicLink()).toBe(true); + expect(await readlink(link)).toBe('missing.wav'); + expect(await readdir(directory)).toEqual(['dangling.wav', 'saved.wav']); +}); + + +it.skipIf(process.platform === 'win32' || process.getuid?.() === 0).each(requests)('rejects a read-only existing export without replacing it in %s', async (channel, request) => { + await chmod(controls.destination, 0o444); + await expect(controls.handlers.get(channel)!(event, request)).rejects.toMatchObject({ code: expect.stringMatching(/^(EACCES|EPERM)$/) }); + expect(await readFile(controls.destination, 'utf8')).toBe('previous complete export'); + expect((await stat(controls.destination)).mode & 0o777).toBe(0o444); + expect(await readdir(directory)).toEqual(['saved.wav']); +}); + +it('rejects a read-only symlink target without replacing the target or link', async (context) => { + if (process.platform === 'win32' || process.getuid?.() === 0) { + context.skip(); + return; + } + const target = controls.destination; + const link = join(directory, 'linked.wav'); + await symlink('saved.wav', link); + await chmod(target, 0o444); + controls.destination = link; + await expect(controls.handlers.get(CHANNELS.filesSaveData)!(event, requests[0][1])).rejects.toMatchObject({ code: expect.stringMatching(/^(EACCES|EPERM)$/) }); + expect(await readFile(target, 'utf8')).toBe('previous complete export'); + expect(await readlink(link)).toBe('saved.wav'); + expect(await readdir(directory)).toEqual(['linked.wav', 'saved.wav']); +}); + + +it.skipIf(process.platform === 'win32' || process.getuid?.() === 0).each(requests)('leaves the existing export intact when the parent cannot stage a replacement in %s', async (channel, request) => { + await chmod(directory, 0o500); + try { + // The existing inode remains writable even though its directory is not. + await writeFile(controls.destination, 'previous complete export'); + await expect(controls.handlers.get(channel)!(event, request)).rejects.toMatchObject({ code: expect.stringMatching(/^(EACCES|EPERM)$/) }); + expect(await readFile(controls.destination, 'utf8')).toBe('previous complete export'); + expect(await readdir(directory)).toEqual(['saved.wav']); + } finally { + await chmod(directory, 0o700); + } +}); + + +it.skipIf(process.platform === 'win32').each(requests)('refuses a changed destination inode without overwriting either file in %s', async (channel, request) => { + controls.swapAfterProbeOpen = true; + await expect(controls.handlers.get(channel)!(event, request)).rejects.toMatchObject({ code: 'ESTALE' }); + expect((await stat(controls.destination)).mode & 0o777).toBe(0o644); + expect(await readFile(controls.destination, 'utf8')).toBe('unrelated replacement'); + expect(await readFile(join(directory, 'authorized-original.wav'), 'utf8')).toBe('previous complete export'); + expect(await readdir(directory)).toEqual(['authorized-original.wav', 'saved.wav']); +}); + + +it.skipIf(process.platform === 'win32').each(requests)('refuses a symlink substituted after authorization without overwriting its target in %s', async (channel, request) => { + controls.swapAfterProbeOpen = true; + controls.swapWithSymlink = true; + await expect(controls.handlers.get(channel)!(event, request)).rejects.toMatchObject({ code: 'ESTALE' }); + expect(await readFile(join(directory, 'unrelated.wav'), 'utf8')).toBe('unrelated target'); + expect((await stat(join(directory, 'unrelated.wav'))).mode & 0o777).toBe(0o444); + expect(await readFile(join(directory, 'authorized-original.wav'), 'utf8')).toBe('previous complete export'); + expect((await lstat(controls.destination)).isSymbolicLink()).toBe(true); + expect(await readlink(controls.destination)).toBe('unrelated.wav'); + expect(await readdir(directory)).toEqual(['authorized-original.wav', 'saved.wav', 'unrelated.wav']); +}); + + +it.each(requests)('preserves a destination that appears while a new export is staged in %s', async (channel, request) => { + await rm(controls.destination); + controls.appearWhileStaging = true; + await expect(controls.handlers.get(channel)!(event, request)).rejects.toMatchObject({ code: 'EEXIST' }); + expect(await readFile(controls.destination, 'utf8')).toBe('concurrent new export'); + expect(await readdir(directory)).toEqual(['saved.wav']); +}); diff --git a/electron/src/main/atomic-export.ts b/electron/src/main/atomic-export.ts new file mode 100644 index 000000000..bfc0e309f --- /dev/null +++ b/electron/src/main/atomic-export.ts @@ -0,0 +1,69 @@ +import { randomUUID } from 'node:crypto'; +import { constants, type BigIntStats } from 'node:fs'; +import { lstat, open, realpath, rename, rm } from 'node:fs/promises'; +import { dirname, join } from 'node:path'; + +/** Keep a selected existing export intact until all replacement bytes are written. */ +export async function writeExportAtomically(path: string, data: Uint8Array): Promise { + // Bind symlink resolution before authorizing the destination. Resolving the + // selected path again after open could follow a newly substituted symlink. + const destination = await realpath(path).catch((error: NodeJS.ErrnoException) => { + if (error.code === 'ENOENT') return path; + throw error; + }); + let missing: NodeJS.ErrnoException | undefined; + const existing = await open(destination, constants.O_WRONLY).catch((error: NodeJS.ErrnoException) => { + if (error.code !== 'ENOENT') throw error; + missing = error; + return null; + }); + let authorized: BigIntStats | undefined; + let mode = 0o666; + if (existing) { + try { + authorized = await existing.stat({ bigint: true }); + mode = Number(authorized.mode & 0o777n); + } finally { + await existing.close(); + } + } else { + const present = await lstat(path).catch((error: NodeJS.ErrnoException) => { + if (error.code === 'ENOENT') return null; + throw error; + }); + // A dangling symlink (or a newly appeared path) must not be overwritten + // after the failed authorization open. + if (present) throw missing; + } + const temporary = join(dirname(destination), `.voicestudio-${randomUUID()}.tmp`); + // Exclusive creation owns this temporary file, even if a later write fails. + const file = await open(temporary, 'wx', existing ? 0o600 : 0o666); + try { + try { + await file.writeFile(data); + // open() applies umask; chmod restores the existing mode exactly. + if (existing) await file.chmod(mode); + await file.sync(); + } finally { + await file.close(); + } + const current = await lstat(destination, { bigint: true }).catch((error: NodeJS.ErrnoException) => { + if (error.code === 'ENOENT') return null; + throw error; + }); + if (authorized) { + // Refuse observed substitutions instead of copying the selected export + // over another inode or following a symlink we did not authorize. + if (!current || current.isSymbolicLink() || current.dev !== authorized.dev || current.ino !== authorized.ino) { + throw Object.assign(new Error('ESTALE'), { code: 'ESTALE' }); + } + } else if (current) { + throw Object.assign(new Error('EEXIST'), { code: 'EEXIST' }); + } + // The identity check is not an atomic compare-and-swap with rename. A + // concurrently modified hostile directory needs native OS protection. + await rename(temporary, destination); + } finally { + await rm(temporary, { force: true }); + } +} diff --git a/electron/src/main/data-relocation.test.ts b/electron/src/main/data-relocation.test.ts index eed31855f..e89e2d380 100644 --- a/electron/src/main/data-relocation.test.ts +++ b/electron/src/main/data-relocation.test.ts @@ -1,5 +1,5 @@ // @vitest-environment node -import { existsSync } from 'node:fs'; +import { existsSync, readdirSync, writeFileSync } from 'node:fs'; import { mkdtemp, mkdir, readFile, realpath, rm, writeFile } from 'node:fs/promises'; import { tmpdir } from 'node:os'; import { dirname, join } from 'node:path'; @@ -140,3 +140,49 @@ it('records nothing without a longform cache and is idempotent', async () => { ); expect(roots).toEqual([join(source, 'voices')]); }); + +it('relocates data into an existing empty native-picked directory', async () => { + const { root, source, target } = await fixture(); + await mkdir(target); + const env = join(root, 'config', 'env'); + const stages: string[] = []; + const result = await relocateDataDirectory(source, target, { + environmentPath: env, + stop: async () => undefined, + start: async () => undefined, + verify: async (path) => (await readFile(join(path, 'omnivoice.db'), 'utf8')) === 'database', + progress: (stage) => stages.push(stage), + }); + expect(result.removed_source).toBe(true); + expect(await readFile(join(target, 'voices', 'sample.wav'), 'utf8')).toBe('voice'); + expect(await readDataDirectorySetting(env)).toBe(target); + expect(existsSync(source)).toBe(false); + expect(stages).toContain('done'); +}); + +it('preserves a destination populated after inspection while staging the copy', async () => { + const { source, target } = await fixture(); + await mkdir(target); + // Observe the real staging directory while asynchronous filesystem work + // yields; neither copy nor directory removal is replaced by a test double. + let populated = false; + let observing = true; + const observe = () => { + if (!observing || populated) return; + if (readdirSync(dirname(target)).some((name) => name.includes('.voicestudio-moving-'))) { + writeFileSync(join(target, 'personal.txt'), 'keep'); + populated = true; + } else { + setImmediate(observe); + } + }; + setImmediate(observe); + try { + await expect(prepareDataRelocation(source, target)).rejects.toThrow(); + expect(populated).toBe(true); + expect(await readFile(join(target, 'personal.txt'), 'utf8')).toBe('keep'); + expect(await readFile(join(source, 'omnivoice.db'), 'utf8')).toBe('database'); + } finally { + observing = false; + } +}); diff --git a/electron/src/main/data-relocation.ts b/electron/src/main/data-relocation.ts index 5069d5c8c..c9d71cdcd 100644 --- a/electron/src/main/data-relocation.ts +++ b/electron/src/main/data-relocation.ts @@ -9,6 +9,7 @@ import { realpath, rename, rm, + rmdir, statfs, writeFile, } from 'node:fs/promises'; @@ -168,7 +169,7 @@ export async function prepareDataRelocation( throw new Error('verification_failed'); } try { - await rm(plan.target, { recursive: false }); + await rmdir(plan.target); } catch (error) { if ((error as NodeJS.ErrnoException).code !== 'ENOENT') throw error; } diff --git a/electron/src/main/ipc.ts b/electron/src/main/ipc.ts index 0f342924a..9b20c9f7a 100644 --- a/electron/src/main/ipc.ts +++ b/electron/src/main/ipc.ts @@ -8,7 +8,7 @@ import { authorizeModelsDirectory, } from './media-authorization'; import { isTrustedRenderer } from './trusted-renderer'; -import { writeFile } from 'node:fs/promises'; +import { writeExportAtomically } from './atomic-export'; import { homedir, tmpdir } from 'node:os'; import { dirname, isAbsolute, join, resolve } from 'node:path'; import { randomUUID } from 'node:crypto'; @@ -583,7 +583,7 @@ export function registerIpc( bypassCustomProtocolHandlers: true, }); if (!res.ok) throw new Error(`Could not download the audio (HTTP ${res.status})`); - await writeFile(picked.filePath, Buffer.from(await res.arrayBuffer())); + await writeExportAtomically(picked.filePath, Buffer.from(await res.arrayBuffer())); return { canceled: false, path: picked.filePath }; }); @@ -596,7 +596,7 @@ export function registerIpc( filters: saveFiltersFor(req.suggestedName), }); if (picked.canceled || !picked.filePath) return { canceled: true }; - await writeFile(picked.filePath, req.data); + await writeExportAtomically(picked.filePath, req.data); return { canceled: false, path: picked.filePath }; }); diff --git a/electron/src/main/remote-backend.test.ts b/electron/src/main/remote-backend.test.ts index 11f0945bc..2b843acf1 100644 --- a/electron/src/main/remote-backend.test.ts +++ b/electron/src/main/remote-backend.test.ts @@ -1,4 +1,7 @@ +// @vitest-environment node import { expect, it, vi } from 'vitest'; +import { createServer, type ServerResponse } from 'node:http'; +import { once } from 'node:events'; import { normalizeRemoteUrl, probeRemoteBackend, remoteWebSocketUrl } from './remote-backend'; function json(body: unknown, init: ResponseInit = {}): Response { @@ -8,6 +11,34 @@ function json(body: unknown, init: ResponseInit = {}): Response { }); } +it.each(['/ws/transcribe', '/ws/events', '/ws/tts'] as const)( + 'preserves remote base prefixes and canonical ticket binding for %s', + async (path) => { + for (const base of ['https://gpu-box:3900/voice', 'https://gpu-box:3900/nested/voice/']) { + const session = { token: `ovs_admin_session_${'a'.repeat(43)}`, expiresAt: 3601 }; + const fetcher = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { + expect(String(input)).toBe(`${base.replace(/\/+$/, '')}/api/auth/ws-ticket`); + expect(init?.body).toBe(JSON.stringify({ path })); + expect(new Headers(init?.headers).get('authorization')).toBe(`Bearer ${session.token}`); + return json({ ticket: `ovs_ws_ticket_${'b'.repeat(43)}`, expires_in: 30 }, { status: 201 }); + }); + const expected = `wss://gpu-box:3900${new URL(base).pathname.replace(/\/+$/, '')}${path}`; + const anonymous = await remoteWebSocketUrl(base, path, null); + expect(anonymous).toBe(expected); + const authorized = new URL(await remoteWebSocketUrl(base, path, session, { + fetcher, now: () => 1000, + })); + expect(authorized.origin + authorized.pathname).toBe(expected); + expect([...authorized.searchParams.keys()]).toEqual(['ws_ticket']); + expect(authorized.searchParams.get('ws_ticket')).toBe(`ovs_ws_ticket_${'b'.repeat(43)}`); + expect(authorized.href).not.toContain(session.token); + } + expect(await remoteWebSocketUrl('http://gpu-box:3900/voice', path, null)).toBe( + `ws://gpu-box:3900/voice${path}`, + ); + }, +); + it('accepts only credential-free absolute HTTP backend bases', () => { expect(normalizeRemoteUrl(' https://gpu-box:3900/ ')).toBe('https://gpu-box:3900'); for (const value of [ @@ -113,3 +144,96 @@ it.each(['/ws/transcribe', '/ws/events', '/ws/tts'] as const)( ); }, ); + + +const nativeResponses: Record = { + '/health': { status: 'ok', version: 'fixture', device: 'cpu' }, + '/system/info': { app_version: 'fixture' }, + '/api/auth/session': { token: `ovs_admin_session_${'a'.repeat(43)}`, expires_in: 3600 }, + '/api/auth/ws-ticket': { ticket: `ovs_ws_ticket_${'b'.repeat(43)}`, expires_in: 30 }, +}; + +async function nativeBackend( + handle: (path: string, response: ServerResponse) => void, + run: (base: string) => Promise, +) { + const server = createServer((request, response) => handle(request.url!, response)); + server.listen(0, '127.0.0.1'); + await once(server, 'listening'); + const address = server.address(); + if (!address || typeof address === 'string') throw new Error('No fixture port'); + try { + await run(`http://127.0.0.1:${address.port}`); + } finally { + server.closeAllConnections(); + await new Promise((resolve) => server.close(() => resolve())); + } +} + +it.each(Object.keys(nativeResponses))('bounds a stalled remote %s body after headers', async (path) => { + await nativeBackend((requested, response) => { + const body = JSON.stringify(nativeResponses[requested]); + response.writeHead(requested.includes('/auth/') ? 201 : 200, { + 'Content-Type': 'application/json', + }); + if (requested === path) { + response.write(body.slice(0, 1)); + // Release the original implementation too, so a RED run cannot hang. + const timer = setTimeout(() => response.end(body.slice(1)), 200); + response.once('close', () => clearTimeout(timer)); + } else response.end(body); + }, async (base) => { + if (path === '/api/auth/ws-ticket') { + await expect(remoteWebSocketUrl(base, '/ws/tts', { + token: `ovs_admin_session_${'a'.repeat(43)}`, + expiresAt: Date.now() / 1000 + 60, + }, { timeoutMs: 50 })).rejects.toMatchObject({ name: 'AbortError' }); + } else { + expect(await probeRemoteBackend(base, path === '/api/auth/session' ? 'fixture-key' : '', { + timeoutMs: 50, + })).toMatchObject({ ok: false, kind: 'timeout' }); + } + }); +}); + +it('keeps immediate authenticated remote responses and header-only auth failures', async () => { + let deny = false; + await nativeBackend((path, response) => { + if (deny && path === '/system/info') { + response.writeHead(401, { 'Content-Type': 'application/json' }); + response.write('{'); // Error classification does not require this body. + } else { + response.writeHead(path.includes('/auth/') ? 201 : 200, { + 'Content-Type': 'application/json', + }); + response.end(JSON.stringify(nativeResponses[path])); + } + }, async (base) => { + expect(await probeRemoteBackend(base, 'fixture-key', { timeoutMs: 100 })).toMatchObject({ + ok: true, session: { token: `ovs_admin_session_${'a'.repeat(43)}` }, + }); + deny = true; + expect(await probeRemoteBackend(base, '', { timeoutMs: 100 })).toMatchObject({ + ok: false, kind: 'auth', status: 401, + }); + }); +}); + +it('keeps remote session exchanges from following redirects', async () => { + const paths: string[] = []; + await nativeBackend((path, response) => { + paths.push(path); + if (path === '/health') { + response.writeHead(200, { 'Content-Type': 'application/json' }); + response.end(JSON.stringify(nativeResponses[path])); + } else { + response.writeHead(302, { Location: '/trap' }); + response.end(); + } + }, async (base) => { + expect(await probeRemoteBackend(base, 'fixture-key', { timeoutMs: 100 })).toMatchObject({ + ok: false, kind: 'network', + }); + expect(paths).toEqual(['/health', '/api/auth/session']); + }); +}); diff --git a/electron/src/main/remote-backend.ts b/electron/src/main/remote-backend.ts index d53b88eb8..8be7172ec 100644 --- a/electron/src/main/remote-backend.ts +++ b/electron/src/main/remote-backend.ts @@ -75,16 +75,22 @@ async function fetchWithin( url: string, init: RequestInit, timeoutMs: number, -): Promise { + expectedStatus?: number, +): Promise<{ ok: boolean; status: number; payload: Record | null }> { const controller = new AbortController(); const timer = setTimeout(() => controller.abort(), timeoutMs); try { - return await fetcher(url, { + const response = await fetcher(url, { ...init, cache: 'no-store', redirect: 'error', signal: controller.signal, }); + // Fetch resolves at headers; keep the deadline armed through JSON reads. + const accepted = + expectedStatus === undefined ? response.ok : response.status === expectedStatus; + const payload = accepted ? await readObject(response) : null; + return { ok: response.ok, status: response.status, payload }; } finally { clearTimeout(timer); } @@ -107,7 +113,7 @@ export async function probeRemoteBackend( if (!healthResponse.ok) { return { ok: false, kind: 'http', status: healthResponse.status, target }; } - const health = await readObject(healthResponse); + const health = healthResponse.payload; if ( !health || health.status !== 'ok' || @@ -133,6 +139,7 @@ export async function probeRemoteBackend( body: JSON.stringify({ transport: 'bearer' }), }, timeoutMs, + 201, ); if (exchange.status !== 201) { return { @@ -142,7 +149,7 @@ export async function probeRemoteBackend( target, }; } - const payload = await readObject(exchange); + const payload = exchange.payload; const token = payload?.token; const relative = payload?.expires_in; if ( @@ -173,7 +180,7 @@ export async function probeRemoteBackend( target, }; } - const info = await readObject(infoResponse); + const info = infoResponse.payload; if (!info || typeof info.app_version !== 'string') { return { ok: false, kind: 'wrong_port', target }; } @@ -195,7 +202,7 @@ export async function remoteWebSocketUrl( { fetcher = fetch, now = Date.now, timeoutMs = 5000 }: ProbeOptions = {}, ): Promise { const target = normalizeRemoteUrl(rawUrl); - const url = new URL(path, `${target}/`); + const url = new URL(path.slice(1), `${target}/`); url.protocol = url.protocol === 'https:' ? 'wss:' : 'ws:'; if (!session) return url.toString(); if (session.expiresAt <= now() / 1000) throw new Error('Remote session expired'); @@ -211,10 +218,11 @@ export async function remoteWebSocketUrl( body: JSON.stringify({ path }), }, timeoutMs, + 201, ); if (response.status !== 201) throw new Error(`Could not authorize WebSocket (HTTP ${response.status})`); - const payload = await readObject(response); + const payload = response.payload; const expiresIn = payload?.expires_in; if ( typeof payload?.ticket !== 'string' || diff --git a/electron/src/main/setup-progress.test.ts b/electron/src/main/setup-progress.test.ts index 5ca761283..689dca96f 100644 --- a/electron/src/main/setup-progress.test.ts +++ b/electron/src/main/setup-progress.test.ts @@ -71,3 +71,47 @@ describe('SetupProgressTracker', () => { expect(cleanProcessLine('\u001b[2K Downloaded torch\r')).toBe('Downloaded torch'); }); }); + + +describe('concurrent package byte updates', () => { + it.each([ + ['torch', 'torchvision'], + ['torchvision', 'torch'], + ])('uses the closest planned identifier in a multi-package prefix (%s first)', (first, second) => { + const tracker = new SetupProgressTracker(); + for (const name of [first, second]) { + tracker.ingest(`Downloading ${name} (${name === 'torch' ? 10 : 20} MiB)`, 0); + } + expect(tracker.ingest('torch ... torchvision 2 MiB / 20 MiB', 1000)).toMatchObject({ + activePackage: 'torchvision', + totalBytes: 30 * 1024 ** 2, + downloadedBytes: 2 * 1024 ** 2, + }); + }); + it.each([ + ['torch', 'torchvision'], + ['torchvision', 'torch'], + ['pydantic', 'pydantic-core'], + ['lib', 'lib.v2'], + ])('matches the complete %s / %s package identifiers', (first, second) => { + const tracker = new SetupProgressTracker(); + tracker.ingest(`Downloading ${first} (10 MiB)`, 0); + tracker.ingest(`Downloading ${second} (20 MiB)`, 0); + expect(tracker.ingest(`${second} 2 MiB / 20 MiB`, 1000)).toMatchObject({ + activePackage: second, + totalBytes: 30 * 1024 ** 2, + downloadedBytes: 2 * 1024 ** 2, + }); + expect(tracker.ingest(`Downloaded ${first}`, 1001)).toMatchObject({ + totalBytes: 30 * 1024 ** 2, + downloadedBytes: 12 * 1024 ** 2, + }); + }); + + it('does not attribute an unannounced package to a matching substring', () => { + const tracker = new SetupProgressTracker(); + tracker.ingest('Downloading torch (10 MiB)', 0); + expect(tracker.ingest('torchvision 2 MiB / 20 MiB', 1000)).toBeNull(); + expect(tracker.snapshot()).toMatchObject({totalBytes: 10 * 1024 ** 2, downloadedBytes: 0}); + }); +}); diff --git a/electron/src/main/setup-progress.ts b/electron/src/main/setup-progress.ts index 2560d4cbb..658917481 100644 --- a/electron/src/main/setup-progress.ts +++ b/electron/src/main/setup-progress.ts @@ -96,7 +96,9 @@ export class SetupProgressTracker { if (bytePair) { const received = parseByteSize(bytePair[1], bytePair[2]); const total = parseByteSize(bytePair[3], bytePair[4]); - const name = [...this.planned.keys()].find((candidate) => line.includes(candidate)); + // Match complete package identifiers, so torchvision cannot update torch. + const identifiers = line.slice(0, bytePair.index).match(/[a-zA-Z0-9][a-zA-Z0-9_.-]*/g) ?? []; + const name = identifiers.reverse().find((candidate) => this.planned.has(candidate)); if (name) { this.planned.set(name, total); this.received.set(name, Math.min(received, total)); diff --git a/electron/src/renderer/src/features/projects/projects-page.tsx b/electron/src/renderer/src/features/projects/projects-page.tsx index aa177a990..98866ed63 100644 --- a/electron/src/renderer/src/features/projects/projects-page.tsx +++ b/electron/src/renderer/src/features/projects/projects-page.tsx @@ -50,6 +50,7 @@ import { runRendererTask } from '@/lib/global-error-recovery'; import { isImeComposing } from '@/lib/ime'; import { useProfiles } from '@/hooks/use-profiles'; import { useHistory } from '@/hooks/use-history'; +import { reuseTake as restoreTake } from '@/lib/store/takes'; import { patchCloneSettings } from '@/lib/store/clone-settings'; import { selectCloneProfile } from '@/lib/store/reference'; import type { HistoryItem, Profile } from '@/lib/api/types'; @@ -463,12 +464,7 @@ export function ProjectsPage() { await navigate({ to: '/clone' }); }; const reuseTake = async (take: HistoryItem) => { - patchCloneSettings({ - text: take.text, - language: take.language || 'Auto', - selectedProfileId: take.profile_id, - instruct: take.instruct || '', - }); + await restoreTake(take); await navigate({ to: '/clone' }); }; const reveal = async (record: ExportRecord) => { diff --git a/electron/src/renderer/src/features/projects/projects-take-reuse.test.tsx b/electron/src/renderer/src/features/projects/projects-take-reuse.test.tsx new file mode 100644 index 000000000..8d510f3ff --- /dev/null +++ b/electron/src/renderer/src/features/projects/projects-take-reuse.test.tsx @@ -0,0 +1,47 @@ +import { afterEach, expect, it, vi } from 'vitest'; +import { cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react'; +import { QueryClient, QueryClientProvider } from '@tanstack/react-query'; +import { cloneSettingsStore, DEFAULT_CLONE_SETTINGS, patchCloneSettings } from '@/lib/store/clone-settings'; +import { rememberTake } from '@/lib/store/takes'; +import type { HistoryItem } from '@/lib/api/types'; + +const mocks = vi.hoisted(() => ({ navigate: vi.fn(), history: [] as HistoryItem[] })); +vi.mock('react-i18next', () => ({ useTranslation: () => ({ t: (key: string) => key }) })); +vi.mock('@tanstack/react-router', () => ({ useNavigate: () => mocks.navigate })); +vi.mock('lucide-react', () => { + const Icon = () => null; + return Object.fromEntries(['AudioLines', 'BookOpen', 'Clock', 'Download', 'FileText', 'Film', 'Fingerprint', 'FolderOpen', 'Grid2X2', 'List', 'Mic', 'Pencil', 'Save', 'Search', 'Trash'].map(name => [name + 'Icon', Icon])); +}); +vi.mock('@/components/workspace-sidebar', () => ({ SecondarySidebar: ({ children }: any) => })); +vi.mock('@/components/app-shell/workspace-header', () => ({ WorkspaceHeader: ({ children }: any) =>
{children}
})); +vi.mock('@/components/ui/button', () => ({ Button: ({ children, variant: _variant, size: _size, ...props }: any) => })); +vi.mock('@/components/ui/input', () => ({ Input: (props: any) => })); +vi.mock('@/components/ui/dialog', () => ({ Dialog: () => null, DialogContent: () => null, DialogHeader: () => null, DialogTitle: () => null, DialogDescription: () => null, DialogFooter: () => null })); +vi.mock('@/components/audio-preview-button', () => ({ AudioPreviewButton: () => null })); +vi.mock('@/components/pipeline-failure', () => ({ PipelineFailure: () => null })); +vi.mock('@/components/profile-avatar', () => ({ ProfileAvatar: () => null })); +vi.mock('@/components/bridge', () => ({ getBridge: () => null })); +vi.mock('@/hooks/use-profiles', () => ({ useProfiles: () => ({ data: [] }), useDeleteProfile: () => ({ mutateAsync: vi.fn() }) })); +vi.mock('@/hooks/use-history', () => ({ useHistory: () => ({ data: mocks.history }) })); +vi.mock('@/lib/api/client', () => ({ apiJson: async (path: string) => path === '/longform/jobs' ? { jobs: [] } : [], apiPath: (path: string) => path, describeError: String })); +vi.mock('@/lib/global-error-recovery', () => ({ runRendererTask: vi.fn() })); +vi.mock('../dub/dub-session', () => ({ useDubSession: () => ({ phase: 'idle' }), openDubProject: vi.fn(), attachDubProject: vi.fn(), detachDubProject: vi.fn() })); +vi.mock('../longform/longform-session', () => ({ useLongformSession: () => ({ active: false }), longformSession: { state: { active: false, drafts: {} } }, blankLongformDraft: () => ({}), editLongform: vi.fn() })); +vi.mock('../longform/project-library', () => ({ projectLibrary: { list: async () => [], remove: vi.fn(), rename: vi.fn() } })); +vi.mock('./render-details', () => ({ RenderDetails: () => null, renderRecipe: () => null })); +import { ProjectsPage } from './projects-page'; + +let client: QueryClient; +afterEach(() => { cleanup(); client?.clear(); localStorage.clear(); vi.clearAllMocks(); }); +it('restores saved Clone quality through the Projects Reuse button', async () => { + const item: HistoryItem = { id: 'saved', text: 'Saved take', mode: 'clone', language: 'English', instruct: '', profile_id: 'voice', audio_path: 'saved.wav', duration_seconds: 1, generation_time: 1, seed: null, starred: false, created_at: 1 }; + localStorage.clear(); + mocks.history = [item]; + rememberTake(item.id, { ...DEFAULT_CLONE_SETTINGS, wavBits: 24, effectPreset: 'raw', speed: 1.5 }); + patchCloneSettings({ ...DEFAULT_CLONE_SETTINGS, wavBits: 16, effectPreset: 'broadcast', autoPlay: true }); + client = new QueryClient({ defaultOptions: { queries: { retry: false } } }); + render(); + fireEvent.click(await screen.findByRole('button', { name: 'clone.history_reuse' })); + await waitFor(() => expect(mocks.navigate).toHaveBeenCalledWith({ to: '/clone' })); + expect(cloneSettingsStore.state).toMatchObject({ wavBits: 24, effectPreset: 'raw', speed: 1.5, autoPlay: true, text: 'Saved take', selectedProfileId: 'voice' }); +}); diff --git a/electron/src/renderer/src/features/tools/compare-voices.test.tsx b/electron/src/renderer/src/features/tools/compare-voices.test.tsx index 2b2d459e0..a1279ac6f 100644 --- a/electron/src/renderer/src/features/tools/compare-voices.test.tsx +++ b/electron/src/renderer/src/features/tools/compare-voices.test.tsx @@ -2,7 +2,8 @@ import { clearComparison } from './comparison-state'; import { cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react'; import { QueryClient, QueryClientProvider } from '@tanstack/react-query'; import { afterEach, expect, it, vi } from 'vitest'; -const mock = vi.hoisted(() => ({ generate: vi.fn() })); +const mock = vi.hoisted(() => ({ generate: vi.fn(), warning: vi.fn() })); +vi.mock('sonner', () => ({ toast: { warning: mock.warning } })); vi.mock('@/lib/api/generate', () => ({ generateClone: mock.generate })); vi.mock('@/hooks/use-tts-readiness', () => ({ useTtsReadiness: () => null })); vi.mock('@/hooks/use-profiles', () => ({ @@ -13,7 +14,12 @@ vi.mock('@/hooks/use-profiles', () => ({ ], }), })); -vi.mock('react-i18next', () => ({ useTranslation: () => ({ t: (key: string) => key }) })); +vi.mock('react-i18next', () => ({ + useTranslation: () => ({ + t: (key: string, values?: { count: number; text?: string }) => + values ? `${key}:${values.count}:${values.text ?? ''}` : key, + }), +})); vi.mock('@/components/waveform-player', () => ({ WaveformPlayer: ({ source }: { source: string }) =>
, })); @@ -57,9 +63,43 @@ it('generates both selected voices sequentially with one text and preserves edit seed: 12, }); expect(cloneSettingsStore.state).toBe(settings); + expect(mock.warning).not.toHaveBeenCalled(); +}); +it.each([ + { count: 1, text: ' Missing sentence. ', message: 'tts.droppedChunksWithText:1:Missing sentence.' }, + { count: 3, text: ' ', message: 'tts.droppedChunks:3:' }, + { count: 2, text: 'x'.repeat(150), message: `tts.droppedChunksWithText:2:${'x'.repeat(120)}` }, +])( + 'warns about omitted comparison speech while retaining both previews ($count)', + async ({ count, text, message }) => { + mock.generate + .mockResolvedValueOnce({ blob: new Blob(['partial audio']), dropped: { count, text } }) + .mockResolvedValueOnce({ blob: new Blob(['complete audio']) }); + mount(); + fireEvent.click(screen.getByRole('button', { name: 'compare.compare_btn' })); + await screen.findByTestId('compare-1'); + expect(screen.getByTestId('compare-0')).toBeInTheDocument(); + expect(mock.generate).toHaveBeenCalledTimes(2); + expect(mock.warning).toHaveBeenCalledExactlyOnceWith(message, { duration: 8000 }); + }, +); +it('also reports omitted speech from the second voice', async () => { + mock.generate + .mockResolvedValueOnce({ blob: new Blob(['complete audio']) }) + .mockResolvedValueOnce({ + blob: new Blob(['partial audio']), + dropped: { count: 1, text: 'Missing tail.' }, + }); + mount(); + fireEvent.click(screen.getByRole('button', { name: 'compare.compare_btn' })); + await screen.findByTestId('compare-1'); + expect(mock.warning).toHaveBeenCalledExactlyOnceWith( + 'tts.droppedChunksWithText:1:Missing tail.', + { duration: 8000 }, + ); }); it('does not start the second voice after leaving the comparison', async () => { - let finish!: (value: { blob: Blob }) => void; + let finish!: (value: { blob: Blob; dropped: { count: number; text: string } }) => void; mock.generate.mockImplementation( () => new Promise((resolve) => { @@ -70,8 +110,9 @@ it('does not start the second voice after leaving the comparison', async () => { fireEvent.click(screen.getByRole('button', { name: 'compare.compare_btn' })); view.unmount(); expect(mock.generate.mock.calls[0][1].signal.aborted).toBe(true); - finish({ blob: new Blob() }); + finish({ blob: new Blob(), dropped: { count: 1, text: 'Cancelled result.' } }); await waitFor(() => expect(mock.generate).toHaveBeenCalledTimes(1)); + expect(mock.warning).not.toHaveBeenCalled(); }); it('keeps completed comparison text, voices and playback across remounts', async () => { diff --git a/electron/src/renderer/src/features/tools/compare-voices.tsx b/electron/src/renderer/src/features/tools/compare-voices.tsx index 58dfe0eda..095841c0c 100644 --- a/electron/src/renderer/src/features/tools/compare-voices.tsx +++ b/electron/src/renderer/src/features/tools/compare-voices.tsx @@ -16,6 +16,7 @@ import { describeError } from '@/lib/api/client'; import { PRESETS } from '@shared/utils/constants'; import { beginAppActivity } from '@/lib/app-activity'; import { useTtsReadiness } from '@/hooks/use-tts-readiness'; +import { toast } from 'sonner'; export function CompareVoices() { const { t } = useTranslation(); @@ -90,6 +91,16 @@ export function CompareVoices() { { signal: controller.signal }, ); if (controller.signal.aborted) break; + if (result.dropped) { + const preview = result.dropped.text.trim().slice(0, 120); + const count = result.dropped.count; + toast.warning( + preview + ? t('tts.droppedChunksWithText', { count, text: preview }) + : t('tts.droppedChunks', { count }), + { duration: 8000 }, + ); + } const url = URL.createObjectURL(result.blob); setUrls((current) => current.map((old, index) => (index === side ? url : old))); void client.invalidateQueries({ queryKey: queryKeys.history }); diff --git a/electron/src/renderer/src/features/transcriptions/history.test.ts b/electron/src/renderer/src/features/transcriptions/history.test.ts index a07774737..870abfecb 100644 --- a/electron/src/renderer/src/features/transcriptions/history.test.ts +++ b/electron/src/renderer/src/features/transcriptions/history.test.ts @@ -1,11 +1,14 @@ -import { beforeEach, expect, it, vi } from 'vitest'; +import { afterEach, beforeEach, expect, it, vi } from 'vitest'; import { addTranscription, loadTranscriptions, + removeTranscription, + subscribeTranscriptions, TRANSCRIPTIONS_KEY, TRANSCRIPTION_EVENT, } from '@shared/utils/transcriptionsStore'; beforeEach(() => localStorage.clear()); +afterEach(() => vi.restoreAllMocks()); it('preserves existing history and publishes a complete entry with nullable timings', () => { localStorage.setItem(TRANSCRIPTIONS_KEY, JSON.stringify([{ id: 1, text: 'Existing' }])); const listener = vi.fn(); @@ -38,3 +41,67 @@ it('keeps raw and refined transcripts separately', () => { addTranscription({ text: 'um original', refined_text: 'Original.' }); expect(loadTranscriptions()[0]).toMatchObject({ text: 'um original', refined_text: 'Original.' }); }); + +it('deleting one same-clock utterance preserves the others and the selected row identity', () => { + vi.spyOn(Date, 'now').mockReturnValue(1000); + const rows = Array.from({ length: 5 }, (_, index) => addTranscription({ text: `Utterance ${index}` })); + const selected = rows[2]!; + const updates = vi.fn(); + const unsubscribe = subscribeTranscriptions(updates); + try { + expect(new Set(rows.map(row => row.id)).size).toBe(5); + removeTranscription(rows[4]!.id); + expect(loadTranscriptions().map(row => row.text)).toEqual(['Utterance 3', 'Utterance 2', 'Utterance 1', 'Utterance 0']); + expect(loadTranscriptions().find(row => row.id === selected.id)?.text).toBe(selected.text); + expect(updates).toHaveBeenCalledOnce(); + expect(updates.mock.calls[0]![0]).toHaveLength(4); + } finally { + unsubscribe(); + } +}); +it('reads persisted identities once and leaves prior records unchanged when avoiding a collision', () => { + vi.spyOn(Date, 'now').mockReturnValue(1000); + const existing = [{ id: 1000, text: 'Earlier' }, { id: 1001, text: 'Next' }, { id: 50, text: 'Legacy' }]; + localStorage.setItem(TRANSCRIPTIONS_KEY, JSON.stringify(existing)); + const reader = vi.spyOn(Storage.prototype, 'getItem'); + const saved = addTranscription({ text: 'Latest' }); + expect(reader).toHaveBeenCalledTimes(1); + expect(saved.id).toBe(1002); + expect(loadTranscriptions().slice(1)).toEqual(existing); +}); +it('keeps a distinct numeric identity after wall-clock rollback', () => { + const clock = vi.spyOn(Date, 'now').mockReturnValue(1000); + const first = addTranscription({ text: 'First' }); + clock.mockReturnValue(999); + const second = addTranscription({ text: 'Second' }); + const third = addTranscription({ text: 'Third' }); + expect([first.id, second.id, third.id]).toEqual([1000, 999, 1001]); + removeTranscription(second.id); + expect(loadTranscriptions().map(row => row.text)).toEqual(['Third', 'First']); +}); +it('keeps the 200-entry bound when every persisted timestamp identity is occupied', () => { + vi.spyOn(Date, 'now').mockReturnValue(1000); + localStorage.setItem(TRANSCRIPTIONS_KEY, JSON.stringify(Array.from({ length: 200 }, (_, offset) => ({ id: 1000 + offset, text: `Old ${offset}` })))); + const saved = addTranscription({ text: 'Latest' }); + const history = loadTranscriptions(); + expect(saved.id).toBe(1200); + expect(history).toHaveLength(200); + expect(new Set(history.map(row => row.id)).size).toBe(200); + expect(history[199]!.text).toBe('Old 198'); +}); + +it('does not publish a history update when saving a distinct identity fails', () => { + vi.spyOn(Date, 'now').mockReturnValue(1000); + const existing = [{ id: 1000, text: 'Existing' }]; + localStorage.setItem(TRANSCRIPTIONS_KEY, JSON.stringify(existing)); + const listener = vi.fn(); + window.addEventListener(TRANSCRIPTION_EVENT, listener); + vi.spyOn(Storage.prototype, 'setItem').mockImplementation(() => { throw new Error('storage full'); }); + try { + expect(() => addTranscription({ text: 'New' })).toThrow('storage full'); + expect(loadTranscriptions()).toEqual(existing); + expect(listener).not.toHaveBeenCalled(); + } finally { + window.removeEventListener(TRANSCRIPTION_EVENT, listener); + } +}); diff --git a/electron/src/renderer/src/features/transcriptions/live-dictation.test.ts b/electron/src/renderer/src/features/transcriptions/live-dictation.test.ts index 3f7ff24f8..dd8fa731e 100644 --- a/electron/src/renderer/src/features/transcriptions/live-dictation.test.ts +++ b/electron/src/renderer/src/features/transcriptions/live-dictation.test.ts @@ -205,3 +205,58 @@ it('preserves a main-issued ticket for authenticated remote dictation', async () expect(Socket.instances[0].url.searchParams.get('ws_ticket')).toMatch(/^ovs_ws_ticket_/); expect(Socket.instances[0].url.searchParams.get('model')).toBe('sherpa-test'); }); + +function delayedMainTicket() { + let resolve!: (url: string) => void; + let reject!: (error: Error) => void; + const ticket = new Promise((yes, no) => { resolve = yes; reject = no; }); + const websocketUrl = vi.fn() + .mockReturnValueOnce(ticket) + .mockResolvedValue('ws://127.0.0.1:3900/ws/transcribe?session=current'); + mocks.bridge = { backend: { websocketUrl } }; + vi.stubGlobal('window', { + location: { href: 'app://voicestudio/index.html#/capture', protocol: 'app:' }, + }); + return { websocketUrl, resolve, reject }; +} + +it('does not create a socket when a cancelled main ticket arrives late', async () => { + const ticket = delayedMainTicket(); + const pending = live.start(vi.fn()); + await vi.waitFor(() => expect(ticket.websocketUrl).toHaveBeenCalledOnce()); + live.cancel(); + ticket.resolve('ws://127.0.0.1:3900/ws/transcribe?session=cancelled'); + await pending; + expect(Socket.instances).toHaveLength(0); + expect(stopTrack).toHaveBeenCalledOnce(); + expect(live.getSnapshot().stage).toBe('idle'); +}); + +it('keeps EOF on the current socket after a superseded ticket resolves', async () => { + const ticket = delayedMainTicket(); + const old = live.start(vi.fn()); + await vi.waitFor(() => expect(ticket.websocketUrl).toHaveBeenCalledOnce()); + live.cancel(); + await live.start(vi.fn()); + const current = Socket.instances[0]!; + ticket.resolve('ws://127.0.0.1:3900/ws/transcribe?session=cancelled'); + await old; + expect(Socket.instances).toHaveLength(1); + expect(live.getSnapshot().stage).toBe('recording'); + await live.stop(); + expect(current.send).toHaveBeenLastCalledWith('EOF'); + expect(current.close).not.toHaveBeenCalled(); +}); + +it('keeps the replacement recording after a cancelled ticket rejects', async () => { + const ticket = delayedMainTicket(); + const old = live.start(vi.fn()); + await vi.waitFor(() => expect(ticket.websocketUrl).toHaveBeenCalledOnce()); + live.cancel(); + await live.start(vi.fn()); + ticket.reject(new Error('Expired cancelled ticket')); + await old; + expect(Socket.instances).toHaveLength(1); + expect(live.getSnapshot()).toMatchObject({ stage: 'recording', issue: undefined }); + expect(Socket.instances[0]!.close).not.toHaveBeenCalled(); +}); diff --git a/electron/src/renderer/src/features/transcriptions/live-dictation.ts b/electron/src/renderer/src/features/transcriptions/live-dictation.ts index 7a241efa8..5aaff27ed 100644 --- a/electron/src/renderer/src/features/transcriptions/live-dictation.ts +++ b/electron/src/renderer/src/features/transcriptions/live-dictation.ts @@ -112,6 +112,7 @@ export class LiveDictation { this.stream = stream; issue = 'connection'; const url = new URL(await backendWebSocketUrl('/ws/transcribe')); + if (!current()) return; url.searchParams.set('model', prefs.model_id); url.searchParams.set('pcm', '1'); url.searchParams.set('sr', '16000'); diff --git a/electron/src/renderer/src/i18n/locales/ar.json b/electron/src/renderer/src/i18n/locales/ar.json index 2a20084b7..e18a16f82 100644 --- a/electron/src/renderer/src/i18n/locales/ar.json +++ b/electron/src/renderer/src/i18n/locales/ar.json @@ -819,6 +819,8 @@ "qc_result": "{{flagged}} من {{total}} قد تحتاج الأسطر إلى إعادة الاستماع", "qc_clean": "جميع أسطر {{total}} تتطابق مع النص", "qc_failed": "فشل التحقق من التوقيت: {{message}}", + "qc_track_changed": "تغيرت الدبلجة أثناء فحص الجودة. أعد الفحص على المسار الحالي.", + "qc_identity_missing": "هويات المقاطع غير واضحة. أعد إنشاء هذا المسار بمعرّفات فريدة للمقاطع.", "paste_translation_btn": "لصق ترجمة", "paste_translation_title": "لصق ترجمة", "paste_translation_desc": "الصق ترجمة أعددتها في مكان آخر (ChatGPT أو DeepL أو مترجم بشري). ستُطابَق مع المقاطع الموجودة لديك — تبقى التوقيتات والنص الأصلي دون تغيير.", diff --git a/electron/src/renderer/src/i18n/locales/de.json b/electron/src/renderer/src/i18n/locales/de.json index f57560190..c893426fb 100644 --- a/electron/src/renderer/src/i18n/locales/de.json +++ b/electron/src/renderer/src/i18n/locales/de.json @@ -809,6 +809,8 @@ "qc_result": "{{flagged}} von {{total}} Zeilen müssen möglicherweise erneut abgehört werden", "qc_clean": "Alle {{total}}-Zeilen stimmen mit dem Skript überein", "qc_failed": "Zeitprüfung fehlgeschlagen: {{message}}", + "qc_track_changed": "Die Synchronisation wurde während der Qualitätsprüfung geändert. Prüfe die aktuelle Spur erneut.", + "qc_identity_missing": "Die Segmentzuordnung ist uneindeutig. Erzeuge diese Spur mit eindeutigen Segment-IDs neu.", "paste_translation_btn": "Übersetzung einfügen", "paste_translation_title": "Übersetzung einfügen", "paste_translation_desc": "Füge eine anderswo erstellte Übersetzung ein (ChatGPT, DeepL, ein menschlicher Übersetzer). Sie wird den vorhandenen Segmenten zugeordnet — Timings und Originaltranskript bleiben unverändert.", diff --git a/electron/src/renderer/src/i18n/locales/en.json b/electron/src/renderer/src/i18n/locales/en.json index dd47c8642..83f713c11 100644 --- a/electron/src/renderer/src/i18n/locales/en.json +++ b/electron/src/renderer/src/i18n/locales/en.json @@ -965,6 +965,8 @@ "qc_result": "{{flagged}} of {{total}} lines may need a re-listen", "qc_clean": "All {{total}} lines match the script", "qc_failed": "Timing check failed: {{message}}", + "qc_track_changed": "The dub changed during the quality check. Run the check again on the current track.", + "qc_identity_missing": "Segment identities are ambiguous. Regenerate this track with unique segment IDs.", "export_btn": "Export…", "prep_stop": "Stop", "install_progress": "Installing {{engine}}…", diff --git a/electron/src/renderer/src/i18n/locales/es.json b/electron/src/renderer/src/i18n/locales/es.json index 73f85da36..1eeb88e5b 100644 --- a/electron/src/renderer/src/i18n/locales/es.json +++ b/electron/src/renderer/src/i18n/locales/es.json @@ -811,6 +811,8 @@ "qc_result": "Es posible que sea necesario volver a escuchar {{flagged}} de {{total}} líneas", "qc_clean": "Todas las líneas {{total}} coinciden con el guión.", "qc_failed": "Error en la verificación de tiempo: {{message}}", + "qc_track_changed": "El doblaje cambió durante la comprobación de calidad. Repite la comprobación en la pista actual.", + "qc_identity_missing": "La identidad de los segmentos es ambigua. Regenera esta pista con identificadores de segmento únicos.", "paste_translation_btn": "Pegar traducción", "paste_translation_title": "Pegar una traducción", "paste_translation_desc": "Pega una traducción hecha en otro sitio (ChatGPT, DeepL, un traductor humano). Se asigna a los segmentos que ya tienes: los tiempos y la transcripción original no cambian.", diff --git a/electron/src/renderer/src/i18n/locales/fr.json b/electron/src/renderer/src/i18n/locales/fr.json index b267c06c7..ac15faa50 100644 --- a/electron/src/renderer/src/i18n/locales/fr.json +++ b/electron/src/renderer/src/i18n/locales/fr.json @@ -811,6 +811,8 @@ "qc_result": "{{flagged}} lignes sur {{total}} peuvent nécessiter une réécoute", "qc_clean": "Toutes les lignes {{total}} correspondent au script", "qc_failed": "Échec de la vérification du timing : {{message}}", + "qc_track_changed": "Le doublage a changé pendant le contrôle qualité. Relancez le contrôle sur la piste actuelle.", + "qc_identity_missing": "Les segments ne sont pas identifiés de façon univoque. Régénérez cette piste avec des identifiants de segment uniques.", "paste_translation_btn": "Coller une traduction", "paste_translation_title": "Coller une traduction", "paste_translation_desc": "Collez une traduction réalisée ailleurs (ChatGPT, DeepL, un traducteur humain). Elle est appliquée aux segments existants — les timings et la transcription d'origine restent intacts.", diff --git a/electron/src/renderer/src/i18n/locales/hi.json b/electron/src/renderer/src/i18n/locales/hi.json index 57eb1d30f..cb929d8a4 100644 --- a/electron/src/renderer/src/i18n/locales/hi.json +++ b/electron/src/renderer/src/i18n/locales/hi.json @@ -809,6 +809,8 @@ "qc_result": "{{total}} पंक्तियों में से {{flagged}} को दोबारा सुनने की आवश्यकता हो सकती है", "qc_clean": "सभी {{total}} पंक्तियाँ स्क्रिप्ट से मेल खाती हैं", "qc_failed": "समय की जाँच विफल: {{message}}", + "qc_track_changed": "गुणवत्ता जाँच के दौरान डबिंग बदल गई। मौजूदा ट्रैक पर जाँच दोबारा चलाएँ।", + "qc_identity_missing": "सेगमेंट की पहचान स्पष्ट नहीं है। हर सेगमेंट को अलग आईडी देकर इस ट्रैक को फिर से जनरेट करें।", "paste_translation_btn": "अनुवाद चिपकाएँ", "paste_translation_title": "अनुवाद चिपकाएँ", "paste_translation_desc": "कहीं और तैयार किया गया अनुवाद चिपकाएँ (ChatGPT, DeepL, कोई मानव अनुवादक)। यह आपके मौजूदा सेगमेंट पर लागू होगा — टाइमिंग और मूल ट्रांसक्रिप्ट अछूते रहते हैं।", diff --git a/electron/src/renderer/src/i18n/locales/id.json b/electron/src/renderer/src/i18n/locales/id.json index b40787d69..ddcbfca96 100644 --- a/electron/src/renderer/src/i18n/locales/id.json +++ b/electron/src/renderer/src/i18n/locales/id.json @@ -811,6 +811,8 @@ "qc_result": "{{flagged}} dari {{total}} baris mungkin perlu didengarkan ulang", "qc_clean": "Semua baris {{total}} cocok dengan skrip", "qc_failed": "Pemeriksaan waktu gagal: {{message}}", + "qc_track_changed": "Sulih suara berubah selama pemeriksaan kualitas. Jalankan kembali pemeriksaan pada trek saat ini.", + "qc_identity_missing": "Identitas segmen tidak jelas. Buat ulang trek ini dengan ID segmen yang unik.", "paste_translation_btn": "Tempel terjemahan", "paste_translation_title": "Tempel terjemahan", "paste_translation_desc": "Tempel terjemahan yang kamu buat di tempat lain (ChatGPT, DeepL, penerjemah manusia). Terjemahan itu dipetakan ke segmen yang sudah ada — pewaktuan dan transkrip asli tidak diubah.", diff --git a/electron/src/renderer/src/i18n/locales/it.json b/electron/src/renderer/src/i18n/locales/it.json index 86cadbd14..47d63d000 100644 --- a/electron/src/renderer/src/i18n/locales/it.json +++ b/electron/src/renderer/src/i18n/locales/it.json @@ -811,6 +811,8 @@ "qc_result": "{{flagged}} delle righe {{total}} potrebbe richiedere un nuovo ascolto", "qc_clean": "Tutte le righe {{total}} corrispondono allo script", "qc_failed": "Controllo cronometraggio fallito: {{message}}", + "qc_track_changed": "Il doppiaggio è cambiato durante il controllo qualità. Ripeti il controllo sulla traccia attuale.", + "qc_identity_missing": "Le identità dei segmenti sono ambigue. Rigenera questa traccia con identificativi di segmento univoci.", "paste_translation_btn": "Incolla traduzione", "paste_translation_title": "Incolla una traduzione", "paste_translation_desc": "Incolla una traduzione prodotta altrove (ChatGPT, DeepL, un traduttore umano). Viene mappata sui segmenti che hai già: tempi e trascrizione originale restano invariati.", diff --git a/electron/src/renderer/src/i18n/locales/ja.json b/electron/src/renderer/src/i18n/locales/ja.json index 8e700c47e..49f2605ac 100644 --- a/electron/src/renderer/src/i18n/locales/ja.json +++ b/electron/src/renderer/src/i18n/locales/ja.json @@ -811,6 +811,8 @@ "qc_result": "{{total}} 行中 {{flagged}} 行は再試行が必要な可能性があります", "qc_clean": "すべての {{total}} 行がスクリプトと一致します", "qc_failed": "タイミング チェックに失敗しました: {{message}}", + "qc_track_changed": "品質チェック中に吹き替えが変更されました。現在のトラックで再度チェックしてください。", + "qc_identity_missing": "セグメントを一意に識別できません。各セグメントに固有のIDを付けて、このトラックを再生成してください。", "paste_translation_btn": "翻訳を貼り付け", "paste_translation_title": "翻訳を貼り付け", "paste_translation_desc": "他所で用意した翻訳(ChatGPT、DeepL、人間の翻訳者)を貼り付けます。既存のセグメントに割り当てられ、タイミングと元の文字起こしはそのまま残ります。", diff --git a/electron/src/renderer/src/i18n/locales/ko.json b/electron/src/renderer/src/i18n/locales/ko.json index ef5bd37dd..76415777e 100644 --- a/electron/src/renderer/src/i18n/locales/ko.json +++ b/electron/src/renderer/src/i18n/locales/ko.json @@ -809,6 +809,8 @@ "qc_result": "{{total}} 줄 중 {{flagged}} 줄을 다시 들어야 할 수 있습니다.", "qc_clean": "모든 {{total}} 줄은 스크립트와 일치합니다.", "qc_failed": "타이밍 확인 실패: {{message}}", + "qc_track_changed": "품질 검사 중 더빙이 변경되었습니다. 현재 트랙에서 검사를 다시 실행하세요.", + "qc_identity_missing": "세그먼트를 명확하게 구분할 수 없습니다. 각 세그먼트에 고유한 ID를 지정하여 이 트랙을 다시 생성하세요.", "paste_translation_btn": "번역 붙여넣기", "paste_translation_title": "번역 붙여넣기", "paste_translation_desc": "다른 곳에서 만든 번역(ChatGPT, DeepL, 사람 번역가)을 붙여넣으세요. 이미 있는 세그먼트에 매핑되며 타이밍과 원본 전사는 그대로 유지됩니다.", diff --git a/electron/src/renderer/src/i18n/locales/nl.json b/electron/src/renderer/src/i18n/locales/nl.json index 4e54205b5..6a5549c6f 100644 --- a/electron/src/renderer/src/i18n/locales/nl.json +++ b/electron/src/renderer/src/i18n/locales/nl.json @@ -809,6 +809,8 @@ "qc_result": "{{flagged}} van {{total}} regels moeten mogelijk opnieuw worden beluisterd", "qc_clean": "Alle {{total}} regels komen overeen met het script", "qc_failed": "Timingcontrole mislukt: {{message}}", + "qc_track_changed": "De nasynchronisatie is tijdens de kwaliteitscontrole gewijzigd. Voer de controle opnieuw uit op het huidige spoor.", + "qc_identity_missing": "De segmenten zijn niet eenduidig te identificeren. Genereer dit spoor opnieuw met unieke segment-ID’s.", "paste_translation_btn": "Vertaling plakken", "paste_translation_title": "Een vertaling plakken", "paste_translation_desc": "Plak een elders gemaakte vertaling (ChatGPT, DeepL, een menselijke vertaler). Die wordt op je bestaande segmenten toegepast — timings en het originele transcript blijven ongewijzigd.", diff --git a/electron/src/renderer/src/i18n/locales/pl.json b/electron/src/renderer/src/i18n/locales/pl.json index 70a0e0199..af04d422f 100644 --- a/electron/src/renderer/src/i18n/locales/pl.json +++ b/electron/src/renderer/src/i18n/locales/pl.json @@ -813,6 +813,8 @@ "qc_result": "{{flagged}} z {{total}} wierszy może wymagać ponownego przesłuchania", "qc_clean": "Wszystkie linie {{total}} odpowiadają skryptowi", "qc_failed": "Kontrola czasu nie powiodła się: {{message}}", + "qc_track_changed": "Dubbing zmienił się podczas kontroli jakości. Uruchom kontrolę ponownie dla bieżącej ścieżki.", + "qc_identity_missing": "Tożsamość segmentów jest niejednoznaczna. Wygeneruj tę ścieżkę ponownie z unikatowymi identyfikatorami segmentów.", "paste_translation_btn": "Wklej tłumaczenie", "paste_translation_title": "Wklej tłumaczenie", "paste_translation_desc": "Wklej tłumaczenie przygotowane gdzie indziej (ChatGPT, DeepL, tłumacz). Zostanie dopasowane do istniejących segmentów — czasy i oryginalna transkrypcja pozostają bez zmian.", diff --git a/electron/src/renderer/src/i18n/locales/pt.json b/electron/src/renderer/src/i18n/locales/pt.json index 54d946758..4769b1a27 100644 --- a/electron/src/renderer/src/i18n/locales/pt.json +++ b/electron/src/renderer/src/i18n/locales/pt.json @@ -811,6 +811,8 @@ "qc_result": "{{flagged}} de {{total}} linhas podem precisar de uma nova escuta", "qc_clean": "Todas as linhas {{total}} correspondem ao script", "qc_failed": "Falha na verificação de tempo: {{message}}", + "qc_track_changed": "A dublagem mudou durante a verificação de qualidade. Execute a verificação novamente na faixa atual.", + "qc_identity_missing": "A identificação dos segmentos é ambígua. Gere esta faixa novamente com identificadores de segmento exclusivos.", "paste_translation_btn": "Colar tradução", "paste_translation_title": "Colar uma tradução", "paste_translation_desc": "Cole uma tradução feita noutro lugar (ChatGPT, DeepL, um tradutor humano). Ela é mapeada nos segmentos que já existem — os tempos e a transcrição original ficam intactos.", diff --git a/electron/src/renderer/src/i18n/locales/ru.json b/electron/src/renderer/src/i18n/locales/ru.json index d1a693b2a..933008f60 100644 --- a/electron/src/renderer/src/i18n/locales/ru.json +++ b/electron/src/renderer/src/i18n/locales/ru.json @@ -813,6 +813,8 @@ "qc_result": "{{flagged}} из {{total}} строк может потребоваться повторное прослушивание", "qc_clean": "Все строки {{total}} соответствуют сценарию", "qc_failed": "Проверка времени не удалась: {{message}}", + "qc_track_changed": "Дубляж изменился во время проверки качества. Повторите проверку текущей дорожки.", + "qc_identity_missing": "Сегменты невозможно однозначно определить. Создайте эту дорожку заново с уникальными идентификаторами сегментов.", "paste_translation_btn": "Вставить перевод", "paste_translation_title": "Вставить перевод", "paste_translation_desc": "Вставьте перевод, сделанный в другом месте (ChatGPT, DeepL, живой переводчик). Он ляжет на уже имеющиеся сегменты — тайминги и исходная расшифровка не изменятся.", diff --git a/electron/src/renderer/src/i18n/locales/sv.json b/electron/src/renderer/src/i18n/locales/sv.json index 140ae24b6..05d44a470 100644 --- a/electron/src/renderer/src/i18n/locales/sv.json +++ b/electron/src/renderer/src/i18n/locales/sv.json @@ -811,6 +811,8 @@ "qc_result": "{{flagged}} av {{total}} rader kan behöva lyssnas om", "qc_clean": "Alla {{total}} rader matchar skriptet", "qc_failed": "Tidskontroll misslyckades: {{message}}", + "qc_track_changed": "Dubbningen ändrades under kvalitetskontrollen. Kör kontrollen igen på det aktuella spåret.", + "qc_identity_missing": "Segmenten kan inte identifieras entydigt. Generera om spåret med unika segment-ID:n.", "paste_translation_btn": "Klistra in översättning", "paste_translation_title": "Klistra in en översättning", "paste_translation_desc": "Klistra in en översättning som gjorts någon annanstans (ChatGPT, DeepL, en mänsklig översättare). Den mappas mot segmenten du redan har — tidkoder och originaltranskriptet rörs inte.", diff --git a/electron/src/renderer/src/i18n/locales/th.json b/electron/src/renderer/src/i18n/locales/th.json index 68e9b14ed..ec217321f 100644 --- a/electron/src/renderer/src/i18n/locales/th.json +++ b/electron/src/renderer/src/i18n/locales/th.json @@ -811,6 +811,8 @@ "qc_result": "{{flagged}} จาก {{total}} บรรทัดอาจต้องฟังซ้ำ", "qc_clean": "{{total}} บรรทัดทั้งหมดตรงกับสคริปต์", "qc_failed": "การตรวจสอบเวลาล้มเหลว: {{message}}", + "qc_track_changed": "เสียงพากย์เปลี่ยนแปลงระหว่างการตรวจสอบคุณภาพ โปรดตรวจสอบแทร็กปัจจุบันอีกครั้ง", + "qc_identity_missing": "ไม่สามารถระบุแต่ละช่วงได้อย่างชัดเจน โปรดสร้างแทร็กนี้ใหม่โดยใช้รหัสที่ไม่ซ้ำกันสำหรับแต่ละช่วง", "paste_translation_btn": "วางคำแปล", "paste_translation_title": "วางคำแปล", "paste_translation_desc": "วางคำแปลที่คุณทำไว้จากที่อื่น (ChatGPT, DeepL หรือผู้แปลที่เป็นคน) ระบบจะจับคู่กับเซกเมนต์ที่มีอยู่แล้ว โดยเวลาและถอดความต้นฉบับยังคงเดิม", diff --git a/electron/src/renderer/src/i18n/locales/tr.json b/electron/src/renderer/src/i18n/locales/tr.json index 7845be0ee..e2e2d0abd 100644 --- a/electron/src/renderer/src/i18n/locales/tr.json +++ b/electron/src/renderer/src/i18n/locales/tr.json @@ -811,6 +811,8 @@ "qc_result": "{{flagged}} / {{total}} satırların yeniden dinlenmesi gerekebilir", "qc_clean": "Tüm {{total}} satırları komut dosyasıyla eşleşiyor", "qc_failed": "Zamanlama kontrolü başarısız oldu: {{message}}", + "qc_track_changed": "Kalite kontrolü sırasında dublaj değişti. Geçerli parçada kontrolü yeniden çalıştırın.", + "qc_identity_missing": "Bölümler kesin olarak tanımlanamıyor. Bu parçayı benzersiz bölüm kimlikleriyle yeniden oluşturun.", "paste_translation_btn": "Çeviriyi yapıştır", "paste_translation_title": "Bir çeviri yapıştır", "paste_translation_desc": "Başka bir yerde hazırladığın çeviriyi yapıştır (ChatGPT, DeepL, bir insan çevirmen). Mevcut segmentlere eşlenir — zamanlamalar ve özgün döküm olduğu gibi kalır.", diff --git a/electron/src/renderer/src/i18n/locales/uk.json b/electron/src/renderer/src/i18n/locales/uk.json index 38038cc78..06a199daa 100644 --- a/electron/src/renderer/src/i18n/locales/uk.json +++ b/electron/src/renderer/src/i18n/locales/uk.json @@ -815,6 +815,8 @@ "qc_result": "{{flagged}} з {{total}} рядків може знадобитися переслухати", "qc_clean": "Усі {{total}} рядки відповідають сценарію", "qc_failed": "Помилка перевірки часу: {{message}}", + "qc_track_changed": "Дубляж змінився під час перевірки якості. Повторіть перевірку поточної доріжки.", + "qc_identity_missing": "Сегменти неможливо однозначно визначити. Створіть цю доріжку заново з унікальними ідентифікаторами сегментів.", "paste_translation_btn": "Вставити переклад", "paste_translation_title": "Вставити переклад", "paste_translation_desc": "Вставте переклад, зроблений деінде (ChatGPT, DeepL, живий перекладач). Він накладеться на наявні сегменти — таймінги й початкова транскрипція лишаться незмінними.", diff --git a/electron/src/renderer/src/i18n/locales/vi.json b/electron/src/renderer/src/i18n/locales/vi.json index 71510fd87..4887838e1 100644 --- a/electron/src/renderer/src/i18n/locales/vi.json +++ b/electron/src/renderer/src/i18n/locales/vi.json @@ -811,6 +811,8 @@ "qc_result": "{{flagged}} trong số {{total}} dòng có thể cần nghe lại", "qc_clean": "Tất cả các dòng {{total}} đều khớp với tập lệnh", "qc_failed": "Kiểm tra thời gian không thành công: {{message}}", + "qc_track_changed": "Bản lồng tiếng đã thay đổi trong khi kiểm tra chất lượng. Hãy kiểm tra lại bản âm thanh hiện tại.", + "qc_identity_missing": "Không thể xác định rõ từng đoạn. Hãy tạo lại bản âm thanh này với mã định danh riêng cho mỗi đoạn.", "paste_translation_btn": "Dán bản dịch", "paste_translation_title": "Dán một bản dịch", "paste_translation_desc": "Dán bản dịch bạn đã làm ở nơi khác (ChatGPT, DeepL, người dịch). Nó sẽ được ánh xạ vào các phân đoạn sẵn có — thời điểm và bản ghi gốc giữ nguyên.", diff --git a/electron/src/renderer/src/i18n/locales/zh-CN.json b/electron/src/renderer/src/i18n/locales/zh-CN.json index a34391f82..3b3b8f604 100644 --- a/electron/src/renderer/src/i18n/locales/zh-CN.json +++ b/electron/src/renderer/src/i18n/locales/zh-CN.json @@ -836,6 +836,8 @@ "qc_result": "{{flagged}} 行(共 {{total}} 行)可能需要重新收听", "qc_clean": "所有 {{total}} 行都与脚本匹配", "qc_failed": "时序检查失败:{{message}}", + "qc_track_changed": "配音在质量检查期间发生了变化。请重新检查当前音轨。", + "qc_identity_missing": "无法唯一识别各个片段。请使用唯一的片段 ID 重新生成此音轨。", "paste_translation_btn": "粘贴译文", "paste_translation_title": "粘贴译文", "paste_translation_desc": "粘贴你在别处完成的译文(ChatGPT、DeepL 或人工译者)。它会映射到已有的片段上——时间轴和原始转写保持不变。", diff --git a/electron/src/renderer/src/i18n/locales/zh-TW.json b/electron/src/renderer/src/i18n/locales/zh-TW.json index f761bf075..404ebe73b 100644 --- a/electron/src/renderer/src/i18n/locales/zh-TW.json +++ b/electron/src/renderer/src/i18n/locales/zh-TW.json @@ -811,6 +811,8 @@ "qc_result": "{{flagged}} 行(共 {{total}} 行)可能需要重新聆聽", "qc_clean": "所有 {{total}} 行都與腳本匹配", "qc_failed": "計時檢查失敗:{{message}}", + "qc_track_changed": "配音在品質檢查期間發生了變更。請重新檢查目前的音軌。", + "qc_identity_missing": "無法唯一識別各個片段。請使用唯一的片段 ID 重新產生此音軌。", "paste_translation_btn": "貼上譯文", "paste_translation_title": "貼上譯文", "paste_translation_desc": "貼上你在別處完成的譯文(ChatGPT、DeepL 或真人譯者)。它會對應到既有的片段上——時間軸與原始逐字稿維持不變。", diff --git a/electron/src/renderer/src/lib/api/client.test.ts b/electron/src/renderer/src/lib/api/client.test.ts index 55516966b..cc0e5e930 100644 --- a/electron/src/renderer/src/lib/api/client.test.ts +++ b/electron/src/renderer/src/lib/api/client.test.ts @@ -28,6 +28,18 @@ afterEach(() => { }); describe('errorFromResponse', () => { + it.each([ + ['dub_qc_track_changed', 'dub.qc_track_changed'], + ['dub_qc_timing_identity_missing', 'dub.qc_identity_missing'], + ['dub_segment_identity_conflict', 'dub.qc_identity_missing'], + ])('localizes %s recovery guidance and retains diagnostics', async (code, key) => { + const translate = vi.spyOn(i18next, 't').mockReturnValue('Localized recovery guidance'); + const detail = { code, message: 'Raw backend diagnostic' }; + const err = await errorFromResponse(new Response(JSON.stringify({ detail }), { status: 409 })); + expect(err.detail).toBe('Localized recovery guidance'); + expect(translate).toHaveBeenCalledWith(key); + expect(err.payload?.detail).toEqual(detail); + }); it('localizes Argos runtime errors and retains diagnostics', async () => { const translate = vi.spyOn(i18next, 't').mockReturnValue('Localized recovery guidance'); const detail = { code: 'argos_runtime_unavailable', message: 'Raw native diagnostic' }; diff --git a/electron/src/renderer/src/lib/api/client.ts b/electron/src/renderer/src/lib/api/client.ts index 15799b848..705589f56 100644 --- a/electron/src/renderer/src/lib/api/client.ts +++ b/electron/src/renderer/src/lib/api/client.ts @@ -74,6 +74,15 @@ export function describeError(err: unknown): string { function detailToString(detail: unknown): string { const localized = languageRejectionMessage(detail, tr) || generationFailureMessage(detail, tr); if (localized) return localized; + if (detail && typeof detail === 'object' && 'code' in detail) { + if (detail.code === 'dub_qc_track_changed') return tr('dub.qc_track_changed'); + if ( + detail.code === 'dub_qc_timing_identity_missing' || + detail.code === 'dub_segment_identity_conflict' + ) { + return tr('dub.qc_identity_missing'); + } + } if ( detail && typeof detail === 'object' && diff --git a/electron/src/renderer/src/lib/audio/streaming-preview.test.ts b/electron/src/renderer/src/lib/audio/streaming-preview.test.ts new file mode 100644 index 000000000..0caee7280 --- /dev/null +++ b/electron/src/renderer/src/lib/audio/streaming-preview.test.ts @@ -0,0 +1,109 @@ +import { afterEach, expect, it, vi } from 'vitest'; +import { createStreamingPreview } from './streaming-preview'; +import { stopActivePlayback } from '@/lib/audio/playback'; +class Context { + static latest: Context; + currentTime = 0; state = 'running'; destination = {}; + sources: Array<{ startAt: number; stop: ReturnType }> = []; + gains: Array<{ changes: number[]; times: number[] }> = []; + close = vi.fn(async () => {}); + constructor(readonly options?: { sampleRate: number }) { Context.latest = this; } + createBuffer(_n: number, length: number) { return { getChannelData: () => new Float32Array(length) }; } + createBufferSource() { + const node = { startAt: 0, connect() {}, disconnect() {}, start(at: number) { node.startAt = at; }, stop: vi.fn() }; + this.sources.push(node); return node; + } + createGain() { + const changes: number[] = []; const times: number[] = []; this.gains.push({ changes, times }); + return { connect() {}, disconnect() {}, gain: { setValueAtTime(v: number, at: number) { changes.push(v); times.push(at); }, linearRampToValueAtTime(v: number, at: number) { changes.push(v); times.push(at); } } }; + } +} +afterEach(() => { stopActivePlayback(); vi.useRealTimers(); vi.unstubAllGlobals(); }); +it('does not stop the final chunk while its scheduled audio is still playing', async () => { + vi.useFakeTimers(); vi.stubGlobal('AudioContext', Context); + const player = createStreamingPreview(24000); + player.appendPcm16Bytes(new Uint8Array(4800).buffer); // 100 ms of PCM + player.finalize(); + const context = Context.latest; + context.currentTime = context.sources[0]!.startAt + 0.09; // 10 ms still scheduled + await vi.advanceTimersByTimeAsync(100); + expect(context.close, 'remaining PCM must drain before completion').not.toHaveBeenCalled(); +}); +it('does not fade a recovered chunk into silence after the preceding source ended', () => { + vi.stubGlobal('AudioContext', Context); + const player = createStreamingPreview(24000, 20); + const pcm = new Uint8Array(4800).buffer; + player.appendPcm16Bytes(pcm); + const context = Context.latest; + context.currentTime = 1; // backend paused; the first 100-ms source has ended + player.appendPcm16Bytes(pcm); + expect(context.sources[1]!.startAt).toBeGreaterThan(context.currentTime); + expect(context.gains[1]!.changes, 'no preceding audio exists to crossfade').not.toContain(0); +}); + +it('preserves an ordinary overlapping crossfade and the PCM device rate', () => { + vi.stubGlobal('AudioContext', Context); + const player = createStreamingPreview(24000, 20); + const pcm = new Uint8Array(4800).buffer; + player.appendPcm16Bytes(pcm); + player.appendPcm16Bytes(pcm); + const context = Context.latest; + expect(context.options?.sampleRate).toBe(24000); + expect(context.sources[0]!.startAt).toBeCloseTo(0.08); + expect(context.sources[1]!.startAt).toBeCloseTo(0.16); + expect(context.gains[0]!.changes).toEqual([1, 0]); + expect(context.gains[1]!.changes).toEqual([0, 1]); +}); +it('limits a recovered crossfade to the preceding source actual remaining overlap', () => { + vi.stubGlobal('AudioContext', Context); + const player = createStreamingPreview(24000, 20); + const pcm = new Uint8Array(4800).buffer; + player.appendPcm16Bytes(pcm); + const context = Context.latest; + context.currentTime = 0.151; + player.appendPcm16Bytes(pcm); + expect(context.sources[1]!.startAt).toBeCloseTo(0.171); + expect(context.gains[1]!.changes).toEqual([0, 1]); + expect(context.gains[1]!.times[1]).toBeCloseTo(0.18); +}); +it('naturally completes after the final scheduled buffer and padding drain', async () => { + vi.useFakeTimers(); vi.stubGlobal('AudioContext', Context); + const done = vi.fn(); + const player = createStreamingPreview(24000, 0, done); + player.appendPcm16Bytes(new Uint8Array(4800).buffer); + player.finalize(); + const context = Context.latest; + context.currentTime = context.sources[0]!.startAt + 0.1; + await vi.advanceTimersByTimeAsync(100); + expect(done).not.toHaveBeenCalled(); + context.currentTime += 0.06; + await vi.advanceTimersByTimeAsync(100); + expect(context.close).toHaveBeenCalledOnce(); + expect(done).toHaveBeenCalledOnce(); +}); +it('keeps explicit cancellation immediate and idempotent', () => { + vi.stubGlobal('AudioContext', Context); + const done = vi.fn(); + const player = createStreamingPreview(24000, 0, done); + player.appendPcm16Bytes(new Uint8Array(4800).buffer); + stopActivePlayback(); player.fail(); + expect(Context.latest.close).toHaveBeenCalledOnce(); + expect(Context.latest.sources[0]!.stop).toHaveBeenCalledOnce(); + expect(done).toHaveBeenCalledOnce(); +}); +it('completes an empty stream without waiting for a device-clock deadline', () => { + vi.stubGlobal('AudioContext', Context); + const done = vi.fn(); + createStreamingPreview(24000, 0, done).finalize(); + expect(Context.latest.close).toHaveBeenCalledOnce(); + expect(done).toHaveBeenCalledOnce(); +}); + +it('anchors the first chunk after a delayed PCM response with the same scheduling lead', () => { + vi.stubGlobal('AudioContext', Context); + const player = createStreamingPreview(24000); + const context = Context.latest; + context.currentTime = 4; + player.appendPcm16Bytes(new Uint8Array(4800).buffer); + expect(context.sources[0]!.startAt - context.currentTime).toBeCloseTo(0.08); +}); diff --git a/electron/src/renderer/src/lib/audio/streaming-preview.ts b/electron/src/renderer/src/lib/audio/streaming-preview.ts index 542078015..4d5c45149 100644 --- a/electron/src/renderer/src/lib/audio/streaming-preview.ts +++ b/electron/src/renderer/src/lib/audio/streaming-preview.ts @@ -42,9 +42,11 @@ export function createStreamingPreview( const Context = window.AudioContext || (window as AudioContextWindow).webkitAudioContext; if (!Context) throw new Error('Web Audio is unavailable'); const context = new Context({ sampleRate }); - const nodes: Array<{ source: AudioBufferSourceNode; gain: GainNode }> = []; + const nodes: Array<{ source: AudioBufferSourceNode; gain: GainNode; endsAt: number }> = []; const crossfadeSeconds = Math.max(0, crossfadeMs) / 1000; - let nextStart = context.currentTime + 0.03; + const startLeadSeconds = 0.08; + const endPaddingSeconds = 0.05; + let nextStart = context.currentTime; let previousDuration = 0; let finished = false; let complete = false; @@ -73,7 +75,7 @@ export function createStreamingPreview( release = claimPlayback(stop, 'output'); if (context.state === 'suspended') void context.resume().catch(() => {}); timer = setInterval(() => { - if (complete && context.currentTime >= nextStart - 0.02) stop(); + if (complete && context.currentTime >= nextStart + endPaddingSeconds) stop(); }, 100); return { @@ -94,9 +96,12 @@ export function createStreamingPreview( if (finished) return; if (!samples.length) return; const duration = samples.length / sampleRate; - const fade = nodes.length ? Math.min(crossfadeSeconds, previousDuration, duration) : 0; - let start = nodes.length ? nextStart - fade : nextStart; + let fade = nodes.length ? Math.min(crossfadeSeconds, previousDuration, duration) : 0; + let start = nodes.length ? nextStart - fade : context.currentTime + startLeadSeconds; if (start < context.currentTime + 0.01) start = context.currentTime + 0.02; + const previous = nodes.at(-1); + // A late chunk can crossfade only against audio still scheduled at its start. + fade = Math.min(fade, Math.max(0, (previous?.endsAt ?? start) - start)); const buffer = context.createBuffer(1, samples.length, sampleRate); buffer.getChannelData(0).set(samples); @@ -106,14 +111,13 @@ export function createStreamingPreview( source.connect(gain); gain.connect(context.destination); if (fade > 0) { - const previous = nodes.at(-1); previous?.gain.gain.setValueAtTime(1, start); previous?.gain.gain.linearRampToValueAtTime(0, start + fade); gain.gain.setValueAtTime(0, start); gain.gain.linearRampToValueAtTime(1, start + fade); } source.start(start); - nodes.push({ source, gain }); + nodes.push({ source, gain, endsAt: start + duration }); previousDuration = duration; nextStart = start + duration; } diff --git a/electron/src/renderer/src/lib/store/takes.test.ts b/electron/src/renderer/src/lib/store/takes.test.ts index 68709b6fa..7a093398d 100644 --- a/electron/src/renderer/src/lib/store/takes.test.ts +++ b/electron/src/renderer/src/lib/store/takes.test.ts @@ -1,6 +1,7 @@ import { beforeEach, describe, expect, it } from 'vitest'; -import { DEFAULT_CLONE_SETTINGS } from './clone-settings'; -import { rememberTake, takeSettings } from './takes'; +import { cloneSettingsStore, DEFAULT_CLONE_SETTINGS, patchCloneSettings } from './clone-settings'; +import { rememberTake, reuseTake, takeSettings } from './takes'; +import { toGenerateForm } from '@/lib/api/generate'; import type { HistoryItem } from '@/lib/api/types'; const item = { id: 'take', @@ -9,7 +10,10 @@ const item = { instruct: '', profile_id: 'voice', } as HistoryItem; -beforeEach(() => localStorage.clear()); +beforeEach(() => { + localStorage.clear(); + patchCloneSettings({ ...DEFAULT_CLONE_SETTINGS }); +}); describe('take metadata', () => { it('restores original generation controls without overwriting application preferences', () => { rememberTake('take', { ...DEFAULT_CLONE_SETTINGS, speed: 1.5, steps: 24, autoPlay: false }); @@ -31,4 +35,37 @@ describe('take metadata', () => { selectedProfileId: 'voice', }); }); + it.each([ + [16, 'broadcast'], + [24, 'raw'], + [32, 'raw'], + ] as const)('restores %i-bit %s quality through the next generation request', async (wavBits, effectPreset) => { + rememberTake(item.id, { ...DEFAULT_CLONE_SETTINGS, wavBits, effectPreset, speed: 1.5 }); + patchCloneSettings({ + wavBits: wavBits === 16 ? 32 : 16, + effectPreset: effectPreset === 'broadcast' ? 'raw' : 'broadcast', + autoPlay: true, + showOverrides: true, + }); + await reuseTake(item); + const settings = cloneSettingsStore.state; + const body = toGenerateForm({ ...settings, profileId: settings.selectedProfileId }); + expect(body.get('wav_bits')).toBe(String(wavBits)); + expect(body.get('effect_preset')).toBe(effectPreset); + expect(settings.speed).toBe(1.5); + expect(settings.autoPlay).toBe(true); + expect(settings.showOverrides).toBe(true); + }); + it.each([ + null, + JSON.stringify({ take: { steps: 24 } }), + JSON.stringify({ take: { wavBits: 8, effectPreset: 'unknown' } }), + JSON.stringify({ take: { wavBits: '24', effectPreset: false } }), + ])('preserves current quality when stored quality is absent or invalid: %s', async (metadata) => { + if (metadata !== null) localStorage.setItem('voicestudio.take-settings.v1', metadata); + patchCloneSettings({ wavBits: 32, effectPreset: 'raw' }); + await reuseTake(item); + expect(cloneSettingsStore.state.wavBits).toBe(32); + expect(cloneSettingsStore.state.effectPreset).toBe('raw'); + }); }); diff --git a/electron/src/renderer/src/lib/store/takes.ts b/electron/src/renderer/src/lib/store/takes.ts index 45e81e07e..4131ea465 100644 --- a/electron/src/renderer/src/lib/store/takes.ts +++ b/electron/src/renderer/src/lib/store/takes.ts @@ -55,6 +55,10 @@ export function takeSettings(item: HistoryItem): Partial { if (typeof raw[name] === 'boolean') safe[name] = raw[name]; for (const name of ['refText', 'duration']) if (typeof raw[name] === 'string') safe[name] = raw[name]; + if (raw.wavBits === 16 || raw.wavBits === 24 || raw.wavBits === 32) + safe.wavBits = raw.wavBits; + if (raw.effectPreset === 'broadcast' || raw.effectPreset === 'raw') + safe.effectPreset = raw.effectPreset; return { ...safe, ...fallback }; } catch { return fallback; diff --git a/electron/src/shared/i18n/locales/ar.json b/electron/src/shared/i18n/locales/ar.json index 3c078a5ca..69e115710 100644 --- a/electron/src/shared/i18n/locales/ar.json +++ b/electron/src/shared/i18n/locales/ar.json @@ -1327,6 +1327,8 @@ "qc_result": "{{flagged}} من {{total}} قد تحتاج الأسطر إلى إعادة الاستماع", "qc_clean": "جميع أسطر {{total}} تتطابق مع النص", "qc_failed": "فشل التحقق من التوقيت: {{message}}", + "qc_track_changed": "تغيرت الدبلجة أثناء فحص الجودة. أعد الفحص على المسار الحالي.", + "qc_identity_missing": "هويات المقاطع غير واضحة. أعد إنشاء هذا المسار بمعرّفات فريدة للمقاطع.", "paste_translation_btn": "لصق ترجمة", "paste_translation_title": "لصق ترجمة", "paste_translation_desc": "الصق ترجمة أعددتها في مكان آخر (ChatGPT أو DeepL أو مترجم بشري). ستُطابَق مع المقاطع الموجودة لديك — تبقى التوقيتات والنص الأصلي دون تغيير.", diff --git a/electron/src/shared/i18n/locales/de.json b/electron/src/shared/i18n/locales/de.json index 2539c92bf..0f1403301 100644 --- a/electron/src/shared/i18n/locales/de.json +++ b/electron/src/shared/i18n/locales/de.json @@ -1325,6 +1325,8 @@ "qc_result": "{{flagged}} von {{total}} Zeilen müssen möglicherweise erneut abgehört werden", "qc_clean": "Alle {{total}}-Zeilen stimmen mit dem Skript überein", "qc_failed": "Zeitprüfung fehlgeschlagen: {{message}}", + "qc_track_changed": "Die Synchronisation wurde während der Qualitätsprüfung geändert. Prüfe die aktuelle Spur erneut.", + "qc_identity_missing": "Die Segmentzuordnung ist uneindeutig. Erzeuge diese Spur mit eindeutigen Segment-IDs neu.", "paste_translation_btn": "Übersetzung einfügen", "paste_translation_title": "Übersetzung einfügen", "paste_translation_desc": "Füge eine anderswo erstellte Übersetzung ein (ChatGPT, DeepL, ein menschlicher Übersetzer). Sie wird den vorhandenen Segmenten zugeordnet — Timings und Originaltranskript bleiben unverändert.", diff --git a/electron/src/shared/i18n/locales/en.json b/electron/src/shared/i18n/locales/en.json index 663e90a9f..d78b9e758 100644 --- a/electron/src/shared/i18n/locales/en.json +++ b/electron/src/shared/i18n/locales/en.json @@ -1540,6 +1540,8 @@ "qc_result": "{{flagged}} of {{total}} lines may need a re-listen", "qc_clean": "All {{total}} lines match the script", "qc_failed": "Timing check failed: {{message}}", + "qc_track_changed": "The dub changed during the quality check. Run the check again on the current track.", + "qc_identity_missing": "Segment identities are ambiguous. Regenerate this track with unique segment IDs.", "export_btn": "Export…", "prep_stop": "Stop", "install_progress": "Installing {{engine}}…", diff --git a/electron/src/shared/i18n/locales/es.json b/electron/src/shared/i18n/locales/es.json index 84ae65593..d9b754dc9 100644 --- a/electron/src/shared/i18n/locales/es.json +++ b/electron/src/shared/i18n/locales/es.json @@ -1325,6 +1325,8 @@ "qc_result": "Es posible que sea necesario volver a escuchar {{flagged}} de {{total}} líneas", "qc_clean": "Todas las líneas {{total}} coinciden con el guión.", "qc_failed": "Error en la verificación de tiempo: {{message}}", + "qc_track_changed": "El doblaje cambió durante la comprobación de calidad. Repite la comprobación en la pista actual.", + "qc_identity_missing": "La identidad de los segmentos es ambigua. Regenera esta pista con identificadores de segmento únicos.", "paste_translation_btn": "Pegar traducción", "paste_translation_title": "Pegar una traducción", "paste_translation_desc": "Pega una traducción hecha en otro sitio (ChatGPT, DeepL, un traductor humano). Se asigna a los segmentos que ya tienes: los tiempos y la transcripción original no cambian.", diff --git a/electron/src/shared/i18n/locales/fr.json b/electron/src/shared/i18n/locales/fr.json index bb4a8f102..9bb824f66 100644 --- a/electron/src/shared/i18n/locales/fr.json +++ b/electron/src/shared/i18n/locales/fr.json @@ -1325,6 +1325,8 @@ "qc_result": "{{flagged}} lignes sur {{total}} peuvent nécessiter une réécoute", "qc_clean": "Toutes les lignes {{total}} correspondent au script", "qc_failed": "Échec de la vérification du timing : {{message}}", + "qc_track_changed": "Le doublage a changé pendant le contrôle qualité. Relancez le contrôle sur la piste actuelle.", + "qc_identity_missing": "Les segments ne sont pas identifiés de façon univoque. Régénérez cette piste avec des identifiants de segment uniques.", "paste_translation_btn": "Coller une traduction", "paste_translation_title": "Coller une traduction", "paste_translation_desc": "Collez une traduction réalisée ailleurs (ChatGPT, DeepL, un traducteur humain). Elle est appliquée aux segments existants — les timings et la transcription d'origine restent intacts.", diff --git a/electron/src/shared/i18n/locales/hi.json b/electron/src/shared/i18n/locales/hi.json index 531f0720c..f5cc74b68 100644 --- a/electron/src/shared/i18n/locales/hi.json +++ b/electron/src/shared/i18n/locales/hi.json @@ -1325,6 +1325,8 @@ "qc_result": "{{total}} पंक्तियों में से {{flagged}} को दोबारा सुनने की आवश्यकता हो सकती है", "qc_clean": "सभी {{total}} पंक्तियाँ स्क्रिप्ट से मेल खाती हैं", "qc_failed": "समय की जाँच विफल: {{message}}", + "qc_track_changed": "गुणवत्ता जाँच के दौरान डबिंग बदल गई। मौजूदा ट्रैक पर जाँच दोबारा चलाएँ।", + "qc_identity_missing": "सेगमेंट की पहचान स्पष्ट नहीं है। हर सेगमेंट को अलग आईडी देकर इस ट्रैक को फिर से जनरेट करें।", "paste_translation_btn": "अनुवाद चिपकाएँ", "paste_translation_title": "अनुवाद चिपकाएँ", "paste_translation_desc": "कहीं और तैयार किया गया अनुवाद चिपकाएँ (ChatGPT, DeepL, कोई मानव अनुवादक)। यह आपके मौजूदा सेगमेंट पर लागू होगा — टाइमिंग और मूल ट्रांसक्रिप्ट अछूते रहते हैं।", diff --git a/electron/src/shared/i18n/locales/id.json b/electron/src/shared/i18n/locales/id.json index 639e3adb2..6bbfded69 100644 --- a/electron/src/shared/i18n/locales/id.json +++ b/electron/src/shared/i18n/locales/id.json @@ -1327,6 +1327,8 @@ "qc_result": "{{flagged}} dari {{total}} baris mungkin perlu didengarkan ulang", "qc_clean": "Semua baris {{total}} cocok dengan skrip", "qc_failed": "Pemeriksaan waktu gagal: {{message}}", + "qc_track_changed": "Sulih suara berubah selama pemeriksaan kualitas. Jalankan kembali pemeriksaan pada trek saat ini.", + "qc_identity_missing": "Identitas segmen tidak jelas. Buat ulang trek ini dengan ID segmen yang unik.", "paste_translation_btn": "Tempel terjemahan", "paste_translation_title": "Tempel terjemahan", "paste_translation_desc": "Tempel terjemahan yang kamu buat di tempat lain (ChatGPT, DeepL, penerjemah manusia). Terjemahan itu dipetakan ke segmen yang sudah ada — pewaktuan dan transkrip asli tidak diubah.", diff --git a/electron/src/shared/i18n/locales/it.json b/electron/src/shared/i18n/locales/it.json index 14ecb2a0a..8fac3eaed 100644 --- a/electron/src/shared/i18n/locales/it.json +++ b/electron/src/shared/i18n/locales/it.json @@ -1325,6 +1325,8 @@ "qc_result": "{{flagged}} delle righe {{total}} potrebbe richiedere un nuovo ascolto", "qc_clean": "Tutte le righe {{total}} corrispondono allo script", "qc_failed": "Controllo cronometraggio fallito: {{message}}", + "qc_track_changed": "Il doppiaggio è cambiato durante il controllo qualità. Ripeti il controllo sulla traccia attuale.", + "qc_identity_missing": "Le identità dei segmenti sono ambigue. Rigenera questa traccia con identificativi di segmento univoci.", "paste_translation_btn": "Incolla traduzione", "paste_translation_title": "Incolla una traduzione", "paste_translation_desc": "Incolla una traduzione prodotta altrove (ChatGPT, DeepL, un traduttore umano). Viene mappata sui segmenti che hai già: tempi e trascrizione originale restano invariati.", diff --git a/electron/src/shared/i18n/locales/ja.json b/electron/src/shared/i18n/locales/ja.json index dfde39e6c..4eab222fe 100644 --- a/electron/src/shared/i18n/locales/ja.json +++ b/electron/src/shared/i18n/locales/ja.json @@ -1327,6 +1327,8 @@ "qc_result": "{{total}} 行中 {{flagged}} 行は再試行が必要な可能性があります", "qc_clean": "すべての {{total}} 行がスクリプトと一致します", "qc_failed": "タイミング チェックに失敗しました: {{message}}", + "qc_track_changed": "品質チェック中に吹き替えが変更されました。現在のトラックで再度チェックしてください。", + "qc_identity_missing": "セグメントを一意に識別できません。各セグメントに固有のIDを付けて、このトラックを再生成してください。", "paste_translation_btn": "翻訳を貼り付け", "paste_translation_title": "翻訳を貼り付け", "paste_translation_desc": "他所で用意した翻訳(ChatGPT、DeepL、人間の翻訳者)を貼り付けます。既存のセグメントに割り当てられ、タイミングと元の文字起こしはそのまま残ります。", diff --git a/electron/src/shared/i18n/locales/ko.json b/electron/src/shared/i18n/locales/ko.json index 5bb1887a6..190c2d426 100644 --- a/electron/src/shared/i18n/locales/ko.json +++ b/electron/src/shared/i18n/locales/ko.json @@ -1616,6 +1616,8 @@ "qc_result": "{{total}} 줄 중 {{flagged}} 줄을 다시 들어야 할 수 있습니다.", "qc_clean": "모든 {{total}} 줄은 스크립트와 일치합니다.", "qc_failed": "타이밍 확인 실패: {{message}}", + "qc_track_changed": "품질 검사 중 더빙이 변경되었습니다. 현재 트랙에서 검사를 다시 실행하세요.", + "qc_identity_missing": "세그먼트를 명확하게 구분할 수 없습니다. 각 세그먼트에 고유한 ID를 지정하여 이 트랙을 다시 생성하세요.", "paste_translation_btn": "번역 붙여넣기", "paste_translation_title": "번역 붙여넣기", "paste_translation_desc": "다른 곳에서 만든 번역(ChatGPT, DeepL, 사람 번역가)을 붙여넣으세요. 이미 있는 세그먼트에 매핑되며 타이밍과 원본 전사는 그대로 유지됩니다.", diff --git a/electron/src/shared/i18n/locales/nl.json b/electron/src/shared/i18n/locales/nl.json index 483d81781..2054a81f0 100644 --- a/electron/src/shared/i18n/locales/nl.json +++ b/electron/src/shared/i18n/locales/nl.json @@ -1325,6 +1325,8 @@ "qc_result": "{{flagged}} van {{total}} regels moeten mogelijk opnieuw worden beluisterd", "qc_clean": "Alle {{total}} regels komen overeen met het script", "qc_failed": "Timingcontrole mislukt: {{message}}", + "qc_track_changed": "De nasynchronisatie is tijdens de kwaliteitscontrole gewijzigd. Voer de controle opnieuw uit op het huidige spoor.", + "qc_identity_missing": "De segmenten zijn niet eenduidig te identificeren. Genereer dit spoor opnieuw met unieke segment-ID’s.", "paste_translation_btn": "Vertaling plakken", "paste_translation_title": "Een vertaling plakken", "paste_translation_desc": "Plak een elders gemaakte vertaling (ChatGPT, DeepL, een menselijke vertaler). Die wordt op je bestaande segmenten toegepast — timings en het originele transcript blijven ongewijzigd.", diff --git a/electron/src/shared/i18n/locales/pl.json b/electron/src/shared/i18n/locales/pl.json index 474071a09..dc729d5e6 100644 --- a/electron/src/shared/i18n/locales/pl.json +++ b/electron/src/shared/i18n/locales/pl.json @@ -1325,6 +1325,8 @@ "qc_result": "{{flagged}} z {{total}} wierszy może wymagać ponownego przesłuchania", "qc_clean": "Wszystkie linie {{total}} odpowiadają skryptowi", "qc_failed": "Kontrola czasu nie powiodła się: {{message}}", + "qc_track_changed": "Dubbing zmienił się podczas kontroli jakości. Uruchom kontrolę ponownie dla bieżącej ścieżki.", + "qc_identity_missing": "Tożsamość segmentów jest niejednoznaczna. Wygeneruj tę ścieżkę ponownie z unikatowymi identyfikatorami segmentów.", "paste_translation_btn": "Wklej tłumaczenie", "paste_translation_title": "Wklej tłumaczenie", "paste_translation_desc": "Wklej tłumaczenie przygotowane gdzie indziej (ChatGPT, DeepL, tłumacz). Zostanie dopasowane do istniejących segmentów — czasy i oryginalna transkrypcja pozostają bez zmian.", diff --git a/electron/src/shared/i18n/locales/pt.json b/electron/src/shared/i18n/locales/pt.json index 8ef42a99d..405202d1f 100644 --- a/electron/src/shared/i18n/locales/pt.json +++ b/electron/src/shared/i18n/locales/pt.json @@ -1325,6 +1325,8 @@ "qc_result": "{{flagged}} de {{total}} linhas podem precisar de uma nova escuta", "qc_clean": "Todas as linhas {{total}} correspondem ao script", "qc_failed": "Falha na verificação de tempo: {{message}}", + "qc_track_changed": "A dublagem mudou durante a verificação de qualidade. Execute a verificação novamente na faixa atual.", + "qc_identity_missing": "A identificação dos segmentos é ambígua. Gere esta faixa novamente com identificadores de segmento exclusivos.", "paste_translation_btn": "Colar tradução", "paste_translation_title": "Colar uma tradução", "paste_translation_desc": "Cole uma tradução feita noutro lugar (ChatGPT, DeepL, um tradutor humano). Ela é mapeada nos segmentos que já existem — os tempos e a transcrição original ficam intactos.", diff --git a/electron/src/shared/i18n/locales/ru.json b/electron/src/shared/i18n/locales/ru.json index 57ec11c3f..acda02ecf 100644 --- a/electron/src/shared/i18n/locales/ru.json +++ b/electron/src/shared/i18n/locales/ru.json @@ -1325,6 +1325,8 @@ "qc_result": "{{flagged}} из {{total}} строк может потребоваться повторное прослушивание", "qc_clean": "Все строки {{total}} соответствуют сценарию", "qc_failed": "Проверка времени не удалась: {{message}}", + "qc_track_changed": "Дубляж изменился во время проверки качества. Повторите проверку текущей дорожки.", + "qc_identity_missing": "Сегменты невозможно однозначно определить. Создайте эту дорожку заново с уникальными идентификаторами сегментов.", "paste_translation_btn": "Вставить перевод", "paste_translation_title": "Вставить перевод", "paste_translation_desc": "Вставьте перевод, сделанный в другом месте (ChatGPT, DeepL, живой переводчик). Он ляжет на уже имеющиеся сегменты — тайминги и исходная расшифровка не изменятся.", diff --git a/electron/src/shared/i18n/locales/sv.json b/electron/src/shared/i18n/locales/sv.json index d68822c9c..bc4de5684 100644 --- a/electron/src/shared/i18n/locales/sv.json +++ b/electron/src/shared/i18n/locales/sv.json @@ -1327,6 +1327,8 @@ "qc_result": "{{flagged}} av {{total}} rader kan behöva lyssnas om", "qc_clean": "Alla {{total}} rader matchar skriptet", "qc_failed": "Tidskontroll misslyckades: {{message}}", + "qc_track_changed": "Dubbningen ändrades under kvalitetskontrollen. Kör kontrollen igen på det aktuella spåret.", + "qc_identity_missing": "Segmenten kan inte identifieras entydigt. Generera om spåret med unika segment-ID:n.", "paste_translation_btn": "Klistra in översättning", "paste_translation_title": "Klistra in en översättning", "paste_translation_desc": "Klistra in en översättning som gjorts någon annanstans (ChatGPT, DeepL, en mänsklig översättare). Den mappas mot segmenten du redan har — tidkoder och originaltranskriptet rörs inte.", diff --git a/electron/src/shared/i18n/locales/th.json b/electron/src/shared/i18n/locales/th.json index 93f38ec32..f8ec751ca 100644 --- a/electron/src/shared/i18n/locales/th.json +++ b/electron/src/shared/i18n/locales/th.json @@ -1327,6 +1327,8 @@ "qc_result": "{{flagged}} จาก {{total}} บรรทัดอาจต้องฟังซ้ำ", "qc_clean": "{{total}} บรรทัดทั้งหมดตรงกับสคริปต์", "qc_failed": "การตรวจสอบเวลาล้มเหลว: {{message}}", + "qc_track_changed": "เสียงพากย์เปลี่ยนแปลงระหว่างการตรวจสอบคุณภาพ โปรดตรวจสอบแทร็กปัจจุบันอีกครั้ง", + "qc_identity_missing": "ไม่สามารถระบุแต่ละช่วงได้อย่างชัดเจน โปรดสร้างแทร็กนี้ใหม่โดยใช้รหัสที่ไม่ซ้ำกันสำหรับแต่ละช่วง", "paste_translation_btn": "วางคำแปล", "paste_translation_title": "วางคำแปล", "paste_translation_desc": "วางคำแปลที่คุณทำไว้จากที่อื่น (ChatGPT, DeepL หรือผู้แปลที่เป็นคน) ระบบจะจับคู่กับเซกเมนต์ที่มีอยู่แล้ว โดยเวลาและถอดความต้นฉบับยังคงเดิม", diff --git a/electron/src/shared/i18n/locales/tr.json b/electron/src/shared/i18n/locales/tr.json index 1e3301f0f..aca1d35d7 100644 --- a/electron/src/shared/i18n/locales/tr.json +++ b/electron/src/shared/i18n/locales/tr.json @@ -1327,6 +1327,8 @@ "qc_result": "{{flagged}} / {{total}} satırların yeniden dinlenmesi gerekebilir", "qc_clean": "Tüm {{total}} satırları komut dosyasıyla eşleşiyor", "qc_failed": "Zamanlama kontrolü başarısız oldu: {{message}}", + "qc_track_changed": "Kalite kontrolü sırasında dublaj değişti. Geçerli parçada kontrolü yeniden çalıştırın.", + "qc_identity_missing": "Bölümler kesin olarak tanımlanamıyor. Bu parçayı benzersiz bölüm kimlikleriyle yeniden oluşturun.", "paste_translation_btn": "Çeviriyi yapıştır", "paste_translation_title": "Bir çeviri yapıştır", "paste_translation_desc": "Başka bir yerde hazırladığın çeviriyi yapıştır (ChatGPT, DeepL, bir insan çevirmen). Mevcut segmentlere eşlenir — zamanlamalar ve özgün döküm olduğu gibi kalır.", diff --git a/electron/src/shared/i18n/locales/uk.json b/electron/src/shared/i18n/locales/uk.json index 714d599de..c268f6e6b 100644 --- a/electron/src/shared/i18n/locales/uk.json +++ b/electron/src/shared/i18n/locales/uk.json @@ -1327,6 +1327,8 @@ "qc_result": "{{flagged}} з {{total}} рядків може знадобитися переслухати", "qc_clean": "Усі {{total}} рядки відповідають сценарію", "qc_failed": "Помилка перевірки часу: {{message}}", + "qc_track_changed": "Дубляж змінився під час перевірки якості. Повторіть перевірку поточної доріжки.", + "qc_identity_missing": "Сегменти неможливо однозначно визначити. Створіть цю доріжку заново з унікальними ідентифікаторами сегментів.", "paste_translation_btn": "Вставити переклад", "paste_translation_title": "Вставити переклад", "paste_translation_desc": "Вставте переклад, зроблений деінде (ChatGPT, DeepL, живий перекладач). Він накладеться на наявні сегменти — таймінги й початкова транскрипція лишаться незмінними.", diff --git a/electron/src/shared/i18n/locales/vi.json b/electron/src/shared/i18n/locales/vi.json index 4111af3e9..963c6dbdd 100644 --- a/electron/src/shared/i18n/locales/vi.json +++ b/electron/src/shared/i18n/locales/vi.json @@ -1327,6 +1327,8 @@ "qc_result": "{{flagged}} trong số {{total}} dòng có thể cần nghe lại", "qc_clean": "Tất cả các dòng {{total}} đều khớp với tập lệnh", "qc_failed": "Kiểm tra thời gian không thành công: {{message}}", + "qc_track_changed": "Bản lồng tiếng đã thay đổi trong khi kiểm tra chất lượng. Hãy kiểm tra lại bản âm thanh hiện tại.", + "qc_identity_missing": "Không thể xác định rõ từng đoạn. Hãy tạo lại bản âm thanh này với mã định danh riêng cho mỗi đoạn.", "paste_translation_btn": "Dán bản dịch", "paste_translation_title": "Dán một bản dịch", "paste_translation_desc": "Dán bản dịch bạn đã làm ở nơi khác (ChatGPT, DeepL, người dịch). Nó sẽ được ánh xạ vào các phân đoạn sẵn có — thời điểm và bản ghi gốc giữ nguyên.", diff --git a/electron/src/shared/i18n/locales/zh-CN.json b/electron/src/shared/i18n/locales/zh-CN.json index 3acaf2e90..eda26df59 100644 --- a/electron/src/shared/i18n/locales/zh-CN.json +++ b/electron/src/shared/i18n/locales/zh-CN.json @@ -1578,6 +1578,8 @@ "qc_result": "{{flagged}} 行(共 {{total}} 行)可能需要重新收听", "qc_clean": "所有 {{total}} 行都与脚本匹配", "qc_failed": "时序检查失败:{{message}}", + "qc_track_changed": "配音在质量检查期间发生了变化。请重新检查当前音轨。", + "qc_identity_missing": "无法唯一识别各个片段。请使用唯一的片段 ID 重新生成此音轨。", "paste_translation_btn": "粘贴译文", "paste_translation_title": "粘贴译文", "paste_translation_desc": "粘贴你在别处完成的译文(ChatGPT、DeepL 或人工译者)。它会映射到已有的片段上——时间轴和原始转写保持不变。", diff --git a/electron/src/shared/i18n/locales/zh-TW.json b/electron/src/shared/i18n/locales/zh-TW.json index eb5380f44..f0931b0e6 100644 --- a/electron/src/shared/i18n/locales/zh-TW.json +++ b/electron/src/shared/i18n/locales/zh-TW.json @@ -1327,6 +1327,8 @@ "qc_result": "{{flagged}} 行(共 {{total}} 行)可能需要重新聆聽", "qc_clean": "所有 {{total}} 行都與腳本匹配", "qc_failed": "計時檢查失敗:{{message}}", + "qc_track_changed": "配音在品質檢查期間發生了變更。請重新檢查目前的音軌。", + "qc_identity_missing": "無法唯一識別各個片段。請使用唯一的片段 ID 重新產生此音軌。", "paste_translation_btn": "貼上譯文", "paste_translation_title": "貼上譯文", "paste_translation_desc": "貼上你在別處完成的譯文(ChatGPT、DeepL 或真人譯者)。它會對應到既有的片段上——時間軸與原始逐字稿維持不變。", diff --git a/electron/src/shared/utils/audioTrim.js b/electron/src/shared/utils/audioTrim.js index 238e770a2..fc08f3b8c 100644 --- a/electron/src/shared/utils/audioTrim.js +++ b/electron/src/shared/utils/audioTrim.js @@ -166,45 +166,14 @@ export async function computePeaksAsync(channel, buckets = DEFAULT_PEAK_BUCKETS, return peaks; } -async function probeDuration(file) { - return new Promise((resolve, reject) => { - const url = URL.createObjectURL(file); - const a = new Audio(); - a.preload = 'metadata'; - const cleanup = () => { - try { - URL.revokeObjectURL(url); - } catch {} - }; - a.addEventListener( - 'loadedmetadata', - () => { - const d = a.duration; - cleanup(); - resolve(isFinite(d) ? d : 0); - }, - { once: true }, - ); - a.addEventListener( - 'error', - () => { - cleanup(); - reject(new Error('metadata failed')); - }, - { once: true }, - ); - a.src = url; - }); -} - +/** Decode directly: media-element metadata may be unavailable for otherwise decodable audio. */ export async function decodeToMonoLowRate(file, targetSR = 22050) { - const duration = await probeDuration(file); const arr = await file.arrayBuffer(); - const len = Math.max(1, Math.ceil(Math.max(0.001, duration) * targetSR)); const Offline = window.OfflineAudioContext || window.webkitOfflineAudioContext; - const offline = new Offline(1, len, targetSR); - const buf = await offline.decodeAudioData(arr); - return buf; + // decodeAudioData returns the complete decoded buffer, independently of the + // context's render length, and resamples it to the context's sample rate. + const offline = new Offline(1, 1, targetSR); + return offline.decodeAudioData(arr); } // Playhead position for a selection [startSec, endSec] that has been playing diff --git a/electron/src/shared/utils/audioTrimDecode.test.ts b/electron/src/shared/utils/audioTrimDecode.test.ts new file mode 100644 index 000000000..e9987cad2 --- /dev/null +++ b/electron/src/shared/utils/audioTrimDecode.test.ts @@ -0,0 +1,43 @@ +// @vitest-environment jsdom +import { afterEach, expect, it, vi } from 'vitest'; +import { decodeToMonoLowRate } from './audioTrim'; +afterEach(() => vi.unstubAllGlobals()); + +function sourceBlob(bytes: number[]) { + const data = new Uint8Array(bytes); + const file = new Blob([data]); + // jsdom lacks Blob.arrayBuffer; keep the byte source controlled while using + // the renderer's real DOM setup, just as the hosted shared suite does. + Object.defineProperty(file, 'arrayBuffer', { value: async () => data.buffer }); + return file; +} + +it('decodes a trim source without waiting for a media-element metadata event', async () => { + vi.useFakeTimers(); + try { + const metadata = vi.fn(); + vi.stubGlobal('Audio', class { constructor() { metadata(); } addEventListener() {} }); + const decoded = { duration: 2, length: 44100, sampleRate: 22050 }; + const decode = vi.fn(async () => decoded); + const context = vi.fn(function (this: unknown) { return { decodeAudioData: decode }; }); + vi.stubGlobal('window', { OfflineAudioContext: context }); + const file = sourceBlob([1, 2, 3]); + const pending = decodeToMonoLowRate(file); + const bounded = Promise.race([pending, new Promise((resolve) => setTimeout(() => resolve('metadata stalled'), 100))]); + await vi.advanceTimersByTimeAsync(100); + expect(await bounded).toBe(decoded); + expect(context).toHaveBeenCalledWith(1, expect.any(Number), 22050); + expect(decode).toHaveBeenCalledWith(await file.arrayBuffer()); + expect(metadata).not.toHaveBeenCalled(); + } finally { vi.useRealTimers(); } +}); + +it('uses the configured sample rate and reports actual decoder failures', async () => { + const failure = new Error('Invalid audio data'); + const decode = vi.fn(async () => { throw failure; }); + const context = vi.fn(function (this: unknown) { return { decodeAudioData: decode }; }); + vi.stubGlobal('window', { webkitOfflineAudioContext: context }); + vi.stubGlobal('Audio', class { duration = 1; addEventListener(name: string, callback: () => void) { if (name === 'loadedmetadata') queueMicrotask(callback); } }); + await expect(decodeToMonoLowRate(sourceBlob([0]), 16000)).rejects.toBe(failure); + expect(context).toHaveBeenCalledWith(1, expect.any(Number), 16000); +}); diff --git a/electron/src/shared/utils/transcriptionsStore.js b/electron/src/shared/utils/transcriptionsStore.js index 22f212f85..f10aaebdb 100644 --- a/electron/src/shared/utils/transcriptionsStore.js +++ b/electron/src/shared/utils/transcriptionsStore.js @@ -23,8 +23,12 @@ export function loadTranscriptions() { /** Append a completed transcript using the existing 200-entry storage contract. */ export function addTranscription(entry) { + const history = loadTranscriptions(); + const ids = new Set(history.map((row) => row?.id)); + let id = Date.now(); + while (ids.has(id)) id += 1; const newEntry = { - id: Date.now(), + id, text: entry.text || '', language: entry.language || 'unknown', duration_s: entry.duration_s || 0, @@ -34,7 +38,7 @@ export function addTranscription(entry) { ? { refined_text: entry.refined_text } : {}), }; - const list = [newEntry, ...loadTranscriptions()].slice(0, 200); + const list = [newEntry, ...history].slice(0, 200); localStorage.setItem(TRANSCRIPTIONS_KEY, JSON.stringify(list)); window.dispatchEvent(new CustomEvent(TRANSCRIPTION_EVENT, { detail: newEntry })); return newEntry; diff --git a/tests/test_asr_device_aware_autodetect.py b/tests/test_asr_device_aware_autodetect.py index de54b04d7..21092b195 100644 --- a/tests/test_asr_device_aware_autodetect.py +++ b/tests/test_asr_device_aware_autodetect.py @@ -140,6 +140,32 @@ def align(segs, model, meta, audio, device, **kw): assert out == aligned # timing preserved +def test_alignment_load_failure_on_mps_retries_the_cpu_model(monkeypatch): + """A failed device transfer during loading must not skip the CPU retry.""" + import sys + import types + loaded = [] + segments = [{"text": "hi", "start": 0.0, "end": 1.0}] + aligned = [dict(segments[0], words=[{"word": "hi", "start": 0.1, "end": 0.8}])] + + def load_align_model(language_code, device): + loaded.append(device) + if device == "mps": + raise RuntimeError("MPS device transfer failed") + return "cpu model", "metadata" + + monkeypatch.setattr(ab, "_mps_available", lambda: True) + monkeypatch.setitem(sys.modules, "whisperx", types.SimpleNamespace( + load_align_model=load_align_model, + align=lambda *a, **kw: {"segments": aligned}, + )) + assert ab.forced_align(segments, object(), "en") == aligned + assert loaded == ["mps", "cpu"] + # The failed MPS load is cached; another call still reaches cached CPU. + assert ab.forced_align(segments, object(), "en") == aligned + assert loaded == ["mps", "cpu"] + + def test_a_language_with_no_aligner_keeps_its_native_timestamps(monkeypatch): """~20 languages have wav2vec2 aligners. The other 626 must still transcribe.""" segments = [{"text": "x", "start": 0.0, "end": 1.0, "words": [{"word": "x", "start": 0.0}]}] diff --git a/tests/test_audiobook_cancel.py b/tests/test_audiobook_cancel.py index bab2e95c4..96379d86f 100644 --- a/tests/test_audiobook_cancel.py +++ b/tests/test_audiobook_cancel.py @@ -44,6 +44,8 @@ def _plan(*chapters): def _drive(plan, monkeypatch, outputs_dir, *, is_disconnected=None, **kw): + from core.db import init_db + init_db() from api.routers import audiobook monkeypatch.setattr(audiobook, "_build_synth", _stub_build_synth()) monkeypatch.setattr("core.config.OUTPUTS_DIR", str(outputs_dir)) @@ -87,6 +89,8 @@ def test_disconnect_stops_early_and_preserves_resume(tmp_path, monkeypatch): assert "assembling" not in types and "done" not in types assert types[-1] == "stopped" assert events[-1]["rendered"] == 1 and events[-1]["total"] == 3 + from core import job_store + assert job_store.get(events[0]["job_id"])["status"] == "cancelled" # The resume manifest is preserved (NOT cleared), so the finished chapter is # offered for resume — Create-again picks up the rest from the cache. @@ -101,6 +105,8 @@ def test_no_disconnect_renders_all_chapters(tmp_path, monkeypatch): # stopped — the normal completion path is untouched. out = tmp_path / "outputs" out.mkdir() + from core.db import init_db + init_db() async def _connected(): return False @@ -120,3 +126,283 @@ async def _run(): assert "stopped" not in types assert types.count("chapter") == 3 assert types[-1] == "done" + from core import job_store + assert job_store.get(events[0]["job_id"])["status"] == "done" + + +@pytest.mark.parametrize("close_kind", ["close", "cancel"]) +def test_cancelled_native_stream_leaves_terminal_job_and_checkpoint(close_kind): + if find_ffmpeg() is None: + pytest.skip("ffmpeg required to reach the native renderer lifecycle") + from core import job_store + from core.db import init_db + from api.routers import audiobook + from services import longform_resume + + init_db() + + async def run(): + entered = asyncio.Event() + release = asyncio.Event() + + async def transport_wait(): + entered.set() + await release.wait() + return False + + stream = audiobook._render_longform_sse( + _plan(("One", "First chapter.")), default_voice=None, + is_disconnected=transport_wait, + ) + event = json.loads((await anext(stream))[len("data:"):].strip()) + assert event["type"] == "started" + job_id = event["job_id"] + assert job_store.get(job_id)["status"] == "running" + if close_kind == "cancel": + task = asyncio.create_task(anext(stream)) + await asyncio.wait_for(entered.wait(), timeout=5) + task.cancel() + with pytest.raises(asyncio.CancelledError): + await task + await stream.aclose() + assert job_store.get(job_id)["status"] == "cancelled" + assert longform_resume.has_manifest("audiobook", job_id) + + asyncio.run(run()) + + +@pytest.mark.parametrize("endpoint", ["audiobook", "longform", "resume"]) +@pytest.mark.parametrize("abort_kind", ["task_cancel", "send_error", "old_disconnect", "anyio_cancel"]) +def test_public_http_transport_cancel_retires_running_job(endpoint, abort_kind): + if find_ffmpeg() is None: + pytest.skip("ffmpeg required to reach the native renderer lifecycle") + from fastapi import FastAPI + from api.routers.audiobook import router + from core import job_store + from core.db import init_db + from services import longform_resume + from starlette.requests import ClientDisconnect + import anyio + + init_db() + app = FastAPI() + app.include_router(router) + events = [] + path = "/audiobook" + payload = {"text": "# One\nFirst chapter."} + if endpoint == "longform": + path = "/longform/render" + payload = {"chapters": [{"title": "One", "spans": [{"text": "First chapter."}]}]} + elif endpoint == "resume": + longform_resume.write_manifest(longform_resume.build_manifest( + job_id="previous", job_type="audiobook", title="One", + plan_chapters=[{"title": "One", "spans": [{"voice_id": None, "text": "First chapter."}]}], + params={"default_voice": None})) + path = "/audiobook/resume/previous" + payload = {} + + async def run(): + request_body = json.dumps(payload).encode() + received_request = False + started = asyncio.Event() + + async def receive(): + nonlocal received_request + if not received_request: + received_request = True + return {"type": "http.request", "body": request_body, "more_body": False} + await started.wait() + return {"type": "http.disconnect"} + + async def send(message): + if message["type"] == "http.response.body" and message.get("body"): + events.append(json.loads(message["body"].decode()[len("data:"):].strip())) + started.set() + if abort_kind == "task_cancel": + raise asyncio.CancelledError + if abort_kind == "send_error": + raise OSError("Client connection closed") + if abort_kind == "anyio_cancel": + cancel_scope.cancel() + await anyio.sleep(0) + await asyncio.Future() # Old ASGI disconnect listener cancels this task. + + scope = {"type": "http", "asgi": {"version": "3.0", "spec_version": "2.3" if abort_kind == "old_disconnect" else "2.4"}, + "http_version": "1.1", "method": "POST", "scheme": "http", + "path": path, "raw_path": path.encode(), "query_string": b"", + "root_path": "", "headers": [(b"content-type", b"application/json")], + "client": ("127.0.0.1", 12345), "server": ("testserver", 80)} + if abort_kind == "anyio_cancel": + with anyio.CancelScope() as cancel_scope: + await app(scope, receive, send) + elif abort_kind == "old_disconnect": + await app(scope, receive, send) + else: + with pytest.raises(asyncio.CancelledError if abort_kind == "task_cancel" else ClientDisconnect): + await app(scope, receive, send) + # Assert BEFORE event-loop shutdown can finalize any lazy iterators. + assert [e["type"] for e in events] == ["started"] + job_id = events[0]["job_id"] + assert job_store.get(job_id)["status"] == "cancelled" + assert longform_resume.has_manifest("story" if endpoint == "longform" else "audiobook", job_id) + if endpoint == "resume": + assert longform_resume.has_manifest("audiobook", "previous") + + asyncio.run(run()) + + +@pytest.mark.parametrize("terminal", ["done", "failed"]) +def test_closed_stream_preserves_existing_terminal_status(terminal): + if find_ffmpeg() is None: + pytest.skip("ffmpeg required to reach the native renderer lifecycle") + from api.routers import audiobook + from core import job_store + from core.db import init_db + + init_db() + + async def run(): + stream = audiobook._render_longform_sse( + _plan(("One", "First chapter.")), default_voice=None, + ) + event = json.loads((await anext(stream))[len("data:"):].strip()) + assert event["type"] == "started" + job_id = event["job_id"] + # Finalization can run after another owner recorded the terminal state. + if terminal == "done": + job_store.mark_done(job_id) + else: + job_store.mark_failed(job_id, "Stopped by the owning job") + await stream.aclose() + assert job_store.get(job_id)["status"] == terminal + + asyncio.run(run()) + + +@pytest.mark.skipif(find_ffmpeg() is None, reason="ffmpeg required to reach native render lifecycle") +def test_failed_render_keeps_failed_status(monkeypatch): + from api.routers import audiobook + from core import job_store + from core.db import init_db + from services import longform_resume + + init_db() + + def factory(*args, **kwargs): + spec = _stub_build_synth()(*args, **kwargs) + def unavailable_model(*_args, **_kwargs): + raise RuntimeError("Local synthesis unavailable") + spec["synth"] = unavailable_model + return spec + + monkeypatch.setattr(audiobook, "_build_synth", factory) # External synthesis boundary only. + + async def run(): + return [json.loads(frame[len("data:"):].strip()) + async for frame in audiobook._render_longform_sse( + _plan(("One", "Unique failed-render control.")), default_voice=None)] + + events = asyncio.run(run()) + assert events[0]["type"] == "started" + assert events[-1]["type"] == "error" + job_id = events[0]["job_id"] + assert job_store.get(job_id)["status"] == "failed" + assert longform_resume.has_manifest("audiobook", job_id) + + +@pytest.mark.parametrize("terminal", [None, "done", "failed"]) +def test_public_iterator_closure_retires_job_before_loop_shutdown(terminal): + if find_ffmpeg() is None: + pytest.skip("ffmpeg required to reach native render lifecycle") + from api.routers import audiobook + from core import job_store + from core.db import init_db + from services import longform_resume + init_db() + + async def run(): + stream = audiobook._public_longform_stream( + _plan(("One", "First chapter.")), default_voice=None) + event = json.loads((await anext(stream))[len("data:"):].strip()) + assert event["type"] == "started" + job_id = event["job_id"] + assert job_store.get(job_id)["status"] == "running" + if terminal == "done": + job_store.mark_done(job_id) + elif terminal == "failed": + job_store.mark_failed(job_id, "Stopped by the owning job") + await stream.aclose() + # Check while this loop and the inner generator are still alive. + assert job_store.get(job_id)["status"] == (terminal or "cancelled") + assert longform_resume.has_manifest("audiobook", job_id) + + asyncio.run(run()) + + +@pytest.mark.parametrize("case", ["empty_chapters", "empty_spans", "missing_ffmpeg"]) +def test_finite_public_setup_error_retires_job_and_keeps_checkpoint(case, monkeypatch): + from fastapi import FastAPI + from fastapi.testclient import TestClient + from api.routers.audiobook import router + from core import job_store + from core.db import init_db + from services import longform_resume + init_db() + app = FastAPI() + app.include_router(router) + payload = {"chapters": []} + if case == "empty_spans": + payload = {"chapters": [{"title": "One", "spans": [{"text": ""}]}]} + elif case == "missing_ffmpeg": + payload = {"chapters": [{"title": "One", "spans": [{"text": "Hello."}]}]} + # External executable availability only; routes, history and checkpoint + # processing are native. No synthesis is reached on this setup path. + monkeypatch.setattr("services.ffmpeg_utils.find_ffmpeg", lambda: None) + previous = {row["id"] for row in job_store.list_jobs(limit=100000)} + client = TestClient(app, client=("127.0.0.1", 50000)) + try: + response = client.post("/longform/render", json=payload) + finally: + client.close() + assert response.status_code == 200 + frames = [json.loads(frame[len("data:"):].strip()) + for frame in response.text.strip().split("\n\n")] + assert len(frames) == 1 and frames[0]["type"] == "error" + expected = "ffmpeg not available" if case == "missing_ffmpeg" else "nothing to render" + assert expected in frames[0]["error"] + jobs = [row for row in job_store.list_jobs(limit=100000) if row["id"] not in previous] + assert len(jobs) == 1 + assert jobs[0]["status"] == "failed" + assert jobs[0]["finished_at"] is not None + assert jobs[0]["error"] == frames[0]["error"] + assert longform_resume.has_manifest("story", jobs[0]["id"]) + + +def test_actual_cache_directory_error_retires_public_job(tmp_path, monkeypatch): + from fastapi import FastAPI + from fastapi.testclient import TestClient + from api.routers.audiobook import router + from core import job_store + from core.db import init_db + from services.longform_render import LONGFORM_CACHE_SUBDIR + from services import longform_resume + init_db() + outputs = tmp_path / "outputs" + outputs.mkdir() + (outputs / LONGFORM_CACHE_SUBDIR).write_text("A file occupies the cache directory name") + monkeypatch.setattr("core.config.OUTPUTS_DIR", str(outputs)) + previous = {row["id"] for row in job_store.list_jobs(limit=100000)} + app = FastAPI() + app.include_router(router) + client = TestClient(app, client=("127.0.0.1", 50000)) + try: + response = client.post("/longform/render", json={"chapters": [{"title": "One", "spans": [{"text": "Hello."}]}]}) + finally: + client.close() + assert response.status_code == 200 + assert '"type": "error"' in response.text + assert "cache directory" not in response.text # Setup exception details stay local. + jobs = [row for row in job_store.list_jobs(limit=100000) if row["id"] not in previous] + assert len(jobs) == 1 and jobs[0]["status"] == "failed" + assert jobs[0]["finished_at"] is not None + assert longform_resume.has_manifest("story", jobs[0]["id"]) diff --git a/tests/test_batch_retry_custody.py b/tests/test_batch_retry_custody.py new file mode 100644 index 000000000..eedb830fc --- /dev/null +++ b/tests/test_batch_retry_custody.py @@ -0,0 +1,231 @@ +"""Public batch retry requests must not duplicate or delete queued work.""" +import asyncio +import threading + +import httpx +import pytest +from fastapi import FastAPI + + +@pytest.fixture +def queue(tmp_path, monkeypatch): + from api.routers import batch + from services import asr_backend, translation_engines + + monkeypatch.setattr(batch, "DATA_DIR", str(tmp_path)) + monkeypatch.setattr(batch, "_jobs", {}) + monkeypatch.setattr(batch, "_queue", asyncio.Queue()) + monkeypatch.setattr(batch, "_processing_job_ids", set()) + monkeypatch.setattr(batch, "_ensure_queue", lambda: None) + # Model/provider admission is covered separately. No inference or external + # provider is needed to race two requests over the real queue and files. + monkeypatch.setattr(asr_backend, "asr_model_missing_error", lambda: None) + monkeypatch.setattr(translation_engines, "is_ready", lambda _provider: True) + video = tmp_path / "input.mp4" + video.write_bytes(b"original input") + outputs = tmp_path / "batch" / "job" + outputs.mkdir(parents=True) + (outputs / "prior.mp4").write_bytes(b"prior output") + batch._jobs["job"] = { + "id": "job", "status": "failed", "video_path": str(video), + "translation_provider": "argos", "langs": ["es"], "attempts": 1, + "created_at": 0, "filename": "input.mp4", "voice_id": None, + } + app = FastAPI() + app.include_router(batch.router) + return batch, app, video, outputs + + +def test_two_retry_requests_enqueue_one_attempt(queue, monkeypatch): + batch, app, video, _outputs = queue + entered, release = threading.Event(), threading.Event() + original = batch._batch_voice + + def gated_voice(voice_id): + entered.set() + assert release.wait(5), "preflight was never released" + return original(voice_id) + + monkeypatch.setattr(batch, "_batch_voice", gated_voice) + + async def scenario(): + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=app), base_url="http://testserver") as client: + first = asyncio.create_task(client.post("/batch/jobs/job/retry")) + assert await asyncio.to_thread(entered.wait, 5) + try: + second = await asyncio.wait_for(client.post("/batch/jobs/job/retry"), 5) + assert second.status_code == 409 + finally: + release.set() + responses = [await first, second] + assert sorted(r.status_code for r in responses) == [200, 409] + assert batch._queue.qsize() == 1 + assert batch._jobs["job"]["attempts"] == 2 + assert video.read_bytes() == b"original input" + + asyncio.run(scenario()) + + +def test_delete_cannot_remove_input_during_retry_admission(queue, monkeypatch): + batch, app, video, outputs = queue + entered, release = threading.Event(), threading.Event() + original = batch._batch_voice + + def gated_voice(voice_id): + entered.set() + assert release.wait(5) + return original(voice_id) + + monkeypatch.setattr(batch, "_batch_voice", gated_voice) + + async def scenario(): + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=app), base_url="http://testserver") as client: + retry = asyncio.create_task(client.post("/batch/jobs/job/retry")) + assert await asyncio.to_thread(entered.wait, 5) + deletion = await client.delete("/batch/jobs/job") + # Always settle the task, including a failing regression assertion. + release.set() + response = await retry + assert deletion.status_code == 409 + assert response.status_code == 200 + assert video.read_bytes() == b"original input" + assert "job" in batch._jobs + assert batch._queue.qsize() == 1 + + asyncio.run(scenario()) + + +@pytest.mark.parametrize("status,processing", [("queued", False), ("running", True), ("cancelled", True)]) +def test_delete_keeps_active_or_still_stopping_job_files(queue, status, processing): + batch, app, video, outputs = queue + batch._jobs["job"]["status"] = status + if processing: + batch._processing_job_ids.add("job") + + async def scenario(): + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=app), base_url="http://testserver") as client: + response = await client.delete("/batch/jobs/job") + assert response.status_code == 409 + assert video.exists() and outputs.exists() + assert "job" in batch._jobs + + asyncio.run(scenario()) + + +def test_settled_terminal_job_can_still_be_deleted(queue): + batch, app, video, outputs = queue + + async def scenario(): + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=app), base_url="http://testserver") as client: + response = await client.delete("/batch/jobs/job") + assert response.status_code == 200 + assert not video.exists() and not outputs.exists() + assert "job" not in batch._jobs + + asyncio.run(scenario()) + + +def test_failed_preflight_releases_job_for_a_later_retry(queue, monkeypatch): + batch, app, video, outputs = queue + from services import translation_engines + monkeypatch.setattr(translation_engines, "is_ready", lambda _provider: False) + + async def scenario(): + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=app), base_url="http://testserver") as client: + failed = await client.post("/batch/jobs/job/retry") + assert failed.status_code == 409 + assert batch._jobs["job"]["attempts"] == 1 + assert video.exists() and outputs.exists() + monkeypatch.setattr(translation_engines, "is_ready", lambda _provider: True) + retried = await client.post("/batch/jobs/job/retry") + assert retried.status_code == 200 + assert batch._queue.qsize() == 1 + assert batch._jobs["job"]["attempts"] == 2 + + asyncio.run(scenario()) + + +def test_cancelled_retry_keeps_reservation_until_output_cleanup_finishes(queue, monkeypatch): + batch, app, video, outputs = queue + entered, release = threading.Event(), threading.Event() + original = batch.shutil.rmtree + + def slow_cleanup(path): + entered.set() + assert release.wait(5) + return original(path) + + monkeypatch.setattr(batch.shutil, "rmtree", slow_cleanup) + + async def scenario(): + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=app), base_url="http://testserver") as client: + retry = asyncio.create_task(client.post("/batch/jobs/job/retry")) + assert await asyncio.to_thread(entered.wait, 5) + retry.cancel() + delivered = asyncio.Event() + # Task.cancel schedules delivery before this callback. Await its + # explicit turn boundary before attempting the conflicting action. + asyncio.get_running_loop().call_soon(delivered.set) + await delivered.wait() + try: + response = await asyncio.wait_for(client.delete("/batch/jobs/job"), 5) + assert response.status_code == 409 + finally: + release.set() + result = (await asyncio.gather(retry, return_exceptions=True))[0] + assert isinstance(result, asyncio.CancelledError) + assert response.status_code == 409 + assert video.read_bytes() == b"original input" + assert batch._jobs["job"]["status"] == "failed" + assert batch._jobs["job"]["attempts"] == 1 + assert batch._queue.qsize() == 0 + assert not outputs.exists() + # Cancellation is propagated after cleanup, and does not strand + # the job reservation or prevent an explicit later attempt. + later = await client.post("/batch/jobs/job/retry") + assert later.status_code == 200 + assert batch._queue.qsize() == 1 + + asyncio.run(scenario()) + + +@pytest.mark.parametrize("cancelled", [False, True]) +def test_cleanup_error_releases_custody_and_preserves_cancellation(queue, monkeypatch, cancelled): + batch, app, video, outputs = queue + entered, release = threading.Event(), threading.Event() + original = batch.shutil.rmtree + + def failing_cleanup(_path): + entered.set() + assert release.wait(5) + # Portable filesystem-error boundary; an independent native proof + # also reproduces this with a real read-only output directory. + raise PermissionError("output is in use") + + monkeypatch.setattr(batch.shutil, "rmtree", failing_cleanup) + + async def scenario(): + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=app), base_url="http://testserver") as client: + retry = asyncio.create_task(client.post("/batch/jobs/job/retry")) + assert await asyncio.to_thread(entered.wait, 5) + if cancelled: + retry.cancel() + delivered = asyncio.Event() + asyncio.get_running_loop().call_soon(delivered.set) + await delivered.wait() + release.set() + result = (await asyncio.gather(retry, return_exceptions=True))[0] + if cancelled: + assert isinstance(result, asyncio.CancelledError) + else: + assert result.status_code == 500 + assert "Could not reset" in result.json()["detail"] + assert video.exists() and outputs.exists() + assert batch._jobs["job"]["attempts"] == 1 + assert batch._queue.qsize() == 0 + monkeypatch.setattr(batch.shutil, "rmtree", original) + later = await client.post("/batch/jobs/job/retry") + assert later.status_code == 200 + assert batch._queue.qsize() == 1 + + asyncio.run(scenario()) diff --git a/tests/test_delete_after_commit_regression.py b/tests/test_delete_after_commit_regression.py index b791a76f6..5319685b5 100644 --- a/tests/test_delete_after_commit_regression.py +++ b/tests/test_delete_after_commit_regression.py @@ -193,7 +193,7 @@ def test_first_lock_failure_leaves_no_orphan_take(profile, monkeypatch): assert sorted(p.name for p in profile.iterdir() if "locked" in p.name) == [] -def test_successful_relock_installs_new_take_and_drops_backup(profile, monkeypatch): +def test_successful_relock_installs_new_take_and_retains_prior_version(profile, monkeypatch): import asyncio from core import db from api.routers import profiles @@ -207,9 +207,12 @@ def test_successful_relock_installs_new_take_and_drops_backup(profile, monkeypat conn.execute( "INSERT INTO generation_history(id, text, audio_path) VALUES('h','t','take.wav')" ) - asyncio.run(profiles.lock_profile("voice", history_id="h", seed=1)) - assert (profile / "voice_locked.wav").read_bytes() == b"new-take" - assert sorted(p.name for p in profile.iterdir() if "locked" in p.name) == ["voice_locked.wav"] + result = asyncio.run(profiles.lock_profile("voice", history_id="h", seed=1)) + current = result["locked_audio_path"] + assert current != "voice_locked.wav" + assert (profile / current).read_bytes() == b"new-take" + assert (profile / "voice_locked.wav").read_bytes() == b"old-locked" + assert sorted(p.name for p in profile.iterdir() if "locked" in p.name) == sorted([current, "voice_locked.wav"]) def test_install_staged_restores_previous_file(tmp_path): diff --git a/tests/test_dub_complete_audio.py b/tests/test_dub_complete_audio.py index 49ad19550..8bc35598a 100644 --- a/tests/test_dub_complete_audio.py +++ b/tests/test_dub_complete_audio.py @@ -1,5 +1,6 @@ """A published dub must contain every requested spoken segment, without early clipping.""" import asyncio +import copy import json from types import SimpleNamespace @@ -133,6 +134,73 @@ def fail(): raise RuntimeError('failure') assert render_dub.job['dubbed_tracks']['en']['path'] == str(previous) +@pytest.mark.parametrize("segment_ids", [["duplicate", "duplicate"], ["seg_1"]]) +def test_conflicting_regeneration_preserves_saved_audio_and_text(render_dub, monkeypatch, segment_ids): + from api.routers import dub_generate as dg + from fastapi import HTTPException + + previous = render_dub.path / 'dubbed_en.wav' + sf.write(previous, [.2] * 24000, 24000) + previous_bytes = previous.read_bytes() + render_dub.job.update({ + 'segments': [{'id': 'a', 'start': 0, 'end': 1, 'text': 'saved first'}, + {'id': 'b', 'start': 1, 'end': 2, 'text': 'saved second'}], + 'segments_i18n': {'en': {'a': 'saved first', 'b': 'saved second'}}, + 'dubbed_tracks': {'en': {'path': str(previous)}}, + }) + saved = copy.deepcopy(render_dub.job) + persisted = render_dub.path / 'job.json' + persisted.write_text(json.dumps(saved)) + monkeypatch.setattr(dg, '_save_job', lambda _, job: persisted.write_text(json.dumps(job))) + async def forbidden_resolution(): + pytest.fail('Conflicting identities must be rejected before loading a backend') + monkeypatch.setattr(dg, '_resolve_dub_execution', forbidden_resolution) + + with pytest.raises(HTTPException) as error: + render_dub.run(segments=[dict(start=0, end=1, text='replacement first'), + dict(start=1, end=2, text='replacement second')], + segment_ids=segment_ids) + assert error.value.status_code == 409 + assert error.value.detail['code'] == 'dub_segment_identity_conflict' + assert previous.read_bytes() == previous_bytes + assert json.loads(persisted.read_text()) == saved + assert render_dub.job == saved + assert not render_dub.generated + + +def test_partial_ids_render_without_reusing_a_saved_segment(render_dub): + render_dub.job['segments'] = [ + {'id': 'a', 'start': 0, 'end': 1, 'text': 'original a'}, + {'id': 'b', 'start': 1, 'end': 2, 'text': 'original b'}, + ] + events = render_dub.run(segments=[dict(start=0, end=1, text='translated b'), + dict(start=1, end=2, text='new segment')], + segment_ids=['b']) + assert any(e['type'] == 'done' for e in events) + assert [row['id'] for row in render_dub.job['segments']] == ['b', 'seg_1'] + assert render_dub.job['segments_i18n']['en'] == {'b': 'translated b', 'seg_1': 'new segment'} + assert sf.info(render_dub.path / 'dubbed_en.wav').frames > 0 + + +def test_explicit_later_id_keeps_saved_metadata_after_render(render_dub): + render_dub.job['segments'] = [ + {'id': 'a', 'start': 0, 'end': 1, 'text': 'original a', 'speaker_id': 'speaker-a'}, + {'id': 'b', 'start': 1, 'end': 2, 'text': 'original b', 'speaker_id': 'speaker-b'}, + ] + events = render_dub.run(segments=[dict(start=0, end=1, text='new x'), + dict(start=1, end=2, text='translated a')], + segment_ids=['x', 'a']) + assert any(e['type'] == 'done' for e in events) + rows = render_dub.job['segments'] + assert [row['id'] for row in rows] == ['x', 'a'] + assert 'speaker_id' not in rows[0] + assert rows[0]['text_original'] == '' + assert rows[1]['speaker_id'] == 'speaker-a' + assert rows[1]['text_original'] == 'original a' + assert render_dub.job['segments_i18n']['en'] == {'x': 'new x', 'a': 'translated a'} + assert sf.info(render_dub.path / 'dubbed_en.wav').frames > 0 + + def test_timing_trims_edge_silence_but_keeps_internal_pauses(): from services.audio_dsp import trim_speech_padding wav = torch.cat((torch.zeros(1, 1000), torch.ones(1, 200)*.1, diff --git a/tests/test_dub_qc_concurrency.py b/tests/test_dub_qc_concurrency.py new file mode 100644 index 000000000..e900a9c39 --- /dev/null +++ b/tests/test_dub_qc_concurrency.py @@ -0,0 +1,76 @@ +"""A QC result belongs to exactly the track and transcript that ASR read.""" +import asyncio +from unittest.mock import Mock + +import pytest + + +@pytest.mark.parametrize('change', ['regenerate', 'text', 'timing', 'audio', 'delete', 'replace', None]) +def test_qc_does_not_publish_after_selected_render_changes(monkeypatch, tmp_path, change): + from api.routers import dub_export, dub_generate + from fastapi import HTTPException + from schemas.requests import DubRequest + from services import asr_backend, dub_pipeline, model_manager + + audio = tmp_path / 'dubbed_es.wav' + audio.write_bytes(b'old rendered track') + job = { + 'segments': [{'id': 'a', 'start': 0., 'end': 2., 'text': 'hola mundo'}], + 'segments_i18n': {'es': {'a': 'hola mundo'}}, + 'dubbed_tracks': {'es': {'path': str(audio), 'timing_strategy': 'strict_slot', + 'source_segments': [{'id': 'a', 'start': 0., 'end': 2.}]}}, + 'seg_order': ['a'], + } + monkeypatch.setattr(dub_pipeline, '_dub_jobs', {'qc-race': job}) + monkeypatch.setattr(dub_pipeline, '_withdrawn_jobs', {}) + saved = Mock() + monkeypatch.setattr(dub_pipeline, 'save_job', saved) + monkeypatch.setattr(dub_export, '_job_dir_or_400', lambda _: None) + monkeypatch.setattr(dub_export, '_get_job', lambda _: dub_pipeline._dub_jobs.get('qc-race')) + monkeypatch.setattr(dub_export, '_dub_artifact', lambda path, *a, **kw: path) + monkeypatch.setattr(asr_backend, 'asr_model_missing_error', lambda: None) + monkeypatch.setattr(model_manager, '_get_gpu_pool', lambda: None) + + class Backend: + id = 'test-local-asr' + def transcribe(self, path, **kwargs): + assert audio.read_bytes() == b'old rendered track' + return {'segments': [{'start': 0., 'end': 2., 'text': 'hola mundo'}]} + monkeypatch.setattr(asr_backend, 'load_active_asr_backend', Backend) + + async def guarded(pool, fn, **kwargs): + recognition = fn() + # These edits happen while an actual ASR worker releases the event loop. + if change == 'regenerate': + dub_generate._sync_job_segments(job, DubRequest( + segments=[{'start': 0., 'end': 2., 'text': 'buenos dias'}], + segment_ids=['a'], language_code='es')) + job['dubbed_tracks']['es'] = dict(job['dubbed_tracks']['es']) + elif change == 'text': + job['segments_i18n']['es']['a'] = 'buenos dias' + elif change == 'timing': + job['dubbed_tracks']['es']['source_segments'][0]['end'] = 20. + elif change == 'audio': + # A render overwrites the same path before publishing new metadata. + audio.write_bytes(b'new rendered audio with different bytes') + elif change == 'delete': + dub_pipeline.purge_jobs(['qc-race'], delete_rows=lambda: None) + elif change == 'replace': + dub_pipeline.put_job('qc-race', dict(job)) + return recognition + monkeypatch.setattr(asr_backend, 'run_transcribe_guarded', guarded) + + run = lambda: asyncio.run(dub_export.dub_qc_pass('qc-race', lang='es', drift_threshold=.5)) + if change is None: + assert run()['flagged_count'] == 0 + assert job['segments'][0]['qc_drift'] == 0 + saved.assert_called_once() + else: + with pytest.raises(HTTPException) as error: + run() + assert error.value.status_code == 409 + assert error.value.detail['code'] == 'dub_qc_track_changed' + assert all('qc_drift' not in segment for segment in job['segments']) + saved.assert_not_called() + if change == 'delete': + assert 'qc-race' not in dub_pipeline._dub_jobs diff --git a/tests/test_dub_qc_track_language.py b/tests/test_dub_qc_track_language.py new file mode 100644 index 000000000..d0e8669d6 --- /dev/null +++ b/tests/test_dub_qc_track_language.py @@ -0,0 +1,399 @@ +"""QC scores the track actually selected, rather than the last generated text.""" +import asyncio +import copy + +import pytest + + +def test_sync_consumes_existing_segment_only_once(): + from api.routers.dub_generate import _sync_job_segments + from schemas.requests import DubRequest + + job = {'segments': [ + {'id': 'a', 'start': 0, 'end': 1, 'text': 'original a', 'speaker_id': 'speaker-a'}, + {'id': 'b', 'start': 1, 'end': 2, 'text': 'original b', 'speaker_id': 'speaker-b'}, + ], 'seg_order': ['b', 'seg_1']} + req = DubRequest(segments=[dict(start=0, end=1, text='translated b'), + dict(start=1, end=2, text='new segment')], + segment_ids=['b'], language_code='en') + _sync_job_segments(job, req) + assert [row['id'] for row in job['segments']] == ['b', 'seg_1'] + assert job['segments'][0]['speaker_id'] == 'speaker-b' + assert job['segments'][0]['text_original'] == 'original b' + assert 'speaker_id' not in job['segments'][1] + assert job['segments'][1]['text_original'] == '' + assert job['segments_i18n']['en'] == {'b': 'translated b', 'seg_1': 'new segment'} + + +def test_sync_reserves_saved_rows_for_later_explicit_ids(): + from api.routers.dub_generate import _sync_job_segments + from schemas.requests import DubRequest + + job = {'segments': [ + {'id': 'a', 'start': 0, 'end': 1, 'text': 'original a', 'speaker_id': 'speaker-a'}, + {'id': 'b', 'start': 1, 'end': 2, 'text': 'original b', 'speaker_id': 'speaker-b'}, + ], 'seg_order': ['x', 'a']} + req = DubRequest(segments=[dict(start=0, end=1, text='new x'), + dict(start=1, end=2, text='translated a')], + segment_ids=['x', 'a'], language_code='en') + _sync_job_segments(job, req) + assert [row['id'] for row in job['segments']] == ['x', 'a'] + assert 'speaker_id' not in job['segments'][0] + assert job['segments'][0]['text_original'] == '' + assert job['segments'][1]['speaker_id'] == 'speaker-a' + assert job['segments'][1]['text_original'] == 'original a' + assert job['segments_i18n']['en'] == {'x': 'new x', 'a': 'translated a'} + + +@pytest.mark.parametrize("requested_lang", ["es", None, "unknown", "bn"]) +def test_qc_uses_selected_track_authoritative_text(monkeypatch, requested_lang): + from api.routers import dub_export + from services import asr_backend, dub_pipeline, model_manager + + job = { + "segments": [{"id": "a", "start": 0.0, "end": 2.0, "text": "ohe prithibi"}], + "segments_i18n": {"es": {"a": "hola mundo"}, "bn": {"a": "ohe prithibi"}}, + "dubbed_tracks": {"es": {"path": "spanish.wav"}, "bn": {"path": "bengali.wav"}}, + "seg_order": ["a"], + } + original = copy.deepcopy(job) + heard = [] + selected = "bn" if requested_lang == "bn" else "es" + expected_text = "ohe prithibi" if selected == "bn" else "hola mundo" + + class Backend: + id = "test-local-asr" + def transcribe(self, path, **kwargs): + heard.append(path) + return {"segments": [{"start": 0.0, "end": 2.0, "text": expected_text}]} + + async def guarded(pool, fn, **kwargs): + return fn() + + monkeypatch.setattr(dub_export, "_job_dir_or_400", lambda job_id: None) + monkeypatch.setattr(dub_export, "_get_job", lambda job_id: job) + monkeypatch.setattr(dub_export, "_dub_artifact", lambda path, *a, **kw: path) + monkeypatch.setattr(asr_backend, "asr_model_missing_error", lambda: None) + monkeypatch.setattr(asr_backend, "load_active_asr_backend", Backend) + monkeypatch.setattr(asr_backend, "run_transcribe_guarded", guarded) + monkeypatch.setattr(model_manager, "_get_gpu_pool", lambda: None) + monkeypatch.setattr(dub_pipeline, "put_job", lambda *a: None) + monkeypatch.setattr(dub_pipeline, "save_job", lambda *a: None) + + result = asyncio.run(dub_export.dub_qc_pass("test", lang=requested_lang, drift_threshold=0.5)) + assert heard == ["bengali.wav" if selected == "bn" else "spanish.wav"] + assert result["flagged_count"] == 0 + assert result["segments"][0]["drift"] == 0.0 + assert job["segments"][0]["text"] == original["segments"][0]["text"] + assert job["segments_i18n"] == original["segments_i18n"] + assert job["segments"][0]["qc_recognized"] == expected_text + + +@pytest.mark.parametrize("strategy,with_plan", [ + ("smart_fit", True), + ("stretch_video", True), + ("strict_slot", True), + ("smart_fit", False), + ("stretch_video", False), + ("legacy_source", False), +]) +@pytest.mark.parametrize("requested_lang", ["es", None]) +@pytest.mark.parametrize("reordered", [False, True]) +def test_qc_matches_selected_track_rendered_times(monkeypatch, strategy, with_plan, requested_lang, reordered): + from api.routers import dub_export + from services import asr_backend, dub_pipeline, model_manager + + source = [{"id": "a", "start": 0.0, "end": 1.0}, + {"id": "b", "start": 1.0, "end": 2.0}] + fitted = [{"id": "a", "start": 0.0, "end": 3.0}, + {"id": "b", "start": 3.0, "end": 4.0}] + plan = [{"orig_start": 0.0, "orig_end": 1.0, "new_start": 0.0, "new_end": 3.0}, + {"orig_start": 1.0, "orig_end": 2.0, "new_start": 3.0, "new_end": 4.0}] + if strategy == "legacy_source": + source = [{"start": s["start"], "end": s["end"]} for s in source] + job = { + "segments": [{"id": "a", "start": 10.0, "end": 11.0, "text": "bengali first"}, + {"id": "b", "start": 11.0, "end": 12.0, "text": "bengali second"}], + "segments_i18n": {"es": {"a": "hola", "b": "adios"}}, + # The most recently generated track has a different timeline/strategy. + "timing_strategy": "concise", + "dubbed_tracks": {"es": {"path": "spanish.wav", "timing_strategy": "strict_slot" if strategy == "legacy_source" else strategy, + "source_segments": source}, + "bn": {"path": "bengali.wav", "timing_strategy": "concise"}}, + "seg_order": ["a", "b"], + } + if with_plan: + # Retained plans must not override a selected strict-slot track. + job["fit_plans"] = {"es": {"fitted_segments": fitted}} + job["video_stretch_plans"] = {"es": {"plan": plan}} + timing = (job["segments"] if strategy == "legacy_source" else + fitted if with_plan and strategy != "strict_slot" else source) + if strategy == "legacy_source": + timing = source + recognized = [dict(t, text=text) for t, text in zip(timing, ["hola", "adios"])] + if reordered: + job["segments"].reverse() + job["seg_order"].reverse() + original = copy.deepcopy(job) + asr_calls = [] + + class Backend: + id = "test-local-asr" + def transcribe(self, path, **kwargs): + asr_calls.append(path) + assert path == "spanish.wav" + return {"segments": recognized} + + async def guarded(pool, fn, **kwargs): + return fn() + + monkeypatch.setattr(dub_export, "_job_dir_or_400", lambda job_id: None) + monkeypatch.setattr(dub_export, "_get_job", lambda job_id: job) + monkeypatch.setattr(dub_export, "_dub_artifact", lambda path, *a, **kw: path) + monkeypatch.setattr(asr_backend, "asr_model_missing_error", lambda: None) + monkeypatch.setattr(asr_backend, "load_active_asr_backend", Backend) + monkeypatch.setattr(asr_backend, "run_transcribe_guarded", guarded) + monkeypatch.setattr(model_manager, "_get_gpu_pool", lambda: None) + monkeypatch.setattr(dub_pipeline, "put_job", lambda *a: None) + monkeypatch.setattr(dub_pipeline, "save_job", lambda *a: None) + + if strategy == "legacy_source": + from fastapi import HTTPException + with pytest.raises(HTTPException) as error: + asyncio.run(dub_export.dub_qc_pass("test", lang=requested_lang, drift_threshold=0.5)) + assert error.value.status_code == 409 + assert error.value.detail["code"] == "dub_qc_timing_identity_missing" + assert asr_calls == [] + assert job == original + return + result = asyncio.run(dub_export.dub_qc_pass("test", lang=requested_lang, drift_threshold=0.5)) + assert result["flagged_count"] == 0 + assert [s["drift"] for s in result["segments"]] == [0.0, 0.0] + for current, old in zip(job["segments"], original["segments"]): + assert (current["start"], current["end"], current["text"]) == (old["start"], old["end"], old["text"]) + assert job["dubbed_tracks"] == original["dubbed_tracks"] + assert job["segments_i18n"] == original["segments_i18n"] + + +def test_render_source_snapshot_preserves_manifest_identity_after_reorder(): + from api.routers.dub_generate import _track_source_segments, _sync_job_segments + from schemas.requests import DubRequest, DubSegment + + first = DubRequest(language_code="es", segment_ids=["a", "b"], segments=[ + DubSegment(start=0.0, end=1.0, text="hola"), + DubSegment(start=2.0, end=3.0, text="adios"), + ]) + job = {"seg_order": ["a", "b"]} + _sync_job_segments(job, first) + snapshot = _track_source_segments(job) + later = DubRequest(language_code="bn", segment_ids=["b", "a"], segments=[ + DubSegment(start=10.0, end=11.0, text="later b"), + DubSegment(start=12.0, end=13.0, text="later a"), + ]) + job["seg_order"] = later.segment_ids + _sync_job_segments(job, later) + from api.routers.dub_export import _apply_fitted_times + restored = _apply_fitted_times(job["segments"], snapshot) + assert [(s["id"], s["start"], s["end"]) for s in restored] == [("b", 2.0, 3.0), ("a", 0.0, 1.0)] + assert [s["id"] for s in _track_source_segments(job)] == ["b", "a"] + + +def test_render_source_snapshot_uses_preserved_ids_for_legacy_request(): + from api.routers.dub_generate import _track_source_segments, _sync_job_segments + from schemas.requests import DubRequest, DubSegment + job = {"segments": [{"id": "saved", "start": 0.0, "end": 1.0, "text": "original"}]} + req = DubRequest(language_code="es", segments=[DubSegment(start=2.0, end=3.0, text="hola")]) + _sync_job_segments(job, req) + assert _track_source_segments(job) == [{"id": "saved", "start": 2.0, "end": 3.0}] + + +def _legacy_qc_route(monkeypatch, job, recognized): + from api.routers import dub_export + from services import asr_backend, dub_pipeline, model_manager + calls = [] + class Backend: + id = "test-local-asr" + def transcribe(self, path, **kwargs): + calls.append(path) + return {"segments": recognized} + async def guarded(pool, fn, **kwargs): + return fn() + monkeypatch.setattr(dub_export, "_job_dir_or_400", lambda _: None) + monkeypatch.setattr(dub_export, "_get_job", lambda _: job) + monkeypatch.setattr(dub_export, "_dub_artifact", lambda path, *a, **kw: path) + monkeypatch.setattr(asr_backend, "asr_model_missing_error", lambda: None) + monkeypatch.setattr(asr_backend, "load_active_asr_backend", Backend) + monkeypatch.setattr(asr_backend, "run_transcribe_guarded", guarded) + monkeypatch.setattr(model_manager, "_get_gpu_pool", lambda: None) + monkeypatch.setattr(dub_pipeline, "put_job", lambda *a: None) + monkeypatch.setattr(dub_pipeline, "save_job", lambda *a: None) + return lambda: asyncio.run(dub_export.dub_qc_pass("test", lang="es", drift_threshold=0.5)), calls + + +@pytest.mark.parametrize("fit_ids", [["a", "b"], ["b", "a"], ["a", "a"], ["a"], ["a", None], ["a", "foreign"]]) +def test_legacy_source_uses_only_complete_smart_fit_identity(monkeypatch, fit_ids): + from fastapi import HTTPException + timings = {"a": (0.0, 1.0), "b": (2.0, 3.0)} + fitted = [dict(id=sid, start=timings.get(sid, (0.0, 1.0))[0], + end=timings.get(sid, (0.0, 1.0))[1]) for sid in fit_ids] + job = {"segments": [{"id": "b", "start": 10.0, "end": 11.0, "text": "later b"}, + {"id": "a", "start": 12.0, "end": 13.0, "text": "later a"}], + "segments_i18n": {"es": {"a": "hola", "b": "adios"}}, "seg_order": ["b", "a"], + "dubbed_tracks": {"es": {"path": "spanish.wav", "timing_strategy": "smart_fit", + "source_segments": [{"start": 0.0, "end": 1.0}, {"start": 2.0, "end": 3.0}]}}, + "fit_plans": {"es": {"fitted_segments": fitted}}} + recognized = [{"start": 0.0, "end": 1.0, "text": "hola"}, + {"start": 2.0, "end": 3.0, "text": "adios"}] + run, calls = _legacy_qc_route(monkeypatch, job, recognized) + if set(fit_ids) == {"a", "b"} and len(fit_ids) == 2: + assert run()["flagged_count"] == 0 + assert calls == ["spanish.wav"] + else: + original = copy.deepcopy(job) + with pytest.raises(HTTPException) as error: + run() + assert error.value.status_code == 409 + assert error.value.detail["code"] == "dub_qc_timing_identity_missing" + assert calls == [] + assert job == original + + +@pytest.mark.parametrize("map_id", ["a", "foreign"]) +def test_legacy_single_line_requires_matching_selected_track_identity(monkeypatch, map_id): + from fastapi import HTTPException + job = {"segments": [{"id": "a", "start": 10.0, "end": 11.0, "text": "later"}], + "segments_i18n": {"es": {map_id: "hola"}}, "seg_order": ["a"], + "dubbed_tracks": {"es": {"path": "spanish.wav", "timing_strategy": "strict_slot", + "source_segments": [{"start": 0.0, "end": 1.0}]}}} + run, calls = _legacy_qc_route(monkeypatch, job, [{"start": 0.0, "end": 1.0, "text": "hola"}]) + if map_id == "a": + assert run()["flagged_count"] == 0 + assert calls == ["spanish.wav"] + assert job["segments"][0]["start"] == 10.0 + else: + with pytest.raises(HTTPException) as error: + run() + assert error.value.status_code == 409 + assert calls == [] + + +def test_legacy_source_does_not_infer_strategy_from_another_track(): + from api.routers.dub_export import _qc_source_segments + from fastapi import HTTPException + job = {"timing_strategy": "smart_fit", + "dubbed_tracks": {"es": {"source_segments": [{"start": 0.0, "end": 1.0}]}}, + "fit_plans": {"es": {"fitted_segments": [{"id": "a", "start": 0.0, "end": 1.0}]}}} + with pytest.raises(HTTPException) as error: + _qc_source_segments(job, "es", [{"id": "a", "start": 10.0, "end": 11.0}]) + assert error.value.status_code == 409 + + +@pytest.mark.parametrize("existing", [[], [{"start": 0.0, "end": 1.0, "text": "old a"}, + {"start": 2.0, "end": 3.0, "text": "old b"}]]) +def test_regenerate_without_request_ids_persists_current_manifest_identity(monkeypatch, existing): + from api.routers.dub_generate import _sync_job_segments, _track_source_segments + from schemas.requests import DubRequest, DubSegment + req = DubRequest(language_code="es", segments=[DubSegment(start=2.0, end=3.0, text="hola"), + DubSegment(start=4.0, end=5.0, text="adios")]) + # Actual current-render contract: dub_generate seeds these same positional + # fallback names before rendering and synchronizing an ID-less request. + job = {"segments": existing, "seg_order": ["seg_0", "seg_1"]} + _sync_job_segments(job, req) + job["dubbed_tracks"] = {"es": {"path": "es.wav", "timing_strategy": "strict_slot", + "source_segments": _track_source_segments(job)}} + run, calls = _legacy_qc_route(monkeypatch, job, [{"start": 2.0, "end": 3.0, "text": "hola"}, + {"start": 4.0, "end": 5.0, "text": "adios"}]) + assert run()["flagged_count"] == 0 + assert calls == ["es.wav"] + assert [s["id"] for s in job["segments"]] == ["seg_0", "seg_1"] + assert job["segments_i18n"]["es"] == {"seg_0": "hola", "seg_1": "adios"} + + +@pytest.mark.parametrize("order", [None, [], ["seg_0"], ["seg_0", "seg_0"], + ["old_a", "old_b"], [0, 1], ["seg_0", "seg_1", "seg_2"], + ("seg_0", "seg_1")]) +def test_sync_does_not_invent_ids_from_invalid_render_order(order): + from api.routers.dub_generate import _sync_job_segments, _track_source_segments + from schemas.requests import DubRequest, DubSegment + req = DubRequest(language_code="es", segments=[DubSegment(start=0.0, end=1.0, text="hola"), + DubSegment(start=2.0, end=3.0, text="adios")]) + job = {"segments": [{"text": "old a"}, {"text": "old b"}], "seg_order": order} + _sync_job_segments(job, req) + assert all(s.get("id") is None for s in job["segments"]) + assert all(s["id"] is None for s in _track_source_segments(job)) + + +def test_sync_request_and_existing_identity_precede_current_manifest_fallback(): + from api.routers.dub_generate import _sync_job_segments + from schemas.requests import DubRequest, DubSegment + req = DubRequest(language_code="es", segment_ids=["explicit"], segments=[ + DubSegment(start=0.0, end=1.0, text="hola"), DubSegment(start=2.0, end=3.0, text="adios")]) + job = {"segments": [{"id": "old", "text": "old a"}, {"id": "preserved", "text": "old b"}], + "seg_order": ["explicit", "seg_1"]} + _sync_job_segments(job, req) + assert [s["id"] for s in job["segments"]] == ["explicit", "preserved"] + + +@pytest.mark.parametrize("owner_index", [0, 1]) +def test_sync_refuses_fallback_collision_and_qc_refuses_partial_identity(monkeypatch, owner_index): + from api.routers.dub_generate import _sync_job_segments, _track_source_segments + from schemas.requests import DubRequest, DubSegment + from fastapi import HTTPException + missing_index = 1 - owner_index + owner_id = f"seg_{missing_index}" + existing = [{"text": "old a"}, {"text": "old b"}] + existing[owner_index]["id"] = owner_id + job = {"segments": existing, "seg_order": ["seg_0", "seg_1"]} + req = DubRequest(language_code="es", segments=[DubSegment(start=0.0, end=1.0, text="hola"), + DubSegment(start=2.0, end=3.0, text="adios")]) + _sync_job_segments(job, req) + ids = [row.get("id") for row in job["segments"]] + assert ids[owner_index] == owner_id + assert ids[missing_index] is None + assert set(job["segments_i18n"]["es"].values()) == {"hola", "adios"} + job["dubbed_tracks"] = {"es": {"path": "es.wav", "timing_strategy": "strict_slot", + "source_segments": _track_source_segments(job)}} + original = copy.deepcopy(job) + run, calls = _legacy_qc_route(monkeypatch, job, [{"start": 0.0, "end": 1.0, "text": "hola"}, + {"start": 2.0, "end": 3.0, "text": "adios"}]) + with pytest.raises(HTTPException) as error: + run() + assert error.value.status_code == 409 + assert calls == [] + assert job == original + + +@pytest.mark.parametrize("existing", [[{"id": "dup", "text": "first"}, {"id": "dup", "text": "second"}], + [{"id": "2", "text": "first"}, {"id": "seg_2", "text": "second"}, {"text": "third"}]]) +def test_sync_rejects_colliding_text_keys_before_metadata_mutation(existing): + from api.routers.dub_generate import _sync_job_segments + from schemas.requests import DubRequest, DubSegment + from fastapi import HTTPException + job = {"segments": existing, "seg_order": [f"seg_{i}" for i in range(len(existing))], + "segments_i18n": {"es": {"prior": "keep"}}, + "segments_i18n_cue_sources": {"es": {"prior": "keep-source"}}, + "dubbed_tracks": {"es": {"path": "prior.wav"}}} + original = copy.deepcopy(job) + req = DubRequest(language_code="es", segments=[ + DubSegment(start=float(i*2), end=float(i*2+1), text=f"text {i}") for i in range(len(existing))]) + with pytest.raises(HTTPException) as error: + _sync_job_segments(job, req) + assert error.value.status_code == 409 + assert job == original + + +@pytest.mark.parametrize("source_ids", [["a", "a"], ["a", None], ["a", "foreign"], ["a"]]) +def test_qc_rejects_incomplete_or_duplicate_source_identity(monkeypatch, source_ids): + from fastapi import HTTPException + source = [{"id": sid, "start": float(i*2), "end": float(i*2+1)} for i, sid in enumerate(source_ids)] + job = {"segments": [{"id": "a", "start": 0.0, "end": 1.0, "text": "hola"}, + {"id": "b", "start": 2.0, "end": 3.0, "text": "adios"}], + "segments_i18n": {"es": {"a": "hola", "b": "adios"}}, "seg_order": ["a", "b"], + "dubbed_tracks": {"es": {"path": "es.wav", "timing_strategy": "strict_slot", "source_segments": source}}} + run, calls = _legacy_qc_route(monkeypatch, job, [{"start": 0.0, "end": 1.0, "text": "hola"}]) + original = copy.deepcopy(job) + with pytest.raises(HTTPException) as error: + run() + assert error.value.status_code == 409 + assert calls == [] + assert job == original diff --git a/tests/test_dub_translate.py b/tests/test_dub_translate.py index 1a8a1915b..45350d0d9 100644 --- a/tests/test_dub_translate.py +++ b/tests/test_dub_translate.py @@ -944,3 +944,15 @@ def packs(source, targets): await batch.retry_batch_job("retry-script") assert err.value.status_code == 422 assert "NLLB" in err.value.detail + + +@pytest.mark.parametrize("text", ["東京都新宿区都庁前駅", "会議は東京駅前で開催します。", "𠮷野家", "々"]) +def test_japanese_kanji_translation_passes_primary_script_guard(text): + from api.routers.dub_translate import _script_ratio, _looks_like_target + assert _script_ratio(text, "ja") == 1.0 + assert _looks_like_target(text, "ja") + + +def test_japanese_primary_script_guard_still_rejects_latin_only_output(): + from api.routers.dub_translate import _looks_like_target + assert not _looks_like_target("this is English", "ja") diff --git a/tests/test_longform_render.py b/tests/test_longform_render.py index bca5190be..a4e17d9be 100644 --- a/tests/test_longform_render.py +++ b/tests/test_longform_render.py @@ -113,6 +113,52 @@ def test_ffmetadata_escapes_special_chars(): assert r"title=a\=b\;c\#d" in doc +@pytest.mark.parametrize("newline", ["\n", "\r\n", "\r"]) +def test_ffmetadata_normalizes_line_endings_before_escaping(newline): + value = newline.join(["Opening =;#\\ paragraph.", "Closing paragraph."]) + expected = build_ffmetadata([("Opening =;#\\ paragraph.\nClosing paragraph.", 1000)], + {"description": "Opening =;#\\ paragraph.\nClosing paragraph."}) + assert build_ffmetadata([(value, 1000)], {"description": value}) == expected + assert "\r" not in expected + + +@pytest.mark.parametrize("newline", ["\n", "\r\n", "\r"]) +@pytest.mark.parametrize("fmt", ["m4b", "mp3"]) +def test_rendered_metadata_retains_description_paragraphs(tmp_path, newline, fmt): + """Exercise the shipped chapter-WAV → mux path without synthesis.""" + import json + import subprocess + import wave + + from services.ffmpeg_utils import find_ffmpeg, find_ffprobe + + ffmpeg, ffprobe = find_ffmpeg(), find_ffprobe() + if not ffmpeg or not ffprobe: + pytest.skip("ffmpeg and ffprobe required for metadata mux round trip") + audio = tmp_path / "chapter.wav" + with wave.open(str(audio), "wb") as handle: + handle.setnchannels(1) + handle.setsampwidth(2) + handle.setframerate(16000) + handle.writeframes(b"\0\0" * 16000) + concat = tmp_path / "chapters.txt" + concat.write_text(build_concat_list([str(audio)]), encoding="utf-8") + metadata = tmp_path / "chapters.ffmeta" + metadata.write_text(build_ffmetadata([("Chapter one", 1000)], { + "title": "My =;#\\ book", + "description": newline.join(["Opening =;#\\ paragraph.", "Closing paragraph."]), + }), encoding="utf-8", newline="") + output = tmp_path / f"book.{fmt}" + subprocess.run(build_render_cmd(ffmpeg, str(concat), str(metadata), str(output), fmt=fmt), + check=True, capture_output=True, timeout=30) + probe = subprocess.run([ffprobe, "-v", "error", "-show_entries", "format_tags=title,comment", + "-of", "json", str(output)], + check=True, capture_output=True, timeout=30) + tags = json.loads(probe.stdout)["format"]["tags"] + assert tags["title"] == "My =;#\\ book" + assert tags["comment"] == "Opening =;#\\ paragraph.\nClosing paragraph." + + # ── concat list ───────────────────────────────────────────────────────────── def test_concat_list_quotes_and_escapes(): @@ -373,3 +419,53 @@ def test_render_cmd_off_emits_no_af_even_with_stray_measured(): cmd2 = build_render_cmd("ffmpeg", "c.txt", "m.ff", "o.m4b") # default: no loudness/measured assert "-af" not in cmd2 + + +@pytest.mark.parametrize("fmt", ["m4b", "mp3"]) +def test_public_renderer_retains_metadata_with_windows_newline_defaults(tmp_path, monkeypatch, fmt): + """Native full renderer, with Windows text-mode translation on any host.""" + import asyncio + import builtins + import json + import subprocess + torch = pytest.importorskip("torch") + from api.routers import audiobook + from core.db import init_db + from services.audiobook import AudiobookPlan, Chapter, Span + from services.ffmpeg_utils import find_ffmpeg, find_ffprobe + + if not find_ffmpeg() or not find_ffprobe(): + pytest.skip("ffmpeg and ffprobe required for metadata mux round trip") + init_db() + outputs = tmp_path / "outputs" + outputs.mkdir() + monkeypatch.setattr("core.config.OUTPUTS_DIR", str(outputs)) + + def synth_factory(*args, **kwargs): + return {"mode": "generic", "engine_id": "stub", "sample_rate": 24000, + "resolve": lambda _: {"ref_audio": None, "ref_text": None, + "instruct": None, "seed": None}, + "synth": lambda *args, **kwargs: torch.zeros(2400)} + monkeypatch.setattr(audiobook, "_build_synth", synth_factory) # Model boundary only. + native_open = builtins.open + + def windows_text_open(file, mode="r", *args, **kwargs): + if "w" in mode and "b" not in mode: + kwargs.setdefault("newline", "\r\n") + return native_open(file, mode, *args, **kwargs) + monkeypatch.setattr(audiobook, "open", windows_text_open, raising=False) + + async def render(): + plan = AudiobookPlan(chapters=[Chapter(title="One", spans=[Span(voice_id=None, text="Hello.")])]) + return [json.loads(frame[len("data:"):].strip()) + async for frame in audiobook._render_longform_sse( + plan, default_voice=None, fmt=fmt, + metadata={"description": "Opening paragraph.\r\nClosing paragraph."})] + + events = asyncio.run(render()) + assert events[-1]["type"] == "done", events + output = outputs / events[-1]["output"] + probe = subprocess.run([find_ffprobe(), "-v", "error", "-show_entries", "format_tags=comment", + "-of", "json", str(output)], + check=True, capture_output=True, timeout=30) + assert json.loads(probe.stdout)["format"]["tags"]["comment"] == "Opening paragraph.\nClosing paragraph." diff --git a/tests/test_longform_segment_cache.py b/tests/test_longform_segment_cache.py index 2bb02c302..e2cfe3c5d 100644 --- a/tests/test_longform_segment_cache.py +++ b/tests/test_longform_segment_cache.py @@ -89,6 +89,41 @@ def test_chapter_key_golden_unchanged_by_segment_layer(): assert key == "ce2accacf51a70d04da0" +@pytest.mark.parametrize("keep_chapter", [True, False]) +@pytest.mark.parametrize("languages", [("English", "French"), (None, "English")]) +def test_language_change_invalidates_chapter_and_segment_audio(tmp_path, keep_chapter, languages): + """Identical normalized words can be synthesized in different languages.""" + import soundfile as sf + + chapter = _chapter("Hello.") + calls = [] + + def render(language, amplitude): + def synth(text, voice_id, speed=None): + calls.append((language, text)) + return torch.full((2400,), amplitude) + return _render_chapter_cached(chapter, synth, _SR, "eng", _resolve, + str(tmp_path), language=language) + + first_language, second_language = languages + first, *_ = render(first_language, 0.1) + if not keep_chapter: + os.remove(first) # Force the second render to consult its segment layer. + second, _dur, cached, stats = render(second_language, 0.2) + assert cached is False + assert stats == {"total": 1, "cached": 0} + assert calls == [(first_language, "Hello."), (second_language, "Hello.")] + assert first != second + audio, _ = sf.read(second) + # Compare the middle of the 2400-sample take: resampling can overshoot + # its edges, and the chapter may append silence after it. + assert audio[600:1800].mean() == pytest.approx(0.2, abs=0.001) + _path, _dur, cached, stats = render(second_language, 0.3) + assert cached is True + assert stats is None + assert len(calls) == 2 + + # ── one-sentence edit → one segment re-renders ────────────────────────────── def test_one_sentence_edit_rerenders_exactly_one_segment(tmp_path): diff --git a/tests/test_mcp_binding_concurrent_updates.py b/tests/test_mcp_binding_concurrent_updates.py new file mode 100644 index 000000000..61f21fb13 --- /dev/null +++ b/tests/test_mcp_binding_concurrent_updates.py @@ -0,0 +1,103 @@ +"""Concurrent Settings edits must preserve independently updated voice fields.""" +from concurrent.futures import ThreadPoolExecutor +from contextlib import contextmanager +import sqlite3 +import threading + +import pytest + + +@pytest.fixture +def bindings(tmp_path, monkeypatch): + from services import mcp_bindings + + path = tmp_path / "bindings.sqlite" + with sqlite3.connect(path) as conn: + conn.execute("CREATE TABLE mcp_client_bindings (client_id TEXT PRIMARY KEY, " + "label TEXT, profile_id TEXT, default_engine TEXT, " + "last_seen_at REAL, created_at REAL)") + + @contextmanager + def connect(): + conn = sqlite3.connect(path, timeout=5) + conn.row_factory = sqlite3.Row + try: + yield conn + conn.commit() + except BaseException: + conn.rollback() + raise + finally: + conn.close() + + monkeypatch.setattr(mcp_bindings, "db_conn", connect) + return mcp_bindings, connect + + +@pytest.mark.parametrize("existing", [True, False]) +def test_concurrent_partial_upserts_preserve_both_edits(bindings, monkeypatch, existing): + service, connect = bindings + if existing: + service.upsert_binding("agent", label="before", profile_id="before-voice") + first_write, second_write = threading.Event(), threading.Event() + actor = threading.local() + + class Connection: + def __init__(self, conn): + self.conn = conn + + def execute(self, sql, params=()): + write = sql.startswith(("INSERT", "UPDATE", "BEGIN IMMEDIATE")) + if actor.name == "second" and write: + # Signal admission before SQLite can block on the first writer. + second_write.set() + if actor.name == "first" and sql.startswith(("INSERT", "UPDATE")): + first_write.set() + assert second_write.wait(5), "second writer never reached SQLite" + return self.conn.execute(sql, params) + + @contextmanager + def overlapping_connect(): + with connect() as conn: + yield Connection(conn) + + monkeypatch.setattr(service, "db_conn", overlapping_connect) + + def update(name, **fields): + actor.name = name + return service.upsert_binding("agent", **fields) + + with ThreadPoolExecutor(max_workers=2) as pool: + first = pool.submit(update, "first", label="new label") + assert first_write.wait(5), "first writer never reached SQLite" + second = pool.submit(update, "second", profile_id="new voice") + assert first.result(timeout=10)["label"] == "new label" + assert second.result(timeout=10)["profile_id"] == "new voice" + monkeypatch.setattr(service, "db_conn", connect) + row = service.get_binding("agent") + assert row["label"] == "new label" + assert row["profile_id"] == "new voice" + + +def test_omitted_fields_preserve_values_and_empty_fields_clear_them(bindings): + service, _ = bindings + original = service.upsert_binding(" agent ", label="named", profile_id="voice", default_engine="engine") + updated = service.upsert_binding("agent", label="renamed") + assert updated["profile_id"] == "voice" + assert updated["default_engine"] == "engine" + assert updated["created_at"] == original["created_at"] + cleared = service.upsert_binding("agent", profile_id="", default_engine="") + assert cleared["profile_id"] is None + assert cleared["default_engine"] is None + assert cleared["label"] == "renamed" + + +def test_failed_update_rolls_back_the_entire_binding(bindings): + service, connect = bindings + before = service.upsert_binding("agent", label="before", profile_id="voice") + with connect() as conn: + conn.execute("CREATE TRIGGER reject_edit BEFORE UPDATE ON mcp_client_bindings " + "BEGIN SELECT RAISE(ABORT, 'edit rejected'); END") + with pytest.raises(sqlite3.IntegrityError, match="edit rejected"): + service.upsert_binding("agent", label="after", profile_id="after-voice") + assert service.get_binding("agent") == before diff --git a/tests/test_models_dir_setting.py b/tests/test_models_dir_setting.py index 33e32606c..4b8ccb71c 100644 --- a/tests/test_models_dir_setting.py +++ b/tests/test_models_dir_setting.py @@ -148,7 +148,7 @@ def test_get_shape(env): def test_path_with_spaces_survives_the_full_persistence_chain(env, tmp_path, monkeypatch): """#1186 class: the wizard/Settings dirs regularly contain spaces ('D:\\Program Data\\OmniVoice\\Model Cache'). The durable env file stores - the value as an UNQUOTED dotenv line and main.py re-reads it through + the value as a dotenv-compatible line and main.py re-reads it through python-dotenv, so a writer/parser quoting regression would truncate at the first space and silently redirect every model download while Settings still shows the chosen folder. Pin the whole chain byte-for-byte: @@ -167,6 +167,28 @@ def test_path_with_spaces_survives_the_full_persistence_chain(env, tmp_path, mon assert s.get_models_dir()["configured"] == abs_target +@pytest.mark.parametrize("folder", ["Books #1", "Model's #1", r"Models\fonts #1"]) +def test_models_directory_comment_characters_survive_route_and_startup(env, tmp_path, monkeypatch, folder): + from fastapi.testclient import TestClient + from api.routers.settings import router + from core import user_env + + target = str(tmp_path / folder) + body = _body(target) + app = fastapi.FastAPI() + app.include_router(router) + with TestClient(app, client=("127.0.0.1", 50000)) as client: + response = client.put("/api/settings/storage/models-dir", json={"authorization": body.authorization}) + assert response.status_code == 200 + assert response.json()["configured"] == target + assert client.get("/api/settings/storage/models-dir").json()["configured"] == target + monkeypatch.setenv("OMNIVOICE_CACHE_DIR", "/stale/launcher/value") + assert user_env.load_into_environ() is True + assert os.environ["OMNIVOICE_CACHE_DIR"] == target + assert client.get("/api/settings/storage/models-dir").json()["configured"] == target + assert not os.path.exists(target.split(" #", 1)[0]) + + def test_windows_drive_paths_with_spaces_round_trip_verbatim(env): """#1186 class, non-system-drive half: 'D:\\…' paths with spaces (and '&') must survive persist → read-back → the exact dotenv parse that diff --git a/tests/test_profile_design_save_decouple.py b/tests/test_profile_design_save_decouple.py index 3e8537f2a..54241c233 100644 --- a/tests/test_profile_design_save_decouple.py +++ b/tests/test_profile_design_save_decouple.py @@ -19,6 +19,7 @@ import importlib import json import os +from pathlib import Path import pytest @@ -88,6 +89,59 @@ def test_design_save_creates_row_when_model_unavailable(iso, monkeypatch): assert stored == {**_VD, "Style": "Auto", "EnglishAccent": "Auto", "ChineseDialect": "Auto"} +def test_cold_design_save_does_not_load_engine(iso, monkeypatch): + """An offline/cold renderer cannot hold the persistence request open.""" + _, db, prof = iso + from api.routers import archetypes as arch + from services import model_manager + monkeypatch.setattr(model_manager, "get_model_status", lambda: {"loaded": False}) + renders = [] + + async def blocked_render(*args): + renders.append(args) + await asyncio.Event().wait() + + monkeypatch.setattr(arch, "_render_archetype_wav", blocked_render) + + async def save(): + return await asyncio.wait_for(prof.create_profile( + name="Offline design", ref_audio=None, ref_text="", instruct="female", + language="English", seed=None, personality="", kind="design", + vd_states=json.dumps(_VD), image=None, + ), timeout=1) + + result = asyncio.run(save()) + assert not renders + with db.db_conn() as conn: + row = conn.execute("SELECT ref_audio_path FROM voice_profiles WHERE id=?", (result["id"],)).fetchone() + assert row is not None and not row["ref_audio_path"] + + +def test_warm_design_save_preserves_identity_sample(iso, monkeypatch): + """A resident engine still renders the reference used by later synthesis.""" + cfg, db, prof = iso + from api.routers import archetypes as arch + from services import model_manager + monkeypatch.setattr(model_manager, "get_model_status", lambda: {"loaded": True}) + renders = [] + + async def render(recipe, out_path): + renders.append(recipe) + out_path.parent.mkdir(parents=True, exist_ok=True) + out_path.write_bytes(b"identity sample") + + monkeypatch.setattr(arch, "_render_archetype_wav", render) + result = asyncio.run(prof.create_profile( + name="Warm design", ref_audio=None, ref_text="Sample line", instruct="female", + language="English", seed=None, personality="", kind="design", + vd_states=json.dumps(_VD), image=None, + )) + assert len(renders) == 1 and renders[0]["sample_script"] == "Sample line" + with db.db_conn() as conn: + row = conn.execute("SELECT ref_audio_path FROM voice_profiles WHERE id=?", (result["id"],)).fetchone() + assert (Path(cfg.VOICES_DIR) / row["ref_audio_path"]).read_bytes() == b"identity sample" + + def test_all_auto_design_is_saveable(iso, monkeypatch): """An all-Auto design (empty instruct) saves; it isn't gated on instruct.""" _, db, prof = iso diff --git a/tests/test_profile_relock_reference.py b/tests/test_profile_relock_reference.py new file mode 100644 index 000000000..7d66da8b3 --- /dev/null +++ b/tests/test_profile_relock_reference.py @@ -0,0 +1,364 @@ +"""Re-locking a voice gives its replacement reference a new cache identity.""" +import os +from pathlib import Path + +import pytest +import soundfile as sf +import torch +from fastapi import FastAPI +from fastapi.testclient import TestClient + + +@pytest.fixture +def profile(tmp_path, monkeypatch): + from api.routers import profiles + from core import db, config + + VOICES_DIR, OUTPUTS_DIR = str(tmp_path / 'voices'), str(tmp_path / 'outputs') + for module in (profiles, config): + monkeypatch.setattr(module, 'VOICES_DIR', VOICES_DIR) + monkeypatch.setattr(module, 'OUTPUTS_DIR', OUTPUTS_DIR) + + monkeypatch.setattr(db, "DB_PATH", str(tmp_path / "profiles.db")) + db.init_db() + os.makedirs(VOICES_DIR, exist_ok=True) + os.makedirs(OUTPUTS_DIR, exist_ok=True) + with db.db_conn() as conn: + conn.execute("INSERT INTO voice_profiles (id,name,ref_audio_path) VALUES ('voice','Voice','')") + for take, amplitude in [("first", 0.1), ("second", 0.2)]: + sf.write(os.path.join(OUTPUTS_DIR, f"{take}.wav"), + torch.full((2400,), amplitude).numpy(), 24000) + conn.execute("INSERT INTO generation_history (id,text,profile_id,audio_path) VALUES (?,?,?,?)", + (take, "Same reference text", "voice", f"{take}.wav")) + app = FastAPI() + app.include_router(profiles.router) + with TestClient(app) as client: + yield client, db, VOICES_DIR + + +@pytest.mark.parametrize("keep_chapter", [True, False]) +def test_relock_does_not_reuse_previous_take_audio(profile, tmp_path, keep_chapter): + from api.routers.audiobook import _render_chapter_cached, _resolve_voice + from services.audiobook import Chapter, Span + + client, _db, voices = profile + chapter = Chapter(title="C", spans=[Span(voice_id="voice", text="Hello.", pause_ms_after=0)]) + calls = [] + + def synth(text, voice_id, speed=None): + reference = _resolve_voice(voice_id)["ref_audio"] + samples, _ = sf.read(reference) + calls.append(reference) + return torch.tensor(samples, dtype=torch.float32) + + response = client.post('/profiles/voice/lock', data={'history_id': 'first', 'seed': '7'}) + assert response.status_code == 200, response.text + first_reference = response.json()['locked_audio_path'] + first, *_ = _render_chapter_cached(chapter, synth, 24000, 'eng', _resolve_voice, str(tmp_path / 'cache')) + if not keep_chapter: + os.remove(first) + response = client.post('/profiles/voice/lock', data={'history_id': 'second', 'seed': '7'}) + assert response.status_code == 200, response.text + second_reference = response.json()['locked_audio_path'] + second, _duration, cached, stats = _render_chapter_cached(chapter, synth, 24000, 'eng', _resolve_voice, + str(tmp_path / 'cache')) + assert cached is False + assert stats == {'total': 1, 'cached': 0} + assert len(calls) == 2 + assert first != second + assert first_reference != second_reference + assert os.path.exists(os.path.join(voices, first_reference)) + # Compare the middle of the 2400-sample take: resampling can overshoot + # its edges, and the chapter may append silence after it. + assert sf.read(second)[0][600:1800].mean() == pytest.approx(0.2, abs=0.001) + served = client.get('/profiles/voice/audio') + assert served.status_code == 200 + assert served.content == Path(voices, second_reference).read_bytes() + + +def test_failed_relock_keeps_original_reference_and_cleans_new_copy(profile): + client, db, voices = profile + initial = client.post('/profiles/voice/lock', data={'history_id': 'first'}).json()['locked_audio_path'] + original = Path(voices, initial).read_bytes() + with db.db_conn() as conn: + conn.execute("CREATE TRIGGER refuse_lock BEFORE UPDATE OF locked_audio_path ON voice_profiles " + "BEGIN SELECT RAISE(ABORT, 'lock update failed'); END") + import sqlite3 + with pytest.raises(sqlite3.IntegrityError, match='lock update failed'): + client.post('/profiles/voice/lock', data={'history_id': 'second'}) + with db.db_conn() as conn: + assert conn.execute("SELECT locked_audio_path FROM voice_profiles WHERE id='voice'").fetchone()[0] == initial + assert Path(voices, initial).read_bytes() == original + assert os.listdir(voices) == [initial] + + +def test_relock_keeps_a_reference_still_used_by_another_profile(profile): + """Cleanup must preserve references shared by imported profile rows.""" + client, db, voices = profile + initial = client.post('/profiles/voice/lock', data={'history_id': 'first'}).json()['locked_audio_path'] + original = Path(voices, initial).read_bytes() + with db.db_conn() as conn: + conn.execute("INSERT INTO voice_profiles (id,name,ref_audio_path) VALUES (?,?,?)", + ('shared', 'Shared reference', initial)) + response = client.post('/profiles/voice/lock', data={'history_id': 'second'}) + assert response.status_code == 200 + assert response.json()['locked_audio_path'] != initial + assert Path(voices, initial).read_bytes() == original + + +def test_existing_longform_voice_snapshot_survives_relock(profile): + from api.routers.audiobook import _build_synth + client, _db, _voices = profile + first = client.post('/profiles/voice/lock', data={'history_id': 'first'}).json() + # Real longform resolver caches the reference before a worker reads it. + running = _build_synth(default_voice='voice') + before = running['resolve']('voice') + old_bytes = Path(before['ref_audio']).read_bytes() + response = client.post('/profiles/voice/lock', data={'history_id': 'second'}) + assert response.status_code == 200 + assert response.json()['locked_audio_path'] != first['locked_audio_path'] + saved = running['resolve']('voice') + assert saved['ref_audio'] == before['ref_audio'] + assert Path(saved['ref_audio']).read_bytes() == old_bytes + assert sf.read(saved['ref_audio'])[0].mean() == pytest.approx(0.1, abs=0.001) + + +def test_explicit_profile_deletion_reclaims_retained_locked_versions(profile): + client, _db, voices = profile + versions = [] + for take in ('first', 'second'): + versions.append(client.post('/profiles/voice/lock', data={'history_id': take}).json()['locked_audio_path']) + assert all(Path(voices, name).exists() for name in versions) + unrelated = Path(voices, 'voice_locked_not-a-generated-version.wav') + unrelated.write_bytes(b'unrelated') + response = client.delete('/profiles/voice') + assert response.status_code == 200, response.text + assert not any(Path(voices, name).exists() for name in versions) + assert unrelated.read_bytes() == b'unrelated' + + +def test_explicit_deletion_preserves_a_retained_version_shared_by_another_profile(profile): + client, db, voices = profile + old = client.post('/profiles/voice/lock', data={'history_id': 'first'}).json()['locked_audio_path'] + with db.db_conn() as conn: + conn.execute("INSERT INTO voice_profiles(id,name,ref_audio_path) VALUES(?,?,?)", ('shared', 'Shared', old)) + latest = client.post('/profiles/voice/lock', data={'history_id': 'second'}).json()['locked_audio_path'] + response = client.delete('/profiles/voice') + assert response.status_code == 200 + assert Path(voices, old).exists() + assert not Path(voices, latest).exists() + assert client.get('/profiles/shared/audio').status_code == 200 + + +@pytest.mark.parametrize('relock', [False, True]) +def test_profile_deletion_waits_for_live_longform_reference(profile, relock): + from api.routers.audiobook import _build_synth + client, db, _voices = profile + client.post('/profiles/voice/lock', data={'history_id': 'first'}) + running = _build_synth(default_voice='voice') + reference = running['resolve']('voice')['ref_audio'] + original = Path(reference).read_bytes() + if relock: + client.post('/profiles/voice/lock', data={'history_id': 'second'}) + response = client.delete('/profiles/voice') + assert response.status_code == 409, response.text + assert Path(reference).read_bytes() == original + with db.db_conn() as conn: + assert conn.execute("SELECT id FROM voice_profiles WHERE id='voice'").fetchone() + # Dropping the actual cached resolver lets the settled profile be deleted. + del running + response = client.delete('/profiles/voice') + assert response.status_code == 200, response.text + assert not Path(reference).exists() + + +def test_profile_deletion_waits_for_every_cached_reader_and_ignores_other_profiles(profile): + from api.routers.audiobook import _build_synth + client, db, _voices = profile + client.post('/profiles/voice/lock', data={'history_id': 'first'}) + first = _build_synth(default_voice='voice') + old = first['resolve']('voice')['ref_audio'] + client.post('/profiles/voice/lock', data={'history_id': 'second'}) + second = _build_synth(default_voice='voice') + latest = second['resolve']('voice')['ref_audio'] + with db.db_conn() as conn: + conn.execute("INSERT INTO voice_profiles(id,name,ref_audio_path) VALUES('other','Other','')") + assert client.delete('/profiles/other').status_code == 200 + assert client.delete('/profiles/voice').status_code == 409 + del first + assert client.delete('/profiles/voice').status_code == 409 + assert Path(old).exists() and Path(latest).exists() + del second + assert client.delete('/profiles/voice').status_code == 200 + assert not Path(old).exists() and not Path(latest).exists() + + +def test_pending_render_worker_keeps_reference_custody_after_request_owner_drops(profile): + import threading + from concurrent.futures import ThreadPoolExecutor + from api.routers.audiobook import _build_synth + client, _db, _voices = profile + client.post('/profiles/voice/lock', data={'history_id': 'first'}) + running = _build_synth(default_voice='voice') + path = running['resolve']('voice')['ref_audio'] + original = Path(path).read_bytes() + entered, release = threading.Event(), threading.Event() + + def worker(resolve): + entered.set() + assert release.wait(5) + return Path(resolve('voice')['ref_audio']).read_bytes() + + with ThreadPoolExecutor(max_workers=1) as pool: + result = pool.submit(worker, running['resolve']) + assert entered.wait(5) + del running # The HTTP request can finish before its worker does. + try: + assert client.delete('/profiles/voice').status_code == 409 + finally: + release.set() + assert result.result(timeout=5) == original + assert client.delete('/profiles/voice').status_code == 200 + assert not Path(path).exists() + + +def test_live_shared_reference_does_not_block_deleting_its_original_profile(profile): + from api.routers.audiobook import _build_synth + client, db, _voices = profile + locked = client.post('/profiles/voice/lock', data={'history_id': 'first'}).json()['locked_audio_path'] + running = _build_synth(default_voice='voice') + path = running['resolve']('voice')['ref_audio'] + with db.db_conn() as conn: + conn.execute("INSERT INTO voice_profiles(id,name,ref_audio_path) VALUES('shared','Shared',?)", (locked,)) + assert client.delete('/profiles/voice').status_code == 200 + assert Path(path).exists() + assert client.delete('/profiles/shared').status_code == 409 + del running + assert client.delete('/profiles/shared').status_code == 200 + assert not Path(path).exists() + + +def test_delete_does_not_adopt_a_shared_path_after_its_precommit_guard(profile, monkeypatch): + from contextlib import contextmanager + from api.routers import profiles + from api.routers.audiobook import _build_synth + + client, db, voices = profile + locked = client.post('/profiles/voice/lock', data={'history_id': 'first'}).json()['locked_audio_path'] + running = _build_synth(default_voice='voice') + path = running['resolve']('voice')['ref_audio'] + original = Path(path).read_bytes() + replacement = Path(voices, 'shared-replacement.wav') + replacement.write_bytes(original) + with db.db_conn() as conn: + conn.execute("INSERT INTO voice_profiles(id,name,ref_audio_path) VALUES('shared','Shared',?)", (locked,)) + + moved = False + real_db_conn = db.db_conn + + @contextmanager + def change_other_profile_after_delete_commit(): + nonlocal moved + with real_db_conn() as conn: + yield conn + if not moved: + with real_db_conn() as conn: + deleted = conn.execute("SELECT 1 FROM voice_profiles WHERE id='voice'").fetchone() is None + if deleted: + # Actual SQLite update at the committed-delete boundary. This + # models the public replacement writer's ref/path reset without + # invoking that writer's separate, pre-existing cleanup path. + moved = True + with real_db_conn() as conn: + conn.execute("UPDATE voice_profiles SET ref_audio_path=?, locked_audio_path='', consent_audio_path='' WHERE id='shared'", (replacement.name,)) + + monkeypatch.setattr(profiles, 'db_conn', change_other_profile_after_delete_commit) + response = client.delete('/profiles/voice') + assert response.status_code == 200, response.text + assert moved + assert Path(path).read_bytes() == original + assert running['resolve']('voice')['ref_audio'] == path + + +@pytest.mark.parametrize("shared_column", ["ref_audio_path", "locked_audio_path", "consent_audio_path"]) +def test_unlock_preserves_locked_reference_shared_by_another_profile(profile, shared_column): + client, db, voices = profile + locked = client.post('/profiles/voice/lock', data={'history_id': 'first'}).json()['locked_audio_path'] + original = Path(voices, locked).read_bytes() + with db.db_conn() as conn: + conn.execute(f"INSERT INTO voice_profiles(id,name,{shared_column}) VALUES('shared','Shared',?)", (locked,)) + response = client.post('/profiles/voice/unlock') + assert response.status_code == 200, response.text + assert Path(voices, locked).read_bytes() == original + with db.db_conn() as conn: + row = conn.execute("SELECT is_locked,locked_audio_path FROM voice_profiles WHERE id='voice'").fetchone() + assert tuple(row) == (0, '') + + +@pytest.mark.parametrize("retained_column", ["ref_audio_path", "consent_audio_path"]) +def test_unlock_preserves_locked_file_retained_by_same_profile(profile, retained_column): + client, db, voices = profile + locked = client.post('/profiles/voice/lock', data={'history_id': 'first'}).json()['locked_audio_path'] + original = Path(voices, locked).read_bytes() + with db.db_conn() as conn: + conn.execute(f"UPDATE voice_profiles SET {retained_column}=? WHERE id='voice'", (locked,)) + response = client.post('/profiles/voice/unlock') + assert response.status_code == 200, response.text + assert Path(voices, locked).read_bytes() == original + + +def test_unlock_keeps_pending_render_reference_then_deletion_reclaims_it(profile): + import threading + from concurrent.futures import ThreadPoolExecutor + from api.routers.audiobook import _build_synth + + client, _db, voices = profile + locked = client.post('/profiles/voice/lock', data={'history_id': 'first'}).json()['locked_audio_path'] + running = _build_synth(default_voice='voice') + path = running['resolve']('voice')['ref_audio'] + original = Path(path).read_bytes() + entered, release = threading.Event(), threading.Event() + + def worker(resolve): + entered.set() + assert release.wait(5) + return Path(resolve('voice')['ref_audio']).read_bytes() + + with ThreadPoolExecutor(max_workers=1) as pool: + result = pool.submit(worker, running['resolve']) + assert entered.wait(5) + del running + try: + response = client.post('/profiles/voice/unlock') + assert response.status_code == 200, response.text + assert Path(path).read_bytes() == original + assert client.delete('/profiles/voice').status_code == 409 + finally: + release.set() + assert result.result(timeout=5) == original + assert client.delete('/profiles/voice').status_code == 200 + assert not Path(voices, locked).exists() + + +def test_unlock_reclaims_unused_locked_file(profile): + client, _db, voices = profile + locked = client.post('/profiles/voice/lock', data={'history_id': 'first'}).json()['locked_audio_path'] + response = client.post('/profiles/voice/unlock') + assert response.status_code == 200, response.text + assert not Path(voices, locked).exists() + + +def test_failed_unlock_preserves_locked_reference_and_row(profile): + import sqlite3 + client, db, voices = profile + locked = client.post('/profiles/voice/lock', data={'history_id': 'first'}).json()['locked_audio_path'] + original = Path(voices, locked).read_bytes() + with db.db_conn() as conn: + conn.execute("CREATE TRIGGER refuse_unlock BEFORE UPDATE OF locked_audio_path ON voice_profiles " + "WHEN NEW.locked_audio_path = '' BEGIN SELECT RAISE(ABORT, 'unlock update failed'); END") + with pytest.raises(sqlite3.IntegrityError, match='unlock update failed'): + client.post('/profiles/voice/unlock') + assert Path(voices, locked).read_bytes() == original + with db.db_conn() as conn: + row = conn.execute("SELECT is_locked,locked_audio_path FROM voice_profiles WHERE id='voice'").fetchone() + assert tuple(row) == (1, locked) diff --git a/tests/test_profile_unification.py b/tests/test_profile_unification.py index d0bc23bc4..8f43a07e4 100644 --- a/tests/test_profile_unification.py +++ b/tests/test_profile_unification.py @@ -52,6 +52,8 @@ async def _fake(a, out_path): out_path.write_bytes(_FAKE_AUDIO) from api.routers import archetypes as _arch + from services import model_manager + monkeypatch.setattr(model_manager, "get_model_status", lambda: {"loaded": True}) monkeypatch.setattr(_arch, "_render_archetype_wav", _fake) return _fake diff --git a/tests/test_pronunciation_backup_order.py b/tests/test_pronunciation_backup_order.py new file mode 100644 index 000000000..6cad92e34 --- /dev/null +++ b/tests/test_pronunciation_backup_order.py @@ -0,0 +1,83 @@ +"""Dictionary backups retain chronological and same-import precedence.""" +import pytest + + +@pytest.fixture +def client(tmp_path, monkeypatch): + from fastapi import FastAPI + from fastapi.testclient import TestClient + from api.routers.pronunciation import router + from core import db + monkeypatch.setattr(db, "DB_PATH", str(tmp_path / "pronunciation.db")) + db.init_db() + app = FastAPI() + app.include_router(router) + with TestClient(app, client=("127.0.0.1", 50000)) as native_client: + yield native_client + + +def seed_bulk_import_rows(last_term): + from core.db import db_conn + # Actual UUID prefixes captured from a native public bulk import. A single + # import assigns both rows the same timestamp; UUID ordering is unrelated + # to their order in the authored JSON dictionary. + with db_conn() as conn: + conn.executemany("INSERT INTO pronunciation_entries " + "(id,term,replacement,type,language,enabled,created_at) VALUES (?,?,?,?,?,?,?)", [ + ("8ee4ed4d-dc6", "GIF", "first", "respelling", "*", 1, 1000.0), + ("41eb97e3-95c", last_term, "last", "respelling", "*", 1, 1000.0), + ]) + + +@pytest.mark.parametrize("last_term", ["gif", "GIF"]) +def test_preview_and_synthesis_loader_agree_on_last_imported_entry(client, last_term): + from services.pronunciation import apply_pronunciation, load_entries_from_db + seed_bulk_import_rows(last_term) + assert client.post("/pronunciation/test", json={"text": "GIF"}).json()["substituted"] == "last" + assert apply_pronunciation("GIF", load_entries_from_db()) == "last" + + +@pytest.mark.parametrize("last_term", ["gif", "GIF"]) +def test_dictionary_list_and_export_keep_bulk_import_order(client, last_term): + seed_bulk_import_rows(last_term) + assert [e["replacement"] for e in client.get("/pronunciation").json()] == ["first", "last"] + assert [e["replacement"] for e in client.get("/pronunciation/export").json()["entries"]] == ["first", "last"] + + +@pytest.mark.parametrize("last_term", ["gif", "GIF"]) +def test_repeated_backup_restore_keeps_pronunciation_precedence(client, last_term): + from services.pronunciation import apply_pronunciation, load_entries_from_db + seed_bulk_import_rows(last_term) + before = client.post("/pronunciation/test", json={"text": "GIF"}).json()["substituted"] + assert before == "last" + for _ in range(3): + backup = client.get("/pronunciation/export").json() + assert client.post("/pronunciation/import", json={**backup, "replace": True}).json()["imported"] == 2 + assert client.post("/pronunciation/test", json={"text": "GIF"}).json()["substituted"] == before + assert apply_pronunciation("GIF", load_entries_from_db()) == before + + +def test_creation_time_still_precedes_insertion_tie_break(client): + from core.db import db_conn + from services.pronunciation import apply_pronunciation, load_entries_from_db + with db_conn() as conn: + conn.executemany("INSERT INTO pronunciation_entries " + "(id,term,replacement,type,language,enabled,created_at) VALUES (?,?,?,?,?,?,?)", [ + ("a", "GIF", "newer", "respelling", "*", 1, 20.0), + ("b", "gif", "older", "respelling", "*", 1, 10.0), + ]) + assert apply_pronunciation("GIF", load_entries_from_db()) == "newer" + assert client.post("/pronunciation/test", json={"text": "GIF"}).json()["substituted"] == "newer" + assert [e["replacement"] for e in client.get("/pronunciation/export").json()["entries"]] == ["older", "newer"] + + +def test_unique_global_scoped_and_disabled_entries_roundtrip(client): + entries = [{"term": "GIF", "replacement": "global", "language": "*"}, + {"term": "GIF", "replacement": "Spanish", "language": "es"}, + {"term": "WAV", "replacement": "disabled", "language": "*", "enabled": False}] + assert client.post("/pronunciation/import", json={"entries": entries, "replace": True}).status_code == 200 + for _ in range(3): + assert client.post("/pronunciation/test", json={"text": "GIF WAV", "language": "es"}).json()["substituted"] == "Spanish WAV" + assert client.post("/pronunciation/test", json={"text": "GIF WAV", "language": "en"}).json()["substituted"] == "global WAV" + backup = client.get("/pronunciation/export").json() + assert client.post("/pronunciation/import", json={**backup, "replace": True}).status_code == 200 diff --git a/tests/test_pronunciation_language_scopes.py b/tests/test_pronunciation_language_scopes.py new file mode 100644 index 000000000..64d66b923 --- /dev/null +++ b/tests/test_pronunciation_language_scopes.py @@ -0,0 +1,238 @@ +"""Native pronunciation scopes match the language names used by synthesis.""" +import pytest + + +@pytest.fixture +def client(tmp_path, monkeypatch): + from fastapi import FastAPI + from fastapi.testclient import TestClient + from core import db + from api.routers.pronunciation import router + + monkeypatch.setattr(db, "DB_PATH", str(tmp_path / "pronunciation.db")) + db.init_db() + app = FastAPI() + app.include_router(router) + native_client = TestClient(app, client=("127.0.0.1", 50000)) + try: + yield native_client + finally: + native_client.close() + + +@pytest.mark.parametrize("code,name", [("es", "Spanish"), ("de", "German"), + ("pt", "Portuguese"), ("nl", "Dutch")]) +def test_dictionary_code_matches_picker_name(client, code, name): + created = client.post("/pronunciation", json={"term": "GIF", "replacement": "respelling", + "language": code}) + assert created.status_code == 200 + for language in (code, name, f"{code}-XX", f"{code}_XX"): + result = client.post("/pronunciation/test", json={"text": "GIF", "language": language}) + assert result.status_code == 200 + assert result.json()["substituted"] == "respelling" + + +def test_spanish_dictionary_does_not_apply_to_estonian(client): + assert client.post("/pronunciation", json={"term": "GIF", "replacement": "Spanish", + "language": "es"}).status_code == 200 + for language in ("Estonian", "et", "et-EE"): + result = client.post("/pronunciation/test", json={"text": "GIF", "language": language}) + assert result.json()["substituted"] == "GIF" + assert result.json()["applied_terms"] == [] + + +@pytest.mark.parametrize("operation", ["create", "update", "import"]) +def test_saved_display_name_uses_same_canonical_scope(client, operation): + entry = {"term": "GIF", "replacement": "respelling", "language": "Spanish"} + if operation == "create": + result = client.post("/pronunciation", json=entry) + elif operation == "update": + initial = client.post("/pronunciation", json={**entry, "language": "*"}).json() + result = client.put(f"/pronunciation/{initial['id']}", json={"language": "Spanish"}) + else: + result = client.post("/pronunciation/import", json={"entries": [entry]}) + assert result.status_code == 200 + exported = client.get("/pronunciation/export").json()["entries"] + assert exported[0]["language"] == "es" + assert client.post("/pronunciation/test", json={"text": "GIF", "language": "es"}).json()["substituted"] == "respelling" + + +def test_three_letter_language_ids_are_not_collapsed(client): + # Abadi=kbt and Abron=abr come from the same bundled picker map as synthesis. + client.post("/pronunciation", json={"term": "GIF", "replacement": "Abadi", "language": "kbt"}) + assert client.get("/pronunciation/export").json()["entries"][0]["language"] == "kbt" + assert client.post("/pronunciation/test", json={"text": "GIF", "language": "Abadi"}).json()["substituted"] == "Abadi" + assert client.post("/pronunciation/test", json={"text": "GIF", "language": "Abron"}).json()["substituted"] == "GIF" + + +def test_global_auto_and_legacy_literals_keep_their_meaning(client): + for entry in [{"term": "GIF", "replacement": "global", "language": "*"}, + {"term": "GIF", "replacement": "legacy", "language": "po"}, + {"term": "GIF", "replacement": "Spanish", "language": "es"}]: + assert client.post("/pronunciation", json=entry).status_code == 200 + for language, expected in [(None, "global"), ("Auto", "global"), ("*", "global"), + ("French", "global"), ("Portuguese", "global"), + ("Polish", "global"), ("po", "legacy"), ("Spanish", "Spanish")]: + assert client.post("/pronunciation/test", json={"text": "GIF", "language": language}).json()["substituted"] == expected + + +def test_inert_entry_uses_same_canonical_scope(client): + assert client.post("/pronunciation", json={"term": "GIF", "replacement": "dʒɪf", + "type": "ipa", "language": "es"}).status_code == 200 + spanish = client.post("/pronunciation/test", json={"text": "GIF", "language": "Spanish"}).json() + estonian = client.post("/pronunciation/test", json={"text": "GIF", "language": "Estonian"}).json() + assert spanish["substituted"] == "GIF" + assert spanish["inert_entries"] == [{"term": "GIF", "type": "ipa"}] + assert estonian["inert_entries"] == [] + + +@pytest.mark.parametrize("name,code", [("Mandarin", "zh"), ("Arabic", "ar"), ("Tagalog", "tl")]) +def test_engine_language_aliases_match_dictionary_scopes(client, name, code): + assert client.post("/pronunciation", json={"term": "GIF", "replacement": "alias", + "language": name}).status_code == 200 + assert client.get("/pronunciation/export").json()["entries"][0]["language"] == code + assert client.post("/pronunciation/test", json={"text": "GIF", "language": code}).json()["substituted"] == "alias" + + +@pytest.mark.parametrize("operation", ["create", "update", "import"]) +def test_unknown_regional_scopes_remain_distinct_literals(client, operation): + for language, replacement in [("spa-MX", "literal Mexico"), ("spa-ES", "literal Spain")]: + entry = {"term": "GIF", "replacement": replacement, "language": language} + if operation == "create": + result = client.post("/pronunciation", json=entry) + elif operation == "update": + initial = client.post("/pronunciation", json={**entry, "language": "*"}).json() + result = client.put(f"/pronunciation/{initial['id']}", json={"language": language}) + else: + result = client.post("/pronunciation/import", json={"entries": [entry]}) + assert result.status_code == 200 + exported = client.get("/pronunciation/export").json()["entries"] + assert {entry["language"] for entry in exported} == {"spa-mx", "spa-es"} + for language, expected in [("spa-MX", "literal Mexico"), ("spa-ES", "literal Spain"), + ("spa", "GIF"), ("Spanish", "GIF"), ("es-MX", "GIF")]: + result = client.post("/pronunciation/test", json={"text": "GIF", "language": language}) + assert result.status_code == 200 + assert result.json()["substituted"] == expected + + +@pytest.mark.parametrize("operation", ["create", "update", "import"]) +@pytest.mark.parametrize("code", ["cmn", "zho"]) +def test_known_chinese_iso_scopes_match_script_tags(client, operation, code): + entry = {"term": "GIF", "replacement": "Chinese", "language": code} + if operation == "create": + result = client.post("/pronunciation", json=entry) + elif operation == "update": + initial = client.post( + "/pronunciation", json={**entry, "language": "*"}, + ).json() + result = client.put( + f"/pronunciation/{initial['id']}", json={"language": code}, + ) + else: + result = client.post("/pronunciation/import", json={"entries": [entry]}) + assert result.status_code == 200 + for language in (code, f"{code}-Hans", f"{code}-Hant", f"{code}_Hans"): + result = client.post( + "/pronunciation/test", json={"text": "GIF", "language": language}, + ) + assert result.status_code == 200 + assert result.json()["substituted"] == "Chinese" + unrelated = client.post( + "/pronunciation/test", json={"text": "GIF", "language": "spa-MX"}, + ) + assert unrelated.json()["substituted"] == "GIF" + + +@pytest.mark.parametrize("operation", ["create", "update", "import"]) +@pytest.mark.parametrize("code", ["cmn", "zho"]) +@pytest.mark.parametrize("reverse", [False, True]) +def test_chinese_script_scopes_keep_identity_and_precedence( + client, operation, code, reverse, +): + entries = [ + {"term": "GIF", "replacement": "global", "language": "*"}, + {"term": "gif", "replacement": "base", "language": code}, + {"term": "GIF", "replacement": "simplified", "language": f"{code}-Hans"}, + {"term": "gif", "replacement": "traditional", "language": f"{code}_Hant"}, + ] + for entry in reversed(entries) if reverse else entries: + if operation == "create": + result = client.post("/pronunciation", json=entry) + elif operation == "update": + initial = client.post( + "/pronunciation", json={**entry, "language": "*"}, + ).json() + result = client.put( + f"/pronunciation/{initial['id']}", + json={"language": entry["language"]}, + ) + else: + result = client.post("/pronunciation/import", json={"entries": [entry]}) + assert result.status_code == 200 + exported = client.get("/pronunciation/export").json()["entries"] + assert {entry["language"] for entry in exported} == { + "*", code, f"{code}-hans", f"{code}-hant", + } + # Export/import retains the explicit script identities, not just live rows. + for entry in client.get("/pronunciation").json(): + assert client.delete(f"/pronunciation/{entry['id']}").status_code == 200 + assert client.post( + "/pronunciation/import", json={"entries": exported}, + ).status_code == 200 + from services.pronunciation import apply_lexicon, load_dict_for_request + + for language, expected in [ + (code, "base"), (f"{code}-Hans", "simplified"), + (f"{code}_Hant", "traditional"), ("spa-MX", "global"), + ]: + result = client.post( + "/pronunciation/test", json={"text": "GIF gif", "language": language}, + ) + assert result.status_code == 200 + assert result.json()["substituted"] == f"{expected} {expected}" + assert apply_lexicon( + "GIF gif", load_dict_for_request(language), + ) == f"{expected} {expected}" + + +@pytest.mark.parametrize("code", ["cmn", "zho"]) +def test_unknown_chinese_suffixes_remain_distinct_literals(client, code): + for suffix in ("custom", "another"): + entry = {"term": "GIF", "replacement": suffix, "language": f"{code}-{suffix}"} + assert client.post("/pronunciation", json=entry).status_code == 200 + exported = client.get("/pronunciation/export").json()["entries"] + assert {entry["language"] for entry in exported} == { + f"{code}-custom", f"{code}-another", + } + for suffix in ("custom", "another"): + result = client.post( + "/pronunciation/test", json={"text": "GIF", "language": f"{code}-{suffix}"}, + ) + assert result.json()["substituted"] == suffix + for language in (code, f"{code}-Hans", f"{code}-Hant"): + result = client.post( + "/pronunciation/test", json={"text": "GIF", "language": language}, + ) + assert result.json()["substituted"] == "GIF" + + +@pytest.mark.parametrize("code", ["cmn", "zho"]) +def test_inert_chinese_rows_share_supported_scope_matching(client, code): + for term, language, enabled in [ + ("global", "*", True), ("base", code, True), + ("simplified", f"{code}-Hans", True), + ("traditional", f"{code}-Hant", True), + ("disabled", f"{code}-Hans", False), + ]: + entry = { + "term": term, "replacement": "dʒɪf", "type": "ipa", + "language": language, "enabled": enabled, + } + assert client.post("/pronunciation", json=entry).status_code == 200 + result = client.post( + "/pronunciation/test", json={"text": "GIF", "language": f"{code}-Hans"}, + ).json() + assert result["substituted"] == "GIF" + assert {entry["term"] for entry in result["inert_entries"]} == { + "global", "base", "simplified", + } diff --git a/tests/test_segmentation.py b/tests/test_segmentation.py index f8bf526a0..b2090ac84 100644 --- a/tests/test_segmentation.py +++ b/tests/test_segmentation.py @@ -94,6 +94,35 @@ def test_prefers_word_level_when_available(self): assert [w.text for w in words] == ["hello", "world", "foo"] assert words[0].start == 0.0 + @pytest.mark.parametrize("untimed_first", [False, True]) + def test_partial_word_timing_preserves_untimed_segment(self, untimed_first): + timed = {"text": "hello", "start": 0.0, "end": 1.0, + "words": [{"word": "hello", "start": 0.1, "end": 0.8}]} + untimed = {"text": "world again", "start": 2.0, "end": 4.0, "words": []} + segments = [untimed, timed] if untimed_first else [timed, untimed] + result = {"segments": segments, "chunks": [ + {"text": s["text"], "timestamp": (s["start"], s["end"])} for s in segments + ]} + words = _words(result) + assert [w.text for w in words] == ( + ["world", "again", "hello"] if untimed_first else ["hello", "world", "again"] + ) + hello = next(w for w in words if w.text == "hello") + assert (hello.start, hello.end) == (0.1, 0.8) + fallback = [w for w in words if w.text != "hello"] + assert [(w.start, w.end) for w in fallback] == [(2.0, 3.0), (3.0, 4.0)] + + def test_all_untimed_segments_keep_the_legacy_chunk_timing_fallback(self): + result = { + "segments": [{"text": "hello world", "start": 0.0, "end": None, "words": []}], + "chunks": [{"text": "hello", "timestamp": (0.0, 1.0)}, + {"text": "world", "timestamp": (1.0, 3.0)}], + } + words = _words(result) + assert [(w.text, w.start, w.end) for w in words] == [ + ("hello", 0.0, 1.0), ("world", 1.0, 3.0), + ] + def test_falls_back_to_chunks(self): result = _chunks(("one two three", 0.0, 3.0)) words = _words(result) diff --git a/tests/test_storage_wide_deadline.py b/tests/test_storage_wide_deadline.py new file mode 100644 index 000000000..9ca018f5c --- /dev/null +++ b/tests/test_storage_wide_deadline.py @@ -0,0 +1,112 @@ +"""A flat directory must not defeat the storage report's category budget.""" +import os + +import pytest + + +def test_directory_budget_is_checked_between_files(tmp_path, monkeypatch): + from services import storage_report + root = tmp_path / "wide" + root.mkdir() + for index in range(30): + (root / f"{index}.bin").write_bytes(b"1234") + now = [0.0] + visited = [] + original = os.lstat + + def slow_stat(path, *args, **kwargs): + if os.path.dirname(os.fspath(path)) == str(root): + visited.append(path) + now[0] += 1.0 + return original(path, *args, **kwargs) + + monkeypatch.setattr(storage_report.time, "monotonic", lambda: now[0]) + monkeypatch.setattr(storage_report.os, "lstat", slow_stat) + size, complete, unreadable = storage_report._dir_size(str(root), deadline=3.0) + assert len(visited) == 3 + assert size == 12 + assert complete is False + assert unreadable is None + + +def test_expired_budget_does_not_stat_a_remaining_file(tmp_path, monkeypatch): + from services import storage_report + file = tmp_path / "remaining.bin" + file.write_bytes(b"1234") + monkeypatch.setattr(storage_report.time, "monotonic", lambda: 3.0) + assert storage_report._dir_size(str(file), deadline=2.0) == (0, False, None) + + +def test_loose_data_files_report_partial_usage_after_deadline(tmp_path, monkeypatch): + from services import storage_report + data = tmp_path / "data" + data.mkdir() + for index in range(30): + (data / f"{index}.bin").write_bytes(b"1234") + now = [0.0] + visited = [] + original = os.scandir + + class Entry: + def __init__(self, entry): + self.entry = entry + self.name, self.path = entry.name, entry.path + + def is_dir(self, **kwargs): + return self.entry.is_dir(**kwargs) + + def stat(self, **kwargs): + visited.append(self.path) + now[0] += 1.0 + return self.entry.stat(**kwargs) + + class Scan: + def __enter__(self): + self.scan = original(data) + return (Entry(entry) for entry in self.scan) + + def __exit__(self, *args): + self.scan.close() + + monkeypatch.setattr(storage_report.os, "scandir", lambda path: Scan() if os.fspath(path) == str(data) else original(path)) + monkeypatch.setattr(storage_report.time, "monotonic", lambda: now[0]) + report = storage_report.build_report(data_dir=str(data), hf_cache_dir=str(tmp_path / "hf"), + engines_dir=str(tmp_path / "engines"), app_venv=None, temp_root=str(tmp_path / "temp"), category_timeout=3.0) + category = next(c for c in report["categories"] if c["id"] == "data") + other = next(c for c in category["children"] if c["id"] == "other") + assert len(visited) == 3 + assert category["bytes"] == other["bytes"] == 12 + assert category["complete"] is False + assert other["complete"] is False + assert any(w.get("category_id") == "data" and w.get("reason") == "timeout" for w in report["warnings"]) + + +@pytest.mark.parametrize("failure", ["enumeration", "classification"]) +def test_other_is_incomplete_when_managed_engine_ownership_is_unknown(tmp_path, monkeypatch, failure): + from services import storage_report + data = tmp_path / "data" + engines = data / "engines" + installed = engines / "installed" + (installed / ".venv").mkdir(parents=True) + (installed / "weights.bin").write_bytes(b"not measured") + original_scandir, original_stat = os.scandir, os.stat + + def scandir(path): + if failure == "enumeration" and os.fspath(path) == str(engines): + raise PermissionError("engine listing unavailable") + return original_scandir(path) + + def stat(path, *args, **kwargs): + if failure == "classification" and os.fspath(path) == str(installed / ".venv"): + raise PermissionError("venv ownership unavailable") + return original_stat(path, *args, **kwargs) + + monkeypatch.setattr(storage_report.os, "scandir", scandir) + monkeypatch.setattr(storage_report.os, "stat", stat) + report = storage_report.build_report(data_dir=str(data), engines_dir=str(engines), + hf_cache_dir=str(tmp_path / "hf"), temp_root=str(tmp_path / "temp")) + category = next(c for c in report["categories"] if c["id"] == "data") + other = next(c for c in category["children"] if c["id"] == "other") + assert category["complete"] is False + assert other["complete"] is False + assert any(w.get("category_id") == "data" and w.get("reason") == "permission" for w in report["warnings"]) diff --git a/tests/test_translator.py b/tests/test_translator.py index 35730a44d..3688f641d 100644 --- a/tests/test_translator.py +++ b/tests/test_translator.py @@ -400,3 +400,15 @@ def test_chat_keeps_a_closing_tag_quoted_from_the_source(monkeypatch): monkeypatch.setattr(tr, "_llm_model", lambda: "test-model") out = tr._chat(client, system="s", user="Use to close the block.") assert out == "Usa para cerrar el bloque." + + +@pytest.mark.parametrize("text", ["東京都新宿区都庁前駅", "会議は東京駅前で開催します。", "𠮷野家", "々"]) +def test_japanese_kanji_translation_passes_script_guard(text): + from services.translator import _looks_like_target_script, refine_output_ok + assert _looks_like_target_script(text, "ja") + assert refine_output_ok(text, text, "ja") == (True, None) + + +def test_japanese_script_guard_still_rejects_latin_only_output(): + from services.translator import _looks_like_target_script + assert not _looks_like_target_script("this is English", "ja") diff --git a/tests/test_video_context_temp_custody.py b/tests/test_video_context_temp_custody.py new file mode 100644 index 000000000..4b46a81bf --- /dev/null +++ b/tests/test_video_context_temp_custody.py @@ -0,0 +1,93 @@ +"""Visual-analysis workers own their frames through completion and cancellation.""" +import asyncio +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path +import subprocess +import tempfile +import threading + +import pytest + + +@pytest.fixture +def analysis(tmp_path, monkeypatch): + from services import ffmpeg_utils, video_context + + original = tempfile.mkdtemp + owned = [] + + def directory(suffix=None, prefix=None, dir=None): + path = original(suffix or "", prefix or "", str(tmp_path)) + owned.append(Path(path)) + return path + + monkeypatch.setattr(video_context.tempfile, "mkdtemp", directory) + monkeypatch.setattr(ffmpeg_utils, "find_ffmpeg", lambda: "fixture-ffmpeg") + + def extract(cmd, **kwargs): + Path(cmd[-1]).write_bytes(b"fixture frame") + return subprocess.CompletedProcess(cmd, 0, b"", b"") + + monkeypatch.setattr(subprocess, "run", extract) + monkeypatch.setattr(video_context, "_analyse_frame_basic", lambda path: { + "brightness": "normal", "mood": "calm", "complexity": "simple"}) + return video_context, owned, tmp_path + + +@pytest.mark.asyncio +async def test_success_removes_the_whole_owned_frame_directory(analysis): + module, owned, root = analysis + unrelated = root / "unrelated.jpg" + unrelated.write_bytes(b"keep") + context = await module.analyse_video("clip.mp4", [{"start": 0, "end": 2}]) + assert context.global_mood == "calm" + assert context.frame_analyses + assert owned and all(not path.exists() for path in owned) + assert unrelated.read_bytes() == b"keep" + + +@pytest.mark.asyncio +async def test_analysis_exception_removes_frames_and_directory(analysis, monkeypatch): + module, owned, _ = analysis + + def fail(*args): + raise RuntimeError("analysis failed") + + monkeypatch.setattr(module, "_build_segment_context", fail) + with pytest.raises(RuntimeError, match="analysis failed"): + await module.analyse_video("clip.mp4", [{"start": 0, "end": 2}]) + assert owned and all(not path.exists() for path in owned) + + +@pytest.mark.asyncio +@pytest.mark.parametrize("phase", ["extraction", "analysis"]) +async def test_cancelled_request_leaves_cleanup_with_the_native_worker(analysis, monkeypatch, phase): + module, owned, _ = analysis + entered, release = threading.Event(), threading.Event() + original = subprocess.run if phase == "extraction" else module._analyse_frame_basic + + def blocked(*args, **kwargs): + value = original(*args, **kwargs) + entered.set() + assert release.wait(5) + return value + + if phase == "extraction": + monkeypatch.setattr(subprocess, "run", blocked) + else: + monkeypatch.setattr(module, "_analyse_frame_basic", blocked) + with ThreadPoolExecutor(max_workers=1) as pool: + monkeypatch.setattr(module, "_analysis_pool", pool) + task = asyncio.create_task(module.analyse_video("clip.mp4", [{"start": 0, "end": 2}])) + try: + assert await asyncio.to_thread(entered.wait, 5) + task.cancel() + with pytest.raises(asyncio.CancelledError): + await task + # A cancelled waiter must not delete files while the worker owns them. + assert owned and any(list(path.glob("*.jpg")) for path in owned) + finally: + release.set() + # A one-worker queue makes this a completion fence, without sleeps. + await asyncio.get_running_loop().run_in_executor(pool, lambda: None) + assert all(not path.exists() for path in owned) From 4cd65a72d050604658257efe51fa7900f62560e9 Mon Sep 17 00:00:00 2001 From: debpalash <4178343+debpalash@users.noreply.github.com> Date: Sat, 3 Oct 2026 04:16:00 +0530 Subject: [PATCH 02/10] fix: close issue sweep review races and recovery gaps --- CHANGELOG.md | 8 ++-- backend/api/routers/archetypes.py | 6 ++- backend/api/routers/dub_generate.py | 18 ++++++-- backend/api/routers/profiles.py | 18 +++++--- backend/core/voice_reference_snapshots.py | 6 +++ backend/services/model_manager.py | 7 ++- docs/desktop-build.md | 2 +- docs/electron-dubbing.md | 2 + docs/electron-storage.md | 2 +- docs/electron-workflows.md | 2 +- docs/voice-design.md | 2 + electron/src/main/atomic-export.test.ts | 36 +++++++++++++++ electron/src/main/atomic-export.ts | 38 +++++++++------ electron/src/main/setup-progress.test.ts | 15 +++++- electron/src/main/setup-progress.ts | 10 ++-- .../src/renderer/src/lib/api/failure.test.ts | 16 +++++++ electron/src/renderer/src/lib/api/failure.ts | 12 +++-- tests/test_dub_complete_audio.py | 35 ++++++++++++++ tests/test_models_dir_setting.py | 2 +- tests/test_profile_design_save_decouple.py | 46 ++++++++++++++++++- tests/test_profile_relock_reference.py | 32 +++++++++++++ tests/test_profile_replace_audio.py | 21 +++++++++ tests/test_profile_unification.py | 2 +- 23 files changed, 293 insertions(+), 45 deletions(-) create mode 100644 electron/src/renderer/src/lib/api/failure.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index 96553aa9a..8b2a5a493 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -88,7 +88,7 @@ metadata and the backend fallback mirror it. ### Fixed -- Saving a voice design skips cold engine loading and downloads (#2583) — thanks @simoncheese! +- Saving a voice design skips cold engine loading and downloads, including when a warm engine unloads during the save (#2583) — thanks @simoncheese! - Electron streaming previews drain the final PCM chunk and crossfade recovered chunks only while audio overlaps (#2518) — thanks @rudycelekli! - Allow application-data relocation into existing empty folders without removing files added during copying (#2521) — thanks @rudycelekli! @@ -99,7 +99,7 @@ metadata and the backend fallback mirror it. - Preserve CRLF and CR metadata paragraphs in longform audio exports (#2528) — thanks @rudycelekli! - Cancelled dictation starts cannot replace the next session after a delayed connection ticket arrives (#2533) — thanks @rudycelekli! - Electron remote WebSockets retain the selected backend path prefix and path-bound tickets (#2537) — thanks @rudycelekli! -- Refresh longform audio after locking a voice to another take (#2535) — thanks @rudycelekli! +- Refresh longform audio after changing a voice reference and keep prior clips usable until active renders finish (#2535) — thanks @rudycelekli! - Retire cancelled longform streams and failed setup jobs while preserving resume checkpoints (#2536) — thanks @rudycelekli! - Buffered dictation utterances receive distinct saved-history IDs so deleting one preserves the others (#2538) — thanks @rudycelekli! - Compare voices warns when a generated preview omits speech, while keeping the surviving audio playable (#2548) — thanks @rudycelekli! @@ -107,8 +107,8 @@ metadata and the backend fallback mirror it. - Keep batch retries and deletion from racing over active job files (#2547) — thanks @rudycelekli! - Dictionary backups preserve duplicate-entry pronunciation order across preview, synthesis and restore (#2552) — thanks @rudycelekli! - Gallery trimming no longer stalls waiting for audio metadata before decoding (#2558) — thanks @rudycelekli! -- Native exports preserve existing files when a replacement write fails (#2560) — thanks @rudycelekli! -- Runtime setup keeps concurrent package download progress separate (#2562) — thanks @rudycelekli! +- Native exports preserve existing files when a replacement write fails and retry temporary file locks (#2560) — thanks @rudycelekli! +- Runtime setup keeps concurrent package download progress separate across equivalent package spellings (#2562) — thanks @rudycelekli! - Honor storage scan budgets in large flat directories (#2564) — thanks @rudycelekli! - Clean visual-context frame directories after worker completion (#2566) — thanks @rudycelekli! - Preserve concurrent partial MCP binding edits (#2568) — thanks @rudycelekli! diff --git a/backend/api/routers/archetypes.py b/backend/api/routers/archetypes.py index f5c6256e7..bb30d2129 100644 --- a/backend/api/routers/archetypes.py +++ b/backend/api/routers/archetypes.py @@ -306,7 +306,7 @@ def _is_unusable_audio(audio_tensor) -> bool: return flatness is not None and flatness < _DEGENERATE_FLATNESS -async def _render_archetype_wav(a: dict, out_path: Path) -> None: +async def _render_archetype_wav(a: dict, out_path: Path, *, allow_model_load: bool = True) -> None: """Render an archetype's sample script to ``out_path`` using the live engine. Reuses generation.py's inference primitives so there is exactly one TTS code @@ -321,7 +321,9 @@ async def _render_archetype_wav(a: dict, out_path: Path) -> None: _safe_torchaudio_save, ) - model = await get_model() + # Save-time samples are optional: an unload after the residency probe must + # not turn persistence into a cold load. Explicit previews retain loading. + model = await get_model() if allow_model_load else await get_model(allow_load=False) language = a["language"] if language in (None, "", "Auto"): language = None diff --git a/backend/api/routers/dub_generate.py b/backend/api/routers/dub_generate.py index 7a93442ea..0d120ce3b 100644 --- a/backend/api/routers/dub_generate.py +++ b/backend/api/routers/dub_generate.py @@ -1663,6 +1663,16 @@ def _render_batch() -> list[torch.Tensor]: yield f"data: {json.dumps({'type': 'assembling'})}\n\n" + # Resolve the final text/identity snapshot before publishing hashes or + # replacing the previous track. A job may have changed while synthesis + # was running; late identity conflicts belong in the task stream. + publication = {"segments": job.get("segments"), "seg_order": expected_order} + try: + _sync_job_segments(publication, req) + except HTTPException as exc: + yield f"data: {json.dumps({'type': 'error', 'error_code': exc.detail['code'], 'error': exc.detail['message']})}\n\n" + return + # ── Batch metadata phase ────────────────────────────────────── # Per-segment WAVs were written during the loop to keep RAM bounded. # Flush only lightweight fingerprints/quality metadata here. @@ -2041,9 +2051,11 @@ def _render_batch() -> list[torch.Tensor]: # mux step needs this to know whether to use the original video as-is # or stretch it per the plan. track_dur = total_samples / sr if total_samples > 0 else 0.0 - # Validate synchronized identity/text keys before installing the new - # track metadata; a collision must not publish a misleading snapshot. - _sync_job_segments(job, req) + # Publish the already validated snapshot with the completed track. + # Preserve other languages that may have been added during rendering. + job["segments"] = publication["segments"] + for key in ("segments_i18n", "segments_i18n_cue_sources"): + job.setdefault(key, {}).update(publication[key]) job["dubbed_tracks"][lang_code] = { "path": track_path, "language": req.language, diff --git a/backend/api/routers/profiles.py b/backend/api/routers/profiles.py index 25def9b26..61e068480 100644 --- a/backend/api/routers/profiles.py +++ b/backend/api/routers/profiles.py @@ -204,11 +204,11 @@ async def create_profile( "instruct": instruct, }, Path(audio_path), + allow_model_load=False, ) audio_filename = f"{profile_id}.wav" except Exception: # OOM / inference failure — defer the sample, clearing partials. - import logging logging.getLogger("omnivoice.profiles").info( "Design profile %s saved with sample pending — " "voice engine not ready; will render on preview", profile_id, @@ -616,8 +616,11 @@ async def replace_profile_audio( with contextlib.suppress(OSError): os.remove(leftover) raise - for column in ("ref_audio_path", "locked_audio_path", "consent_audio_path"): - _remove_voice_file(row[column], keep=new_filename) + with _voice_file_lock: + for column in ("ref_audio_path", "locked_audio_path", "consent_audio_path"): + path = _voices_path(row[column]) if row[column] else None + if path and not references_in_use([path]): + _remove_voice_file(row[column], keep=new_filename) event_bus.emit("profiles", {"action": "updated", "id": profile_id}) return _profile_record(updated) @@ -1027,9 +1030,12 @@ def _delete_profile(profile_id: str): if path: paths.append(path) if row: - # Relocking retains immutable versions for already-admitted renders. - # Explicit deletion reclaims only this profile's generated names. - versions = re.compile(re.escape(profile_id) + r"_locked(?:_[0-9a-f]{32})?\.wav") + # Relocking/replacement retains versions for admitted renders. + # Reclaim this profile's generated reference/locked/consent names, + # including a first upload with no filename extension. + versions = re.compile(re.escape(profile_id) + + r"(?:_locked(?:_[0-9a-f]{32})?\.wav|_consent\.[^./\\]+|" + r"(?:-[0-9a-f]{8})?(?:\.[^./\\]+)?)") if os.path.isdir(VOICES_DIR): for filename in os.listdir(VOICES_DIR): if versions.fullmatch(filename): diff --git a/backend/core/voice_reference_snapshots.py b/backend/core/voice_reference_snapshots.py index e6d3d2563..7648a1014 100644 --- a/backend/core/voice_reference_snapshots.py +++ b/backend/core/voice_reference_snapshots.py @@ -1,5 +1,6 @@ """Track reference paths while longform resolvers can still read them.""" import os +import gc import threading import weakref @@ -29,4 +30,9 @@ def retain(self, path): def references_in_use(paths): """Called under voice_file_lock before committing a profile deletion.""" targets = {os.path.realpath(path) for path in paths} + if not any(targets.intersection(snapshot.paths) for snapshot in _snapshots): + return False + # Finished failed workers can remain in traceback/frame cycles. Collect + # unreachable owners before refusing deletion; live workers stay rooted. + gc.collect() return any(targets.intersection(snapshot.paths) for snapshot in _snapshots) diff --git a/backend/services/model_manager.py b/backend/services/model_manager.py index 4ec0e091d..a63e5acbe 100644 --- a/backend/services/model_manager.py +++ b/backend/services/model_manager.py @@ -3104,7 +3104,7 @@ async def _load_model_with_timeout(): ) from exc -async def get_model(): +async def get_model(*, allow_load: bool = True): global model, _last_used _last_used = time.time() if model is not None: @@ -3125,6 +3125,11 @@ async def get_model(): await asyncio.get_running_loop().run_in_executor(None, make_room_before_generate) return model + # Opportunistic profile samples must never load weights, including when + # ASR or idle cleanup evicted them after the caller's residency check. + if not allow_load: + raise RuntimeError("VoiceStudio model is not loaded") + if running_on_gpu_pool(): # Same reasoning as _heal_tts_placement below, applied to the COLD # path it never covered (#1417). We are on a pool worker, reached from diff --git a/docs/desktop-build.md b/docs/desktop-build.md index e28c50f64..61f8f171b 100644 --- a/docs/desktop-build.md +++ b/docs/desktop-build.md @@ -217,4 +217,4 @@ One session per platform. Ordered by payoff: and smoke-tested. Phase F skeleton committed (CI workflow, primary target only — non-arm64 rows parked). Phases A, B, D, E pending. -Runtime download byte progress matches complete package identifiers, so concurrent downloads such as `torch` and `torchvision` retain separate received bytes and totals. Unknown package progress does not change another package’s planned size. +Runtime download byte progress matches complete package identifiers, so concurrent downloads such as `torch` and `torchvision` retain separate received bytes and totals. Package names use the same case-insensitive, hyphen/underscore/dot normalization for planned, active and finished downloads. Unknown package progress does not change another package’s planned size. diff --git a/docs/electron-dubbing.md b/docs/electron-dubbing.md index c37dcdd64..a661854c6 100644 --- a/docs/electron-dubbing.md +++ b/docs/electron-dubbing.md @@ -409,3 +409,5 @@ Advanced QC keeps the selected track's text and timing fixed while recognition runs. If the track, transcript, timing, or audio changes during that pass, QC asks you to run it again instead of publishing stale scores. Deleting the job during QC also discards the result and keeps it out of history. + +Generation revalidates its segment text and identity snapshot after synthesis, before publishing fingerprints or replacing the previous track. An identity collision caused by a concurrent edit is reported through the task stream; the previous track and published metadata remain available. diff --git a/docs/electron-storage.md b/docs/electron-storage.md index b8b5d2c52..235f6c1cf 100644 --- a/docs/electron-storage.md +++ b/docs/electron-storage.md @@ -26,4 +26,4 @@ Interrupted sidecar installs without an environment remain included in applicati An unreadable engine directory or entry produces an incomplete report and a warning; unavailable bytes are not presented as a complete empty footprint. -Re-locking keeps prior immutable locked-reference clips for renders that already captured their filenames. These clips remain in the voices folder for the lifetime of the profile; explicit profile deletion reclaims its generated versions after committing the record deletion, while preserving versions still referenced by another profile. Deletion returns a conflict while a longform render still holds a cached reference that would be removed; the profile record and files remain intact, and deletion can be retried once all those render workers finish. Unlocking still succeeds while a render holds the current locked clip, but retains that immutable clip until explicit profile deletion can safely reclaim it. Unlocking also preserves clips referenced by any profile, including the unlocked profile’s other audio fields; an unused, unshared current clip is removed after the unlock commits. Repeated locks can therefore use additional local storage until the profile is deleted. +Re-locking keeps prior immutable locked-reference clips for renders that already captured their filenames. These clips remain in the voices folder for the lifetime of the profile; explicit profile deletion reclaims its generated versions after committing the record deletion, while preserving versions still referenced by another profile. Deletion returns a conflict while a longform render still holds a cached reference that would be removed; the profile record and files remain intact, and deletion can be retried once all those render workers finish. Unlocking still succeeds while a render holds the current locked clip, but retains that immutable clip until explicit profile deletion can safely reclaim it. Unlocking also preserves clips referenced by any profile, including the unlocked profile’s other audio fields; an unused, unshared current clip is removed after the unlock commits. Replacing a clone sample likewise retains any previous reference still captured by a render; profile deletion reclaims these retained uploads too. Unreachable traceback cycles from finished failed renders do not keep a profile busy. Repeated locks or replacements during renders can therefore use additional local storage until the profile is deleted. diff --git a/docs/electron-workflows.md b/docs/electron-workflows.md index 22c8d0a15..7579a37e7 100644 --- a/docs/electron-workflows.md +++ b/docs/electron-workflows.md @@ -50,4 +50,4 @@ A branch that goes directly from a Condition to End keeps the incoming text as its exportable result. Conditions have one phrase editor; the generic Instructions field is hidden for those steps. -Native saves stage the complete replacement in the selected destination directory before replacing an existing export. A write or replacement failure leaves the previous file intact. Working symlink targets are resolved before a non-truncating write-authorization open; permission bits are read from that opened file descriptor. A changed destination inode or newly appeared file observed before replacement is refused. This identity check is not an atomic compare-and-swap and does not protect against all concurrent changes in a hostile directory. Existing POSIX permission bits and working symbolic links are preserved; a dangling symbolic link is left unchanged and reports a filesystem error. Atomic replacement requires write permission on the parent directory, even when the existing file is writable. It replaces the original file inode: existing ACLs, ownership, and relationships to other hard links are not preserved. This does not provide a power-loss recovery guarantee. +Native saves stage the complete replacement in the selected destination directory before replacing an existing export. A write or replacement failure leaves the previous file intact. Windows sharing violations receive three bounded retries, with destination identity checked again each time. If another application keeps the file locked, close it and retry the export or choose another destination; the previous export is retained. Working symlink targets are resolved before a non-truncating write-authorization open; permission bits are read from that opened file descriptor. A changed destination inode or newly appeared file observed before replacement is refused. This identity check is not an atomic compare-and-swap and does not protect against all concurrent changes in a hostile directory. Existing POSIX permission bits and working symbolic links are preserved; a dangling symbolic link is left unchanged and reports a filesystem error. Atomic replacement requires write permission on the parent directory, even when the existing file is writable. It replaces the original file inode: existing ACLs, ownership, and relationships to other hard links are not preserved. This does not provide a power-loss recovery guarantee. diff --git a/docs/voice-design.md b/docs/voice-design.md index eb2a11361..43c2fb646 100644 --- a/docs/voice-design.md +++ b/docs/voice-design.md @@ -6,6 +6,8 @@ generates a matching voice on the fly. Saving a design does not load or download a voice engine. If the engine is already loaded, Save also renders its identity sample. Otherwise the design is saved with its attributes and the sample is generated when you preview it. +If the engine unloads before rendering starts, Save keeps the sample pending +instead of loading the engine again. Until that preview exists, synthesis uses the saved attributes directly. ## Quick Example diff --git a/electron/src/main/atomic-export.test.ts b/electron/src/main/atomic-export.test.ts index 0b899fbb1..5e5985dbc 100644 --- a/electron/src/main/atomic-export.test.ts +++ b/electron/src/main/atomic-export.test.ts @@ -8,6 +8,7 @@ import { afterEach, beforeEach, expect, it, vi } from 'vitest'; const controls = vi.hoisted(() => ({ handlers: new Map Promise>(), destination: '', failWrite: false, failRename: false, swapAfterProbeOpen: false, swapWithSymlink: false, appearWhileStaging: false, + sharingFailures: 0, renameAttempts: 0, sharingCode: 'EBUSY', swapAfterSharingFailure: false, })); vi.mock('electron', () => ({ app: {}, @@ -61,6 +62,15 @@ vi.mock('node:fs/promises', async () => { return file; }, rename: async (...args: Parameters) => { + controls.renameAttempts += 1; + if (controls.sharingFailures-- > 0) { + if (controls.swapAfterSharingFailure) { + controls.swapAfterSharingFailure = false; + await real.rename(controls.destination, join(directory, 'authorized-original.wav')); + await real.writeFile(controls.destination, 'concurrent replacement'); + } + throw Object.assign(new Error('export is locked'), { code: controls.sharingCode }); + } if (controls.failRename) throw Object.assign(new Error('injected rename failure'), { code: 'EACCES' }); return real.rename(...args); }, @@ -73,6 +83,7 @@ const owner = { webContents: { mainFrame: frame } }; const event = { sender: owner.webContents, senderFrame: frame }; let directory: string; beforeEach(async () => { + controls.sharingFailures = 0; controls.renameAttempts = 0; controls.sharingCode = 'EBUSY'; controls.swapAfterSharingFailure = false; controls.failWrite = false; controls.failRename = false; controls.swapAfterProbeOpen = false; controls.swapWithSymlink = false; controls.appearWhileStaging = false; controls.handlers.clear(); directory = await mkdtemp(join(tmpdir(), 'voicestudio-export-')); controls.destination = join(directory, 'saved.wav'); @@ -99,6 +110,31 @@ it.each(requests)('preserves the old export after a failed replacement in %s', a expect(await readdir(directory)).toEqual(['saved.wav']); }); +it.each(['EBUSY', 'EPERM'])('retries transient %s without exposing partial output', async (code) => { + controls.sharingCode = code; + controls.sharingFailures = 2; + await expect(controls.handlers.get(CHANNELS.filesSaveData)!(event, requests[0][1])).resolves.toMatchObject({ canceled: false }); + expect(controls.renameAttempts).toBe(3); + expect([...await readFile(controls.destination)]).toEqual([9, 8, 7, 6]); + expect(await readdir(directory)).toEqual(['saved.wav']); +}); + +it('preserves the old export after bounded retries of a persistent sharing lock', async () => { + controls.sharingFailures = 100; + await expect(controls.handlers.get(CHANNELS.filesSaveData)!(event, requests[0][1])).rejects.toMatchObject({ code: 'EBUSY' }); + expect(controls.renameAttempts).toBe(4); + expect(await readFile(controls.destination, 'utf8')).toBe('previous complete export'); + expect(await readdir(directory)).toEqual(['saved.wav']); +}); + +it('revalidates destination identity before retrying a sharing violation', async () => { + controls.sharingFailures = 1; + controls.swapAfterSharingFailure = true; + await expect(controls.handlers.get(CHANNELS.filesSaveData)!(event, requests[0][1])).rejects.toMatchObject({ code: 'ESTALE' }); + expect(await readFile(controls.destination, 'utf8')).toBe('concurrent replacement'); + expect(await readFile(join(directory, 'authorized-original.wav'), 'utf8')).toBe('previous complete export'); +}); + it.each(requests)('replaces the complete export and preserves its permissions in %s', async (channel, request) => { await expect(controls.handlers.get(channel)!(event, request)).resolves.toEqual({ canceled: false, path: controls.destination }); expect([...await readFile(controls.destination)]).toEqual([9, 8, 7, 6]); diff --git a/electron/src/main/atomic-export.ts b/electron/src/main/atomic-export.ts index bfc0e309f..d3e84dd1c 100644 --- a/electron/src/main/atomic-export.ts +++ b/electron/src/main/atomic-export.ts @@ -2,6 +2,7 @@ import { randomUUID } from 'node:crypto'; import { constants, type BigIntStats } from 'node:fs'; import { lstat, open, realpath, rename, rm } from 'node:fs/promises'; import { dirname, join } from 'node:path'; +import { setTimeout as delay } from 'node:timers/promises'; /** Keep a selected existing export intact until all replacement bytes are written. */ export async function writeExportAtomically(path: string, data: Uint8Array): Promise { @@ -47,22 +48,31 @@ export async function writeExportAtomically(path: string, data: Uint8Array): Pro } finally { await file.close(); } - const current = await lstat(destination, { bigint: true }).catch((error: NodeJS.ErrnoException) => { - if (error.code === 'ENOENT') return null; - throw error; - }); - if (authorized) { - // Refuse observed substitutions instead of copying the selected export - // over another inode or following a symlink we did not authorize. - if (!current || current.isSymbolicLink() || current.dev !== authorized.dev || current.ino !== authorized.ino) { - throw Object.assign(new Error('ESTALE'), { code: 'ESTALE' }); + for (let attempt = 0; ; attempt += 1) { + const current = await lstat(destination, { bigint: true }).catch((error: NodeJS.ErrnoException) => { + if (error.code === 'ENOENT') return null; + throw error; + }); + if (authorized) { + // Recheck after every wait: another exporter may replace the destination + // while a transient Windows sharing violation prevents this rename. + if (!current || current.isSymbolicLink() || current.dev !== authorized.dev || current.ino !== authorized.ino) { + throw Object.assign(new Error('ESTALE'), { code: 'ESTALE' }); + } + } else if (current) { + throw Object.assign(new Error('EEXIST'), { code: 'EEXIST' }); + } + // The identity check is not an atomic compare-and-swap with rename. A + // concurrently modified hostile directory needs native OS protection. + try { + await rename(temporary, destination); + break; + } catch (error) { + const code = (error as NodeJS.ErrnoException).code; + if (attempt >= 3 || (code !== 'EPERM' && code !== 'EBUSY')) throw error; + await delay(50 * (attempt + 1)); } - } else if (current) { - throw Object.assign(new Error('EEXIST'), { code: 'EEXIST' }); } - // The identity check is not an atomic compare-and-swap with rename. A - // concurrently modified hostile directory needs native OS protection. - await rename(temporary, destination); } finally { await rm(temporary, { force: true }); } diff --git a/electron/src/main/setup-progress.test.ts b/electron/src/main/setup-progress.test.ts index 689dca96f..e15147bfa 100644 --- a/electron/src/main/setup-progress.test.ts +++ b/electron/src/main/setup-progress.test.ts @@ -74,6 +74,19 @@ describe('SetupProgressTracker', () => { describe('concurrent package byte updates', () => { + it('joins equivalent package spellings across planning, progress and completion', () => { + const tracker = new SetupProgressTracker(); + tracker.ingest('Downloading Pydantic_Core (20 MiB)', 0); + expect(tracker.ingest('pydantic.core 2 MiB / 20 MiB', 1000)).toMatchObject({ + totalBytes: 20 * 1024 ** 2, + downloadedBytes: 2 * 1024 ** 2, + }); + expect(tracker.ingest('Downloaded PYDANTIC-core', 2000)).toMatchObject({ + completedDownloads: 1, + downloadsComplete: true, + downloadedBytes: 20 * 1024 ** 2, + }); + }); it.each([ ['torch', 'torchvision'], ['torchvision', 'torch'], @@ -98,7 +111,7 @@ describe('concurrent package byte updates', () => { tracker.ingest(`Downloading ${first} (10 MiB)`, 0); tracker.ingest(`Downloading ${second} (20 MiB)`, 0); expect(tracker.ingest(`${second} 2 MiB / 20 MiB`, 1000)).toMatchObject({ - activePackage: second, + activePackage: second.replace(/[-_.]+/g, '-'), totalBytes: 30 * 1024 ** 2, downloadedBytes: 2 * 1024 ** 2, }); diff --git a/electron/src/main/setup-progress.ts b/electron/src/main/setup-progress.ts index 658917481..d76a17b71 100644 --- a/electron/src/main/setup-progress.ts +++ b/electron/src/main/setup-progress.ts @@ -18,6 +18,10 @@ export interface SetupProgress { const ANSI_ESCAPE = /\x1b(?:\[[0-?]*[ -/]*[@-~]|\][^\x07]*(?:\x07|\x1b\\))/g; const SIZE = '(\\d+(?:\\.\\d+)?)\\s*(B|KB|KiB|MB|MiB|GB|GiB)'; +function packageKey(name: string): string { + return name.trim().toLowerCase().replace(/[-_.]+/g, '-'); +} + export function cleanProcessLine(line: string): string { return line.replace(ANSI_ESCAPE, '').trim(); } @@ -73,7 +77,7 @@ export class SetupProgressTracker { const starting = line.match(new RegExp(`^Downloading\\s+(.+?)\\s+\\(${SIZE}\\)`, 'i')); if (starting) { - const name = starting[1].trim(); + const name = packageKey(starting[1]); this.planned.set(name, parseByteSize(starting[2], starting[3])); this.completed.delete(name); this.progress.downloadsComplete = false; @@ -83,7 +87,7 @@ export class SetupProgressTracker { const finished = line.match(/^Downloaded\s+(.+?)\s*$/i); if (finished) { - const name = finished[1].trim(); + const name = packageKey(finished[1]); this.completed.add(name); const size = this.planned.get(name); if (size !== undefined) this.received.set(name, size); @@ -98,7 +102,7 @@ export class SetupProgressTracker { const total = parseByteSize(bytePair[3], bytePair[4]); // Match complete package identifiers, so torchvision cannot update torch. const identifiers = line.slice(0, bytePair.index).match(/[a-zA-Z0-9][a-zA-Z0-9_.-]*/g) ?? []; - const name = identifiers.reverse().find((candidate) => this.planned.has(candidate)); + const name = identifiers.reverse().map(packageKey).find((candidate) => this.planned.has(candidate)); if (name) { this.planned.set(name, total); this.received.set(name, Math.min(received, total)); diff --git a/electron/src/renderer/src/lib/api/failure.test.ts b/electron/src/renderer/src/lib/api/failure.test.ts new file mode 100644 index 000000000..1bb774bfd --- /dev/null +++ b/electron/src/renderer/src/lib/api/failure.test.ts @@ -0,0 +1,16 @@ +import i18next from 'i18next'; +import { afterEach, expect, it, vi } from 'vitest'; +import { publicFailureFromEvent } from './failure'; + +afterEach(() => vi.restoreAllMocks()); + +it('localizes a segment identity conflict delivered after generation starts', () => { + const translate = vi.spyOn(i18next, 't').mockReturnValue('Localized identity conflict'); + const failure = publicFailureFromEvent({ + type: 'error', + error_code: 'dub_segment_identity_conflict', + error: 'Server fallback', + }, 'Task failed'); + expect(failure.reason).toBe('Localized identity conflict'); + expect(translate).toHaveBeenCalledWith('dub.qc_identity_missing'); +}); diff --git a/electron/src/renderer/src/lib/api/failure.ts b/electron/src/renderer/src/lib/api/failure.ts index 5ceda8dae..d8a54a31a 100644 --- a/electron/src/renderer/src/lib/api/failure.ts +++ b/electron/src/renderer/src/lib/api/failure.ts @@ -19,11 +19,13 @@ export function publicFailureFromEvent( const localized = generationFailureMessage(event, i18next.t); return { reason: - (event.error_code === 'dub_speech_missing' - ? i18next.t('dubIntegrity.missingSpeech') - : event.error_code === 'dub_timing_overflow' - ? i18next.t('dubIntegrity.timingOverflow') - : undefined) || + (event.error_code === 'dub_segment_identity_conflict' + ? i18next.t('dub.qc_identity_missing') + : event.error_code === 'dub_speech_missing' + ? i18next.t('dubIntegrity.missingSpeech') + : event.error_code === 'dub_timing_overflow' + ? i18next.t('dubIntegrity.timingOverflow') + : undefined) || localized || text(event.reason) || text(event.detail) || diff --git a/tests/test_dub_complete_audio.py b/tests/test_dub_complete_audio.py index 8bc35598a..ed9d44d1d 100644 --- a/tests/test_dub_complete_audio.py +++ b/tests/test_dub_complete_audio.py @@ -208,3 +208,38 @@ def test_timing_trims_edge_silence_but_keeps_internal_pauses(): result = trim_speech_padding(wav, 1000) assert result.shape[-1] == 700 assert torch.equal(result[..., 250:450], torch.zeros(1,200)) + + +def test_identity_change_during_render_preserves_previous_publication(render_dub, monkeypatch): + from api.routers import dub_generate as dg + + previous = render_dub.path / 'dubbed_en.wav' + previous.write_bytes(b'previous complete track') + render_dub.job.update({ + 'segments': [{'id': 'a', 'start': 0, 'end': 1, 'text': 'saved first'}, + {'id': 'b', 'start': 1, 'end': 2, 'text': 'saved second'}], + 'segments_i18n': {'en': {'a': 'saved first', 'b': 'saved second'}}, + 'seg_hashes': {'a': 'saved hash'}, + 'seg_hashes_by_lang': {'en': {'a': 'saved hash'}}, + 'dubbed_tracks': {'en': {'path': str(previous)}}, + }) + saved = copy.deepcopy(render_dub.job) + persisted = render_dub.path / 'job.json' + persisted.write_text(json.dumps(saved)) + monkeypatch.setattr(dg, '_save_job', lambda _, job: persisted.write_text(json.dumps(job))) + + def change_identity(): + render_dub.job['segments'][0]['id'] = 'duplicate' + render_dub.job['segments'][1]['id'] = 'duplicate' + return torch.ones(1, 24000) * .1 + render_dub.output[0] = change_identity + events = render_dub.run(segments=[dict(start=0, end=1, text='replacement first'), + dict(start=1, end=2, text='replacement second')], + segment_ids=[]) + assert any(e['type'] == 'error' and e.get('error_code') == 'dub_segment_identity_conflict' + for e in events) + assert not any(e['type'] == 'done' for e in events) + assert previous.read_bytes() == b'previous complete track' + assert json.loads(persisted.read_text()) == saved + for key in ('seg_hashes', 'seg_hashes_by_lang', 'segments_i18n', 'dubbed_tracks'): + assert render_dub.job[key] == saved[key] diff --git a/tests/test_models_dir_setting.py b/tests/test_models_dir_setting.py index 4b8ccb71c..89fc64332 100644 --- a/tests/test_models_dir_setting.py +++ b/tests/test_models_dir_setting.py @@ -186,7 +186,7 @@ def test_models_directory_comment_characters_survive_route_and_startup(env, tmp_ assert user_env.load_into_environ() is True assert os.environ["OMNIVOICE_CACHE_DIR"] == target assert client.get("/api/settings/storage/models-dir").json()["configured"] == target - assert not os.path.exists(target.split(" #", 1)[0]) + assert os.path.isdir(target) def test_windows_drive_paths_with_spaces_round_trip_verbatim(env): diff --git a/tests/test_profile_design_save_decouple.py b/tests/test_profile_design_save_decouple.py index 54241c233..cd7873bce 100644 --- a/tests/test_profile_design_save_decouple.py +++ b/tests/test_profile_design_save_decouple.py @@ -125,7 +125,8 @@ def test_warm_design_save_preserves_identity_sample(iso, monkeypatch): monkeypatch.setattr(model_manager, "get_model_status", lambda: {"loaded": True}) renders = [] - async def render(recipe, out_path): + async def render(recipe, out_path, *, allow_model_load): + assert allow_model_load is False renders.append(recipe) out_path.parent.mkdir(parents=True, exist_ok=True) out_path.write_bytes(b"identity sample") @@ -142,6 +143,49 @@ async def render(recipe, out_path): assert (Path(cfg.VOICES_DIR) / row["ref_audio_path"]).read_bytes() == b"identity sample" +@pytest.mark.parametrize("on_gpu_worker", [False, True]) +def test_design_save_cannot_cold_load_after_residency_check(iso, monkeypatch, on_gpu_worker): + """ASR/idle unload between admission and rendering must keep save local.""" + _, db, prof = iso + from api.routers import archetypes, generation + from services import model_manager + monkeypatch.setattr(model_manager, "model", object()) + monkeypatch.setattr(model_manager, "get_model_status", lambda: {"loaded": True}) + monkeypatch.setattr(model_manager, "running_on_gpu_pool", lambda: on_gpu_worker) + monkeypatch.setattr(generation, "get_model", model_manager.get_model) + real_render = archetypes._render_archetype_wav + loads = [] + + async def blocked_load(): + loads.append(True) + await asyncio.Event().wait() + + def inline_load(): + loads.append(True) + raise RuntimeError("Unexpected cold load on GPU worker") + + async def unload_then_render(*args, **kwargs): + model_manager.model = None + return await real_render(*args, **kwargs) + + monkeypatch.setattr(model_manager, "_load_model_with_timeout", blocked_load) + monkeypatch.setattr(model_manager, "_load_model_exclusive", inline_load) + monkeypatch.setattr(archetypes, "_render_archetype_wav", unload_then_render) + + async def save(): + return await asyncio.wait_for(prof.create_profile( + name="Unloaded during save", ref_audio=None, ref_text="", instruct="female", + language="English", seed=None, personality="", kind="design", + vd_states=json.dumps(_VD), image=None, + ), timeout=1) + + result = asyncio.run(save()) + assert not loads + with db.db_conn() as conn: + row = conn.execute("SELECT ref_audio_path FROM voice_profiles WHERE id=?", (result["id"],)).fetchone() + assert row is not None and not row["ref_audio_path"] + + def test_all_auto_design_is_saveable(iso, monkeypatch): """An all-Auto design (empty instruct) saves; it isn't gated on instruct.""" _, db, prof = iso diff --git a/tests/test_profile_relock_reference.py b/tests/test_profile_relock_reference.py index 7d66da8b3..99bf8e3ee 100644 --- a/tests/test_profile_relock_reference.py +++ b/tests/test_profile_relock_reference.py @@ -362,3 +362,35 @@ def test_failed_unlock_preserves_locked_reference_and_row(profile): with db.db_conn() as conn: row = conn.execute("SELECT is_locked,locked_audio_path FROM voice_profiles WHERE id='voice'").fetchone() assert tuple(row) == (1, locked) + + +def test_finished_failed_render_does_not_keep_profile_busy_until_automatic_gc(profile): + import gc + from api.routers.audiobook import _build_synth + + client, _db, voices = profile + locked = client.post('/profiles/voice/lock', data={'history_id': 'first'}).json()['locked_audio_path'] + + def failed_chapter(): + # Like _render_longform_sse's last_chapter_exc: the exception owns its + # frame, which owns both the exception and the now-finished resolver. + running = _build_synth(default_voice='voice') + running['resolve']('voice') + last_chapter_exc = None + try: + raise RuntimeError('completed chapter failure') + except RuntimeError as error: + last_chapter_exc = error + assert last_chapter_exc is not None + + enabled = gc.isenabled() + gc.disable() + try: + failed_chapter() + response = client.delete('/profiles/voice') + assert response.status_code == 200, response.text + assert not Path(voices, locked).exists() + finally: + if enabled: + gc.enable() + gc.collect() diff --git a/tests/test_profile_replace_audio.py b/tests/test_profile_replace_audio.py index 4845e4246..f97b0737c 100644 --- a/tests/test_profile_replace_audio.py +++ b/tests/test_profile_replace_audio.py @@ -3,6 +3,7 @@ import io import os import wave +from pathlib import Path import pytest from fastapi import FastAPI @@ -84,6 +85,26 @@ def test_replace_writes_new_versioned_file_and_removes_old(env): assert client.get("/profiles").json()[0]["audio_url"] == updated["audio_url"] +@pytest.mark.parametrize("versioned", [False, True]) +def test_replace_keeps_reference_owned_by_render_until_profile_deletion(env, monkeypatch, versioned): + from core import config + from api.routers.audiobook import _build_synth + client, _profiles, _db, voices, created, _transcribed, _events = env + monkeypatch.setattr(config, 'VOICES_DIR', str(voices)) + if versioned: + created = replace(client, created['id'], text='first replacement').json() + running = _build_synth(default_voice=created['id']) + path = running['resolve'](created['id'])['ref_audio'] + original = Path(path).read_bytes() + response = replace(client, created['id'], text='new reference') + assert response.status_code == 200, response.text + assert Path(path).read_bytes() == original + assert client.delete(f"/profiles/{created['id']}").status_code == 409 + del running + assert client.delete(f"/profiles/{created['id']}").status_code == 200 + assert not os.path.exists(path) + + def test_blank_transcript_is_auto_transcribed_not_carried_over(env): client, _profiles, _db, voices, created, transcribed, _events = env updated = replace(client, created["id"], name="take.FLAC", text=" ").json() diff --git a/tests/test_profile_unification.py b/tests/test_profile_unification.py index 8f43a07e4..497e05d1d 100644 --- a/tests/test_profile_unification.py +++ b/tests/test_profile_unification.py @@ -47,7 +47,7 @@ def app_client(tmp_path_factory): @pytest.fixture() def fake_render(monkeypatch): """Stub the design sample renderer — CI has no TTS engine.""" - async def _fake(a, out_path): + async def _fake(a, out_path, **kwargs): out_path.parent.mkdir(parents=True, exist_ok=True) out_path.write_bytes(_FAKE_AUDIO) From 103a8e53c3512ffb56a09270657b2386ee35edf8 Mon Sep 17 00:00:00 2001 From: debpalash <4178343+debpalash@users.noreply.github.com> Date: Sat, 3 Oct 2026 04:39:42 +0530 Subject: [PATCH 03/10] fix: stage dub publication and preserve concurrent subtitle edits --- CHANGELOG.md | 2 +- backend/api/routers/dub_generate.py | 316 +++++++++++------- backend/core/tasks.py | 37 +- backend/schemas/requests.py | 2 +- docs/electron-dubbing.md | 2 +- .../src/renderer/src/i18n/locales/ar.json | 1 + .../src/renderer/src/i18n/locales/de.json | 1 + .../src/renderer/src/i18n/locales/en.json | 1 + .../src/renderer/src/i18n/locales/es.json | 1 + .../src/renderer/src/i18n/locales/fr.json | 1 + .../src/renderer/src/i18n/locales/hi.json | 1 + .../src/renderer/src/i18n/locales/id.json | 1 + .../src/renderer/src/i18n/locales/it.json | 1 + .../src/renderer/src/i18n/locales/ja.json | 1 + .../src/renderer/src/i18n/locales/ko.json | 1 + .../src/renderer/src/i18n/locales/nl.json | 1 + .../src/renderer/src/i18n/locales/pl.json | 1 + .../src/renderer/src/i18n/locales/pt.json | 1 + .../src/renderer/src/i18n/locales/ru.json | 1 + .../src/renderer/src/i18n/locales/sv.json | 1 + .../src/renderer/src/i18n/locales/th.json | 1 + .../src/renderer/src/i18n/locales/tr.json | 1 + .../src/renderer/src/i18n/locales/uk.json | 1 + .../src/renderer/src/i18n/locales/vi.json | 1 + .../src/renderer/src/i18n/locales/zh-CN.json | 1 + .../src/renderer/src/i18n/locales/zh-TW.json | 1 + .../src/renderer/src/lib/api/failure.test.ts | 9 + electron/src/renderer/src/lib/api/failure.ts | 16 +- electron/src/shared/i18n/locales/ar.json | 1 + electron/src/shared/i18n/locales/de.json | 1 + electron/src/shared/i18n/locales/en.json | 1 + electron/src/shared/i18n/locales/es.json | 1 + electron/src/shared/i18n/locales/fr.json | 1 + electron/src/shared/i18n/locales/hi.json | 1 + electron/src/shared/i18n/locales/id.json | 1 + electron/src/shared/i18n/locales/it.json | 1 + electron/src/shared/i18n/locales/ja.json | 1 + electron/src/shared/i18n/locales/ko.json | 1 + electron/src/shared/i18n/locales/nl.json | 1 + electron/src/shared/i18n/locales/pl.json | 1 + electron/src/shared/i18n/locales/pt.json | 1 + electron/src/shared/i18n/locales/ru.json | 1 + electron/src/shared/i18n/locales/sv.json | 1 + electron/src/shared/i18n/locales/th.json | 1 + electron/src/shared/i18n/locales/tr.json | 1 + electron/src/shared/i18n/locales/uk.json | 1 + electron/src/shared/i18n/locales/vi.json | 1 + electron/src/shared/i18n/locales/zh-CN.json | 1 + electron/src/shared/i18n/locales/zh-TW.json | 1 + tests/test_dub_complete_audio.py | 115 +++++++ tests/test_dub_subtitles_309.py | 2 +- tests/test_router_smoke.py | 2 +- tests/test_task_stream_failure.py | 31 ++ 53 files changed, 422 insertions(+), 154 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 8b2a5a493..551d2e745 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -114,7 +114,7 @@ metadata and the backend fallback mirror it. - Preserve concurrent partial MCP binding edits (#2568) — thanks @rudycelekli! - Preserve CPU forced-alignment fallback when loading the MPS aligner fails (#2570) — thanks @rudycelekli! - Keep untimed transcript segments alongside precise word-timed speech (#2572) — thanks @rudycelekli! -- Score dub quality against the selected track’s saved language text, reject stale checks, and preserve audio when segment identities conflict (#2574) — thanks @rudycelekli! +- Score dub quality against the selected track’s saved language text, reject stale checks, and preserve audio and subtitle edits when generation conflicts (#2574) — thanks @rudycelekli! - Accept Japanese Han letters in translation and refinement script checks (#2576) — thanks @rudycelekli! - Keep audiobook chapter boundaries when importing CR-only manuscripts (#2508) — thanks @rudycelekli! - Preserve busy sidecars during engine-level unload instead of terminating their active operation (#2507) — thanks @Anuj04432 and @rudycelekli! diff --git a/backend/api/routers/dub_generate.py b/backend/api/routers/dub_generate.py index 0d120ce3b..626486b4c 100644 --- a/backend/api/routers/dub_generate.py +++ b/backend/api/routers/dub_generate.py @@ -5,10 +5,15 @@ import logging import time import asyncio +import copy +import contextlib +import shutil +import tempfile import torch from fastapi import APIRouter, HTTPException from core.db import db_conn +from core.logging_utils import log_safe from core.config import DUB_DIR, VOICES_DIR, dub_seg_path from core.tasks import task_manager from schemas.requests import DubRequest @@ -16,7 +21,7 @@ from services.srt_parser import vouch_cue_source from services.tts_backend import TTSBackend, resolve_generation_backend, active_backend_id from services.dub_batching import batch_timeout_s, native_batch_width -from services import gpu_gateway +from services import gpu_gateway, dub_pipeline from services.audio_dsp import apply_mastering, normalize_audio, apply_effects_chain, get_effect_chain from services.audio_io import atomic_save_wav, audio_info, _safe_torchaudio_save from services.ffmpeg_utils import ( @@ -39,6 +44,41 @@ logger = logging.getLogger("omnivoice.dub") +@contextlib.contextmanager +def _install_dub_artifacts(staged: dict[str, str]): + """Install a render's WAVs together, rolling back ordinary commit failures. + + Paths are on the same filesystem. Hard-link backups avoid duplicating long + tracks where supported; copying is the portable fallback. This is rollback + for a failed publication, not a multi-file power-loss transaction. + """ + backups = {} + installed = [] + try: + for destination, source in staged.items(): + backup = source + ".previous" + if os.path.exists(destination): + try: + os.link(destination, backup) + except OSError: + shutil.copy2(destination, backup) + backups[destination] = backup + else: + backups[destination] = None + for destination, source in staged.items(): + os.replace(source, destination) + installed.append(destination) + yield + except BaseException: + for destination in reversed(installed): + backup = backups[destination] + if backup is not None: + os.replace(backup, destination) + else: + os.unlink(destination) + raise + + class _RemoteDubBackend: """Sample-rate carrier while Dubbing runs without local TTS weights.""" @@ -200,7 +240,7 @@ def _sync_job_segments(job: dict, req: DubRequest) -> None: # by a later row. Those stable-ID matches take priority over index fallback. reserved_existing = {by_id[str(sid)] for sid in seg_ids if sid is not None and str(sid) in by_id} - # The current render seeds exactly this manifest before synthesis. Only a + # The current render supplies exactly this manifest in its projection. Only a # complete matching vector can supply an otherwise missing row identity. render_order = job.get("seg_order") expected_order = [seg_ids[i] if i < len(seg_ids) else f"seg_{i}" @@ -583,6 +623,10 @@ async def dub_generate(job_id: str, req: DubRequest): "seg_order": expected_order, }, req) + # Subtitle imports can replace the source while synthesis or fitting awaits. + # Keep the admission snapshot so publishing cannot overwrite those edits. + source_segments = copy.deepcopy(job.get("segments")) + # ── Engine resolution (issue #312 class) ──────────────────────────────── # Every rendered segment clones either source speech or a saved profile, so # local execution still requires a cloning-capable engine. Remote execution @@ -626,7 +670,7 @@ async def dub_generate(job_id: str, req: DubRequest): _job_num_step = req.num_step if req.num_step is not None else _profile_defaults.get("num_step", 16) _job_postprocess = _profile_defaults.get("postprocess_output", True) - async def _stream(task_id): + async def _render_stream(task_id, staging_dir): total = len(req.segments) all_segment_wavs = [] sync_scores = [] @@ -643,6 +687,14 @@ def _seg_lang_path(seg_key) -> str: # remain readable via the gated fallback (_legacy_seg_cache_ok). return dub_seg_path(job_id, f"{lang_code}_{seg_key}") + staged_segments = {} + + def _staged_segment_path(seg_key) -> str: + destination = _seg_lang_path(seg_key) + source = os.path.join(staging_dir, os.path.basename(destination)) + staged_segments[destination] = source + return source + # Throttle the device cache flush. empty_cache() is a synchronous # device stall, so calling it every segment (as the old code did) # serialised the GPU loop; the batched-I/O design it replaced kept @@ -684,7 +736,7 @@ def _store_mix_wav(start: float, end: float, wav: torch.Tensor, sr: int, seg_key """ if wav.shape[-1] <= 0: return (start, end, torch.zeros(1, 0), sr) - path = dub_seg_path(job_id, seg_key) + path = os.path.join(staging_dir, os.path.basename(dub_seg_path(job_id, seg_key))) os.makedirs(os.path.dirname(path), exist_ok=True) atomic_save_wav(path, wav.detach().cpu(), sr) if seg_key.startswith("mix_"): @@ -725,7 +777,6 @@ def _write_memmap_wav_atomic(target_path: str, samples, sample_rate: int) -> Non time (see the seg-write path below), exactly as ``main`` does. Re-marking here would double-mark every segment in the final mix. """ - import tempfile import wave import numpy as np @@ -811,10 +862,8 @@ def _write_memmap_wav_atomic(target_path: str, samples, sample_rate: int) -> Non if not intact: regen_only.add(sid) _validate_render_languages(backend, req, seg_ids, regen_only) - # Manifest: stable segment id per current index. Per-segment WAVs are - # named by stable id (dub_seg_path) so regen reuses the right audio after - # reorder; index-keyed readers (preview/export) resolve via this manifest. - job["seg_order"] = list(expected_order) + # Keep the new manifest private until the track is complete. Assembly + # uses expected_order; preview/export retain the previous successful one. # Per-segment metadata to persist after the hot loop. Audio itself is # written immediately and only file paths are kept, so long videos don't @@ -1600,7 +1649,7 @@ def _render_batch() -> list[torch.Tensor]: # RVC needs the WAV on disk, so write it immediately only # when RVC is active (uncommon path). if rvc_is_enabled(): - seg_wav_path = _seg_lang_path(seg_id) + seg_wav_path = _staged_segment_path(seg_id) atomic_save_wav(seg_wav_path, audio_tensor, backend.sample_rate) try: await loop.run_in_executor(_gpu_pool, apply_rvc, seg_wav_path) @@ -1624,28 +1673,11 @@ def _render_batch() -> list[torch.Tensor]: audio_tensor = mark_synthetic(audio_tensor, backend.sample_rate, context="dub_generate.segment") - seg_wav_path = _seg_lang_path(seg_id) - try: - # Keep the existing per-segment WAV contract for previews - # and partial regeneration, but do not keep the tensor in RAM. - atomic_save_wav(seg_wav_path, audio_tensor, backend.sample_rate) - except Exception as e: - logger.warning("seg write failed for %s: %s", seg_id, e) - # If the durable segment write fails, still preserve a mix - # copy so this generation can finish. - all_segment_wavs.append(_store_mix_wav(seg.start, seg.end, audio_tensor, backend.sample_rate, f"mix_{seg_id}")) - try: - del audio_tensor - except Exception: - pass - _release_audio_tensors() - else: - all_segment_wavs.append((seg.start, seg.end, seg_wav_path, backend.sample_rate)) - try: - del audio_tensor - except Exception: - pass - _release_audio_tensors() + seg_wav_path = _staged_segment_path(seg_id) + atomic_save_wav(seg_wav_path, audio_tensor, backend.sample_rate) + all_segment_wavs.append((seg.start, seg.end, seg_wav_path, backend.sample_rate)) + del audio_tensor + _release_audio_tensors() except Exception as e: # A task-stream error bypasses the global exception handler. # Never publish engine exception text here: allocator errors @@ -1673,34 +1705,6 @@ def _render_batch() -> list[torch.Tensor]: yield f"data: {json.dumps({'type': 'error', 'error_code': exc.detail['code'], 'error': exc.detail['message']})}\n\n" return - # ── Batch metadata phase ────────────────────────────────────── - # Per-segment WAVs were written during the loop to keep RAM bounded. - # Flush only lightweight fingerprints/quality metadata here. - _t_diskw_0 = time.perf_counter() - # P1.3 — fingerprints live per language so each track's staleness is - # judged against ITS OWN last generate. The flat job["seg_hashes"] is - # kept as a mirror of the CURRENT track's map: every existing consumer - # (the `done` event, dub-history restore, older frontends) already - # treats it as "the hashes of the language generated last", which is - # exactly what it now provably contains. - hashes = _seg_hashes_by_lang(job).setdefault(lang_code, {}) - quality_map = job.setdefault("seg_num_step", {}) - for (_si, _sr, _sid, _fp, _nstep) in _pending_seg_writes: - if _fp is not None: - hashes[_sid] = _fp - quality_map[_sid] = _nstep - job["seg_hashes"] = dict(hashes) - # Duration-planner calibration: per-language (chars, natural dur) - # records. update() (not replace) so partial regens keep accumulating - # samples from earlier runs of this track. - if _natural_dur_records: - job.setdefault("seg_natural_durs_by_lang", {}).setdefault( - lang_code, {}, - ).update(_natural_dur_records) - # Single job flush instead of one per 8 segments. - _save_job(job_id, job) - _t_diskw = time.perf_counter() - _t_diskw_0 - sr = backend.sample_rate slot_fit = (req.slot_fit or "time_stretch").lower() overflow_budget_s = max(0.0, float(req.overflow_budget_s or 0.0)) @@ -1765,7 +1769,7 @@ def _render_batch() -> list[torch.Tensor]: allow_video_retime=bool(_fo.allow_video_retime) if _fo is not None and _fo.allow_video_retime is not None else _fit_defaults.allow_video_retime, min_audio_rate=_underrun_min_rate(), ) - _seg_order = job.get("seg_order") or [] + _seg_order = expected_order fit_plan = plan_fit( [ { @@ -1792,7 +1796,6 @@ def _render_batch() -> list[torch.Tensor]: os.makedirs(os.path.dirname(track_path), exist_ok=True) import gc - import tempfile import numpy as np mix_samples = max(total_samples, 1) @@ -1911,7 +1914,7 @@ def _render_batch() -> list[torch.Tensor]: slot_samples_eff = int(max(0.0, (effective_end - start)) * sr) if slot_samples_eff > 0 and wl > slot_samples_eff: overflow_s = (wl - slot_samples_eff) / sr - yield f"data: {json.dumps({'type': 'error', 'segment': i, 'segment_id': job['seg_order'][i], 'error_code': 'dub_timing_overflow', 'error': 'Speech exceeds its time slot. Shorten the translation or choose Strict Slot or Stretch Video before exporting.', 'overflow_s': round(overflow_s, 3)})}\n\n" + yield f"data: {json.dumps({'type': 'error', 'segment': i, 'segment_id': expected_order[i], 'error_code': 'dub_timing_overflow', 'error': 'Speech exceeds its time slot. Shorten the translation or choose Strict Slot or Stretch Video before exporting.', 'overflow_s': round(overflow_s, 3)})}\n\n" return else: fit_status.append({"status": "fits"}) @@ -2016,9 +2019,15 @@ def _render_batch() -> list[torch.Tensor]: pass _release_audio_tensors() + # No await separates this final check from track/metadata publication. + # In particular, a subtitle import during asynchronous fitting wins. + if _get_job(job_id) is not job or job.get("segments") != source_segments: + yield f"data: {json.dumps({'type': 'error', 'error_code': 'dub_source_changed', 'error': 'Subtitles changed during generation. Generate again to use the current subtitles.'})}\n\n" + return _t_save_0 = time.perf_counter() mix_audio.flush() - _write_memmap_wav_atomic(track_path, mix_audio[:mix_samples], sr) + staged_track = os.path.join(staging_dir, os.path.basename(track_path)) + _write_memmap_wav_atomic(staged_track, mix_audio[:mix_samples], sr) _t_save = time.perf_counter() - _t_save_0 _t_mix = _t_save_0 - _t_loop_end finally: @@ -2051,66 +2060,115 @@ def _render_batch() -> list[torch.Tensor]: # mux step needs this to know whether to use the original video as-is # or stretch it per the plan. track_dur = total_samples / sr if total_samples > 0 else 0.0 - # Publish the already validated snapshot with the completed track. - # Preserve other languages that may have been added during rendering. - job["segments"] = publication["segments"] - for key in ("segments_i18n", "segments_i18n_cue_sources"): - job.setdefault(key, {}).update(publication[key]) - job["dubbed_tracks"][lang_code] = { - "path": track_path, - "language": req.language, - "language_code": lang_code, - "duration": round(track_dur, 4), - "timing_strategy": strategy, - } - - # Persist the timing strategy + (for Mode B) the per-segment stretch - # plan so dub_export can build the matching video pipeline at mux - # time. Plans are keyed by language code because each language gets - # its own dub track with its own natural-rate audio layout. - job["language"] = req.language - job["language_code"] = lang_code - job["timing_strategy"] = strategy - # Keep job segments in lock-step with what was just rendered so - # subtitle export / burn-in use the translated text (#309). - job["dubbed_tracks"][lang_code]["source_segments"] = _track_source_segments(job) - if strategy == "stretch_video": - stretch_plans = job.setdefault("video_stretch_plans", {}) - stretch_plans[lang_code] = { - "plan": video_stretch_plan, - "total_duration": round(track_dur, 4), - "orig_duration": round(orig_total_dur, 4), - } - elif strategy == "smart_fit" and fit_plan is not None: - # video_stretch_plans stays untouched — smart_fit persists its - # own keyspace so a job can carry both without clobbering. - _fit_params_payload = { - "timing_strategy": strategy, - "max_audio_only_rate": fit_params.max_audio_only_rate, - "audio_rate_cap": fit_params.audio_rate_cap, - "video_slow_cap": fit_params.video_slow_cap, - "gap_guard_s": fit_params.gap_guard_s, - "allow_video_retime": fit_params.allow_video_retime, - } - fit_fp = fit_fingerprint(_fit_params_payload) - job.setdefault("fit_plans", {})[lang_code] = { - # Same dict shape _build_video_stretch_filter_graph consumes. - "plan": fit_plan.video_plan, - # Cue times from actual stretched sample positions — for - # subtitle export on the fitted timeline. - "fitted_segments": fitted_cues, - "total_duration": round(track_dur, 4), - "orig_duration": round(orig_total_dur, 4), - "params": _fit_params_payload, - "fit_fp": fit_fp, - } - job["dubbed_tracks"][lang_code]["fit_fp"] = fit_fp - # Every new cache preserves natural speech. Old slotted caches must - # be regenerated once because their missing tails cannot be recovered. - _kind = "natural" - job.setdefault("seg_wav_kind_by_lang", {})[lang_code] = _kind - job["seg_wav_kind"] = _kind - _save_job(job_id, job) + def publish(): + with dub_pipeline._dub_jobs_lock: + if _get_job(job_id) is not job or job.get("segments") != source_segments: + return False + published_job = copy.deepcopy(job) + artifacts = {**staged_segments, track_path: staged_track} + with _install_dub_artifacts(artifacts): + # ── Batch metadata phase ────────────────────────────────────── + # Per-segment WAVs were written during the loop to keep RAM bounded. + # Flush only lightweight fingerprints/quality metadata here. + # P1.3 — fingerprints live per language so each track's staleness is + # judged against ITS OWN last generate. The flat published_job["seg_hashes"] is + # kept as a mirror of the CURRENT track's map: every existing consumer + # (the `done` event, dub-history restore, older frontends) already + # treats it as "the hashes of the language generated last", which is + # exactly what it now provably contains. + hashes = _seg_hashes_by_lang(published_job).setdefault(lang_code, {}) + quality_map = published_job.setdefault("seg_num_step", {}) + for (_si, _sr, _sid, _fp, _nstep) in _pending_seg_writes: + if _fp is not None: + hashes[_sid] = _fp + quality_map[_sid] = _nstep + published_job["seg_hashes"] = dict(hashes) + # Duration-planner calibration: per-language (chars, natural dur) + # records. update() (not replace) so partial regens keep accumulating + # samples from earlier runs of this track. + if _natural_dur_records: + published_job.setdefault("seg_natural_durs_by_lang", {}).setdefault( + lang_code, {}, + ).update(_natural_dur_records) + published_job["seg_order"] = list(expected_order) + # Publish the already validated snapshot with the completed track. + # Preserve other languages that may have been added during rendering. + published_job["segments"] = publication["segments"] + for key in ("segments_i18n", "segments_i18n_cue_sources"): + published_job.setdefault(key, {}).update(publication[key]) + published_job["dubbed_tracks"][lang_code] = { + "path": track_path, + "language": req.language, + "language_code": lang_code, + "duration": round(track_dur, 4), + "timing_strategy": strategy, + } + + # Persist the timing strategy + (for Mode B) the per-segment stretch + # plan so dub_export can build the matching video pipeline at mux + # time. Plans are keyed by language code because each language gets + # its own dub track with its own natural-rate audio layout. + published_job["language"] = req.language + published_job["language_code"] = lang_code + published_job["timing_strategy"] = strategy + # Keep published_job segments in lock-step with what was just rendered so + # subtitle export / burn-in use the translated text (#309). + published_job["dubbed_tracks"][lang_code]["source_segments"] = _track_source_segments(published_job) + if strategy == "stretch_video": + stretch_plans = published_job.setdefault("video_stretch_plans", {}) + stretch_plans[lang_code] = { + "plan": video_stretch_plan, + "total_duration": round(track_dur, 4), + "orig_duration": round(orig_total_dur, 4), + } + elif strategy == "smart_fit" and fit_plan is not None: + # video_stretch_plans stays untouched — smart_fit persists its + # own keyspace so a published_job can carry both without clobbering. + _fit_params_payload = { + "timing_strategy": strategy, + "max_audio_only_rate": fit_params.max_audio_only_rate, + "audio_rate_cap": fit_params.audio_rate_cap, + "video_slow_cap": fit_params.video_slow_cap, + "gap_guard_s": fit_params.gap_guard_s, + "allow_video_retime": fit_params.allow_video_retime, + } + fit_fp = fit_fingerprint(_fit_params_payload) + published_job.setdefault("fit_plans", {})[lang_code] = { + # Same dict shape _build_video_stretch_filter_graph consumes. + "plan": fit_plan.video_plan, + # Cue times from actual stretched sample positions — for + # subtitle export on the fitted timeline. + "fitted_segments": fitted_cues, + "total_duration": round(track_dur, 4), + "orig_duration": round(orig_total_dur, 4), + "params": _fit_params_payload, + "fit_fp": fit_fp, + } + published_job["dubbed_tracks"][lang_code]["fit_fp"] = fit_fp + # Every new cache preserves natural speech. Old slotted caches must + # be regenerated once because their missing tails cannot be recovered. + _kind = "natural" + published_job.setdefault("seg_wav_kind_by_lang", {})[lang_code] = _kind + published_job["seg_wav_kind"] = _kind + _save_job(job_id, published_job) + + job.clear() + job.update(published_job) + return True + + _t_diskw_0 = time.perf_counter() + try: + committed = publish() + except Exception as exc: + from core.public_errors import stream_generation_failure + logger.exception("Dub publication failed for job %s", log_safe(job_id)) + detail = stream_generation_failure(exc)["detail"] + yield f"data: {json.dumps({'type': 'error', 'error': detail})}\n\n" + return + if not committed: + yield f"data: {json.dumps({'type': 'error', 'error_code': 'dub_source_changed', 'error': 'Subtitles changed during generation. Generate again to use the current subtitles.'})}\n\n" + return + _t_diskw = time.perf_counter() - _t_diskw_0 _t_total = time.perf_counter() - _t_start logger.info( @@ -2121,6 +2179,16 @@ def _render_batch() -> list[torch.Tensor]: yield f"data: {json.dumps({'type': 'done', 'segments_processed': total, 'language_code': lang_code, 'tracks': list(job['dubbed_tracks'].keys()), 'sync_scores': sync_scores, 'fit_status': fit_status, 'timing_strategy': strategy, 'seg_hashes': job.get('seg_hashes', {}), 'seg_num_step': job.get('seg_num_step', {})})}\n\n" + async def _stream(task_id): + job_dir = os.path.join(DUB_DIR, job_id) + os.makedirs(job_dir, exist_ok=True) + # The context also cleans staged WAVs on rejection, cancellation and + # generator close; no rejected speech can become a partial-regen cache. + with tempfile.TemporaryDirectory(prefix=".render-", dir=job_dir) as staging_dir: + async with contextlib.aclosing(_render_stream(task_id, staging_dir)) as render: + async for event in render: + yield event + task_id = f"dub_{job_id}_{int(time.time())}" await task_manager.add_task(task_id, "dub_generate", _stream, task_id) return {"task_id": task_id} diff --git a/backend/core/tasks.py b/backend/core/tasks.py index 00d483f13..703bf58ae 100644 --- a/backend/core/tasks.py +++ b/backend/core/tasks.py @@ -2,6 +2,7 @@ import time import json import logging +from contextlib import aclosing from core import job_store from core import failure @@ -138,24 +139,24 @@ async def worker(self): import inspect res = func(*args, **kwargs) if inspect.isasyncgen(res): - async for update in res: - if t.get("cancelled"): - await self._push_event(task_id, f"data: {json.dumps({'type': 'cancelled'})}\n\n") - t["status"] = "cancelled" - try: job_store.mark_cancelled(task_id) - except Exception: logger.exception("job_store.mark_cancelled failed") - break - await self._push_event(task_id, update) - stream_error = _stream_failure(update) - if stream_error is not None: - t["status"] = "failed" - t["error"] = stream_error - try: - job_store.mark_failed(task_id, stream_error) - except Exception: - logger.exception("job_store.mark_failed failed") - await res.aclose() - break + async with aclosing(res): + async for update in res: + if t.get("cancelled"): + await self._push_event(task_id, f"data: {json.dumps({'type': 'cancelled'})}\n\n") + t["status"] = "cancelled" + try: job_store.mark_cancelled(task_id) + except Exception: logger.exception("job_store.mark_cancelled failed") + break + await self._push_event(task_id, update) + stream_error = _stream_failure(update) + if stream_error is not None: + t["status"] = "failed" + t["error"] = stream_error + try: + job_store.mark_failed(task_id, stream_error) + except Exception: + logger.exception("job_store.mark_failed failed") + break elif inspect.iscoroutine(res): await res if t["status"] not in {"cancelled", "failed"}: diff --git a/backend/schemas/requests.py b/backend/schemas/requests.py index e41ba535d..5516bd5b0 100644 --- a/backend/schemas/requests.py +++ b/backend/schemas/requests.py @@ -60,7 +60,7 @@ class FitOptions(BaseModel): allow_video_retime: Optional[bool] = None # default True class DubRequest(BaseModel): - segments: List[DubSegment] + segments: List[DubSegment] = Field(min_length=1) language: str = "Auto" language_code: str = "und" # ISO 639-1 for ffmpeg metadata (e.g. "es", "fr", "de") instruct: str = "" diff --git a/docs/electron-dubbing.md b/docs/electron-dubbing.md index a661854c6..ba16857d7 100644 --- a/docs/electron-dubbing.md +++ b/docs/electron-dubbing.md @@ -410,4 +410,4 @@ runs. If the track, transcript, timing, or audio changes during that pass, QC asks you to run it again instead of publishing stale scores. Deleting the job during QC also discards the result and keeps it out of history. -Generation revalidates its segment text and identity snapshot after synthesis, before publishing fingerprints or replacing the previous track. An identity collision caused by a concurrent edit is reported through the task stream; the previous track and published metadata remain available. +Generation revalidates its segment text and identity snapshot after synthesis, before publishing fingerprints or replacing the previous track. An identity collision caused by a concurrent edit is reported through the task stream; the previous track and published metadata remain available. The source snapshot is checked again after asynchronous fitting and before replacing audio: subtitle imports made during generation or assembly survive, and the user can regenerate from the corrected subtitles. Fresh segment WAVs and the assembled track stay in a private staging directory until this check passes. Rejection or cancellation removes the staging files and preserves the reusable segment cache. Publication backs up existing files and rolls back ordinary installation failures; it does not promise a multi-file transaction across power loss. Fingerprints and the segment manifest publish only with a completed track. Empty generation requests are rejected before loading a voice engine. diff --git a/electron/src/renderer/src/i18n/locales/ar.json b/electron/src/renderer/src/i18n/locales/ar.json index e18a16f82..ac936ccd5 100644 --- a/electron/src/renderer/src/i18n/locales/ar.json +++ b/electron/src/renderer/src/i18n/locales/ar.json @@ -821,6 +821,7 @@ "qc_failed": "فشل التحقق من التوقيت: {{message}}", "qc_track_changed": "تغيرت الدبلجة أثناء فحص الجودة. أعد الفحص على المسار الحالي.", "qc_identity_missing": "هويات المقاطع غير واضحة. أعد إنشاء هذا المسار بمعرّفات فريدة للمقاطع.", + "source_changed": "تغيّرت الترجمة أثناء التوليد. أعد التوليد لاستخدام الترجمة الحالية.", "paste_translation_btn": "لصق ترجمة", "paste_translation_title": "لصق ترجمة", "paste_translation_desc": "الصق ترجمة أعددتها في مكان آخر (ChatGPT أو DeepL أو مترجم بشري). ستُطابَق مع المقاطع الموجودة لديك — تبقى التوقيتات والنص الأصلي دون تغيير.", diff --git a/electron/src/renderer/src/i18n/locales/de.json b/electron/src/renderer/src/i18n/locales/de.json index c893426fb..36fe4da31 100644 --- a/electron/src/renderer/src/i18n/locales/de.json +++ b/electron/src/renderer/src/i18n/locales/de.json @@ -811,6 +811,7 @@ "qc_failed": "Zeitprüfung fehlgeschlagen: {{message}}", "qc_track_changed": "Die Synchronisation wurde während der Qualitätsprüfung geändert. Prüfe die aktuelle Spur erneut.", "qc_identity_missing": "Die Segmentzuordnung ist uneindeutig. Erzeuge diese Spur mit eindeutigen Segment-IDs neu.", + "source_changed": "Die Untertitel wurden während der Generierung geändert. Generiere erneut, um die aktuellen Untertitel zu verwenden.", "paste_translation_btn": "Übersetzung einfügen", "paste_translation_title": "Übersetzung einfügen", "paste_translation_desc": "Füge eine anderswo erstellte Übersetzung ein (ChatGPT, DeepL, ein menschlicher Übersetzer). Sie wird den vorhandenen Segmenten zugeordnet — Timings und Originaltranskript bleiben unverändert.", diff --git a/electron/src/renderer/src/i18n/locales/en.json b/electron/src/renderer/src/i18n/locales/en.json index 83f713c11..636c44181 100644 --- a/electron/src/renderer/src/i18n/locales/en.json +++ b/electron/src/renderer/src/i18n/locales/en.json @@ -967,6 +967,7 @@ "qc_failed": "Timing check failed: {{message}}", "qc_track_changed": "The dub changed during the quality check. Run the check again on the current track.", "qc_identity_missing": "Segment identities are ambiguous. Regenerate this track with unique segment IDs.", + "source_changed": "Subtitles changed during generation. Generate again to use the current subtitles.", "export_btn": "Export…", "prep_stop": "Stop", "install_progress": "Installing {{engine}}…", diff --git a/electron/src/renderer/src/i18n/locales/es.json b/electron/src/renderer/src/i18n/locales/es.json index 1eeb88e5b..09c3c5404 100644 --- a/electron/src/renderer/src/i18n/locales/es.json +++ b/electron/src/renderer/src/i18n/locales/es.json @@ -813,6 +813,7 @@ "qc_failed": "Error en la verificación de tiempo: {{message}}", "qc_track_changed": "El doblaje cambió durante la comprobación de calidad. Repite la comprobación en la pista actual.", "qc_identity_missing": "La identidad de los segmentos es ambigua. Regenera esta pista con identificadores de segmento únicos.", + "source_changed": "Los subtítulos cambiaron durante la generación. Genera de nuevo para usar los subtítulos actuales.", "paste_translation_btn": "Pegar traducción", "paste_translation_title": "Pegar una traducción", "paste_translation_desc": "Pega una traducción hecha en otro sitio (ChatGPT, DeepL, un traductor humano). Se asigna a los segmentos que ya tienes: los tiempos y la transcripción original no cambian.", diff --git a/electron/src/renderer/src/i18n/locales/fr.json b/electron/src/renderer/src/i18n/locales/fr.json index ac15faa50..bd7d9d0ee 100644 --- a/electron/src/renderer/src/i18n/locales/fr.json +++ b/electron/src/renderer/src/i18n/locales/fr.json @@ -813,6 +813,7 @@ "qc_failed": "Échec de la vérification du timing : {{message}}", "qc_track_changed": "Le doublage a changé pendant le contrôle qualité. Relancez le contrôle sur la piste actuelle.", "qc_identity_missing": "Les segments ne sont pas identifiés de façon univoque. Régénérez cette piste avec des identifiants de segment uniques.", + "source_changed": "Les sous-titres ont changé pendant la génération. Relancez la génération pour utiliser les sous-titres actuels.", "paste_translation_btn": "Coller une traduction", "paste_translation_title": "Coller une traduction", "paste_translation_desc": "Collez une traduction réalisée ailleurs (ChatGPT, DeepL, un traducteur humain). Elle est appliquée aux segments existants — les timings et la transcription d'origine restent intacts.", diff --git a/electron/src/renderer/src/i18n/locales/hi.json b/electron/src/renderer/src/i18n/locales/hi.json index cb929d8a4..2e8544dcf 100644 --- a/electron/src/renderer/src/i18n/locales/hi.json +++ b/electron/src/renderer/src/i18n/locales/hi.json @@ -811,6 +811,7 @@ "qc_failed": "समय की जाँच विफल: {{message}}", "qc_track_changed": "गुणवत्ता जाँच के दौरान डबिंग बदल गई। मौजूदा ट्रैक पर जाँच दोबारा चलाएँ।", "qc_identity_missing": "सेगमेंट की पहचान स्पष्ट नहीं है। हर सेगमेंट को अलग आईडी देकर इस ट्रैक को फिर से जनरेट करें।", + "source_changed": "जनरेशन के दौरान सबटाइटल बदल गए। मौजूदा सबटाइटल इस्तेमाल करने के लिए फिर से जनरेट करें।", "paste_translation_btn": "अनुवाद चिपकाएँ", "paste_translation_title": "अनुवाद चिपकाएँ", "paste_translation_desc": "कहीं और तैयार किया गया अनुवाद चिपकाएँ (ChatGPT, DeepL, कोई मानव अनुवादक)। यह आपके मौजूदा सेगमेंट पर लागू होगा — टाइमिंग और मूल ट्रांसक्रिप्ट अछूते रहते हैं।", diff --git a/electron/src/renderer/src/i18n/locales/id.json b/electron/src/renderer/src/i18n/locales/id.json index ddcbfca96..3be04a0e1 100644 --- a/electron/src/renderer/src/i18n/locales/id.json +++ b/electron/src/renderer/src/i18n/locales/id.json @@ -813,6 +813,7 @@ "qc_failed": "Pemeriksaan waktu gagal: {{message}}", "qc_track_changed": "Sulih suara berubah selama pemeriksaan kualitas. Jalankan kembali pemeriksaan pada trek saat ini.", "qc_identity_missing": "Identitas segmen tidak jelas. Buat ulang trek ini dengan ID segmen yang unik.", + "source_changed": "Subtitel berubah selama pembuatan. Buat ulang untuk menggunakan subtitel saat ini.", "paste_translation_btn": "Tempel terjemahan", "paste_translation_title": "Tempel terjemahan", "paste_translation_desc": "Tempel terjemahan yang kamu buat di tempat lain (ChatGPT, DeepL, penerjemah manusia). Terjemahan itu dipetakan ke segmen yang sudah ada — pewaktuan dan transkrip asli tidak diubah.", diff --git a/electron/src/renderer/src/i18n/locales/it.json b/electron/src/renderer/src/i18n/locales/it.json index 47d63d000..fe6a5b767 100644 --- a/electron/src/renderer/src/i18n/locales/it.json +++ b/electron/src/renderer/src/i18n/locales/it.json @@ -813,6 +813,7 @@ "qc_failed": "Controllo cronometraggio fallito: {{message}}", "qc_track_changed": "Il doppiaggio è cambiato durante il controllo qualità. Ripeti il controllo sulla traccia attuale.", "qc_identity_missing": "Le identità dei segmenti sono ambigue. Rigenera questa traccia con identificativi di segmento univoci.", + "source_changed": "I sottotitoli sono cambiati durante la generazione. Genera di nuovo per usare i sottotitoli attuali.", "paste_translation_btn": "Incolla traduzione", "paste_translation_title": "Incolla una traduzione", "paste_translation_desc": "Incolla una traduzione prodotta altrove (ChatGPT, DeepL, un traduttore umano). Viene mappata sui segmenti che hai già: tempi e trascrizione originale restano invariati.", diff --git a/electron/src/renderer/src/i18n/locales/ja.json b/electron/src/renderer/src/i18n/locales/ja.json index 49f2605ac..506f4e729 100644 --- a/electron/src/renderer/src/i18n/locales/ja.json +++ b/electron/src/renderer/src/i18n/locales/ja.json @@ -813,6 +813,7 @@ "qc_failed": "タイミング チェックに失敗しました: {{message}}", "qc_track_changed": "品質チェック中に吹き替えが変更されました。現在のトラックで再度チェックしてください。", "qc_identity_missing": "セグメントを一意に識別できません。各セグメントに固有のIDを付けて、このトラックを再生成してください。", + "source_changed": "生成中に字幕が変更されました。現在の字幕を使用するには、もう一度生成してください。", "paste_translation_btn": "翻訳を貼り付け", "paste_translation_title": "翻訳を貼り付け", "paste_translation_desc": "他所で用意した翻訳(ChatGPT、DeepL、人間の翻訳者)を貼り付けます。既存のセグメントに割り当てられ、タイミングと元の文字起こしはそのまま残ります。", diff --git a/electron/src/renderer/src/i18n/locales/ko.json b/electron/src/renderer/src/i18n/locales/ko.json index 76415777e..1bdfbf282 100644 --- a/electron/src/renderer/src/i18n/locales/ko.json +++ b/electron/src/renderer/src/i18n/locales/ko.json @@ -811,6 +811,7 @@ "qc_failed": "타이밍 확인 실패: {{message}}", "qc_track_changed": "품질 검사 중 더빙이 변경되었습니다. 현재 트랙에서 검사를 다시 실행하세요.", "qc_identity_missing": "세그먼트를 명확하게 구분할 수 없습니다. 각 세그먼트에 고유한 ID를 지정하여 이 트랙을 다시 생성하세요.", + "source_changed": "생성 중에 자막이 변경되었습니다. 현재 자막을 사용하려면 다시 생성하세요.", "paste_translation_btn": "번역 붙여넣기", "paste_translation_title": "번역 붙여넣기", "paste_translation_desc": "다른 곳에서 만든 번역(ChatGPT, DeepL, 사람 번역가)을 붙여넣으세요. 이미 있는 세그먼트에 매핑되며 타이밍과 원본 전사는 그대로 유지됩니다.", diff --git a/electron/src/renderer/src/i18n/locales/nl.json b/electron/src/renderer/src/i18n/locales/nl.json index 6a5549c6f..5aac46018 100644 --- a/electron/src/renderer/src/i18n/locales/nl.json +++ b/electron/src/renderer/src/i18n/locales/nl.json @@ -811,6 +811,7 @@ "qc_failed": "Timingcontrole mislukt: {{message}}", "qc_track_changed": "De nasynchronisatie is tijdens de kwaliteitscontrole gewijzigd. Voer de controle opnieuw uit op het huidige spoor.", "qc_identity_missing": "De segmenten zijn niet eenduidig te identificeren. Genereer dit spoor opnieuw met unieke segment-ID’s.", + "source_changed": "De ondertitels zijn gewijzigd tijdens het genereren. Genereer opnieuw om de huidige ondertitels te gebruiken.", "paste_translation_btn": "Vertaling plakken", "paste_translation_title": "Een vertaling plakken", "paste_translation_desc": "Plak een elders gemaakte vertaling (ChatGPT, DeepL, een menselijke vertaler). Die wordt op je bestaande segmenten toegepast — timings en het originele transcript blijven ongewijzigd.", diff --git a/electron/src/renderer/src/i18n/locales/pl.json b/electron/src/renderer/src/i18n/locales/pl.json index af04d422f..69082f844 100644 --- a/electron/src/renderer/src/i18n/locales/pl.json +++ b/electron/src/renderer/src/i18n/locales/pl.json @@ -815,6 +815,7 @@ "qc_failed": "Kontrola czasu nie powiodła się: {{message}}", "qc_track_changed": "Dubbing zmienił się podczas kontroli jakości. Uruchom kontrolę ponownie dla bieżącej ścieżki.", "qc_identity_missing": "Tożsamość segmentów jest niejednoznaczna. Wygeneruj tę ścieżkę ponownie z unikatowymi identyfikatorami segmentów.", + "source_changed": "Napisy zmieniły się podczas generowania. Wygeneruj ponownie, aby użyć aktualnych napisów.", "paste_translation_btn": "Wklej tłumaczenie", "paste_translation_title": "Wklej tłumaczenie", "paste_translation_desc": "Wklej tłumaczenie przygotowane gdzie indziej (ChatGPT, DeepL, tłumacz). Zostanie dopasowane do istniejących segmentów — czasy i oryginalna transkrypcja pozostają bez zmian.", diff --git a/electron/src/renderer/src/i18n/locales/pt.json b/electron/src/renderer/src/i18n/locales/pt.json index 4769b1a27..b089a1926 100644 --- a/electron/src/renderer/src/i18n/locales/pt.json +++ b/electron/src/renderer/src/i18n/locales/pt.json @@ -813,6 +813,7 @@ "qc_failed": "Falha na verificação de tempo: {{message}}", "qc_track_changed": "A dublagem mudou durante a verificação de qualidade. Execute a verificação novamente na faixa atual.", "qc_identity_missing": "A identificação dos segmentos é ambígua. Gere esta faixa novamente com identificadores de segmento exclusivos.", + "source_changed": "As legendas mudaram durante a geração. Gere novamente para usar as legendas atuais.", "paste_translation_btn": "Colar tradução", "paste_translation_title": "Colar uma tradução", "paste_translation_desc": "Cole uma tradução feita noutro lugar (ChatGPT, DeepL, um tradutor humano). Ela é mapeada nos segmentos que já existem — os tempos e a transcrição original ficam intactos.", diff --git a/electron/src/renderer/src/i18n/locales/ru.json b/electron/src/renderer/src/i18n/locales/ru.json index 933008f60..66b728aff 100644 --- a/electron/src/renderer/src/i18n/locales/ru.json +++ b/electron/src/renderer/src/i18n/locales/ru.json @@ -815,6 +815,7 @@ "qc_failed": "Проверка времени не удалась: {{message}}", "qc_track_changed": "Дубляж изменился во время проверки качества. Повторите проверку текущей дорожки.", "qc_identity_missing": "Сегменты невозможно однозначно определить. Создайте эту дорожку заново с уникальными идентификаторами сегментов.", + "source_changed": "Субтитры изменились во время генерации. Запустите генерацию заново, чтобы использовать текущие субтитры.", "paste_translation_btn": "Вставить перевод", "paste_translation_title": "Вставить перевод", "paste_translation_desc": "Вставьте перевод, сделанный в другом месте (ChatGPT, DeepL, живой переводчик). Он ляжет на уже имеющиеся сегменты — тайминги и исходная расшифровка не изменятся.", diff --git a/electron/src/renderer/src/i18n/locales/sv.json b/electron/src/renderer/src/i18n/locales/sv.json index 05d44a470..48ded5174 100644 --- a/electron/src/renderer/src/i18n/locales/sv.json +++ b/electron/src/renderer/src/i18n/locales/sv.json @@ -813,6 +813,7 @@ "qc_failed": "Tidskontroll misslyckades: {{message}}", "qc_track_changed": "Dubbningen ändrades under kvalitetskontrollen. Kör kontrollen igen på det aktuella spåret.", "qc_identity_missing": "Segmenten kan inte identifieras entydigt. Generera om spåret med unika segment-ID:n.", + "source_changed": "Undertexterna ändrades under genereringen. Generera igen för att använda de aktuella undertexterna.", "paste_translation_btn": "Klistra in översättning", "paste_translation_title": "Klistra in en översättning", "paste_translation_desc": "Klistra in en översättning som gjorts någon annanstans (ChatGPT, DeepL, en mänsklig översättare). Den mappas mot segmenten du redan har — tidkoder och originaltranskriptet rörs inte.", diff --git a/electron/src/renderer/src/i18n/locales/th.json b/electron/src/renderer/src/i18n/locales/th.json index ec217321f..4a2f5ba9b 100644 --- a/electron/src/renderer/src/i18n/locales/th.json +++ b/electron/src/renderer/src/i18n/locales/th.json @@ -813,6 +813,7 @@ "qc_failed": "การตรวจสอบเวลาล้มเหลว: {{message}}", "qc_track_changed": "เสียงพากย์เปลี่ยนแปลงระหว่างการตรวจสอบคุณภาพ โปรดตรวจสอบแทร็กปัจจุบันอีกครั้ง", "qc_identity_missing": "ไม่สามารถระบุแต่ละช่วงได้อย่างชัดเจน โปรดสร้างแทร็กนี้ใหม่โดยใช้รหัสที่ไม่ซ้ำกันสำหรับแต่ละช่วง", + "source_changed": "คำบรรยายเปลี่ยนแปลงระหว่างการสร้าง โปรดสร้างอีกครั้งเพื่อใช้คำบรรยายปัจจุบัน", "paste_translation_btn": "วางคำแปล", "paste_translation_title": "วางคำแปล", "paste_translation_desc": "วางคำแปลที่คุณทำไว้จากที่อื่น (ChatGPT, DeepL หรือผู้แปลที่เป็นคน) ระบบจะจับคู่กับเซกเมนต์ที่มีอยู่แล้ว โดยเวลาและถอดความต้นฉบับยังคงเดิม", diff --git a/electron/src/renderer/src/i18n/locales/tr.json b/electron/src/renderer/src/i18n/locales/tr.json index e2e2d0abd..fb5685891 100644 --- a/electron/src/renderer/src/i18n/locales/tr.json +++ b/electron/src/renderer/src/i18n/locales/tr.json @@ -813,6 +813,7 @@ "qc_failed": "Zamanlama kontrolü başarısız oldu: {{message}}", "qc_track_changed": "Kalite kontrolü sırasında dublaj değişti. Geçerli parçada kontrolü yeniden çalıştırın.", "qc_identity_missing": "Bölümler kesin olarak tanımlanamıyor. Bu parçayı benzersiz bölüm kimlikleriyle yeniden oluşturun.", + "source_changed": "Oluşturma sırasında altyazılar değişti. Güncel altyazıları kullanmak için yeniden oluşturun.", "paste_translation_btn": "Çeviriyi yapıştır", "paste_translation_title": "Bir çeviri yapıştır", "paste_translation_desc": "Başka bir yerde hazırladığın çeviriyi yapıştır (ChatGPT, DeepL, bir insan çevirmen). Mevcut segmentlere eşlenir — zamanlamalar ve özgün döküm olduğu gibi kalır.", diff --git a/electron/src/renderer/src/i18n/locales/uk.json b/electron/src/renderer/src/i18n/locales/uk.json index 06a199daa..a09bca0a0 100644 --- a/electron/src/renderer/src/i18n/locales/uk.json +++ b/electron/src/renderer/src/i18n/locales/uk.json @@ -817,6 +817,7 @@ "qc_failed": "Помилка перевірки часу: {{message}}", "qc_track_changed": "Дубляж змінився під час перевірки якості. Повторіть перевірку поточної доріжки.", "qc_identity_missing": "Сегменти неможливо однозначно визначити. Створіть цю доріжку заново з унікальними ідентифікаторами сегментів.", + "source_changed": "Субтитри змінилися під час генерації. Запустіть генерацію знову, щоб використати поточні субтитри.", "paste_translation_btn": "Вставити переклад", "paste_translation_title": "Вставити переклад", "paste_translation_desc": "Вставте переклад, зроблений деінде (ChatGPT, DeepL, живий перекладач). Він накладеться на наявні сегменти — таймінги й початкова транскрипція лишаться незмінними.", diff --git a/electron/src/renderer/src/i18n/locales/vi.json b/electron/src/renderer/src/i18n/locales/vi.json index 4887838e1..4e2bbbc8e 100644 --- a/electron/src/renderer/src/i18n/locales/vi.json +++ b/electron/src/renderer/src/i18n/locales/vi.json @@ -813,6 +813,7 @@ "qc_failed": "Kiểm tra thời gian không thành công: {{message}}", "qc_track_changed": "Bản lồng tiếng đã thay đổi trong khi kiểm tra chất lượng. Hãy kiểm tra lại bản âm thanh hiện tại.", "qc_identity_missing": "Không thể xác định rõ từng đoạn. Hãy tạo lại bản âm thanh này với mã định danh riêng cho mỗi đoạn.", + "source_changed": "Phụ đề đã thay đổi trong quá trình tạo. Hãy tạo lại để sử dụng phụ đề hiện tại.", "paste_translation_btn": "Dán bản dịch", "paste_translation_title": "Dán một bản dịch", "paste_translation_desc": "Dán bản dịch bạn đã làm ở nơi khác (ChatGPT, DeepL, người dịch). Nó sẽ được ánh xạ vào các phân đoạn sẵn có — thời điểm và bản ghi gốc giữ nguyên.", diff --git a/electron/src/renderer/src/i18n/locales/zh-CN.json b/electron/src/renderer/src/i18n/locales/zh-CN.json index 3b3b8f604..6b3e31058 100644 --- a/electron/src/renderer/src/i18n/locales/zh-CN.json +++ b/electron/src/renderer/src/i18n/locales/zh-CN.json @@ -838,6 +838,7 @@ "qc_failed": "时序检查失败:{{message}}", "qc_track_changed": "配音在质量检查期间发生了变化。请重新检查当前音轨。", "qc_identity_missing": "无法唯一识别各个片段。请使用唯一的片段 ID 重新生成此音轨。", + "source_changed": "生成过程中字幕已更改。请重新生成以使用当前字幕。", "paste_translation_btn": "粘贴译文", "paste_translation_title": "粘贴译文", "paste_translation_desc": "粘贴你在别处完成的译文(ChatGPT、DeepL 或人工译者)。它会映射到已有的片段上——时间轴和原始转写保持不变。", diff --git a/electron/src/renderer/src/i18n/locales/zh-TW.json b/electron/src/renderer/src/i18n/locales/zh-TW.json index 404ebe73b..c55ed5f2f 100644 --- a/electron/src/renderer/src/i18n/locales/zh-TW.json +++ b/electron/src/renderer/src/i18n/locales/zh-TW.json @@ -813,6 +813,7 @@ "qc_failed": "計時檢查失敗:{{message}}", "qc_track_changed": "配音在品質檢查期間發生了變更。請重新檢查目前的音軌。", "qc_identity_missing": "無法唯一識別各個片段。請使用唯一的片段 ID 重新產生此音軌。", + "source_changed": "生成過程中字幕已變更。請重新生成以使用目前的字幕。", "paste_translation_btn": "貼上譯文", "paste_translation_title": "貼上譯文", "paste_translation_desc": "貼上你在別處完成的譯文(ChatGPT、DeepL 或真人譯者)。它會對應到既有的片段上——時間軸與原始逐字稿維持不變。", diff --git a/electron/src/renderer/src/lib/api/failure.test.ts b/electron/src/renderer/src/lib/api/failure.test.ts index 1bb774bfd..c317e3a96 100644 --- a/electron/src/renderer/src/lib/api/failure.test.ts +++ b/electron/src/renderer/src/lib/api/failure.test.ts @@ -14,3 +14,12 @@ it('localizes a segment identity conflict delivered after generation starts', () expect(failure.reason).toBe('Localized identity conflict'); expect(translate).toHaveBeenCalledWith('dub.qc_identity_missing'); }); + +it('localizes a source edit that interrupts dub publication', () => { + const translate = vi.spyOn(i18next, 't').mockReturnValue('Localized subtitle change'); + const failure = publicFailureFromEvent({ + type: 'error', error_code: 'dub_source_changed', error: 'Server fallback', + }, 'Task failed'); + expect(failure.reason).toBe('Localized subtitle change'); + expect(translate).toHaveBeenCalledWith('dub.source_changed'); +}); diff --git a/electron/src/renderer/src/lib/api/failure.ts b/electron/src/renderer/src/lib/api/failure.ts index d8a54a31a..081c688ad 100644 --- a/electron/src/renderer/src/lib/api/failure.ts +++ b/electron/src/renderer/src/lib/api/failure.ts @@ -19,13 +19,15 @@ export function publicFailureFromEvent( const localized = generationFailureMessage(event, i18next.t); return { reason: - (event.error_code === 'dub_segment_identity_conflict' - ? i18next.t('dub.qc_identity_missing') - : event.error_code === 'dub_speech_missing' - ? i18next.t('dubIntegrity.missingSpeech') - : event.error_code === 'dub_timing_overflow' - ? i18next.t('dubIntegrity.timingOverflow') - : undefined) || + (event.error_code === 'dub_source_changed' + ? i18next.t('dub.source_changed') + : event.error_code === 'dub_segment_identity_conflict' + ? i18next.t('dub.qc_identity_missing') + : event.error_code === 'dub_speech_missing' + ? i18next.t('dubIntegrity.missingSpeech') + : event.error_code === 'dub_timing_overflow' + ? i18next.t('dubIntegrity.timingOverflow') + : undefined) || localized || text(event.reason) || text(event.detail) || diff --git a/electron/src/shared/i18n/locales/ar.json b/electron/src/shared/i18n/locales/ar.json index 69e115710..aec87306c 100644 --- a/electron/src/shared/i18n/locales/ar.json +++ b/electron/src/shared/i18n/locales/ar.json @@ -1329,6 +1329,7 @@ "qc_failed": "فشل التحقق من التوقيت: {{message}}", "qc_track_changed": "تغيرت الدبلجة أثناء فحص الجودة. أعد الفحص على المسار الحالي.", "qc_identity_missing": "هويات المقاطع غير واضحة. أعد إنشاء هذا المسار بمعرّفات فريدة للمقاطع.", + "source_changed": "تغيّرت الترجمة أثناء التوليد. أعد التوليد لاستخدام الترجمة الحالية.", "paste_translation_btn": "لصق ترجمة", "paste_translation_title": "لصق ترجمة", "paste_translation_desc": "الصق ترجمة أعددتها في مكان آخر (ChatGPT أو DeepL أو مترجم بشري). ستُطابَق مع المقاطع الموجودة لديك — تبقى التوقيتات والنص الأصلي دون تغيير.", diff --git a/electron/src/shared/i18n/locales/de.json b/electron/src/shared/i18n/locales/de.json index 0f1403301..af3a4d46b 100644 --- a/electron/src/shared/i18n/locales/de.json +++ b/electron/src/shared/i18n/locales/de.json @@ -1327,6 +1327,7 @@ "qc_failed": "Zeitprüfung fehlgeschlagen: {{message}}", "qc_track_changed": "Die Synchronisation wurde während der Qualitätsprüfung geändert. Prüfe die aktuelle Spur erneut.", "qc_identity_missing": "Die Segmentzuordnung ist uneindeutig. Erzeuge diese Spur mit eindeutigen Segment-IDs neu.", + "source_changed": "Die Untertitel wurden während der Generierung geändert. Generiere erneut, um die aktuellen Untertitel zu verwenden.", "paste_translation_btn": "Übersetzung einfügen", "paste_translation_title": "Übersetzung einfügen", "paste_translation_desc": "Füge eine anderswo erstellte Übersetzung ein (ChatGPT, DeepL, ein menschlicher Übersetzer). Sie wird den vorhandenen Segmenten zugeordnet — Timings und Originaltranskript bleiben unverändert.", diff --git a/electron/src/shared/i18n/locales/en.json b/electron/src/shared/i18n/locales/en.json index d78b9e758..f0b069457 100644 --- a/electron/src/shared/i18n/locales/en.json +++ b/electron/src/shared/i18n/locales/en.json @@ -1542,6 +1542,7 @@ "qc_failed": "Timing check failed: {{message}}", "qc_track_changed": "The dub changed during the quality check. Run the check again on the current track.", "qc_identity_missing": "Segment identities are ambiguous. Regenerate this track with unique segment IDs.", + "source_changed": "Subtitles changed during generation. Generate again to use the current subtitles.", "export_btn": "Export…", "prep_stop": "Stop", "install_progress": "Installing {{engine}}…", diff --git a/electron/src/shared/i18n/locales/es.json b/electron/src/shared/i18n/locales/es.json index d9b754dc9..f7d89d85f 100644 --- a/electron/src/shared/i18n/locales/es.json +++ b/electron/src/shared/i18n/locales/es.json @@ -1327,6 +1327,7 @@ "qc_failed": "Error en la verificación de tiempo: {{message}}", "qc_track_changed": "El doblaje cambió durante la comprobación de calidad. Repite la comprobación en la pista actual.", "qc_identity_missing": "La identidad de los segmentos es ambigua. Regenera esta pista con identificadores de segmento únicos.", + "source_changed": "Los subtítulos cambiaron durante la generación. Genera de nuevo para usar los subtítulos actuales.", "paste_translation_btn": "Pegar traducción", "paste_translation_title": "Pegar una traducción", "paste_translation_desc": "Pega una traducción hecha en otro sitio (ChatGPT, DeepL, un traductor humano). Se asigna a los segmentos que ya tienes: los tiempos y la transcripción original no cambian.", diff --git a/electron/src/shared/i18n/locales/fr.json b/electron/src/shared/i18n/locales/fr.json index 9bb824f66..2cb03c2e7 100644 --- a/electron/src/shared/i18n/locales/fr.json +++ b/electron/src/shared/i18n/locales/fr.json @@ -1327,6 +1327,7 @@ "qc_failed": "Échec de la vérification du timing : {{message}}", "qc_track_changed": "Le doublage a changé pendant le contrôle qualité. Relancez le contrôle sur la piste actuelle.", "qc_identity_missing": "Les segments ne sont pas identifiés de façon univoque. Régénérez cette piste avec des identifiants de segment uniques.", + "source_changed": "Les sous-titres ont changé pendant la génération. Relancez la génération pour utiliser les sous-titres actuels.", "paste_translation_btn": "Coller une traduction", "paste_translation_title": "Coller une traduction", "paste_translation_desc": "Collez une traduction réalisée ailleurs (ChatGPT, DeepL, un traducteur humain). Elle est appliquée aux segments existants — les timings et la transcription d'origine restent intacts.", diff --git a/electron/src/shared/i18n/locales/hi.json b/electron/src/shared/i18n/locales/hi.json index f5cc74b68..c7f6dc701 100644 --- a/electron/src/shared/i18n/locales/hi.json +++ b/electron/src/shared/i18n/locales/hi.json @@ -1327,6 +1327,7 @@ "qc_failed": "समय की जाँच विफल: {{message}}", "qc_track_changed": "गुणवत्ता जाँच के दौरान डबिंग बदल गई। मौजूदा ट्रैक पर जाँच दोबारा चलाएँ।", "qc_identity_missing": "सेगमेंट की पहचान स्पष्ट नहीं है। हर सेगमेंट को अलग आईडी देकर इस ट्रैक को फिर से जनरेट करें।", + "source_changed": "जनरेशन के दौरान सबटाइटल बदल गए। मौजूदा सबटाइटल इस्तेमाल करने के लिए फिर से जनरेट करें।", "paste_translation_btn": "अनुवाद चिपकाएँ", "paste_translation_title": "अनुवाद चिपकाएँ", "paste_translation_desc": "कहीं और तैयार किया गया अनुवाद चिपकाएँ (ChatGPT, DeepL, कोई मानव अनुवादक)। यह आपके मौजूदा सेगमेंट पर लागू होगा — टाइमिंग और मूल ट्रांसक्रिप्ट अछूते रहते हैं।", diff --git a/electron/src/shared/i18n/locales/id.json b/electron/src/shared/i18n/locales/id.json index 6bbfded69..92e6d45cd 100644 --- a/electron/src/shared/i18n/locales/id.json +++ b/electron/src/shared/i18n/locales/id.json @@ -1329,6 +1329,7 @@ "qc_failed": "Pemeriksaan waktu gagal: {{message}}", "qc_track_changed": "Sulih suara berubah selama pemeriksaan kualitas. Jalankan kembali pemeriksaan pada trek saat ini.", "qc_identity_missing": "Identitas segmen tidak jelas. Buat ulang trek ini dengan ID segmen yang unik.", + "source_changed": "Subtitel berubah selama pembuatan. Buat ulang untuk menggunakan subtitel saat ini.", "paste_translation_btn": "Tempel terjemahan", "paste_translation_title": "Tempel terjemahan", "paste_translation_desc": "Tempel terjemahan yang kamu buat di tempat lain (ChatGPT, DeepL, penerjemah manusia). Terjemahan itu dipetakan ke segmen yang sudah ada — pewaktuan dan transkrip asli tidak diubah.", diff --git a/electron/src/shared/i18n/locales/it.json b/electron/src/shared/i18n/locales/it.json index 8fac3eaed..665a7b868 100644 --- a/electron/src/shared/i18n/locales/it.json +++ b/electron/src/shared/i18n/locales/it.json @@ -1327,6 +1327,7 @@ "qc_failed": "Controllo cronometraggio fallito: {{message}}", "qc_track_changed": "Il doppiaggio è cambiato durante il controllo qualità. Ripeti il controllo sulla traccia attuale.", "qc_identity_missing": "Le identità dei segmenti sono ambigue. Rigenera questa traccia con identificativi di segmento univoci.", + "source_changed": "I sottotitoli sono cambiati durante la generazione. Genera di nuovo per usare i sottotitoli attuali.", "paste_translation_btn": "Incolla traduzione", "paste_translation_title": "Incolla una traduzione", "paste_translation_desc": "Incolla una traduzione prodotta altrove (ChatGPT, DeepL, un traduttore umano). Viene mappata sui segmenti che hai già: tempi e trascrizione originale restano invariati.", diff --git a/electron/src/shared/i18n/locales/ja.json b/electron/src/shared/i18n/locales/ja.json index 4eab222fe..92abded17 100644 --- a/electron/src/shared/i18n/locales/ja.json +++ b/electron/src/shared/i18n/locales/ja.json @@ -1329,6 +1329,7 @@ "qc_failed": "タイミング チェックに失敗しました: {{message}}", "qc_track_changed": "品質チェック中に吹き替えが変更されました。現在のトラックで再度チェックしてください。", "qc_identity_missing": "セグメントを一意に識別できません。各セグメントに固有のIDを付けて、このトラックを再生成してください。", + "source_changed": "生成中に字幕が変更されました。現在の字幕を使用するには、もう一度生成してください。", "paste_translation_btn": "翻訳を貼り付け", "paste_translation_title": "翻訳を貼り付け", "paste_translation_desc": "他所で用意した翻訳(ChatGPT、DeepL、人間の翻訳者)を貼り付けます。既存のセグメントに割り当てられ、タイミングと元の文字起こしはそのまま残ります。", diff --git a/electron/src/shared/i18n/locales/ko.json b/electron/src/shared/i18n/locales/ko.json index 190c2d426..3ff65344a 100644 --- a/electron/src/shared/i18n/locales/ko.json +++ b/electron/src/shared/i18n/locales/ko.json @@ -1618,6 +1618,7 @@ "qc_failed": "타이밍 확인 실패: {{message}}", "qc_track_changed": "품질 검사 중 더빙이 변경되었습니다. 현재 트랙에서 검사를 다시 실행하세요.", "qc_identity_missing": "세그먼트를 명확하게 구분할 수 없습니다. 각 세그먼트에 고유한 ID를 지정하여 이 트랙을 다시 생성하세요.", + "source_changed": "생성 중에 자막이 변경되었습니다. 현재 자막을 사용하려면 다시 생성하세요.", "paste_translation_btn": "번역 붙여넣기", "paste_translation_title": "번역 붙여넣기", "paste_translation_desc": "다른 곳에서 만든 번역(ChatGPT, DeepL, 사람 번역가)을 붙여넣으세요. 이미 있는 세그먼트에 매핑되며 타이밍과 원본 전사는 그대로 유지됩니다.", diff --git a/electron/src/shared/i18n/locales/nl.json b/electron/src/shared/i18n/locales/nl.json index 2054a81f0..84387bf59 100644 --- a/electron/src/shared/i18n/locales/nl.json +++ b/electron/src/shared/i18n/locales/nl.json @@ -1327,6 +1327,7 @@ "qc_failed": "Timingcontrole mislukt: {{message}}", "qc_track_changed": "De nasynchronisatie is tijdens de kwaliteitscontrole gewijzigd. Voer de controle opnieuw uit op het huidige spoor.", "qc_identity_missing": "De segmenten zijn niet eenduidig te identificeren. Genereer dit spoor opnieuw met unieke segment-ID’s.", + "source_changed": "De ondertitels zijn gewijzigd tijdens het genereren. Genereer opnieuw om de huidige ondertitels te gebruiken.", "paste_translation_btn": "Vertaling plakken", "paste_translation_title": "Een vertaling plakken", "paste_translation_desc": "Plak een elders gemaakte vertaling (ChatGPT, DeepL, een menselijke vertaler). Die wordt op je bestaande segmenten toegepast — timings en het originele transcript blijven ongewijzigd.", diff --git a/electron/src/shared/i18n/locales/pl.json b/electron/src/shared/i18n/locales/pl.json index dc729d5e6..947b34d54 100644 --- a/electron/src/shared/i18n/locales/pl.json +++ b/electron/src/shared/i18n/locales/pl.json @@ -1327,6 +1327,7 @@ "qc_failed": "Kontrola czasu nie powiodła się: {{message}}", "qc_track_changed": "Dubbing zmienił się podczas kontroli jakości. Uruchom kontrolę ponownie dla bieżącej ścieżki.", "qc_identity_missing": "Tożsamość segmentów jest niejednoznaczna. Wygeneruj tę ścieżkę ponownie z unikatowymi identyfikatorami segmentów.", + "source_changed": "Napisy zmieniły się podczas generowania. Wygeneruj ponownie, aby użyć aktualnych napisów.", "paste_translation_btn": "Wklej tłumaczenie", "paste_translation_title": "Wklej tłumaczenie", "paste_translation_desc": "Wklej tłumaczenie przygotowane gdzie indziej (ChatGPT, DeepL, tłumacz). Zostanie dopasowane do istniejących segmentów — czasy i oryginalna transkrypcja pozostają bez zmian.", diff --git a/electron/src/shared/i18n/locales/pt.json b/electron/src/shared/i18n/locales/pt.json index 405202d1f..89b57e4f1 100644 --- a/electron/src/shared/i18n/locales/pt.json +++ b/electron/src/shared/i18n/locales/pt.json @@ -1327,6 +1327,7 @@ "qc_failed": "Falha na verificação de tempo: {{message}}", "qc_track_changed": "A dublagem mudou durante a verificação de qualidade. Execute a verificação novamente na faixa atual.", "qc_identity_missing": "A identificação dos segmentos é ambígua. Gere esta faixa novamente com identificadores de segmento exclusivos.", + "source_changed": "As legendas mudaram durante a geração. Gere novamente para usar as legendas atuais.", "paste_translation_btn": "Colar tradução", "paste_translation_title": "Colar uma tradução", "paste_translation_desc": "Cole uma tradução feita noutro lugar (ChatGPT, DeepL, um tradutor humano). Ela é mapeada nos segmentos que já existem — os tempos e a transcrição original ficam intactos.", diff --git a/electron/src/shared/i18n/locales/ru.json b/electron/src/shared/i18n/locales/ru.json index acda02ecf..476e70bdf 100644 --- a/electron/src/shared/i18n/locales/ru.json +++ b/electron/src/shared/i18n/locales/ru.json @@ -1327,6 +1327,7 @@ "qc_failed": "Проверка времени не удалась: {{message}}", "qc_track_changed": "Дубляж изменился во время проверки качества. Повторите проверку текущей дорожки.", "qc_identity_missing": "Сегменты невозможно однозначно определить. Создайте эту дорожку заново с уникальными идентификаторами сегментов.", + "source_changed": "Субтитры изменились во время генерации. Запустите генерацию заново, чтобы использовать текущие субтитры.", "paste_translation_btn": "Вставить перевод", "paste_translation_title": "Вставить перевод", "paste_translation_desc": "Вставьте перевод, сделанный в другом месте (ChatGPT, DeepL, живой переводчик). Он ляжет на уже имеющиеся сегменты — тайминги и исходная расшифровка не изменятся.", diff --git a/electron/src/shared/i18n/locales/sv.json b/electron/src/shared/i18n/locales/sv.json index bc4de5684..c04334388 100644 --- a/electron/src/shared/i18n/locales/sv.json +++ b/electron/src/shared/i18n/locales/sv.json @@ -1329,6 +1329,7 @@ "qc_failed": "Tidskontroll misslyckades: {{message}}", "qc_track_changed": "Dubbningen ändrades under kvalitetskontrollen. Kör kontrollen igen på det aktuella spåret.", "qc_identity_missing": "Segmenten kan inte identifieras entydigt. Generera om spåret med unika segment-ID:n.", + "source_changed": "Undertexterna ändrades under genereringen. Generera igen för att använda de aktuella undertexterna.", "paste_translation_btn": "Klistra in översättning", "paste_translation_title": "Klistra in en översättning", "paste_translation_desc": "Klistra in en översättning som gjorts någon annanstans (ChatGPT, DeepL, en mänsklig översättare). Den mappas mot segmenten du redan har — tidkoder och originaltranskriptet rörs inte.", diff --git a/electron/src/shared/i18n/locales/th.json b/electron/src/shared/i18n/locales/th.json index f8ec751ca..d906cce99 100644 --- a/electron/src/shared/i18n/locales/th.json +++ b/electron/src/shared/i18n/locales/th.json @@ -1329,6 +1329,7 @@ "qc_failed": "การตรวจสอบเวลาล้มเหลว: {{message}}", "qc_track_changed": "เสียงพากย์เปลี่ยนแปลงระหว่างการตรวจสอบคุณภาพ โปรดตรวจสอบแทร็กปัจจุบันอีกครั้ง", "qc_identity_missing": "ไม่สามารถระบุแต่ละช่วงได้อย่างชัดเจน โปรดสร้างแทร็กนี้ใหม่โดยใช้รหัสที่ไม่ซ้ำกันสำหรับแต่ละช่วง", + "source_changed": "คำบรรยายเปลี่ยนแปลงระหว่างการสร้าง โปรดสร้างอีกครั้งเพื่อใช้คำบรรยายปัจจุบัน", "paste_translation_btn": "วางคำแปล", "paste_translation_title": "วางคำแปล", "paste_translation_desc": "วางคำแปลที่คุณทำไว้จากที่อื่น (ChatGPT, DeepL หรือผู้แปลที่เป็นคน) ระบบจะจับคู่กับเซกเมนต์ที่มีอยู่แล้ว โดยเวลาและถอดความต้นฉบับยังคงเดิม", diff --git a/electron/src/shared/i18n/locales/tr.json b/electron/src/shared/i18n/locales/tr.json index aca1d35d7..b249090ce 100644 --- a/electron/src/shared/i18n/locales/tr.json +++ b/electron/src/shared/i18n/locales/tr.json @@ -1329,6 +1329,7 @@ "qc_failed": "Zamanlama kontrolü başarısız oldu: {{message}}", "qc_track_changed": "Kalite kontrolü sırasında dublaj değişti. Geçerli parçada kontrolü yeniden çalıştırın.", "qc_identity_missing": "Bölümler kesin olarak tanımlanamıyor. Bu parçayı benzersiz bölüm kimlikleriyle yeniden oluşturun.", + "source_changed": "Oluşturma sırasında altyazılar değişti. Güncel altyazıları kullanmak için yeniden oluşturun.", "paste_translation_btn": "Çeviriyi yapıştır", "paste_translation_title": "Bir çeviri yapıştır", "paste_translation_desc": "Başka bir yerde hazırladığın çeviriyi yapıştır (ChatGPT, DeepL, bir insan çevirmen). Mevcut segmentlere eşlenir — zamanlamalar ve özgün döküm olduğu gibi kalır.", diff --git a/electron/src/shared/i18n/locales/uk.json b/electron/src/shared/i18n/locales/uk.json index c268f6e6b..bb5b4c3be 100644 --- a/electron/src/shared/i18n/locales/uk.json +++ b/electron/src/shared/i18n/locales/uk.json @@ -1329,6 +1329,7 @@ "qc_failed": "Помилка перевірки часу: {{message}}", "qc_track_changed": "Дубляж змінився під час перевірки якості. Повторіть перевірку поточної доріжки.", "qc_identity_missing": "Сегменти неможливо однозначно визначити. Створіть цю доріжку заново з унікальними ідентифікаторами сегментів.", + "source_changed": "Субтитри змінилися під час генерації. Запустіть генерацію знову, щоб використати поточні субтитри.", "paste_translation_btn": "Вставити переклад", "paste_translation_title": "Вставити переклад", "paste_translation_desc": "Вставте переклад, зроблений деінде (ChatGPT, DeepL, живий перекладач). Він накладеться на наявні сегменти — таймінги й початкова транскрипція лишаться незмінними.", diff --git a/electron/src/shared/i18n/locales/vi.json b/electron/src/shared/i18n/locales/vi.json index 963c6dbdd..0e9f0bc61 100644 --- a/electron/src/shared/i18n/locales/vi.json +++ b/electron/src/shared/i18n/locales/vi.json @@ -1329,6 +1329,7 @@ "qc_failed": "Kiểm tra thời gian không thành công: {{message}}", "qc_track_changed": "Bản lồng tiếng đã thay đổi trong khi kiểm tra chất lượng. Hãy kiểm tra lại bản âm thanh hiện tại.", "qc_identity_missing": "Không thể xác định rõ từng đoạn. Hãy tạo lại bản âm thanh này với mã định danh riêng cho mỗi đoạn.", + "source_changed": "Phụ đề đã thay đổi trong quá trình tạo. Hãy tạo lại để sử dụng phụ đề hiện tại.", "paste_translation_btn": "Dán bản dịch", "paste_translation_title": "Dán một bản dịch", "paste_translation_desc": "Dán bản dịch bạn đã làm ở nơi khác (ChatGPT, DeepL, người dịch). Nó sẽ được ánh xạ vào các phân đoạn sẵn có — thời điểm và bản ghi gốc giữ nguyên.", diff --git a/electron/src/shared/i18n/locales/zh-CN.json b/electron/src/shared/i18n/locales/zh-CN.json index eda26df59..c4c15e24d 100644 --- a/electron/src/shared/i18n/locales/zh-CN.json +++ b/electron/src/shared/i18n/locales/zh-CN.json @@ -1580,6 +1580,7 @@ "qc_failed": "时序检查失败:{{message}}", "qc_track_changed": "配音在质量检查期间发生了变化。请重新检查当前音轨。", "qc_identity_missing": "无法唯一识别各个片段。请使用唯一的片段 ID 重新生成此音轨。", + "source_changed": "生成过程中字幕已更改。请重新生成以使用当前字幕。", "paste_translation_btn": "粘贴译文", "paste_translation_title": "粘贴译文", "paste_translation_desc": "粘贴你在别处完成的译文(ChatGPT、DeepL 或人工译者)。它会映射到已有的片段上——时间轴和原始转写保持不变。", diff --git a/electron/src/shared/i18n/locales/zh-TW.json b/electron/src/shared/i18n/locales/zh-TW.json index f0931b0e6..e33f5f182 100644 --- a/electron/src/shared/i18n/locales/zh-TW.json +++ b/electron/src/shared/i18n/locales/zh-TW.json @@ -1329,6 +1329,7 @@ "qc_failed": "計時檢查失敗:{{message}}", "qc_track_changed": "配音在品質檢查期間發生了變更。請重新檢查目前的音軌。", "qc_identity_missing": "無法唯一識別各個片段。請使用唯一的片段 ID 重新產生此音軌。", + "source_changed": "生成過程中字幕已變更。請重新生成以使用目前的字幕。", "paste_translation_btn": "貼上譯文", "paste_translation_title": "貼上譯文", "paste_translation_desc": "貼上你在別處完成的譯文(ChatGPT、DeepL 或真人譯者)。它會對應到既有的片段上——時間軸與原始逐字稿維持不變。", diff --git a/tests/test_dub_complete_audio.py b/tests/test_dub_complete_audio.py index ed9d44d1d..7525f1c1a 100644 --- a/tests/test_dub_complete_audio.py +++ b/tests/test_dub_complete_audio.py @@ -223,6 +223,10 @@ def test_identity_change_during_render_preserves_previous_publication(render_dub 'seg_hashes_by_lang': {'en': {'a': 'saved hash'}}, 'dubbed_tracks': {'en': {'path': str(previous)}}, }) + cached = [render_dub.path / f'seg_en_seg_{i}.wav' for i in range(2)] + for cache in cached: + sf.write(cache, [.2] * 24000, 24000) + cached_bytes = [cache.read_bytes() for cache in cached] saved = copy.deepcopy(render_dub.job) persisted = render_dub.path / 'job.json' persisted.write_text(json.dumps(saved)) @@ -239,7 +243,118 @@ def change_identity(): assert any(e['type'] == 'error' and e.get('error_code') == 'dub_segment_identity_conflict' for e in events) assert not any(e['type'] == 'done' for e in events) + assert [cache.read_bytes() for cache in cached] == cached_bytes + assert not list(render_dub.path.glob('.render-*')) assert previous.read_bytes() == b'previous complete track' assert json.loads(persisted.read_text()) == saved for key in ('seg_hashes', 'seg_hashes_by_lang', 'segments_i18n', 'dubbed_tracks'): assert render_dub.job[key] == saved[key] + + +def test_subtitle_import_during_assembly_keeps_edits_and_previous_track(render_dub, monkeypatch): + import io + from fastapi import UploadFile + from api.routers import dub_core, dub_generate as dg + + previous = render_dub.path / 'dubbed_en.wav' + previous.write_bytes(b'previous complete track') + render_dub.job.update({ + 'segments': [{'id': 'a', 'start': 0, 'end': 1, 'text': 'saved text'}], + 'seg_order': ['a'], + 'seg_hashes': {'a': 'saved hash'}, + 'seg_hashes_by_lang': {'en': {'a': 'saved hash'}}, + 'dubbed_tracks': {'en': {'path': str(previous)}}, + }) + saved_hashes = copy.deepcopy(render_dub.job['seg_hashes_by_lang']) + persisted = render_dub.path / 'job.json' + def persist(_, job): + persisted.write_text(json.dumps(job)) + persist('job', render_dub.job) + monkeypatch.setattr(dg, '_save_job', persist) + monkeypatch.setattr(dub_core, '_save_job', persist) + monkeypatch.setattr(dub_core, '_get_job', lambda _: render_dub.job) + imported = [] + async def stretch(wav, target, sr): + result = await dub_core.dub_import_srt('job', UploadFile( + filename='corrected.srt', + file=io.BytesIO(b'1\n00:00:00,000 --> 00:00:01,000\nCorrected subtitle\n'), + )) + imported.extend(copy.deepcopy(result['segments'])) + return torch.nn.functional.interpolate(wav.unsqueeze(0), size=target, mode='linear').squeeze(0) + monkeypatch.setattr(dg, '_pitch_preserving_stretch', stretch) + render_dub.output[0] = lambda: torch.ones(1, 48000) * .1 + events = render_dub.run(timing_strategy='strict_slot') + assert imported + assert any(e['type'] == 'error' and e.get('error_code') == 'dub_source_changed' for e in events) + assert not any(e['type'] == 'done' for e in events) + assert render_dub.job['segments'] == imported + assert render_dub.job['seg_hashes_by_lang'] == saved_hashes + assert previous.read_bytes() == b'previous complete track' + assert json.loads(persisted.read_text()) == render_dub.job + + +def test_empty_dub_request_rejected_before_backend_resolution(monkeypatch): + from fastapi import FastAPI + from fastapi.testclient import TestClient + from api.routers import dub_generate as dg + + async def forbidden(): + pytest.fail('Empty requests must not load the synthesis backend') + monkeypatch.setattr(dg, '_resolve_dub_execution', forbidden) + monkeypatch.setattr(dg, '_get_job', lambda _: {'segments': []}) + app = FastAPI() + app.include_router(dg.router) + with TestClient(app) as client: + response = client.post('/dub/generate/job', json={'segments': []}) + assert response.status_code == 422 + assert any(error['loc'] == ['body', 'segments'] for error in response.json()['detail']) + + +@pytest.mark.parametrize('failure', ['install', 'persist']) +def test_failed_publication_restores_track_and_segment_cache(render_dub, monkeypatch, failure): + from api.routers import dub_generate as dg + + previous = render_dub.path / 'dubbed_en.wav' + previous.write_bytes(b'previous complete track') + cache = render_dub.path / 'seg_en_a.wav' + sf.write(cache, [.2] * 24000, 24000) + cache_bytes = cache.read_bytes() + render_dub.job['dubbed_tracks']['en'] = {'path': str(previous)} + original = copy.deepcopy(render_dub.job) + if failure == 'install': + replace = dg.os.replace + def fail_track_install(source, destination): + if str(destination) == str(previous): + raise OSError('injected track installation failure') + return replace(source, destination) + monkeypatch.setattr(dg.os, 'replace', fail_track_install) + else: + def fail_save(*_): + raise OSError('injected persistence failure') + monkeypatch.setattr(dg, '_save_job', fail_save) + events = render_dub.run() + assert any(e['type'] == 'error' for e in events) + assert not any(e['type'] == 'done' for e in events) + assert previous.read_bytes() == b'previous complete track' + assert cache.read_bytes() == cache_bytes + assert render_dub.job == original + assert not list(render_dub.path.glob('.render-*')) + + +def test_cancelled_render_discards_staged_cache(render_dub, monkeypatch): + from api.routers import dub_generate as dg + + cache = render_dub.path / 'seg_en_a.wav' + sf.write(cache, [.2] * 24000, 24000) + cache_bytes = cache.read_bytes() + calls = [] + def cancelled(_): + calls.append(True) + return len(calls) >= 3 # after the first segment, before the second + monkeypatch.setattr(dg.task_manager, 'is_cancelled', cancelled) + events = render_dub.run(segments=[dict(start=0, end=1, text='first'), + dict(start=1, end=2, text='second')], segment_ids=['a', 'b']) + assert any(e['type'] == 'cancelled' for e in events) + assert cache.read_bytes() == cache_bytes + assert not (render_dub.path / 'seg_en_b.wav').exists() + assert not list(render_dub.path.glob('.render-*')) diff --git a/tests/test_dub_subtitles_309.py b/tests/test_dub_subtitles_309.py index f63a5b9b2..6cb596e7f 100644 --- a/tests/test_dub_subtitles_309.py +++ b/tests/test_dub_subtitles_309.py @@ -130,7 +130,7 @@ def test_empty_request_keeps_existing_segments(self): from api.routers.dub_generate import _sync_job_segments from schemas.requests import DubRequest job = {"segments": [{"id": "a1", "start": 0.0, "end": 1.0, "text": "keep me"}]} - _sync_job_segments(job, DubRequest(segments=[])) + _sync_job_segments(job, DubRequest.model_construct(segments=[])) assert job["segments"][0]["text"] == "keep me" def test_request_timing_wins(self): diff --git a/tests/test_router_smoke.py b/tests/test_router_smoke.py index 13b6a2a1d..8f6a6947c 100644 --- a/tests/test_router_smoke.py +++ b/tests/test_router_smoke.py @@ -255,7 +255,7 @@ def test_dub_generate_unknown_job(client): # Hitting /dub/generate/{id} with a non-existent id should surface the # rewritten 404 copy. r = client.post("/dub/generate/__nonexistent__", json={ - "segments": [], + "segments": [{"start": 0, "end": 1, "text": "hello"}], "language": "Auto", "language_code": "und", "num_step": 16, diff --git a/tests/test_task_stream_failure.py b/tests/test_task_stream_failure.py index 36c2afa59..fe8317601 100644 --- a/tests/test_task_stream_failure.py +++ b/tests/test_task_stream_failure.py @@ -40,3 +40,34 @@ async def run(): def test_warnings_remain_non_terminal(): assert _stream_failure('data: {"type":"warning","error":"retrying"}\n\n') is None + + +def test_cancelled_stream_closes_before_worker_waits_for_next_task(monkeypatch): + job_store = TaskManager.worker.__globals__["job_store"] + run_sentinel = TaskManager.worker.__globals__["run_sentinel"] + for name in ['create', 'mark_running', 'mark_cancelled', 'append_event']: + monkeypatch.setattr(job_store, name, lambda *a, **kw: None) + monkeypatch.setattr(run_sentinel, 'touch_activity', lambda *a: None) + closed = [] + + async def run(): + manager = TaskManager() + async def stream(): + try: + manager.cancel_task('test') + yield 'data: {"type":"progress"}\n\n' + finally: + closed.append(True) + await manager.add_task('test', 'dub_generate', stream) + worker = asyncio.create_task(manager.worker()) + try: + await asyncio.wait_for(manager.queue.join(), 2) + assert manager.active_tasks['test']['status'] == 'cancelled' + assert closed == [True] + finally: + worker.cancel() + try: + await worker + except asyncio.CancelledError: + pass + asyncio.run(run()) From 78bde527a68097fc0992cbc593649621587d03ca Mon Sep 17 00:00:00 2001 From: debpalash <4178343+debpalash@users.noreply.github.com> Date: Sat, 3 Oct 2026 04:50:39 +0530 Subject: [PATCH 04/10] fix(dub): preserve QC annotations during render publication --- CHANGELOG.md | 1 + backend/api/routers/dub_generate.py | 22 ++++++++++++++++------ docs/electron-dubbing.md | 11 +++++++++++ tests/test_dub_complete_audio.py | 18 ++++++++++++++++++ 4 files changed, 46 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 551d2e745..78141a5b1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -87,6 +87,7 @@ metadata and the backend fallback mirror it. - The Twilio guide and integration directory describe the guided setup and in-app integration pages (#2304) ### Fixed +- Dubbing finishes when quality-check annotations arrive during assembly, while preserving those annotations and still protecting subtitle edits (#2585) - Saving a voice design skips cold engine loading and downloads, including when a warm engine unloads during the save (#2583) — thanks @simoncheese! diff --git a/backend/api/routers/dub_generate.py b/backend/api/routers/dub_generate.py index 626486b4c..4f0fa0103 100644 --- a/backend/api/routers/dub_generate.py +++ b/backend/api/routers/dub_generate.py @@ -210,6 +210,16 @@ def _underrun_min_rate() -> float: GAP_OVERFLOW_BUFFER_S = 0.05 +def _render_source_segments(job: dict): + """Snapshot source edits without treating derived QC annotations as edits.""" + segments = job.get("segments") + if segments is None: + return None + return [{key: copy.deepcopy(value) for key, value in row.items() + if not key.startswith("qc_")} if isinstance(row, dict) else copy.deepcopy(row) + for row in segments] + + def _track_source_segments(job: dict) -> list[dict]: """Snapshot source times by the identities persisted for this render.""" return [{"id": seg.get("id"), "start": seg["start"], "end": seg["end"]} @@ -625,7 +635,7 @@ async def dub_generate(job_id: str, req: DubRequest): # Subtitle imports can replace the source while synthesis or fitting awaits. # Keep the admission snapshot so publishing cannot overwrite those edits. - source_segments = copy.deepcopy(job.get("segments")) + source_segments = _render_source_segments(job) # ── Engine resolution (issue #312 class) ──────────────────────────────── # Every rendered segment clones either source speech or a saved profile, so @@ -2021,7 +2031,7 @@ def _render_batch() -> list[torch.Tensor]: # No await separates this final check from track/metadata publication. # In particular, a subtitle import during asynchronous fitting wins. - if _get_job(job_id) is not job or job.get("segments") != source_segments: + if _get_job(job_id) is not job or _render_source_segments(job) != source_segments: yield f"data: {json.dumps({'type': 'error', 'error_code': 'dub_source_changed', 'error': 'Subtitles changed during generation. Generate again to use the current subtitles.'})}\n\n" return _t_save_0 = time.perf_counter() @@ -2062,7 +2072,7 @@ def _render_batch() -> list[torch.Tensor]: track_dur = total_samples / sr if total_samples > 0 else 0.0 def publish(): with dub_pipeline._dub_jobs_lock: - if _get_job(job_id) is not job or job.get("segments") != source_segments: + if _get_job(job_id) is not job or _render_source_segments(job) != source_segments: return False published_job = copy.deepcopy(job) artifacts = {**staged_segments, track_path: staged_track} @@ -2093,9 +2103,9 @@ def publish(): published_job["seg_order"] = list(expected_order) # Publish the already validated snapshot with the completed track. # Preserve other languages that may have been added during rendering. - published_job["segments"] = publication["segments"] - for key in ("segments_i18n", "segments_i18n_cue_sources"): - published_job.setdefault(key, {}).update(publication[key]) + # Rebuild from the current snapshot under the lock so QC + # annotations completed during assembly are not overwritten. + _sync_job_segments(published_job, req) published_job["dubbed_tracks"][lang_code] = { "path": track_path, "language": req.language, diff --git a/docs/electron-dubbing.md b/docs/electron-dubbing.md index ba16857d7..6c45d36d2 100644 --- a/docs/electron-dubbing.md +++ b/docs/electron-dubbing.md @@ -411,3 +411,14 @@ asks you to run it again instead of publishing stale scores. Deleting the job during QC also discards the result and keeps it out of history. Generation revalidates its segment text and identity snapshot after synthesis, before publishing fingerprints or replacing the previous track. An identity collision caused by a concurrent edit is reported through the task stream; the previous track and published metadata remain available. The source snapshot is checked again after asynchronous fitting and before replacing audio: subtitle imports made during generation or assembly survive, and the user can regenerate from the corrected subtitles. Fresh segment WAVs and the assembled track stay in a private staging directory until this check passes. Rejection or cancellation removes the staging files and preserves the reusable segment cache. Publication backs up existing files and rolls back ordinary installation failures; it does not promise a multi-file transaction across power loss. Fingerprints and the segment manifest publish only with a completed track. Empty generation requests are rejected before loading a voice engine. + +QC annotations arriving during generation do not count as source edits: a completed +render can publish while retaining those annotations. Source text, timing, identity, +voice bindings, and imported-cue changes still invalidate the admitted snapshot. + +Cancelling or failing a render preserves the previous committed track and its +segment cache. Newly synthesized, unpublished segments are discarded and must be +synthesized again on retry. Resume reconnects to an existing running task; it does +not recover a cancelled task's unpublished speech. Reusing that speech would require +a separate resume cache with validated engine and reference revisions, rather than +replacing the committed cache with partial output. diff --git a/tests/test_dub_complete_audio.py b/tests/test_dub_complete_audio.py index 7525f1c1a..a4e5576c1 100644 --- a/tests/test_dub_complete_audio.py +++ b/tests/test_dub_complete_audio.py @@ -358,3 +358,21 @@ def cancelled(_): assert cache.read_bytes() == cache_bytes assert not (render_dub.path / 'seg_en_b.wav').exists() assert not list(render_dub.path.glob('.render-*')) + + +def test_qc_annotations_during_assembly_do_not_discard_render(render_dub, monkeypatch): + from api.routers import dub_generate as dg + + render_dub.job['segments'] = [{'id': 'a', 'start': 0, 'end': 1, 'text': 'original'}] + async def stretch(wav, target, sr): + render_dub.job['segments'][0].update(qc_drift=0.2, qc_flagged=True, + qc_recognized='measured speech', + qc_measured_start=0.1, qc_measured_end=0.9) + return torch.nn.functional.interpolate(wav.unsqueeze(0), size=target, mode='linear').squeeze(0) + monkeypatch.setattr(dg, '_pitch_preserving_stretch', stretch) + render_dub.output[0] = lambda: torch.ones(1, 48000) * .1 + events = render_dub.run(timing_strategy='strict_slot') + assert any(e['type'] == 'done' for e in events) + assert not any(e.get('error_code') == 'dub_source_changed' for e in events) + assert render_dub.job['segments'][0]['qc_recognized'] == 'measured speech' + assert render_dub.job['segments'][0]['text'] == 'hello' From 385bfe00f35e854786023ea4343cfd3cd471b030 Mon Sep 17 00:00:00 2001 From: debpalash <4178343+debpalash@users.noreply.github.com> Date: Sat, 3 Oct 2026 05:08:10 +0530 Subject: [PATCH 05/10] fix: invalidate old audio QC after successful dub publication --- CHANGELOG.md | 2 +- backend/api/routers/dub_generate.py | 6 ++-- docs/electron-dubbing.md | 3 +- .../src/features/dub/dub-session.test.ts | 28 +++++++++++++++++++ .../renderer/src/features/dub/dub-session.ts | 2 +- tests/test_dub_complete_audio.py | 4 ++- 6 files changed, 39 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 78141a5b1..47eb3c8a6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -87,7 +87,7 @@ metadata and the backend fallback mirror it. - The Twilio guide and integration directory describe the guided setup and in-app integration pages (#2304) ### Fixed -- Dubbing finishes when quality-check annotations arrive during assembly, while preserving those annotations and still protecting subtitle edits (#2585) +- Dubbing finishes when quality-check annotations arrive during assembly and clears measurements of replaced audio while still protecting subtitle edits (#2585) - Saving a voice design skips cold engine loading and downloads, including when a warm engine unloads during the save (#2583) — thanks @simoncheese! diff --git a/backend/api/routers/dub_generate.py b/backend/api/routers/dub_generate.py index 4f0fa0103..cd3fb83cc 100644 --- a/backend/api/routers/dub_generate.py +++ b/backend/api/routers/dub_generate.py @@ -2103,9 +2103,11 @@ def publish(): published_job["seg_order"] = list(expected_order) # Publish the already validated snapshot with the completed track. # Preserve other languages that may have been added during rendering. - # Rebuild from the current snapshot under the lock so QC - # annotations completed during assembly are not overwritten. + # Rebuild from the current snapshot under the lock, but clear + # QC measurements of the audio being replaced. A failed + # publication retains the original job and its valid QC. _sync_job_segments(published_job, req) + published_job["segments"] = _render_source_segments(published_job) published_job["dubbed_tracks"][lang_code] = { "path": track_path, "language": req.language, diff --git a/docs/electron-dubbing.md b/docs/electron-dubbing.md index 6c45d36d2..5824fc2ac 100644 --- a/docs/electron-dubbing.md +++ b/docs/electron-dubbing.md @@ -413,7 +413,8 @@ during QC also discards the result and keeps it out of history. Generation revalidates its segment text and identity snapshot after synthesis, before publishing fingerprints or replacing the previous track. An identity collision caused by a concurrent edit is reported through the task stream; the previous track and published metadata remain available. The source snapshot is checked again after asynchronous fitting and before replacing audio: subtitle imports made during generation or assembly survive, and the user can regenerate from the corrected subtitles. Fresh segment WAVs and the assembled track stay in a private staging directory until this check passes. Rejection or cancellation removes the staging files and preserves the reusable segment cache. Publication backs up existing files and rolls back ordinary installation failures; it does not promise a multi-file transaction across power loss. Fingerprints and the segment manifest publish only with a completed track. Empty generation requests are rejected before loading a voice engine. QC annotations arriving during generation do not count as source edits: a completed -render can publish while retaining those annotations. Source text, timing, identity, +render can still publish. Publishing replacement audio clears QC annotations measured +against the old audio; a failed publication retains the old track and its QC. Source text, timing, identity, voice bindings, and imported-cue changes still invalidate the admitted snapshot. Cancelling or failing a render preserves the previous committed track and its diff --git a/electron/src/renderer/src/features/dub/dub-session.test.ts b/electron/src/renderer/src/features/dub/dub-session.test.ts index 59632b75d..d5296c71d 100644 --- a/electron/src/renderer/src/features/dub/dub-session.test.ts +++ b/electron/src/renderer/src/features/dub/dub-session.test.ts @@ -790,3 +790,31 @@ it.each([ resetDubSession(); } }); + + +it.each(['done', 'error'] as const)('replaces old audio QC only after generation succeeds (%s)', async (terminal) => { + const segment = { + id: 'qc-row', start: 0, end: 1, text: 'Hello', text_original: 'Hello', + qc_drift: 0.4, qc_flagged: true, qc_recognized: 'Old audio', + qc_measured_start: 0.2, qc_measured_end: 1.4, + }; + dubSession.setState((current) => ({ + ...current, jobId: 'qc-replacement', phase: 'editing', recovery: null, + quality: 'fast', segments: [segment], + })); + vi.mocked(apiJson).mockReset().mockResolvedValueOnce({ task_id: 'qc-render' }); + vi.mocked(consumeTaskStream).mockReset().mockImplementationOnce(async (_path, emit) => { + emit(terminal === 'done' + ? { type: 'done', tracks: ['en'], sync_scores: [1] } + : { type: 'error', message: 'Render failed' }); + }); + await generateDub('English', 'en'); + const row = dubSession.state.segments[0]; + expect(row.text).toBe('Hello'); + if (terminal === 'done') { + expect(dubSession.state.phase).toBe('done'); + expect(Object.keys(row).filter((key) => key.startsWith('qc_'))).toEqual([]); + } else { + expect(row).toMatchObject(segment); + } +}); diff --git a/electron/src/renderer/src/features/dub/dub-session.ts b/electron/src/renderer/src/features/dub/dub-session.ts index 0dc7679cd..597018f81 100644 --- a/electron/src/renderer/src/features/dub/dub-session.ts +++ b/electron/src/renderer/src/features/dub/dub-session.ts @@ -1364,7 +1364,7 @@ async function watchGeneration(taskId: string, signal: AbortSignal) { tracks: Array.isArray(event.tracks) ? (event.tracks as string[]) : [], generatedTiming: dubSession.state.pendingTiming || 'strict_slot', segments: dubSession.state.segments.map((segment, index) => ({ - ...segment, + ...invalidateQc(segment), sync_ratio: typeof syncScores[index] === 'number' ? (syncScores[index] as number) : undefined, fit_status: diff --git a/tests/test_dub_complete_audio.py b/tests/test_dub_complete_audio.py index a4e5576c1..28b77507c 100644 --- a/tests/test_dub_complete_audio.py +++ b/tests/test_dub_complete_audio.py @@ -320,6 +320,8 @@ def test_failed_publication_restores_track_and_segment_cache(render_dub, monkeyp sf.write(cache, [.2] * 24000, 24000) cache_bytes = cache.read_bytes() render_dub.job['dubbed_tracks']['en'] = {'path': str(previous)} + render_dub.job['segments'] = [{'id': 'a', 'start': 0, 'end': 1, 'text': 'old speech', + 'qc_flagged': True, 'qc_recognized': 'old speech'}] original = copy.deepcopy(render_dub.job) if failure == 'install': replace = dg.os.replace @@ -374,5 +376,5 @@ async def stretch(wav, target, sr): events = render_dub.run(timing_strategy='strict_slot') assert any(e['type'] == 'done' for e in events) assert not any(e.get('error_code') == 'dub_source_changed' for e in events) - assert render_dub.job['segments'][0]['qc_recognized'] == 'measured speech' + assert not any(key.startswith('qc_') for key in render_dub.job['segments'][0]) assert render_dub.job['segments'][0]['text'] == 'hello' From a2641f5ec4005f6b7f79f23fdb0a1641e618a153 Mon Sep 17 00:00:00 2001 From: debpalash <4178343+debpalash@users.noreply.github.com> Date: Sat, 3 Oct 2026 06:02:43 +0530 Subject: [PATCH 06/10] fix(dub): offload publication and coordinate source updates --- CHANGELOG.md | 1 + backend/api/routers/dub_core.py | 278 +++++++++++--------- backend/api/routers/dub_export.py | 12 +- backend/api/routers/dub_generate.py | 38 ++- backend/services/dub_pipeline.py | 16 +- docs/electron-dubbing.md | 20 +- tests/test_dub_complete_audio.py | 241 ++++++++++++++++- tests/test_dub_import_srt_voice_metadata.py | 21 ++ tests/test_dub_no_tts_load_for_asr.py | 46 ++++ 9 files changed, 527 insertions(+), 146 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 47eb3c8a6..f511c5068 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -87,6 +87,7 @@ metadata and the backend fallback mirror it. - The Twilio guide and integration directory describe the guided setup and in-app integration pages (#2304) ### Fixed +- Dub publication keeps file and database work off the event loop, waits safely on cancellation, and restores audio after save failures (#2585) - Dubbing finishes when quality-check annotations arrive during assembly and clears measurements of replaced audio while still protecting subtitle edits (#2585) - Saving a voice design skips cold engine loading and downloads, including when a warm engine unloads during the save (#2583) — thanks @simoncheese! diff --git a/backend/api/routers/dub_core.py b/backend/api/routers/dub_core.py index 41ecae941..3293dbd5f 100644 --- a/backend/api/routers/dub_core.py +++ b/backend/api/routers/dub_core.py @@ -270,8 +270,7 @@ async def dub_import_srt(job_id: str, file: UploadFile = File(...)): re-time (overlap shifts). The caller surfaces these so the user knows if the import wasn't lossless. """ - job = _get_job(job_id) - if not job: + if not _get_job(job_id): raise HTTPException(status_code=404, detail="Job not found") try: raw_bytes = await file.read() @@ -296,68 +295,78 @@ async def dub_import_srt(job_id: str, file: UploadFile = File(...)): ), ) - # Clamp cues that run past the source media's known duration. Pipeline - # downstream code assumes segment.end <= duration; without this, dub - # generation would try to time-stretch into negative slack. - duration = float(job.get("duration") or 0.0) - clamped = 0 - if duration > 0: - kept = [] - for seg in result.segments: - if seg["start"] >= duration: - continue - if seg["end"] > duration: - seg = {**seg, "end": round(duration, 3)} - clamped += 1 - kept.append(seg) - # Re-id after clamp drops. - segments = [{**s, "id": i} for i, s in enumerate(kept)] - else: - segments = result.segments - - prior_segments = [ - segment for segment in (job.get("segments") or []) if isinstance(segment, dict) - ] - segments, segment_clones = _carry_srt_voice_metadata( - segments, - prior_segments, - job.get("segment_clones"), - job.get("speaker_clones"), - ) - job["segments"] = segments - job["segment_clones"] = segment_clones - # A pooled speaker clone is keyed only by a display label. Replacement - # cues can reuse that label without overlapping the original speaker, so - # retain matched pooled references as segment-specific clones above and - # drop the global map before rebuilding the cast. - job["speaker_clones"] = {} - if segment_clones: - from services.speaker_clone import build_cast_sources - - job["cast_sources"] = build_cast_sources( + # Upload reads may suspend while a render publishes. Resolve the current + # job only when applying parsed cues, with the same lock as publication. + return await asyncio.to_thread(_apply_imported_srt, job_id, result) + + +def _apply_imported_srt(job_id, result): + with dub_pipeline._dub_jobs_lock: + job = _get_job(job_id) + if not job: + raise HTTPException(status_code=404, detail="Job not found") + # Clamp cues that run past the source media's known duration. Pipeline + # downstream code assumes segment.end <= duration; without this, dub + # generation would try to time-stretch into negative slack. + duration = float(job.get("duration") or 0.0) + clamped = 0 + if duration > 0: + kept = [] + for seg in result.segments: + if seg["start"] >= duration: + continue + if seg["end"] > duration: + seg = {**seg, "end": round(duration, 3)} + clamped += 1 + kept.append(seg) + # Re-id after clamp drops. + segments = [{**s, "id": i} for i, s in enumerate(kept)] + else: + segments = result.segments + + prior_segments = [ + segment for segment in (job.get("segments") or []) if isinstance(segment, dict) + ] + segments, segment_clones = _carry_srt_voice_metadata( segments, - None, - segment_clones, + prior_segments, + job.get("segment_clones"), + job.get("speaker_clones"), ) - else: - job.pop("cast_sources", None) - # `source_lang` stays whatever the user (or the upload step) set; we - # don't try to language-detect off the cue text — that's noisy and the - # user usually knows what their .srt is. - _save_job(job_id, job) - logger.info( - "Imported %d cue(s) from .srt for job %s (skipped=%d, overlap_shifted=%d, clamped=%d)", - len(segments), log_safe(job_id), result.skipped_cues, result.dropped_overlaps, clamped, - ) - return { - "segments": segments, - "stats": { - "imported": len(segments), - "skipped_malformed": result.skipped_cues, - "dropped_overlap": result.dropped_overlaps, - "clamped_to_duration": clamped, - }, - } + job["segments"] = segments + job["segment_clones"] = segment_clones + # A pooled speaker clone is keyed only by a display label. Replacement + # cues can reuse that label without overlapping the original speaker, so + # retain matched pooled references as segment-specific clones above and + # drop the global map before rebuilding the cast. + job["speaker_clones"] = {} + if segment_clones: + from services.speaker_clone import build_cast_sources + + job["cast_sources"] = build_cast_sources( + segments, + None, + segment_clones, + ) + else: + job.pop("cast_sources", None) + # `source_lang` stays whatever the user (or the upload step) set; we + # don't try to language-detect off the cue text — that's noisy and the + # user usually knows what their .srt is. + _save_job(job_id, job) + logger.info( + "Imported %d cue(s) from .srt for job %s (skipped=%d, overlap_shifted=%d, clamped=%d)", + len(segments), log_safe(job_id), result.skipped_cues, result.dropped_overlaps, clamped, + ) + return { + "segments": segments, + "stats": { + "imported": len(segments), + "skipped_malformed": result.skipped_cues, + "dropped_overlap": result.dropped_overlaps, + "clamped_to_duration": clamped, + }, + } def _select_downloaded_caption_track( @@ -468,42 +477,43 @@ def remove_repeated_prefix(previous: str, current: str, whole: bool = False) -> @router.post("/dub/use-downloaded-captions/{job_id}") def dub_use_downloaded_captions(job_id: str): """Seed a prepared Dub job from its downloaded caption track.""" - job = _get_job(job_id) - if not job: - raise HTTPException(status_code=404, detail="Job not found") - tracks = job.get("youtube_subs") - if not isinstance(tracks, dict): - raise HTTPException(status_code=404, detail="No downloaded captions are available") - caption_lang = _select_downloaded_caption_track( - tracks, - job.get("source_lang_override") or job.get("source_lang"), - ) - if caption_lang is None: - raise HTTPException(status_code=404, detail="No downloaded captions are available") - segments = _prepare_downloaded_caption_segments( - tracks[caption_lang], - float(job.get("duration") or 0.0), - ) - if not segments: - raise HTTPException(status_code=422, detail="Downloaded captions contain no usable cues") - - source_lang = job.get("source_lang_override") or _detected_source_lang(caption_lang) - job["segments"] = segments - job["source_lang"] = source_lang - job["full_transcript"] = " ".join(segment["text"] for segment in segments) - # Caption files contain timing and text, but no trustworthy speaker or - # reference-audio attribution. Never retain stale clone maps from a prior - # transcript on the same job. - job["segment_clones"] = {} - job["speaker_clones"] = {} - job.pop("cast_sources", None) - _save_job(job_id, job) - return { - "segments": segments, - "source_lang": source_lang, - "caption_lang": caption_lang, - "available": sorted(tracks.keys()), - } + with dub_pipeline._dub_jobs_lock: + job = _get_job(job_id) + if not job: + raise HTTPException(status_code=404, detail="Job not found") + tracks = job.get("youtube_subs") + if not isinstance(tracks, dict): + raise HTTPException(status_code=404, detail="No downloaded captions are available") + caption_lang = _select_downloaded_caption_track( + tracks, + job.get("source_lang_override") or job.get("source_lang"), + ) + if caption_lang is None: + raise HTTPException(status_code=404, detail="No downloaded captions are available") + segments = _prepare_downloaded_caption_segments( + tracks[caption_lang], + float(job.get("duration") or 0.0), + ) + if not segments: + raise HTTPException(status_code=422, detail="Downloaded captions contain no usable cues") + + source_lang = job.get("source_lang_override") or _detected_source_lang(caption_lang) + job["segments"] = segments + job["source_lang"] = source_lang + job["full_transcript"] = " ".join(segment["text"] for segment in segments) + # Caption files contain timing and text, but no trustworthy speaker or + # reference-audio attribution. Never retain stale clone maps from a prior + # transcript on the same job. + job["segment_clones"] = {} + job["speaker_clones"] = {} + job.pop("cast_sources", None) + _save_job(job_id, job) + return { + "segments": segments, + "source_lang": source_lang, + "caption_lang": caption_lang, + "available": sorted(tracks.keys()), + } @router.post("/dub/cleanup-segments/{job_id}") @@ -516,17 +526,18 @@ def dub_cleanup_segments(job_id: str, req: Optional[CleanupSegmentsRequest] = No like any other, and the job keeps the segments its existing audio and subtitle exports were generated from until the next generation. """ - job = _get_job(job_id) - if not job: - raise HTTPException(status_code=404, detail="Job not found") - if req is not None: - cleaned = clean_up_segments(req.segments) - return {"segments": cleaned, "before": len(req.segments), "after": len(cleaned)} - segments = job.get("segments") or [] - cleaned = clean_up_segments(segments) - job["segments"] = cleaned - _save_job(job_id, job) - return {"segments": cleaned, "before": len(segments), "after": len(cleaned)} + with dub_pipeline._dub_jobs_lock: + job = _get_job(job_id) + if not job: + raise HTTPException(status_code=404, detail="Job not found") + if req is not None: + cleaned = clean_up_segments(req.segments) + return {"segments": cleaned, "before": len(req.segments), "after": len(cleaned)} + segments = job.get("segments") or [] + cleaned = clean_up_segments(segments) + job["segments"] = cleaned + _save_job(job_id, job) + return {"segments": cleaned, "before": len(segments), "after": len(cleaned)} @router.post("/dub/abort/{job_id}") @@ -543,9 +554,10 @@ def dub_abort(job_id: str): status_code=503, detail="The dub could not be fully aborted. Retry the abort operation.", ) from exc - job = _dub_jobs.get(job_id) - if job is not None: - job["aborted"] = True + with dub_pipeline._dub_jobs_lock: + job = _dub_jobs.get(job_id) + if job is not None: + job["aborted"] = True # Cancellation is idempotent: a missing active task means it already # stopped between the renderer aborting its stream and this request. return { @@ -2043,7 +2055,7 @@ def _use_turns(crash: Exception | None = None, err_sentinel=None): from services.segmentation import deduplicate_chunk_segments final_segs = deduplicate_chunk_segments(final_segs) - job["segments"] = final_segs + source_updates = {"segments": final_segs} # Auto-speaker-clone: sample each detected speaker's voice from the # Demucs-isolated vocals track and assign `auto:speaker_N` as the @@ -2120,7 +2132,7 @@ def _use_turns(crash: Exception | None = None, err_sentinel=None): # per-speaker clone below. Default on; the user can force # per-speaker by disabling it (job["per_segment_refs"]). seg_clones = {} - job["per_segment_refs"] = per_segment_refs + source_updates["per_segment_refs"] = per_segment_refs if per_segment_refs: try: from services.speaker_clone import extract_segment_refs @@ -2159,15 +2171,15 @@ def _use_turns(crash: Exception | None = None, err_sentinel=None): logger.warning( "segment ref-text refine timed out; keeping original ref_text: %s", e ) - job["segment_clones"] = seg_clones + source_updates["segment_clones"] = seg_clones except Exception as e: logger.warning("per-segment clone refs skipped: %s", e) cast_sources = build_cast_sources(final_segs, clones, seg_clones) - job["cast_sources"] = cast_sources + source_updates["cast_sources"] = cast_sources if cast_sources: if clones: - job["speaker_clones"] = clones + source_updates["speaker_clones"] = clones # Default each segment's profile_id to its detected speaker's # auto-clone — but only if the user hasn't already assigned # something. (#486) @@ -2195,12 +2207,14 @@ def _use_turns(crash: Exception | None = None, err_sentinel=None): except Exception as e: logger.warning("speaker_clone extraction skipped: %s", e) - job["source_lang"] = job.get("source_lang_override") or _detected_source_lang( - detected_lang - ) - job["full_transcript"] = " ".join(s.get("text", "") for s in final_segs) - job["transcription_complete"] = True - _save_job(job_id, job) + with dub_pipeline._dub_jobs_lock: + if _get_job(job_id) is not job: + raise HTTPException(status_code=404, detail="Job not found") + source_updates["source_lang"] = job.get("source_lang_override") or _detected_source_lang(detected_lang) + source_updates["full_transcript"] = " ".join(s.get("text", "") for s in final_segs) + source_updates["transcription_complete"] = True + job.update(source_updates) + _save_job(job_id, job) # Restore TTS model to GPU now that ASR is done. unload() blocks # (gc.collect + CUDA cache drop) — run it on the GPU pool so the @@ -2337,6 +2351,7 @@ async def dub_transcribe(job_id: str, num_speakers: Optional[int] = None): # OMNIVOICE_PRELOAD_TTS_ASR — and when it is off, that branch raises "fallback # is not preloaded" anyway. Loading ~3 GB to reach a None attribute (and then # having offload_tts_for_asr free it) was pure cost. + source_updates = {} _model = await get_model() if should_preload_tts_asr() else None # TTS-only install: no ASR model on disk → typed 409 with a download CTA, @@ -2407,7 +2422,7 @@ def _transcribe(): except Exception as e: logger.warning("Failed to unload ASR backend: %s", e) - job["source_lang"] = job.get("source_lang_override") or _detected_source_lang( + source_updates["source_lang"] = job.get("source_lang_override") or _detected_source_lang( detected_lang ) @@ -2452,7 +2467,7 @@ def _transcribe(): for s in segments: s.setdefault("text_original", s.get("text", "")) - job["full_transcript"] = " ".join(s["text"] for s in segments) + source_updates["full_transcript"] = " ".join(s["text"] for s in segments) # Transcription is done with the resident TTS model still offloaded; # release the accelerator cache the offload freed, on whichever @@ -2471,15 +2486,20 @@ def _transcribe(): # retry cannot overlap it (#1669). segments_result = await run_transcribe_guarded(_gpu_pool, _transcribe, what="Dub") except asyncio.CancelledError: - job["aborted"] = True + with dub_pipeline._dub_jobs_lock: + job["aborted"] = True raise if job.get("aborted"): raise HTTPException(status_code=499, detail="Transcription aborted") from services.segmentation import deduplicate_chunk_segments segments_result = deduplicate_chunk_segments(segments_result) - job["segments"] = segments_result - source_lang = job.get("source_lang") - _save_job(job_id, job) + with dub_pipeline._dub_jobs_lock: + if _get_job(job_id) is not job: + raise HTTPException(status_code=404, detail="Job not found") + source_updates["segments"] = segments_result + job.update(source_updates) + source_lang = job.get("source_lang") + _save_job(job_id, job) return { "job_id": job_id, "segments": segments_result, diff --git a/backend/api/routers/dub_export.py b/backend/api/routers/dub_export.py index e743c2d49..5792cabbc 100644 --- a/backend/api/routers/dub_export.py +++ b/backend/api/routers/dub_export.py @@ -18,6 +18,7 @@ from core.tasks import task_manager from fastapi import APIRouter, Header, HTTPException, Query, Request, Response from fastapi.responses import FileResponse, StreamingResponse +from services.dub_pipeline import _dub_jobs_lock from services.ffmpeg_utils import ( bed_mix_filter, explain_ffmpeg_failure, @@ -862,7 +863,8 @@ async def dub_download( ) # A fresh export is a fresh user intent — clear any sticky abort flag # from a previous /dub/abort so it can't kill this run's first batch. - job.pop("aborted", None) + with _dub_jobs_lock: + job.pop("aborted", None) # realpath-normalised + containment-checked inline at the sink (the # file's established pattern — CodeQL does not track the guard # through a helper's return value). @@ -888,7 +890,8 @@ async def dub_download( raise HTTPException(status_code=409, detail="Export aborted") from core.failure import build_failure retime_warning = build_failure(e, stage="video-retime", include_diagnostic=False) - job["last_export_warning"] = {"type": "video_retime_fallback", **retime_warning} + with _dub_jobs_lock: + job["last_export_warning"] = {"type": "video_retime_fallback", **retime_warning} logger.exception( "Smart Fit video retime failed for job %s — exporting " "without per-segment retime", @@ -1242,7 +1245,8 @@ async def _mux_preview(): smart_track_dur = float( retime_entry.get("total_duration") or track_info.get("duration") or 0.0 ) - job.pop("aborted", None) # fresh user intent — clear sticky abort + with _dub_jobs_lock: + job.pop("aborted", None) # fresh user intent — clear sticky abort # realpath-normalised + containment-checked inline at the sink # (same pattern as preview_path above — _base is the realpath # of DUB_DIR from the top of this endpoint). @@ -1688,7 +1692,7 @@ async def dub_qc_pass(job_id: str, lang: str = Query(None), drift_threshold: flo re-dub. The generated text stays authoritative (design delta from pyvideotrans, which overwrites subtitles).""" from services import dub_qc - from services.dub_pipeline import _dub_jobs_lock, put_and_save_job + from services.dub_pipeline import put_and_save_job _job_dir_or_400(job_id) lang = _safe_lang_or_400(lang) diff --git a/backend/api/routers/dub_generate.py b/backend/api/routers/dub_generate.py index cd3fb83cc..5b9fb5267 100644 --- a/backend/api/routers/dub_generate.py +++ b/backend/api/routers/dub_generate.py @@ -7,6 +7,7 @@ import asyncio import copy import contextlib +import contextvars import shutil import tempfile import torch @@ -39,7 +40,8 @@ from services.watermark import mark_synthetic from services.speaker_clone import auto_profile_id from services.segment_bundle import extract_segment_wavs -from api.routers.dub_core import _get_job, _save_job +from api.routers.dub_core import _get_job +from services.dub_pipeline import save_job_strict as _save_job from omnivoice.utils.voice_design import heal_design_instruct logger = logging.getLogger("omnivoice.dub") @@ -79,6 +81,29 @@ def _install_dub_artifacts(staged: dict[str, str]): raise +async def _finish_publication(publish): + """Keep blocking commit work off-loop and retain its files until it settles.""" + # Use an executor Future, not a detached Task: cancellation (including loop + # shutdown cancelling all Tasks) must not mark this work done while its + # thread is still installing or rolling back files in the staging directory. + pending = asyncio.get_running_loop().run_in_executor( + None, contextvars.copy_context().run, publish, + ) + try: + return await asyncio.shield(pending) + except (asyncio.CancelledError, GeneratorExit): + while not pending.done(): + try: + await asyncio.shield(pending) + except asyncio.CancelledError: + continue # repeated cancellation still cannot release live files + except Exception: + break # rollback finished; preserve the caller's cancellation + if not pending.cancelled(): + pending.exception() # observe a failure even while the caller exits + raise + + class _RemoteDubBackend: """Sample-rate carrier while Dubbing runs without local TTS weights.""" @@ -2029,8 +2054,8 @@ def _render_batch() -> list[torch.Tensor]: pass _release_audio_tensors() - # No await separates this final check from track/metadata publication. - # In particular, a subtitle import during asynchronous fitting wins. + # Reject changed source before writing the staged track. The worker + # revalidates under the job lock immediately before publication. if _get_job(job_id) is not job or _render_source_segments(job) != source_segments: yield f"data: {json.dumps({'type': 'error', 'error_code': 'dub_source_changed', 'error': 'Subtitles changed during generation. Generate again to use the current subtitles.'})}\n\n" return @@ -2072,6 +2097,8 @@ def _render_batch() -> list[torch.Tensor]: track_dur = total_samples / sr if total_samples > 0 else 0.0 def publish(): with dub_pipeline._dub_jobs_lock: + if task_manager.is_cancelled(task_id): + return None # cancellation won before publication started if _get_job(job_id) is not job or _render_source_segments(job) != source_segments: return False published_job = copy.deepcopy(job) @@ -2170,13 +2197,16 @@ def publish(): _t_diskw_0 = time.perf_counter() try: - committed = publish() + committed = await _finish_publication(publish) except Exception as exc: from core.public_errors import stream_generation_failure logger.exception("Dub publication failed for job %s", log_safe(job_id)) detail = stream_generation_failure(exc)["detail"] yield f"data: {json.dumps({'type': 'error', 'error': detail})}\n\n" return + if committed is None: + yield f"data: {json.dumps({'type': 'cancelled', 'segments_processed': total})}\n\n" + return if not committed: yield f"data: {json.dumps({'type': 'error', 'error_code': 'dub_source_changed', 'error': 'Subtitles changed during generation. Generate again to use the current subtitles.'})}\n\n" return diff --git a/backend/services/dub_pipeline.py b/backend/services/dub_pipeline.py index 4893c4b1f..26392d867 100644 --- a/backend/services/dub_pipeline.py +++ b/backend/services/dub_pipeline.py @@ -488,9 +488,10 @@ def purge_jobs(job_ids, *, delete_rows, include_inflight: bool = False) -> None: _expire_withdrawn(now, protected=len(targets)) -def save_job(job_id: str, job: dict, filename: str = "", duration: float = 0.0, content_hash: str = "") -> None: +def save_job(job_id: str, job: dict, filename: str = "", duration: float = 0.0, content_hash: str = "", *, strict: bool = False) -> None: """Persist dub job state to SQLite so it survives restarts. Uses UPSERT on `id` so repeated saves in a session keep the latest snapshot. + ``strict`` propagates failures when the caller must roll back audio files. language / language_code / content_hash only update when the incoming value is non-empty: the ingest-time insert runs before the target @@ -507,14 +508,21 @@ def save_job(job_id: str, job: dict, filename: str = "", duration: float = 0.0, # post-ingest save able to resurrect a dub the user deleted mid-render. # One choke point closes the class and the ninth caller inherits it. if job_id in _withdrawn_jobs: + if strict: + raise RuntimeError("Cannot publish a withdrawn dub job") logger.info( "Dub job %s was deleted while it was still running — not persisting", log_safe(job_id), ) return - _persist_job(job_id, job, filename, duration, content_hash) + _persist_job(job_id, job, filename, duration, content_hash, strict=strict) -def _persist_job(job_id: str, job: dict, filename: str, duration: float, content_hash: str) -> None: +def save_job_strict(job_id: str, job: dict) -> None: + """Persist a publication or raise so its staged artifacts can roll back.""" + save_job(job_id, job, strict=True) + + +def _persist_job(job_id: str, job: dict, filename: str, duration: float, content_hash: str, *, strict: bool = False) -> None: """The actual write. Callers go through :func:`save_job`, which gates it.""" try: segments = job.get("segments") or [] @@ -540,6 +548,8 @@ def _persist_job(job_id: str, job: dict, filename: str, duration: float, content ) except Exception as exc: logger.error("Failed to persist dub job %s: %s", log_safe(job_id), log_safe(exc)) + if strict: + raise return event_bus.emit("dub_history", {"action": "saved", "id": job_id}) diff --git a/docs/electron-dubbing.md b/docs/electron-dubbing.md index 5824fc2ac..38c571c8f 100644 --- a/docs/electron-dubbing.md +++ b/docs/electron-dubbing.md @@ -410,16 +410,30 @@ runs. If the track, transcript, timing, or audio changes during that pass, QC asks you to run it again instead of publishing stale scores. Deleting the job during QC also discards the result and keeps it out of history. -Generation revalidates its segment text and identity snapshot after synthesis, before publishing fingerprints or replacing the previous track. An identity collision caused by a concurrent edit is reported through the task stream; the previous track and published metadata remain available. The source snapshot is checked again after asynchronous fitting and before replacing audio: subtitle imports made during generation or assembly survive, and the user can regenerate from the corrected subtitles. Fresh segment WAVs and the assembled track stay in a private staging directory until this check passes. Rejection or cancellation removes the staging files and preserves the reusable segment cache. Publication backs up existing files and rolls back ordinary installation failures; it does not promise a multi-file transaction across power loss. Fingerprints and the segment manifest publish only with a completed track. Empty generation requests are rejected before loading a voice engine. +Generation revalidates its segment text and identity snapshot after synthesis, before publishing fingerprints or replacing the previous track. An identity collision caused by a concurrent edit is reported through the task stream; the previous track and published metadata remain available. The source snapshot is checked again after asynchronous fitting and before replacing audio: subtitle imports made during generation or assembly survive, and the user can regenerate from the corrected subtitles. Fresh segment WAVs and the assembled track stay in a private staging directory until this check passes. Rejection or cancellation before publication removes the staging files and preserves the reusable segment cache. Publication backs up existing files and rolls back ordinary installation failures; it does not promise a multi-file transaction across power loss. Fingerprints and the segment manifest publish only with a completed track. Empty generation requests are rejected before loading a voice engine. QC annotations arriving during generation do not count as source edits: a completed render can still publish. Publishing replacement audio clears QC annotations measured against the old audio; a failed publication retains the old track and its QC. Source text, timing, identity, voice bindings, and imported-cue changes still invalidate the admitted snapshot. -Cancelling or failing a render preserves the previous committed track and its -segment cache. Newly synthesized, unpublished segments are discarded and must be +Cancellation before publication or a failed render preserves the previous +committed track and its segment cache. Newly synthesized, unpublished segments are discarded and must be synthesized again on retry. Resume reconnects to an existing running task; it does not recover a cancelled task's unpublished speech. Reusing that speech would require a separate resume cache with validated engine and reference revisions, rather than replacing the committed cache with partial output. + +Track publication runs in a worker thread so file backups and SQLite waits do +not occupy the async event loop. Source validation, file installation and the +strict database save share the job lock; a failed save rolls back the audio +replacement. Once that publication transaction has started, cancellation +waits for its commit or rollback before removing staging files. A completed +commit stays published even if cancellation arrives during it. Deleting a job +serializes with publication and cannot leave a resurrected history row. + +Subtitle imports re-read the current job when applying uploaded cues. Imports, +caption cleanup, transcription source updates, and QC share the publication +lock, so an edit arriving during publication is applied afterward. Transcription +keeps new source fields private until completion and refuses to publish into a +deleted or replaced job. Concurrently completed dub tracks are preserved. diff --git a/tests/test_dub_complete_audio.py b/tests/test_dub_complete_audio.py index 28b77507c..ab0efa136 100644 --- a/tests/test_dub_complete_audio.py +++ b/tests/test_dub_complete_audio.py @@ -14,6 +14,7 @@ @pytest.fixture def render_dub(monkeypatch, tmp_path): import api.routers.dub_generate as dg + real_save = dg._save_job job = {'duration': 4.0, 'dubbed_tracks': {}, 'segments': [], 'seg_wav_kind_by_lang': {'en': 'natural'}} path = tmp_path / 'job' path.mkdir() @@ -41,12 +42,15 @@ async def add_task(self, tid, kind, func, *args): monkeypatch.setattr(dg, 'mark_synthetic', lambda a, *args, **kw: a) monkeypatch.setattr(dg, 'get_effect_chain', lambda _: None) monkeypatch.setattr(dg, 'normalize_audio', lambda a, **kw: a) - def run(**kwargs): + async def arun(**kwargs): body = dict(segments=[dict(start=0, end=1, text='hello')], segment_ids=['a'], language_code='en', num_step=4) body.update(kwargs) - asyncio.run(dg.dub_generate('job', DubRequest(**body))) + await dg.dub_generate('job', DubRequest(**body)) return events - return SimpleNamespace(run=run, output=output, job=job, path=path, generated=generated) + def run(**kwargs): + return asyncio.run(arun(**kwargs)) + return SimpleNamespace(run=run, arun=arun, real_save=real_save, output=output, + job=job, path=path, generated=generated) def test_failed_segment_does_not_publish_complete_track(render_dub): @@ -378,3 +382,234 @@ async def stretch(wav, target, sr): assert not any(e.get('error_code') == 'dub_source_changed' for e in events) assert not any(key.startswith('qc_') for key in render_dub.job['segments'][0]) assert render_dub.job['segments'][0]['text'] == 'hello' + + +def test_publication_keeps_event_loop_responsive(render_dub, monkeypatch): + import threading + from api.routers import dub_generate as dg + + release = threading.Event() + async def exercise(): + entered = asyncio.Event() + loop = asyncio.get_running_loop() + publishing_threads = [] + def slow_save(*_): + publishing_threads.append(threading.get_ident()) + loop.call_soon_threadsafe(entered.set) + assert release.wait(2), 'publisher blocked the event-loop releasing it' + monkeypatch.setattr(dg, '_save_job', slow_save) + render = asyncio.create_task(render_dub.arun()) + try: + await asyncio.wait_for(entered.wait(), 10) + assert not render.done(), 'event loop resumed only after blocking publication ended' + assert len(publishing_threads) == 1 + assert publishing_threads[0] != threading.get_ident() + release.set() + events = await render + assert any(event['type'] == 'done' for event in events) + finally: + release.set() + await asyncio.gather(render, return_exceptions=True) + asyncio.run(exercise()) + + +@pytest.mark.parametrize('persist_fails', [False, True]) +def test_cancellation_waits_for_publication_before_staging_cleanup(render_dub, monkeypatch, persist_fails): + import threading + from api.routers import dub_generate as dg + + previous = render_dub.path / 'dubbed_en.wav' + previous.write_bytes(b'previous complete track') + cache = render_dub.path / 'seg_en_a.wav' + sf.write(cache, [.2] * 24000, 24000) + cache_bytes = cache.read_bytes() + render_dub.job['dubbed_tracks']['en'] = {'path': str(previous)} + original = copy.deepcopy(render_dub.job) + release = threading.Event() + async def exercise(): + entered = asyncio.Event() + loop = asyncio.get_running_loop() + def slow_save(*_): + loop.call_soon_threadsafe(entered.set) + assert release.wait(2), 'publisher blocked cancellation handling' + if persist_fails: + raise OSError('injected publication failure during cancellation') + monkeypatch.setattr(dg, '_save_job', slow_save) + render = asyncio.create_task(render_dub.arun()) + try: + await asyncio.wait_for(entered.wait(), 10) + assert not render.done() + for _ in range(2): + render.cancel() + await asyncio.sleep(0) + assert not render.done(), 'cancellation detached the in-flight publisher' + assert list(render_dub.path.glob('.render-*')), 'staging deleted under the publisher' + release.set() + with pytest.raises(asyncio.CancelledError): + await render + finally: + release.set() + await asyncio.gather(render, return_exceptions=True) + asyncio.run(exercise()) + assert not list(render_dub.path.glob('.render-*')) + if persist_fails: + assert previous.read_bytes() == b'previous complete track' + assert cache.read_bytes() == cache_bytes + assert render_dub.job == original + else: + assert sf.info(previous).duration == 4 + assert cache.read_bytes() != cache_bytes + assert render_dub.job['segments'][0]['text'] == 'hello' + + +def test_sqlite_publication_failure_rolls_back_audio_and_metadata(render_dub, monkeypatch, tmp_path): + from collections import OrderedDict + from core import db + from services import dub_pipeline as dp + from api.routers import dub_generate as dg + + monkeypatch.setattr(db, 'DB_PATH', str(tmp_path / 'publication.sqlite')) + monkeypatch.setattr(dp, '_withdrawn_jobs', OrderedDict()) + with db.db_conn() as conn: + conn.executescript(db._BASE_SCHEMA) + previous = render_dub.path / 'dubbed_en.wav' + previous.write_bytes(b'previous complete track') + cache = render_dub.path / 'seg_en_a.wav' + sf.write(cache, [.2] * 24000, 24000) + cache_bytes = cache.read_bytes() + render_dub.job['dubbed_tracks']['en'] = {'path': str(previous)} + original = copy.deepcopy(render_dub.job) + dp.save_job('job', original) + with db.db_conn() as conn: + conn.execute("CREATE TRIGGER reject_render BEFORE UPDATE ON dub_history BEGIN SELECT RAISE(ABORT, 'injected SQLite failure'); END") + dp.save_job('job', original) # normal callers still log and tolerate save failures + monkeypatch.setattr(dg, '_save_job', render_dub.real_save) + events = render_dub.run() + assert any(event['type'] == 'error' for event in events) + assert not any(event['type'] == 'done' for event in events) + assert previous.read_bytes() == b'previous complete track' + assert cache.read_bytes() == cache_bytes + assert render_dub.job == original + with db.db_conn() as conn: + assert json.loads(conn.execute('SELECT job_data FROM dub_history WHERE id=?', ('job',)).fetchone()[0]) == original + assert not list(render_dub.path.glob('.render-*')) + + +@pytest.mark.parametrize('delete_before_publication', [False, True]) +def test_delete_serializes_with_publication_without_resurrecting_job( + render_dub, monkeypatch, tmp_path, delete_before_publication, +): + import threading + from collections import OrderedDict + from core import db + from services import dub_pipeline as dp + from api.routers import dub_generate as dg + + monkeypatch.setattr(db, 'DB_PATH', str(tmp_path / 'delete-publication.sqlite')) + monkeypatch.setattr(dp, '_dub_jobs', {'job': render_dub.job}) + monkeypatch.setattr(dp, '_withdrawn_jobs', OrderedDict()) + with db.db_conn() as conn: + conn.executescript(db._BASE_SCHEMA) + dp.save_job('job', render_dub.job) + monkeypatch.setattr(dg, '_get_job', dp.get_job) + release = threading.Event() + def delete(): + def delete_rows(): + with db.db_conn() as conn: + conn.execute('DELETE FROM dub_history WHERE id=?', ('job',)) + dp.purge_jobs(['job'], delete_rows=delete_rows) + async def exercise(): + loop = asyncio.get_running_loop() + entered = asyncio.Event() + def slow_save(*args): + loop.call_soon_threadsafe(entered.set) + assert release.wait(2) + render_dub.real_save(*args) + monkeypatch.setattr(dg, '_save_job', slow_save) + if delete_before_publication: + finish = dg._finish_publication + async def delete_first(publish): + delete() + return await finish(publish) + monkeypatch.setattr(dg, '_finish_publication', delete_first) + events = await render_dub.arun() + assert any(event.get('error_code') == 'dub_source_changed' for event in events) + assert not (render_dub.path / 'dubbed_en.wav').exists() + else: + render = asyncio.create_task(render_dub.arun()) + deletion = None + try: + await asyncio.wait_for(entered.wait(), 10) + deletion = loop.run_in_executor(None, delete) + await asyncio.sleep(0) + assert not deletion.done(), 'delete crossed the publication lock' + release.set() + await asyncio.gather(render, deletion) + finally: + release.set() + await asyncio.gather(render, *([deletion] if deletion else []), return_exceptions=True) + asyncio.run(exercise()) + assert 'job' not in dp._dub_jobs + with db.db_conn() as conn: + assert conn.execute('SELECT id FROM dub_history WHERE id=?', ('job',)).fetchone() is None + assert not list(render_dub.path.glob('.render-*')) + + +def test_cancel_before_publication_starts_discards_staging(render_dub, monkeypatch): + from api.routers import dub_generate as dg + + cancelled = [False] + monkeypatch.setattr(dg.task_manager, 'is_cancelled', lambda _: cancelled[0]) + finish = dg._finish_publication + async def cancel_first(publish): + cancelled[0] = True + return await finish(publish) + monkeypatch.setattr(dg, '_finish_publication', cancel_first) + events = render_dub.run() + assert any(event['type'] == 'cancelled' for event in events) + assert not any(event['type'] == 'done' for event in events) + assert not render_dub.job['dubbed_tracks'] + assert not (render_dub.path / 'dubbed_en.wav').exists() + assert not (render_dub.path / 'seg_en_a.wav').exists() + assert not list(render_dub.path.glob('.render-*')) + + +def test_import_finishing_during_publication_preserves_new_subtitles(render_dub, monkeypatch): + import threading + from api.routers import dub_core, dub_generate as dg + + release = threading.Event() + async def exercise(): + loop = asyncio.get_running_loop() + publishing = asyncio.Event() + reading = asyncio.Event() + read_finished = asyncio.Event() + class Upload: + async def read(self): + reading.set() + await publishing.wait() + read_finished.set() + return b'1\n00:00:00,000 --> 00:00:01,000\nreplacement subtitle\n' + def slow_save(*_): + loop.call_soon_threadsafe(publishing.set) + assert release.wait(2) + monkeypatch.setattr(dg, '_save_job', slow_save) + monkeypatch.setattr(dub_core, '_get_job', lambda _: render_dub.job) + monkeypatch.setattr(dub_core, '_save_job', lambda *_: None) + imported = asyncio.create_task(dub_core.dub_import_srt('job', Upload())) + await reading.wait() # endpoint has started reading before publication + render = asyncio.create_task(render_dub.arun()) + try: + await asyncio.wait_for(read_finished.wait(), 10) + await asyncio.sleep(0) + assert not imported.done(), 'import mutated the job during publication' + release.set() + events, result = await asyncio.gather(render, imported) + assert any(event['type'] == 'done' for event in events) + assert result['segments'][0]['text'] == 'replacement subtitle' + assert render_dub.job['segments'][0]['text'] == 'replacement subtitle' + assert 'en' in render_dub.job['dubbed_tracks'] + finally: + release.set() + await asyncio.gather(render, imported, return_exceptions=True) + asyncio.run(exercise()) diff --git a/tests/test_dub_import_srt_voice_metadata.py b/tests/test_dub_import_srt_voice_metadata.py index 742e6f975..b08194e85 100644 --- a/tests/test_dub_import_srt_voice_metadata.py +++ b/tests/test_dub_import_srt_voice_metadata.py @@ -189,3 +189,24 @@ def test_import_srt_scopes_matched_speaker_clone_to_the_matched_cue(monkeypatch) assert job["segment_clones"] == {"0": speaker_clone} assert "1" not in job["segment_clones"] assert job["cast_sources"]["Speaker 1"]["kind"] == "segment" + + +def test_import_does_not_restore_job_deleted_during_upload(monkeypatch): + import pytest + from fastapi import HTTPException + from api.routers import dub_core + + job = {'duration': 2, 'segments': []} + current = [job] + saved = [] + monkeypatch.setattr(dub_core, '_get_job', lambda _: current[0]) + monkeypatch.setattr(dub_core, '_save_job', lambda *args: saved.append(args)) + class Upload: + async def read(self): + current[0] = None + return b'1\n00:00:00,000 --> 00:00:01,000\nNew text\n' + with pytest.raises(HTTPException) as error: + asyncio.run(dub_core.dub_import_srt('deleted', Upload())) + assert error.value.status_code == 404 + assert job['segments'] == [] + assert saved == [] diff --git a/tests/test_dub_no_tts_load_for_asr.py b/tests/test_dub_no_tts_load_for_asr.py index 3911df05d..28c90e912 100644 --- a/tests/test_dub_no_tts_load_for_asr.py +++ b/tests/test_dub_no_tts_load_for_asr.py @@ -143,3 +143,49 @@ def test_preflight_error_does_not_leave_asr_on_vocals_unbound(dub, monkeypatch): body = _drain(dc, job_id) assert "NameError" not in body assert "No audio available" in body + + +@pytest.mark.parametrize('outcome', ['complete', 'deleted', 'replaced', 'failed', 'cancelled']) +def test_transcription_publishes_private_source_without_replacing_tracks(dub, monkeypatch, outcome): + from fastapi import HTTPException + + dc, job_id, _ = dub + job = dc._dub_jobs[job_id] + original_segments = [{'id': 0, 'start': 0, 'end': .5, 'text': 'previous source'}] + job.update(segments=original_segments, source_lang='old', full_transcript='previous source') + monkeypatch.setattr(dc, 'should_preload_tts_asr', lambda: False) + monkeypatch.setattr(dc, '_save_job', lambda *_: None) + async def guarded(_pool, transcribe, **_kwargs): + result = transcribe() # actual ASR closure with the fixture's fake engine + assert job['segments'] == original_segments + assert job['source_lang'] == 'old', 'ASR exposed source metadata before commit' + assert job['full_transcript'] == 'previous source' + # A render can complete during ASR; keep its newly committed track. + job['dubbed_tracks']['en'] = {'path': 'concurrently-published.wav'} + if outcome == 'deleted': + dc._dub_jobs.pop(job_id) + monkeypatch.setattr(dc, '_get_job', lambda _: None) + elif outcome == 'replaced': + dc._dub_jobs[job_id] = {'segments': [], 'replacement': True} + elif outcome == 'failed': + raise RuntimeError('injected ASR completion failure') + elif outcome == 'cancelled': + raise asyncio.CancelledError() + return result + monkeypatch.setattr(dc, 'run_transcribe_guarded', guarded) + if outcome == 'complete': + result = asyncio.run(dc.dub_transcribe(job_id)) + assert result['source_lang'] == 'en' + assert result['full_transcript'] == 'hi' + assert job['segments'] != original_segments + assert job['dubbed_tracks']['en']['path'] == 'concurrently-published.wav' + else: + error_type = asyncio.CancelledError if outcome == 'cancelled' else HTTPException + with pytest.raises(error_type) as error: + asyncio.run(dc.dub_transcribe(job_id)) + if outcome in ('deleted', 'replaced'): + assert error.value.status_code == 404 + assert job['segments'] == original_segments + assert job['source_lang'] == 'old' + if outcome == 'replaced': + assert dc._dub_jobs[job_id] == {'segments': [], 'replacement': True} From dcdbf540a7c6f698c4735fac42f24c0ae753b6c7 Mon Sep 17 00:00:00 2001 From: debpalash <4178343+debpalash@users.noreply.github.com> Date: Sat, 3 Oct 2026 06:14:05 +0530 Subject: [PATCH 07/10] fix(dub): retain source fields throughout publication --- CHANGELOG.md | 2 +- backend/api/routers/dub_generate.py | 3 ++- docs/electron-dubbing.md | 4 +++- tests/test_dub_complete_audio.py | 28 ++++++++++++++++++++++++++++ 4 files changed, 34 insertions(+), 3 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f337b3940..fb63f07b4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -90,7 +90,7 @@ metadata and the backend fallback mirror it. - The Twilio guide and integration directory describe the guided setup and in-app integration pages (#2304) ### Fixed -- Dub publication keeps file and database work off the event loop, waits safely on cancellation, and restores audio after save failures (#2585) +- Dub publication keeps file and database work off the event loop, preserves source metadata, waits safely on cancellation, and restores audio after save failures (#2585) - Dubbing finishes when quality-check annotations arrive during assembly and clears measurements of replaced audio while still protecting subtitle edits (#2585) - Saving a voice design skips cold engine loading and downloads, including when a warm engine unloads during the save (#2583) — thanks @simoncheese! diff --git a/backend/api/routers/dub_generate.py b/backend/api/routers/dub_generate.py index 5b9fb5267..42f631db1 100644 --- a/backend/api/routers/dub_generate.py +++ b/backend/api/routers/dub_generate.py @@ -2191,7 +2191,8 @@ def publish(): published_job["seg_wav_kind"] = _kind _save_job(job_id, published_job) - job.clear() + # Keep the shared job populated for readers already holding a + # reference. Publication only adds/replaces top-level fields. job.update(published_job) return True diff --git a/docs/electron-dubbing.md b/docs/electron-dubbing.md index 38c571c8f..5e113aaf4 100644 --- a/docs/electron-dubbing.md +++ b/docs/electron-dubbing.md @@ -425,7 +425,9 @@ a separate resume cache with validated engine and reference revisions, rather th replacing the committed cache with partial output. Track publication runs in a worker thread so file backups and SQLite waits do -not occupy the async event loop. Source validation, file installation and the +not occupy the async event loop. The shared job retains its source fields while +completed metadata is applied, including for readers holding an existing job +reference. Source validation, file installation and the strict database save share the job lock; a failed save rolls back the audio replacement. Once that publication transaction has started, cancellation waits for its commit or rollback before removing staging files. A completed diff --git a/tests/test_dub_complete_audio.py b/tests/test_dub_complete_audio.py index ab0efa136..ca984ad59 100644 --- a/tests/test_dub_complete_audio.py +++ b/tests/test_dub_complete_audio.py @@ -384,6 +384,34 @@ async def stretch(wav, target, sr): assert render_dub.job['segments'][0]['text'] == 'hello' +def test_publication_keeps_source_fields_visible_to_existing_job_readers(render_dub, monkeypatch): + from api.routers import dub_generate as dg + + observed = [] + + class ObservedJob(dict): + # Record each state exposed by an in-place publication step. Other + # tasks can already hold this dictionary without reacquiring the lock. + def clear(self): + super().clear() + observed.append((self.get('duration'), 'segments' in self)) + + def update(self, *args, **kwargs): + super().update(*args, **kwargs) + observed.append((self.get('duration'), 'segments' in self)) + + def __deepcopy__(self, memo): + return copy.deepcopy(dict(self), memo) + + job = ObservedJob(render_dub.job) + monkeypatch.setattr(dg, '_get_job', lambda _: job) + events = render_dub.run() + + assert any(event['type'] == 'done' for event in events) + assert observed and all(state == (4.0, True) for state in observed) + assert job['dubbed_tracks']['en']['path'].endswith('dubbed_en.wav') + + def test_publication_keeps_event_loop_responsive(render_dub, monkeypatch): import threading from api.routers import dub_generate as dg From 79a1e50d666c10a9611be832d9ce9185369cfe7d Mon Sep 17 00:00:00 2001 From: debpalash <4178343+debpalash@users.noreply.github.com> Date: Sat, 3 Oct 2026 06:43:14 +0530 Subject: [PATCH 08/10] fix(dub): coordinate async job access and committed outcomes --- CHANGELOG.md | 2 + backend/api/routers/dub_core.py | 90 ++++++++--- backend/api/routers/dub_export.py | 142 ++++++++++-------- backend/api/routers/dub_generate.py | 29 +--- backend/api/routers/dub_translate.py | 13 +- backend/api/routers/engines.py | 2 +- backend/api/routers/tools.py | 2 +- backend/core/tasks.py | 15 +- backend/services/dub_pipeline.py | 42 +++++- docs/electron-dubbing.md | 9 +- .../src/renderer/src/i18n/locales/ar.json | 1 + .../src/renderer/src/i18n/locales/de.json | 1 + .../src/renderer/src/i18n/locales/en.json | 1 + .../src/renderer/src/i18n/locales/es.json | 1 + .../src/renderer/src/i18n/locales/fr.json | 1 + .../src/renderer/src/i18n/locales/hi.json | 1 + .../src/renderer/src/i18n/locales/id.json | 1 + .../src/renderer/src/i18n/locales/it.json | 1 + .../src/renderer/src/i18n/locales/ja.json | 1 + .../src/renderer/src/i18n/locales/ko.json | 1 + .../src/renderer/src/i18n/locales/nl.json | 1 + .../src/renderer/src/i18n/locales/pl.json | 1 + .../src/renderer/src/i18n/locales/pt.json | 1 + .../src/renderer/src/i18n/locales/ru.json | 1 + .../src/renderer/src/i18n/locales/sv.json | 1 + .../src/renderer/src/i18n/locales/th.json | 1 + .../src/renderer/src/i18n/locales/tr.json | 1 + .../src/renderer/src/i18n/locales/uk.json | 1 + .../src/renderer/src/i18n/locales/vi.json | 1 + .../src/renderer/src/i18n/locales/zh-CN.json | 1 + .../src/renderer/src/i18n/locales/zh-TW.json | 1 + .../src/renderer/src/lib/api/failure.test.ts | 9 ++ electron/src/renderer/src/lib/api/failure.ts | 16 +- electron/src/shared/i18n/locales/ar.json | 1 + electron/src/shared/i18n/locales/de.json | 1 + electron/src/shared/i18n/locales/en.json | 1 + electron/src/shared/i18n/locales/es.json | 1 + electron/src/shared/i18n/locales/fr.json | 1 + electron/src/shared/i18n/locales/hi.json | 1 + electron/src/shared/i18n/locales/id.json | 1 + electron/src/shared/i18n/locales/it.json | 1 + electron/src/shared/i18n/locales/ja.json | 1 + electron/src/shared/i18n/locales/ko.json | 1 + electron/src/shared/i18n/locales/nl.json | 1 + electron/src/shared/i18n/locales/pl.json | 1 + electron/src/shared/i18n/locales/pt.json | 1 + electron/src/shared/i18n/locales/ru.json | 1 + electron/src/shared/i18n/locales/sv.json | 1 + electron/src/shared/i18n/locales/th.json | 1 + electron/src/shared/i18n/locales/tr.json | 1 + electron/src/shared/i18n/locales/uk.json | 1 + electron/src/shared/i18n/locales/vi.json | 1 + electron/src/shared/i18n/locales/zh-CN.json | 1 + electron/src/shared/i18n/locales/zh-TW.json | 1 + tests/test_dub_complete_audio.py | 104 ++++++++++++- tests/test_dub_job_deleted_mid_ingest.py | 12 +- tests/test_dub_no_tts_load_for_asr.py | 62 +++++++- tests/test_dub_pipeline_state.py | 54 +++++++ tests/test_dub_qc_concurrency.py | 46 +++++- 59 files changed, 542 insertions(+), 149 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 247583991..bbe53522d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -101,6 +101,8 @@ metadata and the backend fallback mirror it. - License notice: commercial use is free under the AGPL; the paid licence is for closed-source use, with Pro plans linked (#2578) ### Fixed + +- Preserve subtitle edits made during transcription, keep dub lock waits off the event loop, and report a committed track as complete after late cancellation (#2585) - Dub publication keeps file and database work off the event loop, preserves source metadata, waits safely on cancellation, and restores audio after save failures (#2585) - Dubbing finishes when quality-check annotations arrive during assembly and clears measurements of replaced audio while still protecting subtitle edits (#2585) diff --git a/backend/api/routers/dub_core.py b/backend/api/routers/dub_core.py index 9fa18c596..50322e5ef 100644 --- a/backend/api/routers/dub_core.py +++ b/backend/api/routers/dub_core.py @@ -2,6 +2,8 @@ import errno import uuid import asyncio +import copy +import threading import logging import shutil import subprocess @@ -258,6 +260,58 @@ def dub_parse_subtitle_text(req: ParseSubtitleTextRequest): } +def _transcription_source(job): + """Source edits invalidate ASR, while derived QC and completed tracks do not.""" + snapshot = {key: copy.deepcopy(job.get(key)) for key in ( + "segments", "source_lang_override", "segment_clones", "speaker_clones", + "cast_sources", "per_segment_refs", + )} + if isinstance(snapshot["segments"], list): + snapshot["segments"] = [ + {key: value for key, value in row.items() if not key.startswith("qc_")} + if isinstance(row, dict) else row for row in snapshot["segments"] + ] + return snapshot + + +def _transcription_snapshot(job_id): + with dub_pipeline._dub_jobs_lock: + job = _get_job(job_id) + return job, _transcription_source(job) if job else None + + +def _publish_transcription(job_id, job, source_snapshot, updates, cancelled): + with dub_pipeline._dub_jobs_lock: + if cancelled.is_set(): + return + if _get_job(job_id) is not job: + raise HTTPException(status_code=404, detail="Job not found") + if job.get("aborted"): + raise HTTPException(status_code=499, detail="Transcription aborted") + if _transcription_source(job) != source_snapshot: + raise HTTPException(status_code=409, detail={ + "code": "dub_transcription_source_changed", + "message": "Subtitles changed during transcription. Your edits were kept.", + }) + job.update(updates) + _save_job(job_id, job) + + +async def _save_transcription(job_id, job, source_snapshot, updates): + """Wait for the publication lock off-loop; cancelled queued work stays private.""" + cancelled = threading.Event() + try: + await asyncio.to_thread(_publish_transcription, job_id, job, source_snapshot, updates, cancelled) + except asyncio.CancelledError: + cancelled.set() + raise + + +def _mark_job_aborted(job): + with dub_pipeline._dub_jobs_lock: + job["aborted"] = True + + @router.post("/dub/import-srt/{job_id}") async def dub_import_srt(job_id: str, file: UploadFile = File(...)): """Replace `job["segments"]` with timestamps + text parsed from an SRT @@ -268,7 +322,7 @@ async def dub_import_srt(job_id: str, file: UploadFile = File(...)): re-time (overlap shifts). The caller surfaces these so the user knows if the import wasn't lossless. """ - if not _get_job(job_id): + if not await asyncio.to_thread(_get_job, job_id): raise HTTPException(status_code=404, detail="Job not found") try: raw_bytes = await file.read() @@ -1279,7 +1333,7 @@ async def _gen_body(): from core.run_sentinel import touch_activity touch_activity("transcribe", "dub") - job = _get_job(job_id) + job, source_snapshot = await asyncio.to_thread(_transcription_snapshot, job_id) # The durable job is written before the terminal SSE events below. If # the renderer, proxy, or backend connection drops in that narrow @@ -2205,14 +2259,10 @@ def _use_turns(crash: Exception | None = None, err_sentinel=None): except Exception as e: logger.warning("speaker_clone extraction skipped: %s", e) - with dub_pipeline._dub_jobs_lock: - if _get_job(job_id) is not job: - raise HTTPException(status_code=404, detail="Job not found") - source_updates["source_lang"] = job.get("source_lang_override") or _detected_source_lang(detected_lang) - source_updates["full_transcript"] = " ".join(s.get("text", "") for s in final_segs) - source_updates["transcription_complete"] = True - job.update(source_updates) - _save_job(job_id, job) + source_updates["source_lang"] = job.get("source_lang_override") or _detected_source_lang(detected_lang) + source_updates["full_transcript"] = " ".join(s.get("text", "") for s in final_segs) + source_updates["transcription_complete"] = True + await _save_transcription(job_id, job, source_snapshot, source_updates) # Restore TTS model to GPU now that ASR is done. unload() blocks # (gc.collect + CUDA cache drop) — run it on the GPU pool so the @@ -2277,7 +2327,10 @@ async def gen(): except Exception as exc: # noqa: BLE001 — last-resort stream finalizer logger.error("Transcription stream failed unexpectedly (class=%s)", type(exc).__name__) from core.public_errors import stream_failure, transcription_failure_code - yield _sse_event("error", stream_failure(transcription_failure_code(exc))) + if isinstance(exc, HTTPException) and isinstance(exc.detail, dict) and exc.detail.get("code") == "dub_transcription_source_changed": + yield _sse_event("error", {"error_code": exc.detail["code"], "error": exc.detail["message"]}) + else: + yield _sse_event("error", stream_failure(transcription_failure_code(exc))) yield _sse_event("done", {}) finally: _asr_work.stop() @@ -2341,7 +2394,7 @@ async def dub_transcribe(job_id: str, num_speakers: Optional[int] = None): heuristic when pyannote is unavailable. None → auto-detect. """ num_speakers = _clamp_num_speakers(num_speakers) - job = _get_job(job_id) + job, source_snapshot = await asyncio.to_thread(_transcription_snapshot, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") # Same as the streaming preflight: the only use of the TTS core here is the @@ -2484,20 +2537,15 @@ def _transcribe(): # retry cannot overlap it (#1669). segments_result = await run_transcribe_guarded(_gpu_pool, _transcribe, what="Dub") except asyncio.CancelledError: - with dub_pipeline._dub_jobs_lock: - job["aborted"] = True + await asyncio.to_thread(_mark_job_aborted, job) raise if job.get("aborted"): raise HTTPException(status_code=499, detail="Transcription aborted") from services.segmentation import deduplicate_chunk_segments segments_result = deduplicate_chunk_segments(segments_result) - with dub_pipeline._dub_jobs_lock: - if _get_job(job_id) is not job: - raise HTTPException(status_code=404, detail="Job not found") - source_updates["segments"] = segments_result - job.update(source_updates) - source_lang = job.get("source_lang") - _save_job(job_id, job) + source_updates["segments"] = segments_result + await _save_transcription(job_id, job, source_snapshot, source_updates) + source_lang = job.get("source_lang") return { "job_id": job_id, "segments": segments_result, diff --git a/backend/api/routers/dub_export.py b/backend/api/routers/dub_export.py index 5792cabbc..2c41a2ada 100644 --- a/backend/api/routers/dub_export.py +++ b/backend/api/routers/dub_export.py @@ -356,7 +356,7 @@ async def cancel_task(task_id: str): @router.get("/dub/tracks/{job_id}") async def dub_list_tracks(job_id: str): - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") return {"tracks": job.get("dubbed_tracks", {})} @@ -374,7 +374,7 @@ async def dub_segments_text(job_id: str, lang: str = Query(...)): transcript. Empty map when the job predates segments_i18n or the track was never generated — the client keeps whatever it has. """ - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") i18n = job.get("segments_i18n") or {} @@ -686,7 +686,7 @@ async def dub_download( # path or ffmpeg argv (export dir, retime work path, slice paths). Real # job ids are short uuid slices — alnum/hyphen/underscore only. job_dir = _job_dir_or_400(job_id) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") @@ -863,8 +863,7 @@ async def dub_download( ) # A fresh export is a fresh user intent — clear any sticky abort flag # from a previous /dub/abort so it can't kill this run's first batch. - with _dub_jobs_lock: - job.pop("aborted", None) + await asyncio.to_thread(_clear_export_abort, job) # realpath-normalised + containment-checked inline at the sink (the # file's established pattern — CodeQL does not track the guard # through a helper's return value). @@ -890,8 +889,7 @@ async def dub_download( raise HTTPException(status_code=409, detail="Export aborted") from core.failure import build_failure retime_warning = build_failure(e, stage="video-retime", include_diagnostic=False) - with _dub_jobs_lock: - job["last_export_warning"] = {"type": "video_retime_fallback", **retime_warning} + await asyncio.to_thread(_set_export_warning, job, retime_warning) logger.exception( "Smart Fit video retime failed for job %s — exporting " "without per-segment retime", @@ -1120,7 +1118,7 @@ async def dub_download( @router.api_route("/dub/media/{job_id}", methods=["GET", "HEAD"]) async def dub_get_media(job_id: str, request: Request): _job_dir_or_400(job_id) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") video_path = _dub_artifact(job["video_path"], job_id, missing_detail="Media file not found") @@ -1173,7 +1171,7 @@ async def dub_preview_video( # boundary check as dub_download. job_dir = _job_dir_or_400(job_id) lang = _safe_lang_or_400(lang) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") @@ -1245,8 +1243,7 @@ async def _mux_preview(): smart_track_dur = float( retime_entry.get("total_duration") or track_info.get("duration") or 0.0 ) - with _dub_jobs_lock: - job.pop("aborted", None) # fresh user intent — clear sticky abort + await asyncio.to_thread(_clear_export_abort, job) # realpath-normalised + containment-checked inline at the sink # (same pattern as preview_path above — _base is the realpath # of DUB_DIR from the top of this endpoint). @@ -1443,7 +1440,7 @@ async def dub_get_onsets(job_id: str): newer than the cache (e.g. re-ingest into the same job dir). """ job_dir = _job_dir_or_400(job_id) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") @@ -1508,7 +1505,7 @@ async def dub_prosody_mirror(job_id: str, req: ProsodyMirrorRequest): about the job changes. """ _job_dir_or_400(job_id) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") @@ -1544,7 +1541,7 @@ async def dub_prosody_mirror(job_id: str, req: ProsodyMirrorRequest): async def dub_get_thumb(job_id: str): """Serve the extracted dub video thumbnail (jpg). 404 if not generated.""" job_dir = _job_dir_or_400(job_id) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") # Resolve under DUB_DIR to prevent traversal. @@ -1556,7 +1553,7 @@ async def dub_get_thumb(job_id: str): @router.get("/dub/audio/{job_id}") async def dub_get_audio(job_id: str): _job_dir_or_400(job_id) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") audio = _dub_artifact(job.get("audio_path"), job_id, missing_detail="Audio file not found") @@ -1608,7 +1605,7 @@ def _existing_segment_artifact(job_id: str, candidate_ids: list) -> str | None: async def dub_preview_segment(job_id: str, segment_index: int, lang: str = Query(None)): _job_dir_or_400(job_id) lang = _safe_lang_or_400(lang) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") # Resolve the stable-id-named WAV via the render manifest — language-keyed @@ -1683,6 +1680,16 @@ def _qc_source_segments(job: dict, lang: str, segments: list[dict]) -> list[dict }) +def _clear_export_abort(job): + with _dub_jobs_lock: + job.pop("aborted", None) + + +def _set_export_warning(job, warning): + with _dub_jobs_lock: + job["last_export_warning"] = {"type": "video_retime_fallback", **warning} + + @router.post("/dub/qc/{job_id}") async def dub_qc_pass(job_id: str, lang: str = Query(None), drift_threshold: float = Query(0.5)): """Re-recognize the dubbed audio and flag lines whose recognized text @@ -1696,20 +1703,22 @@ async def dub_qc_pass(job_id: str, lang: str = Query(None), drift_threshold: flo _job_dir_or_400(job_id) lang = _safe_lang_or_400(lang) - job = _get_job(job_id) - if not job: - raise HTTPException(status_code=404, detail="Job not found") - tracks = job.get("dubbed_tracks", {}) - if not tracks: - raise HTTPException(status_code=400, detail="No dubbed audio track generated yet") - # Resolve the text against the same track chosen for recognition, including - # the legacy first-track fallback when no matching language is requested. - selected_lang = lang if lang and lang in tracks else next(iter(tracks)) - wav_path = _dub_artifact(tracks[selected_lang].get("path"), job_id, missing_detail="Dubbed audio file not found") - live_job = job - with _dub_jobs_lock: - job = _qc_snapshot(live_job, selected_lang) - audio_revision = _qc_audio_revision(wav_path) + + def snapshot_track(): + with _dub_jobs_lock: + job = _get_job(job_id) + if not job: + raise HTTPException(status_code=404, detail="Job not found") + tracks = job.get("dubbed_tracks", {}) + if not tracks: + raise HTTPException(status_code=400, detail="No dubbed audio track generated yet") + # Resolve the text against the same track chosen for recognition, including + # the legacy first-track fallback when no matching language is requested. + selected_lang = lang if lang and lang in tracks else next(iter(tracks)) + wav_path = _dub_artifact(tracks[selected_lang].get("path"), job_id, missing_detail="Dubbed audio file not found") + return job, _qc_snapshot(job, selected_lang), selected_lang, wav_path, _qc_audio_revision(wav_path) + + live_job, job, selected_lang, wav_path, audio_revision = await asyncio.to_thread(snapshot_track) tracks = job["dubbed_tracks"] segments = job.get("segments") or [] if not segments: @@ -1791,34 +1800,37 @@ def _recognize(): # Scoring uses the frozen inputs, but a render/edit/deletion can complete # while ASR runs. Publish only into that same revision, under the same lock # as history deletion so QC cannot resurrect a withdrawn job. - with _dub_jobs_lock: - current = _get_job(job_id) - if (current is not live_job - or _qc_snapshot(current, selected_lang) != job - or _qc_audio_revision(wav_path) != audio_revision): - raise HTTPException(status_code=409, detail={ - "code": "dub_qc_track_changed", - "message": "The dub changed during QC. Run QC again on the current track.", - }) - # Annotate each segment (non-destructive — content text untouched). - by_id = {q.seg_id: q for q in scored} - for i, s in enumerate(segments): - sid = str(seg_ids[i]) if i < len(seg_ids) else str(s.get("id", i)) - q = by_id.get(sid) - if q is None: - continue - s["qc_drift"] = q.drift - s["qc_flagged"] = q.flagged - s["qc_recognized"] = q.recognized_text - if q.new_start is not None: - s["qc_measured_start"] = q.new_start - s["qc_measured_end"] = q.new_end - current["segments"] = segments - if not put_and_save_job(job_id, current): - raise HTTPException(status_code=409, detail={ - "code": "dub_qc_track_changed", - "message": "The dub changed during QC. Run QC again on the current track.", - }) + def publish_qc(): + with _dub_jobs_lock: + current = _get_job(job_id) + if (current is not live_job + or _qc_snapshot(current, selected_lang) != job + or _qc_audio_revision(wav_path) != audio_revision): + raise HTTPException(status_code=409, detail={ + "code": "dub_qc_track_changed", + "message": "The dub changed during QC. Run QC again on the current track.", + }) + # Annotate each segment (non-destructive — content text untouched). + by_id = {q.seg_id: q for q in scored} + for i, s in enumerate(segments): + sid = str(seg_ids[i]) if i < len(seg_ids) else str(s.get("id", i)) + q = by_id.get(sid) + if q is None: + continue + s["qc_drift"] = q.drift + s["qc_flagged"] = q.flagged + s["qc_recognized"] = q.recognized_text + if q.new_start is not None: + s["qc_measured_start"] = q.new_start + s["qc_measured_end"] = q.new_end + current["segments"] = segments + if not put_and_save_job(job_id, current): + raise HTTPException(status_code=409, detail={ + "code": "dub_qc_track_changed", + "message": "The dub changed during QC. Run QC again on the current track.", + }) + + await asyncio.to_thread(publish_qc) flagged = [q for q in scored if q.flagged] payload = json.dumps({"event": "qc_done", "engine": engine_id, @@ -1854,7 +1866,7 @@ async def dub_download_audio( ): job_dir = _existing_job_dir_or_404(job_id) lang = _safe_lang_or_400(lang) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") @@ -1963,7 +1975,7 @@ async def dub_export_srt( ): _job_dir_or_400(job_id) lang = _safe_lang_or_400(lang) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") @@ -2014,7 +2026,7 @@ async def dub_export_vtt( ): _job_dir_or_400(job_id) lang = _safe_lang_or_400(lang) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") @@ -2074,7 +2086,7 @@ async def dub_export_ass( """ _job_dir_or_400(job_id) lang = _safe_lang_or_400(lang) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") @@ -2110,7 +2122,7 @@ async def dub_export_segments_zip(job_id: str, lang: str = Query(None)): import zipfile _job_dir_or_400(job_id) lang = _safe_lang_or_400(lang) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") @@ -2154,7 +2166,7 @@ async def dub_download_mp3( ): job_dir = _existing_job_dir_or_404(job_id) lang = _safe_lang_or_400(lang) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") @@ -2238,7 +2250,7 @@ async def dub_export_stems(job_id: str, lang: str = Query(None)): import zipfile _job_dir_or_400(job_id) lang = _safe_lang_or_400(lang) - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") diff --git a/backend/api/routers/dub_generate.py b/backend/api/routers/dub_generate.py index 42f631db1..1d8d577af 100644 --- a/backend/api/routers/dub_generate.py +++ b/backend/api/routers/dub_generate.py @@ -7,7 +7,6 @@ import asyncio import copy import contextlib -import contextvars import shutil import tempfile import torch @@ -83,25 +82,7 @@ def _install_dub_artifacts(staged: dict[str, str]): async def _finish_publication(publish): """Keep blocking commit work off-loop and retain its files until it settles.""" - # Use an executor Future, not a detached Task: cancellation (including loop - # shutdown cancelling all Tasks) must not mark this work done while its - # thread is still installing or rolling back files in the staging directory. - pending = asyncio.get_running_loop().run_in_executor( - None, contextvars.copy_context().run, publish, - ) - try: - return await asyncio.shield(pending) - except (asyncio.CancelledError, GeneratorExit): - while not pending.done(): - try: - await asyncio.shield(pending) - except asyncio.CancelledError: - continue # repeated cancellation still cannot release live files - except Exception: - break # rollback finished; preserve the caller's cancellation - if not pending.cancelled(): - pending.exception() # observe a failure even while the caller exits - raise + return await dub_pipeline.run_job_operation(publish) class _RemoteDubBackend: @@ -634,7 +615,7 @@ def _decode_remote_dub(result: gpu_gateway.RemoteResult) -> dict[int, str]: @router.post("/dub/generate/{job_id}") async def dub_generate(job_id: str, req: DubRequest): """Adds a dub generation job to the async batch task pool.""" - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException( status_code=404, @@ -2056,7 +2037,7 @@ def _render_batch() -> list[torch.Tensor]: # Reject changed source before writing the staged track. The worker # revalidates under the job lock immediately before publication. - if _get_job(job_id) is not job or _render_source_segments(job) != source_segments: + if await asyncio.to_thread(_get_job, job_id) is not job or _render_source_segments(job) != source_segments: yield f"data: {json.dumps({'type': 'error', 'error_code': 'dub_source_changed', 'error': 'Subtitles changed during generation. Generate again to use the current subtitles.'})}\n\n" return _t_save_0 = time.perf_counter() @@ -2220,7 +2201,7 @@ def publish(): f" regen={len(regen_only)}" if regen_only is not None else "", ) - yield f"data: {json.dumps({'type': 'done', 'segments_processed': total, 'language_code': lang_code, 'tracks': list(job['dubbed_tracks'].keys()), 'sync_scores': sync_scores, 'fit_status': fit_status, 'timing_strategy': strategy, 'seg_hashes': job.get('seg_hashes', {}), 'seg_num_step': job.get('seg_num_step', {})})}\n\n" + yield f"data: {json.dumps({'type': 'done', 'committed': True, 'segments_processed': total, 'language_code': lang_code, 'tracks': list(job['dubbed_tracks'].keys()), 'sync_scores': sync_scores, 'fit_status': fit_status, 'timing_strategy': strategy, 'seg_hashes': job.get('seg_hashes', {}), 'seg_num_step': job.get('seg_num_step', {})})}\n\n" async def _stream(task_id): job_dir = os.path.join(DUB_DIR, job_id) @@ -2271,7 +2252,7 @@ async def preview_segment(job_id: str, req: SegmentPreviewRequest): so it carries the same invisible provenance mark as every other producer (#1169 — "no watermark" here used to be an exemption). """ - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") diff --git a/backend/api/routers/dub_translate.py b/backend/api/routers/dub_translate.py index 0acddfd82..7a0f029a7 100644 --- a/backend/api/routers/dub_translate.py +++ b/backend/api/routers/dub_translate.py @@ -342,8 +342,11 @@ def _resolve_translation_context(req, client, model_name: str, timeout: float, ctx = {**ctx, "fingerprint": fp} if job is not None: try: - job.setdefault("translation_context", {})[req.target_lang] = ctx - _save_job(req.job_id, job) + from services.dub_pipeline import _dub_jobs_lock + with _dub_jobs_lock: + if _get_job(req.job_id) is job: + job.setdefault("translation_context", {})[req.target_lang] = ctx + _save_job(req.job_id, job) except Exception: # noqa: BLE001 — persistence is best-effort logger.debug("translation context persist skipped", exc_info=True) return ctx @@ -392,7 +395,7 @@ async def dub_translate(req: TranslateRequest): lang_code = TRANSLATE_CODES.get(req.target_lang, req.target_lang) api_key = os.environ.get("TRANSLATE_API_KEY", "") loop = asyncio.get_running_loop() - src_lang = _resolve_source_lang(req) + src_lang = await asyncio.to_thread(_resolve_source_lang, req) # Offline NLLB Transformer Translation if provider == "nllb": @@ -1047,7 +1050,7 @@ async def _apply_condense_pass(rows, req, loop) -> None: calib = None if getattr(req, "job_id", None): - job = _get_job(req.job_id) + job = await asyncio.to_thread(_get_job, req.job_id) if job: calib = calibration_from_job(job, req.target_lang) source_by_id = {str(s.id): s.text for s in req.segments} @@ -1086,7 +1089,7 @@ async def _one(row): async def _finalize_duration_plan(rows, req, loop) -> None: """Stamp plan verdicts on the FINAL row texts, then (opt-in) condense.""" - _stamp_duration_plan(rows, req) + await asyncio.to_thread(_stamp_duration_plan, rows, req) if getattr(req, "condense", False): await _apply_condense_pass(rows, req, loop) diff --git a/backend/api/routers/engines.py b/backend/api/routers/engines.py index d631db2ce..87f0e4696 100644 --- a/backend/api/routers/engines.py +++ b/backend/api/routers/engines.py @@ -329,7 +329,7 @@ def argos_pack_status(request: ArgosPackRequest): dependencies=[Depends(require_admin)], ) async def install_argos_packs(request: ArgosPackRequest): - source, targets = _argos_pack_request(request) + source, targets = await asyncio.to_thread(_argos_pack_request, request) try: return await asyncio.to_thread( translation_engines.install_argos_packs, diff --git a/backend/api/routers/tools.py b/backend/api/routers/tools.py index 70909a0d2..f96def368 100644 --- a/backend/api/routers/tools.py +++ b/backend/api/routers/tools.py @@ -185,7 +185,7 @@ async def analyse_video_context(job_id: str): job_dir = resolve_within(DUB_DIR, job_id) except UnsafePath as exc: raise HTTPException(status_code=400, detail="Invalid job id") from exc - job = _get_job(job_id) + job = await asyncio.to_thread(_get_job, job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") diff --git a/backend/core/tasks.py b/backend/core/tasks.py index 703bf58ae..d22782bd0 100644 --- a/backend/core/tasks.py +++ b/backend/core/tasks.py @@ -32,6 +32,19 @@ def _stream_failure(update): return detail if isinstance(detail, str) and detail else "Task failed" +def _stream_committed(update): + """A completed durable commit wins over a cancellation that arrived late.""" + if isinstance(update, bytes): + update = update.decode("utf-8", errors="replace") + if not isinstance(update, str): + return False + try: + payload = json.loads("\n".join(line[5:].strip() for line in update.splitlines() if line.startswith("data:"))) + except (ValueError, TypeError): + return False + return isinstance(payload, dict) and payload.get("type") == "done" and payload.get("committed") is True + + class TaskManager: """In-memory task dispatcher with SQLite-backed metadata. @@ -141,7 +154,7 @@ async def worker(self): if inspect.isasyncgen(res): async with aclosing(res): async for update in res: - if t.get("cancelled"): + if t.get("cancelled") and not _stream_committed(update): await self._push_event(task_id, f"data: {json.dumps({'type': 'cancelled'})}\n\n") t["status"] = "cancelled" try: job_store.mark_cancelled(task_id) diff --git a/backend/services/dub_pipeline.py b/backend/services/dub_pipeline.py index 26392d867..dbb9859ec 100644 --- a/backend/services/dub_pipeline.py +++ b/backend/services/dub_pipeline.py @@ -27,6 +27,8 @@ from __future__ import annotations import asyncio +import contextvars +import functools import hashlib import json import logging @@ -303,6 +305,32 @@ def find_cached_job(content_hash: str, exclude_job_id: str) -> Optional[dict]: # ── Job state (in-memory + SQLite fallback) ──────────────────────────────── +async def run_job_operation(operation, *args, **kwargs): + """Wait for blocking job work to settle before cancellation can clean up. + + The shared lock may be held during audio replacement and SQLite writes. + Async callers must acquire it in a worker, retaining that worker through + cancellation so it cannot publish after the caller removes its files. + """ + work = functools.partial(operation, *args, **kwargs) + pending = asyncio.get_running_loop().run_in_executor( + None, contextvars.copy_context().run, work, + ) + try: + return await asyncio.shield(pending) + except (asyncio.CancelledError, GeneratorExit): + while not pending.done(): + try: + await asyncio.shield(pending) + except asyncio.CancelledError: + continue + except Exception: + break + if not pending.cancelled(): + pending.exception() + raise + + def get_job(job_id: str) -> Optional[dict]: """Look up a job. Checks the in-memory cache first, then falls back to `dub_history.job_data` so saved projects still resolve after restart. @@ -448,8 +476,8 @@ def merge_and_save_job( WAL, but up to sqlite3's 5 s default busy timeout if another writer is holding the write lock. The alternative — releasing the lock before the write — is the resurrection race this exists to close, so a rare latency - blip is the better trade. No locked region here calls another locked - function, so the plain (non-reentrant) ``_dub_jobs_lock`` cannot deadlock. + blip is the better trade. Async callers dispatch this operation to a + worker; the reentrant job lock also covers the nested save. """ with _dub_jobs_lock: job = _dub_jobs.get(job_id) @@ -1297,8 +1325,8 @@ async def ingest_pipeline( input_type = (source.get("input_type") or "video").lower() # Declare the run so a "clear history" arriving before this job's first # persistence can still withdraw it (#1252 review). - begin_ingest(job_id) try: + await run_job_operation(begin_ingest, job_id) if source.get("kind") == "url": url = source["url"] fetch_subs = bool(source.get("fetch_subs")) @@ -1500,7 +1528,7 @@ def _yt_progress(d: dict) -> None: "input_type": input_type, "source_lang_override": source.get("source_lang"), } - if not put_and_save_job( + if not await run_job_operation(put_and_save_job, job_id, full_job, filename=filename, duration=dur, content_hash=content_hash, ): logger.info("Dub job %s was deleted during ingest — discarding its result", log_safe(job_id)) @@ -1529,7 +1557,7 @@ def _yt_progress(d: dict) -> None: "input_type": input_type, "source_lang_override": source.get("source_lang"), } - if not put_and_save_job( + if not await run_job_operation(put_and_save_job, job_id, partial, filename=filename, duration=dur, content_hash=content_hash, ): logger.info("Dub job %s was deleted during ingest — discarding its result", log_safe(job_id)) @@ -1635,7 +1663,7 @@ def _yt_progress(d: dict) -> None: # Merge and persist as one step: a delete landing BETWEEN them # would remove the row and then have it written straight back, so # the dub the user deleted reappears in history (#1252 review). - if not merge_and_save_job( + if not await run_job_operation(merge_and_save_job, job_id, { "vocals_path": vocals_path, @@ -1674,6 +1702,6 @@ def _yt_progress(d: dict) -> None: # cancellation; never copy it into the project/job directory. cookie_file = source.get("cookie_file") _delete_cookie_export(cookie_file) - end_ingest(job_id) + await run_job_operation(end_ingest, job_id) with _active_procs_lock: _active_procs.pop(job_id, None) diff --git a/docs/electron-dubbing.md b/docs/electron-dubbing.md index 78ccfd1ae..d625cade0 100644 --- a/docs/electron-dubbing.md +++ b/docs/electron-dubbing.md @@ -437,11 +437,16 @@ reference. Source validation, file installation and the strict database save share the job lock; a failed save rolls back the audio replacement. Once that publication transaction has started, cancellation waits for its commit or rollback before removing staging files. A completed -commit stays published even if cancellation arrives during it. Deleting a job +commit stays published and reports a completed task even if cancellation arrives +during it. Async reads, export/QC updates and ingest persistence wait for the +shared lock in workers, so they cannot prevent the event loop from handling +cancellation while publication is waiting on disk. Deleting a job serializes with publication and cannot leave a resurrected history row. Subtitle imports re-read the current job when applying uploaded cues. Imports, caption cleanup, transcription source updates, and QC share the publication lock, so an edit arriving during publication is applied afterward. Transcription keeps new source fields private until completion and refuses to publish into a -deleted or replaced job. Concurrently completed dub tracks are preserved. +deleted or replaced job. If source text, timing or speaker assignments change +during transcription, the result is rejected with a localized message and the +newer edits remain intact. Concurrently completed dub tracks are preserved. diff --git a/electron/src/renderer/src/i18n/locales/ar.json b/electron/src/renderer/src/i18n/locales/ar.json index c49228596..58544705c 100644 --- a/electron/src/renderer/src/i18n/locales/ar.json +++ b/electron/src/renderer/src/i18n/locales/ar.json @@ -822,6 +822,7 @@ "qc_track_changed": "تغيرت الدبلجة أثناء فحص الجودة. أعد الفحص على المسار الحالي.", "qc_identity_missing": "هويات المقاطع غير واضحة. أعد إنشاء هذا المسار بمعرّفات فريدة للمقاطع.", "source_changed": "تغيّرت الترجمة أثناء التوليد. أعد التوليد لاستخدام الترجمة الحالية.", + "transcription_source_changed": "تغيّرت الترجمة النصية أثناء التفريغ. تم الاحتفاظ بتعديلاتك.", "paste_translation_btn": "لصق ترجمة", "paste_translation_title": "لصق ترجمة", "paste_translation_desc": "الصق ترجمة أعددتها في مكان آخر (ChatGPT أو DeepL أو مترجم بشري). ستُطابَق مع المقاطع الموجودة لديك — تبقى التوقيتات والنص الأصلي دون تغيير.", diff --git a/electron/src/renderer/src/i18n/locales/de.json b/electron/src/renderer/src/i18n/locales/de.json index badd3ddab..2d8253155 100644 --- a/electron/src/renderer/src/i18n/locales/de.json +++ b/electron/src/renderer/src/i18n/locales/de.json @@ -812,6 +812,7 @@ "qc_track_changed": "Die Synchronisation wurde während der Qualitätsprüfung geändert. Prüfe die aktuelle Spur erneut.", "qc_identity_missing": "Die Segmentzuordnung ist uneindeutig. Erzeuge diese Spur mit eindeutigen Segment-IDs neu.", "source_changed": "Die Untertitel wurden während der Generierung geändert. Generiere erneut, um die aktuellen Untertitel zu verwenden.", + "transcription_source_changed": "Die Untertitel wurden während der Transkription geändert. Deine Änderungen wurden beibehalten.", "paste_translation_btn": "Übersetzung einfügen", "paste_translation_title": "Übersetzung einfügen", "paste_translation_desc": "Füge eine anderswo erstellte Übersetzung ein (ChatGPT, DeepL, ein menschlicher Übersetzer). Sie wird den vorhandenen Segmenten zugeordnet — Timings und Originaltranskript bleiben unverändert.", diff --git a/electron/src/renderer/src/i18n/locales/en.json b/electron/src/renderer/src/i18n/locales/en.json index 6e32596df..518607e60 100644 --- a/electron/src/renderer/src/i18n/locales/en.json +++ b/electron/src/renderer/src/i18n/locales/en.json @@ -968,6 +968,7 @@ "qc_track_changed": "The dub changed during the quality check. Run the check again on the current track.", "qc_identity_missing": "Segment identities are ambiguous. Regenerate this track with unique segment IDs.", "source_changed": "Subtitles changed during generation. Generate again to use the current subtitles.", + "transcription_source_changed": "Subtitles changed during transcription. Your edits were kept.", "export_btn": "Export…", "prep_stop": "Stop", "install_progress": "Installing {{engine}}…", diff --git a/electron/src/renderer/src/i18n/locales/es.json b/electron/src/renderer/src/i18n/locales/es.json index fa77fe26c..a0981ae63 100644 --- a/electron/src/renderer/src/i18n/locales/es.json +++ b/electron/src/renderer/src/i18n/locales/es.json @@ -814,6 +814,7 @@ "qc_track_changed": "El doblaje cambió durante la comprobación de calidad. Repite la comprobación en la pista actual.", "qc_identity_missing": "La identidad de los segmentos es ambigua. Regenera esta pista con identificadores de segmento únicos.", "source_changed": "Los subtítulos cambiaron durante la generación. Genera de nuevo para usar los subtítulos actuales.", + "transcription_source_changed": "Los subtítulos cambiaron durante la transcripción. Se conservaron tus cambios.", "paste_translation_btn": "Pegar traducción", "paste_translation_title": "Pegar una traducción", "paste_translation_desc": "Pega una traducción hecha en otro sitio (ChatGPT, DeepL, un traductor humano). Se asigna a los segmentos que ya tienes: los tiempos y la transcripción original no cambian.", diff --git a/electron/src/renderer/src/i18n/locales/fr.json b/electron/src/renderer/src/i18n/locales/fr.json index 77fa7de0c..1a1d01ce9 100644 --- a/electron/src/renderer/src/i18n/locales/fr.json +++ b/electron/src/renderer/src/i18n/locales/fr.json @@ -814,6 +814,7 @@ "qc_track_changed": "Le doublage a changé pendant le contrôle qualité. Relancez le contrôle sur la piste actuelle.", "qc_identity_missing": "Les segments ne sont pas identifiés de façon univoque. Régénérez cette piste avec des identifiants de segment uniques.", "source_changed": "Les sous-titres ont changé pendant la génération. Relancez la génération pour utiliser les sous-titres actuels.", + "transcription_source_changed": "Les sous-titres ont changé pendant la transcription. Vos modifications ont été conservées.", "paste_translation_btn": "Coller une traduction", "paste_translation_title": "Coller une traduction", "paste_translation_desc": "Collez une traduction réalisée ailleurs (ChatGPT, DeepL, un traducteur humain). Elle est appliquée aux segments existants — les timings et la transcription d'origine restent intacts.", diff --git a/electron/src/renderer/src/i18n/locales/hi.json b/electron/src/renderer/src/i18n/locales/hi.json index 264252b3c..4b940aebd 100644 --- a/electron/src/renderer/src/i18n/locales/hi.json +++ b/electron/src/renderer/src/i18n/locales/hi.json @@ -812,6 +812,7 @@ "qc_track_changed": "गुणवत्ता जाँच के दौरान डबिंग बदल गई। मौजूदा ट्रैक पर जाँच दोबारा चलाएँ।", "qc_identity_missing": "सेगमेंट की पहचान स्पष्ट नहीं है। हर सेगमेंट को अलग आईडी देकर इस ट्रैक को फिर से जनरेट करें।", "source_changed": "जनरेशन के दौरान सबटाइटल बदल गए। मौजूदा सबटाइटल इस्तेमाल करने के लिए फिर से जनरेट करें।", + "transcription_source_changed": "ट्रांसक्रिप्शन के दौरान उपशीर्षक बदल गए। आपके बदलाव सुरक्षित रखे गए हैं।", "paste_translation_btn": "अनुवाद चिपकाएँ", "paste_translation_title": "अनुवाद चिपकाएँ", "paste_translation_desc": "कहीं और तैयार किया गया अनुवाद चिपकाएँ (ChatGPT, DeepL, कोई मानव अनुवादक)। यह आपके मौजूदा सेगमेंट पर लागू होगा — टाइमिंग और मूल ट्रांसक्रिप्ट अछूते रहते हैं।", diff --git a/electron/src/renderer/src/i18n/locales/id.json b/electron/src/renderer/src/i18n/locales/id.json index ad9bfb9bb..15a9fd96e 100644 --- a/electron/src/renderer/src/i18n/locales/id.json +++ b/electron/src/renderer/src/i18n/locales/id.json @@ -814,6 +814,7 @@ "qc_track_changed": "Sulih suara berubah selama pemeriksaan kualitas. Jalankan kembali pemeriksaan pada trek saat ini.", "qc_identity_missing": "Identitas segmen tidak jelas. Buat ulang trek ini dengan ID segmen yang unik.", "source_changed": "Subtitel berubah selama pembuatan. Buat ulang untuk menggunakan subtitel saat ini.", + "transcription_source_changed": "Subtitel berubah selama transkripsi. Perubahan Anda dipertahankan.", "paste_translation_btn": "Tempel terjemahan", "paste_translation_title": "Tempel terjemahan", "paste_translation_desc": "Tempel terjemahan yang kamu buat di tempat lain (ChatGPT, DeepL, penerjemah manusia). Terjemahan itu dipetakan ke segmen yang sudah ada — pewaktuan dan transkrip asli tidak diubah.", diff --git a/electron/src/renderer/src/i18n/locales/it.json b/electron/src/renderer/src/i18n/locales/it.json index a44e928bf..1406b08ed 100644 --- a/electron/src/renderer/src/i18n/locales/it.json +++ b/electron/src/renderer/src/i18n/locales/it.json @@ -814,6 +814,7 @@ "qc_track_changed": "Il doppiaggio è cambiato durante il controllo qualità. Ripeti il controllo sulla traccia attuale.", "qc_identity_missing": "Le identità dei segmenti sono ambigue. Rigenera questa traccia con identificativi di segmento univoci.", "source_changed": "I sottotitoli sono cambiati durante la generazione. Genera di nuovo per usare i sottotitoli attuali.", + "transcription_source_changed": "I sottotitoli sono cambiati durante la trascrizione. Le tue modifiche sono state mantenute.", "paste_translation_btn": "Incolla traduzione", "paste_translation_title": "Incolla una traduzione", "paste_translation_desc": "Incolla una traduzione prodotta altrove (ChatGPT, DeepL, un traduttore umano). Viene mappata sui segmenti che hai già: tempi e trascrizione originale restano invariati.", diff --git a/electron/src/renderer/src/i18n/locales/ja.json b/electron/src/renderer/src/i18n/locales/ja.json index 244944bd6..0a5007c4e 100644 --- a/electron/src/renderer/src/i18n/locales/ja.json +++ b/electron/src/renderer/src/i18n/locales/ja.json @@ -814,6 +814,7 @@ "qc_track_changed": "品質チェック中に吹き替えが変更されました。現在のトラックで再度チェックしてください。", "qc_identity_missing": "セグメントを一意に識別できません。各セグメントに固有のIDを付けて、このトラックを再生成してください。", "source_changed": "生成中に字幕が変更されました。現在の字幕を使用するには、もう一度生成してください。", + "transcription_source_changed": "文字起こし中に字幕が変更されました。編集内容は保持されています。", "paste_translation_btn": "翻訳を貼り付け", "paste_translation_title": "翻訳を貼り付け", "paste_translation_desc": "他所で用意した翻訳(ChatGPT、DeepL、人間の翻訳者)を貼り付けます。既存のセグメントに割り当てられ、タイミングと元の文字起こしはそのまま残ります。", diff --git a/electron/src/renderer/src/i18n/locales/ko.json b/electron/src/renderer/src/i18n/locales/ko.json index 00e0853cc..03d21deab 100644 --- a/electron/src/renderer/src/i18n/locales/ko.json +++ b/electron/src/renderer/src/i18n/locales/ko.json @@ -812,6 +812,7 @@ "qc_track_changed": "품질 검사 중 더빙이 변경되었습니다. 현재 트랙에서 검사를 다시 실행하세요.", "qc_identity_missing": "세그먼트를 명확하게 구분할 수 없습니다. 각 세그먼트에 고유한 ID를 지정하여 이 트랙을 다시 생성하세요.", "source_changed": "생성 중에 자막이 변경되었습니다. 현재 자막을 사용하려면 다시 생성하세요.", + "transcription_source_changed": "음성을 텍스트로 변환하는 동안 자막이 변경되었습니다. 수정 내용은 유지되었습니다.", "paste_translation_btn": "번역 붙여넣기", "paste_translation_title": "번역 붙여넣기", "paste_translation_desc": "다른 곳에서 만든 번역(ChatGPT, DeepL, 사람 번역가)을 붙여넣으세요. 이미 있는 세그먼트에 매핑되며 타이밍과 원본 전사는 그대로 유지됩니다.", diff --git a/electron/src/renderer/src/i18n/locales/nl.json b/electron/src/renderer/src/i18n/locales/nl.json index 23c0c3605..1a0f106bc 100644 --- a/electron/src/renderer/src/i18n/locales/nl.json +++ b/electron/src/renderer/src/i18n/locales/nl.json @@ -812,6 +812,7 @@ "qc_track_changed": "De nasynchronisatie is tijdens de kwaliteitscontrole gewijzigd. Voer de controle opnieuw uit op het huidige spoor.", "qc_identity_missing": "De segmenten zijn niet eenduidig te identificeren. Genereer dit spoor opnieuw met unieke segment-ID’s.", "source_changed": "De ondertitels zijn gewijzigd tijdens het genereren. Genereer opnieuw om de huidige ondertitels te gebruiken.", + "transcription_source_changed": "De ondertitels zijn tijdens de transcriptie gewijzigd. Je wijzigingen zijn behouden.", "paste_translation_btn": "Vertaling plakken", "paste_translation_title": "Een vertaling plakken", "paste_translation_desc": "Plak een elders gemaakte vertaling (ChatGPT, DeepL, een menselijke vertaler). Die wordt op je bestaande segmenten toegepast — timings en het originele transcript blijven ongewijzigd.", diff --git a/electron/src/renderer/src/i18n/locales/pl.json b/electron/src/renderer/src/i18n/locales/pl.json index 651ee69d4..07b002966 100644 --- a/electron/src/renderer/src/i18n/locales/pl.json +++ b/electron/src/renderer/src/i18n/locales/pl.json @@ -816,6 +816,7 @@ "qc_track_changed": "Dubbing zmienił się podczas kontroli jakości. Uruchom kontrolę ponownie dla bieżącej ścieżki.", "qc_identity_missing": "Tożsamość segmentów jest niejednoznaczna. Wygeneruj tę ścieżkę ponownie z unikatowymi identyfikatorami segmentów.", "source_changed": "Napisy zmieniły się podczas generowania. Wygeneruj ponownie, aby użyć aktualnych napisów.", + "transcription_source_changed": "Napisy zmieniły się podczas transkrypcji. Twoje zmiany zostały zachowane.", "paste_translation_btn": "Wklej tłumaczenie", "paste_translation_title": "Wklej tłumaczenie", "paste_translation_desc": "Wklej tłumaczenie przygotowane gdzie indziej (ChatGPT, DeepL, tłumacz). Zostanie dopasowane do istniejących segmentów — czasy i oryginalna transkrypcja pozostają bez zmian.", diff --git a/electron/src/renderer/src/i18n/locales/pt.json b/electron/src/renderer/src/i18n/locales/pt.json index 1f2357a51..c62e8743f 100644 --- a/electron/src/renderer/src/i18n/locales/pt.json +++ b/electron/src/renderer/src/i18n/locales/pt.json @@ -814,6 +814,7 @@ "qc_track_changed": "A dublagem mudou durante a verificação de qualidade. Execute a verificação novamente na faixa atual.", "qc_identity_missing": "A identificação dos segmentos é ambígua. Gere esta faixa novamente com identificadores de segmento exclusivos.", "source_changed": "As legendas mudaram durante a geração. Gere novamente para usar as legendas atuais.", + "transcription_source_changed": "As legendas mudaram durante a transcrição. Suas alterações foram mantidas.", "paste_translation_btn": "Colar tradução", "paste_translation_title": "Colar uma tradução", "paste_translation_desc": "Cole uma tradução feita noutro lugar (ChatGPT, DeepL, um tradutor humano). Ela é mapeada nos segmentos que já existem — os tempos e a transcrição original ficam intactos.", diff --git a/electron/src/renderer/src/i18n/locales/ru.json b/electron/src/renderer/src/i18n/locales/ru.json index 357fafb97..19e9572c8 100644 --- a/electron/src/renderer/src/i18n/locales/ru.json +++ b/electron/src/renderer/src/i18n/locales/ru.json @@ -816,6 +816,7 @@ "qc_track_changed": "Дубляж изменился во время проверки качества. Повторите проверку текущей дорожки.", "qc_identity_missing": "Сегменты невозможно однозначно определить. Создайте эту дорожку заново с уникальными идентификаторами сегментов.", "source_changed": "Субтитры изменились во время генерации. Запустите генерацию заново, чтобы использовать текущие субтитры.", + "transcription_source_changed": "Субтитры изменились во время расшифровки. Ваши изменения сохранены.", "paste_translation_btn": "Вставить перевод", "paste_translation_title": "Вставить перевод", "paste_translation_desc": "Вставьте перевод, сделанный в другом месте (ChatGPT, DeepL, живой переводчик). Он ляжет на уже имеющиеся сегменты — тайминги и исходная расшифровка не изменятся.", diff --git a/electron/src/renderer/src/i18n/locales/sv.json b/electron/src/renderer/src/i18n/locales/sv.json index 75aca78cd..4771e3906 100644 --- a/electron/src/renderer/src/i18n/locales/sv.json +++ b/electron/src/renderer/src/i18n/locales/sv.json @@ -814,6 +814,7 @@ "qc_track_changed": "Dubbningen ändrades under kvalitetskontrollen. Kör kontrollen igen på det aktuella spåret.", "qc_identity_missing": "Segmenten kan inte identifieras entydigt. Generera om spåret med unika segment-ID:n.", "source_changed": "Undertexterna ändrades under genereringen. Generera igen för att använda de aktuella undertexterna.", + "transcription_source_changed": "Undertexterna ändrades under transkriberingen. Dina ändringar har sparats.", "paste_translation_btn": "Klistra in översättning", "paste_translation_title": "Klistra in en översättning", "paste_translation_desc": "Klistra in en översättning som gjorts någon annanstans (ChatGPT, DeepL, en mänsklig översättare). Den mappas mot segmenten du redan har — tidkoder och originaltranskriptet rörs inte.", diff --git a/electron/src/renderer/src/i18n/locales/th.json b/electron/src/renderer/src/i18n/locales/th.json index dbbbdbe84..3759f1a89 100644 --- a/electron/src/renderer/src/i18n/locales/th.json +++ b/electron/src/renderer/src/i18n/locales/th.json @@ -814,6 +814,7 @@ "qc_track_changed": "เสียงพากย์เปลี่ยนแปลงระหว่างการตรวจสอบคุณภาพ โปรดตรวจสอบแทร็กปัจจุบันอีกครั้ง", "qc_identity_missing": "ไม่สามารถระบุแต่ละช่วงได้อย่างชัดเจน โปรดสร้างแทร็กนี้ใหม่โดยใช้รหัสที่ไม่ซ้ำกันสำหรับแต่ละช่วง", "source_changed": "คำบรรยายเปลี่ยนแปลงระหว่างการสร้าง โปรดสร้างอีกครั้งเพื่อใช้คำบรรยายปัจจุบัน", + "transcription_source_changed": "คำบรรยายเปลี่ยนแปลงระหว่างการถอดเสียง การแก้ไขของคุณยังคงอยู่", "paste_translation_btn": "วางคำแปล", "paste_translation_title": "วางคำแปล", "paste_translation_desc": "วางคำแปลที่คุณทำไว้จากที่อื่น (ChatGPT, DeepL หรือผู้แปลที่เป็นคน) ระบบจะจับคู่กับเซกเมนต์ที่มีอยู่แล้ว โดยเวลาและถอดความต้นฉบับยังคงเดิม", diff --git a/electron/src/renderer/src/i18n/locales/tr.json b/electron/src/renderer/src/i18n/locales/tr.json index dab75f6fc..ce9b545df 100644 --- a/electron/src/renderer/src/i18n/locales/tr.json +++ b/electron/src/renderer/src/i18n/locales/tr.json @@ -814,6 +814,7 @@ "qc_track_changed": "Kalite kontrolü sırasında dublaj değişti. Geçerli parçada kontrolü yeniden çalıştırın.", "qc_identity_missing": "Bölümler kesin olarak tanımlanamıyor. Bu parçayı benzersiz bölüm kimlikleriyle yeniden oluşturun.", "source_changed": "Oluşturma sırasında altyazılar değişti. Güncel altyazıları kullanmak için yeniden oluşturun.", + "transcription_source_changed": "Transkripsiyon sırasında altyazılar değişti. Düzenlemeleriniz korundu.", "paste_translation_btn": "Çeviriyi yapıştır", "paste_translation_title": "Bir çeviri yapıştır", "paste_translation_desc": "Başka bir yerde hazırladığın çeviriyi yapıştır (ChatGPT, DeepL, bir insan çevirmen). Mevcut segmentlere eşlenir — zamanlamalar ve özgün döküm olduğu gibi kalır.", diff --git a/electron/src/renderer/src/i18n/locales/uk.json b/electron/src/renderer/src/i18n/locales/uk.json index 6a28ea001..eb784a66c 100644 --- a/electron/src/renderer/src/i18n/locales/uk.json +++ b/electron/src/renderer/src/i18n/locales/uk.json @@ -818,6 +818,7 @@ "qc_track_changed": "Дубляж змінився під час перевірки якості. Повторіть перевірку поточної доріжки.", "qc_identity_missing": "Сегменти неможливо однозначно визначити. Створіть цю доріжку заново з унікальними ідентифікаторами сегментів.", "source_changed": "Субтитри змінилися під час генерації. Запустіть генерацію знову, щоб використати поточні субтитри.", + "transcription_source_changed": "Субтитри змінилися під час розшифрування. Ваші зміни збережено.", "paste_translation_btn": "Вставити переклад", "paste_translation_title": "Вставити переклад", "paste_translation_desc": "Вставте переклад, зроблений деінде (ChatGPT, DeepL, живий перекладач). Він накладеться на наявні сегменти — таймінги й початкова транскрипція лишаться незмінними.", diff --git a/electron/src/renderer/src/i18n/locales/vi.json b/electron/src/renderer/src/i18n/locales/vi.json index b726a8524..39c29fdba 100644 --- a/electron/src/renderer/src/i18n/locales/vi.json +++ b/electron/src/renderer/src/i18n/locales/vi.json @@ -814,6 +814,7 @@ "qc_track_changed": "Bản lồng tiếng đã thay đổi trong khi kiểm tra chất lượng. Hãy kiểm tra lại bản âm thanh hiện tại.", "qc_identity_missing": "Không thể xác định rõ từng đoạn. Hãy tạo lại bản âm thanh này với mã định danh riêng cho mỗi đoạn.", "source_changed": "Phụ đề đã thay đổi trong quá trình tạo. Hãy tạo lại để sử dụng phụ đề hiện tại.", + "transcription_source_changed": "Phụ đề đã thay đổi trong khi chép lời. Các chỉnh sửa của bạn được giữ lại.", "paste_translation_btn": "Dán bản dịch", "paste_translation_title": "Dán một bản dịch", "paste_translation_desc": "Dán bản dịch bạn đã làm ở nơi khác (ChatGPT, DeepL, người dịch). Nó sẽ được ánh xạ vào các phân đoạn sẵn có — thời điểm và bản ghi gốc giữ nguyên.", diff --git a/electron/src/renderer/src/i18n/locales/zh-CN.json b/electron/src/renderer/src/i18n/locales/zh-CN.json index 7be516cac..fc9dfaa65 100644 --- a/electron/src/renderer/src/i18n/locales/zh-CN.json +++ b/electron/src/renderer/src/i18n/locales/zh-CN.json @@ -835,6 +835,7 @@ "qc_track_changed": "配音在质量检查期间发生了变化。请重新检查当前音轨。", "qc_identity_missing": "无法唯一识别各个片段。请使用唯一的片段 ID 重新生成此音轨。", "source_changed": "生成过程中字幕已更改。请重新生成以使用当前字幕。", + "transcription_source_changed": "转录期间字幕发生了变化。你的编辑已保留。", "paste_translation_btn": "粘贴译文", "paste_translation_title": "粘贴译文", "paste_translation_desc": "粘贴你在别处完成的译文(ChatGPT、DeepL 或人工译者)。它会映射到已有的片段上——时间轴和原始转写保持不变。", diff --git a/electron/src/renderer/src/i18n/locales/zh-TW.json b/electron/src/renderer/src/i18n/locales/zh-TW.json index 7e861319e..d151387a1 100644 --- a/electron/src/renderer/src/i18n/locales/zh-TW.json +++ b/electron/src/renderer/src/i18n/locales/zh-TW.json @@ -814,6 +814,7 @@ "qc_track_changed": "配音在品質檢查期間發生了變更。請重新檢查目前的音軌。", "qc_identity_missing": "無法唯一識別各個片段。請使用唯一的片段 ID 重新產生此音軌。", "source_changed": "生成過程中字幕已變更。請重新生成以使用目前的字幕。", + "transcription_source_changed": "轉錄期間字幕發生了變更。你的編輯已保留。", "paste_translation_btn": "貼上譯文", "paste_translation_title": "貼上譯文", "paste_translation_desc": "貼上你在別處完成的譯文(ChatGPT、DeepL 或真人譯者)。它會對應到既有的片段上——時間軸與原始逐字稿維持不變。", diff --git a/electron/src/renderer/src/lib/api/failure.test.ts b/electron/src/renderer/src/lib/api/failure.test.ts index c317e3a96..72acdb9f3 100644 --- a/electron/src/renderer/src/lib/api/failure.test.ts +++ b/electron/src/renderer/src/lib/api/failure.test.ts @@ -23,3 +23,12 @@ it('localizes a source edit that interrupts dub publication', () => { expect(failure.reason).toBe('Localized subtitle change'); expect(translate).toHaveBeenCalledWith('dub.source_changed'); }); + +it('localizes an edit preserved when transcription finishes', () => { + const translate = vi.spyOn(i18next, 't').mockReturnValue('Localized preserved edit'); + const failure = publicFailureFromEvent({ + type: 'error', error_code: 'dub_transcription_source_changed', error: 'Server fallback', + }, 'Task failed'); + expect(failure.reason).toBe('Localized preserved edit'); + expect(translate).toHaveBeenCalledWith('dub.transcription_source_changed'); +}); diff --git a/electron/src/renderer/src/lib/api/failure.ts b/electron/src/renderer/src/lib/api/failure.ts index 081c688ad..fb983ad7d 100644 --- a/electron/src/renderer/src/lib/api/failure.ts +++ b/electron/src/renderer/src/lib/api/failure.ts @@ -21,13 +21,15 @@ export function publicFailureFromEvent( reason: (event.error_code === 'dub_source_changed' ? i18next.t('dub.source_changed') - : event.error_code === 'dub_segment_identity_conflict' - ? i18next.t('dub.qc_identity_missing') - : event.error_code === 'dub_speech_missing' - ? i18next.t('dubIntegrity.missingSpeech') - : event.error_code === 'dub_timing_overflow' - ? i18next.t('dubIntegrity.timingOverflow') - : undefined) || + : event.error_code === 'dub_transcription_source_changed' + ? i18next.t('dub.transcription_source_changed') + : event.error_code === 'dub_segment_identity_conflict' + ? i18next.t('dub.qc_identity_missing') + : event.error_code === 'dub_speech_missing' + ? i18next.t('dubIntegrity.missingSpeech') + : event.error_code === 'dub_timing_overflow' + ? i18next.t('dubIntegrity.timingOverflow') + : undefined) || localized || text(event.reason) || text(event.detail) || diff --git a/electron/src/shared/i18n/locales/ar.json b/electron/src/shared/i18n/locales/ar.json index b18596d32..9a6f1da4f 100644 --- a/electron/src/shared/i18n/locales/ar.json +++ b/electron/src/shared/i18n/locales/ar.json @@ -1285,6 +1285,7 @@ "qc_track_changed": "تغيرت الدبلجة أثناء فحص الجودة. أعد الفحص على المسار الحالي.", "qc_identity_missing": "هويات المقاطع غير واضحة. أعد إنشاء هذا المسار بمعرّفات فريدة للمقاطع.", "source_changed": "تغيّرت الترجمة أثناء التوليد. أعد التوليد لاستخدام الترجمة الحالية.", + "transcription_source_changed": "تغيّرت الترجمة النصية أثناء التفريغ. تم الاحتفاظ بتعديلاتك.", "paste_translation_btn": "لصق ترجمة", "paste_translation_title": "لصق ترجمة", "paste_translation_desc": "الصق ترجمة أعددتها في مكان آخر (ChatGPT أو DeepL أو مترجم بشري). ستُطابَق مع المقاطع الموجودة لديك — تبقى التوقيتات والنص الأصلي دون تغيير.", diff --git a/electron/src/shared/i18n/locales/de.json b/electron/src/shared/i18n/locales/de.json index 549ae0369..3e3c6dceb 100644 --- a/electron/src/shared/i18n/locales/de.json +++ b/electron/src/shared/i18n/locales/de.json @@ -1283,6 +1283,7 @@ "qc_track_changed": "Die Synchronisation wurde während der Qualitätsprüfung geändert. Prüfe die aktuelle Spur erneut.", "qc_identity_missing": "Die Segmentzuordnung ist uneindeutig. Erzeuge diese Spur mit eindeutigen Segment-IDs neu.", "source_changed": "Die Untertitel wurden während der Generierung geändert. Generiere erneut, um die aktuellen Untertitel zu verwenden.", + "transcription_source_changed": "Die Untertitel wurden während der Transkription geändert. Deine Änderungen wurden beibehalten.", "paste_translation_btn": "Übersetzung einfügen", "paste_translation_title": "Übersetzung einfügen", "paste_translation_desc": "Füge eine anderswo erstellte Übersetzung ein (ChatGPT, DeepL, ein menschlicher Übersetzer). Sie wird den vorhandenen Segmenten zugeordnet — Timings und Originaltranskript bleiben unverändert.", diff --git a/electron/src/shared/i18n/locales/en.json b/electron/src/shared/i18n/locales/en.json index 846e91812..9b4a75716 100644 --- a/electron/src/shared/i18n/locales/en.json +++ b/electron/src/shared/i18n/locales/en.json @@ -1508,6 +1508,7 @@ "qc_track_changed": "The dub changed during the quality check. Run the check again on the current track.", "qc_identity_missing": "Segment identities are ambiguous. Regenerate this track with unique segment IDs.", "source_changed": "Subtitles changed during generation. Generate again to use the current subtitles.", + "transcription_source_changed": "Subtitles changed during transcription. Your edits were kept.", "export_btn": "Export…", "prep_stop": "Stop", "install_progress": "Installing {{engine}}…", diff --git a/electron/src/shared/i18n/locales/es.json b/electron/src/shared/i18n/locales/es.json index a43577b83..e315ff09f 100644 --- a/electron/src/shared/i18n/locales/es.json +++ b/electron/src/shared/i18n/locales/es.json @@ -1283,6 +1283,7 @@ "qc_track_changed": "El doblaje cambió durante la comprobación de calidad. Repite la comprobación en la pista actual.", "qc_identity_missing": "La identidad de los segmentos es ambigua. Regenera esta pista con identificadores de segmento únicos.", "source_changed": "Los subtítulos cambiaron durante la generación. Genera de nuevo para usar los subtítulos actuales.", + "transcription_source_changed": "Los subtítulos cambiaron durante la transcripción. Se conservaron tus cambios.", "paste_translation_btn": "Pegar traducción", "paste_translation_title": "Pegar una traducción", "paste_translation_desc": "Pega una traducción hecha en otro sitio (ChatGPT, DeepL, un traductor humano). Se asigna a los segmentos que ya tienes: los tiempos y la transcripción original no cambian.", diff --git a/electron/src/shared/i18n/locales/fr.json b/electron/src/shared/i18n/locales/fr.json index 8030060de..e842e6adf 100644 --- a/electron/src/shared/i18n/locales/fr.json +++ b/electron/src/shared/i18n/locales/fr.json @@ -1283,6 +1283,7 @@ "qc_track_changed": "Le doublage a changé pendant le contrôle qualité. Relancez le contrôle sur la piste actuelle.", "qc_identity_missing": "Les segments ne sont pas identifiés de façon univoque. Régénérez cette piste avec des identifiants de segment uniques.", "source_changed": "Les sous-titres ont changé pendant la génération. Relancez la génération pour utiliser les sous-titres actuels.", + "transcription_source_changed": "Les sous-titres ont changé pendant la transcription. Vos modifications ont été conservées.", "paste_translation_btn": "Coller une traduction", "paste_translation_title": "Coller une traduction", "paste_translation_desc": "Collez une traduction réalisée ailleurs (ChatGPT, DeepL, un traducteur humain). Elle est appliquée aux segments existants — les timings et la transcription d'origine restent intacts.", diff --git a/electron/src/shared/i18n/locales/hi.json b/electron/src/shared/i18n/locales/hi.json index a2f433c9d..7aded4573 100644 --- a/electron/src/shared/i18n/locales/hi.json +++ b/electron/src/shared/i18n/locales/hi.json @@ -1283,6 +1283,7 @@ "qc_track_changed": "गुणवत्ता जाँच के दौरान डबिंग बदल गई। मौजूदा ट्रैक पर जाँच दोबारा चलाएँ।", "qc_identity_missing": "सेगमेंट की पहचान स्पष्ट नहीं है। हर सेगमेंट को अलग आईडी देकर इस ट्रैक को फिर से जनरेट करें।", "source_changed": "जनरेशन के दौरान सबटाइटल बदल गए। मौजूदा सबटाइटल इस्तेमाल करने के लिए फिर से जनरेट करें।", + "transcription_source_changed": "ट्रांसक्रिप्शन के दौरान उपशीर्षक बदल गए। आपके बदलाव सुरक्षित रखे गए हैं।", "paste_translation_btn": "अनुवाद चिपकाएँ", "paste_translation_title": "अनुवाद चिपकाएँ", "paste_translation_desc": "कहीं और तैयार किया गया अनुवाद चिपकाएँ (ChatGPT, DeepL, कोई मानव अनुवादक)। यह आपके मौजूदा सेगमेंट पर लागू होगा — टाइमिंग और मूल ट्रांसक्रिप्ट अछूते रहते हैं।", diff --git a/electron/src/shared/i18n/locales/id.json b/electron/src/shared/i18n/locales/id.json index 68a35f9aa..e33707d62 100644 --- a/electron/src/shared/i18n/locales/id.json +++ b/electron/src/shared/i18n/locales/id.json @@ -1285,6 +1285,7 @@ "qc_track_changed": "Sulih suara berubah selama pemeriksaan kualitas. Jalankan kembali pemeriksaan pada trek saat ini.", "qc_identity_missing": "Identitas segmen tidak jelas. Buat ulang trek ini dengan ID segmen yang unik.", "source_changed": "Subtitel berubah selama pembuatan. Buat ulang untuk menggunakan subtitel saat ini.", + "transcription_source_changed": "Subtitel berubah selama transkripsi. Perubahan Anda dipertahankan.", "paste_translation_btn": "Tempel terjemahan", "paste_translation_title": "Tempel terjemahan", "paste_translation_desc": "Tempel terjemahan yang kamu buat di tempat lain (ChatGPT, DeepL, penerjemah manusia). Terjemahan itu dipetakan ke segmen yang sudah ada — pewaktuan dan transkrip asli tidak diubah.", diff --git a/electron/src/shared/i18n/locales/it.json b/electron/src/shared/i18n/locales/it.json index 0ad7809e2..3dfec2389 100644 --- a/electron/src/shared/i18n/locales/it.json +++ b/electron/src/shared/i18n/locales/it.json @@ -1283,6 +1283,7 @@ "qc_track_changed": "Il doppiaggio è cambiato durante il controllo qualità. Ripeti il controllo sulla traccia attuale.", "qc_identity_missing": "Le identità dei segmenti sono ambigue. Rigenera questa traccia con identificativi di segmento univoci.", "source_changed": "I sottotitoli sono cambiati durante la generazione. Genera di nuovo per usare i sottotitoli attuali.", + "transcription_source_changed": "I sottotitoli sono cambiati durante la trascrizione. Le tue modifiche sono state mantenute.", "paste_translation_btn": "Incolla traduzione", "paste_translation_title": "Incolla una traduzione", "paste_translation_desc": "Incolla una traduzione prodotta altrove (ChatGPT, DeepL, un traduttore umano). Viene mappata sui segmenti che hai già: tempi e trascrizione originale restano invariati.", diff --git a/electron/src/shared/i18n/locales/ja.json b/electron/src/shared/i18n/locales/ja.json index 8f74110b7..ca52713fd 100644 --- a/electron/src/shared/i18n/locales/ja.json +++ b/electron/src/shared/i18n/locales/ja.json @@ -1285,6 +1285,7 @@ "qc_track_changed": "品質チェック中に吹き替えが変更されました。現在のトラックで再度チェックしてください。", "qc_identity_missing": "セグメントを一意に識別できません。各セグメントに固有のIDを付けて、このトラックを再生成してください。", "source_changed": "生成中に字幕が変更されました。現在の字幕を使用するには、もう一度生成してください。", + "transcription_source_changed": "文字起こし中に字幕が変更されました。編集内容は保持されています。", "paste_translation_btn": "翻訳を貼り付け", "paste_translation_title": "翻訳を貼り付け", "paste_translation_desc": "他所で用意した翻訳(ChatGPT、DeepL、人間の翻訳者)を貼り付けます。既存のセグメントに割り当てられ、タイミングと元の文字起こしはそのまま残ります。", diff --git a/electron/src/shared/i18n/locales/ko.json b/electron/src/shared/i18n/locales/ko.json index 6ab8e8f0a..2800c46b2 100644 --- a/electron/src/shared/i18n/locales/ko.json +++ b/electron/src/shared/i18n/locales/ko.json @@ -1572,6 +1572,7 @@ "qc_track_changed": "품질 검사 중 더빙이 변경되었습니다. 현재 트랙에서 검사를 다시 실행하세요.", "qc_identity_missing": "세그먼트를 명확하게 구분할 수 없습니다. 각 세그먼트에 고유한 ID를 지정하여 이 트랙을 다시 생성하세요.", "source_changed": "생성 중에 자막이 변경되었습니다. 현재 자막을 사용하려면 다시 생성하세요.", + "transcription_source_changed": "음성을 텍스트로 변환하는 동안 자막이 변경되었습니다. 수정 내용은 유지되었습니다.", "paste_translation_btn": "번역 붙여넣기", "paste_translation_title": "번역 붙여넣기", "paste_translation_desc": "다른 곳에서 만든 번역(ChatGPT, DeepL, 사람 번역가)을 붙여넣으세요. 이미 있는 세그먼트에 매핑되며 타이밍과 원본 전사는 그대로 유지됩니다.", diff --git a/electron/src/shared/i18n/locales/nl.json b/electron/src/shared/i18n/locales/nl.json index 4f9928f06..5fd8f46fe 100644 --- a/electron/src/shared/i18n/locales/nl.json +++ b/electron/src/shared/i18n/locales/nl.json @@ -1283,6 +1283,7 @@ "qc_track_changed": "De nasynchronisatie is tijdens de kwaliteitscontrole gewijzigd. Voer de controle opnieuw uit op het huidige spoor.", "qc_identity_missing": "De segmenten zijn niet eenduidig te identificeren. Genereer dit spoor opnieuw met unieke segment-ID’s.", "source_changed": "De ondertitels zijn gewijzigd tijdens het genereren. Genereer opnieuw om de huidige ondertitels te gebruiken.", + "transcription_source_changed": "De ondertitels zijn tijdens de transcriptie gewijzigd. Je wijzigingen zijn behouden.", "paste_translation_btn": "Vertaling plakken", "paste_translation_title": "Een vertaling plakken", "paste_translation_desc": "Plak een elders gemaakte vertaling (ChatGPT, DeepL, een menselijke vertaler). Die wordt op je bestaande segmenten toegepast — timings en het originele transcript blijven ongewijzigd.", diff --git a/electron/src/shared/i18n/locales/pl.json b/electron/src/shared/i18n/locales/pl.json index d42f07c7c..6e113afbe 100644 --- a/electron/src/shared/i18n/locales/pl.json +++ b/electron/src/shared/i18n/locales/pl.json @@ -1283,6 +1283,7 @@ "qc_track_changed": "Dubbing zmienił się podczas kontroli jakości. Uruchom kontrolę ponownie dla bieżącej ścieżki.", "qc_identity_missing": "Tożsamość segmentów jest niejednoznaczna. Wygeneruj tę ścieżkę ponownie z unikatowymi identyfikatorami segmentów.", "source_changed": "Napisy zmieniły się podczas generowania. Wygeneruj ponownie, aby użyć aktualnych napisów.", + "transcription_source_changed": "Napisy zmieniły się podczas transkrypcji. Twoje zmiany zostały zachowane.", "paste_translation_btn": "Wklej tłumaczenie", "paste_translation_title": "Wklej tłumaczenie", "paste_translation_desc": "Wklej tłumaczenie przygotowane gdzie indziej (ChatGPT, DeepL, tłumacz). Zostanie dopasowane do istniejących segmentów — czasy i oryginalna transkrypcja pozostają bez zmian.", diff --git a/electron/src/shared/i18n/locales/pt.json b/electron/src/shared/i18n/locales/pt.json index c50563a51..9d3e2c7e7 100644 --- a/electron/src/shared/i18n/locales/pt.json +++ b/electron/src/shared/i18n/locales/pt.json @@ -1283,6 +1283,7 @@ "qc_track_changed": "A dublagem mudou durante a verificação de qualidade. Execute a verificação novamente na faixa atual.", "qc_identity_missing": "A identificação dos segmentos é ambígua. Gere esta faixa novamente com identificadores de segmento exclusivos.", "source_changed": "As legendas mudaram durante a geração. Gere novamente para usar as legendas atuais.", + "transcription_source_changed": "As legendas mudaram durante a transcrição. Suas alterações foram mantidas.", "paste_translation_btn": "Colar tradução", "paste_translation_title": "Colar uma tradução", "paste_translation_desc": "Cole uma tradução feita noutro lugar (ChatGPT, DeepL, um tradutor humano). Ela é mapeada nos segmentos que já existem — os tempos e a transcrição original ficam intactos.", diff --git a/electron/src/shared/i18n/locales/ru.json b/electron/src/shared/i18n/locales/ru.json index 8d69f1ecc..901f0487e 100644 --- a/electron/src/shared/i18n/locales/ru.json +++ b/electron/src/shared/i18n/locales/ru.json @@ -1283,6 +1283,7 @@ "qc_track_changed": "Дубляж изменился во время проверки качества. Повторите проверку текущей дорожки.", "qc_identity_missing": "Сегменты невозможно однозначно определить. Создайте эту дорожку заново с уникальными идентификаторами сегментов.", "source_changed": "Субтитры изменились во время генерации. Запустите генерацию заново, чтобы использовать текущие субтитры.", + "transcription_source_changed": "Субтитры изменились во время расшифровки. Ваши изменения сохранены.", "paste_translation_btn": "Вставить перевод", "paste_translation_title": "Вставить перевод", "paste_translation_desc": "Вставьте перевод, сделанный в другом месте (ChatGPT, DeepL, живой переводчик). Он ляжет на уже имеющиеся сегменты — тайминги и исходная расшифровка не изменятся.", diff --git a/electron/src/shared/i18n/locales/sv.json b/electron/src/shared/i18n/locales/sv.json index 95a3391b4..6f159864b 100644 --- a/electron/src/shared/i18n/locales/sv.json +++ b/electron/src/shared/i18n/locales/sv.json @@ -1285,6 +1285,7 @@ "qc_track_changed": "Dubbningen ändrades under kvalitetskontrollen. Kör kontrollen igen på det aktuella spåret.", "qc_identity_missing": "Segmenten kan inte identifieras entydigt. Generera om spåret med unika segment-ID:n.", "source_changed": "Undertexterna ändrades under genereringen. Generera igen för att använda de aktuella undertexterna.", + "transcription_source_changed": "Undertexterna ändrades under transkriberingen. Dina ändringar har sparats.", "paste_translation_btn": "Klistra in översättning", "paste_translation_title": "Klistra in en översättning", "paste_translation_desc": "Klistra in en översättning som gjorts någon annanstans (ChatGPT, DeepL, en mänsklig översättare). Den mappas mot segmenten du redan har — tidkoder och originaltranskriptet rörs inte.", diff --git a/electron/src/shared/i18n/locales/th.json b/electron/src/shared/i18n/locales/th.json index 693f7d0ff..aa42e5c4c 100644 --- a/electron/src/shared/i18n/locales/th.json +++ b/electron/src/shared/i18n/locales/th.json @@ -1285,6 +1285,7 @@ "qc_track_changed": "เสียงพากย์เปลี่ยนแปลงระหว่างการตรวจสอบคุณภาพ โปรดตรวจสอบแทร็กปัจจุบันอีกครั้ง", "qc_identity_missing": "ไม่สามารถระบุแต่ละช่วงได้อย่างชัดเจน โปรดสร้างแทร็กนี้ใหม่โดยใช้รหัสที่ไม่ซ้ำกันสำหรับแต่ละช่วง", "source_changed": "คำบรรยายเปลี่ยนแปลงระหว่างการสร้าง โปรดสร้างอีกครั้งเพื่อใช้คำบรรยายปัจจุบัน", + "transcription_source_changed": "คำบรรยายเปลี่ยนแปลงระหว่างการถอดเสียง การแก้ไขของคุณยังคงอยู่", "paste_translation_btn": "วางคำแปล", "paste_translation_title": "วางคำแปล", "paste_translation_desc": "วางคำแปลที่คุณทำไว้จากที่อื่น (ChatGPT, DeepL หรือผู้แปลที่เป็นคน) ระบบจะจับคู่กับเซกเมนต์ที่มีอยู่แล้ว โดยเวลาและถอดความต้นฉบับยังคงเดิม", diff --git a/electron/src/shared/i18n/locales/tr.json b/electron/src/shared/i18n/locales/tr.json index aa3bd1331..5c858caf7 100644 --- a/electron/src/shared/i18n/locales/tr.json +++ b/electron/src/shared/i18n/locales/tr.json @@ -1285,6 +1285,7 @@ "qc_track_changed": "Kalite kontrolü sırasında dublaj değişti. Geçerli parçada kontrolü yeniden çalıştırın.", "qc_identity_missing": "Bölümler kesin olarak tanımlanamıyor. Bu parçayı benzersiz bölüm kimlikleriyle yeniden oluşturun.", "source_changed": "Oluşturma sırasında altyazılar değişti. Güncel altyazıları kullanmak için yeniden oluşturun.", + "transcription_source_changed": "Transkripsiyon sırasında altyazılar değişti. Düzenlemeleriniz korundu.", "paste_translation_btn": "Çeviriyi yapıştır", "paste_translation_title": "Bir çeviri yapıştır", "paste_translation_desc": "Başka bir yerde hazırladığın çeviriyi yapıştır (ChatGPT, DeepL, bir insan çevirmen). Mevcut segmentlere eşlenir — zamanlamalar ve özgün döküm olduğu gibi kalır.", diff --git a/electron/src/shared/i18n/locales/uk.json b/electron/src/shared/i18n/locales/uk.json index 1afbf6688..8243cf744 100644 --- a/electron/src/shared/i18n/locales/uk.json +++ b/electron/src/shared/i18n/locales/uk.json @@ -1285,6 +1285,7 @@ "qc_track_changed": "Дубляж змінився під час перевірки якості. Повторіть перевірку поточної доріжки.", "qc_identity_missing": "Сегменти неможливо однозначно визначити. Створіть цю доріжку заново з унікальними ідентифікаторами сегментів.", "source_changed": "Субтитри змінилися під час генерації. Запустіть генерацію знову, щоб використати поточні субтитри.", + "transcription_source_changed": "Субтитри змінилися під час розшифрування. Ваші зміни збережено.", "paste_translation_btn": "Вставити переклад", "paste_translation_title": "Вставити переклад", "paste_translation_desc": "Вставте переклад, зроблений деінде (ChatGPT, DeepL, живий перекладач). Він накладеться на наявні сегменти — таймінги й початкова транскрипція лишаться незмінними.", diff --git a/electron/src/shared/i18n/locales/vi.json b/electron/src/shared/i18n/locales/vi.json index 42290204d..8e1b9c521 100644 --- a/electron/src/shared/i18n/locales/vi.json +++ b/electron/src/shared/i18n/locales/vi.json @@ -1285,6 +1285,7 @@ "qc_track_changed": "Bản lồng tiếng đã thay đổi trong khi kiểm tra chất lượng. Hãy kiểm tra lại bản âm thanh hiện tại.", "qc_identity_missing": "Không thể xác định rõ từng đoạn. Hãy tạo lại bản âm thanh này với mã định danh riêng cho mỗi đoạn.", "source_changed": "Phụ đề đã thay đổi trong quá trình tạo. Hãy tạo lại để sử dụng phụ đề hiện tại.", + "transcription_source_changed": "Phụ đề đã thay đổi trong khi chép lời. Các chỉnh sửa của bạn được giữ lại.", "paste_translation_btn": "Dán bản dịch", "paste_translation_title": "Dán một bản dịch", "paste_translation_desc": "Dán bản dịch bạn đã làm ở nơi khác (ChatGPT, DeepL, người dịch). Nó sẽ được ánh xạ vào các phân đoạn sẵn có — thời điểm và bản ghi gốc giữ nguyên.", diff --git a/electron/src/shared/i18n/locales/zh-CN.json b/electron/src/shared/i18n/locales/zh-CN.json index edce27765..896d84b7e 100644 --- a/electron/src/shared/i18n/locales/zh-CN.json +++ b/electron/src/shared/i18n/locales/zh-CN.json @@ -1533,6 +1533,7 @@ "qc_track_changed": "配音在质量检查期间发生了变化。请重新检查当前音轨。", "qc_identity_missing": "无法唯一识别各个片段。请使用唯一的片段 ID 重新生成此音轨。", "source_changed": "生成过程中字幕已更改。请重新生成以使用当前字幕。", + "transcription_source_changed": "转录期间字幕发生了变化。你的编辑已保留。", "paste_translation_btn": "粘贴译文", "paste_translation_title": "粘贴译文", "paste_translation_desc": "粘贴你在别处完成的译文(ChatGPT、DeepL 或人工译者)。它会映射到已有的片段上——时间轴和原始转写保持不变。", diff --git a/electron/src/shared/i18n/locales/zh-TW.json b/electron/src/shared/i18n/locales/zh-TW.json index 97e514db3..4158c0b3f 100644 --- a/electron/src/shared/i18n/locales/zh-TW.json +++ b/electron/src/shared/i18n/locales/zh-TW.json @@ -1285,6 +1285,7 @@ "qc_track_changed": "配音在品質檢查期間發生了變更。請重新檢查目前的音軌。", "qc_identity_missing": "無法唯一識別各個片段。請使用唯一的片段 ID 重新產生此音軌。", "source_changed": "生成過程中字幕已變更。請重新生成以使用目前的字幕。", + "transcription_source_changed": "轉錄期間字幕發生了變更。你的編輯已保留。", "paste_translation_btn": "貼上譯文", "paste_translation_title": "貼上譯文", "paste_translation_desc": "貼上你在別處完成的譯文(ChatGPT、DeepL 或真人譯者)。它會對應到既有的片段上——時間軸與原始逐字稿維持不變。", diff --git a/tests/test_dub_complete_audio.py b/tests/test_dub_complete_audio.py index ca984ad59..13ea2f1e4 100644 --- a/tests/test_dub_complete_audio.py +++ b/tests/test_dub_complete_audio.py @@ -11,6 +11,21 @@ from schemas.requests import DubRequest +class ContendedLock: + """Notify only after a worker actually finds the real lock held elsewhere.""" + def __init__(self, lock, loop, contended): + self.lock, self.loop, self.contended = lock, loop, contended + + def __enter__(self): + if not self.lock.acquire(blocking=False): + self.loop.call_soon_threadsafe(self.contended.set) + self.lock.acquire() + return self + + def __exit__(self, *_): + self.lock.release() + + @pytest.fixture def render_dub(monkeypatch, tmp_path): import api.routers.dub_generate as dg @@ -469,7 +484,9 @@ def slow_save(*_): assert not render.done() for _ in range(2): render.cancel() - await asyncio.sleep(0) + cancellation_delivered = asyncio.Event() + loop.call_soon(cancellation_delivered.set) + await cancellation_delivered.wait() assert not render.done(), 'cancellation detached the in-flight publisher' assert list(render_dub.path.glob('.render-*')), 'staging deleted under the publisher' release.set() @@ -541,18 +558,23 @@ def test_delete_serializes_with_publication_without_resurrecting_job( dp.save_job('job', render_dub.job) monkeypatch.setattr(dg, '_get_job', dp.get_job) release = threading.Event() + order = [] def delete(): def delete_rows(): with db.db_conn() as conn: conn.execute('DELETE FROM dub_history WHERE id=?', ('job',)) + order.append('delete') dp.purge_jobs(['job'], delete_rows=delete_rows) async def exercise(): loop = asyncio.get_running_loop() entered = asyncio.Event() + contended = asyncio.Event() + monkeypatch.setattr(dp, '_dub_jobs_lock', ContendedLock(dp._dub_jobs_lock, loop, contended)) def slow_save(*args): loop.call_soon_threadsafe(entered.set) assert release.wait(2) render_dub.real_save(*args) + order.append('save') monkeypatch.setattr(dg, '_save_job', slow_save) if delete_before_publication: finish = dg._finish_publication @@ -569,7 +591,7 @@ async def delete_first(publish): try: await asyncio.wait_for(entered.wait(), 10) deletion = loop.run_in_executor(None, delete) - await asyncio.sleep(0) + await asyncio.wait_for(contended.wait(), 10) assert not deletion.done(), 'delete crossed the publication lock' release.set() await asyncio.gather(render, deletion) @@ -581,6 +603,7 @@ async def delete_first(publish): with db.db_conn() as conn: assert conn.execute('SELECT id FROM dub_history WHERE id=?', ('job',)).fetchone() is None assert not list(render_dub.path.glob('.render-*')) + assert order == (['delete'] if delete_before_publication else ['save', 'delete']) def test_cancel_before_publication_starts_discards_staging(render_dub, monkeypatch): @@ -605,11 +628,15 @@ async def cancel_first(publish): def test_import_finishing_during_publication_preserves_new_subtitles(render_dub, monkeypatch): import threading from api.routers import dub_core, dub_generate as dg + from services import dub_pipeline as dp release = threading.Event() + order = [] async def exercise(): loop = asyncio.get_running_loop() publishing = asyncio.Event() + contended = asyncio.Event() + monkeypatch.setattr(dp, '_dub_jobs_lock', ContendedLock(dp._dub_jobs_lock, loop, contended)) reading = asyncio.Event() read_finished = asyncio.Event() class Upload: @@ -621,15 +648,16 @@ async def read(self): def slow_save(*_): loop.call_soon_threadsafe(publishing.set) assert release.wait(2) + order.append('save') monkeypatch.setattr(dg, '_save_job', slow_save) monkeypatch.setattr(dub_core, '_get_job', lambda _: render_dub.job) - monkeypatch.setattr(dub_core, '_save_job', lambda *_: None) + monkeypatch.setattr(dub_core, '_save_job', lambda *_: order.append('import')) imported = asyncio.create_task(dub_core.dub_import_srt('job', Upload())) await reading.wait() # endpoint has started reading before publication render = asyncio.create_task(render_dub.arun()) try: await asyncio.wait_for(read_finished.wait(), 10) - await asyncio.sleep(0) + await asyncio.wait_for(contended.wait(), 10) assert not imported.done(), 'import mutated the job during publication' release.set() events, result = await asyncio.gather(render, imported) @@ -637,7 +665,75 @@ def slow_save(*_): assert result['segments'][0]['text'] == 'replacement subtitle' assert render_dub.job['segments'][0]['text'] == 'replacement subtitle' assert 'en' in render_dub.job['dubbed_tracks'] + assert order == ['save', 'import'] finally: release.set() await asyncio.gather(render, imported, return_exceptions=True) asyncio.run(exercise()) + + +def test_committed_publication_reports_done_after_late_cancel(render_dub, monkeypatch): + from api.routers import dub_generate as dg + from core.tasks import TaskManager + + store = TaskManager.worker.__globals__['job_store'] + states = [] + for name in ['create', 'append_event', 'mark_running']: + monkeypatch.setattr(store, name, lambda *a, **kw: None) + monkeypatch.setattr(store, 'mark_done', lambda *_: states.append('done')) + monkeypatch.setattr(store, 'mark_cancelled', lambda *_: states.append('cancelled')) + monkeypatch.setattr(TaskManager.worker.__globals__['run_sentinel'], 'touch_activity', lambda *_: None) + async def exercise(): + manager = TaskManager() + monkeypatch.setattr(dg, 'task_manager', manager) + def late_cancel(*_): + manager.cancel_task(next(iter(manager.active_tasks))) + monkeypatch.setattr(dg, '_save_job', late_cancel) + await render_dub.arun() + worker = asyncio.create_task(manager.worker()) + try: + await asyncio.wait_for(manager.queue.join(), 10) + task = next(iter(manager.active_tasks.values())) + assert render_dub.job['dubbed_tracks']['en']['path'] + assert task['status'] == 'done' + events = [json.loads(event[6:]) for event in task['history'] if event.startswith('data: ')] + assert events[-1]['type'] == 'done' + assert not any(event['type'] == 'cancelled' for event in events) + finally: + worker.cancel() + await asyncio.gather(worker, return_exceptions=True) + asyncio.run(exercise()) + assert states == ['done'] + + +def test_track_read_does_not_block_event_loop_behind_publication(render_dub, monkeypatch): + import threading + from api.routers import dub_export + from services import dub_pipeline as dp + + monkeypatch.setattr(dp, '_dub_jobs', {'job': render_dub.job}) + release = threading.Event() + async def exercise(): + loop = asyncio.get_running_loop() + held, reading = asyncio.Event(), asyncio.Event() + def hold_publication_lock(): + with dp._dub_jobs_lock: + loop.call_soon_threadsafe(held.set) + assert release.wait(2), 'a blocked job lookup prevented the loop from releasing publication' + def read_job(job_id): + loop.call_soon_threadsafe(reading.set) + return dp.get_job(job_id) + monkeypatch.setattr(dub_export, '_get_job', read_job) + holding = loop.run_in_executor(None, hold_publication_lock) + await held.wait() + lookup = asyncio.create_task(dub_export.dub_list_tracks('job')) + try: + await asyncio.wait_for(reading.wait(), 10) + assert not lookup.done(), 'the event loop resumed only after the lock wait finished' + release.set() + assert await lookup == {'tracks': {}} + await holding + finally: + release.set() + await asyncio.gather(holding, lookup, return_exceptions=True) + asyncio.run(exercise()) diff --git a/tests/test_dub_job_deleted_mid_ingest.py b/tests/test_dub_job_deleted_mid_ingest.py index 0506718af..2e4201ed5 100644 --- a/tests/test_dub_job_deleted_mid_ingest.py +++ b/tests/test_dub_job_deleted_mid_ingest.py @@ -69,7 +69,7 @@ def test_the_ingest_pipeline_no_longer_blind_subscripts_the_job(): src = inspect.getsource(dub_pipeline.ingest_pipeline) assert "_dub_jobs[job_id].update(" not in src - assert "merge_and_save_job(" in src + assert "run_job_operation(merge_and_save_job," in src # ── the message ────────────────────────────────────────────────────────── @@ -216,7 +216,7 @@ def test_the_create_checkpoints_are_atomic_too(monkeypatch): import inspect src = inspect.getsource(dub_pipeline.ingest_pipeline) - assert "put_and_save_job(" in src + assert "run_job_operation(put_and_save_job," in src assert "put_job(job_id," not in src, "the unlocked pair is the race" saved = [] @@ -244,7 +244,7 @@ def test_the_ingest_pipeline_persists_atomically(): import inspect src = inspect.getsource(dub_pipeline.ingest_pipeline) - assert "merge_and_save_job(" in src + assert "run_job_operation(merge_and_save_job," in src # The two-step form is what the race lived in. assert "save_job(job_id, get_job(" not in src @@ -435,11 +435,11 @@ def test_the_pipeline_registers_and_releases_its_run(): import inspect src = inspect.getsource(dub_pipeline.ingest_pipeline) - assert "begin_ingest(job_id)" in src - assert "end_ingest(job_id)" in src, "a leaked tombstone would block a later run" + assert "run_job_operation(begin_ingest, job_id)" in src + assert "run_job_operation(end_ingest, job_id)" in src, "a leaked tombstone would block a later run" # Released in `finally`, so a crash or cancel can't leak it. finally_block = src[src.rindex("finally:"):] - assert "end_ingest(job_id)" in finally_block + assert "run_job_operation(end_ingest, job_id)" in finally_block def test_clear_history_sweeps_inflight_jobs(): diff --git a/tests/test_dub_no_tts_load_for_asr.py b/tests/test_dub_no_tts_load_for_asr.py index 28c90e912..c92c84d94 100644 --- a/tests/test_dub_no_tts_load_for_asr.py +++ b/tests/test_dub_no_tts_load_for_asr.py @@ -145,7 +145,7 @@ def test_preflight_error_does_not_leave_asr_on_vocals_unbound(dub, monkeypatch): assert "No audio available" in body -@pytest.mark.parametrize('outcome', ['complete', 'deleted', 'replaced', 'failed', 'cancelled']) +@pytest.mark.parametrize('outcome', ['complete', 'deleted', 'replaced', 'failed', 'cancelled', 'edited']) def test_transcription_publishes_private_source_without_replacing_tracks(dub, monkeypatch, outcome): from fastapi import HTTPException @@ -167,6 +167,8 @@ async def guarded(_pool, transcribe, **_kwargs): monkeypatch.setattr(dc, '_get_job', lambda _: None) elif outcome == 'replaced': dc._dub_jobs[job_id] = {'segments': [], 'replacement': True} + elif outcome == 'edited': + job['segments'] = [dict(original_segments[0], text='imported correction')] elif outcome == 'failed': raise RuntimeError('injected ASR completion failure') elif outcome == 'cancelled': @@ -185,7 +187,63 @@ async def guarded(_pool, transcribe, **_kwargs): asyncio.run(dc.dub_transcribe(job_id)) if outcome in ('deleted', 'replaced'): assert error.value.status_code == 404 - assert job['segments'] == original_segments + if outcome == 'edited': + assert error.value.status_code == 409 + assert job['segments'][0]['text'] == 'imported correction' + else: + assert job['segments'] == original_segments assert job['source_lang'] == 'old' if outcome == 'replaced': assert dc._dub_jobs[job_id] == {'segments': [], 'replacement': True} + + +def test_cancelled_transcription_waiting_for_publication_keeps_source_private(dub, monkeypatch): + import threading + + dc, job_id, _ = dub + job = dc._dub_jobs[job_id] + job['segments'] = [{'id': 0, 'start': 0, 'end': .5, 'text': 'original'}] + original = dc._transcription_source(job) + release = threading.Event() + async def exercise(): + loop = asyncio.get_running_loop() + held, contended, finished = asyncio.Event(), asyncio.Event(), asyncio.Event() + original_lock = dc.dub_pipeline._dub_jobs_lock + class Lock: + def __enter__(self): + if not original_lock.acquire(blocking=False): + loop.call_soon_threadsafe(contended.set) + original_lock.acquire() + return self + def __exit__(self, *_): + original_lock.release() + monkeypatch.setattr(dc.dub_pipeline, '_dub_jobs_lock', Lock()) + publish = dc._publish_transcription + def observed_publish(*args): + try: + return publish(*args) + finally: + loop.call_soon_threadsafe(finished.set) + monkeypatch.setattr(dc, '_publish_transcription', observed_publish) + def holder(): + with original_lock: + loop.call_soon_threadsafe(held.set) + assert release.wait(2) + holding = loop.run_in_executor(None, holder) + await held.wait() + saving = asyncio.create_task(dc._save_transcription( + job_id, job, original, {'segments': [{'text': 'cancelled ASR'}]}, + )) + try: + await asyncio.wait_for(contended.wait(), 10) + saving.cancel() + with pytest.raises(asyncio.CancelledError): + await saving + release.set() + await asyncio.wait_for(finished.wait(), 10) + await holding + assert dc._transcription_source(job) == original + finally: + release.set() + await asyncio.gather(saving, holding, return_exceptions=True) + asyncio.run(exercise()) diff --git a/tests/test_dub_pipeline_state.py b/tests/test_dub_pipeline_state.py index cfb73cc97..7e3049b6c 100644 --- a/tests/test_dub_pipeline_state.py +++ b/tests/test_dub_pipeline_state.py @@ -158,3 +158,57 @@ def test_save_job_upsert_does_not_clobber_language_with_empty(): row = _lang_row(jid) assert row["language"] == "Bengali" assert row["language_code"] == "bn" + + +def test_cancelled_ingest_waits_for_locked_worker_before_cleanup(monkeypatch, tmp_path): + import asyncio + import threading + + release = threading.Event() + job_dir = tmp_path / 'queued-ingest' + job_dir.mkdir() + (job_dir / 'input.wav').write_bytes(b'pending input') + events = [] + async def exercise(): + loop = asyncio.get_running_loop() + held, contended = asyncio.Event(), asyncio.Event() + original = dp._dub_jobs_lock + class Lock: + def __enter__(self): + if not original.acquire(blocking=False): + loop.call_soon_threadsafe(contended.set) + original.acquire() + return self + def __exit__(self, *_): + original.release() + monkeypatch.setattr(dp, '_dub_jobs_lock', Lock()) + def publisher(): + with original: + loop.call_soon_threadsafe(held.set) + assert release.wait(2), 'ingest lock wait blocked the event loop' + publishing = loop.run_in_executor(None, publisher) + await held.wait() + async def ingest(): + async for event in dp.ingest_pipeline('queued-ingest', str(job_dir), {'kind': 'upload'}): + events.append(event) + task = asyncio.create_task(ingest()) + try: + await asyncio.wait_for(contended.wait(), 10) + task.cancel() + rendezvous = asyncio.Event() + loop.call_soon(rendezvous.set) + await rendezvous.wait() + assert job_dir.exists(), 'cleanup raced the queued ingest worker' + assert not task.done() + release.set() + with pytest.raises(asyncio.CancelledError): + await task + await publishing + finally: + release.set() + await asyncio.gather(task, publishing, return_exceptions=True) + asyncio.run(exercise()) + assert not job_dir.exists() + assert 'queued-ingest' not in dp._inflight_jobs + assert 'queued-ingest' not in dp._dub_jobs + assert any('cancelled' in event for event in events) diff --git a/tests/test_dub_qc_concurrency.py b/tests/test_dub_qc_concurrency.py index e900a9c39..c995742b6 100644 --- a/tests/test_dub_qc_concurrency.py +++ b/tests/test_dub_qc_concurrency.py @@ -5,7 +5,7 @@ import pytest -@pytest.mark.parametrize('change', ['regenerate', 'text', 'timing', 'audio', 'delete', 'replace', None]) +@pytest.mark.parametrize('change', ['regenerate', 'text', 'timing', 'audio', 'delete', 'replace', 'publication_wait', None]) def test_qc_does_not_publish_after_selected_render_changes(monkeypatch, tmp_path, change): from api.routers import dub_export, dub_generate from fastapi import HTTPException @@ -38,8 +38,37 @@ def transcribe(self, path, **kwargs): return {'segments': [{'start': 0., 'end': 2., 'text': 'hola mundo'}]} monkeypatch.setattr(asr_backend, 'load_active_asr_backend', Backend) + held_worker = [] + release_worker = [] async def guarded(pool, fn, **kwargs): recognition = fn() + if change == 'publication_wait': + import threading + loop = asyncio.get_running_loop() + held, contended = asyncio.Event(), asyncio.Event() + release = threading.Event() + release_worker.append(release) + original = dub_pipeline._dub_jobs_lock + class Lock: + def __enter__(self): + if not original.acquire(blocking=False): + loop.call_soon_threadsafe(contended.set) + original.acquire() + return self + def __exit__(self, *_): + original.release() + monkeypatch.setattr(dub_export, '_dub_jobs_lock', Lock()) + def publisher(): + with original: + loop.call_soon_threadsafe(held.set) + assert release.wait(2), 'QC lock wait blocked the event loop' + held_worker.append(loop.run_in_executor(None, publisher)) + await held.wait() + async def release_from_loop(): + await asyncio.wait_for(contended.wait(), 10) + assert not held_worker[0].done() + release.set() + held_worker.append(asyncio.create_task(release_from_loop())) # These edits happen while an actual ASR worker releases the event loop. if change == 'regenerate': dub_generate._sync_job_segments(job, DubRequest( @@ -60,8 +89,19 @@ async def guarded(pool, fn, **kwargs): return recognition monkeypatch.setattr(asr_backend, 'run_transcribe_guarded', guarded) - run = lambda: asyncio.run(dub_export.dub_qc_pass('qc-race', lang='es', drift_threshold=.5)) - if change is None: + async def exercise(): + try: + result = await dub_export.dub_qc_pass('qc-race', lang='es', drift_threshold=.5) + if held_worker: + await asyncio.gather(*held_worker) + return result + finally: + for release in release_worker: + release.set() + if held_worker: + await asyncio.gather(*held_worker, return_exceptions=True) + run = lambda: asyncio.run(exercise()) + if change in {None, 'publication_wait'}: assert run()['flagged_count'] == 0 assert job['segments'][0]['qc_drift'] == 0 saved.assert_called_once() From 5bfd415368be731d4339c73a9c2161bc1ac6dcda Mon Sep 17 00:00:00 2001 From: debpalash <4178343+debpalash@users.noreply.github.com> Date: Sat, 3 Oct 2026 06:50:59 +0530 Subject: [PATCH 09/10] fix(dub): settle cancellation before withdrawing ingest state --- CHANGELOG.md | 2 +- backend/api/routers/dub_core.py | 9 +- backend/services/dub_pipeline.py | 50 +++++---- docs/electron-dubbing.md | 10 +- tests/test_dub_no_tts_load_for_asr.py | 43 +++++++- tests/test_dub_pipeline_state.py | 148 ++++++++++++++++++++++++++ 6 files changed, 235 insertions(+), 27 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index bbe53522d..48f61fd89 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -102,7 +102,7 @@ metadata and the backend fallback mirror it. ### Fixed -- Preserve subtitle edits made during transcription, keep dub lock waits off the event loop, and report a committed track as complete after late cancellation (#2585) +- Preserve subtitle edits made during transcription, keep dub lock waits off the event loop, report committed tracks as complete after late cancellation, and prevent cancelled ingest history from referencing deleted files (#2585) - Dub publication keeps file and database work off the event loop, preserves source metadata, waits safely on cancellation, and restores audio after save failures (#2585) - Dubbing finishes when quality-check annotations arrive during assembly and clears measurements of replaced audio while still protecting subtitle edits (#2585) diff --git a/backend/api/routers/dub_core.py b/backend/api/routers/dub_core.py index 50322e5ef..9d06e4e3e 100644 --- a/backend/api/routers/dub_core.py +++ b/backend/api/routers/dub_core.py @@ -300,11 +300,10 @@ def _publish_transcription(job_id, job, source_snapshot, updates, cancelled): async def _save_transcription(job_id, job, source_snapshot, updates): """Wait for the publication lock off-loop; cancelled queued work stays private.""" cancelled = threading.Event() - try: - await asyncio.to_thread(_publish_transcription, job_id, job, source_snapshot, updates, cancelled) - except asyncio.CancelledError: - cancelled.set() - raise + await dub_pipeline.run_job_operation( + _publish_transcription, job_id, job, source_snapshot, updates, cancelled, + on_cancel=cancelled.set, + ) def _mark_job_aborted(job): diff --git a/backend/services/dub_pipeline.py b/backend/services/dub_pipeline.py index dbb9859ec..bb4fa0626 100644 --- a/backend/services/dub_pipeline.py +++ b/backend/services/dub_pipeline.py @@ -305,7 +305,7 @@ def find_cached_job(content_hash: str, exclude_job_id: str) -> Optional[dict]: # ── Job state (in-memory + SQLite fallback) ──────────────────────────────── -async def run_job_operation(operation, *args, **kwargs): +async def run_job_operation(operation, *args, on_cancel=None, **kwargs): """Wait for blocking job work to settle before cancellation can clean up. The shared lock may be held during audio replacement and SQLite writes. @@ -319,6 +319,8 @@ async def run_job_operation(operation, *args, **kwargs): try: return await asyncio.shield(pending) except (asyncio.CancelledError, GeneratorExit): + if on_cancel is not None: + on_cancel() while not pending.done(): try: await asyncio.shield(pending) @@ -335,23 +337,26 @@ def get_job(job_id: str) -> Optional[dict]: """Look up a job. Checks the in-memory cache first, then falls back to `dub_history.job_data` so saved projects still resolve after restart. """ + # Keep lookup + hydration in the same transaction boundary as deletion + # and explicit same-id revival. Async callers dispatch this to a worker. with _dub_jobs_lock: if job_id in _dub_jobs: return _dub_jobs[job_id] - with db_conn() as conn: - row = conn.execute("SELECT job_data FROM dub_history WHERE id=?", (job_id,)).fetchone() - if row and row["job_data"]: - try: - job = json.loads(row["job_data"]) - with _dub_jobs_lock: + if job_id in _withdrawn_jobs: + return None + with db_conn() as conn: + row = conn.execute("SELECT job_data FROM dub_history WHERE id=?", (job_id,)).fetchone() + if row and row["job_data"]: + try: + job = json.loads(row["job_data"]) _dub_jobs[job_id] = job - return job - except json.JSONDecodeError: - # job_id arrives from request paths — strip newlines so a crafted - # id can't forge extra log lines (py/log-injection). - safe_id = str(job_id).replace("\r", "").replace("\n", "") - logger.exception("Failed to decode dub_history.job_data for %s", safe_id) - return None + return job + except json.JSONDecodeError: + # job_id arrives from request paths — strip newlines so a crafted + # id can't forge extra log lines (py/log-injection). + safe_id = str(job_id).replace("\r", "").replace("\n", "") + logger.exception("Failed to decode dub_history.job_data for %s", safe_id) + return None def put_job(job_id: str, job: dict) -> None: @@ -1308,6 +1313,18 @@ def _ts(s: str) -> float: return segments +def _discard_cancelled_ingest(job_id: str, job_dir: str) -> None: + """Withdraw persisted state before removing the files it references.""" + def delete_rows(): + with db_conn() as conn: + conn.execute("DELETE FROM dub_history WHERE id=?", (job_id,)) + + with _dub_jobs_lock: + purge_jobs([job_id], delete_rows=delete_rows) + # A database failure leaves the files intact for the retained row. + shutil.rmtree(job_dir, ignore_errors=True) + + async def ingest_pipeline( job_id: str, job_dir: str, @@ -1683,10 +1700,7 @@ def _yt_progress(d: dict) -> None: except asyncio.CancelledError: logger.info("Dub prep cancelled for job %s; killing subprocesses and cleaning up", log_safe(job_id)) kill_job_procs(job_id) - try: - shutil.rmtree(job_dir, ignore_errors=True) - finally: - _dub_jobs.pop(job_id, None) + await run_job_operation(_discard_cancelled_ingest, job_id, job_dir) yield prep_event("cancelled") raise except Exception as e: diff --git a/docs/electron-dubbing.md b/docs/electron-dubbing.md index d625cade0..06920e289 100644 --- a/docs/electron-dubbing.md +++ b/docs/electron-dubbing.md @@ -441,7 +441,11 @@ commit stays published and reports a completed task even if cancellation arrives during it. Async reads, export/QC updates and ingest persistence wait for the shared lock in workers, so they cannot prevent the event loop from handling cancellation while publication is waiting on disk. Deleting a job -serializes with publication and cannot leave a resurrected history row. +serializes with publication and cannot leave a resurrected history row. Cold +reads hold the same lock through SQLite hydration, including when a deleted +ID is explicitly revived for a new ingest. Cancelling +an ingest waits for an in-flight save, then withdraws its history row before +removing files. Subtitle imports re-read the current job when applying uploaded cues. Imports, caption cleanup, transcription source updates, and QC share the publication @@ -449,4 +453,6 @@ lock, so an edit arriving during publication is applied afterward. Transcription keeps new source fields private until completion and refuses to publish into a deleted or replaced job. If source text, timing or speaker assignments change during transcription, the result is rejected with a localized message and the -newer edits remain intact. Concurrently completed dub tracks are preserved. +newer edits remain intact. Queued transcription updates are cancelled before +admission; an already-admitted source commit settles before cancellation +returns. Concurrently completed dub tracks are preserved. diff --git a/tests/test_dub_no_tts_load_for_asr.py b/tests/test_dub_no_tts_load_for_asr.py index c92c84d94..0fd97e0fd 100644 --- a/tests/test_dub_no_tts_load_for_asr.py +++ b/tests/test_dub_no_tts_load_for_asr.py @@ -237,9 +237,13 @@ def holder(): try: await asyncio.wait_for(contended.wait(), 10) saving.cancel() + rendezvous = asyncio.Event() + loop.call_soon(rendezvous.set) + await rendezvous.wait() + assert not saving.done(), "cancelled ASR detached its queued source worker" + release.set() with pytest.raises(asyncio.CancelledError): await saving - release.set() await asyncio.wait_for(finished.wait(), 10) await holding assert dc._transcription_source(job) == original @@ -247,3 +251,40 @@ def holder(): release.set() await asyncio.gather(saving, holding, return_exceptions=True) asyncio.run(exercise()) + + +def test_cancellation_waits_for_admitted_transcription_commit(dub, monkeypatch): + import threading + + dc, job_id, _ = dub + job = dc._dub_jobs[job_id] + original = dc._transcription_source(job) + release = threading.Event() + completed = [] + async def exercise(): + loop = asyncio.get_running_loop() + entered = asyncio.Event() + def save(*_): + loop.call_soon_threadsafe(entered.set) + assert release.wait(2) + completed.append(True) + monkeypatch.setattr(dc, '_save_job', save) + saving = asyncio.create_task(dc._save_transcription( + job_id, job, original, {'full_transcript': 'committed transcript'}, + )) + try: + await asyncio.wait_for(entered.wait(), 10) + saving.cancel() + rendezvous = asyncio.Event() + loop.call_soon(rendezvous.set) + await rendezvous.wait() + assert not saving.done(), 'ASR commit worker escaped cancellation' + release.set() + with pytest.raises(asyncio.CancelledError): + await saving + assert completed == [True] + assert job['full_transcript'] == 'committed transcript' + finally: + release.set() + await asyncio.gather(saving, return_exceptions=True) + asyncio.run(exercise()) diff --git a/tests/test_dub_pipeline_state.py b/tests/test_dub_pipeline_state.py index 7e3049b6c..946430be1 100644 --- a/tests/test_dub_pipeline_state.py +++ b/tests/test_dub_pipeline_state.py @@ -212,3 +212,151 @@ async def ingest(): assert 'queued-ingest' not in dp._inflight_jobs assert 'queued-ingest' not in dp._dub_jobs assert any('cancelled' in event for event in events) + + +def test_cancelled_ingest_with_inflight_save_cannot_reload_deleted_files(monkeypatch, tmp_path): + import asyncio + import threading + from collections import OrderedDict + from types import SimpleNamespace + import soundfile as sf + from core import db + + monkeypatch.setattr(db, 'DB_PATH', str(tmp_path / 'cancelled-save.sqlite')) + monkeypatch.setattr(dp, '_dub_jobs', {}) + monkeypatch.setattr(dp, '_withdrawn_jobs', OrderedDict()) + monkeypatch.setattr(dp, '_inflight_jobs', set()) + with db.db_conn() as conn: + conn.executescript(db._BASE_SCHEMA) + job_dir = tmp_path / 'saving-ingest' + job_dir.mkdir() + source = job_dir / 'input.wav' + source.write_bytes(b'input fixture') + monkeypatch.setattr(dp, 'find_ffmpeg', lambda: 'test-ffmpeg') + monkeypatch.setattr(dp, 'validate_media_source', lambda *_: None) + monkeypatch.setattr(dp, 'require_audio_stream', lambda *_: None) + monkeypatch.setattr(dp, 'find_cached_job', lambda *_: None) + async def extract(cmd, **kwargs): + sf.write(cmd[-2], [.1] * 160, 16000) + return SimpleNamespace(returncode=0), b'', b'' + monkeypatch.setattr(dp, 'run_proc_factory', lambda _: extract) + release = threading.Event() + real_put = dp.put_and_save_job + async def exercise(): + loop = asyncio.get_running_loop() + persisted = asyncio.Event() + def slow_put(*args, **kwargs): + result = real_put(*args, **kwargs) + loop.call_soon_threadsafe(persisted.set) + assert release.wait(2), 'cancellation handling blocked the event loop' + return result + monkeypatch.setattr(dp, 'put_and_save_job', slow_put) + async def ingest(): + async for _ in dp.ingest_pipeline('saving-ingest', str(job_dir), { + 'kind': 'upload', 'input_type': 'audio', 'path': str(source), + }): + pass + task = asyncio.create_task(ingest()) + try: + await asyncio.wait_for(persisted.wait(), 10) + with db.db_conn() as conn: + assert conn.execute('SELECT id FROM dub_history WHERE id=?', ('saving-ingest',)).fetchone() + task.cancel() + rendezvous = asyncio.Event() + loop.call_soon(rendezvous.set) + await rendezvous.wait() + assert job_dir.exists() + assert not task.done() + release.set() + with pytest.raises(asyncio.CancelledError): + await task + finally: + release.set() + await asyncio.gather(task, return_exceptions=True) + asyncio.run(exercise()) + assert not job_dir.exists() + assert 'saving-ingest' not in dp._dub_jobs + assert dp.get_job('saving-ingest') is None, 'cancelled ingest reloaded a row pointing at deleted files' + dp.save_job('saving-ingest', {'filename': 'late-worker.wav'}) + assert dp.get_job('saving-ingest') is None, 'withdrawal must block a late save too' + + +@pytest.mark.parametrize('change', ['delete', 'replace', 'revive']) +def test_cold_job_read_serializes_with_deletion_and_replacement(monkeypatch, tmp_path, change): + import threading + from contextlib import contextmanager + from collections import OrderedDict + from core import db + + monkeypatch.setattr(db, 'DB_PATH', str(tmp_path / 'cold-read.sqlite')) + monkeypatch.setattr(dp, '_dub_jobs', {}) + monkeypatch.setattr(dp, '_withdrawn_jobs', OrderedDict()) + monkeypatch.setattr(dp, '_inflight_jobs', set()) + with db.db_conn() as conn: + conn.executescript(db._BASE_SCHEMA) + dp.save_job('cold-read', {'filename': 'old.wav'}) + read, release, contended = threading.Event(), threading.Event(), threading.Event() + original_lock = dp._dub_jobs_lock + class Lock: + def __enter__(self): + if not original_lock.acquire(blocking=False): + contended.set() + original_lock.acquire() + return self + def __exit__(self, *_): + original_lock.release() + monkeypatch.setattr(dp, '_dub_jobs_lock', Lock()) + @contextmanager + def paused_read(): + with db.db_conn() as conn: + class Conn: + def execute(self, *args): + row = conn.execute(*args).fetchone() + class Result: + def fetchone(self): + read.set() + assert release.wait(3) + order.append("read") + return row + return Result() + yield Conn() + monkeypatch.setattr(dp, 'db_conn', paused_read) + found, order = [], [] + replacement = {'filename': 'current.wav'} + def lookup(): + found.append(dp.get_job('cold-read')) + def mutate(): + if change in {'delete', 'revive'}: + def delete(): + with db.db_conn() as conn: + conn.execute('DELETE FROM dub_history WHERE id=?', ('cold-read',)) + dp.purge_jobs(['cold-read'], delete_rows=delete) + if change == 'revive': + dp.begin_ingest('cold-read') + else: + dp.put_job('cold-read', replacement) + order.append('mutate') + worker = threading.Thread(target=lookup) + mutation = threading.Thread(target=mutate) + worker.start() + try: + assert read.wait(2) + mutation.start() + assert contended.wait(2), 'cold hydration did not serialize with the source mutation' + release.set() + worker.join(2) + mutation.join(2) + assert not worker.is_alive() and not mutation.is_alive() + assert found == [{'filename': 'old.wav'}] + assert order == ['read', 'mutate'] + if change == 'replace': + assert dp._dub_jobs['cold-read'] is replacement + else: + assert 'cold-read' not in dp._dub_jobs + assert dp.get_job('cold-read') is None + assert ('cold-read' in dp._withdrawn_jobs) == (change == 'delete') + finally: + release.set() + worker.join(2) + if mutation.ident is not None: + mutation.join(2) From e87cba0449485b226674961be796b9d3e20ed0a1 Mon Sep 17 00:00:00 2001 From: debpalash <4178343+debpalash@users.noreply.github.com> Date: Sat, 3 Oct 2026 07:06:30 +0530 Subject: [PATCH 10/10] fix(pronunciation): retain unambiguous legacy Spanish scopes --- CHANGELOG.md | 2 + backend/services/pronunciation.py | 8 ++- docs/expressive-speech.md | 2 +- tests/test_pronunciation_language_scopes.py | 59 +++++++++++++++++++++ 4 files changed, 69 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 48f61fd89..d641b76f4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -102,6 +102,8 @@ metadata and the backend fallback mirror it. ### Fixed +- Keep saved legacy Spanish pronunciation scopes active without guessing ambiguous or unrelated language codes (#2585) + - Preserve subtitle edits made during transcription, keep dub lock waits off the event loop, report committed tracks as complete after late cancellation, and prevent cancelled ingest history from referencing deleted files (#2585) - Dub publication keeps file and database work off the event loop, preserves source metadata, waits safely on cancellation, and restores audio after save failures (#2585) - Dubbing finishes when quality-check annotations arrive during assembly and clears measurements of replaced audio while still protecting subtitle edits (#2585) diff --git a/backend/services/pronunciation.py b/backend/services/pronunciation.py index d0a78cbae..d0ace9e12 100644 --- a/backend/services/pronunciation.py +++ b/backend/services/pronunciation.py @@ -178,13 +178,17 @@ def save_lexicon(path, lexicon: Optional[dict]) -> dict[str, str]: _ALL_LANG = "*" _CHINESE_SCRIPT_SCOPES = {"cmn-hans", "cmn-hant", "zho-hans", "zho-hant"} +# Old writers truncated picker names. Spanish is the only bundled sp-prefix +# name, and sp is not an ISO language code. Do not infer more aliases merely +# from picker absence: ak/qu/vo, for example, are valid codes outside the map. +_LEGACY_SCOPE_ALIASES = {"sp": "es"} def normalize_language_scope(language: Optional[str]) -> Optional[str]: """Resolve picker names and ISO region tags to a dictionary language ID. Auto/unset/global requests have no language pin. Unknown values remain - literal: old truncated codes cannot be unambiguously assigned a language. + literal, except explicitly audited unambiguous legacy truncated scopes. """ if not language: return None @@ -196,6 +200,8 @@ def normalize_language_scope(language: Optional[str]) -> Optional[str]: return aliases[value] if value in LANG_NAME_TO_ID: return LANG_NAME_TO_ID[value] + if value in _LEGACY_SCOPE_ALIASES: + return _LEGACY_SCOPE_ALIASES[value] tag = value.replace("_", "-") if tag in _CHINESE_SCRIPT_SCOPES: return tag diff --git a/docs/expressive-speech.md b/docs/expressive-speech.md index c44608824..e2e78d6f9 100644 --- a/docs/expressive-speech.md +++ b/docs/expressive-speech.md @@ -64,7 +64,7 @@ for your engine below. Pronunciation dictionary matching uses Unicode case-insensitive literal matches. Each matched term uses its own respelling; distinct terms such as Straße and STRASSE can have different respellings. Longer terms win overlaps, and later equal-length case variants retain precedence. -Pronunciation scopes accept picker names such as Spanish, their bundled ISO IDs such as `es` or `kbt`, and regional forms such as `es-MX`; names resolve to the existing picker IDs. Explicit Chinese script scopes (`cmn-Hans`/`cmn-Hant` and `zho-Hans`/`zho-Hant`) remain distinct when saved or exported. A script-tagged request uses its own script scope before its base `cmn` or `zho` fallback, then the global scope; underscore spellings of these known script tags are equivalent. Unknown suffixes on `cmn`/`zho` remain literal scopes. Other alternate codes absent from the bundled engine map remain literal scopes (for example `spa` is not remapped to `es`). Global scopes still apply with Auto. Existing ambiguous truncated codes (for example `po`) keep their literal meaning: edit them to the intended name or ISO code rather than relying on an automatic migration. +Pronunciation scopes accept picker names such as Spanish, their bundled ISO IDs such as `es` or `kbt`, and regional forms such as `es-MX`; names resolve to the existing picker IDs. Explicit Chinese script scopes (`cmn-Hans`/`cmn-Hant` and `zho-Hans`/`zho-Hant`) remain distinct when saved or exported. A script-tagged request uses its own script scope before its base `cmn` or `zho` fallback, then the global scope; underscore spellings of these known script tags are equivalent. Unknown suffixes on `cmn`/`zho` remain literal scopes. Other alternate codes absent from the bundled engine map remain literal scopes (for example `spa` is not remapped to `es`). Global scopes still apply with Auto. The old Spanish picker scope `sp` remains compatible with Spanish/`es` requests: it is the only bundled language name with that prefix, and `sp` is not a language code. Reading or exporting existing rows does not rewrite them; saving or importing that scope writes canonical `es`. Other prefixes are not inferred from picker absence, because an absent code can still identify another language. Existing ambiguous truncated codes (for example `po` or `ge`) keep their literal meaning: edit them to the intended name or ISO code rather than relying on an automatic migration. Dictionary lists, previews, synthesis and exports use creation time, then insertion order for tied timestamps. A bulk import therefore keeps its authored entry order, and exporting/restoring the dictionary preserves duplicate and case-variant precedence. diff --git a/tests/test_pronunciation_language_scopes.py b/tests/test_pronunciation_language_scopes.py index 64d66b923..7cf8ff263 100644 --- a/tests/test_pronunciation_language_scopes.py +++ b/tests/test_pronunciation_language_scopes.py @@ -236,3 +236,62 @@ def test_inert_chinese_rows_share_supported_scope_matching(client, code): assert {entry["term"] for entry in result["inert_entries"]} == { "global", "base", "simplified", } + + +@pytest.mark.parametrize('language', ['Spanish', 'es', 'es-MX', 'sp']) +def test_saved_legacy_spanish_scope_still_applies_and_round_trips(client, language): + from core.db import db_conn + from services.pronunciation import apply_lexicon, load_dict_for_request + + # Old POST /pronunciation stored the picker name Spanish as its first two + # letters. Insert the actual old representation, bypassing today's writer. + with db_conn() as conn: + conn.execute( + 'INSERT INTO pronunciation_entries ' + '(id, term, replacement, type, language, enabled, created_at) ' + 'VALUES (?, ?, ?, ?, ?, ?, ?)', + ('legacy-spanish', 'GIF', 'jiff', 'respelling', 'sp', 1, '2026-01-01'), + ) + assert client.post('/pronunciation/test', json={ + 'text': 'GIF', 'language': language, + }).json()['substituted'] == 'jiff' + assert apply_lexicon('GIF', load_dict_for_request(language)) == 'jiff' + exported = client.get('/pronunciation/export').json() + assert exported['entries'][0]['language'] == 'sp', 'reads must not rewrite user rows' + assert client.post('/pronunciation/import', json={**exported, 'replace': True}).status_code == 200 + assert client.get('/pronunciation/export').json()['entries'][0]['language'] == 'es' + assert client.post('/pronunciation/test', json={ + 'text': 'GIF', 'language': language, + }).json()['substituted'] == 'jiff' + + +def test_legacy_spanish_never_leaks_into_another_bundled_language(): + from omnivoice.utils.lang_map import LANG_NAME_TO_ID + from services.pronunciation import apply_pronunciation + + row = {'term': 'GIF', 'replacement': 'jiff', 'type': 'respelling', + 'language': 'sp', 'enabled': 1} + for name, code in LANG_NAME_TO_ID.items(): + expected = 'jiff' if code == 'es' else 'GIF' + assert apply_pronunciation('GIF', [row], name) == expected, name + assert apply_pronunciation('GIF', [row], code) == expected, code + + +@pytest.mark.parametrize('scope,languages', [ + ('ge', ['German', 'Georgian', 'de', 'ka']), + ('po', ['Polish', 'Portuguese', 'pl', 'pt']), + ('ak', ['Akebu', 'keu']), + ('qu', ['Quiotepec Chinantec', 'chq']), + ('vo', ['Votic', 'vot']), +]) +def test_other_legacy_or_literal_codes_are_not_guessed(scope, languages): + from services.pronunciation import apply_pronunciation, normalize_language_scope + + # A unique picker prefix is insufficient: e.g. ak/qu/vo are real ISO codes + # outside this picker, while ge/po have multiple possible source names. + row = {'term': 'GIF', 'replacement': 'literal', 'type': 'respelling', + 'language': scope, 'enabled': 1} + assert normalize_language_scope(scope) == scope + assert apply_pronunciation('GIF', [row], scope) == 'literal' + for language in languages: + assert apply_pronunciation('GIF', [row], language) == 'GIF'