fix: normalize faster-whisper TranscriptionInfo to dict

Colab run 4 (A100): same namedtuple bug as segments — faster-whisper
returns TranscriptionInfo (namedtuple) but batch.py reads
outcome['info'].get('language') -> AttributeError 'TranscriptionInfo'
object has no attribute 'get' on every real transcription, failing the
CLI, API auto-worker job, worker drain, and bench clip alike.

FasterWhisperEngine now normalizes info to a dict at the boundary
(_to_dict_info: _asdict -> dataclasses.asdict -> known-field fallback).

+ 2 unit tests (namedtuple/dict info); 138 tests pass, ruff clean.
This commit is contained in:
2026-08-12 17:39:15 +09:00
parent 20777386fe
commit be5f505410
2 changed files with 61 additions and 3 deletions
+34 -2
View File
@@ -16,17 +16,18 @@ from luke_scribe.engine.faster_whisper_engine import FasterWhisperEngine
class FakeWhisperModel:
"""faster_whisper.WhisperModel 대체 — 생성 인자/세그먼트를 기록한다."""
"""faster_whisper.WhisperModel 대체 — 생성 인자/세그먼트/info를 기록한다."""
calls: list[dict] = []
segments: list = [] # transcribe()가 yield할 세그먼트 (기본: namedtuple)
info: object = None # 기본: TranscriptionInfo(namedtuple) 흉내
def __init__(self, *args, **kwargs) -> None:
self.kwargs = kwargs
self.__class__.calls.append(kwargs)
def transcribe(self, audio_path, **kwargs):
return iter(list(self.__class__.segments)), {}
return iter(list(self.__class__.segments)), self.__class__.info
def _install_fake(monkeypatch) -> None:
@@ -35,6 +36,7 @@ def _install_fake(monkeypatch) -> None:
monkeypatch.setitem(sys.modules, "faster_whisper", mod)
FakeWhisperModel.calls.clear()
FakeWhisperModel.segments = []
FakeWhisperModel.info = None
def _opts(**kw) -> TranscriptionOptions:
@@ -137,3 +139,33 @@ def test_dict_segments_passthrough(monkeypatch):
)
segs = list(outcome.segments)
assert segs == [{"index": 0, "start": 0.0, "end": 1.0, "text": "x"}]
def test_namedtuple_info_normalized_to_dict(monkeypatch):
"""faster-whisper TranscriptionInfo(namedtuple) → dict (GPU 실전 버그)."""
from collections import namedtuple
_install_fake(monkeypatch)
Info = namedtuple(
"TranscriptionInfo",
["language", "language_probability", "duration", "duration_after_vad"],
)
FakeWhisperModel.info = Info(
language="ko", language_probability=0.99, duration=10.464, duration_after_vad=9.088
)
outcome = FasterWhisperEngine().transcribe(
"/tmp/x.wav", _opts(device="cpu", compute_type="int8")
)
# dict 계약: .get() 사용 가능 (batch.py가 이걸로 접근)
assert outcome.info.get("language") == "ko"
assert outcome.info.get("duration") == 10.464
def test_dict_info_passthrough(monkeypatch):
"""이미 dict인 info는 그대로 (mock 계약과 호환)."""
_install_fake(monkeypatch)
FakeWhisperModel.info = {"language": "ko"}
outcome = FasterWhisperEngine().transcribe(
"/tmp/x.wav", _opts(device="cpu", compute_type="int8")
)
assert outcome.info == {"language": "ko"}