- Зачем:
- CTranslate2 по умолчанию использует 4 потока; на многоядерных CPU (8+ ядер)
это неоптимально — --threads 8 даёт +13% ускорения.
- Не было данных по int8_float32/int8_float16 на NVIDIA GPU.
- Что:
- --threads / -t: новый CLI-флаг, пробрасывается через load_model →
backend.create_model(cpu_threads=...) → WhisperModel(cpu_threads=...).
- Валидация min=0 на входе (typer), Backend протокол синхронизирован.
- docs/gpu.md: результаты бенчмарка 6 комбинаций CUDA compute_type
(medium/large-v3 × float16/int8_float32/int8_float16) на двух файлах
(16 мин и 46 мин). Ключевой вывод: float16 — оптимальный дефолт;
large-v3 ненадёжен на длинных записях.
- README: --threads добавлен в таблицу опций.
- Фикс теста: test_resolve_repo_explicit_unsupported_pair_raises обновлён
под добавление medium fp16 модели.
- Проверка:
- uv run pytest: 157 passed.
- transcribe file.mp4 --device cpu --threads 8: 277с vs 320с (дефолт).
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
377 lines
14 KiB
Python
377 lines
14 KiB
Python
"""CLI-точка входа (typer). Single и batch режимы транскрипции."""
|
||
|
||
import sys
|
||
import time
|
||
from pathlib import Path
|
||
|
||
import typer
|
||
from rich.console import Console
|
||
from rich.status import Status
|
||
|
||
from .config import apply_device_defaults, load_config, resolve_defaults
|
||
from .formatter import format_transcript, write_transcript
|
||
from .transcriber import (
|
||
Segment,
|
||
_is_cuda_error,
|
||
_transcribe_file,
|
||
load_model,
|
||
)
|
||
from .utils import (
|
||
build_output_path,
|
||
detect_device,
|
||
expand_globs,
|
||
get_gpu_name,
|
||
get_intel_gpu_name,
|
||
has_existing_transcript,
|
||
validate_input_file,
|
||
)
|
||
|
||
app = typer.Typer()
|
||
console = Console(stderr=True)
|
||
|
||
|
||
def _format_device_info(device_used: str) -> str:
|
||
"""Формирует строку устройства для шапки транскрипта."""
|
||
if device_used == "cuda":
|
||
gpu_name = get_gpu_name()
|
||
return f"CUDA ({gpu_name or 'Unknown GPU'})"
|
||
if device_used == "openvino-gpu":
|
||
gpu_name = get_intel_gpu_name()
|
||
return f"OpenVINO ({gpu_name or 'Intel GPU'})"
|
||
if device_used in ("openvino", "openvino-cpu"):
|
||
return "OpenVINO (CPU)"
|
||
return "CPU"
|
||
|
||
|
||
@app.command()
|
||
def main(
|
||
files: list[Path] = typer.Argument(..., help="Пути к аудио/видеофайлам"),
|
||
model: str | None = typer.Option(
|
||
None, "--model", "-m", show_default=False, help="Модель Whisper [по умолч.: medium]"
|
||
),
|
||
language: str | None = typer.Option(
|
||
None, "--language", "-l", show_default=False, help="Язык [по умолч.: ru]"
|
||
),
|
||
output: Path | None = typer.Option(None, "--output", "-o", help="Путь к выходному файлу"),
|
||
device: str | None = typer.Option(
|
||
None, "--device", "-d", show_default=False,
|
||
help="Устройство (auto|cpu|cuda|openvino|openvino-gpu|openvino-cpu) [по умолч.: auto]"
|
||
),
|
||
compute_type: str | None = typer.Option(
|
||
None, "--compute-type", show_default=False,
|
||
help="Тип вычислений [по умолч.: float16 (CUDA) / int8 (OpenVINO GPU/CPU) / float32 (CPU)]"
|
||
),
|
||
threads: int = typer.Option(
|
||
0, "--threads", "-t", show_default=False, min=0,
|
||
help="Потоки CPU (0 = дефолт библиотеки; рекомендуется = число физ. ядер)"
|
||
),
|
||
verbose: bool = typer.Option(False, "--verbose", "-v", help="Подробный вывод"),
|
||
force: bool = typer.Option(False, "--force", "-f", help="Перезаписать существующие транскрипты"),
|
||
) -> None:
|
||
"""Транскрибирует аудио/видеофайлы в markdown с таймкодами.
|
||
|
||
Каскад приоритетов параметров: CLI-флаги > .transcriber.toml > device-aware дефолты.
|
||
"""
|
||
try:
|
||
config = load_config()
|
||
cli_values = {"model": model, "language": language, "device": device, "compute_type": compute_type}
|
||
defaults = resolve_defaults(cli_values, config)
|
||
|
||
resolved_device = detect_device(defaults["device"])
|
||
defaults = apply_device_defaults(defaults, resolved_device, cli_values, config)
|
||
|
||
ct_explicit = compute_type is not None or "compute_type" in config
|
||
|
||
expanded = expand_globs(files)
|
||
if not expanded:
|
||
console.print("Файлы не найдены.", style="red bold")
|
||
raise SystemExit(1)
|
||
|
||
is_batch = len(expanded) > 1
|
||
if is_batch and output is not None:
|
||
console.print("--output несовместим с несколькими файлами.", style="red bold")
|
||
raise SystemExit(1)
|
||
|
||
if is_batch:
|
||
_run_batch(expanded, defaults, verbose, force, ct_explicit, cpu_threads=threads)
|
||
else:
|
||
_run_single(expanded[0], defaults, output, verbose, ct_explicit, cpu_threads=threads)
|
||
except KeyboardInterrupt:
|
||
console.print("\nПрервано пользователем.", style="yellow")
|
||
raise SystemExit(130)
|
||
except SystemExit:
|
||
raise
|
||
except ValueError as exc:
|
||
console.print(f"Ошибка: {exc}", style="red bold")
|
||
raise SystemExit(1)
|
||
except (FileNotFoundError,) as exc:
|
||
console.print(f"Ошибка: {exc}", style="red bold")
|
||
raise SystemExit(1)
|
||
except Exception as exc:
|
||
if _is_cuda_error(exc) and sys.platform == "win32":
|
||
console.print(
|
||
"GPU на Windows требует CUDA toolkit (включает cuBLAS).\n"
|
||
"Установите одним из способов:\n"
|
||
" choco install cuda\n"
|
||
" winget install -e --id Nvidia.CUDA\n"
|
||
"После установки перезапустите терминал.",
|
||
style="yellow",
|
||
)
|
||
if verbose:
|
||
console.print_exception()
|
||
else:
|
||
console.print(f"Ошибка: {exc}", style="red bold")
|
||
console.print(
|
||
"Запустите с --verbose для полного traceback.", style="dim"
|
||
)
|
||
raise SystemExit(1)
|
||
|
||
|
||
def _run_single(
|
||
file: Path,
|
||
defaults: dict[str, str],
|
||
output: Path | None,
|
||
verbose: bool,
|
||
compute_type_explicit: bool = False,
|
||
cpu_threads: int = 0,
|
||
) -> None:
|
||
"""Пайплайн одного файла: валидация → модель → транскрипция → запись."""
|
||
start = time.monotonic()
|
||
|
||
validated_file = validate_input_file(file)
|
||
requested_device = defaults["device"]
|
||
resolved_device = detect_device(requested_device)
|
||
strict = requested_device != "auto"
|
||
output_path = build_output_path(validated_file, output)
|
||
|
||
console.print(f"Файл: [bold]{validated_file.name}[/bold]")
|
||
|
||
def on_segment(seg: Segment) -> None:
|
||
console.print(f" [{seg.start:.2f}s] {seg.text.strip()}")
|
||
|
||
model_obj, actual_device, backend, model_path = load_model(
|
||
defaults["model"], resolved_device, defaults["compute_type"],
|
||
on_status=lambda msg: console.print(msg), strict_device=strict,
|
||
compute_type_explicit=compute_type_explicit,
|
||
cpu_threads=cpu_threads,
|
||
)
|
||
actual_ct = getattr(backend, "actual_compute_type", defaults["compute_type"]) or defaults["compute_type"]
|
||
console.print(
|
||
f"Модель: [bold]{defaults['model']}[/bold] "
|
||
f"Устройство: [bold]{actual_device}[/bold] "
|
||
f"Compute: [bold]{actual_ct}[/bold]"
|
||
)
|
||
if actual_device == "openvino-gpu" and defaults["model"] != "large-v3":
|
||
console.print(
|
||
"Совет: --model large-v3 даёт лучшее качество на GPU (~2x дольше)",
|
||
style="dim",
|
||
)
|
||
|
||
with Status("Подготавливаю запуск...", console=console) as status:
|
||
tfr = _transcribe_file(
|
||
model=model_obj,
|
||
actual_device=actual_device,
|
||
backend=backend,
|
||
model_path=model_path,
|
||
file_path=validated_file,
|
||
model_name=defaults["model"],
|
||
compute_type=defaults["compute_type"],
|
||
language=defaults["language"] if defaults["language"] != "auto" else None,
|
||
on_segment=on_segment if verbose else None,
|
||
on_status=status.update,
|
||
strict_device=strict,
|
||
cpu_threads=cpu_threads,
|
||
)
|
||
|
||
result = tfr.result
|
||
|
||
if tfr.actual_device != resolved_device:
|
||
if requested_device == "auto":
|
||
console.print(
|
||
f"Определено устройство {resolved_device}, "
|
||
f"но использовано {tfr.actual_device} (fallback)",
|
||
style="yellow",
|
||
)
|
||
else:
|
||
console.print(
|
||
f"Запрошено {requested_device}, использовано {tfr.actual_device}",
|
||
style="yellow",
|
||
)
|
||
|
||
if len(result.segments) == 0:
|
||
console.print(
|
||
f"Речь не обнаружена в файле {validated_file.name}", style="yellow"
|
||
)
|
||
|
||
device_info = _format_device_info(result.device_used)
|
||
language_mode = "detected" if defaults["language"] == "auto" else "forced"
|
||
|
||
content = format_transcript(
|
||
result=result,
|
||
source_filename=validated_file.name,
|
||
model_name=defaults["model"],
|
||
device_info=device_info,
|
||
language_mode=language_mode,
|
||
)
|
||
write_transcript(content, output_path)
|
||
|
||
elapsed = time.monotonic() - start
|
||
console.print(f"Транскрипт сохранён: \"{output_path}\"", style="green")
|
||
console.print(f" Сегментов: {len(result.segments)} Время: {elapsed:.1f}с")
|
||
|
||
|
||
def _run_batch(
|
||
files: list[Path],
|
||
defaults: dict[str, str],
|
||
verbose: bool,
|
||
force: bool,
|
||
compute_type_explicit: bool = False,
|
||
cpu_threads: int = 0,
|
||
) -> None:
|
||
"""Трёхфазный батч-пайплайн: prescan → загрузка модели → транскрипция."""
|
||
# Phase 1: Prescan — fail-fast + skip до загрузки модели (экономим ~2-5 сек)
|
||
to_process: list[Path] = []
|
||
skipped = 0
|
||
invalid = 0
|
||
for file in files:
|
||
try:
|
||
validated = validate_input_file(file)
|
||
except (FileNotFoundError, ValueError) as exc:
|
||
console.print(f" Ошибка: {file.name}: {exc}", style="red")
|
||
invalid += 1
|
||
continue
|
||
if not force and has_existing_transcript(validated):
|
||
console.print(
|
||
f" Пропуск: {file.name} (транскрипт существует)", style="dim"
|
||
)
|
||
skipped += 1
|
||
continue
|
||
to_process.append(validated)
|
||
|
||
if not to_process:
|
||
console.print(
|
||
f"\nИтого: 0 обработано, {skipped} пропущено, {invalid} ошибок"
|
||
)
|
||
if invalid > 0:
|
||
raise SystemExit(1)
|
||
return
|
||
|
||
# Phase 2: Load model (ensure + create в одном вызове)
|
||
requested_device = defaults["device"]
|
||
resolved_device = detect_device(requested_device)
|
||
strict = requested_device != "auto"
|
||
model_obj, actual_device, backend, model_path = load_model(
|
||
defaults["model"], resolved_device, defaults["compute_type"],
|
||
on_status=lambda msg: console.print(msg), strict_device=strict,
|
||
compute_type_explicit=compute_type_explicit,
|
||
cpu_threads=cpu_threads,
|
||
)
|
||
|
||
if actual_device == "openvino-gpu" and defaults["model"] != "large-v3":
|
||
console.print(
|
||
"Совет: --model large-v3 даёт лучшее качество на GPU (~2x дольше)",
|
||
style="dim",
|
||
)
|
||
|
||
if actual_device != resolved_device:
|
||
if requested_device == "auto":
|
||
console.print(
|
||
f"Определено устройство {resolved_device}, "
|
||
f"но используется {actual_device} (fallback)",
|
||
style="yellow",
|
||
)
|
||
else:
|
||
console.print(
|
||
f"Запрошено {requested_device}, используется {actual_device}",
|
||
style="yellow",
|
||
)
|
||
|
||
# Phase 3: Transcribe
|
||
processed = 0
|
||
failed = 0
|
||
language_mode = "detected" if defaults["language"] == "auto" else "forced"
|
||
|
||
batch_start = time.monotonic()
|
||
|
||
for i, file in enumerate(to_process, 1):
|
||
try:
|
||
prefix = f"[{i}/{len(to_process)}] {file.name}"
|
||
console.print(f"{prefix}", style="bold")
|
||
file_start = time.monotonic()
|
||
|
||
def on_segment(seg: Segment) -> None:
|
||
console.print(f" [{seg.start:.2f}s] {seg.text.strip()}")
|
||
|
||
with Status(f"{prefix}...", console=console) as status:
|
||
tfr = _transcribe_file(
|
||
model=model_obj,
|
||
actual_device=actual_device,
|
||
backend=backend,
|
||
model_path=model_path,
|
||
file_path=file,
|
||
model_name=defaults["model"],
|
||
compute_type=defaults["compute_type"],
|
||
language=defaults["language"] if defaults["language"] != "auto" else None,
|
||
on_segment=on_segment if verbose else None,
|
||
on_status=status.update if not verbose else lambda msg: console.print(msg),
|
||
strict_device=strict,
|
||
cpu_threads=cpu_threads,
|
||
)
|
||
|
||
if tfr.actual_device != actual_device:
|
||
console.print(
|
||
f" {file.name}: fallback на {tfr.actual_device} при транскрипции",
|
||
style="yellow",
|
||
)
|
||
# Обновляем после возможного mid-stream fallback
|
||
model_obj = tfr.model
|
||
actual_device = tfr.actual_device
|
||
backend = tfr.backend
|
||
model_path = tfr.model_path
|
||
|
||
result = tfr.result
|
||
|
||
if len(result.segments) == 0:
|
||
console.print(
|
||
f" Речь не обнаружена: {file.name}", style="yellow"
|
||
)
|
||
|
||
device_info = _format_device_info(result.device_used)
|
||
|
||
content = format_transcript(
|
||
result=result,
|
||
source_filename=file.name,
|
||
model_name=defaults["model"],
|
||
device_info=device_info,
|
||
language_mode=language_mode,
|
||
)
|
||
write_transcript(content, build_output_path(file))
|
||
file_elapsed = time.monotonic() - file_start
|
||
console.print(
|
||
f" Готово: {file.name} "
|
||
f"Сегментов: {len(result.segments)} Время: {file_elapsed:.1f}с",
|
||
style="green",
|
||
)
|
||
processed += 1
|
||
except KeyboardInterrupt:
|
||
raise
|
||
except Exception as exc:
|
||
if verbose:
|
||
console.print_exception()
|
||
else:
|
||
console.print(f" Ошибка: {file.name}: {exc}", style="red")
|
||
failed += 1
|
||
|
||
total_failed = invalid + failed
|
||
batch_elapsed = time.monotonic() - batch_start
|
||
console.print(
|
||
f"\nИтого: {processed} обработано, {skipped} пропущено, {total_failed} ошибок"
|
||
f" Время: {batch_elapsed:.1f}с"
|
||
)
|
||
if total_failed > 0:
|
||
raise SystemExit(1)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
app()
|