From 8a86e7e7ac675c89e159a08a09d88448e93bd5c3 Mon Sep 17 00:00:00 2001 From: EJClaw Date: Wed, 23 Sep 2026 03:55:22 +0900 Subject: [PATCH] =?UTF-8?q?fix:=20=ED=8B=B0=EC=96=B4=20=EA=B2=8C=EC=9D=B4?= =?UTF-8?q?=ED=8A=B8=EA=B0=80=20=EC=8B=A4=EC=A0=9C=20=EC=A0=9C=EC=95=BD(au?= =?UTF-8?q?toawq=C2=B7VRAM)=EC=9D=84=20=EB=B3=B4=EB=8F=84=EB=A1=9D=20?= =?UTF-8?q?=EC=88=98=EC=A0=95?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 실제 GPU 로 검증을 돌리다 발견한 버그다. tier_availability() 가 torch 유무만 봐서, RTX 5050(7.5GB)에서도 정밀·극한 티어를 "사용 가능"이라고 답했다. 정밀(AWQ Int4) autoawq 가 있어야 로드된다. requirements.txt 에 없는 선택 의존성이라 대개 없다. 극한(Qwen3-8B) min_vram_gb=16. 7.5GB 카드에서는 영영 불가능하다. 둘 다 "고를 수 있다"고 해놓고 로드에서 죽는다. 사용자는 원인을 알 수 없다. UI 에 min_vram_gb 경고 배지는 있었지만 경고만 하고 선택은 막지 않았다. - tier_availability(tier, gpu=None) 에 autoawq·VRAM 검사 추가. 새 필드를 만들지 않고 이미 있던 min_vram_gb 를 쓴다. free 가 아니라 total 로 본다 — free 로 막으면 다른 프로그램 때문에 "아까는 되던 티어가 지금은 안 보인다"가 된다. CT2 티어(1~3)는 조기 반환이라 detect_gpu 를 아예 부르지 않는다. (포터블에서 torch 도 GPU 도 없이 돌아야 하므로) - GpuInfo 를 인자로 받아 루프에서 nvidia-smi 반복 호출을 막았다. models_page 는 티어 카드 5개를 만들며 매번 조회하던 것을 __init__ 에서 한 번만 하도록 바꿨다. 빌더 함수의 부수효과에 의존하던 것도 제거. - 점검기의 판정 기준도 고쳤다. 기존엔 "4·5티어가 전부 열려야 통과"였는데 VRAM 작은 GPU 에서 막히는 건 정상이라 오탐이었다. 이제 게이트는 판정 없이 보고만 하고, "게이트가 된다고 한 티어가 실제로 올라가는지" 가 판정한다. 게이트가 전부 막았으면 검증 대상이 없다는 사실을 남기고 통과시킨다 — 하드웨어 한계이지 회귀가 아니다. - scripts/gpu-check.sh 추가. CTranslate2 가 libcublas.so.12 를 직접 찾는데 torch 의 nvidia 패키지 안에 있어 LD_LIBRARY_PATH 가 필요하다. 없으면 GPU 번역만 조용히 실패한다. 스크립트가 경로를 자동으로 잡는다. 검증: pytest 190 passed (신규 4개 중 3개는 torch 필요 -> gpu-venv 에서 실제로 실행해 9 passed 확인), ruff clean, RTX 5050 에서 scripts/gpu-check.sh 6/6 통과 (종료코드 0) Co-Authored-By: Claude Opus 4.7 --- docs/WINDOWS-TESTING.md | 23 +++++++ packaging/windows_smoke.py | 95 +++++++++++++---------------- scripts/gpu-check.sh | 34 +++++++++++ src/livesub/ui/pages/models_page.py | 15 +++-- tests/test_portable.py | 51 ++++++++++++++++ 5 files changed, 159 insertions(+), 59 deletions(-) create mode 100755 scripts/gpu-check.sh diff --git a/docs/WINDOWS-TESTING.md b/docs/WINDOWS-TESTING.md index d48250d..2d5547a 100644 --- a/docs/WINDOWS-TESTING.md +++ b/docs/WINDOWS-TESTING.md @@ -22,6 +22,29 @@ Windows 러너를 하나 붙이면 `main` 에 push할 때(또는 Actions 탭에 | **오디오 → 한국어 자막 (전 과정)** | ❌ CPU로 됨 | ✅ | | CUDA 인식·VRAM 상한 | ✅ | ⚠️ 경고만 (빌드 통과) | +### GPU 검증은 `.9`(리눅스)에서 — 이미 돌아갑니다 + +GPU는 `.9`에 고정입니다. 그래서 GPU 검증은 Windows가 아니라 **여기서** 합니다. +한 줄이면 됩니다. + +```bash +scripts/gpu-check.sh +``` + +RTX 5050 실측 결과입니다. + +``` +[OK] CUDA 사용 가능 RTX 5050 · 7.5GB +[OK] VRAM 상한이 실제로 걸리는지 35% 상한 적용됨 +[OK] GPU 로 실제 번역 '적 왼쪽에서 오는' +[OK] 티어 게이트가 뭐라고 하는지 정밀=autoawq 없음 / 극한=VRAM 부족(16GB 필요, 7.5GB) +6/6 통과 +``` + +**이 GPU로는 4·5티어를 쓸 수 없습니다.** 정밀은 `autoawq`가 필요하고 +(`pip install autoawq`), 극한은 16GB가 필요한데 카드가 7.5GB입니다. +극한은 카드를 바꾸지 않는 한 영영 불가능합니다. 1~3티어는 정상 동작합니다. + ### GPU가 필요한 테스트는 어떻게 하나 **GPU는 한 VM만 독점합니다.** 컨슈머 NVIDIA는 vGPU/SR-IOV를 지원하지 않아서 diff --git a/packaging/windows_smoke.py b/packaging/windows_smoke.py index 91d3235..debfb74 100644 --- a/packaging/windows_smoke.py +++ b/packaging/windows_smoke.py @@ -273,67 +273,58 @@ def run_gpu_checks() -> None: raise RuntimeError("상한 적용 실패 — 게임과 같이 쓸 때 VRAM 을 못 막습니다") return "35% 상한 적용됨" - @check("torch 필요 티어(4·5) 게이트") + @check("티어 게이트가 뭐라고 하는지") def _gate(): - """앱이 '이 티어 쓸 수 있다'고 답하는지. + """앱이 각 torch 티어를 쓸 수 있다고 하는지, 아니면 왜 막는지. - 주의: tier_availability() 는 torch 유무만 본다. autoawq 유무나 VRAM - 용량은 보지 않으므로, 여기서 통과해도 실제로 올라간다는 뜻은 아니다. - 그건 아래 항목에서 실제로 올려서 확인한다. + 여기서는 판정하지 않고 그대로 보고만 한다. VRAM 이 작은 GPU 에서 + 4·5티어가 막히는 건 정상이라 실패로 처리하면 안 된다. 진짜 판정은 + 아래 "실제 로드" 가 한다 — 게이트가 된다고 한 걸 못 올리면 그게 버그다. """ - from livesub.models.manager import has_torch, tier_availability - from livesub.models.tiers import TIERS, MTBackend + from livesub.models.manager import detect_gpu, has_torch, tier_availability + from livesub.models.tiers import MTBackend, ordered_tiers if not has_torch(): - raise RuntimeError("torch 없음 — GPU 러너에는 일반 설치가 필요합니다") - blocked = [ - t.name for t in TIERS.values() - if t.mt.backend is MTBackend.TRANSFORMERS and not tier_availability(t)[0] - ] - if blocked: - raise RuntimeError(f"막힌 티어: {', '.join(blocked)}") - return "정밀·극한 게이트 통과 (얕은 검사)" - - @check("torch 티어 실제 로드 + 번역") - def _real_load(): - """게이트가 아니라 **진짜로 모델을 올려서** 번역까지 해본다. - - 티어마다 요구 VRAM 이 다르고(정밀 5.6GB / 극한 11GB) 정밀 티어는 - autoawq 까지 필요하다. 러너 GPU 가 무엇일지 모르므로, 올릴 수 있는 - 것 중 가장 무거운 티어를 골라서 검증한다. - - 게이트는 통과인데 여기서 실패하면 그게 바로 찾아야 할 버그다 — - 앱이 사용자에게 "쓸 수 있다"고 하고선 실제로는 못 올리는 상황이다. - """ - import importlib.util - - import torch - - from livesub.models.glossary import Glossary - from livesub.models.tiers import MTBackend, ordered_tiers - from livesub.models.translator import create_translator - - free_mb = torch.cuda.mem_get_info()[0] / 1024**2 - has_awq = importlib.util.find_spec("awq") is not None - - fits, skipped = [], [] + raise RuntimeError("torch 없음 — GPU 검증에는 일반 설치가 필요합니다") + gpu = detect_gpu() + parts = [] for t in ordered_tiers(): if t.mt.backend is not MTBackend.TRANSFORMERS: continue - if t.mt.compute_type == "int4" and not has_awq: - skipped.append(f"{t.name}(autoawq 없음)") - elif t.mt.vram_mb > free_mb * 0.9: - skipped.append(f"{t.name}({t.mt.vram_mb}MB 필요)") - else: - fits.append(t) + ok, reason = tier_availability(t, gpu) + parts.append(f"{t.name}={'가능' if ok else reason}") + return " / ".join(parts) - if not fits: - raise RuntimeError( - f"여유 VRAM {free_mb:.0f}MB 로 올릴 수 있는 torch 티어가 없습니다. " - f"건너뜀: {', '.join(skipped)}" + @check("게이트가 된다고 한 티어가 실제로 올라가는지") + def _real_load(): + """게이트 판정을 **실물로 검증**한다. + + 게이트가 "가능"이라 한 torch 티어를 실제로 올려 번역까지 시킨다. + 여기서 죽으면 앱이 사용자에게 거짓말을 하고 있다는 뜻이다. + + 게이트가 전부 막았다면 검증할 대상이 없다. 그건 이 GPU 의 한계이지 + 회귀가 아니므로 실패로 처리하지 않고 이유를 남긴다. + """ + from livesub.models.glossary import Glossary + from livesub.models.manager import detect_gpu, tier_availability + from livesub.models.tiers import MTBackend, ordered_tiers + from livesub.models.translator import create_translator + + gpu = detect_gpu() + allowed, blocked = [], [] + for t in ordered_tiers(): + if t.mt.backend is not MTBackend.TRANSFORMERS: + continue + ok, reason = tier_availability(t, gpu) + (allowed if ok else blocked).append(t if ok else f"{t.name}({reason})") + + if not allowed: + return ( + "검증 대상 없음 — 이 GPU 로는 torch 티어를 못 씁니다. " + f"{'; '.join(blocked)}" ) - tier = fits[-1] # 올릴 수 있는 것 중 가장 무거운 것 + tier = allowed[-1] # 열린 것 중 가장 무거운 것 translator = create_translator(tier.mt, device="cuda") translator.load() try: @@ -342,9 +333,7 @@ def run_gpu_checks() -> None: translator.unload() if not any("\uac00" <= ch <= "\ud7a3" for ch in out): raise RuntimeError(f"{tier.name}: 한국어가 아닙니다 — {out!r}") - - note = f" (건너뜀: {', '.join(skipped)})" if skipped else "" - return f"{tier.name} 로드·번역 성공 {out!r}{note}" + return f"{tier.name} 로드·번역 성공 {out!r}" @check("GPU 로 실제 번역 (1~3티어 CUDA 경로)") def _translate(): diff --git a/scripts/gpu-check.sh b/scripts/gpu-check.sh new file mode 100755 index 0000000..f85872f --- /dev/null +++ b/scripts/gpu-check.sh @@ -0,0 +1,34 @@ +#!/usr/bin/env bash +# GPU 검증 — RTX 5050 이 달린 .9(리눅스)에서 돌린다. +# +# GPU 는 한 VM 만 독점하므로 Windows CI VM 에는 GPU 가 없다. 그래서 CUDA +# 인식·VRAM 상한·티어 게이트·실제 GPU 번역은 GPU 가 실제로 있는 여기서 +# 확인한다. Windows 전용 경로(오디오·단축키·exe)는 Windows 러너가 맡는다. +# +# scripts/gpu-check.sh +# +# 종료코드 0 이면 전부 통과. 항목 하나라도 실패하면 1. +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +VENV="${GPU_VENV:-/home/claude/gpu-venv}" + +if [[ ! -x "$VENV/bin/python" ]]; then + echo "GPU 용 파이썬이 없습니다: $VENV/bin/python" >&2 + echo "GPU_VENV 로 경로를 지정하거나 torch(CUDA 빌드)가 깔린 venv 를 만드세요." >&2 + exit 1 +fi + +# CTranslate2 가 libcublas.so.12 를 직접 찾는다. torch 가 딸려온 nvidia +# 패키지 안에 있으므로 경로를 잡아준다. 없으면 +# "Library libcublas.so.12 is not found" 로 GPU 번역만 조용히 실패한다. +NV="$("$VENV/bin/python" -c 'import nvidia,os;print(os.path.dirname(nvidia.__file__))' 2>/dev/null || true)" +if [[ -n "$NV" ]]; then + for lib in cublas cudnn; do + [[ -d "$NV/$lib/lib" ]] && export LD_LIBRARY_PATH="$NV/$lib/lib:${LD_LIBRARY_PATH:-}" + done +fi + +cd "$ROOT" +export PYTHONPATH="src" +exec "$VENV/bin/python" packaging/windows_smoke.py --gpu diff --git a/src/livesub/ui/pages/models_page.py b/src/livesub/ui/pages/models_page.py index c5439d4..5af9efd 100644 --- a/src/livesub/ui/pages/models_page.py +++ b/src/livesub/ui/pages/models_page.py @@ -18,7 +18,7 @@ from PySide6.QtWidgets import ( from ...config import AppConfig from ...models import Tier, detect_gpu, ordered_tiers -from ...models.manager import tier_availability +from ...models.manager import GpuInfo, tier_availability from ..theme import SPACING, palette from ..widgets import Card, PageHeader @@ -26,10 +26,11 @@ from ..widgets import Card, PageHeader class TierCard(QFrame): """티어 하나를 라디오 버튼 카드로.""" - def __init__(self, tier: Tier, gpu_vram_gb: float, parent: QWidget | None = None): + def __init__(self, tier: Tier, gpu: GpuInfo, parent: QWidget | None = None): super().__init__(parent) self.tier = tier - self.available, self.unavailable_reason = tier_availability(tier) + gpu_vram_gb = gpu.total_vram_mb / 1024 if gpu.available else 0.0 + self.available, self.unavailable_reason = tier_availability(tier, gpu) self.setObjectName("Card") p = palette("dark") @@ -102,6 +103,9 @@ class ModelsPage(QWidget): super().__init__(parent) self.config = config self._cards: list[TierCard] = [] + # 티어 카드마다 조회하면 nvidia-smi 를 다섯 번 부른다. 한 번만 본다. + self._gpu = detect_gpu() + self._gpu_vram_gb = self._gpu.total_vram_mb / 1024 if self._gpu.available else 0.0 root = QVBoxLayout(self) root.setContentsMargins(0, 0, 0, 0) @@ -119,7 +123,7 @@ class ModelsPage(QWidget): self._group = QButtonGroup(self) self._group.setExclusive(True) for tier in ordered_tiers(): - card = TierCard(tier, self._gpu_vram_gb) + card = TierCard(tier, self._gpu) self._group.addButton(card.radio, tier.order) card.radio.toggled.connect( lambda checked, t=tier: self._on_tier_selected(t) if checked else None @@ -133,8 +137,7 @@ class ModelsPage(QWidget): # --- 구성 ----------------------------------------------------------- def _build_gpu_card(self) -> Card: - gpu = detect_gpu() - self._gpu_vram_gb = gpu.total_vram_mb / 1024 if gpu.available else 0.0 + gpu = self._gpu p = palette("dark") card = Card("실행 환경") diff --git a/tests/test_portable.py b/tests/test_portable.py index dd65a64..4bd48de 100644 --- a/tests/test_portable.py +++ b/tests/test_portable.py @@ -139,3 +139,54 @@ def test_app_window_builds_without_torch(tmp_path, monkeypatch): win.overlay.close() win.close() _ = app + + +# --- 티어 게이트가 실제 제약을 보는지 ------------------------------------- +# +# 예전에는 torch 유무만 봤다. 그래서 8GB 카드에서도 "극한(16GB 필요) 사용 +# 가능"이라고 답했고, 사용자가 고르면 로드에서 죽었다. 실제 GPU 로 검증하다 +# 발견한 버그다. + + +def _gpu(total_gb: float): + from livesub.models.manager import GpuInfo + + return GpuInfo(available=True, name="test", total_vram_mb=int(total_gb * 1024)) + + +@pytest.mark.skipif(not has_torch(), reason="torch 없으면 어차피 앞단에서 막힌다") +@pytest.mark.parametrize("key", ["precision", "ultimate"]) +def test_vram_이_모자라면_티어를_막는다(key, monkeypatch): + monkeypatch.setattr("livesub.models.manager.has_awq", lambda: True) + ok, reason = tier_availability(TIERS[key], _gpu(7.5)) + assert not ok + assert "VRAM" in reason and "7.5GB" in reason + + +@pytest.mark.skipif(not has_torch(), reason="torch 없으면 어차피 앞단에서 막힌다") +def test_vram_이_충분하면_통과한다(monkeypatch): + monkeypatch.setattr("livesub.models.manager.has_awq", lambda: True) + assert tier_availability(TIERS["ultimate"], _gpu(24))[0] + + +@pytest.mark.skipif(not has_torch(), reason="torch 없으면 어차피 앞단에서 막힌다") +def test_autoawq_가_없으면_int4_티어를_막는다(monkeypatch): + monkeypatch.setattr("livesub.models.manager.has_awq", lambda: False) + ok, reason = tier_availability(TIERS["precision"], _gpu(24)) + assert not ok + assert "autoawq" in reason + + +def test_ct2_티어는_gpu_조회_없이도_통과한다(): + """1~3티어는 torch 도 GPU 도 안 따진다 — 포터블에서 돌아야 하므로.""" + def _boom(): + raise AssertionError("CT2 티어는 detect_gpu 를 부르면 안 된다") + + import livesub.models.manager as mgr + + original, mgr.detect_gpu = mgr.detect_gpu, _boom + try: + for key in ("lightning", "swift", "balance"): + assert tier_availability(TIERS[key])[0] + finally: + mgr.detect_gpu = original