ci: GPU 검증을 별도 러너 라벨로 분리 (windows-gpu)
Some checks failed
Windows / verify (push) Has been cancelled

GPU 는 한 VM 만 독점한다 — 컨슈머 NVIDIA 는 vGPU/SR-IOV 를 지원하지 않아
쪼개 쓸 방법이 없고, 지금 RTX 5050 은 .9(리눅스)에 물려 있다. 따라서 GPU
없는 상시 CI VM 에서는 CUDA·VRAM·4~5티어를 영영 확인할 수 없다.

코드로 우회할 수 없는 제약이므로, 대신 **나중에 GPU 머신이 생겼을 때 그대로
붙도록** 구조를 지금 잡아뒀다. 이 러너를 다른 프로젝트 GPU 테스트에도 쓸
계획이라면 이 구분이 그쪽에도 그대로 적용된다.

  windows:host      GPU 없는 서버 VM   main push 마다 자동
                    테스트·Windows 경로·오디오→자막·포터블 exe
  windows-gpu:host  GPU 달린 머신      수동 실행만
                    CUDA 인식·VRAM 상한·4·5티어·GPU 번역

- windows_smoke.py --gpu
  GPU 있는 머신에서만 의미 있는 항목만 모았다. CUDA 가용성, 앱의 GPU 인식,
  VRAM 상한이 실제로 걸리는지, torch 필요 티어(4·5)가 열리는지, GPU 로 실제
  번역까지. 다른 모드와 달리 **GPU 부재를 치명적 실패로 처리**한다 —
  GPU 를 확인하겠다고 부른 작업이 조용히 통과하면 안 되기 때문이다.

- .gitea/workflows/windows-gpu.yml
  수동 실행 전용. 러너가 없으면 큐에 쌓이기만 하고 아무 일도 없다.
  포터블 CI 와 달리 torch CUDA 빌드를 설치하고, 설치 직후 CPU 빌드가
  깔리지 않았는지 확인한다. nvidia-smi 가 없으면 그 자리에서 멈춘다.

- 문서에 두 러너 구조와 GPU 머신 확보 3가지 방안(서버에 GPU 추가 /
  .9 에서 옮기기 / 필요할 때만 실물 PC) + 슬롯 확인 명령 추가.

검증: pytest 189개 통과, ruff clean, 두 워크플로 파싱·라벨·트리거 확인,
      --gpu 를 torch 없는 환경에서 실행해 의도대로 실패(종료코드 1)하는지 확인

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
EJClaw
2026-09-23 03:40:01 +09:00
parent 8d91087f85
commit bb385c1d7b
3 changed files with 199 additions and 6 deletions

View File

@@ -230,6 +230,83 @@ def run_pipeline_check(timeout_s: int = 900) -> None:
return f"{line.source_text!r} -> {line.translated_text!r}"
# --- GPU 전용 점검 ----------------------------------------------------------
def run_gpu_checks() -> None:
"""GPU 가 있는 머신에서만 의미가 있는 것들.
GPU 는 한 VM 만 독점하므로(컨슈머 NVIDIA 는 vGPU 미지원) GPU 없는 CI VM
에서는 확인할 수 없다. 그래서 별도 러너 라벨(windows-gpu)로 떼어놨다.
여기서는 GPU 가 없으면 **분명히 실패**해야 한다 — GPU 를 확인하겠다고
부른 작업이 조용히 통과하면 안 된다.
"""
print(f"\n=== GPU 전용 (플랫폼: {sys.platform}) ===")
@check("CUDA 사용 가능")
def _cuda():
import torch
if not torch.cuda.is_available():
raise RuntimeError(
"torch 가 CUDA 를 못 씁니다. CPU 빌드가 깔렸는지 확인하세요 "
"(pip install torch --index-url https://download.pytorch.org/whl/cu128)"
)
idx = torch.cuda.current_device()
free, total = torch.cuda.mem_get_info(idx)
return f"{torch.cuda.get_device_name(idx)} · {total / 1024**3:.1f}GB (여유 {free / 1024**3:.1f}GB)"
@check("앱의 GPU 인식")
def _detect():
from livesub.models import detect_gpu
gpu = detect_gpu()
if not gpu.available:
raise RuntimeError(gpu.reason)
return f"{gpu.name} · {gpu.total_vram_mb / 1024:.1f}GB"
@check("VRAM 상한이 실제로 걸리는지")
def _vram():
from livesub.models.manager import apply_vram_limit
if not apply_vram_limit(0.35):
raise RuntimeError("상한 적용 실패 — 게임과 같이 쓸 때 VRAM 을 못 막습니다")
return "35% 상한 적용됨"
@check("torch 필요 티어(4·5)가 열리는지")
def _tiers():
from livesub.models.manager import has_torch, tier_availability
from livesub.models.tiers import TIERS, MTBackend
if not has_torch():
raise RuntimeError("torch 없음 — GPU 러너에는 일반 설치가 필요합니다")
blocked = [
t.name for t in TIERS.values()
if t.mt.backend is MTBackend.TRANSFORMERS and not tier_availability(t)[0]
]
if blocked:
raise RuntimeError(f"막힌 티어: {', '.join(blocked)}")
return "정밀·극한 사용 가능"
@check("GPU 로 실제 번역")
def _translate():
from livesub.models.glossary import Glossary
from livesub.models.packs import build_glossary
from livesub.models.tiers import get_tier
from livesub.models.translator import create_translator
tier = get_tier("swift") # CT2 지만 device=cuda 로 GPU 경로를 탄다
translator = create_translator(tier.mt, device="cuda")
translator.load()
glossary = build_glossary(Glossary(), ["fps-common", "pubg"])
out = translator.translate("Enemy coming from the left.", "en", "ko", glossary)
if not any("\uac00" <= ch <= "\ud7a3" for ch in out):
raise RuntimeError(f"한국어가 아닙니다: {out!r}")
ARTIFACTS.mkdir(exist_ok=True)
(ARTIFACTS / "gpu-translate.txt").write_text(out + "\n", encoding="utf-8")
return repr(out)
# --- 빌드된 exe 점검 --------------------------------------------------------
@@ -269,10 +346,16 @@ def main() -> int:
"--pipeline", action="store_true",
help="오디오 파일로 전 파이프라인 점검 (모델을 내려받으므로 느립니다)",
)
parser.add_argument(
"--gpu", action="store_true",
help="GPU 가 있는 머신에서만 되는 항목 점검 (CUDA·VRAM 상한·4/5티어)",
)
args = parser.parse_args()
if args.exe:
run_exe_check(args.exe)
elif args.gpu:
run_gpu_checks()
elif args.pipeline:
run_pipeline_check()
else:
@@ -288,13 +371,15 @@ def main() -> int:
print(f"\n{len(_results) - len(failed)}/{len(_results)} 통과")
if failed:
print("실패:", ", ".join(failed))
# 오디오 장치·보조 프로그램은 러너 환경에 따라 없을 수 있어 경고로만 둔다.
fatal = [n for n in failed if n not in {
# 오디오 장치·보조 프로그램·GPU 는 러너 환경에 따라 없을 수 있어 경고로만 둔다.
# 다만 --gpu 로 부른 경우엔 GPU 가 없는 것 자체가 실패다.
lenient = {
"프로그램별 캡처 보조 프로그램",
"소리 내는 프로그램 열거 (pycaw)",
"WASAPI 루프백 장치 열거",
"GPU 인식",
}]
}
fatal = failed if args.gpu else [n for n in failed if n not in lenient]
return 1 if fatal else 0