테스트

2026-01-28 20:58:57 +09:00
parent 95c5fb7867
commit 5f828601ce
16 changed files with 115 additions and 82 deletions
--- a/config/ai/call_llm_model.py
+++ b/config/ai/call_llm_model.py
@@ -0,0 +1,53 @@
+from transformers import AutoTokenizer, AutoModelForCausalLM
+
+QWEN_MODEL_PATH = "./models/Qwen3-0.6B"
+
+
+_model = None
+_tokenizer = None
+
+def get_qwen_model():
+    """
+    Qwen 모델과 토크나이저를 로드하거나 캐시된 인스턴스를 반환합니다.
+    torch.compile을 사용하여 추론 속도를 최적화합니다.
+
+    Returns:
+        tuple: (model, tokenizer)
+    """
+    global _model, _tokenizer
+    if _model is not None:
+        return _model, _tokenizer
+
+    # 토크나이저 로드
+    _tokenizer = AutoTokenizer.from_pretrained(
+        QWEN_MODEL_PATH,
+        trust_remote_code=True,
+        local_files_only=True
+    )
+
+    # 모델 로드
+    from sympy.printing.pytorch import torch
+    _model = AutoModelForCausalLM.from_pretrained(
+        QWEN_MODEL_PATH,
+        dtype=torch.bfloat16,  # CPU: bfloat16, GPU: float16 권장
+        device_map="auto",
+        trust_remote_code=True,
+        local_files_only=True
+    )
+
+    # ✅ torch.compile() 적용 (PyTorch 2.0+)
+    if hasattr(torch, 'compile'):
+        try:
+            print("🚀 torch.compile() 적용 중...")
+            # mode="reduce-overhead": 추론 시 추천
+            # dynamic=True: 입력 길이가 유동적인 RAG 환경에 적합
+            _model = torch.compile(
+                _model,
+                mode="reduce-overhead",
+                dynamic=True
+            )
+            print("✅ torch.compile() 성공!")
+        except Exception as e:
+            print(f"⚠️ torch.compile() 실패, 원본 모델 사용: {e}")
+            pass  # 실패하면 원본 사용
+    return _model, _tokenizer