import os import torch import gradio as gr from transformers import ( AutoConfig, AutoTokenizer, AutoModelForCausalLM ) # ================================================== # GOSHAWK AI — Hugging Face Space # ================================================== APP_NAME = "Goshawk AI" MODEL_PATH = os.getenv("MODEL_PATH", "./") MAX_NEW_TOKENS = 256 MAX_CONTEXT = 2048 device = "cuda" if torch.cuda.is_available() else "cpu" dtype = torch.float16 if device == "cuda" else torch.float32 print(f"[{APP_NAME}] Device: {device}") print(f"[{APP_NAME}] Model path: {MODEL_PATH}") tokenizer = None model = None load_error = None try: config = AutoConfig.from_pretrained( MODEL_PATH, local_files_only=True, trust_remote_code=False ) print(f"Model architecture: {config.model_type}") tokenizer = AutoTokenizer.from_pretrained( MODEL_PATH, local_files_only=True, trust_remote_code=False ) model = AutoModelForCausalLM.from_pretrained( MODEL_PATH, config=config, torch_dtype=dtype, low_cpu_mem_usage=True, local_files_only=True, trust_remote_code=False ) model.to(device) model.eval() if tokenizer.pad_token_id is None: tokenizer.pad_token = tokenizer.eos_token print(f"[{APP_NAME}] Model loaded successfully.") except Exception as exc: load_error = f"{type(exc).__name__}: {exc}" print(f"[{APP_NAME}] Loading failed: {load_error}") SYSTEM_PROMPT = ( "You are Goshawk AI, a helpful and precise AI assistant. " "Answer in the user's language. Be transparent about uncertainty. " "Never invent facts, live market data, or sources." ) def build_prompt(message, history): messages = [ {"role": "system", "content": SYSTEM_PROMPT} ] for item in (history or [])[-8:]: if isinstance(item, dict): role = item.get("role") content = item.get("content", "") if role in ("user", "assistant") and isinstance(content, str): messages.append({ "role": role, "content": content }) elif isinstance(item, (list, tuple)) and len(item) == 2: if item[0]: messages.append({ "role": "user", "content": str(item[0]) }) if item[1]: messages.append({ "role": "assistant", "content": str(item[1]) }) messages.append({"role": "user", "content": message}) if hasattr(tokenizer, "apply_chat_template"): try: return tokenizer.apply_chat_template( messages, tokenize=False, add_generation_prompt=True ) except Exception: pass # Fallback for models without a chat template. prompt = f"System: {SYSTEM_PROMPT}\n" for msg in messages[1:]: label = "User" if msg["role"] == "user" else "Assistant" prompt += f"{label}: {msg['content']}\n" return prompt + "Assistant:" def respond(message, history, temperature, max_tokens): if not message or not message.strip(): yield "Lütfen bir mesaj yaz." return if model is None or tokenizer is None: yield ( "Model yüklenemedi.\n\n" f"Hata: {load_error}\n\n" "config.json, model.safetensors ve tokenizer " "dosyalarını kontrol et. Model mimarisi metin " "üretimini desteklemiyor olabilir." ) return try: prompt = build_prompt(message.strip(), history) inputs = tokenizer( prompt, return_tensors="pt", truncation=True, max_length=MAX_CONTEXT ) inputs = {k: v.to(device) for k, v in inputs.items()} input_length = inputs["input_ids"].shape[1] if input_length >= MAX_CONTEXT: yield "Girdi bağlam sınırına ulaştı. Daha kısa bir mesaj dene." return with torch.inference_mode(): output = model.generate( **inputs, max_new_tokens=int(max_tokens), do_sample=float(temperature) > 0, temperature=max(float(temperature), 0.01), top_p=0.9, repetition_penalty=1.08, pad_token_id=tokenizer.pad_token_id, eos_token_id=tokenizer.eos_token_id ) new_tokens = output[0][input_length:] answer = tokenizer.decode( new_tokens, skip_special_tokens=True ).strip() yield answer or "Model boş yanıt üretti." except Exception as exc: yield f"Üretim hatası: {type(exc).__name__}: {exc}" with gr.Blocks(title=APP_NAME) as demo: gr.Markdown( "# 🦅 Goshawk AI\n" "### Yerel model tabanlı yapay zekâ asistanı\n" f"**Cihaz:** `{device}`" ) if load_error: gr.Markdown( "⚠️ Model yüklenemedi. Ayrıntılar sohbet alanında görünür." ) chatbot = gr.ChatInterface( fn=respond, chatbot=gr.Chatbot(height=480), textbox=gr.Textbox( placeholder="Goshawk AI'ye bir soru sor...", lines=2 ), additional_inputs=[ gr.Slider( minimum=0.1, maximum=1.2, value=0.7, step=0.1, label="Yaratıcılık" ), gr.Slider( minimum=32, maximum=512, value=MAX_NEW_TOKENS, step=32, label="Maksimum yeni token" ) ] ) gr.Markdown( "Not: Yanıt kalitesi ve hızı kullanılan modelin " "mimarisine ve donanıma bağlıdır." ) if __name__ == "__main__": demo.queue().launch()