Goshawk_Vi / app.py
GoshawkVortexAI's picture
Create app.py
0cc80cc verified
Raw History Blame Contribute Delete
6.05 kB
import os
import torch
import gradio as gr
from transformers import (
AutoConfig,
AutoTokenizer,
AutoModelForCausalLM
)
# ==================================================
# GOSHAWK AI — Hugging Face Space
# ==================================================
APP_NAME = "Goshawk AI"
MODEL_PATH = os.getenv("MODEL_PATH", "./")
MAX_NEW_TOKENS = 256
MAX_CONTEXT = 2048
device = "cuda" if torch.cuda.is_available() else "cpu"
dtype = torch.float16 if device == "cuda" else torch.float32
print(f"[{APP_NAME}] Device: {device}")
print(f"[{APP_NAME}] Model path: {MODEL_PATH}")
tokenizer = None
model = None
load_error = None
try:
config = AutoConfig.from_pretrained(
MODEL_PATH,
local_files_only=True,
trust_remote_code=False
)
print(f"Model architecture: {config.model_type}")
tokenizer = AutoTokenizer.from_pretrained(
MODEL_PATH,
local_files_only=True,
trust_remote_code=False
)
model = AutoModelForCausalLM.from_pretrained(
MODEL_PATH,
config=config,
torch_dtype=dtype,
low_cpu_mem_usage=True,
local_files_only=True,
trust_remote_code=False
)
model.to(device)
model.eval()
if tokenizer.pad_token_id is None:
tokenizer.pad_token = tokenizer.eos_token
print(f"[{APP_NAME}] Model loaded successfully.")
except Exception as exc:
load_error = f"{type(exc).__name__}: {exc}"
print(f"[{APP_NAME}] Loading failed: {load_error}")
SYSTEM_PROMPT = (
"You are Goshawk AI, a helpful and precise AI assistant. "
"Answer in the user's language. Be transparent about uncertainty. "
"Never invent facts, live market data, or sources."
)
def build_prompt(message, history):
messages = [
{"role": "system", "content": SYSTEM_PROMPT}
]
for item in (history or [])[-8:]:
if isinstance(item, dict):
role = item.get("role")
content = item.get("content", "")
if role in ("user", "assistant") and isinstance(content, str):
messages.append({
"role": role,
"content": content
})
elif isinstance(item, (list, tuple)) and len(item) == 2:
if item[0]:
messages.append({
"role": "user",
"content": str(item[0])
})
if item[1]:
messages.append({
"role": "assistant",
"content": str(item[1])
})
messages.append({"role": "user", "content": message})
if hasattr(tokenizer, "apply_chat_template"):
try:
return tokenizer.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True
)
except Exception:
pass
# Fallback for models without a chat template.
prompt = f"System: {SYSTEM_PROMPT}\n"
for msg in messages[1:]:
label = "User" if msg["role"] == "user" else "Assistant"
prompt += f"{label}: {msg['content']}\n"
return prompt + "Assistant:"
def respond(message, history, temperature, max_tokens):
if not message or not message.strip():
yield "Lütfen bir mesaj yaz."
return
if model is None or tokenizer is None:
yield (
"Model yüklenemedi.\n\n"
f"Hata: {load_error}\n\n"
"config.json, model.safetensors ve tokenizer "
"dosyalarını kontrol et. Model mimarisi metin "
"üretimini desteklemiyor olabilir."
)
return
try:
prompt = build_prompt(message.strip(), history)
inputs = tokenizer(
prompt,
return_tensors="pt",
truncation=True,
max_length=MAX_CONTEXT
)
inputs = {k: v.to(device) for k, v in inputs.items()}
input_length = inputs["input_ids"].shape[1]
if input_length >= MAX_CONTEXT:
yield "Girdi bağlam sınırına ulaştı. Daha kısa bir mesaj dene."
return
with torch.inference_mode():
output = model.generate(
**inputs,
max_new_tokens=int(max_tokens),
do_sample=float(temperature) > 0,
temperature=max(float(temperature), 0.01),
top_p=0.9,
repetition_penalty=1.08,
pad_token_id=tokenizer.pad_token_id,
eos_token_id=tokenizer.eos_token_id
)
new_tokens = output[0][input_length:]
answer = tokenizer.decode(
new_tokens,
skip_special_tokens=True
).strip()
yield answer or "Model boş yanıt üretti."
except Exception as exc:
yield f"Üretim hatası: {type(exc).__name__}: {exc}"
with gr.Blocks(title=APP_NAME) as demo:
gr.Markdown(
"# 🦅 Goshawk AI\n"
"### Yerel model tabanlı yapay zekâ asistanı\n"
f"**Cihaz:** `{device}`"
)
if load_error:
gr.Markdown(
"⚠️ Model yüklenemedi. Ayrıntılar sohbet alanında görünür."
)
chatbot = gr.ChatInterface(
fn=respond,
chatbot=gr.Chatbot(height=480),
textbox=gr.Textbox(
placeholder="Goshawk AI'ye bir soru sor...",
lines=2
),
additional_inputs=[
gr.Slider(
minimum=0.1,
maximum=1.2,
value=0.7,
step=0.1,
label="Yaratıcılık"
),
gr.Slider(
minimum=32,
maximum=512,
value=MAX_NEW_TOKENS,
step=32,
label="Maksimum yeni token"
)
]
)
gr.Markdown(
"Not: Yanıt kalitesi ve hızı kullanılan modelin "
"mimarisine ve donanıma bağlıdır."
)
if __name__ == "__main__":
demo.queue().launch()