Spaces:
Runtime error
Runtime error
Download app.py from changastyle/test: direct link, hf CLI and curl.
- Browser
- Download file 1.37 kB
-
https://huggingface.co/spaces/changastyle/test/resolve/main/app.py
- Command line
-
hf download hf://spaces/changastyle/test/app.py
-
curl -L -o app.py https://huggingface.co/spaces/changastyle/test/resolve/main/app.py
1.37 kB
| import gradio as gr | |
| import spaces | |
| from llama_cpp import Llama | |
| # Descarga y carga el modelo automáticamente desde el repositorio optimizado de la comunidad | |
| # Usamos la versión Q4_K_M (4 bits) que ofrece el mejor equilibrio entre velocidad y calidad para 32B | |
| print("Descargando y cargando Kevin-32B...") | |
| llm = Llama.from_pretrained( | |
| repo_id="lmstudio-community/Kevin-32B-GGUF", | |
| filename="Kevin-32B-GGUF.Q4_K_M.gguf", | |
| n_ctx=40960, # Soporta el contexto largo original del modelo | |
| n_gpu_layers=-1 # Envía todas las capas a la GPU asignada por Hugging Face | |
| ) | |
| # Le pide una GPU potente a Hugging Face para procesar el mensaje | |
| def responder_chat(message, history): | |
| # Formateamos el historial para el modelo | |
| prompt = "" | |
| for user_msg, ai_msg in history: | |
| prompt += f"<|im_start|>user\n{user_msg}<|im_end|>\n<|im_start|>assistant\n{ai_msg}<|im_end|>\n" | |
| prompt += f"<|im_start|>user\n{message}<|im_end|>\n<|im_start|>assistant\n" | |
| # Generamos la respuesta del modelo | |
| output = llm( | |
| prompt, | |
| max_tokens=1024, | |
| stop=["<|im_end|>"], | |
| echo=False | |
| ) | |
| return output["choices"][0]["text"].strip() | |
| # Interfaz gráfica de Chat ejecutable | |
| demo = gr.ChatInterface(fn=responder_chat, title="Mi Servidor Privado de Kevin-32B") | |
| if __name__ == "__main__": | |
| demo.launch() | |