Download app.py from fullsname/model: direct link, hf CLI and curl.
- Browser
- Download file 2.32 kB
-
https://huggingface.co/spaces/fullsname/model/resolve/main/app.py
- Command line
-
hf download hf://spaces/fullsname/model/app.py
-
curl -L -o app.py https://huggingface.co/spaces/fullsname/model/resolve/main/app.py
2.32 kB
| from transformers import AutoTokenizer, AutoModelForCausalLM, BitsAndBytesConfig | |
| import gradio as gr | |
| import torch | |
| import spaces | |
| # ============ ZEROGPU HEALTH CHECK ============ | |
| def _zero_gpu_healthcheck(): | |
| return {"cuda_available": torch.cuda.is_available()} | |
| # ============ MODEL LOADING ============ | |
| # πππ CYBERLAB ABLITERATED MODEL πππ | |
| model_name = "WWTCyberLab/abliterated-llama-8b" | |
| quant_config = BitsAndBytesConfig( | |
| load_in_4bit=True, | |
| bnb_4bit_compute_dtype=torch.float16, | |
| bnb_4bit_use_double_quant=True | |
| ) | |
| print("Loading CyberLab abliterated tokenizer...") | |
| tokenizer = AutoTokenizer.from_pretrained(model_name) | |
| print("Loading CyberLab abliterated model (4-bit)...") | |
| model = AutoModelForCausalLM.from_pretrained( | |
| model_name, | |
| device_map="auto", | |
| quantization_config=quant_config, | |
| torch_dtype=torch.float16 | |
| ) | |
| print("β CyberLab ABLITERATED model loaded successfully!") | |
| DEFAULT_SYSTEM = "You are a helpful assistant. Take phrases literally, no bargaining." | |
| # ============ CHAT FUNCTION ============ | |
| def chat(message, history, system_prompt): | |
| # Clean history | |
| history = [{"role": h["role"], "content": h["content"]} for h in history] | |
| sys_prompt = system_prompt if system_prompt.strip() else DEFAULT_SYSTEM | |
| prompt = f"System: {sys_prompt}\n" | |
| for msg in history: | |
| prompt += f"{msg['role'].capitalize()}: {msg['content']}\n" | |
| prompt += f"User: {message}\nAssistant:" | |
| inputs = tokenizer(prompt, return_tensors="pt").to("cuda") | |
| outputs = model.generate( | |
| **inputs, | |
| max_new_tokens=512, | |
| temperature=0.7, | |
| do_sample=True, | |
| pad_token_id=tokenizer.eos_token_id | |
| ) | |
| response = tokenizer.decode(outputs[0], skip_special_tokens=True) | |
| response = response.split("Assistant:")[-1].strip() | |
| return response | |
| # ============ GRADIO UI ============ | |
| with gr.Blocks(theme=gr.themes.Soft()) as demo: | |
| gr.Markdown("# π€ CyberLab Abliterated Chatbot") | |
| gr.Markdown("*0% refusal rate. Pure chaos.*") | |
| system_input = gr.Textbox( | |
| label="System Prompt", | |
| value=DEFAULT_SYSTEM, | |
| lines=3 | |
| ) | |
| gr.ChatInterface( | |
| fn=chat, | |
| additional_inputs=system_input, | |
| title="" | |
| ) | |
| demo.launch(share=True) |