File size: 2,316 Bytes
db3390f db78e6a db3390f e45771a 4b5d261 e45771a 08e6950 4b5d261 db3390f 4b5d261 08e6950 db3390f 4b5d261 08e6950 db3390f 4b5d261 08e6950 e45771a 08e6950 4b5d261 e45771a db78e6a 08e6950 d1263c8 08e6950 db78e6a db3390f db78e6a 08e6950 db78e6a 08e6950 db78e6a e45771a db78e6a 08e6950 db78e6a e45771a db78e6a 4b5d261 db78e6a | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 | from transformers import AutoTokenizer, AutoModelForCausalLM, BitsAndBytesConfig
import gradio as gr
import torch
import spaces
# ============ ZEROGPU HEALTH CHECK ============
@spaces.GPU
def _zero_gpu_healthcheck():
return {"cuda_available": torch.cuda.is_available()}
# ============ MODEL LOADING ============
# πππ CYBERLAB ABLITERATED MODEL πππ
model_name = "WWTCyberLab/abliterated-llama-8b"
quant_config = BitsAndBytesConfig(
load_in_4bit=True,
bnb_4bit_compute_dtype=torch.float16,
bnb_4bit_use_double_quant=True
)
print("Loading CyberLab abliterated tokenizer...")
tokenizer = AutoTokenizer.from_pretrained(model_name)
print("Loading CyberLab abliterated model (4-bit)...")
model = AutoModelForCausalLM.from_pretrained(
model_name,
device_map="auto",
quantization_config=quant_config,
torch_dtype=torch.float16
)
print("β
CyberLab ABLITERATED model loaded successfully!")
DEFAULT_SYSTEM = "You are a helpful assistant. Take phrases literally, no bargaining."
# ============ CHAT FUNCTION ============
@spaces.GPU
def chat(message, history, system_prompt):
# Clean history
history = [{"role": h["role"], "content": h["content"]} for h in history]
sys_prompt = system_prompt if system_prompt.strip() else DEFAULT_SYSTEM
prompt = f"System: {sys_prompt}\n"
for msg in history:
prompt += f"{msg['role'].capitalize()}: {msg['content']}\n"
prompt += f"User: {message}\nAssistant:"
inputs = tokenizer(prompt, return_tensors="pt").to("cuda")
outputs = model.generate(
**inputs,
max_new_tokens=512,
temperature=0.7,
do_sample=True,
pad_token_id=tokenizer.eos_token_id
)
response = tokenizer.decode(outputs[0], skip_special_tokens=True)
response = response.split("Assistant:")[-1].strip()
return response
# ============ GRADIO UI ============
with gr.Blocks(theme=gr.themes.Soft()) as demo:
gr.Markdown("# π€ CyberLab Abliterated Chatbot")
gr.Markdown("*0% refusal rate. Pure chaos.*")
system_input = gr.Textbox(
label="System Prompt",
value=DEFAULT_SYSTEM,
lines=3
)
gr.ChatInterface(
fn=chat,
additional_inputs=system_input,
title=""
)
demo.launch(share=True) |