Any-to-Any
Transformers
Safetensors
English
qwen2
text-generation
qwen
lora
router
multi-agent
orchestration
gradio
multimodal
text
image
video
audio
Eval Results (legacy)
text-generation-inference
Instructions to use Questionmarkboy/frankenstein-3-0 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use Questionmarkboy/frankenstein-3-0 with Transformers:
# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("Questionmarkboy/frankenstein-3-0") model = AutoModelForCausalLM.from_pretrained("Questionmarkboy/frankenstein-3-0", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download app.py from Questionmarkboy/frankenstein-3-0: direct link, hf CLI and curl.
- Browser
- Download file 20.1 kB
-
https://huggingface.co/Questionmarkboy/frankenstein-3-0/resolve/main/app.py
- Command line
-
hf download hf://Questionmarkboy/frankenstein-3-0/app.py
-
curl -L -o app.py https://huggingface.co/Questionmarkboy/frankenstein-3-0/resolve/main/app.py
20.1 kB
| import gradio as gr | |
| import torch | |
| import gc | |
| import json | |
| import re | |
| import time | |
| import os | |
| from transformers import AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig | |
| import requests | |
| from bs4 import BeautifulSoup | |
| import sympy | |
| import sqlite3 | |
| HAS_CUDA = torch.cuda.is_available() | |
| # ============ PHASE 11: MEMORY SYSTEM (SQLite) ============ | |
| class MemorySystem: | |
| def __init__(self, db_path="frankenstein_memory.db"): | |
| self.conn = sqlite3.connect(db_path, check_same_thread=False) | |
| c = self.conn.cursor() | |
| c.execute(""" | |
| CREATE TABLE IF NOT EXISTS conversations ( | |
| id INTEGER PRIMARY KEY AUTOINCREMENT, | |
| user_input TEXT, specialist TEXT, response TEXT, timestamp REAL | |
| ) | |
| """) | |
| self.conn.commit() | |
| def save_interaction(self, user_input, specialist, response): | |
| c = self.conn.cursor() | |
| c.execute("INSERT INTO conversations (user_input, specialist, response, timestamp) VALUES (?, ?, ?, ?)", | |
| (user_input, specialist, response, time.time())) | |
| self.conn.commit() | |
| def get_recent_context(self, limit=3): | |
| c = self.conn.cursor() | |
| c.execute("SELECT user_input, specialist, response FROM conversations ORDER BY timestamp DESC LIMIT ?", (limit,)) | |
| return c.fetchall()[::-1] | |
| def clear_memory(self): | |
| c = self.conn.cursor() | |
| c.execute("DELETE FROM conversations") | |
| self.conn.commit() | |
| # ============ PHASE 9: INTERNET ACCESS ============ | |
| class InternetSearch: | |
| def search_web(query, max_results=3): | |
| try: | |
| headers = {'User-Agent': 'Mozilla/5.0'} | |
| r = requests.get(f"https://duckduckgo.com/html/?q={query}", headers=headers, timeout=10) | |
| soup = BeautifulSoup(r.text, 'html.parser') | |
| return [{"title": a.get_text(), "url": a.get('href')} | |
| for a in soup.find_all('a', class_='result__a')[:max_results]] | |
| except Exception as e: | |
| return [{"error": str(e)}] | |
| # ============ PHASE 10: TOOL SUITE ============ | |
| class ToolSuite: | |
| def calculate(expression): | |
| try: | |
| result = sympy.sympify(expression) | |
| return float(result) if result.is_number else str(result) | |
| except Exception as e: | |
| return f"Error: {str(e)}" | |
| def execute_python(code, timeout=5): | |
| import subprocess, tempfile | |
| try: | |
| with tempfile.NamedTemporaryFile(mode='w', suffix='.py', delete=False) as f: | |
| f.write(code); temp_path = f.name | |
| result = subprocess.run(['python', temp_path], capture_output=True, text=True, timeout=timeout) | |
| os.unlink(temp_path) | |
| return {"output": result.stdout or "", "error": result.stderr or "", "returncode": result.returncode} | |
| except Exception as e: | |
| return {"error": str(e)} | |
| # ============ PHASE 8: RAG ============ | |
| class RAGSystem: | |
| def __init__(self): | |
| self.documents = [] | |
| def add_document(self, file): | |
| try: | |
| if file.name.endswith('.txt'): | |
| with open(file.name, 'r') as f: content = f.read() | |
| self.documents.append({"filename": os.path.basename(file.name), "content": content}) | |
| return f"Added {os.path.basename(file.name)} to knowledge base" | |
| return "Only .txt files supported" | |
| except Exception as e: | |
| return f"Error: {str(e)}" | |
| def search(self, query, max_results=2): | |
| if not self.documents: return "No documents uploaded yet" | |
| results = []; q = query.lower() | |
| for doc in self.documents: | |
| if q in doc["content"].lower(): | |
| idx = doc["content"].lower().find(q) | |
| snippet = doc["content"][max(0, idx-100):min(len(doc["content"]), idx+300)] | |
| results.append(f"**{doc['filename']}**: ...{snippet}...") | |
| return "\n\n".join(results[:max_results]) if results else "No relevant information found" | |
| # ============ MODEL MANAGER (Dynamic Load/Unload) ============ | |
| class ModelManager: | |
| def __init__(self): | |
| self.resident_model = None; self.resident_tokenizer = None | |
| self.image_model = None; self.video_model = None | |
| self.coder_model = None; self.coder_tokenizer = None | |
| def load_resident(self): | |
| if self.resident_model is None: | |
| print("Loading resident model...", flush=True) | |
| self.resident_tokenizer = AutoTokenizer.from_pretrained("Qwen/Qwen3.5-4B", trust_remote_code=True) | |
| if HAS_CUDA: | |
| quant = BitsAndBytesConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4", | |
| bnb_4bit_compute_dtype=torch.float16, bnb_4bit_use_double_quant=True) | |
| self.resident_model = AutoModelForCausalLM.from_pretrained( | |
| "Qwen/Qwen3.5-4B", quantization_config=quant, device_map="auto", | |
| torch_dtype=torch.float16, trust_remote_code=True) | |
| else: | |
| self.resident_model = AutoModelForCausalLM.from_pretrained( | |
| "Qwen/Qwen3.5-4B", torch_dtype=torch.bfloat16, device_map="cpu", trust_remote_code=True) | |
| print("Resident loaded", flush=True) | |
| return self.resident_model, self.resident_tokenizer | |
| def unload_specialist(self, name): | |
| if name == "image" and self.image_model is not None: | |
| del self.image_model; self.image_model = None | |
| elif name == "video" and self.video_model is not None: | |
| del self.video_model; self.video_model = None | |
| elif name == "coder" and self.coder_model is not None: | |
| del self.coder_model; del self.coder_tokenizer | |
| self.coder_model = None; self.coder_tokenizer = None | |
| gc.collect() | |
| if HAS_CUDA: torch.cuda.empty_cache() | |
| def load_image_model(self): | |
| if not HAS_CUDA: raise RuntimeError("GPU required for image generation") | |
| if self.image_model is None: | |
| from diffusers import StableDiffusionXLPipeline | |
| self.image_model = StableDiffusionXLPipeline.from_pretrained( | |
| "stabilityai/stable-diffusion-xl-base-1.0", torch_dtype=torch.float16) | |
| self.image_model.enable_model_cpu_offload() | |
| return self.image_model | |
| def load_video_model(self): | |
| if not HAS_CUDA: raise RuntimeError("GPU required for video generation") | |
| if self.video_model is None: | |
| from diffsynth.pipelines.minimax_h3_audio_video import MiniMaxH3Pipeline, ModelConfig | |
| vram_config = { | |
| "offload_dtype": torch.bfloat16, "offload_device": "cpu", | |
| "onload_dtype": torch.bfloat16, "onload_device": "cpu", | |
| "preparing_dtype": torch.bfloat16, "preparing_device": "cuda", | |
| "computation_dtype": torch.bfloat16, "computation_device": "cuda", | |
| } | |
| self.video_model = MiniMaxH3Pipeline.from_pretrained( | |
| torch_dtype=torch.bfloat16, device="cuda", | |
| model_configs=[ | |
| ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-fl2va-nf4.safetensors", **vram_config), | |
| ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-text-encoder-nf4.safetensors", **vram_config), | |
| ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="video_vae_nf4.safetensors", **vram_config), | |
| ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="audio_vae_nf4.safetensors", **vram_config), | |
| ], | |
| processor_config=ModelConfig(model_id="MiniMax/MiniMax-H3", origin_file_pattern="FL2VA/processor/")) | |
| return self.video_model | |
| def load_coder(self): | |
| if not HAS_CUDA: return None, None | |
| if self.coder_model is None: | |
| self.coder_tokenizer = AutoTokenizer.from_pretrained("Qwen/Qwen2.5-Coder-14B-Instruct") | |
| quant = BitsAndBytesConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4", | |
| bnb_4bit_compute_dtype=torch.float16, bnb_4bit_use_double_quant=True) | |
| self.coder_model = AutoModelForCausalLM.from_pretrained( | |
| "Qwen/Qwen2.5-Coder-14B-Instruct", quantization_config=quant, | |
| device_map="auto", torch_dtype=torch.float16) | |
| return self.coder_model, self.coder_tokenizer | |
| # ============ INTELLIGENT ROUTER ============ | |
| class FrankensteinRouter: | |
| def __init__(self, model_manager, memory_system): | |
| self.mm = model_manager; self.memory = memory_system | |
| self.resident, self.tokenizer = model_manager.load_resident() | |
| self.device = next(self.resident.parameters()).device | |
| self.internet = InternetSearch(); self.tools = ToolSuite(); self.rag = RAGSystem() | |
| self.brain, self.brain_tok = self._load_brain() | |
| def _load_brain(self): | |
| try: | |
| bt = AutoTokenizer.from_pretrained("Questionmarkboy/frankenstein-3-0", subfolder="routerbrain") | |
| bm = AutoModelForCausalLM.from_pretrained("Questionmarkboy/frankenstein-3.0", subfolder="routerbrain", torch_dtype=torch.float16, device_map="auto" if HAS_CUDA else "cpu") | |
| print("RouterBrain loaded", flush=True) | |
| return bm, bt | |
| except Exception as e: | |
| print("RouterBrain unavailable, using resident:", e, flush=True) | |
| return None, None | |
| def _brain_generate(self, prompt): | |
| text = self.brain_tok.apply_chat_template([{"role": "user", "content": prompt}], tokenize=False, add_generation_prompt=True) | |
| inp = self.brain_tok(text, return_tensors="pt").to(next(self.brain.parameters()).device) | |
| with torch.no_grad(): | |
| out = self.brain.generate(**inp, max_new_tokens=120, pad_token_id=self.brain_tok.eos_token_id) | |
| return self.brain_tok.decode(out[0][inp['input_ids'].shape[1]:], skip_special_tokens=True).strip() | |
| def _generate(self, prompt, max_tokens=150, temperature=0.1): | |
| messages = [{"role": "user", "content": prompt}] | |
| try: | |
| text = self.tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True, enable_thinking=False) | |
| except TypeError: | |
| text = self.tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True) + " /no_think" | |
| inputs = self.tokenizer(text, return_tensors="pt").to(self.device) | |
| with torch.no_grad(): | |
| out = self.resident.generate(**inputs, max_new_tokens=max_tokens, temperature=temperature, | |
| top_p=0.9, do_sample=True, pad_token_id=self.tokenizer.eos_token_id) | |
| return self.tokenizer.decode(out[0][inputs['input_ids'].shape[1]:], skip_special_tokens=True).strip() | |
| def _parse_json(self, response): | |
| try: | |
| match = re.search(r'\{.*\}', response.replace('\n', ''), re.DOTALL) | |
| return json.loads(match.group()) if match else json.loads(response) | |
| except Exception: | |
| return None | |
| def route(self, user_input): | |
| # PRE-FLIGHT CHECKS (regex overrides - 100% accurate for hard patterns) | |
| u_lower = user_input.lower() | |
| if any(kw in u_lower for kw in ["uploaded", "my document", "the file", "my manual", "the pdf", "my notes", "from the doc", "according to my", "my handbook", "the reference"]): | |
| return {"steps": [{"specialist": "rag", "prompt": user_input}], "confidence": 1.0, "reasoning": "Pre-flight: Document reference"} | |
| if " then " in u_lower or " and then " in u_lower or " followed by " in u_lower or " after that " in u_lower: | |
| parts = re.split(r' (then|and then|followed by|after that|next|afterwards) ', user_input, maxsplit=1) | |
| if len(parts) >= 3: | |
| p1, p2 = parts[0], parts[2] | |
| spec1 = "image" if any(kw in p1.lower() for kw in ["draw", "paint", "create an image", "illustrate", "generate a picture"]) else "code" if any(kw in p1.lower() for kw in ["write", "code", "implement", "build", "create"]) else "search" | |
| spec2 = "video" if any(kw in p2.lower() for kw in ["animate", "video", "motion", "clip"]) else "chat" | |
| return {"steps": [{"specialist": spec1, "prompt": p1}, {"specialist": spec2, "prompt": p2}], "confidence": 1.0, "reasoning": f"Pre-flight: Multi-step ({spec1} -> {spec2})"} | |
| if u_lower.startswith("explain how to ") or u_lower.startswith("how do i ") or u_lower.startswith("teach me how to "): | |
| return {"steps": [{"specialist": "chat", "prompt": user_input}], "confidence": 1.0, "reasoning": "Pre-flight: Explanation request"} | |
| # CONTINUE WITH BRAIN ROUTING | |
| recent = self.memory.get_recent_context(3) | |
| context = "" | |
| if recent: | |
| context = "Recent interactions:\n" + "\n".join([f"User: {r[0][:50]}... -> {r[1]}" for r in recent]) + "\n\n" | |
| prompt = f"""You are the routing brain of Frankenstein-3.0. Route to the correct specialist. | |
| {context}SPECIALISTS: | |
| - "chat": Conversation, questions, explanations | |
| - "code": Programming, debugging, scripts, games | |
| - "image": Pictures, art, illustrations, logos, photos | |
| - "video": Animations, clips, motion, films | |
| - "search": Web search, current events, real-time info | |
| - "calc": Mathematical calculations | |
| - "execute": Run Python code | |
| - "rag": Answer questions from uploaded documents | |
| USER: "{user_input}" | |
| Respond ONLY with JSON: | |
| {{"specialist": "<name>", "confidence": <0.0-1.0>, "reasoning": "<brief>", "steps": [{{"specialist": "<name>", "prompt": "<refined prompt>"}}]}}""" | |
| raw = self._brain_generate(prompt) if self.brain is not None else self._generate(prompt, max_tokens=150) | |
| parsed = self._parse_json(raw) | |
| if parsed: | |
| return {"steps": parsed.get("steps", [{"specialist": parsed.get("specialist", "chat"), "prompt": user_input}]), | |
| "confidence": parsed.get("confidence", 0.8), "reasoning": parsed.get("reasoning", "LLM routed")} | |
| return {"steps": [{"specialist": "chat", "prompt": user_input}], "confidence": 0.5, "reasoning": "Fallback"} | |
| def execute_step(self, spec, prompt): | |
| result = ""; media_path = None | |
| try: | |
| if spec == "chat": | |
| result = self._generate(prompt, max_tokens=300, temperature=0.7) | |
| elif spec == "code": | |
| coder, coder_tok = self.mm.load_coder() | |
| if coder is not None: | |
| device = next(coder.parameters()).device | |
| messages = [{"role": "user", "content": f"Write clean Python code for: {prompt}"}] | |
| text = coder_tok.apply_chat_template(messages, tokenize=False, add_generation_prompt=True) | |
| inputs = coder_tok(text, return_tensors="pt").to(device) | |
| with torch.no_grad(): | |
| out = coder.generate(**inputs, max_new_tokens=500, temperature=0.3, pad_token_id=coder_tok.eos_token_id) | |
| code = coder_tok.decode(out[0][inputs['input_ids'].shape[1]:], skip_special_tokens=True) | |
| self.mm.unload_specialist("coder") | |
| result = f"```python\n{code}\n```" | |
| else: | |
| result = "```python\n" + self._generate(f"Write clean Python code for: {prompt}", max_tokens=400, temperature=0.3) + "\n```" | |
| elif spec == "image": | |
| pipe = self.mm.load_image_model() | |
| img = pipe(prompt, num_inference_steps=25, width=768, height=768).images[0] | |
| media_path = f"/tmp/img_{int(time.time())}.png"; img.save(media_path) | |
| self.mm.unload_specialist("image") | |
| result = "Image generated successfully!" | |
| elif spec == "video": | |
| pipe = self.mm.load_video_model() | |
| from diffsynth.utils.data.audio_video import write_video_audio | |
| video, audio = pipe(prompt, height=256, width=448, num_frames=25, num_inference_steps=20, | |
| use_gradient_checkpointing=True, use_gradient_checkpointing_offload=True) | |
| media_path = f"/tmp/vid_{int(time.time())}.mp4" | |
| write_video_audio(video, audio, media_path, fps=24, audio_sample_rate=32000) | |
| self.mm.unload_specialist("video") | |
| result = "Video generated successfully!" | |
| elif spec == "search": | |
| sr = self.internet.search_web(prompt) | |
| result = "**Web Search Results:**\n" + "\n".join([f"- {r['title']}: {r['url']}" for r in sr if 'title' in r]) | |
| elif spec == "calc": | |
| result = f"**Calculation:** {prompt} = {self.tools.calculate(prompt)}" | |
| elif spec == "execute": | |
| er = self.tools.execute_python(prompt) | |
| result = f"**Code Execution:**\nOutput: {er.get('output', 'None')}\nError: {er.get('error', 'None')}" | |
| elif spec == "rag": | |
| result = f"**RAG Search:**\n{self.rag.search(prompt)}" | |
| else: | |
| result = self._generate(prompt, max_tokens=300, temperature=0.7) | |
| except Exception as e: | |
| result = f"Specialist '{spec}' error: {str(e)}" | |
| self.memory.save_interaction(prompt, spec, result[:100]) | |
| return result, media_path | |
| # ============ MAIN ============ | |
| print("Initializing Frankenstein-3.0...", flush=True) | |
| mm = ModelManager(); memory = MemorySystem(); router = FrankensteinRouter(mm, memory) | |
| print("System ready!", flush=True) | |
| def frankenstein_chat(message, history): | |
| decision = router.route(message) | |
| route_path = " -> ".join([s["specialist"] for s in decision["steps"]]) | |
| output = f"**Router:** `{route_path.upper()}` (conf: {decision['confidence']:.2f})\n*{decision['reasoning']}*\n\n---\n\n" | |
| media_files = [] | |
| for step in decision["steps"]: | |
| result, media_path = router.execute_step(step["specialist"], step["prompt"]) | |
| output += result + "\n\n" | |
| if media_path: media_files.append(media_path) | |
| return output, (media_files if media_files else None) | |
| def upload_document(file): return router.rag.add_document(file) | |
| def clear_memory(): | |
| memory.clear_memory(); return "Memory cleared" | |
| with gr.Blocks(theme=gr.themes.Soft(), title="Frankenstein-3.0") as demo: | |
| gr.Markdown("# 🧟 Frankenstein-3.0: Unified AI Entity") | |
| gr.Markdown("**Specialists:** Chat - Code - Image - Video - Web Search - Calculator - Code Execution - RAG") | |
| with gr.Tabs(): | |
| with gr.Tab("Chat"): | |
| chatbot = gr.Chatbot(type="messages", height=500) | |
| msg = gr.Textbox(label="Command", placeholder="Try: Draw a cat, or Write a snake game") | |
| with gr.Row(): | |
| send_btn = gr.Button("Execute", variant="primary"); clr_btn = gr.Button("Clear") | |
| media_gallery = gr.Gallery(label="Generated Media", height=300) | |
| def respond(message, history): | |
| if not message.strip(): return "", history, None | |
| history = history + [{"role": "user", "content": message}] | |
| response, media = frankenstein_chat(message, history) | |
| history = history + [{"role": "assistant", "content": response}] | |
| return "", history, media | |
| msg.submit(respond, [msg, chatbot], [msg, chatbot, media_gallery]) | |
| send_btn.click(respond, [msg, chatbot], [msg, chatbot, media_gallery]) | |
| clr_btn.click(lambda: [], None, chatbot) | |
| with gr.Tab("Upload Documents (RAG)"): | |
| file_upload = gr.File(label="Upload .txt file"); upload_btn = gr.Button("Add to Knowledge Base") | |
| upload_status = gr.Textbox(label="Status") | |
| upload_btn.click(upload_document, [file_upload], upload_status) | |
| with gr.Tab("System"): | |
| mem_btn = gr.Button("Clear Memory"); mem_status = gr.Textbox(label="Status") | |
| mem_btn.click(clear_memory, outputs=mem_status) | |
| demo.launch(share=False, server_name="0.0.0.0", server_port=7860) | |