Spaces:
Paused
Paused
File size: 5,894 Bytes
bdf5f4a | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 | #!/usr/bin/env python3
"""
run.py — Start servers, run agent, non-truncated output.
Usage:
python3 run.py "your question here"
python3 run.py --plan "build a web server"
python3 run.py --server status
python3 run.py --server stop
"""
import sys, os, time, json, signal, subprocess, urllib.request
from pathlib import Path
ROOT = Path(__file__).parent
MODELS = {
"router": {"path": os.path.expanduser("~/models/Llama-3.2-1B-Instruct-Q4_K_M.gguf"), "port": 8080, "ctx": 2048},
"executor": {"path": os.path.expanduser("~/models/hermes-3-3b-Q4_K_M.gguf"), "port": 8081, "ctx": 4096},
"reflector": {"path": os.path.expanduser("~/models/Qwen2.5-1.5B-Instruct-Q4_K_M.gguf"), "port": 8082, "ctx": 4096},
}
def health(port):
try:
req = urllib.request.Request(f"http://localhost:{port}/health")
with urllib.request.urlopen(req, timeout=2) as r:
return json.loads(r.read()).get("status") == "ok"
except:
return False
def kill_port(port):
try:
r = subprocess.run(["lsof", "-ti", f":{port}"], capture_output=True, text=True)
for pid in r.stdout.strip().split():
try: os.kill(int(pid), signal.SIGTERM)
except: pass
if r.stdout.strip(): time.sleep(1)
except: pass
def start_servers():
started = []
for name, cfg in MODELS.items():
if health(cfg["port"]):
print(f" ● {name} already running on port {cfg['port']}")
started.append(name)
continue
kill_port(cfg["port"])
log = open(f"/tmp/llama-{name}.log", "w")
subprocess.Popen(
["llama-server", "-m", cfg["path"], "--port", str(cfg["port"]),
"--ctx-size", str(cfg["ctx"]), "--host", "0.0.0.0"],
stdout=log, stderr=log, start_new_session=True)
for _ in range(30):
time.sleep(1)
if health(cfg["port"]):
print(f" ● {name} started on port {cfg['port']}")
started.append(name)
break
else:
print(f" ✗ {name} failed to start on port {cfg['port']}")
return started
def chat(model, msg, max_tok=512, temp=0.7, grammar=None):
port = MODELS[model]["port"]
body = {"messages": [{"role": "user", "content": msg}],
"max_tokens": max_tok, "temperature": temp, "stream": False}
if grammar:
body["grammar"] = grammar
try:
req = urllib.request.Request(f"http://localhost:{port}/v1/chat/completions",
json.dumps(body).encode(), {"Content-Type": "application/json"})
with urllib.request.urlopen(req, timeout=120) as r:
return json.loads(r.read())["choices"][0]["message"]["content"].strip()
except Exception as e:
return f"[ERROR: {e}]"
def run_agent(prompt, force_plan=False):
# Classify
if force_plan:
cls = "PLAN"
else:
grammar = 'root ::= "SIMPLE" | "PLAN" | "CODE"'
cls = chat("router",
f"Classify as SIMPLE, PLAN, or CODE:\n\n{prompt}\n\nClassification:",
max_tok=5, temp=0.1, grammar=grammar)
cls = cls.strip().upper()
if "PLAN" in cls: cls = "PLAN"
elif "CODE" in cls: cls = "CODE"
else: cls = "SIMPLE"
print(f"[ROUTER → {cls}]")
print("─" * 60)
if cls == "SIMPLE":
resp = chat("executor",
f"You are a capable AI agent. Respond helpfully and directly.\n\n"
f"User: {prompt}\n\nAssistant:",
max_tok=1024, temp=0.7)
print(resp)
return resp
else:
# Plan
plan = chat("executor",
f"You are a planning agent. For the task below, output a JSON plan.\n\n"
f"Task: {prompt}\n\n"
f'Output: {{"steps": [{{"tool": "...", "params": {{}}}}]}}',
max_tok=1024, temp=0.3)
print(f"PLAN:\n{plan}")
print("─" * 60)
# Execute
exec_r = chat("executor",
f"You are an execution agent. Carry out this plan step by step.\n\n"
f"Task: {prompt}\n\nPlan: {plan}\n\n"
f"Describe each step's result:",
max_tok=1024, temp=0.3)
print(f"EXECUTE:\n{exec_r}")
print("─" * 60)
# Reflect
refl = chat("reflector",
f"Task: {prompt}\n\nPlan: {plan}\n\nResult: {exec_r}\n\n"
f"REFLECTION:\nWhat worked? What failed? What should be remembered?",
max_tok=1024, temp=0.5)
print(f"REFLECT:\n{refl}")
return exec_r
if __name__ == "__main__":
if len(sys.argv) < 2:
print("Usage: python3 run.py '<question>'")
print(" python3 run.py --plan '<task>'")
print(" python3 run.py --server status|stop")
sys.exit(1)
cmd = sys.argv[1]
if cmd == "--server":
action = sys.argv[2] if len(sys.argv) > 2 else "status"
if action == "status":
for name, cfg in MODELS.items():
icon = "●" if health(cfg["port"]) else "○"
print(f" {icon} {name}:{cfg['port']}")
elif action == "stop":
for name, cfg in MODELS.items():
kill_port(cfg["port"])
print(f" ✗ {name} stopped")
elif action == "start":
start_servers()
sys.exit(0)
# Ensure servers are running
all_up = all(health(cfg["port"]) for cfg in MODELS.values())
if not all_up:
print("Starting servers...")
start_servers()
running = [name for name, cfg in MODELS.items() if health(cfg["port"])]
if not running:
print("No servers running. Exiting.")
sys.exit(1)
print()
# Run agent
force_plan = False
msg = " ".join(sys.argv[1:])
if msg.startswith("--plan "):
msg = msg[7:]
force_plan = True
run_agent(msg, force_plan) |