Q-TensorFormer / experiments /run_controller_convergence.py
Premchandyadav369
feat(research): Turn Q-TensorFormer into a verified research system with LFS figure assets
0e4850b
Raw History Blame Contribute Delete
6.34 kB
#!/usr/bin/env python3
"""
experiments/run_controller_convergence.py
Component 9: Online Closed-Loop Controller Convergence & Disturbance Rejection Benchmark.
Evaluates PID Dual Subgradient Controller against:
1. Full PID Dual Controller (Kp=0.08, Ki=0.02, Kd=0.01)
2. Proportional-Only Controller (Kp=0.08, Ki=0.0, Kd=0.0)
3. Fixed Multiplier Controller (lambda = const)
4. Static Heuristic Baseline (fixed rank)
Under Step Disturbance:
- Tokens 1-30: nominal load (context = 32)
- Tokens 31-100: sudden step disturbance (context jumps to 128, doubling base compute demand)
Tracks:
- Settling time (steps to recover to +/- 5% SLA band)
- Maximum Overshoot percentage (Mp)
- Steady-state error (e_ss)
- SLA Violation Rate (%)
"""
import sys
import os
import json
import time
from typing import Dict, Any, List
import torch
import numpy as np
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from src.resource_allocator import PIDDualSubgradientController, InformationValueAllocator, AllocationBudget
from src.information_state import TokenInformationState
def simulate_hardware_latency(rank: int, seq_len: int, noise_scale: float = 0.02) -> float:
"""Simulate realistic hardware execution latency (ms) given rank and seq_len."""
# Rank base cost: rank 1 -> 0.4ms, rank 2 -> 0.6ms, rank 4 -> 0.9ms, rank 8 -> 1.5ms
rank_costs = {1: 0.40, 2: 0.60, 4: 0.90, 8: 1.50}
base = rank_costs.get(rank, 0.90)
# Sequence length scaling
scale = 1.0 + (seq_len - 32) * 0.008
noise = np.random.normal(0, noise_scale)
return max(0.2, (base * scale) + noise)
def run_controller_experiment():
np.random.seed(42)
torch.manual_seed(42)
total_steps = 100
step_disturbance_at = 30
target_latency_ms = 1.20 # SLA target
controllers = {
"PID_Dual_Controller": {
"type": "pid",
"kp": 0.08, "ki": 0.02, "kd": 0.01,
},
"P_Only_Controller": {
"type": "pid",
"kp": 0.08, "ki": 0.0, "kd": 0.0,
},
"Fixed_Multiplier": {
"type": "fixed",
"fixed_lambda": 1.0,
},
"Static_Heuristic": {
"type": "static",
"fixed_rank": 8,
},
}
results: Dict[str, Any] = {
"metadata": {
"timestamp": time.strftime("%Y-%m-%d %H:%M:%S"),
"total_steps": total_steps,
"step_disturbance_step": step_disturbance_at,
"target_latency_ms": target_latency_ms,
},
"controllers": {},
}
allocator = InformationValueAllocator(info_dim=8)
info_engine = TokenInformationState(d_model=64)
for c_name, c_cfg in controllers.items():
if c_cfg["type"] == "pid":
ctrl = PIDDualSubgradientController(
target_latency_ms=target_latency_ms,
kp=c_cfg["kp"],
ki=c_cfg["ki"],
kd=c_cfg["kd"],
)
else:
ctrl = None
history = []
cur_lambda = c_cfg.get("fixed_lambda", 1.0)
for t in range(1, total_steps + 1):
# Context length step disturbance
seq_len = 32 if t <= step_disturbance_at else 128
x = torch.randn(1, 1, 64)
z_t = info_engine(x)
if c_cfg["type"] == "static":
chosen_rank = c_cfg["fixed_rank"]
budget = AllocationBudget(max_latency_ms=target_latency_ms)
elif c_cfg["type"] == "fixed":
budget = AllocationBudget(max_latency_ms=target_latency_ms, lambda_latency=cur_lambda)
decisions, _ = allocator(z_t, budget=budget)
chosen_rank = decisions["rank"]
elif c_cfg["type"] == "pid":
budget = ctrl.get_budget()
decisions, _ = allocator(z_t, budget=budget)
chosen_rank = decisions["rank"]
# Observe hardware response
measured_latency = simulate_hardware_latency(chosen_rank, seq_len)
# Update controller
if ctrl is not None:
budget = ctrl.update(measured_latency_ms=measured_latency)
cur_lambda = budget.lambda_latency
history.append({
"step": t,
"seq_len": seq_len,
"chosen_rank": chosen_rank,
"measured_latency_ms": round(measured_latency, 3),
"target_latency_ms": target_latency_ms,
"error": round(measured_latency - target_latency_ms, 3),
"lambda_latency": round(cur_lambda, 4),
})
# Post-disturbance analysis (steps 31 to 100)
post_dist = [h for h in history if h["step"] > step_disturbance_at]
post_errors = [h["error"] for h in post_dist]
violations = sum(1 for e in post_errors if e > 0)
violation_rate = (violations / len(post_errors)) * 100.0
max_overshoot = max(0.0, (max(post_errors) / target_latency_ms) * 100.0)
# Steady state error (last 20 steps)
last_20 = [h["error"] for h in history[-20:]]
steady_state_err = sum(abs(e) for e in last_20) / len(last_20)
# Settling time after step disturbance
band = 0.05 * target_latency_ms
settling_steps = len(post_dist)
for idx in range(len(post_dist)):
if all(abs(e) <= band for e in post_errors[idx:]):
settling_steps = idx + 1
break
results["controllers"][c_name] = {
"settling_steps_post_disturbance": settling_steps,
"violation_rate_pct": round(violation_rate, 2),
"max_overshoot_pct": round(max_overshoot, 2),
"steady_state_error_ms": round(steady_state_err, 4),
"mean_rank_post_disturbance": round(float(np.mean([h["chosen_rank"] for h in post_dist])), 2),
"history_sample": history[::10],
}
out_path = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "outputs", "controller_convergence.json")
with open(out_path, "w") as f:
json.dump(results, f, indent=2)
print(f"Controller convergence results saved to {out_path}")
print(json.dumps(results, indent=2))
if __name__ == "__main__":
run_controller_experiment()