Text Generation
Transformers
English
qtensorformer
tensor-networks
model-compression
adaptive-computation
kv-cache-compression
hardware-aware
energy-aware
quantum-machine-learning
green-ai
Instructions to use Premchan369/Q-TensorFormer with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use Premchan369/Q-TensorFormer with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="Premchan369/Q-TensorFormer")# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("Premchan369/Q-TensorFormer", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use Premchan369/Q-TensorFormer with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "Premchan369/Q-TensorFormer" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Premchan369/Q-TensorFormer", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/Premchan369/Q-TensorFormer
- SGLang
How to use Premchan369/Q-TensorFormer with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "Premchan369/Q-TensorFormer" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Premchan369/Q-TensorFormer", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "Premchan369/Q-TensorFormer" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Premchan369/Q-TensorFormer", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use Premchan369/Q-TensorFormer with Docker Model Runner:
docker model run hf.co/Premchan369/Q-TensorFormer
Premchandyadav369
feat(research): Turn Q-TensorFormer into a verified research system with LFS figure assets
0e4850b Download experiments/run_controller_convergence.py from Premchan369/Q-TensorFormer: direct link, hf CLI and curl.
- Browser
- Download file 6.34 kB
-
https://huggingface.co/Premchan369/Q-TensorFormer/resolve/main/experiments/run_controller_convergence.py
- Command line
-
hf download hf://Premchan369/Q-TensorFormer/experiments/run_controller_convergence.py
-
curl -L -o run_controller_convergence.py https://huggingface.co/Premchan369/Q-TensorFormer/resolve/main/experiments/run_controller_convergence.py
6.34 kB
| #!/usr/bin/env python3 | |
| """ | |
| experiments/run_controller_convergence.py | |
| Component 9: Online Closed-Loop Controller Convergence & Disturbance Rejection Benchmark. | |
| Evaluates PID Dual Subgradient Controller against: | |
| 1. Full PID Dual Controller (Kp=0.08, Ki=0.02, Kd=0.01) | |
| 2. Proportional-Only Controller (Kp=0.08, Ki=0.0, Kd=0.0) | |
| 3. Fixed Multiplier Controller (lambda = const) | |
| 4. Static Heuristic Baseline (fixed rank) | |
| Under Step Disturbance: | |
| - Tokens 1-30: nominal load (context = 32) | |
| - Tokens 31-100: sudden step disturbance (context jumps to 128, doubling base compute demand) | |
| Tracks: | |
| - Settling time (steps to recover to +/- 5% SLA band) | |
| - Maximum Overshoot percentage (Mp) | |
| - Steady-state error (e_ss) | |
| - SLA Violation Rate (%) | |
| """ | |
| import sys | |
| import os | |
| import json | |
| import time | |
| from typing import Dict, Any, List | |
| import torch | |
| import numpy as np | |
| sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) | |
| from src.resource_allocator import PIDDualSubgradientController, InformationValueAllocator, AllocationBudget | |
| from src.information_state import TokenInformationState | |
| def simulate_hardware_latency(rank: int, seq_len: int, noise_scale: float = 0.02) -> float: | |
| """Simulate realistic hardware execution latency (ms) given rank and seq_len.""" | |
| # Rank base cost: rank 1 -> 0.4ms, rank 2 -> 0.6ms, rank 4 -> 0.9ms, rank 8 -> 1.5ms | |
| rank_costs = {1: 0.40, 2: 0.60, 4: 0.90, 8: 1.50} | |
| base = rank_costs.get(rank, 0.90) | |
| # Sequence length scaling | |
| scale = 1.0 + (seq_len - 32) * 0.008 | |
| noise = np.random.normal(0, noise_scale) | |
| return max(0.2, (base * scale) + noise) | |
| def run_controller_experiment(): | |
| np.random.seed(42) | |
| torch.manual_seed(42) | |
| total_steps = 100 | |
| step_disturbance_at = 30 | |
| target_latency_ms = 1.20 # SLA target | |
| controllers = { | |
| "PID_Dual_Controller": { | |
| "type": "pid", | |
| "kp": 0.08, "ki": 0.02, "kd": 0.01, | |
| }, | |
| "P_Only_Controller": { | |
| "type": "pid", | |
| "kp": 0.08, "ki": 0.0, "kd": 0.0, | |
| }, | |
| "Fixed_Multiplier": { | |
| "type": "fixed", | |
| "fixed_lambda": 1.0, | |
| }, | |
| "Static_Heuristic": { | |
| "type": "static", | |
| "fixed_rank": 8, | |
| }, | |
| } | |
| results: Dict[str, Any] = { | |
| "metadata": { | |
| "timestamp": time.strftime("%Y-%m-%d %H:%M:%S"), | |
| "total_steps": total_steps, | |
| "step_disturbance_step": step_disturbance_at, | |
| "target_latency_ms": target_latency_ms, | |
| }, | |
| "controllers": {}, | |
| } | |
| allocator = InformationValueAllocator(info_dim=8) | |
| info_engine = TokenInformationState(d_model=64) | |
| for c_name, c_cfg in controllers.items(): | |
| if c_cfg["type"] == "pid": | |
| ctrl = PIDDualSubgradientController( | |
| target_latency_ms=target_latency_ms, | |
| kp=c_cfg["kp"], | |
| ki=c_cfg["ki"], | |
| kd=c_cfg["kd"], | |
| ) | |
| else: | |
| ctrl = None | |
| history = [] | |
| cur_lambda = c_cfg.get("fixed_lambda", 1.0) | |
| for t in range(1, total_steps + 1): | |
| # Context length step disturbance | |
| seq_len = 32 if t <= step_disturbance_at else 128 | |
| x = torch.randn(1, 1, 64) | |
| z_t = info_engine(x) | |
| if c_cfg["type"] == "static": | |
| chosen_rank = c_cfg["fixed_rank"] | |
| budget = AllocationBudget(max_latency_ms=target_latency_ms) | |
| elif c_cfg["type"] == "fixed": | |
| budget = AllocationBudget(max_latency_ms=target_latency_ms, lambda_latency=cur_lambda) | |
| decisions, _ = allocator(z_t, budget=budget) | |
| chosen_rank = decisions["rank"] | |
| elif c_cfg["type"] == "pid": | |
| budget = ctrl.get_budget() | |
| decisions, _ = allocator(z_t, budget=budget) | |
| chosen_rank = decisions["rank"] | |
| # Observe hardware response | |
| measured_latency = simulate_hardware_latency(chosen_rank, seq_len) | |
| # Update controller | |
| if ctrl is not None: | |
| budget = ctrl.update(measured_latency_ms=measured_latency) | |
| cur_lambda = budget.lambda_latency | |
| history.append({ | |
| "step": t, | |
| "seq_len": seq_len, | |
| "chosen_rank": chosen_rank, | |
| "measured_latency_ms": round(measured_latency, 3), | |
| "target_latency_ms": target_latency_ms, | |
| "error": round(measured_latency - target_latency_ms, 3), | |
| "lambda_latency": round(cur_lambda, 4), | |
| }) | |
| # Post-disturbance analysis (steps 31 to 100) | |
| post_dist = [h for h in history if h["step"] > step_disturbance_at] | |
| post_errors = [h["error"] for h in post_dist] | |
| violations = sum(1 for e in post_errors if e > 0) | |
| violation_rate = (violations / len(post_errors)) * 100.0 | |
| max_overshoot = max(0.0, (max(post_errors) / target_latency_ms) * 100.0) | |
| # Steady state error (last 20 steps) | |
| last_20 = [h["error"] for h in history[-20:]] | |
| steady_state_err = sum(abs(e) for e in last_20) / len(last_20) | |
| # Settling time after step disturbance | |
| band = 0.05 * target_latency_ms | |
| settling_steps = len(post_dist) | |
| for idx in range(len(post_dist)): | |
| if all(abs(e) <= band for e in post_errors[idx:]): | |
| settling_steps = idx + 1 | |
| break | |
| results["controllers"][c_name] = { | |
| "settling_steps_post_disturbance": settling_steps, | |
| "violation_rate_pct": round(violation_rate, 2), | |
| "max_overshoot_pct": round(max_overshoot, 2), | |
| "steady_state_error_ms": round(steady_state_err, 4), | |
| "mean_rank_post_disturbance": round(float(np.mean([h["chosen_rank"] for h in post_dist])), 2), | |
| "history_sample": history[::10], | |
| } | |
| out_path = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "outputs", "controller_convergence.json") | |
| with open(out_path, "w") as f: | |
| json.dump(results, f, indent=2) | |
| print(f"Controller convergence results saved to {out_path}") | |
| print(json.dumps(results, indent=2)) | |
| if __name__ == "__main__": | |
| run_controller_experiment() | |