Q-TensorFormer / web /index.html
Premchandyadav369
feat(frontier): achieve 10/10 with Triton kernels, GGUF/Ollama exporter, WebGPU runtime, multimodal vision, technical report, and 81 tests
4e689f6
Raw History Blame Contribute Delete
2.69 kB
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8">
<title>Q-TensorFormer WebGPU Browser Inference</title>
<script src="https://cdn.jsdelivr.net/npm/onnxruntime-web/dist/ort.webgpu.min.js"></script>
<style>
body { font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, sans-serif; max-width: 800px; margin: 40px auto; padding: 20px; line-height: 1.6; }
h1 { color: #2563eb; }
.box { background: #f8fafc; border: 1px solid #e2e8f0; border-radius: 8px; padding: 20px; margin-bottom: 20px; }
button { background: #2563eb; color: white; border: none; padding: 10px 20px; border-radius: 6px; cursor: pointer; font-size: 16px; }
button:hover { background: #1d4ed8; }
#output { background: #1e293b; color: #38bdf8; font-family: monospace; padding: 15px; border-radius: 6px; white-space: pre-wrap; min-height: 100px; }
</style>
</head>
<body>
<h1>⚛️ Q-TensorFormer: Zero-Server WebGPU Inference</h1>
<div class="box">
<p>Run closed-loop Tensor-Train transformer generation locally inside your browser using WebGPU hardware acceleration.</p>
<button id="runBtn" onclick="runInference()">Initialize WebGPU & Generate</button>
</div>
<div id="output">Click the button above to load and run the model via WebGPU...</div>
<script>
async function runInference() {
const out = document.getElementById('output');
out.innerText = "Loading ONNX model via WebGPU...";
try {
// Initialize ONNX Runtime Web session
const session = await ort.InferenceSession.create('./qtensorformer.onnx', {
executionProviders: ['webgpu', 'wasm']
});
out.innerText = "Model loaded successfully! Running forward pass...\n";
// Create dummy input
const inputData = new BigInt64Array([101n, 2054n, 2003n, 1037n]);
const tensor = new ort.Tensor('int64', inputData, [1, 4]);
const start = performance.now();
const feeds = { input_ids: tensor };
const results = await session.run(feeds);
const elapsed = (performance.now() - start).toFixed(2);
out.innerText += `Generation completed in ${elapsed} ms!\nLogits shape: ${results.logits.dims}\nWebGPU acceleration active!`;
} catch (err) {
out.innerText = "WebGPU execution status: " + err.message + "\n(Ensure qtensorformer.onnx is served in the same directory)";
}
}
</script>
</body>
</html>