duovlm-40m-v1 / code /index.html
Duoia's picture
DuoVLM-40M v1: from-scratch 40M vision-language model (frozen CLIP + MiniPile-pretrained LM)
4afe981 verified
Raw History Blame Contribute Delete
5.91 kB
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>DuoVLM-40M - Ask a Question About an Image</title>
</head>
<body>
<h1>DuoVLM-40M</h1>
<p>
A 39.34M-parameter English vision-language model. One image, one short question, a one-to-three word
answer. Questions must be in English.
</p>
<p id="info">Loading model info...</p>
<hr>
<h2>Step 1 - Choose an image</h2>
<p>
<input type="file" id="file" accept="image/*">
or press Ctrl+V to paste an image from the clipboard.
</p>
<p>Bundled sample images: <span id="samples"></span></p>
<p><img id="preview" src="" alt="" width="320"></p>
<h2>Step 2 - Ask a question</h2>
<p>
<input type="button" value="Describe the image"
onclick="setQuestion('Render a clear and concise summary of the photo.')">
<input type="button" value="How many people?"
onclick="setQuestion('How many people are in the image?')">
<input type="button" value="What is the person doing?"
onclick="setQuestion('What is the person doing?')">
<input type="button" value="What sport is this?"
onclick="setQuestion('What sport is the person performing?')">
<input type="button" value="What is the person holding?"
onclick="setQuestion('What is the person holding?')">
<input type="button" value="Indoor or outdoor?"
onclick="setQuestion('Is this indoor or outdoor?')">
</p>
<p>
<textarea id="question" rows="2" cols="80">Render a clear and concise summary of the photo.</textarea>
</p>
<h2>Step 3 - Options</h2>
<p>
<label><input type="checkbox" id="blind"> Also answer without the image (blind control)</label>
</p>
<p>
<label>Max new tokens:
<input type="number" id="maxnew" value="30" min="1" max="80">
</label>
(12 for questions, 30-40 for captions)
</p>
<p><button id="ask" onclick="ask()">Ask</button></p>
<hr>
<h2>Answer</h2>
<p id="answer">Nothing yet.</p>
<p id="detail"></p>
<p id="blindanswer"></p>
<hr>
<h2>Known limits</h2>
<p>
Colours are unreliable (a brown bench was answered "blue", and the answer is not even stable).
No OCR: it cannot read text in an image. No world knowledge: ask it a fact without an image and it
guesses. English only. Answers stay short. See README.md for the measured numbers and MODEL_CARD.md
for the full list.
</p>
<script>
var currentFile = null;
var currentSample = null;
function setQuestion(t) {
document.getElementById("question").value = t;
}
function showDataUrl(url) {
document.getElementById("preview").src = url;
}
document.getElementById("file").onchange = function (e) {
currentFile = e.target.files[0] || null;
currentSample = null;
if (currentFile) {
var r = new FileReader();
r.onload = function (ev) { showDataUrl(ev.target.result); };
r.readAsDataURL(currentFile);
}
};
document.addEventListener("paste", function (e) {
var items = (e.clipboardData || {}).items || [];
for (var i = 0; i < items.length; i++) {
if (items[i].type && items[i].type.indexOf("image") === 0) {
currentFile = items[i].getAsFile();
currentSample = null;
var r = new FileReader();
r.onload = function (ev) { showDataUrl(ev.target.result); };
r.readAsDataURL(currentFile);
break;
}
}
});
fetch("/api/health").then(function (r) { return r.json(); }).then(function (h) {
document.getElementById("info").textContent =
"Weights ready: " + h.params.toLocaleString() + " parameters, checkpoint step " + h.step +
", device " + h.device + ", vision tower " + h.clip + ". The model is loaded into memory once; " +
"each question costs 0.1-0.3 seconds.";
});
fetch("/api/samples").then(function (r) { return r.json(); }).then(function (s) {
var box = document.getElementById("samples");
s.samples.forEach(function (name, i) {
var b = document.createElement("input");
b.type = "button";
b.value = "sample " + (i + 1);
b.title = name;
b.onclick = function () {
currentSample = name;
currentFile = null;
showDataUrl("/api/sample/" + encodeURIComponent(name));
document.getElementById("answer").textContent = "Sample " + name + " selected. Press Ask.";
};
box.appendChild(b);
box.appendChild(document.createTextNode(" "));
});
});
function post(body) {
document.getElementById("answer").textContent = "Thinking...";
document.getElementById("detail").textContent = "";
document.getElementById("blindanswer").textContent = "";
fetch("/api/ask", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify(body)
}).then(function (r) {
if (!r.ok) { throw new Error("HTTP " + r.status); }
return r.json();
}).then(function (res) {
document.getElementById("answer").textContent = res.answer;
document.getElementById("detail").textContent =
res.tokens + " tokens, " + res.ms + " ms generation, " + (res.emb_ms || 0) +
" ms vision tower, source " + res.source + ", decode stats " + JSON.stringify(res.stats);
if (res.blind_answer !== undefined) {
document.getElementById("blindanswer").textContent =
"Without the image (blind control): " + res.blind_answer;
}
}).catch(function (e) {
document.getElementById("answer").textContent = "Request failed: " + e.message;
});
}
function ask() {
var body = {
question: document.getElementById("question").value,
blind: document.getElementById("blind").checked,
max_new: parseInt(document.getElementById("maxnew").value, 10) || 30
};
if (currentSample) {
body.sample = currentSample;
post(body);
return;
}
if (!currentFile) {
document.getElementById("answer").textContent = "Choose an image first (Step 1).";
return;
}
var r = new FileReader();
r.onload = function (ev) { body.image_b64 = ev.target.result; post(body); };
r.readAsDataURL(currentFile);
}
</script>
</body>
</html>