File size: 5,913 Bytes
4afe981 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 | <!DOCTYPE html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>DuoVLM-40M - Ask a Question About an Image</title>
</head>
<body>
<h1>DuoVLM-40M</h1>
<p>
A 39.34M-parameter English vision-language model. One image, one short question, a one-to-three word
answer. Questions must be in English.
</p>
<p id="info">Loading model info...</p>
<hr>
<h2>Step 1 - Choose an image</h2>
<p>
<input type="file" id="file" accept="image/*">
or press Ctrl+V to paste an image from the clipboard.
</p>
<p>Bundled sample images: <span id="samples"></span></p>
<p><img id="preview" src="" alt="" width="320"></p>
<h2>Step 2 - Ask a question</h2>
<p>
<input type="button" value="Describe the image"
onclick="setQuestion('Render a clear and concise summary of the photo.')">
<input type="button" value="How many people?"
onclick="setQuestion('How many people are in the image?')">
<input type="button" value="What is the person doing?"
onclick="setQuestion('What is the person doing?')">
<input type="button" value="What sport is this?"
onclick="setQuestion('What sport is the person performing?')">
<input type="button" value="What is the person holding?"
onclick="setQuestion('What is the person holding?')">
<input type="button" value="Indoor or outdoor?"
onclick="setQuestion('Is this indoor or outdoor?')">
</p>
<p>
<textarea id="question" rows="2" cols="80">Render a clear and concise summary of the photo.</textarea>
</p>
<h2>Step 3 - Options</h2>
<p>
<label><input type="checkbox" id="blind"> Also answer without the image (blind control)</label>
</p>
<p>
<label>Max new tokens:
<input type="number" id="maxnew" value="30" min="1" max="80">
</label>
(12 for questions, 30-40 for captions)
</p>
<p><button id="ask" onclick="ask()">Ask</button></p>
<hr>
<h2>Answer</h2>
<p id="answer">Nothing yet.</p>
<p id="detail"></p>
<p id="blindanswer"></p>
<hr>
<h2>Known limits</h2>
<p>
Colours are unreliable (a brown bench was answered "blue", and the answer is not even stable).
No OCR: it cannot read text in an image. No world knowledge: ask it a fact without an image and it
guesses. English only. Answers stay short. See README.md for the measured numbers and MODEL_CARD.md
for the full list.
</p>
<script>
var currentFile = null;
var currentSample = null;
function setQuestion(t) {
document.getElementById("question").value = t;
}
function showDataUrl(url) {
document.getElementById("preview").src = url;
}
document.getElementById("file").onchange = function (e) {
currentFile = e.target.files[0] || null;
currentSample = null;
if (currentFile) {
var r = new FileReader();
r.onload = function (ev) { showDataUrl(ev.target.result); };
r.readAsDataURL(currentFile);
}
};
document.addEventListener("paste", function (e) {
var items = (e.clipboardData || {}).items || [];
for (var i = 0; i < items.length; i++) {
if (items[i].type && items[i].type.indexOf("image") === 0) {
currentFile = items[i].getAsFile();
currentSample = null;
var r = new FileReader();
r.onload = function (ev) { showDataUrl(ev.target.result); };
r.readAsDataURL(currentFile);
break;
}
}
});
fetch("/api/health").then(function (r) { return r.json(); }).then(function (h) {
document.getElementById("info").textContent =
"Weights ready: " + h.params.toLocaleString() + " parameters, checkpoint step " + h.step +
", device " + h.device + ", vision tower " + h.clip + ". The model is loaded into memory once; " +
"each question costs 0.1-0.3 seconds.";
});
fetch("/api/samples").then(function (r) { return r.json(); }).then(function (s) {
var box = document.getElementById("samples");
s.samples.forEach(function (name, i) {
var b = document.createElement("input");
b.type = "button";
b.value = "sample " + (i + 1);
b.title = name;
b.onclick = function () {
currentSample = name;
currentFile = null;
showDataUrl("/api/sample/" + encodeURIComponent(name));
document.getElementById("answer").textContent = "Sample " + name + " selected. Press Ask.";
};
box.appendChild(b);
box.appendChild(document.createTextNode(" "));
});
});
function post(body) {
document.getElementById("answer").textContent = "Thinking...";
document.getElementById("detail").textContent = "";
document.getElementById("blindanswer").textContent = "";
fetch("/api/ask", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify(body)
}).then(function (r) {
if (!r.ok) { throw new Error("HTTP " + r.status); }
return r.json();
}).then(function (res) {
document.getElementById("answer").textContent = res.answer;
document.getElementById("detail").textContent =
res.tokens + " tokens, " + res.ms + " ms generation, " + (res.emb_ms || 0) +
" ms vision tower, source " + res.source + ", decode stats " + JSON.stringify(res.stats);
if (res.blind_answer !== undefined) {
document.getElementById("blindanswer").textContent =
"Without the image (blind control): " + res.blind_answer;
}
}).catch(function (e) {
document.getElementById("answer").textContent = "Request failed: " + e.message;
});
}
function ask() {
var body = {
question: document.getElementById("question").value,
blind: document.getElementById("blind").checked,
max_new: parseInt(document.getElementById("maxnew").value, 10) || 30
};
if (currentSample) {
body.sample = currentSample;
post(body);
return;
}
if (!currentFile) {
document.getElementById("answer").textContent = "Choose an image first (Step 1).";
return;
}
var r = new FileReader();
r.onload = function (ev) { body.image_b64 = ev.target.result; post(body); };
r.readAsDataURL(currentFile);
}
</script>
</body>
</html>
|