Spaces:
Running
Running
Download scripts/recommendation_dashboard.html from siddhm11/ResearchIT: direct link, hf CLI and curl.
- Browser
- Download file 20 kB
-
https://huggingface.co/spaces/siddhm11/ResearchIT/resolve/main/scripts/recommendation_dashboard.html
- Command line
-
hf download hf://spaces/siddhm11/ResearchIT/scripts/recommendation_dashboard.html
-
curl -L -o recommendation_dashboard.html https://huggingface.co/spaces/siddhm11/ResearchIT/resolve/main/scripts/recommendation_dashboard.html
20 kB
| <html lang="en"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1"> | |
| <title>ResearchIT · Recommendation audit</title> | |
| <style> | |
| :root{--bg:#f5f5ef;--ink:#172b28;--muted:#5a6964;--line:#d6dfd6;--green:#19694c;--amber:#97630a;--red:#a63131}*{box-sizing:border-box}html{scroll-behavior:smooth}body{margin:0;background:var(--bg);color:var(--ink);font:16px/1.6 system-ui,-apple-system,sans-serif}a{color:var(--green);text-underline-offset:3px}header{background:#173b32;color:white;padding:48px max(24px,calc((100vw - 1180px)/2)) 36px}header p{color:#d1e4d7;max-width:770px}h1{font-size:clamp(30px,5vw,54px);line-height:1.1;letter-spacing:-2px;max-width:850px;margin:18px 0}h2{font-size:27px;letter-spacing:-.8px;margin:0 0 12px}h3{font-size:18px;margin:0 0 10px}.eyebrow{font:12px ui-monospace,monospace;text-transform:uppercase;letter-spacing:2px}nav{background:#e8eee5;border-bottom:1px solid var(--line);padding:13px 24px;display:flex;gap:22px;flex-wrap:wrap;justify-content:center}nav a{text-decoration:none;font-size:14px;font-weight:650}main{max-width:1228px;margin:auto;padding:30px 24px 70px}section{margin:34px 0 50px;scroll-margin-top:20px}.grid{display:grid;grid-template-columns:repeat(3,minmax(0,1fr));gap:16px}.two{grid-template-columns:repeat(2,minmax(0,1fr))}.card{border:1px solid var(--line);background:#fff;border-radius:12px;padding:24px;min-width:0}.metric{font-size:40px;letter-spacing:-2px;line-height:1.2;font-weight:650;margin:14px 0}.muted,small{color:var(--muted)}.pill{display:inline-block;padding:4px 9px;border-radius:5px;font-size:11px;font-weight:750;letter-spacing:.7px;text-transform:uppercase;background:#e8eee5}.pass,.passed,.measured{background:#e1f1e6;color:var(--green)}.blocked,.not_run,.not_configured,.skipped{background:#fff0cd;color:var(--amber)}.fail,.failed,.error{background:#fce5e5;color:var(--red)}.callout{border-left:4px solid #c99531;background:#fff6df;padding:18px 22px;margin:22px 0}.flow{display:grid;grid-template-columns:repeat(5,minmax(0,1fr));gap:10px;margin:20px 0}.flow div{background:#e8eee5;border-radius:8px;padding:16px;font-size:14px}.flow b{display:block;margin-bottom:8px}.table-wrap{overflow:auto}table{width:100%;border-collapse:collapse;text-align:left;font-size:14px}th{font-size:11px;letter-spacing:1px;text-transform:uppercase;color:var(--muted);background:#edf1e9}td,th{padding:12px;border-bottom:1px solid var(--line);vertical-align:top}td:first-child{font-weight:550}code{font:12px ui-monospace,monospace;overflow-wrap:anywhere}pre{white-space:pre-wrap;background:#172b28;color:#e3f3e5;padding:20px;border-radius:8px}details{border:1px solid var(--line);border-radius:8px;padding:14px;background:white;margin:10px 0}summary{cursor:pointer;font-weight:600}.bars{display:flex;height:28px;border-radius:5px;overflow:hidden;margin:12px 0}.bars span{display:flex;align-items:center;justify-content:center;background:#286e51;color:white;font-size:12px}.bars span:nth-child(2){background:#c39a45;color:#192c24}.bars span:nth-child(3){background:#839ec1;color:#172b28}.controls{display:flex;gap:12px;flex-wrap:wrap;margin:20px 0}input,select,button{font:inherit;padding:9px 12px;border:1px solid #9aaa9e;border-radius:6px;background:white;color:var(--ink)}input{flex:1;min-width:180px}button{cursor:pointer}button:hover{background:#e8eee5}button:focus-visible,a:focus-visible,summary:focus-visible{outline:3px solid #b58321;outline-offset:4px}footer{font-size:12px;color:var(--muted);border-top:1px solid var(--line);padding-top:20px}#test-list{max-height:440px;overflow:auto}.task{counter-increment:task}.task h3:before{content:counter(task,decimal-leading-zero)' / ';color:var(--green)}.tasks{counter-reset:task}.caption{font-size:13px;margin:5px 0}.sr{position:absolute;width:1px;height:1px;overflow:hidden;clip:rect(0,0,0,0)}@media(max-width:780px){.grid,.two{grid-template-columns:1fr}.flow{grid-template-columns:1fr 1fr}header{padding:32px 24px}h1{letter-spacing:-1px}td,th{padding:9px}.card{padding:20px}}@media print{nav,.controls{display:none}body{background:white}.card{break-inside:avoid}#test-list{max-height:none}header{background:white;color:#173b32}header p{color:#425a4d}} | |
| </style></head><body> | |
| <header><div class="eyebrow">ResearchIT / Evidence review / {{ report.generated_local[:10] }}</div><h1>Does the feed discover<br>research worth reading?</h1><p>A working recommendation structure. A freshness gap. Semantic quality still needs real-model evidence. This dashboard separates those three conclusions.</p><span class="pill">Local audit · No deployment</span></header> | |
| <nav aria-label="Sections"><a href="#verdict">Where we stand</a><a href="#structure">Engine structure</a><a href="#huggingface">Hugging Face</a><a href="#embeddings">Embeddings</a><a href="#tests">Test results</a><a href="#next">Next steps</a></nav> | |
| <main> | |
| <section id="verdict"><div class="grid"> | |
| <div class="card"><span class="pill {{ 'failed' if report.tests.exit_code else 'passed' }}">Executed tests</span><div class="metric">{{ report.tests.counts.get('passed',0) }} passed</div><p>{{ report.tests.counts.get('skipped',0) }} skipped · {{ report.tests.counts.get('failed',0)+report.tests.counts.get('error',0) }} failures/errors.</p><p class="caption muted">Entire local suite, including existing tests. This count measures correctness checks, not recommendation accuracy.</p></div> | |
| <div class="card"><span class="pill">Recommendation quality</span><div class="metric">Unproven</div><p>The architecture is defensible. We cannot yet say the actual papers are consistently good for real readers.</p><p class="caption muted">No invented 8/10 score. Independent relevance judgments are absent.</p></div> | |
| <div class="card"><span class="pill blocked">Biggest gap</span><div class="metric">Fresh supply</div><p>Citation popularity is not live trending. Refresh can reveal unseen archive papers without discovering new releases.</p><p class="caption muted">A zero-citation paper can be excluded by the current remote popularity query.</p></div></div> | |
| <div class="callout"><strong>What this audit does not prove:</strong> synthetic vectors do not test BGE-M3's understanding of research. A blocked service is not a bad model. Two observed Hugging Face records are not a quality benchmark.</div></section> | |
| <section id="structure"><div class="eyebrow">01 / Structure</div><h2>Keep the core. Separate where papers come from.</h2><p class="muted">Preserve distinct interests and their quotas. Add external discovery before ranking, with explicit source provenance and candidate readiness.</p> | |
| <div class="flow"><div><b>1. Candidate sources</b>Existing corpus + optional HF + new arXiv ingestion</div><div><b>2. Eligibility</b>Canonical ID, metadata, freshness, saved/dismissed state</div><div><b>3. Personal relevance</b>Saved-paper clusters, medoids, profile similarity</div><div><b>4. Composition</b>Interest quotas, heuristic rank, within-cluster diversity</div><div><b>5. Delivery & learning</b>Fresh refresh, explanations, history, explicit feedback</div></div> | |
| <div class="grid">{% for case in report.structural %}<article class="card"><span class="pill {{ case.status }}">{{ case.status }} · {{ case.runs }} runs</span><h3 style="margin-top:14px">{{ case.name }}</h3><p>{{ case.save_counts|join(' / ') }} saved examples across the simulated interests.</p><div class="bars" aria-label="First page composition for the first deterministic simulation">{% for n in case.first_page_counts[0] %}<span style="width:{{ n*10 }}%">{{ n }}</span>{% endfor %}</div><p class="caption">First run: {{ case.first_page_counts[0]|join(' / ') }} papers on page one. Each color is a different simulated interest.</p><details><summary>Inspect all 20 runs</summary><p class="caption">Real Ward, quota, heuristic and MMR functions; generated 1024-dimensional vectors. The dominant profile deliberately favors one topic.</p><code>{{ case.first_page_counts }}</code><p class="caption">{{ case.limitation }}</p></details></article>{% endfor %}</div> | |
| <p><strong>Structural stress result:</strong> {% set failures=report.structural|selectattr('status','equalto','fail')|list %}{{ 'Some scenarios failed; inspect the runs above.' if failures else 'All three simulated profiles retained their interests across 60 seeded runs.' }} This is a controlled composition test, not an end-to-end semantic evaluation.</p></section> | |
| <section id="huggingface"><div class="eyebrow">02 / Fresh discovery</div><h2>Yes to Hugging Face as a candidate source.</h2><div class="grid two"><div class="card"><h3>Built in this change</h3><p>A Daily Papers adapter and scheduled worker now persist dated source observations in separate local SQLite. Metadata enrichment, compatible local embeddings, freshness checks, bounded retries and concurrent-worker exclusion are implemented.</p><p><strong>Shadow evaluation only:</strong> it is not wired into the serving feed. The app now owns scheduling when HF_DISCOVERY_MODE=shadow; /healthz/discovery reports collection, preparation, and serving separately. Missing shared model dependencies preserve candidate retries. These changes have not been deployed. Production indexing and measured trend growth remain next steps. <a href="shadow-demo/comparison.html">Inspect the explicitly synthetic end-to-end comparison →</a></p><a href="https://huggingface.co/api/daily_papers" target="_blank" rel="noopener">Inspected public Daily Papers endpoint ↗</a></div><div class="card"><h3>Learn preferences, not a copy of popularity</h3><p>HF can bootstrap useful candidates while ResearchIT has little traffic. Votes are an attention signal, not a save label for every user.</p><p>Keep source collection separate from the ranker. Reduce HF's contribution only when a source-removal experiment preserves fresh coverage and judged relevance. Training alone will never tell us tomorrow's releases.</p></div></div> | |
| {% if report.discovery_pipeline %}<details><summary>Latest worker execution status (local snapshot)</summary><pre>{{ report.discovery_pipeline|tojson(indent=2) }}</pre><p>The worker writes only its separate shadow database. No production candidate indexing has occurred.</p></details>{% endif %} | |
| {% if report.hf_observed_sample %}<h3 style="margin-top:25px">Actual source observations</h3><p class="caption muted">{{ report.hf_observed_sample.capture_method }} Observed {{ report.hf_observed_sample.observed_on }}. These have not been verified as indexed in ResearchIT.</p><div class="table-wrap"><table><thead><tr><th>Paper</th><th>Publication date</th><th>Votes observed</th><th>What this establishes</th></tr></thead><tbody>{% for row in report.hf_observed_sample.records %}<tr><td><a href="https://huggingface.co/papers/{{ row.paper.id }}" target="_blank" rel="noopener">{{ row.paper.title }}</a><br><small>{{ row.paper.id }}</small></td><td>{{ row.paper.publishedAt[:10] }}</td><td>{{ row.paper.upvotes }}</td><td>Real candidate metadata; no measured trend velocity or relevance grade.</td></tr>{% endfor %}</tbody></table></div>{% endif %} | |
| <div class="callout"><strong>Jev is now identified:</strong> <a href="https://typesafe.ai/blog/introducing-system-one-models-and-jev" target="_blank" rel="noopener">TypeSafe’s official announcement</a> introduces Jev on September 15, 2026 as a model for structured decisions. This is a model-release discovery case, not automatically an arXiv paper. Its inclusion in ResearchIT and its performance claims are not verified by this audit.</div> | |
| <p class="muted">One source snapshot shows current attention, not acceleration. HF primarily covers AI research; it cannot replace candidate sources for every ResearchIT category.</p></section> | |
| <section id="embeddings"><div class="eyebrow">03 / Representation quality</div><h2>Are the embeddings doing their job?</h2><p>There is not enough evidence to answer yes or no. The suite separates model semantics, stored-vector integrity, retrieval coverage, and final ranking so we can locate the cause of a poor result.</p> | |
| <div class="grid two"><div class="card"><h3>Local BGE-M3 semantic probe</h3><span class="pill {{ report.semantics.status }}">{{ report.semantics.status }}</span><p>{{ report.semantics.get('reason',report.semantics.get('note','')) }}</p><p class="caption">Six authored query/positive/hard-negative cases: calibration, retrieval, compression, robotics, medical imaging and safety. Real encoding only; no fallback to random vectors.</p>{% for r in report.semantics.get('results',[]) %}<p>{{ r.case }}: <span class="pill {{ r.status }}">{{ r.status }}</span> margin {{ r.get('margin','n/a') }}</p>{% endfor %}</div> | |
| <div class="card"><h3>Bundled ranker audit</h3><div class="metric">{{ report.model.get('personalization_splits_20_30','?') }} splits</div><p>On personalization features 20–30, across {{ report.model.get('trees','?') }} trees in the bundled model.</p><p class="caption">{{ report.model.get('note','Model unavailable') }} The configured scorer is <strong>{{ report.environment.scorer }}</strong>. A model file with many trees is not evidence of personalized learning.</p></div></div> | |
| <div class="table-wrap" style="margin-top:22px"><table><thead><tr><th>Runtime probe</th><th>Status</th><th>Evidence / limitation</th></tr></thead><tbody>{% for name, value in report.live.items() if value is mapping %}<tr><td>{{ name.replace('_',' ') }}</td><td><span class="pill {{ value.status }}">{{ value.status }}</span></td><td>{{ value.get('reason',value.get('note','Results available in the JSON evidence export.')) }}{% if value.get('points') is not none %} · {{ value.points }} points{% endif %}</td></tr>{% endfor %}{% if report.live.get('status') %}<tr><td>External probes</td><td>{{ report.live.status }}</td><td>{{ report.live.reason }}</td></tr>{% endif %}</tbody></table></div> | |
| <details><summary>Environment and the next embedding checks</summary><p><code>{{ report.environment }}</code></p><ol><li>Inspect dimensions, finite values, vector norms and collapsed duplicates.</li><li>Compare exact nearest neighbors with approximate retrieval to isolate index/quantization losses.</li><li>Compare dense, keyword and hybrid results using the same judged candidate pool.</li><li>Inspect recommendation pages for separate reader interests and recent-paper coverage.</li></ol><p>Stored-vector probes and real HTTP feed collection are executable with reachable configured services. Exact-vs-approximate probes are executable with reachable stores; independently judged ablations still need a labeled dataset.</p></details></section> | |
| <section id="tests"><div class="eyebrow">04 / Reproducible evidence</div><h2>Inspect the actual test cases</h2><p>HTTP refresh and history tests use isolated SQLite with mocked retrieval. Algorithm stress tests use real code with synthetic geometry. The source adapter uses deterministic HTTP fixtures. None is labelled as live relevance.</p> | |
| <div class="controls"><label class="sr" for="search">Filter test names</label><input id="search" type="search" placeholder="Search: quota, refresh, vector, history…"><label class="sr" for="status">Test status</label><select id="status"><option value="all">All statuses</option><option value="passed">Passed</option><option value="skipped">Skipped</option><option value="failed">Failed</option><option value="error">Errors</option></select><button id="export" type="button">Download full evidence JSON</button></div><p id="count" aria-live="polite" class="caption">{{ report.tests.cases|length }} recorded cases. Live/browser marked tests excluded.</p> | |
| <div id="test-list" class="table-wrap" tabindex="0" aria-label="Scrollable test results"><table><thead><tr><th>Case</th><th>Suite</th><th>Status</th></tr></thead><tbody>{% for t in report.tests.cases %}<tr data-test data-status="{{ t.status }}"><td>{{ t.name }}</td><td><small>{{ t.group }}</small></td><td><span class="pill {{ t.status }}">{{ t.status }}</span></td></tr>{% endfor %}</tbody></table></div> | |
| <p class="caption">{{ report.tests.deselected_note }} Pytest exit code: {{ report.tests.exit_code }}. This dashboard reports failures rather than hiding them.</p> | |
| <details><summary>Human-rated recommendation quality</summary> | |
| {% if report.judged_quality %}<pre>{{ report.judged_quality|tojson(indent=2) }}</pre>{% else %}<p>No human judgment file supplied. Precision, NDCG and relevance recall are <strong>not evaluable</strong>, rather than zero or 100%.</p>{% endif %} | |
| <p>The runner accepts <code>--judgments /path/to/judgments.json</code>. It requires evaluation timestamps, rejects future source observations, refuses incomplete top-page labels, and reports pooled recall rather than pretending the entire corpus was judged.</p></details> | |
| <details><summary>How to reproduce</summary><pre>.venv/bin/python scripts/run_recommendation_audit.py --live --encode</pre><p>Writes HTML, JSON and JUnit results into <code>reports/recommendations/</code>. Uses temporary user storage; Turso replication stays disabled. External probes read public papers and configured collections. Encoding can download model weights when its runtime is installed.</p><p>Without flags, it runs local checks and marks external/semantic measurements not run. Missing prerequisites are not counted as passing quality checks.</p></details></section> | |
| <section id="next"><div class="eyebrow">05 / What to do next</div><h2>A practical path to better recommendations</h2><div class="grid tasks"><article class="card task"><h3>Fix fresh candidate delivery</h3><p>Schedule source snapshots. Canonicalize and ingest new papers, including zero-citation papers. Track which have searchable metadata and compatible vectors.</p><span class="pill">First priority</span></article><article class="card task"><h3>Evaluate before promotion</h3><p>Gather dated candidate pools and reader briefs. Blindly judge current, content-only, HF-only and combined feeds. Use an untouched time window and report relevance by interest.</p><span class="pill">Release gate</span></article><article class="card task"><h3>Learn from explicit feedback</h3><p>Record helpful discoveries, saves, dismissals and returns with source attribution. Keep views separate from approval. Trial reviewed video explanations after paper discovery works.</p><span class="pill">After baseline evidence</span></article></div> | |
| <div class="callout"><strong>Recommendation:</strong> retain the current multi-interest architecture and personalized heuristic. Add HF as one replaceable source. Do not retrain or replace BGE-M3 based on synthetic tests, and do not remove all external sources after a fixed number of months.</div></section> | |
| <footer>Generated {{ report.generated_at }} · Code/test fingerprint <code>{{ report.code_sha256[:20] }}</code><br>{{ report.browser_verification }}<br>Standalone report; no external assets, telemetry or server required. Interactive filtering and JSON export run entirely in this tab. Source observations are dated; this is not a live monitor.</footer> | |
| </main> | |
| <script id="evidence" type="application/json"></script> | |
| <script> | |
| ; | |
| const search=document.getElementById('search'),status=document.getElementById('status'); | |
| function filter(){let count=0;document.querySelectorAll('[data-test]').forEach(row=>{const show=(status.value==='all'||row.dataset.status===status.value)&&row.textContent.toLowerCase().includes(search.value.toLowerCase());row.hidden=!show;if(show)count++;});document.getElementById('count').textContent=count+' matching test cases';} | |
| search.addEventListener('input',filter);status.addEventListener('change',filter); | |
| document.getElementById('export').addEventListener('click',()=>{const blob=new Blob([document.getElementById('evidence').textContent],{type:'application/json'}),url=URL.createObjectURL(blob),a=document.createElement('a');a.href=url;a.download='researchit-recommendation-evidence.json';a.click();setTimeout(()=>URL.revokeObjectURL(url),1000);}); | |
| </script></body></html> | |