Spaces:
Running
feat(discovery): balance the starter feed per interest and give it a recent lane
Browse filesMeasured 2026-09-30 for a reader who ticked NLP, CV, ML, AI, Robotics and
Software Engineering, the 200-paper starter pool held 147 papers from 2024,
53 from early 2025 and none from 2026, and Computer Vision got 22 of 200.
- Deal per ticked interest, not per arXiv code. Two-code interests (ML is
cs.LG + stat.ML) were getting twice the share of one-code ones.
- Rank the established lane by citations per month since publication
(floored at three months). Raw totals inside a 24-month window always
favour the oldest papers in it.
- Every third slot per interest comes from a new fresh lane: papers from
the corpus's last three months, most-cited first. A few-weeks-old paper
cannot win on citations however they are ranked.
Onboarding suggestions use the same grouping.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
- app/db.py +14 -0
- app/discovery_svc.py +70 -21
- app/local_meta.py +85 -6
- app/routers/onboarding.py +1 -1
- app/routers/recommendations.py +3 -1
- app/turso_svc.py +24 -0
- tests/test_discovery_journey.py +38 -1
- tests/test_onboarding.py +9 -0
|
@@ -633,6 +633,20 @@ async def get_user_category_filter(user_id: str) -> set[str]:
|
|
| 633 |
return expand_category_groups(state["selected_categories"])
|
| 634 |
|
| 635 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 636 |
# ── Phase 6.5 B3: Cluster snapshot versioning ─────────────────────────────────
|
| 637 |
|
| 638 |
async def save_cluster_snapshot(user_id: str, clusters: list[dict]) -> str:
|
|
|
|
| 633 |
return expand_category_groups(state["selected_categories"])
|
| 634 |
|
| 635 |
|
| 636 |
+
async def get_user_category_groups(user_id: str) -> dict[str, set[str]]:
|
| 637 |
+
"""The reader's selected interests, each mapped to its arXiv codes.
|
| 638 |
+
|
| 639 |
+
Same source as get_user_category_filter, without flattening: starter feeds
|
| 640 |
+
balance per interest, and flattening loses which codes belong together.
|
| 641 |
+
"""
|
| 642 |
+
state = await get_onboarding_state(user_id)
|
| 643 |
+
if state is None:
|
| 644 |
+
return {}
|
| 645 |
+
from app.config import expand_category_groups
|
| 646 |
+
return {key: codes for key in state["selected_categories"]
|
| 647 |
+
if (codes := expand_category_groups([key]))}
|
| 648 |
+
|
| 649 |
+
|
| 650 |
# ── Phase 6.5 B3: Cluster snapshot versioning ─────────────────────────────────
|
| 651 |
|
| 652 |
async def save_cluster_snapshot(user_id: str, clusters: list[dict]) -> str:
|
|
@@ -2,39 +2,88 @@
|
|
| 2 |
from __future__ import annotations
|
| 3 |
|
| 4 |
import asyncio
|
|
|
|
| 5 |
from itertools import zip_longest
|
| 6 |
|
| 7 |
from app import local_meta, turso_svc
|
| 8 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 9 |
|
| 10 |
-
async def starter_papers(categories: set[str], limit: int = 30) -> list[dict]:
|
| 11 |
-
"""Give thin categories room without eight remote full-table scans.
|
| 12 |
|
| 13 |
-
|
| 14 |
-
|
| 15 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 16 |
"""
|
| 17 |
-
|
|
|
|
|
|
|
|
|
|
| 18 |
return []
|
| 19 |
-
|
| 20 |
-
|
|
|
|
| 21 |
semaphore = asyncio.Semaphore(4)
|
| 22 |
|
| 23 |
-
async def
|
| 24 |
async with semaphore:
|
| 25 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 26 |
|
| 27 |
-
|
| 28 |
-
|
| 29 |
-
|
|
|
|
| 30 |
seen: set[str] = set()
|
| 31 |
papers: list[dict] = []
|
| 32 |
-
for
|
| 33 |
-
|
| 34 |
-
|
| 35 |
-
|
| 36 |
-
|
| 37 |
-
|
| 38 |
-
|
| 39 |
-
|
| 40 |
return papers
|
|
|
|
| 2 |
from __future__ import annotations
|
| 3 |
|
| 4 |
import asyncio
|
| 5 |
+
from collections.abc import Mapping
|
| 6 |
from itertools import zip_longest
|
| 7 |
|
| 8 |
from app import local_meta, turso_svc
|
| 9 |
|
| 10 |
+
# Every third starter slot within an interest goes to a paper from the corpus's
|
| 11 |
+
# last three months. Ranked by citations alone, those can never compete: a
|
| 12 |
+
# paper a few weeks old has almost none yet.
|
| 13 |
+
_FRESH_EVERY = 3
|
| 14 |
|
|
|
|
|
|
|
| 15 |
|
| 16 |
+
def _interleave(lists: list[list[dict]]) -> list[dict]:
|
| 17 |
+
return [p for row in zip_longest(*lists) for p in row if p]
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
def _with_fresh(established: list[dict], fresh: list[dict]) -> list[dict]:
|
| 21 |
+
out, e, f = [], iter(established), iter(fresh)
|
| 22 |
+
while True:
|
| 23 |
+
batch = [p for p in (next(e, None) for _ in range(_FRESH_EVERY - 1)) if p]
|
| 24 |
+
batch += [p for p in [next(f, None)] if p]
|
| 25 |
+
if not batch:
|
| 26 |
+
return out + list(e) + list(f)
|
| 27 |
+
out += batch
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
async def starter_papers(
|
| 31 |
+
categories: set[str] | Mapping[str, set[str]], limit: int = 30,
|
| 32 |
+
) -> list[dict]:
|
| 33 |
+
"""Deal starter candidates out evenly across the reader's interests.
|
| 34 |
+
|
| 35 |
+
`categories` is either a mapping of interest -> arXiv codes (what onboarding
|
| 36 |
+
stores) or a flat set of codes, which is treated as one interest per code.
|
| 37 |
+
Balancing per interest rather than per code matters: "Machine Learning"
|
| 38 |
+
spans cs.LG and stat.ML while "Robotics" is only cs.RO, and dealing per code
|
| 39 |
+
gave the two-code interests twice the share. Measured 2026-09-30 for six
|
| 40 |
+
interests, Computer Vision got 22 of 200 starter papers and language-model
|
| 41 |
+
papers 86.
|
| 42 |
+
|
| 43 |
+
Within an interest, codes are round-robined and every third slot is a
|
| 44 |
+
recent paper (see _FRESH_EVERY). These are citation/popularity candidates,
|
| 45 |
+
not measured live trends. Without the sidecar, one cached remote query
|
| 46 |
+
serves everything and there is no fresh lane.
|
| 47 |
"""
|
| 48 |
+
groups = ({k: set(v) for k, v in categories.items() if v}
|
| 49 |
+
if isinstance(categories, Mapping)
|
| 50 |
+
else {code: {code} for code in categories})
|
| 51 |
+
if not groups or limit <= 0:
|
| 52 |
return []
|
| 53 |
+
all_codes = set().union(*groups.values())
|
| 54 |
+
if not local_meta.is_available():
|
| 55 |
+
return await turso_svc.fetch_trending_by_categories(all_codes, limit=limit)
|
| 56 |
semaphore = asyncio.Semaphore(4)
|
| 57 |
|
| 58 |
+
async def guarded(coro):
|
| 59 |
async with semaphore:
|
| 60 |
+
try:
|
| 61 |
+
return await coro
|
| 62 |
+
except Exception as e: # one bad lane must not empty the feed
|
| 63 |
+
print(f"[discovery] starter lane failed: {e}")
|
| 64 |
+
return []
|
| 65 |
+
|
| 66 |
+
codes = sorted(all_codes)
|
| 67 |
+
names = sorted(groups)
|
| 68 |
+
results = await asyncio.gather(
|
| 69 |
+
*(guarded(turso_svc.fetch_trending_by_categories({c}, limit=limit)) for c in codes),
|
| 70 |
+
*(guarded(turso_svc.fetch_fresh_by_categories(groups[g], limit=limit)) for g in names),
|
| 71 |
+
)
|
| 72 |
+
trending = dict(zip(codes, results[:len(codes)]))
|
| 73 |
+
fresh = dict(zip(names, results[len(codes):]))
|
| 74 |
|
| 75 |
+
per_group = [
|
| 76 |
+
_with_fresh(_interleave([trending[c] for c in sorted(groups[g])]), fresh[g])
|
| 77 |
+
for g in names
|
| 78 |
+
]
|
| 79 |
seen: set[str] = set()
|
| 80 |
papers: list[dict] = []
|
| 81 |
+
for paper in _interleave(per_group):
|
| 82 |
+
aid = paper.get("arxiv_id")
|
| 83 |
+
if not aid or aid in seen:
|
| 84 |
+
continue
|
| 85 |
+
seen.add(aid)
|
| 86 |
+
papers.append(paper)
|
| 87 |
+
if len(papers) == limit:
|
| 88 |
+
break
|
| 89 |
return papers
|
|
@@ -209,26 +209,61 @@ def newest_update_date() -> str | None:
|
|
| 209 |
return _max_date
|
| 210 |
|
| 211 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 212 |
def fetch_trending(
|
| 213 |
codes: set[str],
|
| 214 |
limit: int = 10,
|
| 215 |
recency_months: int = 24,
|
| 216 |
) -> list[dict]:
|
| 217 |
"""
|
| 218 |
-
|
| 219 |
|
| 220 |
Ordering by all-time citations returns the same canonical papers to every
|
| 221 |
user forever — for cs.LG that is Adam (2014), scikit-learn (2011) and
|
| 222 |
BatchNorm (2015). Those are famous, not trending, and this feeds Tier 0,
|
| 223 |
-
which is the very first thing a new user sees.
|
| 224 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 225 |
|
| 226 |
Two details that matter:
|
| 227 |
|
| 228 |
* The window is measured back from the newest paper in the corpus, not
|
| 229 |
-
from today
|
| 230 |
-
absolute cutoff would silently empty out; anchoring to the data means
|
| 231 |
-
this keeps working if ingestion is added later.
|
| 232 |
* Categories are wildly uneven (~302k papers in cs.LG vs ~7.9k in
|
| 233 |
q-bio.NC), so a fixed window starves the thin ones. The window widens
|
| 234 |
and finally drops away entirely rather than returning a short list.
|
|
@@ -239,6 +274,25 @@ def fetch_trending(
|
|
| 239 |
if conn is None:
|
| 240 |
return []
|
| 241 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 242 |
ph = ",".join("?" * len(codes))
|
| 243 |
# Read the most-cited papers in these categories in one index-ordered pass,
|
| 244 |
# then apply the publication-date window in Python. The date cannot be
|
|
@@ -292,3 +346,28 @@ def fetch_trending(
|
|
| 292 |
|
| 293 |
# No window could fill the slots — fall back to plain citation order.
|
| 294 |
return (best or candidates)[:limit]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 209 |
return _max_date
|
| 210 |
|
| 211 |
|
| 212 |
+
def _published_since(conn: sqlite3.Connection, codes: set[str],
|
| 213 |
+
cutoff: tuple[int, int]) -> list[tuple[str, int, tuple[int, int]]]:
|
| 214 |
+
"""(arxiv_id, citations, publication month) for papers in `codes` published
|
| 215 |
+
at or after `cutoff`.
|
| 216 |
+
|
| 217 |
+
Filters on the (code, update_date) index first. update_date is the last
|
| 218 |
+
revision, which is never earlier than publication, so it is a superset of
|
| 219 |
+
the window; the identifier then gives the true publication month. Only ids
|
| 220 |
+
and counts are read here, so a 24-month cs.LG window stays cheap.
|
| 221 |
+
"""
|
| 222 |
+
ph = ",".join("?" * len(codes))
|
| 223 |
+
rows = conn.execute(
|
| 224 |
+
f"""SELECT arxiv_id, MAX(citation_count) FROM paper_categories
|
| 225 |
+
WHERE code IN ({ph}) AND update_date >= ?
|
| 226 |
+
GROUP BY arxiv_id""",
|
| 227 |
+
(*codes, f"{cutoff[0]:04d}-{cutoff[1]:02d}-01"),
|
| 228 |
+
).fetchall()
|
| 229 |
+
out = []
|
| 230 |
+
for aid, cit in rows:
|
| 231 |
+
ym = pub_year_month(aid)
|
| 232 |
+
if ym is not None and ym >= cutoff:
|
| 233 |
+
out.append((aid, cit or 0, ym))
|
| 234 |
+
return out
|
| 235 |
+
|
| 236 |
+
|
| 237 |
+
def _rows_in_order(conn: sqlite3.Connection, ids: list[str]) -> list[dict]:
|
| 238 |
+
got = {r["arxiv_id"]: r for r in fetch_rows(ids)}
|
| 239 |
+
return [got[i] for i in ids if i in got]
|
| 240 |
+
|
| 241 |
+
|
| 242 |
def fetch_trending(
|
| 243 |
codes: set[str],
|
| 244 |
limit: int = 10,
|
| 245 |
recency_months: int = 24,
|
| 246 |
) -> list[dict]:
|
| 247 |
"""
|
| 248 |
+
Papers in any of `codes` that are gathering citations fastest.
|
| 249 |
|
| 250 |
Ordering by all-time citations returns the same canonical papers to every
|
| 251 |
user forever — for cs.LG that is Adam (2014), scikit-learn (2011) and
|
| 252 |
BatchNorm (2015). Those are famous, not trending, and this feeds Tier 0,
|
| 253 |
+
which is the very first thing a new user sees.
|
| 254 |
+
|
| 255 |
+
Within the window, papers are ranked by citations per month since
|
| 256 |
+
publication, not by raw totals. Raw totals inside a 24-month window still
|
| 257 |
+
hand the top slots to the oldest papers in it, which have had two years to
|
| 258 |
+
accumulate; measured 2026-09-30, the starter pool for six ML interests held
|
| 259 |
+
147 papers from 2024 and none from 2026. The rate is floored at three
|
| 260 |
+
months of age so a week-old paper with two citations does not outrank
|
| 261 |
+
everything.
|
| 262 |
|
| 263 |
Two details that matter:
|
| 264 |
|
| 265 |
* The window is measured back from the newest paper in the corpus, not
|
| 266 |
+
from today, so a corpus whose ingestion lags keeps working.
|
|
|
|
|
|
|
| 267 |
* Categories are wildly uneven (~302k papers in cs.LG vs ~7.9k in
|
| 268 |
q-bio.NC), so a fixed window starves the thin ones. The window widens
|
| 269 |
and finally drops away entirely rather than returning a short list.
|
|
|
|
| 274 |
if conn is None:
|
| 275 |
return []
|
| 276 |
|
| 277 |
+
anchor = newest_update_date()
|
| 278 |
+
now = _months_before(anchor, 0) if anchor else None
|
| 279 |
+
if now is not None:
|
| 280 |
+
try:
|
| 281 |
+
for months in (recency_months, recency_months * 2, recency_months * 4):
|
| 282 |
+
cutoff = _months_before(anchor, months)
|
| 283 |
+
if cutoff is None:
|
| 284 |
+
break
|
| 285 |
+
window = _published_since(conn, codes, cutoff)
|
| 286 |
+
if len(window) < limit:
|
| 287 |
+
continue
|
| 288 |
+
age = lambda ym: (now[0] - ym[0]) * 12 + (now[1] - ym[1]) + 1
|
| 289 |
+
window.sort(key=lambda r: (-r[1] / max(age(r[2]), 3), -r[1], r[0]))
|
| 290 |
+
picked = _rows_in_order(conn, [r[0] for r in window[:limit]])
|
| 291 |
+
if len(picked) >= limit:
|
| 292 |
+
return picked
|
| 293 |
+
except Exception as e:
|
| 294 |
+
print(f"[local_meta] trending window failed ({e}) — using citation order")
|
| 295 |
+
|
| 296 |
ph = ",".join("?" * len(codes))
|
| 297 |
# Read the most-cited papers in these categories in one index-ordered pass,
|
| 298 |
# then apply the publication-date window in Python. The date cannot be
|
|
|
|
| 346 |
|
| 347 |
# No window could fill the slots — fall back to plain citation order.
|
| 348 |
return (best or candidates)[:limit]
|
| 349 |
+
|
| 350 |
+
|
| 351 |
+
def fetch_fresh(codes: set[str], limit: int = 10, recency_months: int = 3) -> list[dict]:
|
| 352 |
+
"""
|
| 353 |
+
The newest papers in `codes`: published in the last `recency_months` of the
|
| 354 |
+
corpus, most-cited first, newest first among ties.
|
| 355 |
+
|
| 356 |
+
fetch_trending cannot supply these however it ranks: a paper a few weeks
|
| 357 |
+
old has almost no citations yet. This lane exists so a new reader's first
|
| 358 |
+
feed contains the current month of their field at all. Returns [] when the
|
| 359 |
+
sidecar is unavailable; there is no Turso fallback for this lane.
|
| 360 |
+
"""
|
| 361 |
+
conn = connection() if codes else None
|
| 362 |
+
anchor = newest_update_date() if conn is not None else None
|
| 363 |
+
cutoff = _months_before(anchor, recency_months - 1) if anchor else None
|
| 364 |
+
if cutoff is None:
|
| 365 |
+
return []
|
| 366 |
+
try:
|
| 367 |
+
window = _published_since(conn, codes, cutoff)
|
| 368 |
+
window.sort(key=lambda r: r[0], reverse=True) # newest id first...
|
| 369 |
+
window.sort(key=lambda r: -r[1]) # ...then by citations (stable)
|
| 370 |
+
return _rows_in_order(conn, [r[0] for r in window[:limit]])
|
| 371 |
+
except Exception as e:
|
| 372 |
+
print(f"[local_meta] fresh lane failed ({e})")
|
| 373 |
+
return []
|
|
@@ -163,7 +163,7 @@ async def seed_search(
|
|
| 163 |
except Exception as e:
|
| 164 |
print(f"[onboarding] keyword fallback failed: {e}")
|
| 165 |
else:
|
| 166 |
-
categories = await db.
|
| 167 |
if categories:
|
| 168 |
try:
|
| 169 |
papers = await discovery_svc.starter_papers(categories, limit=12)
|
|
|
|
| 163 |
except Exception as e:
|
| 164 |
print(f"[onboarding] keyword fallback failed: {e}")
|
| 165 |
else:
|
| 166 |
+
categories = await db.get_user_category_groups(user_id)
|
| 167 |
if categories:
|
| 168 |
try:
|
| 169 |
papers = await discovery_svc.starter_papers(categories, limit=12)
|
|
@@ -418,7 +418,9 @@ async def _build_feed(
|
|
| 418 |
|
| 419 |
# ── Tier 0: category trending (cold start, Phase 5) ──────────────────
|
| 420 |
if not state.has_enough_for_recs():
|
| 421 |
-
|
|
|
|
|
|
|
| 422 |
# No categories means the reader skipped onboarding. That used to fall
|
| 423 |
# straight through to the empty state and stay there; a reader who told
|
| 424 |
# us nothing still gets a feed, just a broader one. See
|
|
|
|
| 418 |
|
| 419 |
# ── Tier 0: category trending (cold start, Phase 5) ──────────────────
|
| 420 |
if not state.has_enough_for_recs():
|
| 421 |
+
# Grouped by interest, so the starter pool balances interests rather
|
| 422 |
+
# than arXiv codes (see discovery_svc.starter_papers).
|
| 423 |
+
category_filter = await db.get_user_category_groups(user_id)
|
| 424 |
# No categories means the reader skipped onboarding. That used to fall
|
| 425 |
# straight through to the empty state and stay there; a reader who told
|
| 426 |
# us nothing still gets a feed, just a broader one. See
|
|
@@ -551,3 +551,27 @@ async def fetch_trending_by_categories(
|
|
| 551 |
for p in papers:
|
| 552 |
_cache_put(p["arxiv_id"], p)
|
| 553 |
return papers
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 551 |
for p in papers:
|
| 552 |
_cache_put(p["arxiv_id"], p)
|
| 553 |
return papers
|
| 554 |
+
|
| 555 |
+
|
| 556 |
+
async def fetch_fresh_by_categories(categories: set[str], limit: int = 10) -> list[dict]:
|
| 557 |
+
"""
|
| 558 |
+
Papers published in the corpus's last three months in `categories`, from
|
| 559 |
+
the sidecar only (see local_meta.fetch_fresh). Turso has no index that
|
| 560 |
+
makes this affordable, so without the sidecar the lane is simply empty.
|
| 561 |
+
"""
|
| 562 |
+
if not categories:
|
| 563 |
+
return []
|
| 564 |
+
cache_key = ("fresh", tuple(sorted(categories)), limit)
|
| 565 |
+
cached = _TRENDING_CACHE.get(cache_key)
|
| 566 |
+
if cached is not None and (time.time() - cached[0]) < _TRENDING_TTL_SECONDS:
|
| 567 |
+
return cached[1]
|
| 568 |
+
from app import local_meta
|
| 569 |
+
if not local_meta.is_available():
|
| 570 |
+
return []
|
| 571 |
+
rows = await asyncio.to_thread(local_meta.fetch_fresh, set(categories), limit)
|
| 572 |
+
papers = [p for p in (_to_paper_dict(r) for r in rows) if p]
|
| 573 |
+
if papers:
|
| 574 |
+
_TRENDING_CACHE[cache_key] = (time.time(), papers)
|
| 575 |
+
for p in papers:
|
| 576 |
+
_cache_put(p["arxiv_id"], p)
|
| 577 |
+
return papers
|
|
@@ -56,7 +56,7 @@ async def test_starter_suggestions_use_selected_categories_and_mark_saved(client
|
|
| 56 |
'category':'cs.CL','year':2026}])
|
| 57 |
monkeypatch.setattr(discovery_svc, 'starter_papers', starter)
|
| 58 |
r = await client.get('/api/onboarding/seed-search')
|
| 59 |
-
starter.assert_awaited_once_with({'cs.CL','cs.IR'}, limit=12)
|
| 60 |
assert 'An NLP seed' in r.text and 'Saved' in r.text
|
| 61 |
assert 'hx-post=' not in r.text
|
| 62 |
|
|
@@ -167,3 +167,40 @@ async def test_cookieless_paper_visit_records_nothing(monkeypatch):
|
|
| 167 |
async with aiosqlite.connect(db.DB_PATH) as conn:
|
| 168 |
n = (await (await conn.execute("SELECT COUNT(*) FROM interactions")).fetchone())[0]
|
| 169 |
assert n == 0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 56 |
'category':'cs.CL','year':2026}])
|
| 57 |
monkeypatch.setattr(discovery_svc, 'starter_papers', starter)
|
| 58 |
r = await client.get('/api/onboarding/seed-search')
|
| 59 |
+
starter.assert_awaited_once_with({'nlp': {'cs.CL','cs.IR'}}, limit=12)
|
| 60 |
assert 'An NLP seed' in r.text and 'Saved' in r.text
|
| 61 |
assert 'hx-post=' not in r.text
|
| 62 |
|
|
|
|
| 167 |
async with aiosqlite.connect(db.DB_PATH) as conn:
|
| 168 |
n = (await (await conn.execute("SELECT COUNT(*) FROM interactions")).fetchone())[0]
|
| 169 |
assert n == 0
|
| 170 |
+
|
| 171 |
+
|
| 172 |
+
def _lanes(monkeypatch, trending_per_code, fresh_per_group):
|
| 173 |
+
monkeypatch.setattr(discovery_svc.local_meta, 'is_available', lambda: True)
|
| 174 |
+
|
| 175 |
+
async def trending(cats, limit):
|
| 176 |
+
(code,) = cats
|
| 177 |
+
return [{'arxiv_id': f'{code}:{i}'} for i in range(trending_per_code.get(code, 0))]
|
| 178 |
+
|
| 179 |
+
async def fresh(cats, limit):
|
| 180 |
+
key = '+'.join(sorted(cats))
|
| 181 |
+
return [{'arxiv_id': f'fresh:{key}:{i}'} for i in range(fresh_per_group.get(key, 0))]
|
| 182 |
+
|
| 183 |
+
monkeypatch.setattr(turso_svc, 'fetch_trending_by_categories', trending)
|
| 184 |
+
monkeypatch.setattr(turso_svc, 'fetch_fresh_by_categories', fresh)
|
| 185 |
+
|
| 186 |
+
|
| 187 |
+
async def test_starters_balance_interests_not_arxiv_codes(monkeypatch):
|
| 188 |
+
"""ML spans two codes and robotics one; each interest still gets half."""
|
| 189 |
+
_lanes(monkeypatch, {'cs.LG': 50, 'stat.ML': 50, 'cs.RO': 50}, {})
|
| 190 |
+
papers = await discovery_svc.starter_papers(
|
| 191 |
+
{'ml': {'cs.LG', 'stat.ML'}, 'robotics': {'cs.RO'}}, limit=20)
|
| 192 |
+
ids = [p['arxiv_id'] for p in papers]
|
| 193 |
+
assert sum(a.startswith('cs.RO') for a in ids) == 10
|
| 194 |
+
assert sum(a.startswith(('cs.LG', 'stat.ML')) for a in ids) == 10
|
| 195 |
+
|
| 196 |
+
|
| 197 |
+
async def test_every_third_starter_slot_per_interest_is_recent(monkeypatch):
|
| 198 |
+
_lanes(monkeypatch, {'cs.RO': 50}, {'cs.RO': 50})
|
| 199 |
+
papers = await discovery_svc.starter_papers({'robotics': {'cs.RO'}}, limit=9)
|
| 200 |
+
assert [p['arxiv_id'].startswith('fresh:') for p in papers] == [False, False, True] * 3
|
| 201 |
+
|
| 202 |
+
|
| 203 |
+
async def test_starters_fall_back_to_established_when_fresh_lane_is_empty(monkeypatch):
|
| 204 |
+
_lanes(monkeypatch, {'cs.RO': 50}, {})
|
| 205 |
+
papers = await discovery_svc.starter_papers({'robotics': {'cs.RO'}}, limit=9)
|
| 206 |
+
assert len(papers) == 9 and not any(p['arxiv_id'].startswith('fresh:') for p in papers)
|
|
@@ -136,3 +136,12 @@ def test_expand_category_groups_unknown_key():
|
|
| 136 |
assert "cs.CL" in result
|
| 137 |
# unknown key produced nothing extra
|
| 138 |
assert len(result) == 2 # cs.CL + cs.IR
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 136 |
assert "cs.CL" in result
|
| 137 |
# unknown key produced nothing extra
|
| 138 |
assert len(result) == 2 # cs.CL + cs.IR
|
| 139 |
+
|
| 140 |
+
|
| 141 |
+
async def test_get_user_category_groups_keeps_interests_apart(tmp_db):
|
| 142 |
+
import app.db as db
|
| 143 |
+
await db.init_db()
|
| 144 |
+
await db.save_onboarding_categories("u1", ["ml", "robotics"])
|
| 145 |
+
groups = await db.get_user_category_groups("u1")
|
| 146 |
+
assert groups == {"ml": {"cs.LG", "stat.ML"}, "robotics": {"cs.RO"}}
|
| 147 |
+
assert await db.get_user_category_groups("nobody") == {}
|