siddhm11 Claude Opus 5.5 commited on
Commit
f9dfa80
·
1 Parent(s): 9622e3a

feat(discovery): balance the starter feed per interest and give it a recent lane

Browse files

Measured 2026-09-30 for a reader who ticked NLP, CV, ML, AI, Robotics and
Software Engineering, the 200-paper starter pool held 147 papers from 2024,
53 from early 2025 and none from 2026, and Computer Vision got 22 of 200.

- Deal per ticked interest, not per arXiv code. Two-code interests (ML is
cs.LG + stat.ML) were getting twice the share of one-code ones.
- Rank the established lane by citations per month since publication
(floored at three months). Raw totals inside a 24-month window always
favour the oldest papers in it.
- Every third slot per interest comes from a new fresh lane: papers from
the corpus's last three months, most-cited first. A few-weeks-old paper
cannot win on citations however they are ranked.

Onboarding suggestions use the same grouping.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

app/db.py CHANGED
@@ -633,6 +633,20 @@ async def get_user_category_filter(user_id: str) -> set[str]:
633
  return expand_category_groups(state["selected_categories"])
634
 
635
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
636
  # ── Phase 6.5 B3: Cluster snapshot versioning ─────────────────────────────────
637
 
638
  async def save_cluster_snapshot(user_id: str, clusters: list[dict]) -> str:
 
633
  return expand_category_groups(state["selected_categories"])
634
 
635
 
636
+ async def get_user_category_groups(user_id: str) -> dict[str, set[str]]:
637
+ """The reader's selected interests, each mapped to its arXiv codes.
638
+
639
+ Same source as get_user_category_filter, without flattening: starter feeds
640
+ balance per interest, and flattening loses which codes belong together.
641
+ """
642
+ state = await get_onboarding_state(user_id)
643
+ if state is None:
644
+ return {}
645
+ from app.config import expand_category_groups
646
+ return {key: codes for key in state["selected_categories"]
647
+ if (codes := expand_category_groups([key]))}
648
+
649
+
650
  # ── Phase 6.5 B3: Cluster snapshot versioning ─────────────────────────────────
651
 
652
  async def save_cluster_snapshot(user_id: str, clusters: list[dict]) -> str:
app/discovery_svc.py CHANGED
@@ -2,39 +2,88 @@
2
  from __future__ import annotations
3
 
4
  import asyncio
 
5
  from itertools import zip_longest
6
 
7
  from app import local_meta, turso_svc
8
 
 
 
 
 
9
 
10
- async def starter_papers(categories: set[str], limit: int = 30) -> list[dict]:
11
- """Give thin categories room without eight remote full-table scans.
12
 
13
- With the indexed sidecar, query each category and round-robin its ranked
14
- results. Without it, retain the existing single cached remote fallback.
15
- These are citation/popularity candidates, not measured live trends.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
16
  """
17
- if not categories or limit <= 0:
 
 
 
18
  return []
19
- if len(categories) == 1 or not local_meta.is_available():
20
- return await turso_svc.fetch_trending_by_categories(categories, limit=limit)
 
21
  semaphore = asyncio.Semaphore(4)
22
 
23
- async def one(code: str) -> list[dict]:
24
  async with semaphore:
25
- return await turso_svc.fetch_trending_by_categories({code}, limit=limit)
 
 
 
 
 
 
 
 
 
 
 
 
 
26
 
27
- results = await asyncio.gather(*(one(code) for code in sorted(categories)),
28
- return_exceptions=True)
29
- pools = [r for r in results if isinstance(r, list)]
 
30
  seen: set[str] = set()
31
  papers: list[dict] = []
32
- for row in zip_longest(*pools):
33
- for paper in row:
34
- if not paper or not paper.get("arxiv_id") or paper["arxiv_id"] in seen:
35
- continue
36
- seen.add(paper["arxiv_id"])
37
- papers.append(paper)
38
- if len(papers) == limit:
39
- return papers
40
  return papers
 
2
  from __future__ import annotations
3
 
4
  import asyncio
5
+ from collections.abc import Mapping
6
  from itertools import zip_longest
7
 
8
  from app import local_meta, turso_svc
9
 
10
+ # Every third starter slot within an interest goes to a paper from the corpus's
11
+ # last three months. Ranked by citations alone, those can never compete: a
12
+ # paper a few weeks old has almost none yet.
13
+ _FRESH_EVERY = 3
14
 
 
 
15
 
16
+ def _interleave(lists: list[list[dict]]) -> list[dict]:
17
+ return [p for row in zip_longest(*lists) for p in row if p]
18
+
19
+
20
+ def _with_fresh(established: list[dict], fresh: list[dict]) -> list[dict]:
21
+ out, e, f = [], iter(established), iter(fresh)
22
+ while True:
23
+ batch = [p for p in (next(e, None) for _ in range(_FRESH_EVERY - 1)) if p]
24
+ batch += [p for p in [next(f, None)] if p]
25
+ if not batch:
26
+ return out + list(e) + list(f)
27
+ out += batch
28
+
29
+
30
+ async def starter_papers(
31
+ categories: set[str] | Mapping[str, set[str]], limit: int = 30,
32
+ ) -> list[dict]:
33
+ """Deal starter candidates out evenly across the reader's interests.
34
+
35
+ `categories` is either a mapping of interest -> arXiv codes (what onboarding
36
+ stores) or a flat set of codes, which is treated as one interest per code.
37
+ Balancing per interest rather than per code matters: "Machine Learning"
38
+ spans cs.LG and stat.ML while "Robotics" is only cs.RO, and dealing per code
39
+ gave the two-code interests twice the share. Measured 2026-09-30 for six
40
+ interests, Computer Vision got 22 of 200 starter papers and language-model
41
+ papers 86.
42
+
43
+ Within an interest, codes are round-robined and every third slot is a
44
+ recent paper (see _FRESH_EVERY). These are citation/popularity candidates,
45
+ not measured live trends. Without the sidecar, one cached remote query
46
+ serves everything and there is no fresh lane.
47
  """
48
+ groups = ({k: set(v) for k, v in categories.items() if v}
49
+ if isinstance(categories, Mapping)
50
+ else {code: {code} for code in categories})
51
+ if not groups or limit <= 0:
52
  return []
53
+ all_codes = set().union(*groups.values())
54
+ if not local_meta.is_available():
55
+ return await turso_svc.fetch_trending_by_categories(all_codes, limit=limit)
56
  semaphore = asyncio.Semaphore(4)
57
 
58
+ async def guarded(coro):
59
  async with semaphore:
60
+ try:
61
+ return await coro
62
+ except Exception as e: # one bad lane must not empty the feed
63
+ print(f"[discovery] starter lane failed: {e}")
64
+ return []
65
+
66
+ codes = sorted(all_codes)
67
+ names = sorted(groups)
68
+ results = await asyncio.gather(
69
+ *(guarded(turso_svc.fetch_trending_by_categories({c}, limit=limit)) for c in codes),
70
+ *(guarded(turso_svc.fetch_fresh_by_categories(groups[g], limit=limit)) for g in names),
71
+ )
72
+ trending = dict(zip(codes, results[:len(codes)]))
73
+ fresh = dict(zip(names, results[len(codes):]))
74
 
75
+ per_group = [
76
+ _with_fresh(_interleave([trending[c] for c in sorted(groups[g])]), fresh[g])
77
+ for g in names
78
+ ]
79
  seen: set[str] = set()
80
  papers: list[dict] = []
81
+ for paper in _interleave(per_group):
82
+ aid = paper.get("arxiv_id")
83
+ if not aid or aid in seen:
84
+ continue
85
+ seen.add(aid)
86
+ papers.append(paper)
87
+ if len(papers) == limit:
88
+ break
89
  return papers
app/local_meta.py CHANGED
@@ -209,26 +209,61 @@ def newest_update_date() -> str | None:
209
  return _max_date
210
 
211
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
212
  def fetch_trending(
213
  codes: set[str],
214
  limit: int = 10,
215
  recency_months: int = 24,
216
  ) -> list[dict]:
217
  """
218
- Well-cited *recent* papers in any of `codes`.
219
 
220
  Ordering by all-time citations returns the same canonical papers to every
221
  user forever — for cs.LG that is Adam (2014), scikit-learn (2011) and
222
  BatchNorm (2015). Those are famous, not trending, and this feeds Tier 0,
223
- which is the very first thing a new user sees. Restricting to a recent
224
- window over the same data returns Llama 3, DPO and DeepSeek-R1 instead.
 
 
 
 
 
 
 
225
 
226
  Two details that matter:
227
 
228
  * The window is measured back from the newest paper in the corpus, not
229
- from today. The corpus is a static snapshot ending 2025-05-30, so an
230
- absolute cutoff would silently empty out; anchoring to the data means
231
- this keeps working if ingestion is added later.
232
  * Categories are wildly uneven (~302k papers in cs.LG vs ~7.9k in
233
  q-bio.NC), so a fixed window starves the thin ones. The window widens
234
  and finally drops away entirely rather than returning a short list.
@@ -239,6 +274,25 @@ def fetch_trending(
239
  if conn is None:
240
  return []
241
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
242
  ph = ",".join("?" * len(codes))
243
  # Read the most-cited papers in these categories in one index-ordered pass,
244
  # then apply the publication-date window in Python. The date cannot be
@@ -292,3 +346,28 @@ def fetch_trending(
292
 
293
  # No window could fill the slots — fall back to plain citation order.
294
  return (best or candidates)[:limit]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
209
  return _max_date
210
 
211
 
212
+ def _published_since(conn: sqlite3.Connection, codes: set[str],
213
+ cutoff: tuple[int, int]) -> list[tuple[str, int, tuple[int, int]]]:
214
+ """(arxiv_id, citations, publication month) for papers in `codes` published
215
+ at or after `cutoff`.
216
+
217
+ Filters on the (code, update_date) index first. update_date is the last
218
+ revision, which is never earlier than publication, so it is a superset of
219
+ the window; the identifier then gives the true publication month. Only ids
220
+ and counts are read here, so a 24-month cs.LG window stays cheap.
221
+ """
222
+ ph = ",".join("?" * len(codes))
223
+ rows = conn.execute(
224
+ f"""SELECT arxiv_id, MAX(citation_count) FROM paper_categories
225
+ WHERE code IN ({ph}) AND update_date >= ?
226
+ GROUP BY arxiv_id""",
227
+ (*codes, f"{cutoff[0]:04d}-{cutoff[1]:02d}-01"),
228
+ ).fetchall()
229
+ out = []
230
+ for aid, cit in rows:
231
+ ym = pub_year_month(aid)
232
+ if ym is not None and ym >= cutoff:
233
+ out.append((aid, cit or 0, ym))
234
+ return out
235
+
236
+
237
+ def _rows_in_order(conn: sqlite3.Connection, ids: list[str]) -> list[dict]:
238
+ got = {r["arxiv_id"]: r for r in fetch_rows(ids)}
239
+ return [got[i] for i in ids if i in got]
240
+
241
+
242
  def fetch_trending(
243
  codes: set[str],
244
  limit: int = 10,
245
  recency_months: int = 24,
246
  ) -> list[dict]:
247
  """
248
+ Papers in any of `codes` that are gathering citations fastest.
249
 
250
  Ordering by all-time citations returns the same canonical papers to every
251
  user forever — for cs.LG that is Adam (2014), scikit-learn (2011) and
252
  BatchNorm (2015). Those are famous, not trending, and this feeds Tier 0,
253
+ which is the very first thing a new user sees.
254
+
255
+ Within the window, papers are ranked by citations per month since
256
+ publication, not by raw totals. Raw totals inside a 24-month window still
257
+ hand the top slots to the oldest papers in it, which have had two years to
258
+ accumulate; measured 2026-09-30, the starter pool for six ML interests held
259
+ 147 papers from 2024 and none from 2026. The rate is floored at three
260
+ months of age so a week-old paper with two citations does not outrank
261
+ everything.
262
 
263
  Two details that matter:
264
 
265
  * The window is measured back from the newest paper in the corpus, not
266
+ from today, so a corpus whose ingestion lags keeps working.
 
 
267
  * Categories are wildly uneven (~302k papers in cs.LG vs ~7.9k in
268
  q-bio.NC), so a fixed window starves the thin ones. The window widens
269
  and finally drops away entirely rather than returning a short list.
 
274
  if conn is None:
275
  return []
276
 
277
+ anchor = newest_update_date()
278
+ now = _months_before(anchor, 0) if anchor else None
279
+ if now is not None:
280
+ try:
281
+ for months in (recency_months, recency_months * 2, recency_months * 4):
282
+ cutoff = _months_before(anchor, months)
283
+ if cutoff is None:
284
+ break
285
+ window = _published_since(conn, codes, cutoff)
286
+ if len(window) < limit:
287
+ continue
288
+ age = lambda ym: (now[0] - ym[0]) * 12 + (now[1] - ym[1]) + 1
289
+ window.sort(key=lambda r: (-r[1] / max(age(r[2]), 3), -r[1], r[0]))
290
+ picked = _rows_in_order(conn, [r[0] for r in window[:limit]])
291
+ if len(picked) >= limit:
292
+ return picked
293
+ except Exception as e:
294
+ print(f"[local_meta] trending window failed ({e}) — using citation order")
295
+
296
  ph = ",".join("?" * len(codes))
297
  # Read the most-cited papers in these categories in one index-ordered pass,
298
  # then apply the publication-date window in Python. The date cannot be
 
346
 
347
  # No window could fill the slots — fall back to plain citation order.
348
  return (best or candidates)[:limit]
349
+
350
+
351
+ def fetch_fresh(codes: set[str], limit: int = 10, recency_months: int = 3) -> list[dict]:
352
+ """
353
+ The newest papers in `codes`: published in the last `recency_months` of the
354
+ corpus, most-cited first, newest first among ties.
355
+
356
+ fetch_trending cannot supply these however it ranks: a paper a few weeks
357
+ old has almost no citations yet. This lane exists so a new reader's first
358
+ feed contains the current month of their field at all. Returns [] when the
359
+ sidecar is unavailable; there is no Turso fallback for this lane.
360
+ """
361
+ conn = connection() if codes else None
362
+ anchor = newest_update_date() if conn is not None else None
363
+ cutoff = _months_before(anchor, recency_months - 1) if anchor else None
364
+ if cutoff is None:
365
+ return []
366
+ try:
367
+ window = _published_since(conn, codes, cutoff)
368
+ window.sort(key=lambda r: r[0], reverse=True) # newest id first...
369
+ window.sort(key=lambda r: -r[1]) # ...then by citations (stable)
370
+ return _rows_in_order(conn, [r[0] for r in window[:limit]])
371
+ except Exception as e:
372
+ print(f"[local_meta] fresh lane failed ({e})")
373
+ return []
app/routers/onboarding.py CHANGED
@@ -163,7 +163,7 @@ async def seed_search(
163
  except Exception as e:
164
  print(f"[onboarding] keyword fallback failed: {e}")
165
  else:
166
- categories = await db.get_user_category_filter(user_id)
167
  if categories:
168
  try:
169
  papers = await discovery_svc.starter_papers(categories, limit=12)
 
163
  except Exception as e:
164
  print(f"[onboarding] keyword fallback failed: {e}")
165
  else:
166
+ categories = await db.get_user_category_groups(user_id)
167
  if categories:
168
  try:
169
  papers = await discovery_svc.starter_papers(categories, limit=12)
app/routers/recommendations.py CHANGED
@@ -418,7 +418,9 @@ async def _build_feed(
418
 
419
  # ── Tier 0: category trending (cold start, Phase 5) ──────────────────
420
  if not state.has_enough_for_recs():
421
- category_filter = await db.get_user_category_filter(user_id)
 
 
422
  # No categories means the reader skipped onboarding. That used to fall
423
  # straight through to the empty state and stay there; a reader who told
424
  # us nothing still gets a feed, just a broader one. See
 
418
 
419
  # ── Tier 0: category trending (cold start, Phase 5) ──────────────────
420
  if not state.has_enough_for_recs():
421
+ # Grouped by interest, so the starter pool balances interests rather
422
+ # than arXiv codes (see discovery_svc.starter_papers).
423
+ category_filter = await db.get_user_category_groups(user_id)
424
  # No categories means the reader skipped onboarding. That used to fall
425
  # straight through to the empty state and stay there; a reader who told
426
  # us nothing still gets a feed, just a broader one. See
app/turso_svc.py CHANGED
@@ -551,3 +551,27 @@ async def fetch_trending_by_categories(
551
  for p in papers:
552
  _cache_put(p["arxiv_id"], p)
553
  return papers
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
551
  for p in papers:
552
  _cache_put(p["arxiv_id"], p)
553
  return papers
554
+
555
+
556
+ async def fetch_fresh_by_categories(categories: set[str], limit: int = 10) -> list[dict]:
557
+ """
558
+ Papers published in the corpus's last three months in `categories`, from
559
+ the sidecar only (see local_meta.fetch_fresh). Turso has no index that
560
+ makes this affordable, so without the sidecar the lane is simply empty.
561
+ """
562
+ if not categories:
563
+ return []
564
+ cache_key = ("fresh", tuple(sorted(categories)), limit)
565
+ cached = _TRENDING_CACHE.get(cache_key)
566
+ if cached is not None and (time.time() - cached[0]) < _TRENDING_TTL_SECONDS:
567
+ return cached[1]
568
+ from app import local_meta
569
+ if not local_meta.is_available():
570
+ return []
571
+ rows = await asyncio.to_thread(local_meta.fetch_fresh, set(categories), limit)
572
+ papers = [p for p in (_to_paper_dict(r) for r in rows) if p]
573
+ if papers:
574
+ _TRENDING_CACHE[cache_key] = (time.time(), papers)
575
+ for p in papers:
576
+ _cache_put(p["arxiv_id"], p)
577
+ return papers
tests/test_discovery_journey.py CHANGED
@@ -56,7 +56,7 @@ async def test_starter_suggestions_use_selected_categories_and_mark_saved(client
56
  'category':'cs.CL','year':2026}])
57
  monkeypatch.setattr(discovery_svc, 'starter_papers', starter)
58
  r = await client.get('/api/onboarding/seed-search')
59
- starter.assert_awaited_once_with({'cs.CL','cs.IR'}, limit=12)
60
  assert 'An NLP seed' in r.text and 'Saved' in r.text
61
  assert 'hx-post=' not in r.text
62
 
@@ -167,3 +167,40 @@ async def test_cookieless_paper_visit_records_nothing(monkeypatch):
167
  async with aiosqlite.connect(db.DB_PATH) as conn:
168
  n = (await (await conn.execute("SELECT COUNT(*) FROM interactions")).fetchone())[0]
169
  assert n == 0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
56
  'category':'cs.CL','year':2026}])
57
  monkeypatch.setattr(discovery_svc, 'starter_papers', starter)
58
  r = await client.get('/api/onboarding/seed-search')
59
+ starter.assert_awaited_once_with({'nlp': {'cs.CL','cs.IR'}}, limit=12)
60
  assert 'An NLP seed' in r.text and 'Saved' in r.text
61
  assert 'hx-post=' not in r.text
62
 
 
167
  async with aiosqlite.connect(db.DB_PATH) as conn:
168
  n = (await (await conn.execute("SELECT COUNT(*) FROM interactions")).fetchone())[0]
169
  assert n == 0
170
+
171
+
172
+ def _lanes(monkeypatch, trending_per_code, fresh_per_group):
173
+ monkeypatch.setattr(discovery_svc.local_meta, 'is_available', lambda: True)
174
+
175
+ async def trending(cats, limit):
176
+ (code,) = cats
177
+ return [{'arxiv_id': f'{code}:{i}'} for i in range(trending_per_code.get(code, 0))]
178
+
179
+ async def fresh(cats, limit):
180
+ key = '+'.join(sorted(cats))
181
+ return [{'arxiv_id': f'fresh:{key}:{i}'} for i in range(fresh_per_group.get(key, 0))]
182
+
183
+ monkeypatch.setattr(turso_svc, 'fetch_trending_by_categories', trending)
184
+ monkeypatch.setattr(turso_svc, 'fetch_fresh_by_categories', fresh)
185
+
186
+
187
+ async def test_starters_balance_interests_not_arxiv_codes(monkeypatch):
188
+ """ML spans two codes and robotics one; each interest still gets half."""
189
+ _lanes(monkeypatch, {'cs.LG': 50, 'stat.ML': 50, 'cs.RO': 50}, {})
190
+ papers = await discovery_svc.starter_papers(
191
+ {'ml': {'cs.LG', 'stat.ML'}, 'robotics': {'cs.RO'}}, limit=20)
192
+ ids = [p['arxiv_id'] for p in papers]
193
+ assert sum(a.startswith('cs.RO') for a in ids) == 10
194
+ assert sum(a.startswith(('cs.LG', 'stat.ML')) for a in ids) == 10
195
+
196
+
197
+ async def test_every_third_starter_slot_per_interest_is_recent(monkeypatch):
198
+ _lanes(monkeypatch, {'cs.RO': 50}, {'cs.RO': 50})
199
+ papers = await discovery_svc.starter_papers({'robotics': {'cs.RO'}}, limit=9)
200
+ assert [p['arxiv_id'].startswith('fresh:') for p in papers] == [False, False, True] * 3
201
+
202
+
203
+ async def test_starters_fall_back_to_established_when_fresh_lane_is_empty(monkeypatch):
204
+ _lanes(monkeypatch, {'cs.RO': 50}, {})
205
+ papers = await discovery_svc.starter_papers({'robotics': {'cs.RO'}}, limit=9)
206
+ assert len(papers) == 9 and not any(p['arxiv_id'].startswith('fresh:') for p in papers)
tests/test_onboarding.py CHANGED
@@ -136,3 +136,12 @@ def test_expand_category_groups_unknown_key():
136
  assert "cs.CL" in result
137
  # unknown key produced nothing extra
138
  assert len(result) == 2 # cs.CL + cs.IR
 
 
 
 
 
 
 
 
 
 
136
  assert "cs.CL" in result
137
  # unknown key produced nothing extra
138
  assert len(result) == 2 # cs.CL + cs.IR
139
+
140
+
141
+ async def test_get_user_category_groups_keeps_interests_apart(tmp_db):
142
+ import app.db as db
143
+ await db.init_db()
144
+ await db.save_onboarding_categories("u1", ["ml", "robotics"])
145
+ groups = await db.get_user_category_groups("u1")
146
+ assert groups == {"ml": {"cs.LG", "stat.ML"}, "robotics": {"cs.RO"}}
147
+ assert await db.get_user_category_groups("nobody") == {}