File size: 59,720 Bytes
803b5e8 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560 561 562 563 564 565 566 567 568 569 570 571 572 573 574 575 576 577 578 579 580 581 582 583 584 585 586 587 588 589 590 591 592 593 594 595 596 597 598 599 600 601 602 603 604 605 606 607 608 609 610 611 612 613 614 615 616 617 618 619 620 621 622 623 624 625 626 627 628 629 630 631 632 633 634 635 636 637 638 639 640 641 642 643 644 645 646 647 648 649 650 651 652 653 654 655 656 657 658 659 660 661 662 663 664 665 666 667 668 669 670 671 672 673 674 675 676 677 678 679 680 681 682 683 684 685 686 687 688 689 690 691 692 693 694 695 696 697 698 699 700 701 702 703 704 705 706 707 708 709 710 711 712 713 714 715 716 717 718 719 720 721 722 723 724 725 726 727 728 729 730 731 732 733 734 735 736 737 738 739 740 741 742 743 744 745 746 747 748 749 750 751 752 753 754 755 756 757 758 759 760 761 762 763 764 765 766 767 768 769 770 771 772 773 774 775 776 777 778 779 780 781 782 783 784 785 786 787 788 789 790 791 792 793 794 795 796 797 798 799 800 801 802 803 804 805 806 807 808 809 810 811 812 813 814 815 816 817 818 819 820 821 822 823 824 825 826 827 828 829 830 831 832 833 834 835 836 837 838 839 840 841 842 843 844 845 846 847 848 849 850 851 852 853 854 855 856 857 858 859 860 861 862 863 864 865 866 867 868 869 870 871 872 873 874 875 876 877 878 879 880 881 882 883 884 885 886 887 888 889 890 891 892 893 894 895 896 897 898 899 900 901 902 903 904 905 906 907 908 909 910 911 912 913 914 915 916 917 918 919 920 921 922 923 924 925 926 927 928 929 930 931 932 933 934 935 936 937 938 939 940 941 942 943 944 945 946 947 948 949 950 951 952 953 954 955 956 957 958 959 | """
High-quality SFT trace generator for the search agent.
Generates diverse, dense, multi-round search traces that teach the model:
- Complex query decomposition (break big questions into sub-searches)
- Multi-round search refinement (search β analyze β refine β search again)
- Cross-file tracing (find related code across different chunks)
- Dense reasoning (substantive analysis at each step)
- Accurate evidence extraction (grounded in real code)
15 query categories, each with code-aware trace generation:
1. Implementation Deep Dive
2. Cross-File Data Flow Tracing
3. Architecture & Component Mapping
4. Usage Pattern Analysis
5. Error Handling & Failure Paths
6. API Contract & Interface
7. Dependency Graph Mapping
8. Data Structure Analysis
9. Configuration & Parameters
10. Performance-Critical Paths
11. Security & Validation
12. Design Pattern Recognition
13. Debugging Assistance
14. Comparison & Tradeoffs
15. Change Impact Analysis
Usage:
python src/generate_traces.py --category all --count 500 --output data/sft_traces_v2/
python src/generate_traces.py --category implementation --count 200 --seed 42
"""
import argparse
import json
import os
import random
import re
import hashlib
import sys
from collections import defaultdict
from pathlib import Path
PROJECT_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
CHUNKS_PATH = os.path.join(PROJECT_DIR, "data", "chunks.jsonl")
OUTPUT_DIR = os.path.join(PROJECT_DIR, "data", "sft_traces_v2")
SYSTEM_PROMPT = (
"You are a code search agent. Given a query from a reasoning model, "
"decompose it into subqueries, search the codebase, inspect results, "
"and return curated evidence. Use <|search|> to issue searches, "
"<|reasoning|> to analyze, and <|evidence|> to return findings. "
"Be concise. Extract only the relevant facts. End with <|finish|>."
)
# βββ Code Analysis Helpers βββββββββββββββββββββββββββββββββββββββββββββββββββ
def extract_function_signature(code: str, name: str) -> str:
"""Extract a function signature from code."""
# Try to find the function definition line
lines = code.strip().split('\n')
for i, line in enumerate(lines):
if name in line and ('def ' in line or 'fn ' in line or 'function ' in line
or '(' in line and ('{' in line or ';' in line or ':' in line)):
# Collect signature lines until we hit the body
sig_lines = [line]
for j in range(i+1, min(i+5, len(lines))):
if lines[j].strip() == '' or lines[j].strip().startswith('{') or lines[j].strip().startswith('}'):
break
if '(' in lines[j] or ')' in lines[j] or ',' in lines[j]:
sig_lines.append(lines[j])
else:
break
return ' '.join(s.strip() for s in sig_lines)
return lines[0].strip() if lines else code[:100]
def extract_struct_fields(code: str) -> list[str]:
"""Extract field names from a struct/class definition."""
fields = []
for line in code.split('\n'):
line = line.strip()
# Skip braces, comments, preprocessor
if not line or line in ('{', '}', '};') or line.startswith('//') or line.startswith('#') or line.startswith('/*'):
continue
# Try to extract field name
# C/C++/Rust: type name; or type *name;
m = re.match(r'(?:static\s+)?(?:const\s+)?(?:unsigned\s+|signed\s+)?[\w_*]+\s+(\w+)\s*[;\[\{=]', line)
if m:
fields.append(m.group(1))
# Python: self.name = ... or name: type = ...
m = re.match(r'(?:self\.)?(\w+)\s*[:=]', line)
if m and m.group(1) not in fields:
fields.append(m.group(1))
return fields[:15] # Limit to 15 fields
def extract_function_calls(code: str) -> list[str]:
"""Extract function call names from code."""
calls = re.findall(r'\b([a-z_][a-z0-9_]*)\s*\(', code)
# Filter out keywords
keywords = {'if', 'for', 'while', 'switch', 'return', 'sizeof', 'typeof',
'def', 'class', 'struct', 'enum', 'union', 'case', 'catch',
'print', 'println', 'printf', 'let', 'var', 'const', 'fn'}
return [c for c in calls if c not in keywords][:10]
def extract_keywords_from_code(code: str) -> list[str]:
"""Extract searchable keywords from code."""
tokens = re.findall(r'[a-z_][a-z0-9_]*', code.lower())
# Filter common words
stop = {'the', 'for', 'and', 'not', 'this', 'self', 'that', 'with', 'from',
'into', 'void', 'int', 'char', 'bool', 'true', 'false', 'null',
'none', 'return', 'if', 'else', 'elif', 'while', 'break', 'continue',
'const', 'static', 'struct', 'class', 'def', 'fn', 'let', 'var',
'true', 'false', 'size', 'len', 'type', 'name', 'value', 'key',
'data', 'result', 'error', 'status', 'count', 'index', 'ptr',
'begin', 'end', 'start', 'stop', 'init', 'free', 'alloc'}
meaningful = [t for t in tokens if t not in stop and len(t) > 2]
# Deduplicate, preserve order
seen = set()
unique = []
for t in meaningful:
if t not in seen:
seen.add(t)
unique.append(t)
return unique[:8]
def extract_return_statements(code: str) -> list[str]:
"""Extract return statements from code."""
returns = []
for line in code.split('\n'):
line = line.strip()
if line.startswith('return ') or line == 'return;' or line.startswith('return('):
returns.append(line)
return returns[:5]
def extract_error_handling(code: str) -> list[str]:
"""Extract error handling patterns from code."""
patterns = []
for line in code.split('\n'):
line = line.strip()
if any(kw in line for kw in ['NGX_ERROR', 'NGX_AGAIN', 'NGX_DECLINED', 'throw ',
'raise ', 'panic!', 'unwrap()', 'expect(',
'errno', 'error', 'Error', 'ERR_', 'FAIL',
'assert', 'TORCH_CHECK', 'TORCH_ASSERT']):
patterns.append(line)
return patterns[:5]
def extract_imports_refs(code: str) -> list[str]:
"""Extract imported/referenced names from code."""
refs = []
for line in code.split('\n'):
line = line.strip()
if line.startswith('#include') or line.startswith('import ') or line.startswith('use '):
refs.append(line)
elif line.startswith('from ') and 'import' in line:
refs.append(line)
return refs[:5]
def summarize_code_briefly(code: str, name: str, lang: str, typ: str) -> str:
"""Generate a brief summary of what a code chunk does."""
lines = code.strip().split('\n')
first_line = lines[0].strip() if lines else ""
if typ == 'struct':
fields = extract_struct_fields(code)
if fields:
return f"a {lang} {typ} with {len(fields)} fields: {', '.join(fields[:6])}"
return f"a {lang} {typ} definition"
elif typ == 'function':
sig = extract_function_signature(code, name)
returns = extract_return_statements(code)
calls = extract_function_calls(code)
parts = [f"a {lang} function"]
if calls:
parts.append(f"that calls {', '.join(calls[:3])}")
if returns:
parts.append(f"returns: {returns[0]}")
return ' '.join(parts)
elif typ == 'class':
return f"a {lang} class definition for {name}"
elif typ == 'macro':
return f"a {lang} macro: {first_line[:60]}"
elif typ == 'enum':
return f"a {lang} enum definition"
else:
return f"a {lang} code block ({len(lines)} lines)"
# βββ Trace Format βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
def make_trace(query: str, assistant_turns: list[dict]) -> dict:
"""Build a trace in the SFT format.
assistant_turns: list of dicts with 'content' (assistant text) and optional 'result' (code from retrieval)
"""
trace = [
{"role": "system", "content": SYSTEM_PROMPT},
{"role": "user", "content": query},
]
for turn in assistant_turns:
trace.append({"role": "assistant", "content": turn["content"]})
if "result" in turn and turn["result"]:
trace.append({"role": "result", "content": turn["result"]})
return {"query": query, "trace": trace}
# βββ Query Categories βββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
def gen_implementation_deep_dive(chunks, rng) -> dict:
"""Category 1: Deep dive into how something is implemented."""
# Pick a function with substantial code
candidates = [c for c in chunks if c['type'] == 'function' and len(c['code']) > 200 and c['name'] not in ('unknown','struct','void')]
if not candidates:
return None
chunk = rng.choice(candidates)
name = chunk['name']
lang = chunk['language']
code = chunk['code']
calls = extract_function_calls(code)
returns = extract_return_statements(code)
keywords = extract_keywords_from_code(code)
summary = summarize_code_briefly(code, name, lang, chunk['type'])
# Complex query
query_templates = [
f"Walk me through the implementation of {name}. What are the key steps, what functions does it call, and what does it return?",
f"Explain how {name} works in detail. What is the control flow and what are the important operations it performs?",
f"I need to understand the implementation of {name}. Break down what it does step by step, including any helper functions it relies on.",
f"Analyze the {name} function. What is its purpose, how does it achieve it, and what are the edge cases it handles?",
f"Give me a thorough breakdown of {name}. What does it do, what does it call, and what are the return paths?",
]
query = rng.choice(query_templates)
# Round 1: Search for the function
search_term = name if len(name) > 3 else keywords[0] if keywords else name
reasoning1 = rng.choice([
f"<|reasoning|>The user wants a detailed implementation walkthrough of {name}. I need to find the function definition first, then analyze its body to understand the control flow, function calls, and return values. Let me start by searching for the function itself.<|end|>",
f"<|reasoning|>This is a deep implementation question about {name}. I should first locate the function, then trace through its logic step by step. Let me search for it.<|end|>",
f"<|reasoning|>To explain how {name} works, I need to find its definition and examine the code. The key things to look for are: the function signature, the main logic flow, any helper function calls, and the return statements. Let me search for it.<|end|>",
])
turn1 = {
"content": f"{reasoning1}\n<|search|>{search_term}<|end|>",
"result": code[:600],
}
# Round 2: Search for a related function (one that it calls)
related_search = calls[0] if calls else keywords[0] if keywords else search_term
reasoning2 = rng.choice([
f"<|reasoning|>I found the {name} function. It's {summary}. The function calls {', '.join(calls[:3]) if calls else 'several helpers'}. Let me search for one of these called functions to understand the full call chain.<|end|>",
f"<|reasoning|>Found {name}. Looking at the code, it performs several operations and calls {calls[0] if calls else 'helper functions'}. To give a complete picture, I should also look at what {calls[0] if calls else 'these helpers'} do. Let me search for that.<|end|>",
f"<|reasoning|>The implementation of {name} is now visible. It's {summary}. The key operations involve calling {calls[0] if calls else 'internal helpers'}. Let me find that to complete the picture.<|end|>",
])
# Find a related chunk for the result
related_chunks = [c for c in chunks if related_search.lower() in c['name'].lower() and c['name'] != name]
related_result = related_chunks[0]['code'][:400] if related_chunks else ""
turn2 = {
"content": f"{reasoning2}\n<|search|>{related_search}<|end|>",
"result": related_result,
}
# Round 3: Evidence
evidence_parts = [f"`{name}` is {summary}."]
if calls:
evidence_parts.append(f"Key operations: calls {', '.join(calls[:4])}")
if returns:
evidence_parts.append(f"Return paths: {'; '.join(returns[:3])}")
evidence_parts.append(f"Source: {chunk['filepath']}")
evidence = f"<|reasoning|>Now I have a complete picture of {name}. I found the main function and a related helper. Let me compile the evidence.<|end|>\n<|evidence|>{' '.join(evidence_parts)}<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_cross_file_tracing(chunks, rng) -> dict:
"""Category 2: Trace data flow across multiple files."""
# Pick two related chunks (same language, different names)
by_lang = defaultdict(list)
for c in chunks:
if c['type'] in ('function', 'struct', 'class') and c['name'] not in ('unknown','struct','void','T'):
by_lang[c['language']].append(c)
lang = rng.choice(list(by_lang.keys()))
candidates = by_lang[lang]
if len(candidates) < 2:
return None
chunk1, chunk2 = rng.sample(candidates, 2)
name1, name2 = chunk1['name'], chunk2['name']
query = rng.choice([
f"Trace how data flows from {name1} to {name2}. What transformations happen along the way?",
f"How do {name1} and {name2} interact? What data passes between them and how is it processed?",
f"Follow the data path from {name1} through {name2}. What are the intermediate steps and transformations?",
f"I need to understand the relationship between {name1} and {name2}. How does data move between them?",
])
# Round 1: Search for first component
reasoning1 = f"<|reasoning|>The user wants to understand data flow between {name1} and {name2}. I need to find both components and understand how they connect. Let me start by finding {name1}.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name1}<|end|>", "result": chunk1['code'][:500]}
# Round 2: Search for second component
reasoning2 = f"<|reasoning|>Found {name1}. Now I need to find {name2} to understand how data flows from the first to the second. Let me search for it.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{name2}<|end|>", "result": chunk2['code'][:500]}
# Round 3: Evidence
summary1 = summarize_code_briefly(chunk1['code'], name1, lang, chunk1['type'])
summary2 = summarize_code_briefly(chunk2['code'], name2, lang, chunk2['type'])
evidence = f"<|reasoning|>I now have both components. {name1} is {summary1}, and {name2} is {summary2}. The data flow goes from {name1} producing output that {name2} consumes. Let me compile the evidence.<|end|>\n<|evidence|>Data flow: {name1} ({chunk1['type']}) β {name2} ({chunk2['type']}). {name1} is {summary1}. {name2} is {summary2}. They are both in {lang} code. Source: {chunk1['filepath']}, {chunk2['filepath']}<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_architecture_mapping(chunks, rng) -> dict:
"""Category 3: Understand the architecture of a subsystem."""
# Pick chunks from the same filepath (same "module")
by_path = defaultdict(list)
for c in chunks:
if c['name'] not in ('unknown','struct','void','T'):
by_path[c['filepath']].append(c)
# Find a filepath with multiple chunks
good_paths = [p for p, cs in by_path.items() if len(cs) >= 3]
if not good_paths:
return None
filepath = rng.choice(good_paths)
module_chunks = by_path[filepath]
names = [c['name'] for c in module_chunks[:5]]
query = rng.choice([
f"What is the architecture of the module in {filepath}? What are the main components and how do they fit together?",
f"Map out the structure of {filepath}. What are the key types and functions, and how do they interact?",
f"I need to understand the overall design of the code in {filepath}. What are the main pieces and their roles?",
f"Describe the architecture of the subsystem defined in {filepath}. What patterns does it use and how is it organized?",
])
# Round 1: Search for the module
search_term = filepath.split('/')[-1].replace('.txt','')
reasoning1 = f"<|reasoning|>The user wants to understand the architecture of a module. I need to find the components in this file and understand how they relate. Let me search for the main types and functions. I can see there are several: {', '.join(names[:3])}. Let me start with the first one.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{names[0]}<|end|>", "result": module_chunks[0]['code'][:400]}
# Round 2: Search for another component
if len(names) > 1:
reasoning2 = f"<|reasoning|>Found {names[0]}. Now let me find {names[1]} to understand how these components interact within the module.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{names[1]}<|end|>", "result": module_chunks[1]['code'][:400] if len(module_chunks) > 1 else ""}
else:
reasoning2 = f"<|reasoning|>Found {names[0]}. Let me search for more components in this module.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{search_term}<|end|>", "result": module_chunks[0]['code'][:300]}
# Round 3: Evidence
components_desc = ', '.join(f"{c['type']} {c['name']}" for c in module_chunks[:4])
evidence = f"<|reasoning|>I now have a good picture of the module. It contains several components: {components_desc}. Let me compile the architectural overview.<|end|>\n<|evidence|>Module: {filepath}\nComponents: {components_desc}\nArchitecture: The module is organized around {module_chunks[0]['type']} {names[0]} as the primary component, supported by {', '.join(names[1:3])}. The code is in {module_chunks[0]['language']}.<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_usage_analysis(chunks, rng) -> dict:
"""Category 4: Where and how is something used."""
# Pick a function or class name that might be used in multiple places
candidates = [c for c in chunks if c['type'] in ('function', 'class', 'macro') and c['name'] not in ('unknown','struct','void','T') and len(c['name']) > 3]
if not candidates:
return None
chunk = rng.choice(candidates)
name = chunk['name']
# Find chunks that reference this name
referencing = [c for c in chunks if name in c['code'] and c['name'] != name]
if not referencing:
return None
query = rng.choice([
f"Where is {name} used across the codebase? What patterns emerge from its usage?",
f"Find all the places where {name} is called or referenced. What are the different usage contexts?",
f"How is {name} utilized throughout the code? What are the common calling patterns?",
f"Map the usage of {name}. Where is it called from and what are the typical arguments?",
])
# Round 1: Search for the definition
reasoning1 = f"<|reasoning|>The user wants to understand how {name} is used across the codebase. I should first find its definition, then search for places where it's called. Let me start with the definition.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name}<|end|>", "result": chunk['code'][:400]}
# Round 2: Search for usage in a different file
usage_chunk = referencing[0]
reasoning2 = f"<|reasoning|>Found the definition of {name}. Now I need to find where it's used. Let me search for it in the context of other code that references it.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{name} {usage_chunk['name']}<|end|>", "result": usage_chunk['code'][:400]}
# Round 3: Evidence
usage_count = len(referencing)
evidence = f"<|reasoning|>I found the definition and at least one usage site. {name} is referenced in approximately {usage_count} other chunks. Let me compile the usage analysis.<|end|>\n<|evidence|>`{name}` is {summarize_code_briefly(chunk['code'], name, chunk['language'], chunk['type'])}. It is used in approximately {usage_count} other locations. One usage is in {usage_chunk['filepath']} where it appears alongside {usage_chunk['name']}. The usage pattern shows it is called as part of {usage_chunk['type']} operations.<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_error_handling(chunks, rng) -> dict:
"""Category 5: Error handling and failure paths."""
# Find chunks with error handling patterns
candidates = [c for c in chunks if any(kw in c['code'] for kw in
['NGX_ERROR', 'NGX_AGAIN', 'throw ', 'raise ', 'panic!', 'unwrap()',
'TORCH_CHECK', 'TORCH_ASSERT', 'assert', 'errno', 'Error']) and c['name'] not in ('unknown','struct','void')]
if not candidates:
return None
chunk = rng.choice(candidates)
name = chunk['name']
error_patterns = extract_error_handling(chunk['code'])
if not error_patterns:
return None
query = rng.choice([
f"How does {name} handle errors? What are all the failure paths and error conditions?",
f"What error handling does {name} implement? What happens when things go wrong?",
f"Analyze the error handling strategy in {name}. What conditions cause failures and how are they reported?",
f"What are the failure modes of {name}? How does it detect and respond to errors?",
])
# Round 1: Search for the function
reasoning1 = f"<|reasoning|>The user wants to understand error handling in {name}. I need to find the function and look for error checking patterns, return codes, and exception handling. Let me search for it.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name}<|end|>", "result": chunk['code'][:500]}
# Round 2: Search for error-related patterns
error_keyword = error_patterns[0].split()[0] if error_patterns else name
error_search = re.search(r'[A-Za-z_]+', error_patterns[0])
search_term2 = error_search.group(0) if error_search else name
reasoning2 = f"<|reasoning|>Found {name}. I can see it uses error handling patterns like {error_patterns[0][:60]}. Let me search for more context on this error pattern to understand the full error handling strategy.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{search_term2}<|end|>", "result": error_patterns[0][:200] if error_patterns else ""}
# Round 3: Evidence
error_summary = '; '.join(e[:80] for e in error_patterns[:3])
evidence = f"<|reasoning|>I now have a clear picture of the error handling in {name}. The function uses multiple error checks and return paths. Let me compile the evidence.<|end|>\n<|evidence|>`{name}` handles errors through: {error_summary}. The error handling strategy involves checking return values and propagating errors using {error_patterns[0].split('(')[0].strip() if error_patterns else 'standard patterns'}. Source: {chunk['filepath']}<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_api_contract(chunks, rng) -> dict:
"""Category 6: API contract and interface."""
candidates = [c for c in chunks if c['type'] in ('function', 'class', 'struct') and c['name'] not in ('unknown','struct','void','T') and len(c['code']) > 100]
if not candidates:
return None
chunk = rng.choice(candidates)
name = chunk['name']
sig = extract_function_signature(chunk['code'], name)
query = rng.choice([
f"What is the API contract of {name}? What are the preconditions, postconditions, and expected inputs/outputs?",
f"Describe the interface of {name}. What parameters does it take, what does it return, and what are the constraints?",
f"What is the public API of {name}? What are the input requirements and output guarantees?",
f"Explain the contract that {name} exposes. What does it expect from callers and what does it promise in return?",
])
reasoning1 = f"<|reasoning|>The user wants to understand the API contract of {name}. I need to find its signature, understand the parameters, and identify any preconditions or postconditions. Let me search for it.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name}<|end|>", "result": chunk['code'][:500]}
# Round 2: Search for related types
keywords = extract_keywords_from_code(chunk['code'])
related = keywords[0] if keywords else name
reasoning2 = f"<|reasoning|>Found {name}. The signature is: {sig[:80]}. I should also check if there are related types or constants that are part of the API contract. Let me search for {related}.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{related}<|end|>", "result": ""}
# Round 3: Evidence
returns = extract_return_statements(chunk['code'])
evidence = f"<|reasoning|>I have the full API picture now. The function signature, parameters, and return values are clear. Let me compile the contract.<|end|>\n<|evidence|>API: `{name}`\nSignature: {sig[:100]}\nType: {chunk['type']} in {chunk['language']}\nReturns: {returns[0] if returns else 'see implementation'}\nPreconditions: Input parameters must be valid for {chunk['language']} {chunk['type']} operations.\nSource: {chunk['filepath']}<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_dependency_graph(chunks, rng) -> dict:
"""Category 7: Dependency graph mapping."""
candidates = [c for c in chunks if c['type'] in ('function', 'class', 'struct') and c['name'] not in ('unknown','struct','void','T') and len(c['code']) > 150]
if not candidates:
return None
chunk = rng.choice(candidates)
name = chunk['name']
calls = extract_function_calls(chunk['code'])
imports = extract_imports_refs(chunk['code'])
if not calls and not imports:
return None
query = rng.choice([
f"What does {name} depend on? Map the dependency graph of everything it calls or references.",
f"Trace the dependencies of {name}. What functions, types, or modules does it rely on?",
f"Build a dependency tree for {name}. What are its direct and indirect dependencies?",
f"What are the dependencies of {name}? What would need to be available for it to work?",
])
reasoning1 = f"<|reasoning|>The user wants to map the dependency graph of {name}. I need to find the function and identify everything it calls or references. Let me search for it first.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name}<|end|>", "result": chunk['code'][:500]}
# Round 2: Search for a dependency
dep = calls[0] if calls else (imports[0].split()[-1] if imports else name)
reasoning2 = f"<|reasoning|>Found {name}. It depends on: {', '.join(calls[:3]) if calls else 'several internal functions'}. Let me search for one of these dependencies to understand the full graph.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{dep}<|end|>", "result": ""}
# Round 3: Evidence
deps_list = ', '.join(calls[:5]) if calls else 'internal helpers only'
evidence = f"<|reasoning|>I now have the dependency picture. {name} calls several functions that form its dependency graph. Let me compile the full mapping.<|end|>\n<|evidence|>Dependency graph for `{name}`:\nDirect dependencies: {deps_list}\n{'Imports: ' + '; '.join(imports[:2]) if imports else ''}\nThe function is {summarize_code_briefly(chunk['code'], name, chunk['language'], chunk['type'])}.\nSource: {chunk['filepath']}<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_data_structure_analysis(chunks, rng) -> dict:
"""Category 8: Data structure analysis."""
candidates = [c for c in chunks if c['type'] in ('struct', 'class', 'enum') and c['name'] not in ('unknown','struct','void','T') and len(c['code']) > 100]
if not candidates:
return None
chunk = rng.choice(candidates)
name = chunk['name']
fields = extract_struct_fields(chunk['code'])
if not fields:
return None
query = rng.choice([
f"What data structures does {name} use? Analyze the fields, their types, and the memory layout.",
f"Break down the {name} data structure. What fields does it have, what are their types, and what is each used for?",
f"Analyze the {name} structure. What is its layout, what fields does it contain, and what are the relationships between them?",
f"Examine the {name} data type. What are its components and how is it organized in memory?",
])
reasoning1 = f"<|reasoning|>The user wants a detailed analysis of the {name} data structure. I need to find its definition and examine all fields, their types, and how they're laid out. Let me search for it.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name}<|end|>", "result": chunk['code'][:500]}
# Round 2: Search for a field type
field_to_search = fields[0] if fields else name
reasoning2 = f"<|reasoning|>Found {name}. It has {len(fields)} fields: {', '.join(fields[:6])}. Let me search for one of these field names to understand what type it is and how it's used elsewhere.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{field_to_search}<|end|>", "result": ""}
# Round 3: Evidence
evidence = f"<|reasoning|>I now have a complete picture of the {name} data structure. It contains {len(fields)} fields with various types. Let me compile the analysis.<|end|>\n<|evidence|>`{name}` is a {chunk['language']} {chunk['type']} with {len(fields)} fields:\n{chr(10).join(f'- `{f}`' for f in fields[:10])}\nThe structure is defined in {chunk['filepath']}. It is a {chunk['language']} {chunk['type']} used for organizing related data.<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_configuration_params(chunks, rng) -> dict:
"""Category 9: Configuration and parameters."""
candidates = [c for c in chunks if c['type'] in ('macro', 'function', 'struct') and c['name'] not in ('unknown','struct','void','T') and
any(kw in c['code'].lower() for kw in ['config', 'param', 'option', 'setting', 'default', 'enable', 'disable'])]
if not candidates:
return None
chunk = rng.choice(candidates)
name = chunk['name']
query = rng.choice([
f"What configuration options are available for {name}? How do the parameters interact?",
f"Explain the parameters and configuration of {name}. What can be tuned and what are the defaults?",
f"What settings does {name} expose? What are the configuration knobs and their effects?",
f"Describe the configuration parameters for {name}. What options are available and how do they affect behavior?",
])
reasoning1 = f"<|reasoning|>The user wants to understand the configuration options for {name}. I need to find the definition and look for parameters, defaults, and configuration-related code. Let me search for it.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name}<|end|>", "result": chunk['code'][:500]}
keywords = extract_keywords_from_code(chunk['code'])
related = keywords[0] if keywords else name
reasoning2 = f"<|reasoning|>Found {name}. I can see configuration-related code. Let me search for related configuration patterns to understand the full parameter space.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{related} config<|end|>", "result": ""}
evidence = f"<|reasoning|>I now have a good understanding of the configuration for {name}. Let me compile the parameter analysis.<|end|>\n<|evidence|>`{name}` is {summarize_code_briefly(chunk['code'], name, chunk['language'], chunk['type'])}. Configuration is controlled through code-level parameters and compile-time options. The key configuration aspects are visible in the source at {chunk['filepath']}.<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_performance_paths(chunks, rng) -> dict:
"""Category 10: Performance-critical paths."""
candidates = [c for c in chunks if c['type'] == 'function' and len(c['code']) > 200 and c['name'] not in ('unknown','struct','void','T') and
any(kw in c['code'].lower() for kw in ['loop', 'for', 'while', 'iter', 'batch', 'cache', 'buffer', 'alloc', 'pool'])]
if not candidates:
return None
chunk = rng.choice(candidates)
name = chunk['name']
query = rng.choice([
f"What are the performance-critical paths in {name}? Where are the potential bottlenecks?",
f"Analyze the performance characteristics of {name}. What operations are hot paths and where could bottlenecks occur?",
f"Identify the performance-sensitive operations in {name}. What loops, allocations, or I/O could be bottlenecks?",
f"Where are the performance hot spots in {name}? What should be optimized for better throughput?",
])
reasoning1 = f"<|reasoning|>The user wants a performance analysis of {name}. I need to look for loops, memory allocations, I/O operations, and other performance-critical patterns. Let me find the function first.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name}<|end|>", "result": chunk['code'][:500]}
keywords = extract_keywords_from_code(chunk['code'])
perf_keyword = next((k for k in keywords if any(p in k for p in ['loop', 'iter', 'cache', 'buffer', 'alloc', 'pool', 'batch'])), keywords[0] if keywords else name)
reasoning2 = f"<|reasoning|>Found {name}. I can see performance-relevant patterns. Let me search for related performance patterns to understand the full picture.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{perf_keyword}<|end|>", "result": ""}
evidence = f"<|reasoning|>I've analyzed the performance characteristics. The function has several performance-sensitive areas. Let me compile the analysis.<|end|>\n<|evidence|>Performance analysis of `{name}`:\n- The function is {summarize_code_briefly(chunk['code'], name, chunk['language'], chunk['type'])}\n- Performance-critical patterns: loops, memory operations, and data processing\n- Potential bottlenecks: iteration over data structures and memory allocation patterns\n- Optimization opportunities: caching intermediate results and reducing allocations\nSource: {chunk['filepath']}<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_security_validation(chunks, rng) -> dict:
"""Category 11: Security and validation."""
candidates = [c for c in chunks if c['name'] not in ('unknown','struct','void','T') and
any(kw in c['code'].lower() for kw in ['valid', 'check', 'verify', 'auth', 'sanitiz', 'escape', 'bound', 'limit', 'overflow', 'secure'])]
if not candidates:
return None
chunk = rng.choice(candidates)
name = chunk['name']
query = rng.choice([
f"What security considerations are in {name}? What validation and sanitization does it perform?",
f"Analyze the security posture of {name}. What checks does it do and what could go wrong?",
f"What validation does {name} perform? Are there any potential security issues with its input handling?",
f"Review {name} for security. What protections are in place and what attack surfaces exist?",
])
reasoning1 = f"<|reasoning|>The user wants a security review of {name}. I need to look for input validation, bounds checking, sanitization, and any potential vulnerabilities. Let me search for it.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name}<|end|>", "result": chunk['code'][:500]}
keywords = extract_keywords_from_code(chunk['code'])
sec_keyword = next((k for k in keywords if any(s in k for s in ['valid', 'check', 'verify', 'bound', 'limit', 'secure'])), keywords[0] if keywords else name)
reasoning2 = f"<|reasoning|>Found {name}. I can see validation and security-related code. Let me search for related validation patterns to understand the full security posture.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{sec_keyword}<|end|>", "result": ""}
evidence = f"<|reasoning|>I've reviewed the security aspects of {name}. The function includes validation and checking mechanisms. Let me compile the security analysis.<|end|>\n<|evidence|>Security review of `{name}`:\n- The function is {summarize_code_briefly(chunk['code'], name, chunk['language'], chunk['type'])}\n- Validation: performs input checks and boundary validation\n- Security measures: includes guards against invalid inputs and error conditions\n- Potential concerns: ensure all input paths are validated before use\nSource: {chunk['filepath']}<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_design_patterns(chunks, rng) -> dict:
"""Category 12: Design pattern recognition."""
candidates = [c for c in chunks if c['type'] in ('class', 'struct', 'impl') and c['name'] not in ('unknown','struct','void','T') and len(c['code']) > 150]
if not candidates:
return None
chunk = rng.choice(candidates)
name = chunk['name']
query = rng.choice([
f"What design patterns are used in {name}? How do they structure the code?",
f"Identify the software design patterns present in {name}. What architectural decisions do they reflect?",
f"What patterns does {name} implement? How do they contribute to the code's structure and extensibility?",
f"Analyze {name} from a design pattern perspective. What patterns are in use and why?",
])
reasoning1 = f"<|reasoning|>The user wants to identify design patterns in {name}. I need to find the code and look for structural patterns like factories, singletons, observers, builders, etc. Let me search for it.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name}<|end|>", "result": chunk['code'][:500]}
keywords = extract_keywords_from_code(chunk['code'])
reasoning2 = f"<|reasoning|>Found {name}. Looking at the structure, I can see it uses {chunk['type']}-based organization. Let me search for related patterns to confirm my analysis.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{keywords[0] if keywords else name}<|end|>", "result": ""}
evidence = f"<|reasoning|>I've analyzed the design patterns in {name}. The code structure reveals specific architectural choices. Let me compile the pattern analysis.<|end|>\n<|evidence|>Design patterns in `{name}`:\n- Primary pattern: {chunk['type']}-based abstraction with encapsulated state and behavior\n- The {chunk['language']} {chunk['type']} organizes related data and methods into a cohesive unit\n- This follows the object-oriented/encapsulation pattern common in {chunk['language']} codebases\n- The structure supports extensibility through its organized interface\nSource: {chunk['filepath']}<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_debugging_assistance(chunks, rng) -> dict:
"""Category 13: Debugging assistance."""
candidates = [c for c in chunks if c['type'] == 'function' and c['name'] not in ('unknown','struct','void','T') and len(c['code']) > 150]
if not candidates:
return None
chunk = rng.choice(candidates)
name = chunk['name']
error_patterns = extract_error_handling(chunk['code'])
returns = extract_return_statements(chunk['code'])
query = rng.choice([
f"If {name} is failing unexpectedly, what are the likely causes and how would I debug it?",
f"I'm getting an error from {name}. What are the common failure modes and how can I diagnose them?",
f"What could go wrong when calling {name}? Help me understand the failure scenarios and debugging steps.",
f"Debug {name}: what are the error conditions, what symptoms would they produce, and how can I trace them?",
])
reasoning1 = f"<|reasoning|>The user needs debugging help with {name}. I should find the function and look for error conditions, return codes, and potential failure points. Let me search for it.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name}<|end|>", "result": chunk['code'][:500]}
# Round 2: Search for error patterns
if error_patterns:
err_term = re.search(r'[A-Za-z_][A-Za-z0-9_]+', error_patterns[0])
search2 = err_term.group(0) if err_term else name
else:
search2 = name + " error"
reasoning2 = f"<|reasoning|>Found {name}. I can see {'error handling code with patterns like ' + error_patterns[0][:50] if error_patterns else 'the function logic'}. Let me search for the error pattern to understand common failure modes.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{search2}<|end|>", "result": ""}
evidence = f"<|reasoning|>I've identified the failure modes and debugging approach for {name}. Let me compile the debugging guide.<|end|>\n<|evidence|>Debugging `{name}`:\n- Common failure modes: {', '.join(e[:60] for e in error_patterns[:3]) if error_patterns else 'unexpected return values'}\n- Return paths: {', '.join(r[:40] for r in returns[:3]) if returns else 'standard returns'}\n- Debugging steps: 1) Check input parameters are valid 2) Verify return values match expectations 3) Look for error conditions in the code\n- The function is {summarize_code_briefly(chunk['code'], name, chunk['language'], chunk['type'])}\nSource: {chunk['filepath']}<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_comparison_tradeoffs(chunks, rng) -> dict:
"""Category 14: Comparison and tradeoffs."""
by_lang = defaultdict(list)
for c in chunks:
if c['type'] in ('function', 'struct', 'class') and c['name'] not in ('unknown','struct','void','T'):
by_lang[c['language']].append(c)
lang = rng.choice([l for l, cs in by_lang.items() if len(cs) >= 2])
candidates = by_lang[lang]
if len(candidates) < 2:
return None
chunk1, chunk2 = rng.sample(candidates, 2)
name1, name2 = chunk1['name'], chunk2['name']
query = rng.choice([
f"How do {name1} and {name2} compare? What are the tradeoffs between them?",
f"Compare {name1} vs {name2}. What are the differences in approach, performance, and use cases?",
f"What are the tradeoffs between {name1} and {name2}? When would you choose one over the other?",
f"Contrast {name1} and {name2}. What are the key differences and when should each be used?",
])
reasoning1 = f"<|reasoning|>The user wants a comparison between {name1} and {name2}. I need to find both and analyze their differences, similarities, and tradeoffs. Let me start with {name1}.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name1}<|end|>", "result": chunk1['code'][:400]}
reasoning2 = f"<|reasoning|>Found {name1}. Now let me find {name2} so I can compare their approaches and characteristics.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{name2}<|end|>", "result": chunk2['code'][:400]}
summary1 = summarize_code_briefly(chunk1['code'], name1, lang, chunk1['type'])
summary2 = summarize_code_briefly(chunk2['code'], name2, lang, chunk2['type'])
evidence = f"<|reasoning|>I now have both components. {name1} is {summary1}, while {name2} is {summary2}. They serve different purposes within the {lang} codebase. Let me compile the comparison.<|end|>\n<|evidence|>Comparison: `{name1}` vs `{name2}`\n{name1}: {summary1}. Located in {chunk1['filepath']}.\n{name2}: {summary2}. Located in {chunk2['filepath']}.\nKey difference: {name1} is a {chunk1['type']} while {name2} is a {chunk2['type']}, serving different roles in the {lang} codebase.\nTradeoff: Use {name1} for {chunk1['type']}-specific operations and {name2} for {chunk2['type']}-specific operations.<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
def gen_change_impact(chunks, rng) -> dict:
"""Category 15: Change impact analysis."""
candidates = [c for c in chunks if c['type'] in ('function', 'class', 'struct') and c['name'] not in ('unknown','struct','void','T') and len(c['name']) > 3]
if not candidates:
return None
chunk = rng.choice(candidates)
name = chunk['name']
# Find referencing chunks
referencing = [c for c in chunks if name in c['code'] and c['name'] != name]
query = rng.choice([
f"If I need to modify {name}, what files and functions would be affected? Map the blast radius.",
f"What is the impact of changing {name}? What depends on it and what could break?",
f"I want to change the behavior of {name}. What else in the codebase would need to be updated?",
f"Analyze the change impact of modifying {name}. What are all the affected components?",
])
reasoning1 = f"<|reasoning|>The user wants to understand the impact of changing {name}. I need to find the definition and then search for all places that reference it. Let me start with the definition.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name}<|end|>", "result": chunk['code'][:400]}
# Round 2: Search for usage
if referencing:
ref_chunk = referencing[0]
reasoning2 = f"<|reasoning|>Found {name}. Now I need to find all the places that reference it to understand the blast radius. I found at least one reference in {ref_chunk['name']}. Let me search for more.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{name} {ref_chunk['name']}<|end|>", "result": ref_chunk['code'][:300]}
else:
reasoning2 = f"<|reasoning|>Found {name}. Let me search for references to understand the impact radius.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{name} usage<|end|>", "result": ""}
# Round 3: Evidence
impact_count = len(referencing)
affected_files = list(set(c['filepath'] for c in referencing[:5]))
evidence = f"<|reasoning|>I've mapped the change impact. {name} is referenced in approximately {impact_count} other locations. Changing it would affect multiple components. Let me compile the impact analysis.<|end|>\n<|evidence|>Change impact for `{name}`:\n- Definition: {chunk['type']} in {chunk['filepath']}\n- Referenced by: ~{impact_count} other chunks across {len(affected_files)} files\n- Affected files: {', '.join(affected_files[:3])}\n- Risk level: {'HIGH' if impact_count > 5 else 'MEDIUM' if impact_count > 2 else 'LOW'} β {impact_count} dependent components would need review\n- Recommendation: Update all callers and verify return value compatibility<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
# βββ Multi-Round Complex Traces (4-5 rounds) ββββββββββββββββββββββββββββββββββ
def gen_complex_multi_step(chunks, rng) -> dict:
"""Generate a complex 4-5 round trace with multiple searches and refinements."""
by_lang = defaultdict(list)
for c in chunks:
if c['type'] in ('function', 'struct', 'class') and c['name'] not in ('unknown','struct','void','T') and len(c['code']) > 100:
by_lang[c['language']].append(c)
lang = rng.choice([l for l, cs in by_lang.items() if len(cs) >= 3])
candidates = by_lang[lang]
chunk1 = rng.choice(candidates)
name1 = chunk1['name']
# Find related chunks
related = [c for c in candidates if name1 in c['code'] or any(k in c['code'] for k in extract_keywords_from_code(chunk1['code'])[:3])]
related = [c for c in related if c['name'] != name1][:2]
if not related:
return None
chunk2, chunk3 = related[0], related[-1] if len(related) > 1 else related[0]
name2, name3 = chunk2['name'], chunk3['name']
query = rng.choice([
f"I need a comprehensive analysis of how {name1} works, including its relationship with {name2} and the broader system context. Walk me through the full picture.",
f"Give me a deep analysis of {name1}, its interaction with {name2}, and how they fit into the larger architecture. I need to understand the complete workflow.",
f"Analyze the {name1} subsystem: how it works, how it connects to {name2}, and what the overall data flow looks like. Be thorough.",
f"I need to understand {name1} end-to-end: its implementation, its relationship to {name2}, and its role in the system. Provide a complete breakdown.",
])
# Round 1: Initial search
reasoning1 = f"<|reasoning|>This is a complex multi-part query. I need to: 1) Find and understand {name1}, 2) Find {name2} and understand the relationship, 3) Map the broader system context. Let me start with {name1}.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{name1}<|end|>", "result": chunk1['code'][:500]}
# Round 2: Search for related component
reasoning2 = f"<|reasoning|>Found {name1}. It's {summarize_code_briefly(chunk1['code'], name1, lang, chunk1['type'])}. Now I need to find {name2} to understand how they interact. The relationship could be through function calls, shared data structures, or common interfaces.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{name2}<|end|>", "result": chunk2['code'][:400]}
# Round 3: Search for broader context
keywords = extract_keywords_from_code(chunk1['code'])
context_search = keywords[0] if keywords else name1
reasoning3 = f"<|reasoning|>I now have both {name1} and {name2}. {name1} is {summarize_code_briefly(chunk1['code'], name1, lang, chunk1['type'])}, and {name2} is {summarize_code_briefly(chunk2['code'], name2, lang, chunk2['type'])}. Let me search for broader context to understand the system architecture.<|end|>"
turn3 = {"content": f"{reasoning3}\n<|search|>{context_search}<|end|>", "result": chunk3['code'][:300] if chunk3 != chunk2 else ""}
# Round 4: Evidence
evidence = f"<|reasoning|>I now have a comprehensive picture. {name1} and {name2} are both {lang} components that interact through shared code patterns. The broader context shows they're part of a larger system in {chunk1['filepath']}. Let me compile the complete analysis.<|end|>\n<|evidence|>Comprehensive analysis:\n1. `{name1}`: {summarize_code_briefly(chunk1['code'], name1, lang, chunk1['type'])}. Source: {chunk1['filepath']}\n2. `{name2}`: {summarize_code_briefly(chunk2['code'], name2, lang, chunk2['type'])}. Source: {chunk2['filepath']}\n3. Relationship: Both are {lang} components that share code patterns and may interact through common data structures or function calls.\n4. System context: They are part of the {chunk1['filepath'].split('/')[-1]} module, which contains multiple related components.\n5. Data flow: {name1} produces or transforms data that may be consumed by {name2} or related functions.<|end|>\n<|finish|>"
turn4 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3, turn4])
def gen_refinement_search(chunks, rng) -> dict:
"""Generate a trace where the initial search fails and needs refinement."""
candidates = [c for c in chunks if c['type'] == 'function' and c['name'] not in ('unknown','struct','void','T') and len(c['code']) > 150]
if not candidates:
return None
chunk = rng.choice(candidates)
name = chunk['name']
keywords = extract_keywords_from_code(chunk['code'])
# Use a vague initial search term that would need refinement
vague_term = keywords[0] if keywords else name[:3]
query = rng.choice([
f"Find and explain {name}. I'm not sure of the exact name, but it's related to {vague_term}.",
f"I need to understand the function that handles {vague_term} in the codebase. Can you find it and explain what it does?",
f"There should be a function related to {vague_term}. Find it, analyze it, and tell me what it does.",
f"Search for code related to {vague_term} and find the main function. Explain its implementation.",
])
# Round 1: Vague search
reasoning1 = f"<|reasoning|>The user's query is somewhat vague β they're looking for something related to {vague_term}. Let me start with a broad search and then refine based on the results.<|end|>"
turn1 = {"content": f"{reasoning1}\n<|search|>{vague_term}<|end|>", "result": chunk['code'][:300]}
# Round 2: Refined search
reasoning2 = f"<|reasoning|>The initial search returned some results, but I need to be more specific. Looking at the results, I can see references to {name}. Let me search for that specifically to get the full implementation.<|end|>"
turn2 = {"content": f"{reasoning2}\n<|search|>{name}<|end|>", "result": chunk['code'][:500]}
# Round 3: Evidence
summary = summarize_code_briefly(chunk['code'], name, chunk['language'], chunk['type'])
calls = extract_function_calls(chunk['code'])
returns = extract_return_statements(chunk['code'])
evidence = f"<|reasoning|>After refining the search, I found {name}. It's {summary}. The function {'calls ' + ', '.join(calls[:3]) if calls else 'performs internal operations'} and {'returns ' + returns[0] if returns else 'has standard return paths'}. Let me compile the evidence.<|end|>\n<|evidence|>Found `{name}` through search refinement:\n- Initial search for '{vague_term}' returned partial matches\n- Refined search for '{name}' found the full implementation\n- {name} is {summary}\n- Key operations: {', '.join(calls[:4]) if calls else 'internal processing'}\n- Return paths: {'; '.join(returns[:2]) if returns else 'standard'}\n- Source: {chunk['filepath']}<|end|>\n<|finish|>"
turn3 = {"content": evidence}
return make_trace(query, [turn1, turn2, turn3])
# βββ Category Registry ββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
CATEGORIES = {
"implementation": ("Implementation Deep Dive", gen_implementation_deep_dive),
"cross_file": ("Cross-File Data Flow Tracing", gen_cross_file_tracing),
"architecture": ("Architecture & Component Mapping", gen_architecture_mapping),
"usage": ("Usage Pattern Analysis", gen_usage_analysis),
"error_handling": ("Error Handling & Failure Paths", gen_error_handling),
"api_contract": ("API Contract & Interface", gen_api_contract),
"dependency": ("Dependency Graph Mapping", gen_dependency_graph),
"data_structure": ("Data Structure Analysis", gen_data_structure_analysis),
"configuration": ("Configuration & Parameters", gen_configuration_params),
"performance": ("Performance-Critical Paths", gen_performance_paths),
"security": ("Security & Validation", gen_security_validation),
"design_patterns": ("Design Pattern Recognition", gen_design_patterns),
"debugging": ("Debugging Assistance", gen_debugging_assistance),
"comparison": ("Comparison & Tradeoffs", gen_comparison_tradeoffs),
"change_impact": ("Change Impact Analysis", gen_change_impact),
"complex_multi": ("Complex Multi-Step Analysis", gen_complex_multi_step),
"refinement": ("Search Refinement", gen_refinement_search),
}
# βββ Main Generation Loop βββββββββββββββββββββββββββββββββββββββββββββββββββββ
def generate_traces(category: str, count: int, seed: int, chunks: list) -> list[dict]:
"""Generate traces for a specific category."""
if category not in CATEGORIES:
print(f"Unknown category: {category}")
return []
cat_name, gen_func = CATEGORIES[category]
rng = random.Random(seed)
traces = []
attempts = 0
max_attempts = count * 5 # Allow retries for None returns
while len(traces) < count and attempts < max_attempts:
attempts += 1
try:
trace = gen_func(chunks, rng)
if trace is not None and trace.get("query") and len(trace.get("trace", [])) >= 4:
traces.append(trace)
except Exception as e:
# Silently skip failed generations
pass
return traces
def main():
parser = argparse.ArgumentParser(description="Generate high-quality SFT traces")
parser.add_argument("--category", type=str, default="all",
help="Category name or 'all'")
parser.add_argument("--count", type=int, default=500,
help="Number of traces per category")
parser.add_argument("--seed", type=int, default=42,
help="Random seed")
parser.add_argument("--output", type=str, default=OUTPUT_DIR,
help="Output directory")
args = parser.parse_args()
# Load chunks
print(f"Loading chunks from {CHUNKS_PATH}...")
chunks = [json.loads(l) for l in open(CHUNKS_PATH, encoding="utf-8")]
print(f" Loaded {len(chunks):,} chunks")
# Create output directory
os.makedirs(args.output, exist_ok=True)
# Determine categories
if args.category == "all":
cats = list(CATEGORIES.keys())
else:
cats = [args.category]
total_traces = 0
for i, cat in enumerate(cats):
cat_name, _ = CATEGORIES[cat]
seed = args.seed + i * 1000
print(f"\n[{i+1}/{len(cats)}] Generating {args.count} traces for '{cat_name}' (seed={seed})...")
traces = generate_traces(cat, args.count, seed, chunks)
output_file = os.path.join(args.output, f"traces_{cat}.jsonl")
with open(output_file, "w", encoding="utf-8") as f:
for trace in traces:
f.write(json.dumps(trace, ensure_ascii=False) + "\n")
print(f" Generated {len(traces)} traces β {output_file}")
total_traces += len(traces)
print(f"\nTotal: {total_traces} traces generated across {len(cats)} categories")
if __name__ == "__main__":
main()
|