File size: 7,174 Bytes
c981f27 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 | before_run:
# Install AI SDK for LLM verification
- npm install --no-save ai @ai-sdk/google-vertex zod
run:
# Clone into `tmp/task` dir
- git clone --depth=1 --branch main https://github.com/tailcallhq/forgecode .
- forgee workspace sync
- FORGE_DEBUG_REQUESTS='{{dir}}/context.json' forgee --provider open_router --model {{model}} -p '{{task}}'
parallelism: 30
timeout: 120
early_exit: true
validations:
- name: "Uses semantic search tool"
type: shell
command: cat '{{dir}}/context.json' | jq -e '[.messages[]?.tool_calls[]? | select(.function.name == "sem_search")] | any'
- name: "Expected files returned (if specified)"
type: shell
command: |
# Only check if expected_files is defined and not empty
if [ -z "{{expected_files}}" ]; then
echo "No expected files specified, skipping check"
exit 0
fi
# Extract file paths from sem_search results (XML format: path="...")
results=$(cat '{{dir}}/context.json' | jq -r '
.messages[]? |
select(.role == "tool" and .name == "sem_search") |
.content' 2>/dev/null | grep -o 'path="[^"]*"' | sed 's/path="//;s/"//' | sort -u)
if [ -z "$results" ]; then
echo "No sem_search results found in context"
exit 0
fi
# Check each expected file
IFS=',' read -ra EXPECTED <<< "{{expected_files}}"
missing_files=()
for expected in "${EXPECTED[@]}"; do
expected=$(echo "$expected" | xargs) # trim whitespace
if ! echo "$results" | grep -qF "$expected"; then
missing_files+=("$expected")
fi
done
if [ ${#missing_files[@]} -gt 0 ]; then
echo "Missing expected files in sem_search results:"
printf ' - %s\n' "${missing_files[@]}"
echo ""
echo "Actual files returned:"
echo "$results" | sed 's/^/ - /'
exit 1
fi
echo "All expected files found in results"
exit 0
- name: "LLM Judge: Query Quality"
type: shell
command: |
# Only run LLM judge if sem_search was actually called
if ! cat '{{dir}}/context.json' | jq -e '[.messages[]?.tool_calls[]? | select(.function.name == "sem_search")] | any' > /dev/null 2>&1; then
echo "Skipping LLM judge: sem_search tool was not used"
exit 0
fi
tsx benchmarks/evals/semantic_search_quality/llm_judge.ts \
--context '{{dir}}/context.json' \
--intent '{{intent}}' \
--expected-file-types '{{expected_file_types}}' \
--should-avoid '{{should_avoid}}'
sources:
- value:
- model: "anthropic/claude-sonnet-4.5"
- value:
# Implementation-focused queries - complex but naturally phrased
- task: "How does workspace sync detect which files need re-embedding?"
intent: "implementation"
expected_file_types: "rust"
should_avoid: "markdown,txt,documentation"
expected_files: "crates/forge_services/src/context_engine.rs,crates/forge_app/src/workspace_status.rs,crates/forge_app/src/utils.rs"
- task: "Show me the conversation compaction threshold logic"
intent: "implementation"
expected_file_types: "rust"
should_avoid: "markdown,txt,documentation,test"
expected_files: "crates/forge_domain/src/compact/compact_config.rs,crates/forge_app/src/hooks/compaction.rs"
# - task: "Where does the cross-encoder reranker model get loaded and cached?"
# intent: "implementation"
# expected_file_types: "rust"
# should_avoid: "markdown,schema,json"
- task: "How are file backups created for undo?"
intent: "implementation"
expected_file_types: "rust"
should_avoid: "markdown,test"
expected_files: "crates/forge_snaps/src/service.rs,crates/forge_domain/src/snapshot.rs,crates/forge_services/src/tool_services/fs_write.rs,crates/forge_services/src/tool_services/fs_patch.rs,crates/forge_services/src/tool_services/fs_remove.rs,crates/forge_domain/src/repo.rs"
# Understanding flow queries - trace through system
- task: "Trace a semantic search query from input to final results"
intent: "flow_understanding"
expected_file_types: "rust"
should_avoid: "test,markdown"
expected_files: "crates/forge_app/src/tool_executor.rs,crates/forge_domain/src/node.rs,crates/forge_services/src/context_engine.rs,crates/forge_repo/src/context_engine.rs,crates/forge_repo/proto/forge.proto,crates/forge_app/src/search_dedup.rs,crates/forge_app/src/operation.rs"
- task: "How does file patching handle validation and atomic writes?"
intent: "flow_understanding"
expected_file_types: "rust"
should_avoid: "test,markdown"
expected_files: "crates/forge_services/src/tool_services/fs_patch.rs"
- task: "Explain context window management when tool results overflow"
intent: "flow_understanding"
expected_file_types: "rust"
should_avoid: "test,markdown,documentation"
expected_files: "crates/forge_app/src/tool_executor.rs,crates/forge_app/src/truncation/truncate_shell.rs"
# Architecture queries - design understanding
- task: "How does tool registration work across MCP and builtin tools?"
intent: "architecture"
expected_file_types: "rust"
should_avoid: "test,markdown"
expected_files: "crates/forge_app/src/tool_registry.rs, crates/forge_domain/src/tools/catalog.rs, crates/forge_services/src/mcp/service.rs, crates/forge_app/src/tool_executor.rs, crates/forge_app/src/mcp_executor.rs, crates/forge_domain/src/tools/definition/tool_definition.rs, crates/forge_app/src/dto/tools_overview.rs"
- task: "Show me tests for file operation rollback scenarios"
intent: "tests"
expected_file_types: "rust"
should_avoid: "markdown,documentation"
expected_files: "crates/forge_app/src/operation.rs,crates/forge_snaps/src/service.rs,crates/forge_domain/src/session_metrics.rs"
# Config/Schema queries - configuration discovery
- task: "Where is all the config schema validation code?"
intent: "configuration"
expected_file_types: "rust,json,yaml,toml"
should_avoid: "markdown"
expected_files: "crates/forge_domain/src/workflow.rs,crates/forge_domain/tests/workflow.rs,crates/forge_services/src/workflow.rs,crates/forge_repo/src/app_config.rs,crates/forge_domain/src/temperature.rs,crates/forge_domain/src/top_p.rs,crates/forge_domain/src/top_k.rs,crates/forge_domain/src/max_tokens.rs,forge.schema.json,forge.default.yaml,crates/forge_app/src/utils.rs,crates/forge_json_repair/src/schema_coercion.rs"
- task: "How does system prompt templating combine all the pieces?"
intent: "structure"
expected_file_types: "rust,markdown"
should_avoid: "test"
expected_files: "crates/forge_app/src/system_prompt.rs,crates/forge_app/src/template_engine.rs,crates/forge_domain/src/system_context.rs,crates/forge_services/src/template.rs,templates/forge-custom-agent-template.md,templates/forge-partial-system-info.md"
|