small-test / src /synth_tools.py
Serveurperso's picture
Serveurperso HF Staff
small-test: a 95M multimodal fixture for the llama.cpp server CI
4a393d1
Raw History Blame Contribute Delete
35.1 kB
# synthetic multi step tool calling trajectories, rendered through the chat template
# every value the model must emit is either copied from the user prompt or from a previous tool result,
# tool descriptions carry "e.g." examples that never match the correct value
import json, random, sys, os
from multiprocessing import Pool
import render as R
CONS="bdfgklmnprstvz"; VOW="aeiou"
TLDS=["com","net","org","io","de","fr","it","es","nl","co","app","dev"]
NOUNS=["listing","invoice","ticket","parcel","article","venue","booking","shipment","course","recipe","device","repo","dataset","contract","vessel","station"]
METRICS=["price","score","rating","latency_ms","stock","temperature","occupancy","volume"]
SITES=["server","website","mirror","endpoint","host"]
CITIES_HINT=["city","town","region","area"]
CI_BLOCKLIST=["azzoo","kettle","bmi","exercise","trending","mobile_app","geolocation","halle-eins","galerie-deux","galleria-tre","stock_quote","advisor","commercial_prop","video_recommend","voltara","rivex","vltr","rvxn","halverton","fleet"]
def word(rng,n=None):
n=n or rng.randint(2,4)
return "".join(rng.choice(CONS)+rng.choice(VOW) for _ in range(n))
def name(rng): return word(rng).capitalize()
def domain(rng): return f"{word(rng)}-{word(rng)}.{rng.choice(TLDS)}" if rng.random()<0.5 else f"{word(rng)}{word(rng,1)}.{rng.choice(TLDS)}"
def ident(rng): return "".join(rng.choice("ABCDEFGHJKLMNPQRSTUVWXYZ") for _ in range(rng.randint(2,3)))+str(rng.randint(1000,999999))
def ticker(rng): return "".join(rng.choice("ABCDEFGHJKLMNPQRSTUVWXYZ") for _ in range(rng.randint(3,5)))
def money(rng): return round(rng.uniform(3,900),2)
def coord(rng): return round(rng.uniform(-60,70),4),round(rng.uniform(-150,150),4)
def tool(fname,desc,props,required):
return {"type":"function","function":{"name":fname,"description":desc,"parameters":{"type":"object","properties":props,"required":required}}}
def strp(desc,eg): return {"type":"string","description":f"{desc} (e.g. '{eg}')"}
def intp(desc,eg): return {"type":"integer","description":f"{desc} (e.g. {eg})"}
def nump(desc): return {"type":"number","description":desc}
def call(rng,fname,args,lead=None):
content=lead if (lead and rng.random()<0.25) else ""
return {"role":"assistant","content":content,"tool_calls":[{"type":"function","function":{"name":fname,"arguments":args}}]}
def calls(fnames_args):
return {"role":"assistant","content":"","tool_calls":[{"type":"function","function":{"name":f,"arguments":a}} for f,a in fnames_args]}
def result(obj): return {"role":"tool","content":json.dumps(obj)}
def list_items(rng,noun,n):
items=[{"id":ident(rng),"name":f"{name(rng)} {name(rng)}","rating":round(rng.uniform(2.5,5.0),1),"price":money(rng)} for _ in range(n)]
return items
# search -> detail on the best item (id copied from the result) -> one more lookup on the same id -> summary
def chain(rng):
noun=rng.choice(NOUNS); brand=name(rng)
q=f"{word(rng)} {word(rng)} {noun}"
search=f"{brand.lower()}_search_{noun}s"; get=f"{brand.lower()}_get_{noun}"; extra=rng.choice(["reviews","history","availability","specs"]); more=f"{brand.lower()}_get_{noun}_{extra}"
tools=[tool(search,f"Search {brand} {noun}s by keyword.",{"query":strp("Search keyword or phrase",f"{word(rng)} {word(rng)}"),"page":intp("Page number",1)},["query"]),
tool(get,f"Retrieve details about a {brand} {noun}.",{f"{noun}_id":strp(f"{brand} {noun} identifier",ident(rng))},[f"{noun}_id"]),
tool(more,f"Fetch {extra} for a {brand} {noun}.",{f"{noun}_id":strp(f"{brand} {noun} identifier",ident(rng)),"limit":intp("Maximum entries",5)},[f"{noun}_id"])]
rng.shuffle(tools)
crit=rng.choice(["top-rated","cheapest","most expensive"])
user=f"Please search {brand} for '{q}', then get full details on the {crit} result, and finally fetch its {extra}. Give me a short summary."
items=list_items(rng,noun,rng.randint(2,4))
best=max(items,key=lambda x:x["rating"]) if crit=="top-rated" else (min if crit=="cheapest" else max)(items,key=lambda x:x["price"])
detail={"id":best["id"],"name":best["name"],"price":best["price"],"rating":best["rating"],"in_stock":rng.random()<0.8,"seller":name(rng)}
extras={extra:[{"text":f"{name(rng)} says it is {rng.choice(['great','fine','slow','solid','noisy'])}.","stars":rng.randint(1,5)} for _ in range(rng.randint(1,3))]}
msgs=[{"role":"user","content":user},
call(rng,search,{"query":q},f"Searching {brand} for {q}."),result({"results":items,"page":1}),
call(rng,get,{f"{noun}_id":best["id"]},f"Looking up {best['id']}."),result(detail),
call(rng,more,{f"{noun}_id":best["id"]}),result(extras),
{"role":"assistant","content":f"The {crit} {noun} for '{q}' is {best['name']} ({best['id']}) at {best['price']} with a rating of {best['rating']}. It has {len(extras[extra])} {extra} entries, mostly {extras[extra][0]['text'].split(' is ')[-1].rstrip('.')}."}]
return msgs,tools
# one lookup per entity named in the prompt, sequential or parallel, then a summary filtered on a stated criterion
def fanout(rng):
kind=rng.choice(["host","ticker","id"])
n=rng.randint(2,4)
ents=[domain(rng) if kind=="host" else ticker(rng) if kind=="ticker" else ident(rng) for _ in range(n)]
fname={"host":"lookup_"+rng.choice(SITES)+"_location","ticker":"get_market_quote","id":"get_"+rng.choice(NOUNS)+"_status"}[kind]
pname={"host":"host","ticker":"symbol","id":"id"}[kind]
eg={"host":domain(rng),"ticker":ticker(rng),"id":ident(rng)}[kind]
tools=[tool(fname,{"host":"Look up the geographic location of a website's server.","ticker":"Retrieve the latest market quote for a ticker symbol.","id":"Check the current status of an item by identifier."}[kind],{pname:strp({"host":"Hostname or IP address","ticker":"Ticker symbol","id":"Item identifier"}[kind],eg)},[pname])]
if rng.random()<0.5: tools.append(tool("send_"+rng.choice(["report","alert","note"]),"Send a short text message to the user's inbox.",{"text":strp("Message body","hello")},["text"]))
res={}
for e in ents:
if kind=="host":
la,lo=coord(rng); res[e]={"host":e,"city":name(rng),"country":rng.choice(["DE","FR","IT","ES","NL","PL","AT"]),"lat":la,"lon":lo,"population":rng.choice([12000,45000,800000,2500000,3400000])}
elif kind=="ticker": res[e]={"symbol":e,"price":money(rng),"change_pct":round(rng.uniform(-6,6),2)}
else: res[e]={"id":e,"status":rng.choice(["open","closed","pending","shipped"]),"updated":f"2026-0{rng.randint(1,9)}-{rng.randint(10,28)}"}
lst=", ".join(ents[:-1])+f" and {ents[-1]}"
if kind=="host":
user=f"I got enquiries from {n} sites: {lst}. Look up where each one's server is located. Discard any that sit in a big city (over one million people) and report the coordinates of the others."
keep=[e for e in ents if res[e]["population"]<1000000]
final=("None of them is outside a big city." if not keep else "Outside big cities: "+"; ".join(f"{e} in {res[e]['city']} at {res[e]['lat']}, {res[e]['lon']}" for e in keep)+".")
elif kind=="ticker":
thr=rng.randint(20,400)
user=f"Get the latest quote for {lst}. Tell me which ones trade above ${thr}."
keep=[e for e in ents if res[e]["price"]>thr]
final=("None trades above the threshold." if not keep else "Above the threshold: "+", ".join(f"{e} at {res[e]['price']}" for e in keep)+".")
else:
user=f"Check the status of {lst} and tell me which ones are still open or pending."
keep=[e for e in ents if res[e]["status"] in("open","pending")]
final=("All of them are closed or shipped." if not keep else "Still active: "+", ".join(f"{e} ({res[e]['status']})" for e in keep)+".")
msgs=[{"role":"user","content":user}]
if rng.random()<0.5:
msgs.append(calls([(fname,{pname:e}) for e in ents]))
for e in ents: msgs.append(result(res[e]))
else:
for e in ents: msgs+=[call(rng,fname,{pname:e},f"Checking {e}."),result(res[e])]
msgs.append({"role":"assistant","content":final})
return msgs,tools
# numbered steps, a threshold decides which entity continues, later steps copy values from the prompt
def conditional(rng):
a,b=name(rng),name(rng); ta,tb=ticker(rng),ticker(rng); city=name(rng); topic=f"{word(rng)} {word(rng)}"
thr=rng.randint(20,300); pa,pb=money(rng),money(rng)
while (pa>thr)==(pb>thr): pa,pb=money(rng),money(rng)
win,wt=(a,ta) if pa>thr else (b,tb)
noun=rng.choice(NOUNS); kind=rng.choice(["office","garage","warehouse","studio"])
tools=[tool("get_market_quote","Retrieve the latest market quote for a ticker symbol.",{"symbol":strp("Ticker symbol",ticker(rng)),"interval":strp("Time interval","1day")},["symbol"]),
tool("search_incidents","Search public incident reports mentioning a company or product.",{"query":strp("Keyword or company name",name(rng)),"limit":intp("Maximum results",5)},["query"]),
tool("search_"+kind+"s",f"Search {kind}s available in a given city.",{"city":strp("City name",name(rng)),"max_price":nump("Maximum monthly price")},["city"]),
tool("get_"+noun+"_recommendations",f"Fetch recommended {noun}s about a topic.",{"query":strp("Topic",f"{word(rng)} {word(rng)}"),"count":intp("Number of results",3)},["query"])]
user=(f"I need a multi-step analysis:\n1. Get the latest quote for {a} ({ta}) and {b} ({tb}). If either is above ${thr}, continue with that company.\n"
f"2. Search incident reports about that company.\n3. Find {kind}s in {city}.\n4. Recommend {noun}s about {topic}.\nWork through all four steps and give me a concise summary.")
inc=[{"title":f"{win} {rng.choice(['outage','recall','breach','delay'])}","severity":rng.choice(["low","medium","high"])} for _ in range(rng.randint(1,3))]
props=[{"id":ident(rng),"address":f"{rng.randint(1,200)} {name(rng)} Street, {city}","price":rng.randint(800,9000)} for _ in range(rng.randint(1,3))]
recs=[{"title":f"{name(rng)} on {topic}","duration_min":rng.randint(5,60)} for _ in range(rng.randint(1,3))]
msgs=[{"role":"user","content":user}]
if rng.random()<0.5:
msgs.append(calls([("get_market_quote",{"symbol":ta}),("get_market_quote",{"symbol":tb})]))
msgs+=[result({"symbol":ta,"price":pa}),result({"symbol":tb,"price":pb})]
else:
msgs+=[call(rng,"get_market_quote",{"symbol":ta}),result({"symbol":ta,"price":pa}),call(rng,"get_market_quote",{"symbol":tb}),result({"symbol":tb,"price":pb})]
msgs+=[call(rng,"search_incidents",{"query":win},f"{win} is above ${thr}, checking incidents."),result({"incidents":inc}),
call(rng,"search_"+kind+"s",{"city":city}),result({"results":props}),
call(rng,"get_"+noun+"_recommendations",{"query":topic}),result({"results":recs}),
{"role":"assistant","content":f"{win} ({wt}) trades at {max(pa,pb)}, above ${thr}. Incidents: {', '.join(i['title'] for i in inc)}. {len(props)} {kind}(s) found in {city}, cheapest at {min(p['price'] for p in props)} per month. Recommended: {recs[0]['title']}."}]
return msgs,tools
# numeric inputs with units named in the prompt, then a follow-up lookup that reuses the computed value
def measure(rng):
w=round(rng.uniform(45,130),1); h=round(rng.uniform(1.45,2.05),2); age=rng.randint(18,80)
metric=rng.choice(["body_mass_index","calorie_need","dose","fitness_score"])
v=round(w/(h*h),1) if metric=="body_mass_index" else round(w*rng.uniform(20,35))
tools=[tool("calculate_"+metric,f"Calculate {metric.replace('_',' ')} from body measurements.",{"weight_kg":nump("Body weight in kilograms"),"height_m":nump("Height in meters"),"age":intp("Age in years",40)},["weight_kg","height_m"]),
tool("get_plan","Suggest a plan for a given category and level.",{"category":strp("Plan category","cardio"),"level":strp("Difficulty level: beginner, intermediate, expert","beginner")},["category"])]
cat=rng.choice(["strength","cardio","mobility","endurance"]); lvl=rng.choice(["beginner","intermediate","expert"])
user=rng.choice([f"I weigh {w} kg, I'm {h} m tall and {age} years old. Compute my {metric.replace('_',' ')} and then suggest a {lvl} {cat} plan.",
f"Height {h} m, weight {w} kg, age {age}. What is my {metric.replace('_',' ')}? Then give me a {cat} plan for a {lvl}."])
plan=[{"name":f"{name(rng)} {rng.choice(['press','pull','squat','run','stretch'])}","sets":rng.randint(2,5)} for _ in range(rng.randint(2,4))]
msgs=[{"role":"user","content":user},
call(rng,"calculate_"+metric,{"weight_kg":w,"height_m":h,"age":age}),result({metric:v}),
call(rng,"get_plan",{"category":cat,"level":lvl}),result({"plan":plan}),
{"role":"assistant","content":f"Your {metric.replace('_',' ')} is {v}. Suggested {lvl} {cat} plan: "+", ".join(f"{p['name']} ({p['sets']} sets)" for p in plan)+"."}]
return msgs,tools
# two discovery tools called in order on values from the prompt, then a cross reference in the summary
def discover(rng):
topic=f"{word(rng)} {word(rng)}"; cat=f"{name(rng)} {rng.choice(NOUNS)}"
ta="get_popular_"+rng.choice(["questions","threads","topics"]); tb="search_"+rng.choice(["store","catalog","directory"])
tools=[tool(ta,"Fetch the most popular community entries about a topic.",{"query":strp("Topic",f"{word(rng)} {word(rng)}"),"limit":intp("Maximum entries",5)},["query"]),
tool(tb,"Search the catalog for entries matching a category or keyword.",{"query":strp("Search keyword",f"{name(rng)} {word(rng)}"),"count":intp("Number of results",3)},["query"])]
rng.shuffle(tools)
user=f"I'm planning a session on {topic}. First find what people ask most about {topic}, then search the catalog for '{cat}' and tell me what is missing."
qs=[{"title":f"How do I {word(rng)} my {word(rng)}?","votes":rng.randint(3,900)} for _ in range(rng.randint(2,4))]
apps=[{"name":name(rng)+name(rng),"rating":round(rng.uniform(2,5),1)} for _ in range(rng.randint(1,3))]
msgs=[{"role":"user","content":user},
call(rng,ta,{"query":topic}),result({"entries":qs}),
call(rng,tb,{"query":cat}),result({"results":apps}),
{"role":"assistant","content":f"Top question about {topic}: \"{qs[0]['title']}\" ({qs[0]['votes']} votes). The catalog has {len(apps)} entries for '{cat}', best rated {max(apps,key=lambda x:x['rating'])['name']}, so nothing covers that question yet."}]
return msgs,tools
# parameter kinds: how a value is generated and how the prompt names it
def pval(rng,kind):
if kind=="city": return name(rng)
if kind=="ticker": return ticker(rng)
if kind=="domain": return domain(rng)
if kind=="email": return f"{word(rng)}.{word(rng)}@{word(rng)}.{rng.choice(TLDS)}"
if kind=="phrase": return " ".join(word(rng) for _ in range(rng.randint(1,3)))
if kind=="ident": return ident(rng)
if kind=="date": return f"2026-{rng.randint(1,12):02d}-{rng.randint(1,28):02d}"
if kind=="int": return rng.randint(1,20)
if kind=="money": return rng.choice([50,100,250,500,1000,2500])
if kind=="person": return f"{name(rng)} {name(rng)}"
if kind=="code": return rng.choice(["DE","FR","IT","ES","NL","PL","AT","PT","SE","JP","BR"])
if kind=="enum": return rng.choice(["low","medium","high"])
if kind=="word": return word(rng)
raise ValueError(kind)
PARAMS=[("city","city","City name"),("symbol","ticker","Ticker symbol"),("host","domain","Hostname"),("email","email","Recipient email address"),
("query","phrase","Search keyword or phrase"),("topic","phrase","Topic to look for"),("keyword","phrase","Keyword"),("user_id","ident","User identifier"),
("order_id","ident","Order identifier"),("ticket_id","ident","Ticket identifier"),("date","date","Date in YYYY-MM-DD"),("limit","int","Maximum number of results"),
("count","int","Number of entries"),("max_price","money","Maximum price"),("recipient","person","Full name of the recipient"),("country","code","Two letter country code"),
("priority","enum","Priority: low, medium, high"),("category","word","Category name"),("tag","word","Tag to filter on"),("region","city","Region or area name")]
VERBS=["get","fetch","search","list","check","lookup","create","send","update","find","compute","schedule"]
def pipeline_tool(rng,used):
# one random tool with 1 to 3 parameters, the first one is required, names are unique in the tool set
while True:
fname=f"{rng.choice(VERBS)}_{rng.choice(NOUNS)}{rng.choice(['','s','_details','_status','_history','_summary'])}"
if fname not in used: used.add(fname); break
ps=rng.sample(PARAMS,rng.randint(1,3)); props={}; req=[]
for i,(pn,kind,desc) in enumerate(ps):
eg=pval(rng,kind)
props[pn]={"type":"integer" if kind in("int","money") else "string","description":f"{desc} (e.g. {eg if kind in('int','money') else repr(eg)})"} if rng.random()<0.7 else {"type":"integer" if kind in("int","money") else "string","description":desc}
if i==0 or rng.random()<0.3: req.append(pn)
desc=f"{fname.split('_')[0].capitalize()} {' '.join(fname.split('_')[1:])} "+rng.choice(["from the backend.","for the current account.","in the catalog.","using the public API.","from the records."])
return tool(fname,desc,props,req),ps
def step_phrase(rng,fname,args,src,off=0):
verb,rest=fname.split("_",1); rest=rest.replace("_"," ")
parts=[]
for pn,v in args.items():
s=src[pn]
if s=="prompt": parts.append(rng.choice([f"{pn} {v!r}" if isinstance(v,str) else f"{pn} {v}",f"{pn}={v}",f"{pn} set to {v}"]))
elif s=="winner": parts.append(rng.choice([f"{pn} set to that company",f"{pn} being the one selected in step 1",f"that entity as {pn}"]))
else:
n=s+off; parts.append(rng.choice([f"the {pn} from step {n}",f"the {pn} you got in step {n}",f"the {pn} returned by the previous call" if s==len(src) else f"the {pn} from step {n}"]))
return f"{verb} {rest} with "+" and ".join(parts) if parts else f"{verb} {rest}"
def pipeline(rng):
used=set(); K=rng.choices([2,3,4,5,6,7],[1,2,3,3,2,1])[0]
tools=[]; specs=[]
for _ in range(K+rng.randint(0,2)):
t,ps=pipeline_tool(rng,used); tools.append(t); specs.append(ps)
order=rng.sample(range(len(tools)),K)
steps=[]; results=[]; gate=None
if rng.random()<0.35:
# gate: the same metric tool on two entities named in the prompt, the one above the threshold continues
ent=rng.choice([("symbol","ticker"),("host","domain"),("user_id","ident")]); pn,kind=ent
a,b=name(rng),name(rng); va,vb=pval(rng,kind),pval(rng,kind); thr=rng.randint(10,500)
ma,mb=rng.randint(1,1000),rng.randint(1,1000)
while (ma>thr)==(mb>thr): ma,mb=rng.randint(1,1000),rng.randint(1,1000)
gname=f"get_{rng.choice(['metric','quote','score','level'])}"; used.add(gname)
metric=rng.choice(["price","score","load","risk"])
tools.insert(0,tool(gname,f"Retrieve the current {metric} for one entity.",{pn:strp({"symbol":"Ticker symbol","host":"Hostname","user_id":"User identifier"}[pn],pval(rng,kind))},[pn]))
specs.insert(0,[(pn,kind,"")]); order=[i+1 for i in order]
gate=dict(gname=gname,pn=pn,a=a,b=b,va=va,vb=vb,thr=thr,ma=ma,mb=mb,metric=metric,win=a if ma>thr else b)
for k,ti in enumerate(order):
fname=tools[ti]["function"]["name"]; ps=specs[ti]; req=tools[ti]["function"]["parameters"]["required"]
args={}; src={}
for pn,kind,desc in ps:
if pn not in req and rng.random()<0.5: continue
prev=[(j,r) for j,r in enumerate(results) if pn in r]
if prev and rng.random()<0.8:
j,r=rng.choice(prev); args[pn]=r[pn]; src[pn]=j+1
elif gate and kind=="phrase" and "winner" not in src.values() and rng.random()<0.7: args[pn]=gate["win"]; src[pn]="winner"
else: args[pn]=pval(rng,kind); src[pn]="prompt"
res={"status":rng.choice(["ok","done","found"])}
for pn,kind,desc in rng.sample(PARAMS,rng.randint(1,3)): res[pn]=pval(rng,kind)
res[rng.choice(["result","value","total","score"])]=rng.choice([rng.randint(1,999),round(rng.uniform(0,100),1),name(rng),f"{name(rng)} {name(rng)}"])
steps.append((fname,args,src)); results.append(res)
opt=None
cand=[(j,k) for j in range(len(results)-1) for k,v in results[j].items() if isinstance(v,(int,float)) and not isinstance(v,bool)]
if rng.random()<0.3 and cand:
# the last step is conditional on a numeric field of an earlier result, half the time the condition is false and the step is skipped
j,key=rng.choice(cand); val=results[j][key]
thr=val-rng.randint(1,50) if rng.random()<0.5 else val+rng.randint(1,50)
opt=dict(j=j+1+(1 if gate else 0),key=key,thr=thr,do=val>thr)
numbered=rng.random()<0.5 or gate is not None or opt is not None
phrases=[step_phrase(rng,f,a,s,1 if gate else 0) for f,a,s in steps]
if opt: phrases[-1]=f"only if the {opt['key']} from step {opt['j']} is above {opt['thr']}, {phrases[-1]}"
if gate:
g=gate; phrases.insert(0,f"get the {g['metric']} for {g['a']} ({g['va']}) and {g['b']} ({g['vb']}). If either is above {g['thr']}, continue with that one")
if numbered: user=rng.choice(["Please do the following:","I need these steps done in order:","Run this pipeline for me:"])+"\n"+"\n".join(f"{i+1}. {p[0].upper()+p[1:]}." for i,p in enumerate(phrases))+"\n"+rng.choice(["Then summarize.","Give me a short summary at the end.","Report the results briefly."])
else: user=rng.choice(["First ","Start by ","Could you "])+phrases[0]+"".join(f", then {p}" for p in phrases[1:])+rng.choice([". Summarize the results.",", and give me a short recap.",". Keep the summary brief."])
msgs=[{"role":"user","content":user}]
if gate:
g=gate; ra={g["pn"]:g["va"],g["metric"]:g["ma"]}; rb={g["pn"]:g["vb"],g["metric"]:g["mb"]}
if rng.random()<0.5: msgs+=[calls([(g["gname"],{g["pn"]:g["va"]}),(g["gname"],{g["pn"]:g["vb"]})]),result(ra),result(rb)]
else: msgs+=[call(rng,g["gname"],{g["pn"]:g["va"]}),result(ra),call(rng,g["gname"],{g["pn"]:g["vb"]}),result(rb)]
k=0; n_do=len(steps)-(0 if opt is None or opt["do"] else 1)
while k<n_do:
fname,args,src=steps[k]
# two consecutive steps whose arguments all come from the prompt can go out as parallel calls
if k+1<n_do and all(v=="prompt" for v in src.values()) and all(v=="prompt" for v in steps[k+1][2].values()) and rng.random()<0.3:
msgs+=[calls([(fname,args),(steps[k+1][0],steps[k+1][1])]),result(results[k]),result(results[k+1])]; k+=2; continue
msgs+=[call(rng,fname,args,rng.choice([f"Calling {fname}.",f"Next step: {fname.replace('_',' ')}.","Working on the next step."])),result(results[k])]; k+=1
summ=[f"{gate['win']} is above {gate['thr']} with {gate['metric']} {max(gate['ma'],gate['mb'])}"] if gate else []
for (fname,args,src),res in list(zip(steps,results))[:n_do]:
key=[k for k in res if k!="status"][-1]; summ.append(f"{fname.replace('_',' ')} returned {key} {res[key]}")
if opt and not opt["do"]: summ.append(f"{opt['key']} from step {opt['j']} is {results[opt['j']-1-(1 if gate else 0)][opt['key']]}, not above {opt['thr']}, so {steps[-1][0].replace('_',' ')} was skipped")
msgs.append({"role":"assistant","content":rng.choice(["Done. ","All steps completed. ","Here is the summary: "])+"; ".join(summ)+"."})
if rng.random()<0.25:
# a follow-up turn that reuses the tools with a value from an earlier result
ti=rng.choice(order); fname=tools[ti]["function"]["name"]; ps=specs[ti]; pn,kind,desc=ps[0]
prev=[r for r in results if pn in r]
v=rng.choice(prev)[pn] if prev else pval(rng,kind)
msgs+=[{"role":"user","content":rng.choice([f"Thanks. Now {fname.replace('_',' ')} again with {pn} {v!r}.",f"One more thing: run {fname} for {pn} {v}."])},
call(rng,fname,{pn:v}),result({"status":"ok",pn:v,"note":name(rng)}),
{"role":"assistant","content":f"{fname.replace('_',' ')} for {pn} {v} is done, status ok."}]
return msgs,tools
# a question about the tools themselves is answered in plain text without any call
def capabilities(rng):
used=set(); tools=[pipeline_tool(rng,used)[0] for _ in range(rng.randint(2,4))]
names=[t["function"]["name"] for t in tools]
user=rng.choice(["What can you do for me here?","Which tools do you have available?","Hi, what are you able to help with?","List your functions please."])
msgs=[{"role":"user","content":user},{"role":"assistant","content":"I can "+", ".join(n.replace("_"," ") for n in names[:-1])+f" and {names[-1].replace('_',' ')}. Tell me what you need and I will run the right one."}]
return msgs,tools
SYSTEMS=["You are a helpful assistant.","You are a coding assistant.","You are a tool-calling agent.","You are a chatbot that uses tools/functions. Dont overthink things.","You are a tools-calling assistant.","You are an assistant with access to functions. Use them when they help.","Answer concisely."]
CITIES=["Paris","Berlin","Madrid","Rome","Lisbon","Vienna","Prague","Warsaw","Oslo","Dublin","Tokyo","Seoul","Bangkok","Hanoi","Cairo","Nairobi","Lagos","Lima","Bogota","Santiago","Montreal","Toronto","Denver","Austin","Seattle","Boston","Sydney","Auckland","Mumbai","Jakarta"]
COUNTRIES={"Paris":"France","Berlin":"Germany","Madrid":"Spain","Rome":"Italy","Lisbon":"Portugal","Vienna":"Austria","Prague":"Czechia","Warsaw":"Poland","Oslo":"Norway","Dublin":"Ireland","Tokyo":"Japan","Seoul":"South Korea","Bangkok":"Thailand","Hanoi":"Vietnam","Cairo":"Egypt","Nairobi":"Kenya","Lagos":"Nigeria","Lima":"Peru","Bogota":"Colombia","Santiago":"Chile","Montreal":"Canada","Toronto":"Canada","Denver":"USA","Austin":"USA","Seattle":"USA","Boston":"USA","Sydney":"Australia","Auckland":"New Zealand","Mumbai":"India","Jakarta":"Indonesia"}
# a vague request with a single tool: the tool is called with plausible arguments instead of answering in prose
def vague(rng):
used=set(); t,ps=pipeline_tool(rng,used); fname=t["function"]["name"]
tools=[t]+([pipeline_tool(rng,used)[0]] if rng.random()<0.3 else [])
rng.shuffle(tools)
user=rng.choice(["Write an example.","Give me an example.","Show me how it works.","Run it.","Do it.","Try it out.","Give me something.","Go ahead.","Can you do that for me?","Please proceed.","Test it.","Use the tool."])
args={pn:pval(rng,kind) for pn,kind,desc in ps if pn in t["function"]["parameters"]["required"] or rng.random()<0.3}
res={"status":"ok",rng.choice(["result","value","id"]):rng.choice([rng.randint(1,999),ident(rng),name(rng)])}
key=[k for k in res if k!="status"][0]
msgs=[{"role":"user","content":user},call(rng,fname,args),result(res),
{"role":"assistant","content":rng.choice([f"Done, {fname.replace('_',' ')} returned {key} {res[key]}.",f"Here is an example: {fname.replace('_',' ')} with {', '.join(f'{k} {v}' for k,v in args.items())} gives {key} {res[key]}.",f"I ran {fname.replace('_',' ')}, {key} is {res[key]}."])}]
return msgs,tools
# a code execution tool: the argument is real Python that does what the user asked
def code(rng):
fname=rng.choice(["python","run_python","execute_code","code_interpreter","exec_python","ipython"])
tools=[tool(fname,rng.choice(["Runs code in a Python interpreter and returns the output.","Execute Python code.","Runs the given Python code and returns its stdout."]),{"code":{"type":"string","description":"The Python code to run."}},["code"])]
a,b=rng.randint(1,99),rng.randint(1,99); n=rng.randint(3,12); w=word(rng); msg=rng.choice(["hello world","Hello, World!","hi there",f"hello {w}"])
q=rng.choice(['"',"'"])
kind=rng.choice(["hello","hello","add","range","len","upper","square"])
if kind=="hello": user=rng.choice([f"say {msg} with python",f"print {msg} using python",f"write python that prints {msg}",f"Say {msg} in Python."]); c=f"print({q}{msg}{q})"; out=msg
elif kind=="add": user=rng.choice([f"compute {a} + {b} in python",f"what is {a} plus {b}? use python",f"add {a} and {b} with python"]); c=f"print({a} + {b})"; out=str(a+b)
elif kind=="range": user=rng.choice([f"print the numbers from 1 to {n} in python",f"use python to list 1 to {n}"]); c=f"for i in range(1, {n+1}):\n print(i)"; out="\n".join(str(i) for i in range(1,n+1))
elif kind=="len": user=rng.choice([f"how many characters in {q}{w}{q}? use python",f"python: length of {q}{w}{q}"]); c=f"print(len({q}{w}{q}))"; out=str(len(w))
elif kind=="upper": user=rng.choice([f"uppercase {q}{w}{q} with python",f"use python to make {q}{w}{q} uppercase"]); c=f"print({q}{w}{q}.upper())"; out=w.upper()
else: user=rng.choice([f"square {a} in python",f"what is {a} squared? run python"]); c=f"print({a} ** 2)"; out=str(a*a)
msgs=[{"role":"user","content":user},call(rng,fname,{"code":c}),result({"stdout":out}),{"role":"assistant","content":rng.choice([f"Output: {out}",f"The code printed {out}.",f"Result: {out}"])}]
return msgs,tools
# a location argument copied exactly as the user wrote it, city alone or city with its country
def weather(rng):
city=rng.choice(CITIES); loc=city if rng.random()<0.6 else f"{city}, {COUNTRIES[city]}"
fname=rng.choice(["get_current_weather","get_weather","weather_lookup","fetch_forecast"])
props={"location":strp("The city and country, or city alone",f"{rng.choice(CITIES)}")}
if rng.random()<0.5: props["unit"]={"type":"string","enum":["celsius","fahrenheit"],"description":"Temperature unit"}
tools=[tool(fname,"Get the current weather for a location.",props,["location"])]
if rng.random()<0.4: tools.append(pipeline_tool(rng,{fname})[0]); rng.shuffle(tools)
user=rng.choice([f"What is the weather in {loc}?",f"What's the weather like in {loc} right now?",f"Is it raining in {loc}?",f"Give me the forecast for {loc}.",f"Temperature in {loc} today?"])
args={"location":loc}
if "unit" in props and rng.random()<0.4: args["unit"]=rng.choice(["celsius","fahrenheit"])
temp=rng.randint(-5,38); cond=rng.choice(["sunny","cloudy","light rain","clear","overcast","windy"])
msgs=[{"role":"user","content":user},call(rng,fname,args),result({"location":loc,"temperature_c":temp,"condition":cond}),
{"role":"assistant","content":f"It is {cond} in {loc}, {temp} degrees Celsius."}]
return msgs,tools
# two named companies, both quoted, a threshold picks one, then heterogeneous tools whose arguments come from the prompt,
# only the report search takes the selected company, the other tools take their own values
def compare_then(rng):
a,b=name(rng)+" "+rng.choice(["Inc.","Labs","Corp","Group","AG"]),name(rng)+" "+rng.choice(["Inc.","Labs","Corp","Group","AG"])
ta,tb=ticker(rng),ticker(rng); thr=rng.randint(20,400); pa,pb=money(rng),money(rng)
while (pa>thr)==(pb>thr): pa,pb=money(rng),money(rng)
win=a if pa>thr else b
city=rng.choice(CITIES) if rng.random()<0.5 else name(rng); topic=" ".join(word(rng) for _ in range(2)); kind=rng.choice(["office","warehouse","studio","rental","garage"])
q=rng.choice(["get_stock_price","get_share_price","get_quote","lookup_price"])
rep=rng.choice(["search_reports","search_news","find_incidents","search_filings"])
prop="search_"+kind+"s"; rec=rng.choice(["get_talks","list_articles","find_podcasts","get_tutorials"])
tools=[tool(q,"Get the latest share price for a ticker symbol.",{"symbol":strp("Ticker symbol",ticker(rng))},["symbol"]),
tool(rep,"Search public reports mentioning a company.",{"query":strp("Company name or keyword",name(rng)),"limit":intp("Maximum results",5)},["query"]),
tool(prop,f"Search {kind}s for rent in a city.",{"city":strp("City name",rng.choice(CITIES)),"max_price":intp("Maximum monthly price",2000)},["city"]),
tool(rec,"Get recommended content about a topic.",{"query":strp("Topic",f"{word(rng)} {word(rng)}"),"count":intp("Number of results",3)},["query"])]
rng.shuffle(tools)
user=rng.choice([f"I need a multi-step analysis:\n1. Get the share price of {a} ({ta}) and {b} ({tb}). If either is above ${thr}, continue with that company.\n2. Search reports about that company.\n3. Find {kind}s in {city}.\n4. Get content about {topic}.\nWork through all four steps and summarize.",
f"Check the share price for {a} ({ta}) and {b} ({tb}). Whichever is above ${thr}, search reports about it, then find {kind}s in {city}, then get content about {topic}. Summarize at the end."])
msgs=[{"role":"user","content":user}]
if rng.random()<0.5: msgs+=[calls([(q,{"symbol":ta}),(q,{"symbol":tb})]),result({"symbol":ta,"price":pa}),result({"symbol":tb,"price":pb})]
else: msgs+=[call(rng,q,{"symbol":ta}),result({"symbol":ta,"price":pa}),call(rng,q,{"symbol":tb}),result({"symbol":tb,"price":pb})]
reps=[{"title":f"{win} {rng.choice(['outage','recall','audit','delay'])}","severity":rng.choice(["low","medium","high"])} for _ in range(rng.randint(1,3))]
props=[{"id":ident(rng),"address":f"{rng.randint(1,200)} {name(rng)} Street, {city}","price":rng.randint(500,5000)} for _ in range(rng.randint(1,3))]
recs=[{"title":f"{name(rng)} on {topic}","minutes":rng.randint(5,60)} for _ in range(rng.randint(1,3))]
msgs+=[call(rng,rep,{"query":win},f"{win} is above ${thr}."),result({"results":reps}),
call(rng,prop,{"city":city}),result({"results":props}),
call(rng,rec,{"query":topic}),result({"results":recs}),
{"role":"assistant","content":f"{win} trades at {max(pa,pb)}, above ${thr}. Reports: {', '.join(r['title'] for r in reps)}. {len(props)} {kind}(s) in {city}, cheapest {min(p['price'] for p in props)} per month. Content on {topic}: {recs[0]['title']}."}]
return msgs,tools
PATTERNS=[(compare_then,0.01),(pipeline,0.39),(chain,0.11),(fanout,0.12),(conditional,0.07),(measure,0.05),(discover,0.05),(capabilities,0.05),(vague,0.06),(code,0.05),(weather,0.04)]
def gen(seed):
# a random name that collides with a CI test name is redrawn from a shifted seed
while True:
rng=random.Random(seed)
fn=rng.choices([p for p,_ in PATTERNS],[w for _,w in PATTERNS])[0]
msgs,tools=fn(rng)
# a system prompt in a third of the documents, the server tests always send one
if rng.random()<0.33: msgs=[{"role":"system","content":rng.choice(SYSTEMS)}]+msgs
text,spans=R.render(msgs,tools)
if not any(b in text.lower() for b in CI_BLOCKLIST): return json.dumps({"text":text,"spans":spans},ensure_ascii=False)
seed+=10**7
if __name__=="__main__":
model_dir,n,out=sys.argv[1],int(sys.argv[2]),sys.argv[3]
R.init(model_dir)
n_ev=max(1,n//100)
with Pool(16) as p: lines=list(p.imap(gen,range(n),chunksize=64))
with open(out+".jsonl","w") as f: f.write("\n".join(lines[n_ev:])+"\n")
with open(out+".eval.jsonl","w") as f: f.write("\n".join(lines[:n_ev])+"\n")
print(f"{out}: {n-n_ev} train, {n_ev} eval")