# synthetic multi step tool calling trajectories, rendered through the chat template # every value the model must emit is either copied from the user prompt or from a previous tool result, # tool descriptions carry "e.g." examples that never match the correct value import json, random, sys, os from multiprocessing import Pool import render as R CONS="bdfgklmnprstvz"; VOW="aeiou" TLDS=["com","net","org","io","de","fr","it","es","nl","co","app","dev"] NOUNS=["listing","invoice","ticket","parcel","article","venue","booking","shipment","course","recipe","device","repo","dataset","contract","vessel","station"] METRICS=["price","score","rating","latency_ms","stock","temperature","occupancy","volume"] SITES=["server","website","mirror","endpoint","host"] CITIES_HINT=["city","town","region","area"] CI_BLOCKLIST=["azzoo","kettle","bmi","exercise","trending","mobile_app","geolocation","halle-eins","galerie-deux","galleria-tre","stock_quote","advisor","commercial_prop","video_recommend","voltara","rivex","vltr","rvxn","halverton","fleet"] def word(rng,n=None): n=n or rng.randint(2,4) return "".join(rng.choice(CONS)+rng.choice(VOW) for _ in range(n)) def name(rng): return word(rng).capitalize() def domain(rng): return f"{word(rng)}-{word(rng)}.{rng.choice(TLDS)}" if rng.random()<0.5 else f"{word(rng)}{word(rng,1)}.{rng.choice(TLDS)}" def ident(rng): return "".join(rng.choice("ABCDEFGHJKLMNPQRSTUVWXYZ") for _ in range(rng.randint(2,3)))+str(rng.randint(1000,999999)) def ticker(rng): return "".join(rng.choice("ABCDEFGHJKLMNPQRSTUVWXYZ") for _ in range(rng.randint(3,5))) def money(rng): return round(rng.uniform(3,900),2) def coord(rng): return round(rng.uniform(-60,70),4),round(rng.uniform(-150,150),4) def tool(fname,desc,props,required): return {"type":"function","function":{"name":fname,"description":desc,"parameters":{"type":"object","properties":props,"required":required}}} def strp(desc,eg): return {"type":"string","description":f"{desc} (e.g. '{eg}')"} def intp(desc,eg): return {"type":"integer","description":f"{desc} (e.g. {eg})"} def nump(desc): return {"type":"number","description":desc} def call(rng,fname,args,lead=None): content=lead if (lead and rng.random()<0.25) else "" return {"role":"assistant","content":content,"tool_calls":[{"type":"function","function":{"name":fname,"arguments":args}}]} def calls(fnames_args): return {"role":"assistant","content":"","tool_calls":[{"type":"function","function":{"name":f,"arguments":a}} for f,a in fnames_args]} def result(obj): return {"role":"tool","content":json.dumps(obj)} def list_items(rng,noun,n): items=[{"id":ident(rng),"name":f"{name(rng)} {name(rng)}","rating":round(rng.uniform(2.5,5.0),1),"price":money(rng)} for _ in range(n)] return items # search -> detail on the best item (id copied from the result) -> one more lookup on the same id -> summary def chain(rng): noun=rng.choice(NOUNS); brand=name(rng) q=f"{word(rng)} {word(rng)} {noun}" search=f"{brand.lower()}_search_{noun}s"; get=f"{brand.lower()}_get_{noun}"; extra=rng.choice(["reviews","history","availability","specs"]); more=f"{brand.lower()}_get_{noun}_{extra}" tools=[tool(search,f"Search {brand} {noun}s by keyword.",{"query":strp("Search keyword or phrase",f"{word(rng)} {word(rng)}"),"page":intp("Page number",1)},["query"]), tool(get,f"Retrieve details about a {brand} {noun}.",{f"{noun}_id":strp(f"{brand} {noun} identifier",ident(rng))},[f"{noun}_id"]), tool(more,f"Fetch {extra} for a {brand} {noun}.",{f"{noun}_id":strp(f"{brand} {noun} identifier",ident(rng)),"limit":intp("Maximum entries",5)},[f"{noun}_id"])] rng.shuffle(tools) crit=rng.choice(["top-rated","cheapest","most expensive"]) user=f"Please search {brand} for '{q}', then get full details on the {crit} result, and finally fetch its {extra}. Give me a short summary." items=list_items(rng,noun,rng.randint(2,4)) best=max(items,key=lambda x:x["rating"]) if crit=="top-rated" else (min if crit=="cheapest" else max)(items,key=lambda x:x["price"]) detail={"id":best["id"],"name":best["name"],"price":best["price"],"rating":best["rating"],"in_stock":rng.random()<0.8,"seller":name(rng)} extras={extra:[{"text":f"{name(rng)} says it is {rng.choice(['great','fine','slow','solid','noisy'])}.","stars":rng.randint(1,5)} for _ in range(rng.randint(1,3))]} msgs=[{"role":"user","content":user}, call(rng,search,{"query":q},f"Searching {brand} for {q}."),result({"results":items,"page":1}), call(rng,get,{f"{noun}_id":best["id"]},f"Looking up {best['id']}."),result(detail), call(rng,more,{f"{noun}_id":best["id"]}),result(extras), {"role":"assistant","content":f"The {crit} {noun} for '{q}' is {best['name']} ({best['id']}) at {best['price']} with a rating of {best['rating']}. It has {len(extras[extra])} {extra} entries, mostly {extras[extra][0]['text'].split(' is ')[-1].rstrip('.')}."}] return msgs,tools # one lookup per entity named in the prompt, sequential or parallel, then a summary filtered on a stated criterion def fanout(rng): kind=rng.choice(["host","ticker","id"]) n=rng.randint(2,4) ents=[domain(rng) if kind=="host" else ticker(rng) if kind=="ticker" else ident(rng) for _ in range(n)] fname={"host":"lookup_"+rng.choice(SITES)+"_location","ticker":"get_market_quote","id":"get_"+rng.choice(NOUNS)+"_status"}[kind] pname={"host":"host","ticker":"symbol","id":"id"}[kind] eg={"host":domain(rng),"ticker":ticker(rng),"id":ident(rng)}[kind] tools=[tool(fname,{"host":"Look up the geographic location of a website's server.","ticker":"Retrieve the latest market quote for a ticker symbol.","id":"Check the current status of an item by identifier."}[kind],{pname:strp({"host":"Hostname or IP address","ticker":"Ticker symbol","id":"Item identifier"}[kind],eg)},[pname])] if rng.random()<0.5: tools.append(tool("send_"+rng.choice(["report","alert","note"]),"Send a short text message to the user's inbox.",{"text":strp("Message body","hello")},["text"])) res={} for e in ents: if kind=="host": la,lo=coord(rng); res[e]={"host":e,"city":name(rng),"country":rng.choice(["DE","FR","IT","ES","NL","PL","AT"]),"lat":la,"lon":lo,"population":rng.choice([12000,45000,800000,2500000,3400000])} elif kind=="ticker": res[e]={"symbol":e,"price":money(rng),"change_pct":round(rng.uniform(-6,6),2)} else: res[e]={"id":e,"status":rng.choice(["open","closed","pending","shipped"]),"updated":f"2026-0{rng.randint(1,9)}-{rng.randint(10,28)}"} lst=", ".join(ents[:-1])+f" and {ents[-1]}" if kind=="host": user=f"I got enquiries from {n} sites: {lst}. Look up where each one's server is located. Discard any that sit in a big city (over one million people) and report the coordinates of the others." keep=[e for e in ents if res[e]["population"]<1000000] final=("None of them is outside a big city." if not keep else "Outside big cities: "+"; ".join(f"{e} in {res[e]['city']} at {res[e]['lat']}, {res[e]['lon']}" for e in keep)+".") elif kind=="ticker": thr=rng.randint(20,400) user=f"Get the latest quote for {lst}. Tell me which ones trade above ${thr}." keep=[e for e in ents if res[e]["price"]>thr] final=("None trades above the threshold." if not keep else "Above the threshold: "+", ".join(f"{e} at {res[e]['price']}" for e in keep)+".") else: user=f"Check the status of {lst} and tell me which ones are still open or pending." keep=[e for e in ents if res[e]["status"] in("open","pending")] final=("All of them are closed or shipped." if not keep else "Still active: "+", ".join(f"{e} ({res[e]['status']})" for e in keep)+".") msgs=[{"role":"user","content":user}] if rng.random()<0.5: msgs.append(calls([(fname,{pname:e}) for e in ents])) for e in ents: msgs.append(result(res[e])) else: for e in ents: msgs+=[call(rng,fname,{pname:e},f"Checking {e}."),result(res[e])] msgs.append({"role":"assistant","content":final}) return msgs,tools # numbered steps, a threshold decides which entity continues, later steps copy values from the prompt def conditional(rng): a,b=name(rng),name(rng); ta,tb=ticker(rng),ticker(rng); city=name(rng); topic=f"{word(rng)} {word(rng)}" thr=rng.randint(20,300); pa,pb=money(rng),money(rng) while (pa>thr)==(pb>thr): pa,pb=money(rng),money(rng) win,wt=(a,ta) if pa>thr else (b,tb) noun=rng.choice(NOUNS); kind=rng.choice(["office","garage","warehouse","studio"]) tools=[tool("get_market_quote","Retrieve the latest market quote for a ticker symbol.",{"symbol":strp("Ticker symbol",ticker(rng)),"interval":strp("Time interval","1day")},["symbol"]), tool("search_incidents","Search public incident reports mentioning a company or product.",{"query":strp("Keyword or company name",name(rng)),"limit":intp("Maximum results",5)},["query"]), tool("search_"+kind+"s",f"Search {kind}s available in a given city.",{"city":strp("City name",name(rng)),"max_price":nump("Maximum monthly price")},["city"]), tool("get_"+noun+"_recommendations",f"Fetch recommended {noun}s about a topic.",{"query":strp("Topic",f"{word(rng)} {word(rng)}"),"count":intp("Number of results",3)},["query"])] user=(f"I need a multi-step analysis:\n1. Get the latest quote for {a} ({ta}) and {b} ({tb}). If either is above ${thr}, continue with that company.\n" f"2. Search incident reports about that company.\n3. Find {kind}s in {city}.\n4. Recommend {noun}s about {topic}.\nWork through all four steps and give me a concise summary.") inc=[{"title":f"{win} {rng.choice(['outage','recall','breach','delay'])}","severity":rng.choice(["low","medium","high"])} for _ in range(rng.randint(1,3))] props=[{"id":ident(rng),"address":f"{rng.randint(1,200)} {name(rng)} Street, {city}","price":rng.randint(800,9000)} for _ in range(rng.randint(1,3))] recs=[{"title":f"{name(rng)} on {topic}","duration_min":rng.randint(5,60)} for _ in range(rng.randint(1,3))] msgs=[{"role":"user","content":user}] if rng.random()<0.5: msgs.append(calls([("get_market_quote",{"symbol":ta}),("get_market_quote",{"symbol":tb})])) msgs+=[result({"symbol":ta,"price":pa}),result({"symbol":tb,"price":pb})] else: msgs+=[call(rng,"get_market_quote",{"symbol":ta}),result({"symbol":ta,"price":pa}),call(rng,"get_market_quote",{"symbol":tb}),result({"symbol":tb,"price":pb})] msgs+=[call(rng,"search_incidents",{"query":win},f"{win} is above ${thr}, checking incidents."),result({"incidents":inc}), call(rng,"search_"+kind+"s",{"city":city}),result({"results":props}), call(rng,"get_"+noun+"_recommendations",{"query":topic}),result({"results":recs}), {"role":"assistant","content":f"{win} ({wt}) trades at {max(pa,pb)}, above ${thr}. Incidents: {', '.join(i['title'] for i in inc)}. {len(props)} {kind}(s) found in {city}, cheapest at {min(p['price'] for p in props)} per month. Recommended: {recs[0]['title']}."}] return msgs,tools # numeric inputs with units named in the prompt, then a follow-up lookup that reuses the computed value def measure(rng): w=round(rng.uniform(45,130),1); h=round(rng.uniform(1.45,2.05),2); age=rng.randint(18,80) metric=rng.choice(["body_mass_index","calorie_need","dose","fitness_score"]) v=round(w/(h*h),1) if metric=="body_mass_index" else round(w*rng.uniform(20,35)) tools=[tool("calculate_"+metric,f"Calculate {metric.replace('_',' ')} from body measurements.",{"weight_kg":nump("Body weight in kilograms"),"height_m":nump("Height in meters"),"age":intp("Age in years",40)},["weight_kg","height_m"]), tool("get_plan","Suggest a plan for a given category and level.",{"category":strp("Plan category","cardio"),"level":strp("Difficulty level: beginner, intermediate, expert","beginner")},["category"])] cat=rng.choice(["strength","cardio","mobility","endurance"]); lvl=rng.choice(["beginner","intermediate","expert"]) user=rng.choice([f"I weigh {w} kg, I'm {h} m tall and {age} years old. Compute my {metric.replace('_',' ')} and then suggest a {lvl} {cat} plan.", f"Height {h} m, weight {w} kg, age {age}. What is my {metric.replace('_',' ')}? Then give me a {cat} plan for a {lvl}."]) plan=[{"name":f"{name(rng)} {rng.choice(['press','pull','squat','run','stretch'])}","sets":rng.randint(2,5)} for _ in range(rng.randint(2,4))] msgs=[{"role":"user","content":user}, call(rng,"calculate_"+metric,{"weight_kg":w,"height_m":h,"age":age}),result({metric:v}), call(rng,"get_plan",{"category":cat,"level":lvl}),result({"plan":plan}), {"role":"assistant","content":f"Your {metric.replace('_',' ')} is {v}. Suggested {lvl} {cat} plan: "+", ".join(f"{p['name']} ({p['sets']} sets)" for p in plan)+"."}] return msgs,tools # two discovery tools called in order on values from the prompt, then a cross reference in the summary def discover(rng): topic=f"{word(rng)} {word(rng)}"; cat=f"{name(rng)} {rng.choice(NOUNS)}" ta="get_popular_"+rng.choice(["questions","threads","topics"]); tb="search_"+rng.choice(["store","catalog","directory"]) tools=[tool(ta,"Fetch the most popular community entries about a topic.",{"query":strp("Topic",f"{word(rng)} {word(rng)}"),"limit":intp("Maximum entries",5)},["query"]), tool(tb,"Search the catalog for entries matching a category or keyword.",{"query":strp("Search keyword",f"{name(rng)} {word(rng)}"),"count":intp("Number of results",3)},["query"])] rng.shuffle(tools) user=f"I'm planning a session on {topic}. First find what people ask most about {topic}, then search the catalog for '{cat}' and tell me what is missing." qs=[{"title":f"How do I {word(rng)} my {word(rng)}?","votes":rng.randint(3,900)} for _ in range(rng.randint(2,4))] apps=[{"name":name(rng)+name(rng),"rating":round(rng.uniform(2,5),1)} for _ in range(rng.randint(1,3))] msgs=[{"role":"user","content":user}, call(rng,ta,{"query":topic}),result({"entries":qs}), call(rng,tb,{"query":cat}),result({"results":apps}), {"role":"assistant","content":f"Top question about {topic}: \"{qs[0]['title']}\" ({qs[0]['votes']} votes). The catalog has {len(apps)} entries for '{cat}', best rated {max(apps,key=lambda x:x['rating'])['name']}, so nothing covers that question yet."}] return msgs,tools # parameter kinds: how a value is generated and how the prompt names it def pval(rng,kind): if kind=="city": return name(rng) if kind=="ticker": return ticker(rng) if kind=="domain": return domain(rng) if kind=="email": return f"{word(rng)}.{word(rng)}@{word(rng)}.{rng.choice(TLDS)}" if kind=="phrase": return " ".join(word(rng) for _ in range(rng.randint(1,3))) if kind=="ident": return ident(rng) if kind=="date": return f"2026-{rng.randint(1,12):02d}-{rng.randint(1,28):02d}" if kind=="int": return rng.randint(1,20) if kind=="money": return rng.choice([50,100,250,500,1000,2500]) if kind=="person": return f"{name(rng)} {name(rng)}" if kind=="code": return rng.choice(["DE","FR","IT","ES","NL","PL","AT","PT","SE","JP","BR"]) if kind=="enum": return rng.choice(["low","medium","high"]) if kind=="word": return word(rng) raise ValueError(kind) PARAMS=[("city","city","City name"),("symbol","ticker","Ticker symbol"),("host","domain","Hostname"),("email","email","Recipient email address"), ("query","phrase","Search keyword or phrase"),("topic","phrase","Topic to look for"),("keyword","phrase","Keyword"),("user_id","ident","User identifier"), ("order_id","ident","Order identifier"),("ticket_id","ident","Ticket identifier"),("date","date","Date in YYYY-MM-DD"),("limit","int","Maximum number of results"), ("count","int","Number of entries"),("max_price","money","Maximum price"),("recipient","person","Full name of the recipient"),("country","code","Two letter country code"), ("priority","enum","Priority: low, medium, high"),("category","word","Category name"),("tag","word","Tag to filter on"),("region","city","Region or area name")] VERBS=["get","fetch","search","list","check","lookup","create","send","update","find","compute","schedule"] def pipeline_tool(rng,used): # one random tool with 1 to 3 parameters, the first one is required, names are unique in the tool set while True: fname=f"{rng.choice(VERBS)}_{rng.choice(NOUNS)}{rng.choice(['','s','_details','_status','_history','_summary'])}" if fname not in used: used.add(fname); break ps=rng.sample(PARAMS,rng.randint(1,3)); props={}; req=[] for i,(pn,kind,desc) in enumerate(ps): eg=pval(rng,kind) props[pn]={"type":"integer" if kind in("int","money") else "string","description":f"{desc} (e.g. {eg if kind in('int','money') else repr(eg)})"} if rng.random()<0.7 else {"type":"integer" if kind in("int","money") else "string","description":desc} if i==0 or rng.random()<0.3: req.append(pn) desc=f"{fname.split('_')[0].capitalize()} {' '.join(fname.split('_')[1:])} "+rng.choice(["from the backend.","for the current account.","in the catalog.","using the public API.","from the records."]) return tool(fname,desc,props,req),ps def step_phrase(rng,fname,args,src,off=0): verb,rest=fname.split("_",1); rest=rest.replace("_"," ") parts=[] for pn,v in args.items(): s=src[pn] if s=="prompt": parts.append(rng.choice([f"{pn} {v!r}" if isinstance(v,str) else f"{pn} {v}",f"{pn}={v}",f"{pn} set to {v}"])) elif s=="winner": parts.append(rng.choice([f"{pn} set to that company",f"{pn} being the one selected in step 1",f"that entity as {pn}"])) else: n=s+off; parts.append(rng.choice([f"the {pn} from step {n}",f"the {pn} you got in step {n}",f"the {pn} returned by the previous call" if s==len(src) else f"the {pn} from step {n}"])) return f"{verb} {rest} with "+" and ".join(parts) if parts else f"{verb} {rest}" def pipeline(rng): used=set(); K=rng.choices([2,3,4,5,6,7],[1,2,3,3,2,1])[0] tools=[]; specs=[] for _ in range(K+rng.randint(0,2)): t,ps=pipeline_tool(rng,used); tools.append(t); specs.append(ps) order=rng.sample(range(len(tools)),K) steps=[]; results=[]; gate=None if rng.random()<0.35: # gate: the same metric tool on two entities named in the prompt, the one above the threshold continues ent=rng.choice([("symbol","ticker"),("host","domain"),("user_id","ident")]); pn,kind=ent a,b=name(rng),name(rng); va,vb=pval(rng,kind),pval(rng,kind); thr=rng.randint(10,500) ma,mb=rng.randint(1,1000),rng.randint(1,1000) while (ma>thr)==(mb>thr): ma,mb=rng.randint(1,1000),rng.randint(1,1000) gname=f"get_{rng.choice(['metric','quote','score','level'])}"; used.add(gname) metric=rng.choice(["price","score","load","risk"]) tools.insert(0,tool(gname,f"Retrieve the current {metric} for one entity.",{pn:strp({"symbol":"Ticker symbol","host":"Hostname","user_id":"User identifier"}[pn],pval(rng,kind))},[pn])) specs.insert(0,[(pn,kind,"")]); order=[i+1 for i in order] gate=dict(gname=gname,pn=pn,a=a,b=b,va=va,vb=vb,thr=thr,ma=ma,mb=mb,metric=metric,win=a if ma>thr else b) for k,ti in enumerate(order): fname=tools[ti]["function"]["name"]; ps=specs[ti]; req=tools[ti]["function"]["parameters"]["required"] args={}; src={} for pn,kind,desc in ps: if pn not in req and rng.random()<0.5: continue prev=[(j,r) for j,r in enumerate(results) if pn in r] if prev and rng.random()<0.8: j,r=rng.choice(prev); args[pn]=r[pn]; src[pn]=j+1 elif gate and kind=="phrase" and "winner" not in src.values() and rng.random()<0.7: args[pn]=gate["win"]; src[pn]="winner" else: args[pn]=pval(rng,kind); src[pn]="prompt" res={"status":rng.choice(["ok","done","found"])} for pn,kind,desc in rng.sample(PARAMS,rng.randint(1,3)): res[pn]=pval(rng,kind) res[rng.choice(["result","value","total","score"])]=rng.choice([rng.randint(1,999),round(rng.uniform(0,100),1),name(rng),f"{name(rng)} {name(rng)}"]) steps.append((fname,args,src)); results.append(res) opt=None cand=[(j,k) for j in range(len(results)-1) for k,v in results[j].items() if isinstance(v,(int,float)) and not isinstance(v,bool)] if rng.random()<0.3 and cand: # the last step is conditional on a numeric field of an earlier result, half the time the condition is false and the step is skipped j,key=rng.choice(cand); val=results[j][key] thr=val-rng.randint(1,50) if rng.random()<0.5 else val+rng.randint(1,50) opt=dict(j=j+1+(1 if gate else 0),key=key,thr=thr,do=val>thr) numbered=rng.random()<0.5 or gate is not None or opt is not None phrases=[step_phrase(rng,f,a,s,1 if gate else 0) for f,a,s in steps] if opt: phrases[-1]=f"only if the {opt['key']} from step {opt['j']} is above {opt['thr']}, {phrases[-1]}" if gate: g=gate; phrases.insert(0,f"get the {g['metric']} for {g['a']} ({g['va']}) and {g['b']} ({g['vb']}). If either is above {g['thr']}, continue with that one") if numbered: user=rng.choice(["Please do the following:","I need these steps done in order:","Run this pipeline for me:"])+"\n"+"\n".join(f"{i+1}. {p[0].upper()+p[1:]}." for i,p in enumerate(phrases))+"\n"+rng.choice(["Then summarize.","Give me a short summary at the end.","Report the results briefly."]) else: user=rng.choice(["First ","Start by ","Could you "])+phrases[0]+"".join(f", then {p}" for p in phrases[1:])+rng.choice([". Summarize the results.",", and give me a short recap.",". Keep the summary brief."]) msgs=[{"role":"user","content":user}] if gate: g=gate; ra={g["pn"]:g["va"],g["metric"]:g["ma"]}; rb={g["pn"]:g["vb"],g["metric"]:g["mb"]} if rng.random()<0.5: msgs+=[calls([(g["gname"],{g["pn"]:g["va"]}),(g["gname"],{g["pn"]:g["vb"]})]),result(ra),result(rb)] else: msgs+=[call(rng,g["gname"],{g["pn"]:g["va"]}),result(ra),call(rng,g["gname"],{g["pn"]:g["vb"]}),result(rb)] k=0; n_do=len(steps)-(0 if opt is None or opt["do"] else 1) while kthr)==(pb>thr): pa,pb=money(rng),money(rng) win=a if pa>thr else b city=rng.choice(CITIES) if rng.random()<0.5 else name(rng); topic=" ".join(word(rng) for _ in range(2)); kind=rng.choice(["office","warehouse","studio","rental","garage"]) q=rng.choice(["get_stock_price","get_share_price","get_quote","lookup_price"]) rep=rng.choice(["search_reports","search_news","find_incidents","search_filings"]) prop="search_"+kind+"s"; rec=rng.choice(["get_talks","list_articles","find_podcasts","get_tutorials"]) tools=[tool(q,"Get the latest share price for a ticker symbol.",{"symbol":strp("Ticker symbol",ticker(rng))},["symbol"]), tool(rep,"Search public reports mentioning a company.",{"query":strp("Company name or keyword",name(rng)),"limit":intp("Maximum results",5)},["query"]), tool(prop,f"Search {kind}s for rent in a city.",{"city":strp("City name",rng.choice(CITIES)),"max_price":intp("Maximum monthly price",2000)},["city"]), tool(rec,"Get recommended content about a topic.",{"query":strp("Topic",f"{word(rng)} {word(rng)}"),"count":intp("Number of results",3)},["query"])] rng.shuffle(tools) user=rng.choice([f"I need a multi-step analysis:\n1. Get the share price of {a} ({ta}) and {b} ({tb}). If either is above ${thr}, continue with that company.\n2. Search reports about that company.\n3. Find {kind}s in {city}.\n4. Get content about {topic}.\nWork through all four steps and summarize.", f"Check the share price for {a} ({ta}) and {b} ({tb}). Whichever is above ${thr}, search reports about it, then find {kind}s in {city}, then get content about {topic}. Summarize at the end."]) msgs=[{"role":"user","content":user}] if rng.random()<0.5: msgs+=[calls([(q,{"symbol":ta}),(q,{"symbol":tb})]),result({"symbol":ta,"price":pa}),result({"symbol":tb,"price":pb})] else: msgs+=[call(rng,q,{"symbol":ta}),result({"symbol":ta,"price":pa}),call(rng,q,{"symbol":tb}),result({"symbol":tb,"price":pb})] reps=[{"title":f"{win} {rng.choice(['outage','recall','audit','delay'])}","severity":rng.choice(["low","medium","high"])} for _ in range(rng.randint(1,3))] props=[{"id":ident(rng),"address":f"{rng.randint(1,200)} {name(rng)} Street, {city}","price":rng.randint(500,5000)} for _ in range(rng.randint(1,3))] recs=[{"title":f"{name(rng)} on {topic}","minutes":rng.randint(5,60)} for _ in range(rng.randint(1,3))] msgs+=[call(rng,rep,{"query":win},f"{win} is above ${thr}."),result({"results":reps}), call(rng,prop,{"city":city}),result({"results":props}), call(rng,rec,{"query":topic}),result({"results":recs}), {"role":"assistant","content":f"{win} trades at {max(pa,pb)}, above ${thr}. Reports: {', '.join(r['title'] for r in reps)}. {len(props)} {kind}(s) in {city}, cheapest {min(p['price'] for p in props)} per month. Content on {topic}: {recs[0]['title']}."}] return msgs,tools PATTERNS=[(compare_then,0.01),(pipeline,0.39),(chain,0.11),(fanout,0.12),(conditional,0.07),(measure,0.05),(discover,0.05),(capabilities,0.05),(vague,0.06),(code,0.05),(weather,0.04)] def gen(seed): # a random name that collides with a CI test name is redrawn from a shifted seed while True: rng=random.Random(seed) fn=rng.choices([p for p,_ in PATTERNS],[w for _,w in PATTERNS])[0] msgs,tools=fn(rng) # a system prompt in a third of the documents, the server tests always send one if rng.random()<0.33: msgs=[{"role":"system","content":rng.choice(SYSTEMS)}]+msgs text,spans=R.render(msgs,tools) if not any(b in text.lower() for b in CI_BLOCKLIST): return json.dumps({"text":text,"spans":spans},ensure_ascii=False) seed+=10**7 if __name__=="__main__": model_dir,n,out=sys.argv[1],int(sys.argv[2]),sys.argv[3] R.init(model_dir) n_ev=max(1,n//100) with Pool(16) as p: lines=list(p.imap(gen,range(n),chunksize=64)) with open(out+".jsonl","w") as f: f.write("\n".join(lines[n_ev:])+"\n") with open(out+".eval.jsonl","w") as f: f.write("\n".join(lines[:n_ev])+"\n") print(f"{out}: {n-n_ev} train, {n_ev} eval")