cazyundee commited on
Commit
ca39a8e
·
verified ·
1 Parent(s): 5be5b50

Enable Qwen reasoning by default

Browse files
Files changed (1) hide show
  1. app.py +15 -8
app.py CHANGED
@@ -116,7 +116,7 @@ API_RESOURCES = {
116
  "prompt": "string",
117
  "max_new_tokens": "integer (1-256, default 64)",
118
  "temperature": "number (0-2, default 0.7)",
119
- "thinking": "boolean (default false)",
120
  },
121
  "output": "JSON containing generated text and CPU timing",
122
  },
@@ -601,9 +601,17 @@ def _generate_qwen38(prompt, max_new_tokens=64, temperature=0.7, thinking=False)
601
  processor, model = _load_qwen38()
602
  max_new_tokens = max(1, min(256, int(max_new_tokens)))
603
  temperature = max(0.0, min(2.0, float(temperature)))
604
- content = prompt if thinking else prompt + "/no_think"
605
  messages = [{"role": "user", "content": content}]
606
- text = processor.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
 
 
 
 
 
 
 
 
607
  inputs = processor(text=[text], return_tensors="pt")
608
  inputs = {key: value.to("cpu") if hasattr(value, "to") else value for key, value in inputs.items()}
609
  t0 = time.time()
@@ -617,7 +625,7 @@ def _generate_qwen38(prompt, max_new_tokens=64, temperature=0.7, thinking=False)
617
  elapsed = max(time.time() - t0, 0.001)
618
  input_tokens = int(inputs["input_ids"].shape[-1])
619
  generated = output[0, input_tokens:]
620
- answer = processor.tokenizer.decode(generated, skip_special_tokens=True)
621
  return {"text": answer, "prompt_tokens": input_tokens, "completion_tokens": int(generated.shape[-1]), "generation_tokens_per_sec": round(float(generated.shape[-1]) / elapsed, 2), "inference_time_sec": round(elapsed, 3), "device": "cpu", "dtype": "bfloat16", "thinking": bool(thinking)}
622
 
623
 
@@ -746,12 +754,11 @@ def _patched_create_app(blocks, **kwargs):
746
  return JSONResponse({"error": "Missing 'prompt'"}, status_code=400)
747
  job = _start_benchmark_job(
748
  "qwen38",
749
- prompt,
750
- False,
751
  request={
752
- "max_new_tokens": payload.get("max_new_tokens", 64),
753
  "temperature": payload.get("temperature", 0.7),
754
- "thinking": payload.get("thinking", False),
755
  },
756
  )
757
  return JSONResponse({"status": "accepted", "job_id": job, "poll": "/respite/text/qwen38?job_id=" + job}, status_code=202)
 
116
  "prompt": "string",
117
  "max_new_tokens": "integer (1-256, default 64)",
118
  "temperature": "number (0-2, default 0.7)",
119
+ "thinking": "boolean (default true; Qwen3.8 is a reasoning model)",
120
  },
121
  "output": "JSON containing generated text and CPU timing",
122
  },
 
601
  processor, model = _load_qwen38()
602
  max_new_tokens = max(1, min(256, int(max_new_tokens)))
603
  temperature = max(0.0, min(2.0, float(temperature)))
604
+ content = prompt
605
  messages = [{"role": "user", "content": content}]
606
+ try:
607
+ text = processor.apply_chat_template(
608
+ messages,
609
+ tokenize=False,
610
+ add_generation_prompt=True,
611
+ enable_thinking=True,
612
+ )
613
+ except TypeError:
614
+ text = processor.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
615
  inputs = processor(text=[text], return_tensors="pt")
616
  inputs = {key: value.to("cpu") if hasattr(value, "to") else value for key, value in inputs.items()}
617
  t0 = time.time()
 
625
  elapsed = max(time.time() - t0, 0.001)
626
  input_tokens = int(inputs["input_ids"].shape[-1])
627
  generated = output[0, input_tokens:]
628
+ answer = processor.tokenizer.decode(generated, skip_special_tokens=True).strip()
629
  return {"text": answer, "prompt_tokens": input_tokens, "completion_tokens": int(generated.shape[-1]), "generation_tokens_per_sec": round(float(generated.shape[-1]) / elapsed, 2), "inference_time_sec": round(elapsed, 3), "device": "cpu", "dtype": "bfloat16", "thinking": bool(thinking)}
630
 
631
 
 
754
  return JSONResponse({"error": "Missing 'prompt'"}, status_code=400)
755
  job = _start_benchmark_job(
756
  "qwen38",
757
+ None,
 
758
  request={
759
+ "max_new_tokens": payload.get("max_new_tokens", 256),
760
  "temperature": payload.get("temperature", 0.7),
761
+ "thinking": payload.get("thinking", True),
762
  },
763
  )
764
  return JSONResponse({"status": "accepted", "job_id": job, "poll": "/respite/text/qwen38?job_id=" + job}, status_code=202)