Spaces:
Running on Zero
Running on Zero
Enable Qwen reasoning by default
Browse files
app.py
CHANGED
|
@@ -116,7 +116,7 @@ API_RESOURCES = {
|
|
| 116 |
"prompt": "string",
|
| 117 |
"max_new_tokens": "integer (1-256, default 64)",
|
| 118 |
"temperature": "number (0-2, default 0.7)",
|
| 119 |
-
"thinking": "boolean (default
|
| 120 |
},
|
| 121 |
"output": "JSON containing generated text and CPU timing",
|
| 122 |
},
|
|
@@ -601,9 +601,17 @@ def _generate_qwen38(prompt, max_new_tokens=64, temperature=0.7, thinking=False)
|
|
| 601 |
processor, model = _load_qwen38()
|
| 602 |
max_new_tokens = max(1, min(256, int(max_new_tokens)))
|
| 603 |
temperature = max(0.0, min(2.0, float(temperature)))
|
| 604 |
-
content = prompt
|
| 605 |
messages = [{"role": "user", "content": content}]
|
| 606 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 607 |
inputs = processor(text=[text], return_tensors="pt")
|
| 608 |
inputs = {key: value.to("cpu") if hasattr(value, "to") else value for key, value in inputs.items()}
|
| 609 |
t0 = time.time()
|
|
@@ -617,7 +625,7 @@ def _generate_qwen38(prompt, max_new_tokens=64, temperature=0.7, thinking=False)
|
|
| 617 |
elapsed = max(time.time() - t0, 0.001)
|
| 618 |
input_tokens = int(inputs["input_ids"].shape[-1])
|
| 619 |
generated = output[0, input_tokens:]
|
| 620 |
-
answer = processor.tokenizer.decode(generated, skip_special_tokens=True)
|
| 621 |
return {"text": answer, "prompt_tokens": input_tokens, "completion_tokens": int(generated.shape[-1]), "generation_tokens_per_sec": round(float(generated.shape[-1]) / elapsed, 2), "inference_time_sec": round(elapsed, 3), "device": "cpu", "dtype": "bfloat16", "thinking": bool(thinking)}
|
| 622 |
|
| 623 |
|
|
@@ -746,12 +754,11 @@ def _patched_create_app(blocks, **kwargs):
|
|
| 746 |
return JSONResponse({"error": "Missing 'prompt'"}, status_code=400)
|
| 747 |
job = _start_benchmark_job(
|
| 748 |
"qwen38",
|
| 749 |
-
|
| 750 |
-
False,
|
| 751 |
request={
|
| 752 |
-
"max_new_tokens": payload.get("max_new_tokens",
|
| 753 |
"temperature": payload.get("temperature", 0.7),
|
| 754 |
-
"thinking": payload.get("thinking",
|
| 755 |
},
|
| 756 |
)
|
| 757 |
return JSONResponse({"status": "accepted", "job_id": job, "poll": "/respite/text/qwen38?job_id=" + job}, status_code=202)
|
|
|
|
| 116 |
"prompt": "string",
|
| 117 |
"max_new_tokens": "integer (1-256, default 64)",
|
| 118 |
"temperature": "number (0-2, default 0.7)",
|
| 119 |
+
"thinking": "boolean (default true; Qwen3.8 is a reasoning model)",
|
| 120 |
},
|
| 121 |
"output": "JSON containing generated text and CPU timing",
|
| 122 |
},
|
|
|
|
| 601 |
processor, model = _load_qwen38()
|
| 602 |
max_new_tokens = max(1, min(256, int(max_new_tokens)))
|
| 603 |
temperature = max(0.0, min(2.0, float(temperature)))
|
| 604 |
+
content = prompt
|
| 605 |
messages = [{"role": "user", "content": content}]
|
| 606 |
+
try:
|
| 607 |
+
text = processor.apply_chat_template(
|
| 608 |
+
messages,
|
| 609 |
+
tokenize=False,
|
| 610 |
+
add_generation_prompt=True,
|
| 611 |
+
enable_thinking=True,
|
| 612 |
+
)
|
| 613 |
+
except TypeError:
|
| 614 |
+
text = processor.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
|
| 615 |
inputs = processor(text=[text], return_tensors="pt")
|
| 616 |
inputs = {key: value.to("cpu") if hasattr(value, "to") else value for key, value in inputs.items()}
|
| 617 |
t0 = time.time()
|
|
|
|
| 625 |
elapsed = max(time.time() - t0, 0.001)
|
| 626 |
input_tokens = int(inputs["input_ids"].shape[-1])
|
| 627 |
generated = output[0, input_tokens:]
|
| 628 |
+
answer = processor.tokenizer.decode(generated, skip_special_tokens=True).strip()
|
| 629 |
return {"text": answer, "prompt_tokens": input_tokens, "completion_tokens": int(generated.shape[-1]), "generation_tokens_per_sec": round(float(generated.shape[-1]) / elapsed, 2), "inference_time_sec": round(elapsed, 3), "device": "cpu", "dtype": "bfloat16", "thinking": bool(thinking)}
|
| 630 |
|
| 631 |
|
|
|
|
| 754 |
return JSONResponse({"error": "Missing 'prompt'"}, status_code=400)
|
| 755 |
job = _start_benchmark_job(
|
| 756 |
"qwen38",
|
| 757 |
+
None,
|
|
|
|
| 758 |
request={
|
| 759 |
+
"max_new_tokens": payload.get("max_new_tokens", 256),
|
| 760 |
"temperature": payload.get("temperature", 0.7),
|
| 761 |
+
"thinking": payload.get("thinking", True),
|
| 762 |
},
|
| 763 |
)
|
| 764 |
return JSONResponse({"status": "accepted", "job_id": job, "poll": "/respite/text/qwen38?job_id=" + job}, status_code=202)
|