InfiniteDev commited on
Commit
e27ab5d
·
verified ·
1 Parent(s): 5b77310

Upload app.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. app.py +28 -16
app.py CHANGED
@@ -19,8 +19,8 @@ from transformers import AutoModelForCausalLM, AutoTokenizer, TextIteratorStream
19
 
20
  MODEL_ID = "XHToken/Spark-X2.5-4B"
21
  MAX_CONTEXT_TOKENS = 32_768
22
- MIN_NEW_TOKENS = 128
23
- MAX_NEW_TOKENS = 3072
24
 
25
  # The model ships a custom `spark2_5` architecture via `trust_remote_code`.
26
  # Its custom attention path currently requires the eager implementation.
@@ -50,18 +50,19 @@ TASK_PRESETS = {
50
  }
51
 
52
 
53
- def split_reasoning(text: str) -> tuple[str, str]:
54
  """Split a partial/complete generation into (reasoning, answer).
55
 
56
- Handles three states: still thinking (no closing tag yet), finished
57
- thinking, and thinking disabled (template emits `</think>` immediately).
 
58
  """
59
- if "<think>" in text and "</think>" not in text:
60
- return text.split("<think>", 1)[1].strip(), ""
61
  if "</think>" in text:
62
  reasoning, answer = text.split("</think>", 1)
63
- return reasoning.replace("<think>", "").strip(), answer.strip()
64
- return "", text.replace("<think>", "").strip()
65
 
66
 
67
  def history_to_messages(history: list[object] | None) -> list[dict[str, str]]:
@@ -91,7 +92,7 @@ def _estimate_duration(
91
  history=None,
92
  system_prompt=None,
93
  enable_thinking=None,
94
- max_new_tokens=1024,
95
  temperature=None,
96
  top_p=None,
97
  *args,
@@ -173,8 +174,8 @@ def respond(
173
  last_emit = 0.0
174
  for new_text in streamer:
175
  completion += new_text
176
- reasoning, answer = split_reasoning(completion)
177
- display = answer if answer else ("_Thinking…_" if reasoning else "")
178
  now = time.perf_counter()
179
  if now - last_emit >= 0.1:
180
  last_emit = now
@@ -183,9 +184,20 @@ def respond(
183
  )
184
  thread.join()
185
 
186
- reasoning, answer = split_reasoning(completion)
187
- final = base + [{"role": "assistant", "content": answer or reasoning}]
188
- yield final, reasoning if enable_thinking else ""
 
 
 
 
 
 
 
 
 
 
 
189
 
190
 
191
  with gr.Blocks(title="Spark-X2.5-4B Code Assistant") as demo:
@@ -224,7 +236,7 @@ with gr.Blocks(title="Spark-X2.5-4B Code Assistant") as demo:
224
  )
225
  with gr.Row():
226
  max_new_tokens = gr.Slider(
227
- MIN_NEW_TOKENS, MAX_NEW_TOKENS, value=1024, step=128, label="Max new tokens"
228
  )
229
  temperature = gr.Slider(0, 1.5, value=1.0, step=0.05, label="Temperature")
230
  top_p = gr.Slider(0.1, 1.0, value=0.95, step=0.05, label="Top-p")
 
19
 
20
  MODEL_ID = "XHToken/Spark-X2.5-4B"
21
  MAX_CONTEXT_TOKENS = 32_768
22
+ MIN_NEW_TOKENS = 256
23
+ MAX_NEW_TOKENS = 4096
24
 
25
  # The model ships a custom `spark2_5` architecture via `trust_remote_code`.
26
  # Its custom attention path currently requires the eager implementation.
 
50
  }
51
 
52
 
53
+ def split_reasoning(text: str, enable_thinking: bool) -> tuple[str, str]:
54
  """Split a partial/complete generation into (reasoning, answer).
55
 
56
+ With thinking enabled the chat template appends `<think>` to the prompt, so
57
+ the generated text is pure reasoning until it emits `</think>` and then the
58
+ answer. With thinking disabled the model answers directly.
59
  """
60
+ if not enable_thinking:
61
+ return "", text.strip()
62
  if "</think>" in text:
63
  reasoning, answer = text.split("</think>", 1)
64
+ return reasoning.strip(), answer.strip()
65
+ return text.strip(), ""
66
 
67
 
68
  def history_to_messages(history: list[object] | None) -> list[dict[str, str]]:
 
92
  history=None,
93
  system_prompt=None,
94
  enable_thinking=None,
95
+ max_new_tokens=2048,
96
  temperature=None,
97
  top_p=None,
98
  *args,
 
174
  last_emit = 0.0
175
  for new_text in streamer:
176
  completion += new_text
177
+ reasoning, answer = split_reasoning(completion, bool(enable_thinking))
178
+ display = answer or ("_Thinking…_" if reasoning else "")
179
  now = time.perf_counter()
180
  if now - last_emit >= 0.1:
181
  last_emit = now
 
184
  )
185
  thread.join()
186
 
187
+ reasoning, answer = split_reasoning(completion, bool(enable_thinking))
188
+ if answer:
189
+ final_text = answer
190
+ elif reasoning:
191
+ final_text = (
192
+ reasoning
193
+ + "\n\n> ⚠️ The model stopped before finishing its reasoning — "
194
+ "increase **Max new tokens** and try again."
195
+ )
196
+ else:
197
+ final_text = ""
198
+ yield base + [{"role": "assistant", "content": final_text}], (
199
+ reasoning if enable_thinking else ""
200
+ )
201
 
202
 
203
  with gr.Blocks(title="Spark-X2.5-4B Code Assistant") as demo:
 
236
  )
237
  with gr.Row():
238
  max_new_tokens = gr.Slider(
239
+ MIN_NEW_TOKENS, MAX_NEW_TOKENS, value=2048, step=256, label="Max new tokens"
240
  )
241
  temperature = gr.Slider(0, 1.5, value=1.0, step=0.05, label="Temperature")
242
  top_p = gr.Slider(0.1, 1.0, value=0.95, step=0.05, label="Top-p")