ASE-GLM / tests /test_stream.py
Bit-Trading-Company's picture
CI deploy local
17b22d9 verified
Raw History Blame Contribute Delete
4.69 kB
"""The SSE translation, checked against a stream GLM-5.3 actually produced.
The fixture is a recorded response, not a hand-written one: the fields that
matter here (`reasoning_content`, the per-chunk usage block, the trailing
choice-less frame) are exactly the ones a plausible-looking fake would get
subtly wrong.
"""
from __future__ import annotations
import json
from pathlib import Path
import pytest
from server.stream import Accumulated, build_payload, sse, translate
FIXTURE = Path(__file__).parent / "fixtures" / "glm53_stream.sse"
def run_fixture() -> tuple[list[tuple[str, str]], Accumulated]:
acc = Accumulated()
events: list[tuple[str, str]] = []
for line in FIXTURE.read_text().splitlines():
for frame in translate(line, acc):
events.append(parse_frame(frame))
return events, acc
def parse_frame(frame: str) -> tuple[str, str]:
name = "message"
data: list[str] = []
for line in frame.split("\n"):
if line.startswith("event:"):
name = line[6:].strip()
elif line.startswith("data:"):
data.append(line[5:].lstrip(" "))
return name, "\n".join(data)
def test_reassembles_the_answer():
events, acc = run_fixture()
assert acc.content.startswith("Hi there")
assert "```python" in acc.content
assert acc.reasoning, "GLM-5.3 streams a separate reasoning channel"
assert acc.finished
def test_content_and_reasoning_stay_on_separate_channels():
events, _ = run_fixture()
deltas = [json.loads(d) for n, d in events if n == "delta"]
assert any("content" in d for d in deltas)
assert any("reasoning" in d for d in deltas)
# A frame carrying only thinking must not also claim to carry an answer,
# or the transcript prints the model's scratchpad as its reply.
for d in deltas:
if d.get("reasoning") and "content" in d:
assert d["content"], "empty content should be dropped, not forwarded"
def test_reports_every_usage_field_the_provider_sent():
_, acc = run_fixture()
for key in ("promptTokens", "completionTokens", "reasoningTokens",
"cachedTokens", "acceptedTokens", "rejectedTokens"):
assert key in acc.stats, f"{key} was dropped"
assert acc.stats["finishReason"] == "stop"
assert acc.stats["requestId"].startswith("chatcmpl-")
assert acc.stats["model"] == "GLM-5.3"
def test_final_stats_frame_is_emitted_on_done():
events, _ = run_fixture()
assert events[-1][0] == "stats"
def test_ignores_noise_and_partial_lines():
acc = Accumulated()
for junk in ("", " ", ": keep-alive", "data:", "data: {not json"):
assert list(translate(junk, acc)) == []
assert acc.content == ""
def test_surfaces_an_error_delivered_inside_a_200_stream():
acc = Accumulated()
frames = list(translate(
'data: {"error": {"message": "model is overloaded"}}', acc))
assert len(frames) == 1
name, data = parse_frame(frames[0])
assert name == "error"
assert data == "model is overloaded"
def test_sse_frames_survive_a_newline_in_the_payload():
name, data = parse_frame(sse("error", "line one\nline two"))
assert name == "error"
assert data == "line one\nline two"
class TestBuildPayload:
BODY = {"messages": [{"role": "user", "content": "hi"}], "maxTokens": 4096}
def test_always_asks_for_per_chunk_usage(self):
p = build_payload(self.BODY, "zai-org/GLM-5.3", reasoning=True)
assert p["stream"] is True
assert p["stream_options"] == {"include_usage": True}
def test_omits_reasoning_effort_when_max(self):
# The card says the parameter defaults to `max`; sending it changes
# nothing and is one more field a provider can reject.
p = build_payload({**self.BODY, "reasoningEffort": "max"},
"zai-org/GLM-5.3", reasoning=True)
assert "reasoning_effort" not in p
@pytest.mark.parametrize("effort", ["low", "high"])
def test_passes_the_efforts_that_need_passing(self, effort):
p = build_payload({**self.BODY, "reasoningEffort": effort},
"zai-org/GLM-5.3", reasoning=True)
assert p["reasoning_effort"] == effort
def test_clears_thinking_for_chat(self):
p = build_payload(self.BODY, "zai-org/GLM-5.3", reasoning=True)
assert p["chat_template_kwargs"] == {"clear_thinking": True}
def test_sends_no_reasoning_fields_to_a_plain_model(self):
p = build_payload({**self.BODY, "reasoningEffort": "low"},
"some/model", reasoning=False)
assert "reasoning_effort" not in p
assert "chat_template_kwargs" not in p