Spaces:
Paused
Paused
Download tests/test_stream.py from Bit-Trading-Company/ASE-GLM: direct link, hf CLI and curl.
- Browser
- Download file 4.69 kB
-
https://huggingface.co/spaces/Bit-Trading-Company/ASE-GLM/resolve/main/tests/test_stream.py
- Command line
-
hf download hf://spaces/Bit-Trading-Company/ASE-GLM/tests/test_stream.py
-
curl -L -o test_stream.py https://huggingface.co/spaces/Bit-Trading-Company/ASE-GLM/resolve/main/tests/test_stream.py
4.69 kB
| """The SSE translation, checked against a stream GLM-5.3 actually produced. | |
| The fixture is a recorded response, not a hand-written one: the fields that | |
| matter here (`reasoning_content`, the per-chunk usage block, the trailing | |
| choice-less frame) are exactly the ones a plausible-looking fake would get | |
| subtly wrong. | |
| """ | |
| from __future__ import annotations | |
| import json | |
| from pathlib import Path | |
| import pytest | |
| from server.stream import Accumulated, build_payload, sse, translate | |
| FIXTURE = Path(__file__).parent / "fixtures" / "glm53_stream.sse" | |
| def run_fixture() -> tuple[list[tuple[str, str]], Accumulated]: | |
| acc = Accumulated() | |
| events: list[tuple[str, str]] = [] | |
| for line in FIXTURE.read_text().splitlines(): | |
| for frame in translate(line, acc): | |
| events.append(parse_frame(frame)) | |
| return events, acc | |
| def parse_frame(frame: str) -> tuple[str, str]: | |
| name = "message" | |
| data: list[str] = [] | |
| for line in frame.split("\n"): | |
| if line.startswith("event:"): | |
| name = line[6:].strip() | |
| elif line.startswith("data:"): | |
| data.append(line[5:].lstrip(" ")) | |
| return name, "\n".join(data) | |
| def test_reassembles_the_answer(): | |
| events, acc = run_fixture() | |
| assert acc.content.startswith("Hi there") | |
| assert "```python" in acc.content | |
| assert acc.reasoning, "GLM-5.3 streams a separate reasoning channel" | |
| assert acc.finished | |
| def test_content_and_reasoning_stay_on_separate_channels(): | |
| events, _ = run_fixture() | |
| deltas = [json.loads(d) for n, d in events if n == "delta"] | |
| assert any("content" in d for d in deltas) | |
| assert any("reasoning" in d for d in deltas) | |
| # A frame carrying only thinking must not also claim to carry an answer, | |
| # or the transcript prints the model's scratchpad as its reply. | |
| for d in deltas: | |
| if d.get("reasoning") and "content" in d: | |
| assert d["content"], "empty content should be dropped, not forwarded" | |
| def test_reports_every_usage_field_the_provider_sent(): | |
| _, acc = run_fixture() | |
| for key in ("promptTokens", "completionTokens", "reasoningTokens", | |
| "cachedTokens", "acceptedTokens", "rejectedTokens"): | |
| assert key in acc.stats, f"{key} was dropped" | |
| assert acc.stats["finishReason"] == "stop" | |
| assert acc.stats["requestId"].startswith("chatcmpl-") | |
| assert acc.stats["model"] == "GLM-5.3" | |
| def test_final_stats_frame_is_emitted_on_done(): | |
| events, _ = run_fixture() | |
| assert events[-1][0] == "stats" | |
| def test_ignores_noise_and_partial_lines(): | |
| acc = Accumulated() | |
| for junk in ("", " ", ": keep-alive", "data:", "data: {not json"): | |
| assert list(translate(junk, acc)) == [] | |
| assert acc.content == "" | |
| def test_surfaces_an_error_delivered_inside_a_200_stream(): | |
| acc = Accumulated() | |
| frames = list(translate( | |
| 'data: {"error": {"message": "model is overloaded"}}', acc)) | |
| assert len(frames) == 1 | |
| name, data = parse_frame(frames[0]) | |
| assert name == "error" | |
| assert data == "model is overloaded" | |
| def test_sse_frames_survive_a_newline_in_the_payload(): | |
| name, data = parse_frame(sse("error", "line one\nline two")) | |
| assert name == "error" | |
| assert data == "line one\nline two" | |
| class TestBuildPayload: | |
| BODY = {"messages": [{"role": "user", "content": "hi"}], "maxTokens": 4096} | |
| def test_always_asks_for_per_chunk_usage(self): | |
| p = build_payload(self.BODY, "zai-org/GLM-5.3", reasoning=True) | |
| assert p["stream"] is True | |
| assert p["stream_options"] == {"include_usage": True} | |
| def test_omits_reasoning_effort_when_max(self): | |
| # The card says the parameter defaults to `max`; sending it changes | |
| # nothing and is one more field a provider can reject. | |
| p = build_payload({**self.BODY, "reasoningEffort": "max"}, | |
| "zai-org/GLM-5.3", reasoning=True) | |
| assert "reasoning_effort" not in p | |
| def test_passes_the_efforts_that_need_passing(self, effort): | |
| p = build_payload({**self.BODY, "reasoningEffort": effort}, | |
| "zai-org/GLM-5.3", reasoning=True) | |
| assert p["reasoning_effort"] == effort | |
| def test_clears_thinking_for_chat(self): | |
| p = build_payload(self.BODY, "zai-org/GLM-5.3", reasoning=True) | |
| assert p["chat_template_kwargs"] == {"clear_thinking": True} | |
| def test_sends_no_reasoning_fields_to_a_plain_model(self): | |
| p = build_payload({**self.BODY, "reasoningEffort": "low"}, | |
| "some/model", reasoning=False) | |
| assert "reasoning_effort" not in p | |
| assert "chat_template_kwargs" not in p | |