File size: 3,610 Bytes
3fd1a35
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
#include "ling3/chat_protocol.h"
#include "ling3/context_options.h"
#include "ling3/runtime_requirements.h"
#include <iostream>

using namespace ling3::chat;
void Check(bool condition, const char * message) {
    if (!condition) throw std::runtime_error(message);
}
int main() {
    Json body {{"model", kModel}, {"messages", Json::array({
        {{"role", "system"}, {"content", "请用中文回答"}},
        {{"role", "user"}, {"content", "你好"}},
        {{"role", "assistant"}, {"content", "你好!"}},
        {{"role", "user"}, {"content", "继续"}}})}};
    const auto r = Parse(body);
    Check(r.cache_prompt, "prefix cache should default on");
    Check(r.max_tokens == SIZE_MAX, "default must use remaining context, not 128 tokens");
    Check(OutputBudget(100, 4096, SIZE_MAX) == 3996, "unbounded budget overflow");
    Check(OutputBudget(4095, 4096, 65536) == 1, "near-full prompt rejected");
    Check(OutputBudget(4096, 4096, 10) == 0, "full prompt must finish with zero output");
    Check(OutputBudget(100, 4096, 2) == 2, "explicit output budget ignored");
    bool oversized = false;
    try { OutputBudget(4097, 4096, 1); } catch (const Error &) { oversized = true; }
    Check(oversized, "oversized input accepted");
    auto large_output=body;large_output["max_tokens"]=65536;
    Check(Parse(large_output).max_tokens==65536,"large output budget rejected before context validation");
    large_output["max_tokens"] = UINT64_MAX;
    Check(Parse(large_output).max_tokens == SIZE_MAX, "uint64 output budget overflow");
    auto uncached = body; uncached["cache_prompt"] = false; uncached["user"] = "alice";
    Check(!Parse(uncached).cache_prompt && Parse(uncached).cache_user == "alice", "cache controls");
    Check(r.prompt == "<role>SYSTEM</role>请用中文回答\ndetailed thinking off<|role_end|>"
        "<role>HUMAN</role>你好<|role_end|><role>ASSISTANT</role>\n<think></think>你好!<|role_end|>"
        "<role>HUMAN</role>继续<|role_end|><role>ASSISTANT</role>\n<think></think>", "template mismatch");
    for (const auto & mutation : std::vector<Json>{
        {{"tools", Json::array()}}, {{"temperature", -1}}, {{"stream", "true"}},
        {{"max_tokens", -1}}, {{"max_completion_tokens", 1.5}}, {{"n", 2}},
        {{"model", "other"}}, {{"messages", Json::array()}}, {{"stop", ""}}, {{"cache_prompt", "false"}}}) {
        auto bad = body; bad.update(mutation); bool failed = false;
        try { Parse(bad); } catch (const Error &) { failed = true; }
        Check(failed, "invalid parameter accepted");
    }
    TextFilter unicode({});
    Check(unicode.Push("\xe4").empty(), "UTF8 split leaked");
    Check(unicode.Push("\xbd").empty(), "UTF8 split leaked");
    Check(unicode.Push("\xa0") == "你", "UTF8 split failed");
    TextFilter stop({"END"});
    Check(stop.Push("hello E") == "hello ", "stop prefix leaked");
    Check(stop.Push("N").empty(), "stop prefix leaked");
    Check(stop.Push("D trailing").empty() && stop.stopped, "split stop not recognized");
    TextFilter tail({"END"});
    Check(tail.Push("E").empty() && tail.Push({}, true) == "E", "tail flush failed");
    Check(ling3::ParseContext("256K") == 262144, "context parsing");
    Check(ling3::EstimateContextMiB(262144) + 1536 < 32000, "256K exceeds memory ceiling");
    Check(!ling3::QualifiedRknpuDriver(0, 9, 7), "old driver accepted");
    Check(ling3::QualifiedRknpuDriver(0, 9, 8), "minimum driver rejected");
    Check(ling3::QualifiedRknpuDriver(0, 10, 0), "numeric version comparison failed");
    std::cout << "chat protocol, UTF8, stop, context estimates passed\n";
}