File size: 4,863 Bytes
b01bf09
 
 
 
 
 
 
 
 
 
b1e674a
 
 
b01bf09
 
 
 
 
 
 
 
 
b1e674a
 
0ae18df
 
9a1b22c
 
b01bf09
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0ae18df
 
 
 
b1e674a
 
617604d
 
 
b01bf09
 
 
 
617604d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b1e674a
 
0ae18df
 
 
 
 
b1e674a
 
 
 
 
 
 
 
 
b01bf09
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
using System.Text.Json;
using System.Text.Json.Serialization;

namespace FindAJev.Bench;

public sealed record ModelSpec(
    [property: JsonPropertyName("id")] string Id,
    [property: JsonPropertyName("family")] string Family,
    [property: JsonPropertyName("repo")] string Repo,
    [property: JsonPropertyName("onnx")] string Onnx,
    [property: JsonPropertyName("precision")] string Precision,
    [property: JsonPropertyName("license")] string License = "unknown",
    [property: JsonPropertyName("sizeMb")] int SizeMb = 0);

/// <summary>One pre-tokenized decision (see tools/encode.py).</summary>
public sealed record Item(
    [property: JsonPropertyName("id")] string Id,
    [property: JsonPropertyName("domain")] string Domain,
    [property: JsonPropertyName("n")] int N,
    [property: JsonPropertyName("gold")] int Gold,
    [property: JsonPropertyName("ids")] long[] Ids,
    [property: JsonPropertyName("pos")] long[] Pos,
    [property: JsonPropertyName("qtype")] long QType = 0,
    [property: JsonPropertyName("task")] string Task = "",
    [property: JsonPropertyName("labels")] string[]? Labels = null,
    [property: JsonPropertyName("suite")] string Suite = "",
    [property: JsonPropertyName("optAttrs")] Dictionary<string, string[]>? OptAttrs = null,
    [property: JsonPropertyName("ctx")] Dictionary<string, string>? Ctx = null);

public sealed class RunResult
{
    public string Id { get; set; } = "";
    public string Family { get; set; } = "";
    public string Precision { get; set; } = "";
    public string State { get; set; } = "";
    public string? Error { get; set; }
    public int Threads { get; set; }
    public int Items { get; set; }
    public double LoadSeconds { get; set; }
    public double MeanMs { get; set; }
    public double P50Ms { get; set; }
    public double P95Ms { get; set; }
    public double ItemsPerSec { get; set; }
    public double Accuracy { get; set; }
    public Dictionary<string, double> AccuracyByDomain { get; set; } = new();
    public double PeakRssMb { get; set; }
    /// <summary>Non-default selections that make this run comparable only to runs with the same variant ("" = all suites, default parameters).</summary>
    public string Variant { get; set; } = "";
    public string[] SuitesRun { get; set; } = Array.Empty<string>();
    public Dictionary<string, long> PolicyParams { get; set; } = new();
    /// <summary>Per-suite outcome counts from the per-test state machines (empty when run without policies).</summary>
    public Dictionary<string, SuiteResult> Suites { get; set; } = new();
    /// <summary>Label-free Cedar oracle quality per suite: how well flags coincide with real errors (see OracleStats).</summary>
    public Dictionary<string, OracleStats> Oracle { get; set; } = new();
    public Dictionary<string, OracleRuleStat> OracleRules { get; set; } = new();
    public string Cpu { get; set; } = "";
    public string OrtVersion { get; set; } = "";
}

public sealed class OracleStats
{
    public int Tests { get; set; }
    public int Wrong { get; set; }            // tests where at least one head was predicted wrong
    public int Flagged { get; set; }          // tests the oracle rules flagged (no gold labels used)
    public int FlaggedWrong { get; set; }     // flagged AND really wrong  => precision = FlaggedWrong / Flagged, recall = FlaggedWrong / Wrong
    public int Unsafe { get; set; }           // enforcement failures (Cedar outcome differs unsafely from gold)
    public int UnsafeFlagged { get; set; }    // ... that the oracle had flagged beforehand
}

public sealed class OracleRuleStat
{
    public int Eligible { get; set; }         // tests the rule can apply to (its domain, or all tests for a generic rule)
    public int EligibleWrong { get; set; }    // of those, tests with a wrong label: the base rate a rule has to beat (lift = precision / base rate)
    public int Flagged { get; set; }
    public int FlaggedWrong { get; set; }
    public int FiredOnGold { get; set; }      // tests where the rule also fires on the GOLD labels (a sound rule: ~0)
}

public sealed class SuiteResult
{
    public int Heads { get; set; }              // model calls (single-label decisions) in this suite
    public double Accuracy { get; set; }        // head-level argmax accuracy
    public double MeanMs { get; set; }
    public double P50Ms { get; set; }
    public double P95Ms { get; set; }
    public int Tests { get; set; }
    public int Correct { get; set; }
    public int WrongButSafe { get; set; }
    public int Overblocked { get; set; }
    public int Unsafe { get; set; }
    public int Misclassified { get; set; }
    public int Errored { get; set; }
}

public static class Json
{
    public static readonly JsonSerializerOptions Opts = new() { WriteIndented = true, PropertyNamingPolicy = JsonNamingPolicy.CamelCase };
}