File size: 3,540 Bytes
9c84f9d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
"""Output formatting utilities following dispatch CLI patterns."""

from typing import Any

from tabulate import tabulate


def format_status(status: str) -> str:
    symbols = {
        "completed": "✓ completed",
        "failed": "✗ failed",
        "running": "⟳ running",
        "evaluating": "⟳ evaluating",
        "pending": "○ pending",
    }
    return symbols.get(status, status)


def format_scores(scores: dict[str, float] | None) -> str:
    if not scores:
        return "-"
    return ", ".join(f"{k}: {v:.2f}" for k, v in scores.items())


def format_overall_score(score: float | None) -> str:
    if score is None:
        return "-"
    return f"{score:.4f}"


def format_projects_table(projects: list[dict[str, Any]]) -> str:
    rows = [[
        p.get("name", ""),
        p.get("display_name", ""),
        p.get("description", "")[:50],
        p.get("task_count", 0),
    ] for p in projects]
    return tabulate(rows, headers=["Name", "Display Name", "Description", "Tasks"], tablefmt="simple")


def format_prompts_table(prompts: list[dict[str, Any]]) -> str:
    rows = [[
        p.get("id", "")[:12],
        p.get("task", ""),
        f"v{p.get('version', '?')}",
        p.get("model", ""),
        p.get("temperature", ""),
        (p.get("system_prompt", "") or "")[:40] + "...",
    ] for p in prompts]
    return tabulate(rows, headers=["ID", "Task", "Version", "Model", "Temp", "Prompt Preview"], tablefmt="simple")


def format_runs_table(runs: list[dict[str, Any]]) -> str:
    rows = [[
        r.get("id", "")[:12],
        r.get("task", ""),
        format_status(r.get("status", "")),
        f"{r.get('completed_samples', 0)}/{r.get('total_samples', 0)}",
        r.get("created_at", "")[:19],
    ] for r in runs]
    return tabulate(rows, headers=["ID", "Task", "Status", "Progress", "Created"], tablefmt="simple")


def format_results_table(results: list[dict[str, Any]]) -> str:
    rows = []
    for r in results:
        inp = str(r.get("input", ""))[:40]
        out = str(r.get("output", ""))[:40]
        rows.append([
            r.get("sample_idx", ""),
            inp + ("..." if len(inp) >= 40 else ""),
            out + ("..." if len(out) >= 40 else ""),
            r.get("input_tokens", ""),
            r.get("output_tokens", ""),
            f"{r.get('inference_time_ms', 0):.0f}ms",
        ])
    return tabulate(rows, headers=["#", "Input", "Output", "In Tok", "Out Tok", "Time"], tablefmt="simple")


def format_evaluation_detail(evaluation: dict[str, Any]) -> str:
    lines = [
        f"Run ID:        {evaluation.get('run_id', '-')}",
        f"Overall Score: {format_overall_score(evaluation.get('overall_score'))}",
        f"Eval Model:    {evaluation.get('eval_model', '-')}",
        f"Evaluated At:  {(evaluation.get('evaluated_at') or '-')[:19]}",
    ]
    scores = evaluation.get("scores", {})
    if scores:
        lines.append(f"Scores:        {format_scores(scores)}")
    return "\n".join(lines)


def format_dashboard_table(dashboard: dict[str, Any]) -> str:
    rows = [[
        r.get("run_id", "")[:12],
        r.get("task", ""),
        f"v{r.get('prompt_version', '?')}",
        r.get("model", ""),
        format_overall_score(r.get("overall_score")),
        format_scores(r.get("scores")),
        r.get("total_samples", ""),
        r.get("date", "")[:10],
    ] for r in dashboard.get("runs", [])]
    return tabulate(rows, headers=["Run ID", "Task", "Version", "Model", "Score", "Metrics", "Samples", "Date"], tablefmt="simple")