File size: 2,911 Bytes
728caeb
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
"""Drop-in PyTorch SDK adapter for independent-field tree attention."""
import time

import torch

from core.engine_torch import get_torch_engine, gpu_decorator
from tree_decode import FieldPlan, decode_fields, prefill


@gpu_decorator
@torch.inference_mode()
def run_parallel_generation_tree(context, schema, temperature=1.0):
    model, tokenizer, device = get_torch_engine()
    if model.config._attn_implementation not in ('sdpa', 'eager'):
        raise ValueError('Tree attention requires SDPA or eager attention with a 4D mask')

    def sync():
        if str(device).startswith('mps'):
            torch.mps.synchronize()

    sync()
    start = time.perf_counter()
    meta = schema.compile_parallel_metadata(tokenizer)
    suffixes = [list(map(int, row[:n])) for row, n in zip(meta['suffixes_batch'], meta['suffix_lengths'])]
    prompt = (f'<|im_start|>system\nClassify JSON attributes:\n{schema.to_parallel_schema_str()}<|im_end|>\n'
              f'<|im_start|>user\n{context}<|im_end|>\n<|im_start|>assistant\n{{\n')
    ids = tokenizer.encode(prompt, return_tensors='pt').to(device)
    plan = FieldPlan(suffixes, ids.shape[1], device, model.dtype, tokenizer.pad_token_id or 0)
    sync()
    pre_start = time.perf_counter()
    cache = prefill(model, ids)
    sync()
    pre_ms = (time.perf_counter() - pre_start) * 1000
    dec_start = time.perf_counter()
    logits = decode_fields(model, cache, plan, 'tree')[0]
    sync()
    dec_ms = (time.perf_counter() - dec_start) * 1000
    parsed, telemetry = {}, {}
    for i, (name, field) in enumerate(meta['field_items']):
        probs = (logits[i, meta['cands_per_field'][i]].float() / max(temperature, 1e-4)).softmax(-1).tolist()
        winner = max(range(len(probs)), key=probs.__getitem__)
        value = winner == 0 if field.field_type == 'boolean' else field.choices[winner]
        parsed[name] = {'value': value, 'prob': round(probs[winner], 4)}
        choices = [{'choice': c, 'probability': round(p, 4)} for c, p in zip(field.choices, probs)]
        telemetry[name] = dict(value=value, type=field.field_type, confidence=round(probs[winner], 4),
                               cardinality=field.cardinality, top_choices=sorted(choices, key=lambda c: c['probability'], reverse=True)[:5])
    return dict(mode='parallel_constrained_tree', elapsed_ms=round((time.perf_counter()-start)*1000, 2),
                prefill_ms=round(pre_ms, 2), suffix_eval_ms=round(dec_ms, 2), total_tokens_generated=0,
                sequential_forward_passes=1, is_valid_json=True, schema_match=True,
                parsed_json=parsed, field_telemetry=telemetry, has_calibrated_probabilities=False,
                num_fields=len(schema), device=str(device),
                candidate_collision_fields=[name for (name, _), collision in zip(meta['field_items'], meta['has_collisions']) if collision])