File size: 1,416 Bytes
1591c33
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
"""MicroSupra-10k — inferência de demonstração (mesmos prompts do card original)."""
import os
from huggingface_hub import hf_hub_download
from transformers import LlamaForCausalLM, PreTrainedTokenizerFast
import torch

BASE = os.path.dirname(os.path.abspath(__file__))
OUT = f"{BASE}/out"

tok_path = hf_hub_download("SupraLabs/MicroSupra-1k", "tokenizer.json")
tokenizer = PreTrainedTokenizerFast(
    tokenizer_file=tok_path,
    bos_token="<s>", eos_token="</s>", pad_token="<pad>", unk_token="<unk>",
)
model = LlamaForCausalLM.from_pretrained(OUT)
model.eval()
print(f"[*] Parâmetros: {sum(p.numel() for p in model.parameters()):,}", flush=True)

prompts = [
    "My name is ",
    "The main concept of physics is ",
    "Question: What is the capital of France?\nAnswer: ",
]
for prompt in prompts:
    inputs = tokenizer(prompt, return_tensors="pt")
    with torch.no_grad():
        out = model.generate(
            input_ids=inputs["input_ids"],
            attention_mask=inputs["attention_mask"],
            max_new_tokens=120,
            do_sample=True,
            temperature=0.35,
            top_p=0.85,
            repetition_penalty=1.2,
            pad_token_id=tokenizer.pad_token_id,
            eos_token_id=tokenizer.eos_token_id,
        )
    print(f"\nPROMPT: {prompt!r}\nOUTPUT: {tokenizer.decode(out[0], skip_special_tokens=True)!r}", flush=True)
print("\n[*] DONE", flush=True)