devoppro commited on
Commit
9113dc0
·
verified ·
1 Parent(s): db85aa5

Update app.py

Browse files
Files changed (1) hide show
  1. app.py +35 -28
app.py CHANGED
@@ -4,15 +4,9 @@ import torch
4
  from transformers import AutoModelForCausalLM, AutoTokenizer
5
 
6
  MODEL_ID = "devoppro/FastLLM"
7
- TOKENIZER_FALLBACK_ID = "Qwen/Qwen2.5-0.5B" # FastLLM reuses the Qwen2.5 BPE vocab
8
-
9
- try:
10
- tokenizer = AutoTokenizer.from_pretrained(MODEL_ID, trust_remote_code=True)
11
- except Exception as e:
12
- print(f"Could not load tokenizer from {MODEL_ID} ({e}); "
13
- f"falling back to {TOKENIZER_FALLBACK_ID}")
14
- tokenizer = AutoTokenizer.from_pretrained(TOKENIZER_FALLBACK_ID)
15
 
 
16
  model = AutoModelForCausalLM.from_pretrained(
17
  MODEL_ID,
18
  torch_dtype=torch.float16,
@@ -23,19 +17,32 @@ model.eval()
23
 
24
 
25
  @spaces.GPU(duration=30)
26
- def generate(prompt, max_new_tokens, temperature, top_p):
27
- inputs = tokenizer(prompt, return_tensors="pt").to("cuda")
 
 
 
 
28
  with torch.no_grad():
29
- output_ids = model.generate(
30
- **inputs,
31
- max_new_tokens=int(max_new_tokens),
32
- do_sample=True,
33
- temperature=float(temperature),
34
- top_p=float(top_p),
35
- pad_token_id=tokenizer.eos_token_id,
36
- )
37
- full_text = tokenizer.decode(output_ids[0], skip_special_tokens=True)
38
- return full_text
 
 
 
 
 
 
 
 
 
39
 
40
 
41
  with gr.Blocks(title="FastLLM (150M) Demo") as demo:
@@ -43,7 +50,7 @@ with gr.Blocks(title="FastLLM (150M) Demo") as demo:
43
  """
44
  # FastLLM (150M) — Modern Causal Language Model
45
  A ~150M parameter decoder-only model (GQA, SwiGLU, RMSNorm, RoPE) from
46
- [devoppro/FastLLM](https://huggingface.co/devoppro/FastLLM).
47
  Small model, so expect small-model quality — this is a demo, not a chatbot.
48
  """
49
  )
@@ -54,26 +61,26 @@ with gr.Blocks(title="FastLLM (150M) Demo") as demo:
54
  value="Once upon a time,",
55
  lines=4,
56
  )
57
- max_new_tokens = gr.Slider(16, 512, value=128, step=16, label="Max new tokens")
58
  temperature = gr.Slider(0.1, 1.5, value=0.7, step=0.05, label="Temperature")
59
- top_p = gr.Slider(0.1, 1.0, value=0.95, step=0.05, label="Top-p")
60
  run_btn = gr.Button("Generate", variant="primary")
61
  with gr.Column():
62
  output = gr.Textbox(label="Output", lines=12)
63
 
64
  run_btn.click(
65
  fn=generate,
66
- inputs=[prompt, max_new_tokens, temperature, top_p],
67
  outputs=output,
68
  )
69
 
70
  gr.Examples(
71
  examples=[
72
- ["Once upon a time,", 128, 0.7, 0.95],
73
- ["The most important thing about machine learning is", 128, 0.7, 0.95],
74
- ["def fibonacci(n):", 96, 0.5, 0.9],
75
  ],
76
- inputs=[prompt, max_new_tokens, temperature, top_p],
77
  )
78
 
79
  if __name__ == "__main__":
 
4
  from transformers import AutoModelForCausalLM, AutoTokenizer
5
 
6
  MODEL_ID = "devoppro/FastLLM"
7
+ TOKENIZER_ID = "Qwen/Qwen2.5-0.5B" # FastLLM reuses the Qwen2.5 BPE vocab
 
 
 
 
 
 
 
8
 
9
+ tokenizer = AutoTokenizer.from_pretrained(TOKENIZER_ID)
10
  model = AutoModelForCausalLM.from_pretrained(
11
  MODEL_ID,
12
  torch_dtype=torch.float16,
 
17
 
18
 
19
  @spaces.GPU(duration=30)
20
+ def generate(prompt, max_new_tokens, temperature, top_k):
21
+ input_ids = tokenizer.encode(prompt, return_tensors="pt").to("cuda")
22
+ max_new_tokens = int(max_new_tokens)
23
+ temperature = float(temperature)
24
+ top_k = int(top_k)
25
+
26
  with torch.no_grad():
27
+ for _ in range(max_new_tokens):
28
+ outputs = model(input_ids)
29
+ logits = outputs["logits"][:, -1, :]
30
+
31
+ logits = logits / max(temperature, 1e-5)
32
+
33
+ if top_k > 0:
34
+ v, _ = torch.topk(logits, min(top_k, logits.size(-1)))
35
+ logits[logits < v[:, [-1]]] = -float("Inf")
36
+
37
+ probs = torch.softmax(logits, dim=-1)
38
+ next_token = torch.multinomial(probs, num_samples=1)
39
+
40
+ input_ids = torch.cat([input_ids, next_token], dim=-1)
41
+
42
+ if next_token.item() == tokenizer.eos_token_id:
43
+ break
44
+
45
+ return tokenizer.decode(input_ids[0], skip_special_tokens=True)
46
 
47
 
48
  with gr.Blocks(title="FastLLM (150M) Demo") as demo:
 
50
  """
51
  # FastLLM (150M) — Modern Causal Language Model
52
  A ~150M parameter decoder-only model (GQA, SwiGLU, RMSNorm, RoPE) from
53
+ [devoppro/FastLLM](https://huggingface.co/devoppro/FastLLM), still training.
54
  Small model, so expect small-model quality — this is a demo, not a chatbot.
55
  """
56
  )
 
61
  value="Once upon a time,",
62
  lines=4,
63
  )
64
+ max_new_tokens = gr.Slider(16, 256, value=60, step=8, label="Max new tokens")
65
  temperature = gr.Slider(0.1, 1.5, value=0.7, step=0.05, label="Temperature")
66
+ top_k = gr.Slider(0, 100, value=40, step=5, label="Top-k")
67
  run_btn = gr.Button("Generate", variant="primary")
68
  with gr.Column():
69
  output = gr.Textbox(label="Output", lines=12)
70
 
71
  run_btn.click(
72
  fn=generate,
73
+ inputs=[prompt, max_new_tokens, temperature, top_k],
74
  outputs=output,
75
  )
76
 
77
  gr.Examples(
78
  examples=[
79
+ ["Once upon a time,", 60, 0.7, 40],
80
+ ["Who are you?", 60, 0.7, 40],
81
+ ["The most important thing about machine learning is", 60, 0.7, 40],
82
  ],
83
+ inputs=[prompt, max_new_tokens, temperature, top_k],
84
  )
85
 
86
  if __name__ == "__main__":