Spaces:
Running on Zero
Running on Zero
Download app.py from akshayyy1/small-code-assistant: direct link, hf CLI and curl.
- Browser
- Download file 2.63 kB
-
https://huggingface.co/spaces/akshayyy1/small-code-assistant/resolve/main/app.py
- Command line
-
hf download hf://spaces/akshayyy1/small-code-assistant/app.py
-
curl -L -o app.py https://huggingface.co/spaces/akshayyy1/small-code-assistant/resolve/main/app.py
2.63 kB
| import spaces # noqa: F401 must be imported before torch | |
| import torch | |
| import gradio as gr | |
| from threading import Thread | |
| from transformers import AutoModelForCausalLM, AutoTokenizer, TextIteratorStreamer | |
| MODEL_ID = "Qwen/Qwen2.5-Coder-3B-Instruct" | |
| tokenizer = AutoTokenizer.from_pretrained(MODEL_ID) | |
| model = AutoModelForCausalLM.from_pretrained( | |
| MODEL_ID, | |
| torch_dtype=torch.bfloat16, | |
| attn_implementation="sdpa", | |
| ).to("cuda").eval() | |
| def respond( | |
| message: str, | |
| history: list[dict[str, str]], | |
| system_message: str, | |
| max_tokens: int, | |
| temperature: float, | |
| top_p: float, | |
| ) -> str: | |
| """Generate a reply from the coding assistant, streaming it token by token.""" | |
| messages = [{"role": "system", "content": system_message}] | |
| messages += history | |
| messages.append({"role": "user", "content": message}) | |
| # apply_chat_template returns a BatchEncoding on new transformers, a tensor on old | |
| ids = tokenizer.apply_chat_template( | |
| messages, add_generation_prompt=True, return_tensors="pt" | |
| ) | |
| input_ids = (ids["input_ids"] if not torch.is_tensor(ids) else ids).to("cuda") | |
| streamer = TextIteratorStreamer( | |
| tokenizer, skip_prompt=True, skip_special_tokens=True | |
| ) | |
| gen_kwargs = dict( | |
| input_ids=input_ids, streamer=streamer, max_new_tokens=max_tokens | |
| ) | |
| if temperature > 0.1: | |
| gen_kwargs.update(do_sample=True, temperature=temperature, top_p=top_p) | |
| else: | |
| gen_kwargs.update(do_sample=False) | |
| Thread(target=model.generate, kwargs=gen_kwargs, daemon=True).start() | |
| response = "" | |
| for new_text in streamer: | |
| response += new_text | |
| yield response | |
| demo = gr.ChatInterface( | |
| fn=respond, | |
| title="Small Code Assistant", | |
| description="Qwen2.5-Coder-3B-Instruct streaming on ZeroGPU.", | |
| additional_inputs=[ | |
| gr.Textbox( | |
| value="You are a helpful coding assistant.", label="System message" | |
| ), | |
| gr.Slider( | |
| minimum=64, maximum=2048, value=512, step=64, label="Max new tokens" | |
| ), | |
| gr.Slider( | |
| minimum=0.0, maximum=2.0, value=0.7, step=0.1, label="Temperature" | |
| ), | |
| gr.Slider( | |
| minimum=0.1, maximum=1.0, value=0.95, step=0.05, label="Top-p" | |
| ), | |
| ], | |
| examples=[ | |
| ["Write a Python function that checks if a string is a palindrome."], | |
| ["Explain what this regex does: ^\\d{3}-\\d{4}$"], | |
| ["Fix the bug: `for i in range(len(items)): print(items[i+1])`"], | |
| ], | |
| cache_examples=False, | |
| ) | |
| if __name__ == "__main__": | |
| demo.launch(mcp_server=True) | |