Download scripts/test_nova3_7b.py from kings1/Nova: direct link, hf CLI and curl.
- Browser
- Download file 2.78 kB
-
https://huggingface.co/kings1/Nova/resolve/main/scripts/test_nova3_7b.py
- Command line
-
hf download hf://kings1/Nova/scripts/test_nova3_7b.py
-
curl -L -o test_nova3_7b.py https://huggingface.co/kings1/Nova/resolve/main/scripts/test_nova3_7b.py
2.78 kB
| """ | |
| Nova 3.0 DeepReasoning โ 7B Foundation Model Engine (Qwen2.5-7B / DeepSeek-R1) | |
| Runs 7B Foundation Model on AMD ROCm GPU (96 GB VRAM) with | |
| Hierarchical Chain-of-Thought (CoT) and Autonomous Python Tool Execution. | |
| """ | |
| import os | |
| import sys | |
| import time | |
| import torch | |
| from transformers import AutoTokenizer, AutoModelForCausalLM | |
| # Add project root to sys.path | |
| sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))) | |
| from src.inference.tool_engine import parse_and_execute_tools | |
| def run_nova3_7b_engine(prompt: str, model_name: str = "Qwen/Qwen2.5-7B-Instruct"): | |
| token = os.environ.get("HF_TOKEN") | |
| print(f"Loading '{model_name}' onto AMD ROCm GPU (96 GB VRAM)...", flush=True) | |
| t0 = time.time() | |
| tokenizer = AutoTokenizer.from_pretrained(model_name, token=token) | |
| model = AutoModelForCausalLM.from_pretrained( | |
| model_name, | |
| dtype=torch.bfloat16, | |
| device_map="cuda", | |
| token=token | |
| ) | |
| print(f"๐ Loaded {model_name} onto GPU in {time.time()-t0:.2f}s!", flush=True) | |
| messages = [ | |
| {"role": "system", "content": "You are Nova 3.0 DeepReasoning, an advanced AI reasoning assistant. Break down complex math, calculus, logic, and code step-by-step. If code execution is needed, output <python>code here</python>."}, | |
| {"role": "user", "content": prompt} | |
| ] | |
| text_input = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True) | |
| model_inputs = tokenizer([text_input], return_tensors="pt").to("cuda") | |
| start_gen = time.time() | |
| generated_ids = model.generate( | |
| **model_inputs, | |
| max_new_tokens=300, | |
| temperature=0.7, | |
| top_p=0.9, | |
| do_sample=True | |
| ) | |
| generated_ids = [ | |
| output_ids[len(input_ids):] for input_ids, output_ids in zip(model_inputs.input_ids, generated_ids) | |
| ] | |
| response = tokenizer.batch_decode(generated_ids, skip_special_tokens=True)[0] | |
| gen_time = time.time() - start_gen | |
| print(f"\n=======================================================") | |
| print(f"๐ NOVA 3.0 DEEPREASONING GENERATION OUTPUT ({gen_time:.2f}s)") | |
| print(f"=======================================================") | |
| print(response) | |
| # Execute Autonomous Tool Engine if <python> block present | |
| clean_text, tool_results = parse_and_execute_tools(response) | |
| if tool_results: | |
| print("\nโก [Autonomous Tool Output]:") | |
| for res in tool_results: | |
| print(res) | |
| print("=======================================================\n") | |
| if __name__ == "__main__": | |
| test_prompt = "Hello how are you? Can you tell me how many r's are in the word strawberry and write a python code snippet to calculate the sum of 12**2 + 15**2?" | |
| run_nova3_7b_engine(test_prompt) | |