Nova / scripts /test_nova3_7b.py
kings1's picture
Upload folder using huggingface_hub
3bc5889 verified
Raw History Blame Contribute Delete
2.78 kB
"""
Nova 3.0 DeepReasoning โ€” 7B Foundation Model Engine (Qwen2.5-7B / DeepSeek-R1)
Runs 7B Foundation Model on AMD ROCm GPU (96 GB VRAM) with
Hierarchical Chain-of-Thought (CoT) and Autonomous Python Tool Execution.
"""
import os
import sys
import time
import torch
from transformers import AutoTokenizer, AutoModelForCausalLM
# Add project root to sys.path
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..")))
from src.inference.tool_engine import parse_and_execute_tools
def run_nova3_7b_engine(prompt: str, model_name: str = "Qwen/Qwen2.5-7B-Instruct"):
token = os.environ.get("HF_TOKEN")
print(f"Loading '{model_name}' onto AMD ROCm GPU (96 GB VRAM)...", flush=True)
t0 = time.time()
tokenizer = AutoTokenizer.from_pretrained(model_name, token=token)
model = AutoModelForCausalLM.from_pretrained(
model_name,
dtype=torch.bfloat16,
device_map="cuda",
token=token
)
print(f"๐Ÿš€ Loaded {model_name} onto GPU in {time.time()-t0:.2f}s!", flush=True)
messages = [
{"role": "system", "content": "You are Nova 3.0 DeepReasoning, an advanced AI reasoning assistant. Break down complex math, calculus, logic, and code step-by-step. If code execution is needed, output <python>code here</python>."},
{"role": "user", "content": prompt}
]
text_input = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
model_inputs = tokenizer([text_input], return_tensors="pt").to("cuda")
start_gen = time.time()
generated_ids = model.generate(
**model_inputs,
max_new_tokens=300,
temperature=0.7,
top_p=0.9,
do_sample=True
)
generated_ids = [
output_ids[len(input_ids):] for input_ids, output_ids in zip(model_inputs.input_ids, generated_ids)
]
response = tokenizer.batch_decode(generated_ids, skip_special_tokens=True)[0]
gen_time = time.time() - start_gen
print(f"\n=======================================================")
print(f"๐ŸŒŒ NOVA 3.0 DEEPREASONING GENERATION OUTPUT ({gen_time:.2f}s)")
print(f"=======================================================")
print(response)
# Execute Autonomous Tool Engine if <python> block present
clean_text, tool_results = parse_and_execute_tools(response)
if tool_results:
print("\nโšก [Autonomous Tool Output]:")
for res in tool_results:
print(res)
print("=======================================================\n")
if __name__ == "__main__":
test_prompt = "Hello how are you? Can you tell me how many r's are in the word strawberry and write a python code snippet to calculate the sum of 12**2 + 15**2?"
run_nova3_7b_engine(test_prompt)