""" Nova 3.0 DeepReasoning — 7B Foundation Model Engine (Qwen2.5-7B / DeepSeek-R1) Runs 7B Foundation Model on AMD ROCm GPU (96 GB VRAM) with Hierarchical Chain-of-Thought (CoT) and Autonomous Python Tool Execution. """ import os import sys import time import torch from transformers import AutoTokenizer, AutoModelForCausalLM # Add project root to sys.path sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))) from src.inference.tool_engine import parse_and_execute_tools def run_nova3_7b_engine(prompt: str, model_name: str = "Qwen/Qwen2.5-7B-Instruct"): token = os.environ.get("HF_TOKEN") print(f"Loading '{model_name}' onto AMD ROCm GPU (96 GB VRAM)...", flush=True) t0 = time.time() tokenizer = AutoTokenizer.from_pretrained(model_name, token=token) model = AutoModelForCausalLM.from_pretrained( model_name, dtype=torch.bfloat16, device_map="cuda", token=token ) print(f"šŸš€ Loaded {model_name} onto GPU in {time.time()-t0:.2f}s!", flush=True) messages = [ {"role": "system", "content": "You are Nova 3.0 DeepReasoning, an advanced AI reasoning assistant. Break down complex math, calculus, logic, and code step-by-step. If code execution is needed, output code here."}, {"role": "user", "content": prompt} ] text_input = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True) model_inputs = tokenizer([text_input], return_tensors="pt").to("cuda") start_gen = time.time() generated_ids = model.generate( **model_inputs, max_new_tokens=300, temperature=0.7, top_p=0.9, do_sample=True ) generated_ids = [ output_ids[len(input_ids):] for input_ids, output_ids in zip(model_inputs.input_ids, generated_ids) ] response = tokenizer.batch_decode(generated_ids, skip_special_tokens=True)[0] gen_time = time.time() - start_gen print(f"\n=======================================================") print(f"🌌 NOVA 3.0 DEEPREASONING GENERATION OUTPUT ({gen_time:.2f}s)") print(f"=======================================================") print(response) # Execute Autonomous Tool Engine if block present clean_text, tool_results = parse_and_execute_tools(response) if tool_results: print("\n⚔ [Autonomous Tool Output]:") for res in tool_results: print(res) print("=======================================================\n") if __name__ == "__main__": test_prompt = "Hello how are you? Can you tell me how many r's are in the word strawberry and write a python code snippet to calculate the sum of 12**2 + 15**2?" run_nova3_7b_engine(test_prompt)