# DXA-LM Model Configurations v0.2 # Designed for lightweight CPU inference on an 8 GB Intel Mac default_model: baseline models: baseline: id: baseline provider: local backend: baseline model_path: "" context_size: 4096 temperature: 0.0 max_tokens: 1024 description: "Deterministic expert rule baseline engine (Zero-RAM overhead)" mock: id: mock provider: local backend: mock model_path: "" context_size: 2048 temperature: 0.0 max_tokens: 1024 description: "Fast mock backend for unit testing and pipeline verification" dxa_tiny_0_5b: id: dxa_tiny_0_5b provider: local backend: llama_cpp model_path: "models/Qwen2.5-0.5B-Instruct-Q4_K_M.gguf" repo_id: "Qwen/Qwen2.5-0.5B-Instruct-GGUF" filename: "qwen2.5-0.5b-instruct-q4_k_m.gguf" download_url: "https://huggingface.co/Qwen/Qwen2.5-0.5B-Instruct-GGUF/resolve/main/qwen2.5-0.5b-instruct-q4_k_m.gguf" size_mb: 398 context_size: 4096 temperature: 0.0 max_tokens: 1024 threads: 4 gpu_layers: 0 description: "0.5B parameter quantized model for fast edge stack detection" dxa_core_1b: id: dxa_core_1b provider: local backend: llama_cpp model_path: "models/dxa-core-1b-q4_k_m.gguf" context_size: 4096 temperature: 0.0 max_tokens: 1024 threads: 4 gpu_layers: 0 description: "1.0B parameter model for deployment planning and log parsing" dxa_expert_1_5b: id: dxa_expert_1_5b provider: local backend: llama_cpp model_path: "models/Qwen2.5-Coder-1.5B-Instruct-Q4_K_M.gguf" repo_id: "Qwen/Qwen2.5-Coder-1.5B-Instruct-GGUF" filename: "qwen2.5-coder-1.5b-instruct-q4_k_m.gguf" download_url: "https://huggingface.co/Qwen/Qwen2.5-Coder-1.5B-Instruct-GGUF/resolve/main/qwen2.5-coder-1.5b-instruct-q4_k_m.gguf" size_mb: 986 context_size: 4096 temperature: 0.0 max_tokens: 1024 threads: 4 gpu_layers: 0 description: "1.5B parameter specialized code model for multi-cause failure diagnosis" dxa_hybrid_1_5b: id: dxa_hybrid_1_5b provider: local backend: hybrid model_path: "models/Qwen2.5-Coder-1.5B-Instruct-Q4_K_M.gguf" base_llm: dxa_expert_1_5b context_size: 4096 temperature: 0.0 max_tokens: 1024 threads: 4 gpu_layers: 0 description: "Hybrid architecture: Deterministic inspection/detection + LLM reasoning and repair" dxa_reasoner_3b: id: dxa_reasoner_3b provider: local backend: llama_cpp model_path: "models/dxa-reasoner-3b-q4_k_m.gguf" context_size: 8192 temperature: 0.0 max_tokens: 1024 threads: 4 gpu_layers: 0 description: "3.0B parameter repair and configuration reconciliation model"