# 🛠️ Complete Command Reference ## 1. Create Project Directory ```bash mkdir -p /root/ai-tuning cd /root/ai-tuning cat /etc/os-release /opt/python310/bin/python3 -m venv /root/ai-tuning/.venv310 source /root/ai-tuning/.venv310/bin/activate pip install numpy==1.26.4 pip install pyarrow==20.0.0 pip install torch==2.6.0 pip install transformers==4.46.3 pip install datasets==2.21.0 pip install peft==0.13.2 pip install accelerate==1.0.1 pip install requests python - <<'PY' import numpy import pyarrow import torch import transformers import datasets import peft import accelerate print("NumPy :", numpy.__version__) print("PyArrow :", pyarrow.__version__) print("PyTorch :", torch.__version__) print("Transformers:", transformers.__version__) print("Datasets :", datasets.__version__) print("PEFT :", peft.__version__) print("Accelerate :", accelerate.__version__) print("CUDA :", torch.cuda.is_available()) PY . Hugging Face Login hf auth whoami hf auth login Create DevOps Dataset /root/ai-tuning/devops_dataset.jsonl ls -lh /root/ai-tuning/devops_dataset.jsonl wc -l /root/ai-tuning/devops_dataset.jsonl Validate JSONL python3 -m json.tool devops_dataset.jsonl python3 - <<'PY' import json file = "/root/ai-tuning/devops_dataset.jsonl" errors = 0 with open(file, encoding="utf-8") as f: for line_no, line in enumerate(f, 1): line = line.strip() if not line: continue try: json.loads(line) except Exception as e: print(f"Invalid JSON at line {line_no}: {e}") errors += 1 if errors == 0: print("JSONL validation successful") else: print(f"Validation failed: {errors} errors") PY Create Train / Validation Split python3 - <<'PY' import json import random input_file = "/root/ai-tuning/devops_dataset.jsonl" train_file = "/root/ai-tuning/train.jsonl" validation_file = "/root/ai-tuning/validation.jsonl" with open(input_file, encoding="utf-8") as f: data = [json.loads(line) for line in f if line.strip()] random.seed(42) random.shuffle(data) split = int(len(data) * 0.8) train = data[:split] validation = data[split:] with open(train_file, "w", encoding="utf-8") as f: for item in train: f.write(json.dumps(item, ensure_ascii=False) + "\n") with open(validation_file, "w", encoding="utf-8") as f: for item in validation: f.write(json.dumps(item, ensure_ascii=False) + "\n") print("Total :", len(data)) print("Train :", len(train)) print("Validation :", len(validation)) PY Check Training Script ls -lh /root/ai-tuning/train_lora.py less /root/ai-tuning/train_lora.py . Run LoRA Training cd /root/ai-tuning /opt/python310/bin/python3 train_lora.py Check LoRA Output ls -lah /root/ai-tuning/lora-output ls -lh /root/ai-tuning/lora-output/adapter_model.safetensors cat /root/ai-tuning/lora-output/adapter_config.json du -sh /root/ai-tuning/lora-output Test LoRA Adapter find /root/ai-tuning/lora-output -maxdepth 1 -type f -printf "%f\n" Merge LoRA cd /root/ai-tuning /opt/python310/bin/python3 merge_lora.py Check Merged Model du -sh /root/ai-tuning/merged-qwen-devops ls -lh /root/ai-tuning/merged-qwen-devops ls -lh /root/ai-tuning/merged-qwen-devops/*.safetensors Clone llama.cpp cd /root/ai-tuning git clone https://github.com/ggml-org/llama.cpp.git cd /root/ai-tuning/llama.cpp git status Build llama.cpp docker run --rm -it \ -v /root/ai-tuning:/workspace \ ubuntu:22.04 apt-get update apt-get install -y \ build-essential \ cmake \ git \ python3 \ python3-pip \ python3-dev cd /workspace/llama.cpp cmake -B build -DCMAKE_BUILD_TYPE=Release ls -lh build/bin/llama-cli Convert Merged Model to GGUF python3 /workspace/llama.cpp/convert_hf_to_gguf.py \ /workspace/merged-qwen-devops \ --outfile /workspace/qwen-devops-f16.gguf \ --outtype f16 ls -lh /root/ai-tuning/qwen-devops-f16.gguf Quantize F16 → Q4_K_M /workspace/llama.cpp/build/bin/llama-quantize \ /workspace/qwen-devops-f16.gguf \ /workspace/qwen-devops-q4_k_m.gguf \ Q4_K_M ls -lh /workspace/qwen-devops-*.gguf Test GGUF Directly /workspace/llama.cpp/build/bin/llama-cli \ -m /workspace/qwen-devops-q4_k_m.gguf Check Docker docker --version docker ps | grep ollama docker exec $OLLAMA_CONTAINER ollama --version docker exec $OLLAMA_CONTAINER ollama list Create Model Directory docker exec $OLLAMA_CONTAINER mkdir -p /models docker exec $OLLAMA_CONTAINER ls -ld /models Copy GGUF into Ollama Container docker cp \ /root/ai-tuning/qwen-devops-q4_k_m.gguf \ $OLLAMA_CONTAINER:/models/qwen-devops-q4_k_m.gguf docker exec $OLLAMA_CONTAINER \ ls -lh /models/qwen-devops-q4_k_m.gguf docker cp \ /root/ai-tuning/Modelfile-qwen-devops \ $OLLAMA_CONTAINER:/tmp/Modelfile docker exec $OLLAMA_CONTAINER \ cat /tmp/Modelfile Create Ollama Fine-Tuned Model docker exec -it $OLLAMA_CONTAINER \ ollama create devops-qwen \ -f /tmp/Modelfile docker exec $OLLAMA_CONTAINER ollama list Run Fine-Tuned Model docker exec -it $OLLAMA_CONTAINER \ ollama run devops-qwen Ollama API curl http://localhost:11434/api/tags curl http://localhost:11434/api/generate \ -d '{ "model": "devops-qwen", "prompt": "How do I check disk usage in Linux?", "stream": false }' Streaming API curl http://localhost:11434/api/generate \ -d '{ "model": "devops-qwen", "prompt": "How do I troubleshoot Kubernetes CrashLoopBackOff?", "stream": true }' API Health Check curl -s http://localhost:11434/api/tags | python3 -m json.tool Complete Pipeline Commands # 1. Project cd /root/ai-tuning # 2. Activate Python source .venv310/bin/activate # 3. Validate dataset python3 -c "import json; [json.loads(x) for x in open('devops_dataset.jsonl')]; print('OK')" # 4. Train /opt/python310/bin/python3 train_lora.py # 5. Check adapter ls -lh lora-output/adapter_model.safetensors # 6. Merge /opt/python310/bin/python3 merge_lora.py # 7. Check merged model du -sh merged-qwen-devops # 8. Check GGUF ls -lh qwen-devops-*.gguf # 9. Check Ollama docker exec 4333edae24c6 ollama list # 10. Create model docker exec -it 4333edae24c6 \ ollama create devops-qwen -f /tmp/Modelfile # 11. Run docker exec -it 4333edae24c6 \ ollama run devops-qwen # 12. Benchmark /opt/python310/bin/python3 benchmark_ollama.py Final Verification Checklist echo "===== SYSTEM =====" uname -a free -h nproc echo "===== PYTHON =====" /opt/python310/bin/python3 --version echo "===== DATASET =====" wc -l /root/ai-tuning/devops_dataset.jsonl wc -l /root/ai-tuning/train.jsonl wc -l /root/ai-tuning/validation.jsonl echo "===== LORA =====" ls -lh /root/ai-tuning/lora-output/adapter_model.safetensors echo "===== MERGED =====" du -sh /root/ai-tuning/merged-qwen-devops echo "===== GGUF =====" ls -lh /root/ai-tuning/qwen-devops-*.gguf echo "===== DOCKER =====" docker ps echo "===== OLLAMA =====" docker exec 4333edae24c6 ollama list echo "===== BENCHMARK =====" ls -lh /root/ai-tuning/benchmark_results.json Final Result DevOps Dataset │ ▼ Train / Validation │ ▼ Qwen2.5-3B-Instruct │ ▼ LoRA Training │ ▼ LoRA Adapter │ ▼ Merge LoRA │ ▼ Merged Qwen Model │ ▼ GGUF │ ▼ Q4_K_M │ ▼ Ollama / Docker │ ▼ devops-qwen:latest │ ┌───────────┴───────────┐ ▼ ▼ Ollama API Benchmark │ │ ▼ ▼ n8n / RAG Base vs Fine-Tuned │ ▼ DevOps AI Assistant Production Target User │ ▼ Chat Interface │ ▼ n8n / FastAPI │ ▼ RAG Retriever │ ┌──────┴──────┐ ▼ ▼ Vector DB DevOps Docs │ ▼ Relevant Context │ ▼ devops-qwen:latest │ ▼ DevOps AI Response │ ▼ User The final architecture combines: LoRA + Qwen2.5-3B + GGUF + Q4_K_M + Ollama + Docker + RAG + n8n / FastAPI