devarshia5 commited on
Commit
83ec29a
·
verified ·
1 Parent(s): d9da938

Upload 12 files

Browse files
.gitignore ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ # OpenMemory - IDE/Assistant specific rules
2
+ .windsurf\rules\openmemory.md
Dockerfile ADDED
@@ -0,0 +1,78 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Ornith-1.0-9B on a Hugging Face Space (Docker SDK) — OpenAI-compatible API.
2
+ #
3
+ # This is the CPU build: it runs on the FREE tier (2 vCPU / 16 GB, no GPU).
4
+ # A 9B model on CPU is usable, not fast (~5-10 tok/s). For real speed, use GPU
5
+ # hardware — see the notes at the bottom of this file.
6
+ #
7
+ # Speed/robustness choices:
8
+ # * llama.cpp OpenAI-compatible server (prebuilt CPU wheel -> fast build).
9
+ # * Model GGUF baked into the image -> instant cold start (no re-download,
10
+ # which matters because free Spaces have only ephemeral disk).
11
+ # * --chat_format chatml -> applies a proper template, which
12
+ # kills the repetition loop AND enforces <|im_end|> as a stop (no more
13
+ # duplicated answers).
14
+
15
+ FROM python:3.11-slim
16
+
17
+ # curl/ca-certs only; the prebuilt wheel means we need NO compiler toolchain.
18
+ RUN apt-get update && apt-get install -y --no-install-recommends \
19
+ curl ca-certificates \
20
+ && rm -rf /var/lib/apt/lists/*
21
+
22
+ # Hugging Face Spaces run the container as a non-root user with UID 1000.
23
+ RUN useradd -m -u 1000 user
24
+ USER user
25
+ ENV HOME=/home/user \
26
+ PATH=/home/user/.local/bin:$PATH
27
+
28
+ # --- Python deps: prebuilt CPU wheel installs in seconds (no source build) ---
29
+ RUN pip install --no-cache-dir --user "llama-cpp-python[server]" huggingface_hub
30
+
31
+ # --- Model: which GGUF to serve ---------------------------------------------
32
+ # Q4_K_M = good quality/size. For a bit more CPU speed swap to ornith-1.0-9b-Q4_0.gguf
33
+ # (simpler kernels, faster on CPU) or ornith-1.0-9b-Q3_K_M.gguf (smaller, faster).
34
+ # Confirm exact filenames on the repo's "Files" tab before changing.
35
+ ENV MODEL_REPO=deepreinforce-ai/Ornith-1.0-9B-GGUF \
36
+ MODEL_FILE=ornith-1.0-9b-Q4_K_M.gguf \
37
+ MODEL_DIR=/home/user/models
38
+
39
+ # Bake the weights into the image so the Space starts instantly.
40
+ RUN python -c "from huggingface_hub import hf_hub_download; \
41
+ hf_hub_download(repo_id='${MODEL_REPO}', filename='${MODEL_FILE}', local_dir='${MODEL_DIR}')"
42
+
43
+ # --- Serving params ----------------------------------------------------------
44
+ # Free tier = 2 vCPU, so 2 threads. n_ctx kept modest to save CPU RAM.
45
+ ENV N_CTX=8192 \
46
+ N_THREADS=2 \
47
+ OMP_NUM_THREADS=2
48
+
49
+ # HF routes external traffic to this single port.
50
+ EXPOSE 7860
51
+
52
+ # OpenAI-compatible API at http://<space>.hf.space/v1 (chat/completions, models).
53
+ CMD python -m llama_cpp.server \
54
+ --model ${MODEL_DIR}/${MODEL_FILE} \
55
+ --host 0.0.0.0 --port 7860 \
56
+ --n_ctx ${N_CTX} \
57
+ --n_threads ${N_THREADS} \
58
+ --n_batch 512 \
59
+ --chat_format chatml
60
+
61
+ # -----------------------------------------------------------------------------
62
+ # WANT IT ACTUALLY FAST? Free HF has no GPU. Two options:
63
+ #
64
+ # 1) HF GPU hardware (paid, e.g. "T4 small"): change the base image to a CUDA
65
+ # one and install the CUDA wheel, then set --n_gpu_layers -1:
66
+ #
67
+ # FROM nvidia/cuda:12.4.1-runtime-ubuntu22.04
68
+ # ... install python3 ...
69
+ # RUN pip install "llama-cpp-python[server]" \
70
+ # --extra-index-url https://abetlen.github.io/llama-cpp-python/whl/cu124
71
+ # CMD python3 -m llama_cpp.server --model ... --n_gpu_layers -1 \
72
+ # --host 0.0.0.0 --port 7860 --n_ctx 16384 --chat_format chatml
73
+ #
74
+ # On a T4 this jumps to ~40-70 tok/s.
75
+ #
76
+ # 2) ZeroGPU Space (free-ish, needs PRO): that path uses the Gradio SDK +
77
+ # @spaces.GPU with transformers/vLLM — NOT this llama.cpp Dockerfile.
78
+ # -----------------------------------------------------------------------------
README_SPACE.md ADDED
@@ -0,0 +1,61 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ title: Ornith 1.0 9B API
3
+ emoji: 🦅
4
+ colorFrom: indigo
5
+ colorTo: purple
6
+ sdk: docker
7
+ app_port: 7860
8
+ pinned: false
9
+ license: mit
10
+ ---
11
+
12
+ # Ornith-1.0-9B — OpenAI-compatible API (llama.cpp)
13
+
14
+ Serves `deepreinforce-ai/Ornith-1.0-9B` (Q4_K_M GGUF) as an **OpenAI-compatible
15
+ REST API** via llama.cpp's built-in server.
16
+
17
+ > **Note:** This is the CPU build for the free tier (2 vCPU, no GPU). A 9B model
18
+ > on CPU is usable but slow (~5-10 tok/s). See the Dockerfile footer for the GPU
19
+ > variant if you need real speed.
20
+
21
+ ## Endpoints
22
+
23
+ - `GET /v1/models`
24
+ - `POST /v1/chat/completions` (streaming supported)
25
+ - `POST /v1/completions`
26
+
27
+ Base URL: `https://<your-username>-<space-name>.hf.space/v1`
28
+
29
+ ## Use it
30
+
31
+ ```python
32
+ from openai import OpenAI
33
+
34
+ client = OpenAI(
35
+ base_url="https://<your-username>-<space-name>.hf.space/v1",
36
+ api_key="not-needed", # llama.cpp server ignores it by default
37
+ )
38
+
39
+ resp = client.chat.completions.create(
40
+ model="ornith",
41
+ messages=[{"role": "user", "content": "Write a Python LRU cache with a docstring."}],
42
+ temperature=0.6, top_p=0.95,
43
+ stream=True,
44
+ )
45
+ for chunk in resp:
46
+ print(chunk.choices[0].delta.content or "", end="")
47
+ ```
48
+
49
+ Or with curl:
50
+
51
+ ```bash
52
+ curl https://<your-username>-<space-name>.hf.space/v1/chat/completions \
53
+ -H "Content-Type: application/json" \
54
+ -d '{"model":"ornith","messages":[{"role":"user","content":"hello"}],"temperature":0.6}'
55
+ ```
56
+
57
+ ## Deploy
58
+
59
+ 1. Create a new Space → **Docker** (blank template).
60
+ 2. Add this `Dockerfile` and rename this file to `README.md` in the Space repo.
61
+ 3. Push. First build takes a few minutes (it bakes the ~5.5 GB model into the image).
hf-space-rag/.gitignore ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ # Baked into the Docker image at build time — do not commit.
2
+ models/
3
+ .cache/
4
+ __pycache__/
5
+ *.pyc
hf-space-rag/Dockerfile ADDED
@@ -0,0 +1,52 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # CPU + FREE-tier RAG Space (Docker SDK).
2
+ # Embeddings : BAAI/bge-small-en-v1.5 (fastembed / ONNX, no torch)
3
+ # Vector DB : FAISS (in-memory)
4
+ # LLM : Qwen2.5-1.5B-Instruct (llama.cpp GGUF, fast on CPU)
5
+ # API : OpenAI-compatible -> /v1/chat/completions (+ web UI at /)
6
+ #
7
+ # Everything is CPU-only. No GPU, no paid hardware required.
8
+ # Models are baked into the image so the Space cold-starts instantly.
9
+
10
+ FROM python:3.11-slim
11
+
12
+ RUN apt-get update && apt-get install -y --no-install-recommends \
13
+ curl ca-certificates && rm -rf /var/lib/apt/lists/*
14
+
15
+ # Hugging Face Spaces run the container as non-root UID 1000.
16
+ RUN useradd -m -u 1000 user
17
+ USER user
18
+ ENV HOME=/home/user \
19
+ PATH=/home/user/.local/bin:$PATH
20
+ WORKDIR /home/user/app
21
+
22
+ # --- Python deps (prebuilt wheels -> fast, no build toolchain) ---
23
+ COPY --chown=user requirements.txt .
24
+ RUN pip install --no-cache-dir --user -r requirements.txt
25
+
26
+ # --- Model config ------------------------------------------------------------
27
+ ENV LLM_REPO=Qwen/Qwen2.5-1.5B-Instruct-GGUF \
28
+ LLM_FILE=qwen2.5-1.5b-instruct-q4_k_m.gguf \
29
+ MODEL_DIR=/home/user/models \
30
+ EMBED_MODEL=BAAI/bge-small-en-v1.5 \
31
+ FASTEMBED_CACHE=/home/user/.cache/fastembed \
32
+ HF_HOME=/home/user/.cache/huggingface
33
+
34
+ # Bake the LLM (~1 GB) into the image -> instant cold start.
35
+ RUN python -c "from huggingface_hub import hf_hub_download; \
36
+ hf_hub_download(repo_id='${LLM_REPO}', filename='${LLM_FILE}', local_dir='${MODEL_DIR}')"
37
+
38
+ # Bake the embedding model (ONNX ~130 MB) into the image too.
39
+ RUN python -c "import os; from fastembed import TextEmbedding; \
40
+ TextEmbedding(os.environ['EMBED_MODEL'], cache_dir=os.environ['FASTEMBED_CACHE'])"
41
+
42
+ # --- App ---------------------------------------------------------------------
43
+ COPY --chown=user . .
44
+
45
+ # Runtime knobs (free tier = 2 vCPU).
46
+ ENV N_CTX=8192 \
47
+ N_THREADS=2 \
48
+ TOP_K=4 \
49
+ DOCS_DIR=documents
50
+
51
+ EXPOSE 7860
52
+ CMD ["python", "-m", "uvicorn", "app:app", "--host", "0.0.0.0", "--port", "7860"]
hf-space-rag/README.md ADDED
@@ -0,0 +1,67 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ title: CPU RAG Space
3
+ emoji: 🦅
4
+ colorFrom: indigo
5
+ colorTo: purple
6
+ sdk: docker
7
+ app_port: 7860
8
+ pinned: false
9
+ license: mit
10
+ ---
11
+
12
+ # CPU RAG Space — Qwen2.5-1.5B + FAISS (free tier, no GPU)
13
+
14
+ A self-contained Retrieval-Augmented Generation service that runs **entirely on
15
+ CPU** and fits the Hugging Face **free tier** (2 vCPU / 16 GB).
16
+
17
+ | Component | What | Why |
18
+ |-----------|------|-----|
19
+ | Embeddings | `BAAI/bge-small-en-v1.5` via `fastembed` (ONNX) | fast on CPU, no PyTorch, ~130 MB |
20
+ | Vector DB | **FAISS** (in-memory) | tiny, instant search |
21
+ | LLM | `Qwen2.5-1.5B-Instruct` GGUF Q4_K_M via llama.cpp | fast on CPU (~12–20 tok/s), ~1 GB |
22
+ | API | OpenAI-compatible `/v1/chat/completions` + web UI | drop-in for any client |
23
+
24
+ Total footprint ≈ 1.5–2 GB RAM. Retrieval adds ~20 ms; the LLM is the only
25
+ real latency.
26
+
27
+ ## Deploy (drag & drop)
28
+
29
+ 1. Create a new Space → **Docker** (blank template).
30
+ 2. Drag **all files in this folder** into the Space repo (keep the structure —
31
+ `documents/` included).
32
+ 3. Push. First build takes a few minutes (it bakes the ~1 GB LLM and the
33
+ embedder into the image so cold starts are instant).
34
+
35
+ ## Use it
36
+
37
+ Web UI: open the Space URL. Upload `.txt`/`.md` files and ask questions.
38
+
39
+ API (OpenAI-compatible):
40
+
41
+ ```python
42
+ from openai import OpenAI
43
+ client = OpenAI(base_url="https://<user>-<space>.hf.space/v1", api_key="x")
44
+ r = client.chat.completions.create(
45
+ model="cpu-rag",
46
+ messages=[{"role": "user", "content": "How does retrieval work here?"}],
47
+ )
48
+ print(r.choices[0].message.content)
49
+ ```
50
+
51
+ Extra endpoints: `POST /ingest` (upload a doc), `GET /stats`.
52
+
53
+ ## Add your own knowledge
54
+
55
+ - Put `.txt`/`.md` files in `documents/` before pushing (indexed at startup), or
56
+ - Upload them at runtime via the UI / `POST /ingest`.
57
+
58
+ > Note: the free tier has ephemeral storage, so runtime-uploaded docs are lost
59
+ > on restart. For a permanent corpus, commit files into `documents/`.
60
+
61
+ ## Swap the model
62
+
63
+ Change these in the `Dockerfile` (confirm exact filenames on the repo's *Files*
64
+ tab):
65
+
66
+ - Faster / smaller: `Qwen/Qwen2.5-0.5B-Instruct-GGUF`
67
+ - Coding-focused: `Qwen/Qwen2.5-Coder-1.5B-Instruct-GGUF`
hf-space-rag/app.py ADDED
@@ -0,0 +1,208 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ CPU RAG Space — bge-small (fastembed) + FAISS + Qwen2.5-1.5B (llama.cpp),
3
+ served as an OpenAI-compatible API with a small web UI.
4
+
5
+ Everything runs on CPU and fits the Hugging Face free tier (2 vCPU / 16 GB).
6
+ """
7
+
8
+ import glob
9
+ import json
10
+ import os
11
+
12
+ import faiss
13
+ import numpy as np
14
+ from fastapi import FastAPI, File, UploadFile
15
+ from fastapi.responses import HTMLResponse, JSONResponse, StreamingResponse
16
+ from fastembed import TextEmbedding
17
+ from llama_cpp import Llama
18
+ from pydantic import BaseModel
19
+
20
+ # --------------------------------------------------------------------------- #
21
+ # Config (all overridable via Space "Variables")
22
+ # --------------------------------------------------------------------------- #
23
+ MODEL_DIR = os.environ.get("MODEL_DIR", "models")
24
+ LLM_FILE = os.environ.get("LLM_FILE", "qwen2.5-1.5b-instruct-q4_k_m.gguf")
25
+ LLM_PATH = os.path.join(MODEL_DIR, LLM_FILE)
26
+ EMBED_MODEL = os.environ.get("EMBED_MODEL", "BAAI/bge-small-en-v1.5")
27
+ FASTEMBED_CACHE = os.environ.get("FASTEMBED_CACHE")
28
+ N_CTX = int(os.environ.get("N_CTX", "8192"))
29
+ N_THREADS = int(os.environ.get("N_THREADS", str(os.cpu_count() or 2)))
30
+ TOP_K = int(os.environ.get("TOP_K", "4"))
31
+ DOCS_DIR = os.environ.get("DOCS_DIR", "documents")
32
+
33
+ CHUNK_SIZE = 800 # characters per chunk
34
+ CHUNK_OVERLAP = 120
35
+
36
+ RAG_SYSTEM = (
37
+ "You are a helpful assistant. Answer the user's question using ONLY the "
38
+ "context below. If the answer is not in the context, say you don't know. "
39
+ "Cite the source of each fact in square brackets like [filename].\n\n"
40
+ "Context:\n{context}"
41
+ )
42
+
43
+ # --------------------------------------------------------------------------- #
44
+ # Lazily-initialised singletons
45
+ # --------------------------------------------------------------------------- #
46
+ app = FastAPI(title="CPU RAG Space")
47
+
48
+ _embedder = None
49
+ _llm = None
50
+ _index = None # faiss.IndexFlatIP
51
+ _chunks = [] # list[{"text": str, "source": str}]
52
+
53
+
54
+ def embedder():
55
+ global _embedder
56
+ if _embedder is None:
57
+ _embedder = TextEmbedding(EMBED_MODEL, cache_dir=FASTEMBED_CACHE)
58
+ return _embedder
59
+
60
+
61
+ def embed(texts):
62
+ vecs = np.array(list(embedder().embed(list(texts))), dtype="float32")
63
+ faiss.normalize_L2(vecs) # cosine similarity via inner product
64
+ return vecs
65
+
66
+
67
+ def llm():
68
+ global _llm
69
+ if _llm is None:
70
+ _llm = Llama(model_path=LLM_PATH, n_ctx=N_CTX, n_threads=N_THREADS,
71
+ n_batch=512, verbose=False)
72
+ return _llm
73
+
74
+
75
+ # --------------------------------------------------------------------------- #
76
+ # Indexing / retrieval
77
+ # --------------------------------------------------------------------------- #
78
+ def chunk_text(text, source):
79
+ out, i, n = [], 0, len(text)
80
+ step = max(CHUNK_SIZE - CHUNK_OVERLAP, 1)
81
+ while i < n:
82
+ piece = text[i:i + CHUNK_SIZE].strip()
83
+ if piece:
84
+ out.append({"text": piece, "source": source})
85
+ i += step
86
+ return out
87
+
88
+
89
+ def add_chunks(new_chunks):
90
+ global _index, _chunks
91
+ if not new_chunks:
92
+ return 0
93
+ vecs = embed([c["text"] for c in new_chunks])
94
+ if _index is None:
95
+ _index = faiss.IndexFlatIP(vecs.shape[1])
96
+ _index.add(vecs)
97
+ _chunks.extend(new_chunks)
98
+ return len(new_chunks)
99
+
100
+
101
+ def build_index():
102
+ patterns = ("*.txt", "*.md")
103
+ files = []
104
+ for p in patterns:
105
+ files += glob.glob(os.path.join(DOCS_DIR, "**", p), recursive=True)
106
+ all_chunks = []
107
+ for f in files:
108
+ try:
109
+ with open(f, encoding="utf-8") as fh:
110
+ all_chunks += chunk_text(fh.read(), os.path.basename(f))
111
+ except Exception as exc:
112
+ print(f"[rag] skip {f}: {exc}")
113
+ add_chunks(all_chunks)
114
+
115
+
116
+ def retrieve(query, k=TOP_K):
117
+ if _index is None or _index.ntotal == 0:
118
+ return []
119
+ scores, ids = _index.search(embed([query]), min(k, _index.ntotal))
120
+ hits = []
121
+ for score, idx in zip(scores[0], ids[0]):
122
+ if idx < 0:
123
+ continue
124
+ c = _chunks[idx]
125
+ hits.append({"text": c["text"], "source": c["source"], "score": float(score)})
126
+ return hits
127
+
128
+
129
+ # --------------------------------------------------------------------------- #
130
+ # Startup
131
+ # --------------------------------------------------------------------------- #
132
+ @app.on_event("startup")
133
+ def _startup():
134
+ print("[rag] loading embedder + llm ...")
135
+ embedder()
136
+ llm()
137
+ build_index()
138
+ print(f"[rag] ready. indexed_chunks={len(_chunks)}")
139
+
140
+
141
+ # --------------------------------------------------------------------------- #
142
+ # OpenAI-compatible chat endpoint (with RAG)
143
+ # --------------------------------------------------------------------------- #
144
+ class ChatMessage(BaseModel):
145
+ role: str
146
+ content: str
147
+
148
+
149
+ class ChatRequest(BaseModel):
150
+ model: str = "cpu-rag"
151
+ messages: list[ChatMessage]
152
+ temperature: float = 0.3
153
+ top_p: float = 0.9
154
+ max_tokens: int = 512
155
+ stream: bool = False
156
+ use_rag: bool = True
157
+
158
+
159
+ def _augment(req: ChatRequest):
160
+ msgs = [m.model_dump() for m in req.messages]
161
+ users = [m for m in msgs if m["role"] == "user"]
162
+ query = users[-1]["content"] if users else ""
163
+ ctxs = retrieve(query) if req.use_rag else []
164
+ if ctxs:
165
+ context = "\n\n".join(f"[{c['source']}] {c['text']}" for c in ctxs)
166
+ system = {"role": "system", "content": RAG_SYSTEM.format(context=context)}
167
+ msgs = [system] + [m for m in msgs if m["role"] != "system"]
168
+ return msgs, ctxs
169
+
170
+
171
+ @app.post("/v1/chat/completions")
172
+ def chat_completions(req: ChatRequest):
173
+ messages, ctxs = _augment(req)
174
+ params = dict(messages=messages, temperature=req.temperature, top_p=req.top_p,
175
+ max_tokens=req.max_tokens, stop=["<|im_end|>", "<|endoftext|>"])
176
+
177
+ if req.stream:
178
+ def gen():
179
+ for chunk in llm().create_chat_completion(**params, stream=True):
180
+ yield f"data: {json.dumps(chunk)}\n\n"
181
+ yield "data: [DONE]\n\n"
182
+ return StreamingResponse(gen(), media_type="text/event-stream")
183
+
184
+ resp = llm().create_chat_completion(**params)
185
+ resp["sources"] = [{"source": c["source"], "score": round(c["score"], 3)} for c in ctxs]
186
+ return JSONResponse(resp)
187
+
188
+
189
+ # --------------------------------------------------------------------------- #
190
+ # Ingest / stats / UI
191
+ # --------------------------------------------------------------------------- #
192
+ @app.post("/ingest")
193
+ async def ingest(file: UploadFile = File(...)):
194
+ text = (await file.read()).decode("utf-8", "ignore")
195
+ added = add_chunks(chunk_text(text, file.filename))
196
+ return {"file": file.filename, "added_chunks": added, "total_chunks": len(_chunks)}
197
+
198
+
199
+ @app.get("/stats")
200
+ def stats():
201
+ return {"indexed_chunks": len(_chunks), "embed_model": EMBED_MODEL,
202
+ "llm": LLM_FILE, "n_ctx": N_CTX, "threads": N_THREADS, "top_k": TOP_K}
203
+
204
+
205
+ @app.get("/", response_class=HTMLResponse)
206
+ def home():
207
+ with open(os.path.join(os.path.dirname(__file__), "index.html"), encoding="utf-8") as f:
208
+ return f.read()
hf-space-rag/documents/sample.txt ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ CPU RAG Space — sample knowledge document.
2
+
3
+ This Space is a fully CPU, free-tier Retrieval-Augmented Generation service.
4
+
5
+ Architecture:
6
+ - Embedding model: BAAI/bge-small-en-v1.5, run via fastembed (ONNX). It turns
7
+ text into 384-dimensional vectors and needs no GPU or PyTorch.
8
+ - Vector store: FAISS (IndexFlatIP) holds the document vectors in memory and
9
+ returns the most similar chunks for a query using cosine similarity.
10
+ - Language model: Qwen2.5-1.5B-Instruct in GGUF Q4_K_M form, served by
11
+ llama.cpp. It reads the retrieved chunks and writes a grounded answer.
12
+
13
+ How retrieval works:
14
+ 1. Your question is embedded into a vector.
15
+ 2. FAISS finds the top-K most similar document chunks (default K = 4).
16
+ 3. Those chunks are inserted into the model's system prompt as context.
17
+ 4. The model answers using only that context and cites the source file.
18
+
19
+ Why a small model is fine here:
20
+ RAG moves knowledge out of the model's weights and into the retriever, so the
21
+ model only needs to read and summarise the provided context rather than
22
+ memorise facts. That makes a fast 1.5B model a good fit for CPU serving.
23
+
24
+ Replace this file with your own .txt or .md documents, or upload files at
25
+ runtime through the web UI, and the Space will answer questions about them.
hf-space-rag/index.html ADDED
@@ -0,0 +1,97 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ <!doctype html>
2
+ <html lang="en">
3
+ <head>
4
+ <meta charset="utf-8" />
5
+ <meta name="viewport" content="width=device-width, initial-scale=1" />
6
+ <title>CPU RAG · Qwen2.5-1.5B + FAISS</title>
7
+ <style>
8
+ :root { color-scheme: light dark; --bg:#0f1220; --panel:#1a1e33; --acc:#7c8cff; --mut:#8b90a8; }
9
+ * { box-sizing: border-box; }
10
+ body { margin:0; font-family: system-ui, sans-serif; background:var(--bg); color:#e7e9f3; }
11
+ header { padding:14px 18px; border-bottom:1px solid #2a2f4a; display:flex; gap:12px; align-items:center; }
12
+ header h1 { font-size:15px; margin:0; font-weight:600; }
13
+ header .tag { font-size:11px; color:var(--mut); background:var(--panel); padding:3px 8px; border-radius:20px; }
14
+ main { max-width:820px; margin:0 auto; padding:18px; }
15
+ #chat { display:flex; flex-direction:column; gap:12px; min-height:50vh; }
16
+ .msg { padding:11px 14px; border-radius:12px; max-width:88%; white-space:pre-wrap; line-height:1.45; font-size:14px; }
17
+ .user { align-self:flex-end; background:var(--acc); color:#fff; }
18
+ .bot { align-self:flex-start; background:var(--panel); }
19
+ .src { align-self:flex-start; font-size:11px; color:var(--mut); margin-top:-6px; }
20
+ form { display:flex; gap:8px; margin-top:16px; }
21
+ textarea { flex:1; resize:none; padding:11px; border-radius:10px; border:1px solid #2a2f4a; background:#12152a; color:#e7e9f3; font-size:14px; }
22
+ button { padding:0 16px; border:0; border-radius:10px; background:var(--acc); color:#fff; font-weight:600; cursor:pointer; }
23
+ button:disabled { opacity:.5; cursor:default; }
24
+ .bar { display:flex; gap:10px; align-items:center; margin-bottom:14px; font-size:12px; color:var(--mut); }
25
+ .bar input[type=file] { font-size:12px; color:var(--mut); }
26
+ </style>
27
+ </head>
28
+ <body>
29
+ <header>
30
+ <h1>🦅 CPU RAG</h1>
31
+ <span class="tag">Qwen2.5-1.5B · bge-small · FAISS</span>
32
+ <span class="tag" id="stat">…</span>
33
+ </header>
34
+ <main>
35
+ <div class="bar">
36
+ <label>Add a document (.txt/.md):</label>
37
+ <input type="file" id="file" accept=".txt,.md" />
38
+ <button id="up" type="button">Upload</button>
39
+ <span id="upmsg"></span>
40
+ </div>
41
+ <div id="chat"></div>
42
+ <form id="form">
43
+ <textarea id="q" rows="2" placeholder="Ask about your documents…"></textarea>
44
+ <button id="send" type="submit">Send</button>
45
+ </form>
46
+ </main>
47
+ <script>
48
+ const chat = document.getElementById('chat');
49
+ const stat = document.getElementById('stat');
50
+
51
+ function add(cls, text) {
52
+ const d = document.createElement('div');
53
+ d.className = 'msg ' + cls; d.textContent = text;
54
+ chat.appendChild(d); chat.scrollIntoView({block:'end'}); return d;
55
+ }
56
+ async function refreshStats(){
57
+ try { const s = await (await fetch('/stats')).json();
58
+ stat.textContent = s.indexed_chunks + ' chunks indexed'; } catch(e){}
59
+ }
60
+ refreshStats();
61
+
62
+ document.getElementById('up').onclick = async () => {
63
+ const f = document.getElementById('file').files[0];
64
+ if (!f) return;
65
+ document.getElementById('upmsg').textContent = 'uploading…';
66
+ const fd = new FormData(); fd.append('file', f);
67
+ const r = await (await fetch('/ingest', {method:'POST', body:fd})).json();
68
+ document.getElementById('upmsg').textContent = '+' + r.added_chunks + ' chunks';
69
+ refreshStats();
70
+ };
71
+
72
+ document.getElementById('form').onsubmit = async (e) => {
73
+ e.preventDefault();
74
+ const q = document.getElementById('q').value.trim();
75
+ if (!q) return;
76
+ document.getElementById('q').value = '';
77
+ document.getElementById('send').disabled = true;
78
+ add('user', q);
79
+ const thinking = add('bot', '…');
80
+ try {
81
+ const r = await fetch('/v1/chat/completions', {
82
+ method:'POST', headers:{'Content-Type':'application/json'},
83
+ body: JSON.stringify({ messages:[{role:'user', content:q}], temperature:0.3 })
84
+ });
85
+ const data = await r.json();
86
+ thinking.textContent = data.choices?.[0]?.message?.content ?? '(no answer)';
87
+ if (data.sources && data.sources.length) {
88
+ add('src', 'sources: ' + data.sources.map(s => s.source + ' (' + s.score + ')').join(', '));
89
+ }
90
+ } catch (err) {
91
+ thinking.textContent = 'Error: ' + err;
92
+ }
93
+ document.getElementById('send').disabled = false;
94
+ };
95
+ </script>
96
+ </body>
97
+ </html>
hf-space-rag/requirements.txt ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Prebuilt CPU wheel for llama.cpp -> fast build, no compiler needed.
2
+ --extra-index-url https://abetlen.github.io/llama-cpp-python/whl/cpu
3
+ llama-cpp-python
4
+
5
+ # Lightweight CPU embeddings via ONNX (no PyTorch -> small image, fast).
6
+ fastembed
7
+ faiss-cpu
8
+
9
+ # API + serving
10
+ fastapi
11
+ uvicorn[standard]
12
+ python-multipart
13
+ huggingface_hub
14
+ numpy
openmemory.md ADDED
File without changes
ornith_colab.py ADDED
@@ -0,0 +1,508 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ ornith_colab.py — run deepreinforce-ai/Ornith-1.0-9B in Google Colab as an AGENT.
3
+
4
+ Goal: CPU-first, auto-use a T4 GPU if present. Engine: llama.cpp (via
5
+ llama-cpp-python) on a GGUF quant — NOT vLLM (vLLM is slow to spin up and
6
+ wants a big GPU; llama.cpp installs from a prebuilt wheel and runs the SAME
7
+ GGUF on CPU or a 16 GB T4).
8
+
9
+ Ornith-1.0-9B (model card + release, June 2026): ~9B dense reasoning model
10
+ post-trained on Qwen3.5-9B, MIT license, emits <think>...</think> then the
11
+ answer, and is built for *agentic coding / tool use*. Sampling: temp 0.6,
12
+ top_p 0.95, top_k 20.
13
+
14
+ What this file gives you
15
+ ------------------------
16
+ chat(msg) -> str (just the answer)
17
+ generate(msg, history) -> dict ({reasoning, answer, raw})
18
+ stream_chat(msg, history) -> yields chunks (live tokens)
19
+ run_agent(task, tools) -> dict (tool-calling agent loop)
20
+
21
+ Why the agent loop is hand-rolled (researched):
22
+ The Ornith GGUFs have inconsistent tool-aware chat templates (same root
23
+ cause as the known repetition-loop bug when the chat template is missing).
24
+ So instead of relying on llama.cpp's function-calling handler, we inject
25
+ tool schemas Qwen/Hermes-style into the system prompt and parse
26
+ <tool_call>{...}</tool_call> blocks ourselves. This is template-independent
27
+ and works on any Ornith GGUF mirror.
28
+
29
+ Colab usage:
30
+ !python ornith_colab.py # runs chat + agent demos
31
+ or from a cell:
32
+ from ornith_colab import chat, generate, run_agent
33
+ """
34
+
35
+ import json
36
+ import os
37
+ import re
38
+ import shutil
39
+ import subprocess
40
+ import sys
41
+
42
+ # --------------------------------------------------------------------------- #
43
+ # Config
44
+ # --------------------------------------------------------------------------- #
45
+ GGUF_REPO = "AtomicChat/ornith-9b-GGUF" # template-embedded mirror (avoids loop bug)
46
+ GGUF_FILE = "*Q4_K_M*.gguf" # ~5.5 GB; good for T4 or CPU RAM
47
+
48
+ # Device: "cpu" (default — this build showcases CPU capability), "gpu", or
49
+ # "auto". Override from a Colab cell: os.environ["ORNITH_DEVICE"] = "gpu".
50
+ DEVICE = os.environ.get("ORNITH_DEVICE", "cpu").lower()
51
+
52
+ N_CTX = 16384 # agents burn context on tool results — give them room
53
+ TEMPERATURE = 0.6
54
+ TOP_P = 0.95
55
+ TOP_K = 20
56
+ REPEAT_PENALTY = 1.05 # insurance against loops
57
+ MAX_TOKENS = 2048
58
+
59
+ # Char-code tags so raw special-token bytes never sit in source.
60
+ _TAG = lambda *cs: "".join(chr(c) for c in cs)
61
+ _IM_START = _TAG(60, 124, 105, 109, 95, 115, 116, 97, 114, 116, 124, 62) # <|im_start|>
62
+ _IM_END = _TAG(60, 124, 105, 109, 95, 101, 110, 100, 124, 62) # <|im_end|>
63
+ _THINK_CLOSE = _TAG(60, 47, 116, 104, 105, 110, 107, 62) # </think>
64
+
65
+ FALLBACK_CHAT_TEMPLATE = (
66
+ "{%- for m in messages %}"
67
+ f"{_IM_START}" "{{ m['role'] }}\n{{ m['content'] }}" f"{_IM_END}" "\n"
68
+ "{%- endfor %}"
69
+ "{%- if add_generation_prompt %}"
70
+ f"{_IM_START}" "assistant\n"
71
+ "{%- endif %}"
72
+ )
73
+
74
+ _TOOL_CALL_RE = re.compile(r"<tool_call>\s*(\{.*?\})\s*</tool_call>", re.DOTALL)
75
+
76
+ # End-of-turn stops (built from char codes). "<|im_end|>" is the ChatML turn
77
+ # marker; "<|endoftext|>" is the tokenizer EOS.
78
+ DEFAULT_STOP = [_IM_END, _TAG(60, 124, 101, 110, 100, 111, 102, 116, 101, 120, 116, 124, 62)]
79
+
80
+
81
+ # --------------------------------------------------------------------------- #
82
+ # Environment detection + install
83
+ # --------------------------------------------------------------------------- #
84
+ def _has_nvidia_gpu() -> bool:
85
+ if not shutil.which("nvidia-smi"):
86
+ return False
87
+ try:
88
+ subprocess.run(["nvidia-smi"], check=True,
89
+ stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
90
+ return True
91
+ except Exception:
92
+ return False
93
+
94
+
95
+ def _pip(*args: str) -> None:
96
+ subprocess.run([sys.executable, "-m", "pip", "install", "-q", *args], check=True)
97
+
98
+
99
+ def _resolve_use_gpu() -> bool:
100
+ if DEVICE == "cpu":
101
+ return False
102
+ if DEVICE == "gpu":
103
+ return True
104
+ return _has_nvidia_gpu() # "auto"
105
+
106
+
107
+ def _ensure_deps(use_gpu: bool) -> None:
108
+ for pkg in ("huggingface_hub", "psutil"):
109
+ try:
110
+ __import__(pkg)
111
+ except ImportError:
112
+ _pip(pkg)
113
+
114
+ try:
115
+ import llama_cpp # noqa: F401
116
+ return
117
+ except ImportError:
118
+ pass
119
+
120
+ if use_gpu:
121
+ for cu in ("cu124", "cu122", "cu121"):
122
+ try:
123
+ _pip("llama-cpp-python", "--extra-index-url",
124
+ f"https://abetlen.github.io/llama-cpp-python/whl/{cu}")
125
+ import llama_cpp # noqa: F401
126
+ print(f"[ornith] installed llama-cpp-python (CUDA {cu} wheel)")
127
+ return
128
+ except Exception:
129
+ continue
130
+ print("[ornith] no prebuilt CUDA wheel matched; building from source...")
131
+ env = dict(os.environ, CMAKE_ARGS="-DGGML_CUDA=on")
132
+ subprocess.run([sys.executable, "-m", "pip", "install", "-q",
133
+ "--no-cache-dir", "llama-cpp-python"], check=True, env=env)
134
+ else:
135
+ _pip("llama-cpp-python")
136
+ print("[ornith] installed llama-cpp-python (CPU wheel)")
137
+
138
+
139
+ # --------------------------------------------------------------------------- #
140
+ # Model loading
141
+ # --------------------------------------------------------------------------- #
142
+ _LLM = None
143
+
144
+
145
+ def load_model():
146
+ global _LLM
147
+ if _LLM is not None:
148
+ return _LLM
149
+
150
+ use_gpu = _resolve_use_gpu()
151
+ print(f"[ornith] device={DEVICE} use_gpu={use_gpu} -> "
152
+ f"{'offloading all layers to GPU' if use_gpu else f'running on CPU ({os.cpu_count()} threads)'}")
153
+ _ensure_deps(use_gpu)
154
+
155
+ from llama_cpp import Llama
156
+
157
+ kwargs = dict(
158
+ repo_id=GGUF_REPO,
159
+ filename=GGUF_FILE,
160
+ n_ctx=N_CTX,
161
+ n_gpu_layers=(-1 if use_gpu else 0),
162
+ n_threads=(None if use_gpu else (os.cpu_count() or 2)),
163
+ n_batch=512,
164
+ flash_attn=use_gpu, # faster KV attention on the T4
165
+ verbose=False,
166
+ )
167
+ print(f"[ornith] loading {GGUF_REPO} ({GGUF_FILE}) ... first run downloads ~5.5 GB")
168
+ try:
169
+ llm = Llama.from_pretrained(**kwargs)
170
+ except TypeError:
171
+ # older llama-cpp-python without flash_attn kwarg
172
+ kwargs.pop("flash_attn", None)
173
+ llm = Llama.from_pretrained(**kwargs)
174
+ except Exception as exc:
175
+ raise RuntimeError(
176
+ f"Failed to load GGUF from {GGUF_REPO}. Try another mirror "
177
+ f"(e.g. deepreinforce-ai/Ornith-1.0-9B-GGUF). Error: {exc}")
178
+
179
+ meta = getattr(llm, "metadata", {}) or {}
180
+ if not meta.get("tokenizer.chat_template"):
181
+ print("[ornith] no embedded chat template -> applying Qwen3 fallback")
182
+ from llama_cpp.llama_chat_format import Jinja2ChatFormatter
183
+ llm.chat_handler = Jinja2ChatFormatter(
184
+ template=FALLBACK_CHAT_TEMPLATE, eos_token=_IM_END, bos_token="",
185
+ ).to_chat_handler()
186
+
187
+ _LLM = llm
188
+ print("[ornith] model ready")
189
+ return _LLM
190
+
191
+
192
+ # --------------------------------------------------------------------------- #
193
+ # Core generation
194
+ # --------------------------------------------------------------------------- #
195
+ def _messages(user_msg, history):
196
+ msgs = list(history or [])
197
+ if user_msg is not None:
198
+ msgs.append({"role": "user", "content": user_msg})
199
+ return msgs
200
+
201
+
202
+ def stream_chat(user_msg, history=None, max_tokens=MAX_TOKENS, stop=None):
203
+ """Yield text chunks live (includes the <think>...</think> block)."""
204
+ llm = load_model()
205
+ # Always enforce the end-of-turn stops (dedup while preserving order).
206
+ effective_stop = list(dict.fromkeys((stop or []) + DEFAULT_STOP))
207
+ stream = llm.create_chat_completion(
208
+ messages=_messages(user_msg, history),
209
+ max_tokens=max_tokens, temperature=TEMPERATURE, top_p=TOP_P,
210
+ top_k=TOP_K, repeat_penalty=REPEAT_PENALTY, stop=effective_stop, stream=True,
211
+ )
212
+ for chunk in stream:
213
+ delta = chunk["choices"][0]["delta"].get("content")
214
+ if delta:
215
+ yield delta
216
+
217
+
218
+ def _split_think(text):
219
+ if _THINK_CLOSE in text:
220
+ reasoning, answer = text.split(_THINK_CLOSE, 1)
221
+ return reasoning.replace("<think>", "").strip(), answer.strip()
222
+ return "", text.strip()
223
+
224
+
225
+ def generate(user_msg, history=None, max_tokens=MAX_TOKENS, stop=None):
226
+ """Structured, non-streaming call. Returns {reasoning, answer, raw}."""
227
+ raw = "".join(stream_chat(user_msg, history, max_tokens, stop))
228
+ reasoning, answer = _split_think(raw)
229
+ return {"reasoning": reasoning, "answer": answer, "raw": raw}
230
+
231
+
232
+ def chat(user_msg, history=None, max_tokens=MAX_TOKENS) -> str:
233
+ """Just the final answer (reasoning stripped)."""
234
+ return generate(user_msg, history, max_tokens)["answer"]
235
+
236
+
237
+ # --------------------------------------------------------------------------- #
238
+ # Tool-calling agent loop
239
+ # --------------------------------------------------------------------------- #
240
+ def _tools_system_prompt(tools):
241
+ schemas = "\n".join(json.dumps(t["schema"]) for t in tools)
242
+ return (
243
+ "You are Ornith, an agentic assistant that can call tools to act.\n"
244
+ "You have access to these tools (JSON schemas):\n"
245
+ f"{schemas}\n\n"
246
+ "When you need a tool, emit EXACTLY one block per call:\n"
247
+ '<tool_call>{"name": "<tool_name>", "arguments": {<args>}}</tool_call>\n'
248
+ "You may emit multiple tool_call blocks in one turn. After you receive "
249
+ "the tool results, continue reasoning. When the task is fully done and "
250
+ "you need no more tools, reply with the final answer and NO tool_call block."
251
+ )
252
+
253
+
254
+ def _parse_tool_calls(text):
255
+ calls = []
256
+ for m in _TOOL_CALL_RE.finditer(text):
257
+ try:
258
+ obj = json.loads(m.group(1))
259
+ calls.append({"name": obj.get("name"), "arguments": obj.get("arguments", {})})
260
+ except json.JSONDecodeError:
261
+ continue
262
+ return calls
263
+
264
+
265
+ def run_agent(task, tools, max_steps=6, verbose=True):
266
+ """
267
+ Minimal tool-calling agent loop.
268
+
269
+ tools: list of {
270
+ "name": str,
271
+ "schema": {json schema shown to the model},
272
+ "fn": callable(**arguments) -> anything json-serializable,
273
+ }
274
+ Returns {answer, steps, transcript}.
275
+ """
276
+ registry = {t["name"]: t["fn"] for t in tools}
277
+ history = [{"role": "system", "content": _tools_system_prompt(tools)},
278
+ {"role": "user", "content": task}]
279
+ transcript = []
280
+
281
+ for step in range(1, max_steps + 1):
282
+ # Stop right after a tool_call so we can execute promptly.
283
+ out = generate(None, history, stop=["</tool_call>"])
284
+ raw = out["raw"]
285
+ # generate() stripped the closing tag via `stop`; restore it for parsing.
286
+ if "<tool_call>" in raw and "</tool_call>" not in raw:
287
+ raw = raw + "</tool_call>"
288
+ calls = _parse_tool_calls(raw)
289
+ history.append({"role": "assistant", "content": raw})
290
+
291
+ if verbose:
292
+ print(f"\n[agent step {step}] reasoning: {out['reasoning'][:200]}")
293
+ if calls:
294
+ print(f"[agent step {step}] tool calls: {calls}")
295
+
296
+ if not calls:
297
+ answer = out["answer"] or out["raw"].strip()
298
+ transcript.append({"step": step, "type": "final", "content": answer})
299
+ return {"answer": answer, "steps": step, "transcript": transcript}
300
+
301
+ # Execute every requested tool and feed results back as one user turn.
302
+ results = []
303
+ for c in calls:
304
+ fn = registry.get(c["name"])
305
+ if fn is None:
306
+ res = f"ERROR: unknown tool '{c['name']}'"
307
+ else:
308
+ try:
309
+ res = fn(**(c["arguments"] or {}))
310
+ except Exception as exc:
311
+ res = f"ERROR: {exc}"
312
+ results.append({"name": c["name"], "result": res})
313
+ transcript.append({"step": step, "type": "tool", "call": c, "result": res})
314
+ if verbose:
315
+ print(f"[agent step {step}] {c['name']} -> {str(res)[:200]}")
316
+
317
+ tool_msg = "\n".join(
318
+ f"<tool_response>{json.dumps(r, default=str)}</tool_response>" for r in results
319
+ )
320
+ history.append({"role": "user", "content": tool_msg})
321
+
322
+ return {"answer": "(stopped: max_steps reached)", "steps": max_steps,
323
+ "transcript": transcript}
324
+
325
+
326
+ # --------------------------------------------------------------------------- #
327
+ # Performance instrumentation: TPS, RAM, KV cache
328
+ # --------------------------------------------------------------------------- #
329
+ def _meta_int(llm, suffix):
330
+ """Read an int from GGUF metadata by key suffix (arch-agnostic)."""
331
+ for k, v in (getattr(llm, "metadata", {}) or {}).items():
332
+ if k.endswith(suffix):
333
+ try:
334
+ return int(v)
335
+ except (TypeError, ValueError):
336
+ pass
337
+ return None
338
+
339
+
340
+ def kv_cache_report(llm):
341
+ """
342
+ Estimate KV-cache memory from the model's attention geometry.
343
+ KV bytes/token = n_layer * n_head_kv * (key_len + val_len) * bytes_per_elem
344
+ (llama.cpp defaults the KV cache to f16 = 2 bytes/element.)
345
+ """
346
+ n_layer = _meta_int(llm, ".block_count")
347
+ n_embd = _meta_int(llm, ".embedding_length")
348
+ n_head = _meta_int(llm, ".attention.head_count")
349
+ n_head_kv = _meta_int(llm, ".attention.head_count_kv") or n_head
350
+ key_len = _meta_int(llm, ".attention.key_length")
351
+ val_len = _meta_int(llm, ".attention.value_length")
352
+ head_dim = key_len or ((n_embd // n_head) if (n_embd and n_head) else None)
353
+ key_len = key_len or head_dim
354
+ val_len = val_len or head_dim
355
+
356
+ info = {"n_layer": n_layer, "n_head": n_head, "n_head_kv": n_head_kv,
357
+ "head_dim": head_dim, "n_ctx": llm.n_ctx()}
358
+ if not (n_layer and n_head_kv and key_len and val_len):
359
+ info["note"] = "insufficient metadata to size KV cache"
360
+ return info
361
+
362
+ bytes_per_tok = n_layer * n_head_kv * (key_len + val_len) * 2 # f16
363
+ used_tokens = int(getattr(llm, "n_tokens", 0) or 0)
364
+ info.update({
365
+ "kv_bytes_per_token": bytes_per_tok,
366
+ "kv_full_mb": bytes_per_tok * llm.n_ctx() / (1024 ** 2),
367
+ "kv_used_mb": bytes_per_tok * used_tokens / (1024 ** 2),
368
+ "used_tokens": used_tokens,
369
+ })
370
+ return info
371
+
372
+
373
+ def benchmark(prompt="Write a Python function to check if a string is a palindrome, with a docstring.",
374
+ max_tokens=256):
375
+ """Run one generation on the current device and print a metrics table."""
376
+ import time
377
+ import psutil
378
+
379
+ llm = load_model()
380
+ proc = psutil.Process(os.getpid())
381
+
382
+ t0 = time.perf_counter()
383
+ t_first = None
384
+ text = ""
385
+ for piece in stream_chat(prompt, max_tokens=max_tokens):
386
+ if t_first is None:
387
+ t_first = time.perf_counter()
388
+ text += piece
389
+ t_end = time.perf_counter()
390
+ if t_first is None: # produced nothing
391
+ print("[bench] model produced no output"); return {}
392
+
393
+ gen_tokens = len(llm.tokenize(text.encode("utf-8"), add_bos=False))
394
+ used = int(getattr(llm, "n_tokens", 0) or 0)
395
+ prompt_tokens = max(used - gen_tokens, 0)
396
+
397
+ ttft = t_first - t0 # includes prompt prefill
398
+ decode_time = max(t_end - t_first, 1e-9)
399
+ total_time = t_end - t0
400
+ decode_tps = (gen_tokens - 1) / decode_time if gen_tokens > 1 else 0.0
401
+ prefill_tps = prompt_tokens / ttft if (prompt_tokens and ttft > 0) else 0.0
402
+
403
+ rss_gb = proc.memory_info().rss / (1024 ** 3)
404
+ avail_gb = psutil.virtual_memory().available / (1024 ** 3)
405
+ kv = kv_cache_report(llm)
406
+
407
+ use_gpu = _resolve_use_gpu()
408
+ print("\n" + "=" * 70)
409
+ print(f"PERFORMANCE — device={'GPU' if use_gpu else f'CPU ({os.cpu_count()} threads)'}, "
410
+ f"model={GGUF_REPO} {GGUF_FILE}")
411
+ print("=" * 70)
412
+ print(f" decode speed : {decode_tps:6.2f} tok/s <-- the headline TPS")
413
+ print(f" prefill speed : {prefill_tps:6.2f} tok/s (prompt processing)")
414
+ print(f" time to first token : {ttft:6.2f} s")
415
+ print(f" generated tokens : {gen_tokens} in {decode_time:.2f}s")
416
+ print(f" prompt tokens : {prompt_tokens}")
417
+ print(f" overall throughput : {gen_tokens / total_time:6.2f} tok/s (incl. prefill)")
418
+ print("-" * 70)
419
+ print(f" process RAM (RSS) : {rss_gb:6.2f} GB")
420
+ print(f" system RAM free : {avail_gb:6.2f} GB")
421
+ print("-" * 70)
422
+ if "kv_full_mb" in kv:
423
+ pct = 100 * kv["used_tokens"] / kv["n_ctx"] if kv["n_ctx"] else 0
424
+ print(f" context window : {kv['used_tokens']} / {kv['n_ctx']} tokens ({pct:.1f}% used)")
425
+ print(f" KV cache / token : {kv['kv_bytes_per_token'] / 1024:6.2f} KB")
426
+ print(f" KV cache (used) : {kv['kv_used_mb']:6.2f} MB")
427
+ print(f" KV cache (full ctx) : {kv['kv_full_mb']:6.2f} MB reserved for n_ctx={kv['n_ctx']}")
428
+ print(f" attn geometry : {kv['n_layer']} layers, "
429
+ f"{kv['n_head']} heads / {kv['n_head_kv']} KV heads (GQA), head_dim={kv['head_dim']}")
430
+ else:
431
+ print(f" KV cache : {kv.get('note', 'n/a')}")
432
+ print("=" * 70)
433
+
434
+ return {"decode_tps": decode_tps, "prefill_tps": prefill_tps, "ttft_s": ttft,
435
+ "gen_tokens": gen_tokens, "prompt_tokens": prompt_tokens,
436
+ "rss_gb": rss_gb, "kv": kv}
437
+
438
+
439
+ # --------------------------------------------------------------------------- #
440
+ # Demos
441
+ # --------------------------------------------------------------------------- #
442
+ def _demo_chat():
443
+ prompt = "Write a Python function that returns the nth Fibonacci number iteratively, with a docstring."
444
+ print("\n" + "=" * 70 + f"\nCHAT DEMO\nPROMPT: {prompt}\n" + "=" * 70)
445
+
446
+ mode, buf = "reasoning", ""
447
+ print("\n--- reasoning ---")
448
+ for piece in stream_chat(prompt):
449
+ buf += piece
450
+ if mode == "reasoning":
451
+ idx = buf.find(_THINK_CLOSE)
452
+ if idx != -1: # crossed into the answer
453
+ sys.stdout.write(buf[:idx])
454
+ print("\n\n--- answer ---")
455
+ sys.stdout.write(buf[idx + len(_THINK_CLOSE):])
456
+ mode, buf = "answer", ""
457
+ else: # hold back a tail so the tag can't split
458
+ keep = len(_THINK_CLOSE)
459
+ if len(buf) > keep:
460
+ sys.stdout.write(buf[:-keep])
461
+ buf = buf[-keep:]
462
+ else:
463
+ sys.stdout.write(piece)
464
+ sys.stdout.flush()
465
+ if buf:
466
+ sys.stdout.write(buf)
467
+ print("\n" + "=" * 70)
468
+
469
+
470
+ def _demo_agent():
471
+ print("\n" + "=" * 70 + "\nAGENT DEMO (tool calling)\n" + "=" * 70)
472
+
473
+ def calculator(expression: str):
474
+ """Safely evaluate a basic arithmetic expression."""
475
+ if not re.fullmatch(r"[0-9+\-*/().%\s]+", expression or ""):
476
+ return "ERROR: only arithmetic allowed"
477
+ return eval(expression, {"__builtins__": {}}, {}) # sandboxed namespace
478
+
479
+ tools = [{
480
+ "name": "calculator",
481
+ "schema": {
482
+ "name": "calculator",
483
+ "description": "Evaluate a basic arithmetic expression and return the number.",
484
+ "parameters": {
485
+ "type": "object",
486
+ "properties": {"expression": {"type": "string",
487
+ "description": "e.g. '(1234*7) + 89'"}},
488
+ "required": ["expression"],
489
+ },
490
+ },
491
+ "fn": calculator,
492
+ }]
493
+
494
+ result = run_agent(
495
+ "What is (1234 * 7) + 89, and then that result divided by 3? "
496
+ "Use the calculator tool for each arithmetic step.",
497
+ tools, max_steps=6,
498
+ )
499
+ print("\n--- FINAL ANSWER ---")
500
+ print(result["answer"])
501
+ print("=" * 70)
502
+
503
+
504
+ if __name__ == "__main__":
505
+ benchmark() # showcase device capability: TPS, RAM, KV cache
506
+ _demo_chat()
507
+ _demo_agent()
508
+ print("Import chat / generate / run_agent / benchmark from this file.")