File size: 7,120 Bytes
f1bca7d | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 | #!/usr/bin/env python3
r"""Phase G -- GEN-AT-DEPTH: actually RUN a servable chakra-tier (a real GGUF) via
the existing llama-server launcher and GENERATE tokens at a chosen depth (the
"deep slow stream"). Machine-first: ONE tier resident at a time; his VRAM_CAP_MB
(683, "one third") is respected first -- the auto-fit ladder settles at ngl=0
(all weights CPU/mmap) for the big models, which still runs; only if a tier
REFUSES under the third do we disclose and offer a raised cap. Unload after.
TAV chakra -> Adelic tier (grades 1-5). GGUF tiers run direct; the safetensors
tiers (Third Eye 1B / Throat 8B) take the torus-retrofit path (later) or an
on-disk GGUF proxy -- disclosed.
"""
from __future__ import annotations
import sys
import time
sys.path.insert(0, r"D:\Holorites\torus_upgrades")
import phase30_model as PM # noqa: E402
# chakra -> (port, gguf path, adelic tier note). One resident at a time on 4 GB.
TIERS = {
"heart": (8103, r"D:\sneedjak_models\Adelic-Gemma-4-12B-GGUF\adelic-gemma4-12b-Q6_K.gguf",
"grade 3 / 12B / relation-continuity"),
"solar": (8104, r"D:\sneedjak_models\Adelic-Qwen3.6-27B-Topology\adelic-qwen-27b-q8_0.gguf",
"grade 4 / 27B / verify-contradiction"),
"sacral": (8105, r"D:\sneedjak_models\Adelic-Gemma-4-31B-it\adelic-gemma4-31b-Q4_K_M.gguf",
"grade 5 / 31B / deep synthesis"),
# fast tiers are safetensors -> on-disk GGUF proxies until the retrofit path:
"throat": (8102, r"D:\0000_Raw_LLM Models\Qwen3-8B-abliterated.Q4_K_M.gguf",
"grade 2 / 8B proxy / language"),
# the PHONE'S OWN MIND, resident on the desktop for 1:1 evaluation of mobile Tav'iel
"root": (8106, r"D:\0000_Raw_LLM Models\Qwen2.5-1.5B-Instruct-Q4_K_M.gguf",
"grade 0 / 1.5B / the mobile tier"),
"third_eye": (8101, r"D:\0000_Raw_LLM Models\Qwen3-0.6B-Q8_0.gguf",
"grade 1 / 0.6B proxy / recognition-routing"),
}
def open_tier(chakra, cap_mb=683, total_mb=4096, ctx=2048):
"""Start one chakra-tier's llama-server, auto-fitting under the VRAM cap.
Returns a live LlamaModel (its own __enter__ already run) or raises VramRefusal.
Respects his 683 'third' by default; raise cap_mb only when a tier needs it."""
if chakra not in TIERS:
raise KeyError("unknown chakra %r; have %s" % (chakra, list(TIERS)))
port, path, note = TIERS[chakra]
PM.VRAM_CAP_MB = cap_mb # runtime override (this process only)
print("[gen] opening %s tier (%s) on port %d, cap %d MB ..." % (chakra, note, port, cap_mb),
flush=True)
# start the ladder at ngl=0 (all CPU/mmap) so a big GGUF with no MEASURED entry
# does NOT try to load all layers onto the 4 GB card and time out. MEASURED
# tiers (0.6B/8B) ignore this and use their tuned ladder.
#
# RESOURCE-BOUNDED STREAMING for the big tiers: heart 12B / solar 27B / sacral 31B are the
# disk-streamed voices. They are GUARANTEED never to overload the box -- they run at IDLE OS
# priority (only spare CPU cycles), on just a few threads (cores stay free for the RAM voice
# + other apps), with a small batch and a bounded context (bounded KV memory), and mmap keeps
# the weights paging from disk instead of all in RAM. That is the trickle: they still flow
# and contribute, but they physically cannot starve the system.
BIG = {"heart", "solar", "sacral"}
if chakra in BIG:
# below-normal priority (yields to the RAM voice + his foreground apps, so it can never
# overload) but NOT dead-idle, so it still progresses to an answer -- a real trickle that
# reaches B. Capped threads + small batch + bounded ctx + mmap keep it bounded.
m = PM.LlamaModel(path, gpu_layers=0, ctx=min(ctx, 2048), port=port, total_mb=total_mb,
threads=4, priority="belownormal", batch=128)
else:
m = PM.LlamaModel(path, gpu_layers=0, ctx=ctx, port=port, total_mb=total_mb)
m.__enter__() # auto-fit ladder -> ngl; VramRefusal if none
m._chakra = chakra
return m
def generate(m, prompt, grounding="", max_tokens=160, temperature=0.7, system=None):
"""Generate at this tier's depth. `grounding` (lexicon/Bible/etc) is prepended
to the user prompt; `system` (an identity/instruction) goes to the system role
-- the model carries language, the grounding carries the truth."""
full = (grounding.rstrip() + "\n\n" + prompt) if grounding else prompt
t0 = time.time()
text = m.complete(full, max_tokens=max_tokens, temperature=temperature, system=system)
dt = time.time() - t0
# truncate at any chat-template turn marker OR special/multimodal token the model
# echoed past its answer (Gemma-4 is multimodal -- its <image|> token can loop
# at ngl=0; the coherent text prefix before it is the answer).
for mark in ("<|im_end|>", "<|im_start|>", "<end_of_turn>", "<start_of_turn>",
"<image", "<unused", "<pad>", "<eos>"):
i = text.find(mark)
if i != -1:
text = text[:i]
text = text.strip()
ntok = max(1, len(text.split()))
return {"text": text, "seconds": round(dt, 1),
"approx_tok_s": round(ntok / dt, 2) if dt else None,
"ngl": m.gpu_layers, "chakra": m._chakra}
def generate_stream(m, prompt, grounding="", max_tokens=768, temperature=0.7, system=None):
"""Stream generation at this tier's depth -- yields text deltas as they arrive."""
full = (grounding.rstrip() + "\n\n" + prompt) if grounding else prompt
MARKS = ("<|im_end|>", "<|im_start|>", "<end_of_turn>", "<start_of_turn>",
"<image", "<unused", "<pad>", "<eos>")
for delta in m.complete_stream(full, max_tokens=max_tokens, temperature=temperature,
system=system):
for mark in MARKS: # stop at any turn/special marker the model echoes
if mark in delta:
delta = delta.split(mark)[0]
if delta:
yield delta
def close_tier(m):
try:
m.stop()
except Exception:
pass
def run_one(chakra, prompt, cap_mb=683):
"""Open a tier, generate once, unload -- the full machine-first cycle."""
m = None
try:
m = open_tier(chakra, cap_mb=cap_mb)
r = generate(m, prompt)
print("[gen] %s: ngl=%s %.1fs ~%.2f tok/s" % (chakra, r["ngl"], r["seconds"],
r["approx_tok_s"] or 0), flush=True)
print("[gen] OUTPUT:\n" + r["text"], flush=True)
return r
finally:
if m is not None:
close_tier(m)
print("[gen] %s unloaded (nothing left resident)." % chakra, flush=True)
if __name__ == "__main__":
chakra = sys.argv[1] if len(sys.argv) > 1 else "heart"
prompt = sys.argv[2] if len(sys.argv) > 2 else \
"In one sentence, what is the purpose of a vessel that remembers?"
cap = int(sys.argv[3]) if len(sys.argv) > 3 else 683
run_one(chakra, prompt, cap_mb=cap)
|