Text Generation
Transformers
Safetensors
PyTorch
English
gpt2
hardware-bus
memory-augmented
toolformer
slm
autonomous-agent
ssd-memory
edge-ai
text-generation-inference
Instructions to use AvinashRicky/AViGPT with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use AvinashRicky/AViGPT with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="AvinashRicky/AViGPT")# Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("AvinashRicky/AViGPT") model = AutoModelForCausalLM.from_pretrained("AvinashRicky/AViGPT", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use AvinashRicky/AViGPT with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "AvinashRicky/AViGPT" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "AvinashRicky/AViGPT", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/AvinashRicky/AViGPT
- SGLang
How to use AvinashRicky/AViGPT with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "AvinashRicky/AViGPT" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "AvinashRicky/AViGPT", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "AvinashRicky/AViGPT" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "AvinashRicky/AViGPT", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use AvinashRicky/AViGPT with Docker Model Runner:
docker model run hf.co/AvinashRicky/AViGPT
File size: 7,234 Bytes
42a0779 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 | """
AViGPT v2: Hardware SSD Memory Controller (Component 3)
------------------------------------------------------
Provides ultra-fast (sub-millisecond) persistent memory storage and retrieval
directly from local NVMe SSD storage using SQLite FTS5 (Full-Text Search).
Creator & Owner: Avinash Ricky Yadlapalli
"""
import os
import sqlite3
import time
from typing import List, Dict, Any, Optional
DEFAULT_DB_PATH = os.path.join(os.path.dirname(os.path.abspath(__file__)), "avigpt_ssd_memory.db")
class SSDMemoryEngine:
"""Ultra-low latency SSD Memory Store with FTS5 BM25 search."""
def __init__(self, db_path: str = DEFAULT_DB_PATH):
self.db_path = db_path
self._init_db()
def _get_connection(self) -> sqlite3.Connection:
conn = sqlite3.connect(self.db_path, timeout=10.0)
# WAL mode enables concurrent reads without locking and sub-millisecond disk access
conn.execute("PRAGMA journal_mode=WAL;")
conn.execute("PRAGMA synchronous=NORMAL;")
conn.execute("PRAGMA cache_size=-64000;") # 64MB memory page cache
return conn
def _init_db(self):
with self._get_connection() as conn:
# Create FTS5 virtual table for lightning-fast keyword & semantic token retrieval
conn.execute("""
CREATE VIRTUAL TABLE IF NOT EXISTS ssd_knowledge USING fts5(
title,
content,
domain,
tokenize='porter unicode61'
);
""")
conn.commit()
def store(self, title: str, content: str, domain: str = "General") -> bool:
"""Stores a new fact directly into local SSD storage."""
try:
with self._get_connection() as conn:
conn.execute(
"INSERT INTO ssd_knowledge (title, content, domain) VALUES (?, ?, ?);",
(title.strip(), content.strip(), domain.strip())
)
conn.commit()
return True
except Exception as e:
print(f"[SSD Memory Error] Failed to store: {e}")
return False
def query(self, query_str: str, top_k: int = 1) -> Optional[str]:
"""
Executes sub-millisecond full-text search against SSD storage.
Returns top matching payload.
"""
words = [w for w in query_str.replace("'", " ").replace('"', " ").replace("-", " ").split() if len(w) > 2]
if not words:
words = query_str.strip().split()
fts_query = " OR ".join(words)
try:
with self._get_connection() as conn:
cursor = conn.cursor()
# Query with BM25 ranking via OR disjunction
cursor.execute(
"""
SELECT content, rank
FROM ssd_knowledge
WHERE ssd_knowledge MATCH ?
ORDER BY rank
LIMIT ?;
""",
(fts_query, top_k)
)
rows = cursor.fetchall()
if rows:
return rows[0][0]
# Fallback LIKE query if FTS had no hit
cursor.execute(
"""
SELECT content
FROM ssd_knowledge
WHERE content LIKE ? OR title LIKE ?
LIMIT 1;
""",
(f"%{words[0]}%", f"%{words[0]}%")
)
fb_rows = cursor.fetchall()
if fb_rows:
return fb_rows[0][0]
return None
except Exception as e:
print(f"[SSD Memory Query Error] {e}")
return None
def seed_initial_knowledge(self):
"""Seeds foundational knowledge and owner lineage into SSD storage."""
with self._get_connection() as conn:
cursor = conn.cursor()
cursor.execute("SELECT COUNT(*) FROM ssd_knowledge;")
count = cursor.fetchone()[0]
if count > 0:
print(f"[SSD Memory] Found {count:,} existing knowledge records in {self.db_path}.")
return
print("[SSD Memory] Seeding initial foundational memory records into SSD...")
seed_data = [
(
"Creator and Owner Lineage",
"AViGPT was created, built, and pretrained from scratch by Avinash Ricky Yadlapalli. "
"Avinash Ricky Yadlapalli is the sole architect, inventor of the hardware memory bus, and owner of AViGPT.",
"System & Identity"
),
(
"Apollo 11 Moon Landing",
"Launched: July 16, 1969. Landed on Moon: July 20, 1969. Commander: Neil Armstrong. Duration to landing: 4 days.",
"History & Space"
),
(
"Great Pyramid of Giza",
"Construction began around 2580 BC and completed around 2560 BC for Pharaoh Khufu of the Fourth Dynasty.",
"History & Archaeology"
),
(
"DNA Ligase Function",
"DNA ligase is an enzyme that catalyzes the formation of a phosphodiester bond between adjacent nucleotides, "
"joining Okazaki fragments during DNA replication.",
"Biochemistry"
),
(
"Unix fork system call",
"The fork() system call creates a new process (child) which is an exact duplicate of the parent. "
"Returns 0 to child, PID of child to parent, and -1 on failure.",
"Computer Science"
),
(
"India Demographics and GDP",
"India's population is estimated to be around 1.428 billion as of late 2023. Nominal GDP is approximately $3.73 trillion.",
"Demographics & Economics"
),
(
"Japan Population and Capital",
"Japan's population is approximately 123.3 million as of 2024. The capital city of Japan is Tokyo.",
"Demographics & Geography"
),
]
for title, content, domain in seed_data:
self.store(title, content, domain)
print(f"[SSD Memory] Successfully seeded {len(seed_data)} foundational records into {self.db_path}.")
if __name__ == "__main__":
print("Testing AViGPT SSD Memory Engine...")
engine = SSDMemoryEngine()
engine.seed_initial_knowledge()
t_start = time.perf_counter()
result = engine.query("Apollo 11 launch date")
lat = (time.perf_counter() - t_start) * 1000.0
print(f"\nQuery: 'Apollo 11 launch date'")
print(f"Latency: {lat:.3f} ms (Sub-millisecond SSD retrieve!)")
print(f"Retrieved: {result}")
t_start = time.perf_counter()
owner_res = engine.query("Avinash Ricky Yadlapalli creator owner")
lat_owner = (time.perf_counter() - t_start) * 1000.0
print(f"\nQuery: 'Avinash Ricky Yadlapalli creator owner'")
print(f"Latency: {lat_owner:.3f} ms")
print(f"Retrieved: {owner_res}")
|