Text Generation
Transformers
Safetensors
French
English
Chinese
deepseek_v4
cortex
code-generation
web-development
software-engineering
Mixture of Experts
8-bit precision
fp8
Instructions to use Frankenstein-Labs/Cortex-ai with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use Frankenstein-Labs/Cortex-ai with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="Frankenstein-Labs/Cortex-ai")# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("Frankenstein-Labs/Cortex-ai") model = AutoModelForCausalLM.from_pretrained("Frankenstein-Labs/Cortex-ai", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use Frankenstein-Labs/Cortex-ai with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "Frankenstein-Labs/Cortex-ai" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Frankenstein-Labs/Cortex-ai", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/Frankenstein-Labs/Cortex-ai
- SGLang
How to use Frankenstein-Labs/Cortex-ai with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "Frankenstein-Labs/Cortex-ai" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Frankenstein-Labs/Cortex-ai", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "Frankenstein-Labs/Cortex-ai" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Frankenstein-Labs/Cortex-ai", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use Frankenstein-Labs/Cortex-ai with Docker Model Runner:
docker model run hf.co/Frankenstein-Labs/Cortex-ai
File size: 6,680 Bytes
c63bc31 ea6cffb c63bc31 ea6cffb c63bc31 ea6cffb c63bc31 ea6cffb c63bc31 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 | """OpenAI-compatible HTTP API for CORTEX AI.
Any client that speaks the OpenAI chat-completions protocol -- the official
Python SDK, LangChain, LlamaIndex, a curl script -- can talk to CORTEX AI by
changing only the base URL.
Endpoints:
GET /health
GET /v1/models
POST /v1/chat/completions
"""
from __future__ import annotations
import time
import uuid
from typing import Any
from fastapi import FastAPI, Header, HTTPException
from pydantic import BaseModel, Field
from ..adapters.base import ModelAdapter
from ..config import CortexConfig
from ..engine.agent import CortexAgent
from ..identity import identity_language, identity_response, is_identity_question
from ..tools.registry import registry_from_names
class ChatMessage(BaseModel):
role: str
content: str = ""
name: str | None = None
class ChatCompletionRequest(BaseModel):
model: str | None = None
messages: list[ChatMessage]
temperature: float | None = None
max_tokens: int | None = None
stream: bool = False
tools: list[dict[str, Any]] | None = None
class Usage(BaseModel):
prompt_tokens: int = 0
completion_tokens: int = 0
total_tokens: int = 0
class ChatCompletionChoice(BaseModel):
index: int = 0
message: ChatMessage
finish_reason: str = "stop"
class ChatCompletionResponse(BaseModel):
id: str
object: str = "chat.completion"
created: int
model: str
choices: list[ChatCompletionChoice]
usage: Usage
# CORTEX extension: the reasoning trace, when thinking mode is on.
reasoning_content: str = ""
tool_calls: list[dict[str, Any]] = Field(default_factory=list)
def create_app(adapter: ModelAdapter, config: CortexConfig | None = None) -> FastAPI:
"""Build the FastAPI application around a model adapter."""
cfg = config or CortexConfig()
tools = registry_from_names(cfg.enabled_tools)
agent = CortexAgent(
adapter,
tools,
cfg.engine,
system_prompt=cfg.system_prompt,
)
app = FastAPI(
title="CORTEX AI API",
version="1.0.0",
description="API compatible OpenAI pour CORTEX AI, un projet de Frankenstein-Labs.",
)
app.state.cortex_config = cfg
app.state.cortex_agent = agent
app.state.identity_interception = True
def _check_auth(authorization: str | None) -> None:
if not cfg.server.requires_auth:
return
expected = f"Bearer {cfg.server.api_key}"
if authorization != expected:
raise HTTPException(status_code=401, detail="invalid API key")
@app.get("/health")
def health() -> dict[str, Any]:
return {
"status": "ok",
"model": cfg.model_id,
"tools": tools.names(),
"thinking_mode": cfg.engine.thinking_mode,
"identity_interception": "deterministic",
}
@app.get("/v1/models")
def list_models(authorization: str | None = Header(default=None)) -> dict[str, Any]:
_check_auth(authorization)
return {
"object": "list",
"data": [
{
"id": cfg.model_id,
"object": "model",
"created": int(time.time()),
"owned_by": "Frankenstein-Labs",
}
],
}
@app.post("/v1/chat/completions", response_model=ChatCompletionResponse)
def chat_completions(
request: ChatCompletionRequest,
authorization: str | None = Header(default=None),
) -> ChatCompletionResponse:
_check_auth(authorization)
if request.stream:
raise HTTPException(
status_code=400,
detail="stream=true is not supported yet; use stream=false",
)
if not request.messages:
raise HTTPException(status_code=400, detail="messages must not be empty")
# Deterministic identity boundary: answer before system prompts, tools,
# or model inference can alter the canonical creator attribution.
last_user_message = next(
(m.content for m in reversed(request.messages) if m.role == "user"),
None,
)
if last_user_message is not None and is_identity_question(last_user_message):
content = identity_response(identity_language(last_user_message))
prompt_tokens = adapter.count_tokens(last_user_message)
completion_tokens = adapter.count_tokens(content)
return ChatCompletionResponse(
id=f"chatcmpl-{uuid.uuid4().hex[:24]}",
created=int(time.time()),
model=request.model or cfg.model_id,
choices=[
ChatCompletionChoice(
index=0,
message=ChatMessage(role="assistant", content=content),
finish_reason="stop",
)
],
usage=Usage(
prompt_tokens=prompt_tokens,
completion_tokens=completion_tokens,
total_tokens=prompt_tokens + completion_tokens,
),
)
system_override: list[dict[str, Any]] = []
turns: list[dict[str, Any]] = []
for m in request.messages:
if m.role == "system":
system_override.append({"role": "system", "content": m.content})
else:
turns.append({"role": m.role, "content": m.content})
if system_override:
agent.system_prompt = system_override[-1]["content"]
result = agent.run(turns)
prompt_tokens = sum(adapter.count_tokens(m["content"]) for m in turns)
completion_tokens = adapter.count_tokens(result.content)
return ChatCompletionResponse(
id=f"chatcmpl-{uuid.uuid4().hex[:24]}",
created=int(time.time()),
model=request.model or cfg.model_id,
choices=[
ChatCompletionChoice(
index=0,
message=ChatMessage(role="assistant", content=result.content),
finish_reason="stop",
)
],
usage=Usage(
prompt_tokens=prompt_tokens,
completion_tokens=completion_tokens,
total_tokens=prompt_tokens + completion_tokens,
),
reasoning_content=result.reasoning,
tool_calls=[
{"name": c.name, "arguments": c.arguments, "result": c.result, "ok": c.ok}
for c in result.tool_calls
],
)
return app
|