Spaces:
Running
Running
Download main.py from mjpsm/activity-composition-api: direct link, hf CLI and curl.
- Browser
- Download file 10.5 kB
-
https://huggingface.co/spaces/mjpsm/activity-composition-api/resolve/main/main.py
- Command line
-
hf download hf://spaces/mjpsm/activity-composition-api/main.py
-
curl -L -o main.py https://huggingface.co/spaces/mjpsm/activity-composition-api/resolve/main/main.py
10.5 kB
| import json | |
| import logging | |
| import os | |
| import re | |
| import threading | |
| import time | |
| import uuid | |
| from contextlib import asynccontextmanager | |
| from enum import Enum | |
| from typing import List | |
| from fastapi import FastAPI, HTTPException, Request | |
| from fastapi.middleware.cors import CORSMiddleware | |
| from huggingface_hub import hf_hub_download | |
| from llama_cpp import Llama | |
| from pydantic import BaseModel, ConfigDict, Field | |
| # ============================================================ | |
| # Logging | |
| # ============================================================ | |
| logging.basicConfig( | |
| level=os.getenv("LOG_LEVEL", "INFO").upper(), | |
| format="%(asctime)s %(levelname)s %(name)s %(message)s", | |
| ) | |
| logger = logging.getLogger("activity-composition-api") | |
| # ============================================================ | |
| # Model configuration | |
| # ============================================================ | |
| MODEL_REPO = os.getenv( | |
| "MODEL_REPO", | |
| "mjpsm/activity-composition-model-400-qwen3.5-0.8b-gguf", | |
| ) | |
| MODEL_FILENAME = os.getenv( | |
| "MODEL_FILENAME", | |
| "activity-composition-model-400-qwen3.5-0.8b-Q8_0.gguf", | |
| ) | |
| N_CTX = int(os.getenv("N_CTX", "2048")) | |
| N_THREADS = int(os.getenv("N_THREADS", "2")) | |
| MAX_NEW_TOKENS = int(os.getenv("MAX_NEW_TOKENS", "220")) | |
| REPEAT_PENALTY = float(os.getenv("REPEAT_PENALTY", "1.05")) | |
| SEED = int(os.getenv("SEED", "42")) | |
| llm: Llama | None = None | |
| model_lock = threading.Lock() | |
| # ============================================================ | |
| # Exact Composition prompt used during training/evaluation | |
| # ============================================================ | |
| SYSTEM_INSTRUCTION = """You are the MyVillage Activity Composition Model. | |
| The Activity Progression Model has already decided what the villager should do next. | |
| Your only job is to compose that progression into a short activity title and a one-sentence activity description. | |
| Return ONLY valid JSON. | |
| Required output schema: | |
| { | |
| "output": { | |
| "activity_title": "string", | |
| "activity_description": "string" | |
| } | |
| } | |
| Rules: | |
| - Trust the progression input. Do not choose a different next activity. | |
| - Base the title and description only on the progression provided. | |
| - Keep the title short, clear, professional, and villager-friendly. | |
| - The title should represent the progression itself and should avoid unnecessary project or program names when the activity remains clear without them. | |
| - The description must be exactly one explanatory sentence about what the activity is. | |
| - Write the description descriptively, not as a direct command or step-by-step instruction. | |
| - Descriptive noun phrases and gerund constructions are preferred when natural. | |
| - Do not summarize what the villager previously accomplished. | |
| - Do not use phrases such as 'You previously', 'You learned', 'You successfully', 'You discovered', or 'You already'. | |
| - Do not assign work listed in avoid_repeating. | |
| - Do not turn demonstrated_state into new work. | |
| - Do not invent unsupported goals, requirements, tools, deliverables, or outcomes. | |
| - Do not generate instructions, numbered steps, bullets, or commentary. | |
| """ | |
| # ============================================================ | |
| # API schemas | |
| # ============================================================ | |
| class KnowledgeState(str, Enum): | |
| COMPLETED = "COMPLETED" | |
| PARTIAL = "PARTIAL" | |
| UNCLEAR = "UNCLEAR" | |
| BLOCKED = "BLOCKED" | |
| ADVANCED = "ADVANCED" | |
| class ActivityPlan(BaseModel): | |
| model_config = ConfigDict(extra="forbid") | |
| action: str = Field(min_length=1) | |
| target: str = Field(min_length=1) | |
| avoid_repeating: List[str] = Field(default_factory=list) | |
| class Progression(BaseModel): | |
| model_config = ConfigDict(extra="forbid") | |
| knowledge_state: KnowledgeState | |
| demonstrated_state: List[str] = Field(default_factory=list) | |
| next_gap: str = Field(min_length=1) | |
| activity_plan: ActivityPlan | |
| class CompositionRequest(BaseModel): | |
| model_config = ConfigDict(extra="forbid") | |
| progression: Progression | |
| class CompositionOutput(BaseModel): | |
| model_config = ConfigDict(extra="forbid") | |
| activity_title: str = Field(min_length=1) | |
| activity_description: str = Field(min_length=1) | |
| class CompositionResponse(BaseModel): | |
| model_config = ConfigDict(extra="forbid") | |
| output: CompositionOutput | |
| # ============================================================ | |
| # Prompt helpers | |
| # ============================================================ | |
| def compact_json(value: dict) -> str: | |
| return json.dumps( | |
| value, | |
| ensure_ascii=False, | |
| separators=(",", ":"), | |
| ) | |
| def build_prompt(input_obj: dict) -> str: | |
| return ( | |
| SYSTEM_INSTRUCTION.strip() | |
| + "\n\nINPUT:\n" | |
| + compact_json(input_obj) | |
| + "\n\nOUTPUT:\n" | |
| ) | |
| def extract_json(text: str): | |
| decoder = json.JSONDecoder() | |
| for match in re.finditer(r"\{", text): | |
| try: | |
| obj, _ = decoder.raw_decode(text[match.start():]) | |
| if isinstance(obj, dict): | |
| return obj | |
| except json.JSONDecodeError: | |
| pass | |
| return None | |
| # ============================================================ | |
| # Load GGUF directly with llama-cpp-python | |
| # ============================================================ | |
| def load_model() -> Llama: | |
| logger.info( | |
| "Downloading GGUF repo=%s filename=%s", | |
| MODEL_REPO, | |
| MODEL_FILENAME, | |
| ) | |
| model_path = hf_hub_download( | |
| repo_id=MODEL_REPO, | |
| filename=MODEL_FILENAME, | |
| token=os.getenv("HF_TOKEN") or None, | |
| ) | |
| logger.info("Loading GGUF with llama-cpp-python") | |
| model = Llama( | |
| model_path=model_path, | |
| n_ctx=N_CTX, | |
| n_threads=N_THREADS, | |
| n_gpu_layers=0, | |
| seed=SEED, | |
| verbose=False, | |
| ) | |
| logger.info("Model loaded successfully") | |
| return model | |
| async def lifespan(app: FastAPI): | |
| global llm | |
| llm = load_model() | |
| yield | |
| llm = None | |
| # ============================================================ | |
| # FastAPI | |
| # ============================================================ | |
| app = FastAPI( | |
| title="MyVillage Activity Composition API", | |
| version="1.0.0", | |
| description=( | |
| "Converts Activity Progression Model output into an activity title " | |
| "and one-sentence activity description." | |
| ), | |
| lifespan=lifespan, | |
| ) | |
| # ============================================================ | |
| # CORS | |
| # ============================================================ | |
| origins = os.getenv("ALLOWED_ORIGINS", "*").strip() | |
| app.add_middleware( | |
| CORSMiddleware, | |
| allow_origins=["*"] if origins == "*" else [ | |
| origin.strip() | |
| for origin in origins.split(",") | |
| if origin.strip() | |
| ], | |
| allow_credentials=False, | |
| allow_methods=["*"], | |
| allow_headers=["*"], | |
| ) | |
| # ============================================================ | |
| # Request logging | |
| # ============================================================ | |
| async def request_logging(request: Request, call_next): | |
| request_id = request.headers.get("x-request-id") or str(uuid.uuid4()) | |
| started = time.perf_counter() | |
| response = await call_next(request) | |
| elapsed = time.perf_counter() - started | |
| response.headers["x-request-id"] = request_id | |
| logger.info( | |
| "%s %s status=%s request_id=%s elapsed=%.4fs", | |
| request.method, | |
| request.url.path, | |
| response.status_code, | |
| request_id, | |
| elapsed, | |
| ) | |
| return response | |
| # ============================================================ | |
| # Inference | |
| # ============================================================ | |
| def generate_composition(input_obj: dict) -> CompositionResponse: | |
| if llm is None: | |
| raise HTTPException( | |
| status_code=503, | |
| detail="Model is not loaded yet.", | |
| ) | |
| prompt = build_prompt(input_obj) | |
| try: | |
| with model_lock: | |
| result = llm.create_completion( | |
| prompt=prompt, | |
| max_tokens=MAX_NEW_TOKENS, | |
| temperature=0.0, | |
| repeat_penalty=REPEAT_PENALTY, | |
| seed=SEED, | |
| echo=False, | |
| ) | |
| except Exception as exc: | |
| logger.exception("llama-cpp-python inference failed") | |
| raise HTTPException( | |
| status_code=500, | |
| detail="Model inference failed.", | |
| ) from exc | |
| raw_text = result["choices"][0]["text"].strip() | |
| parsed = extract_json(raw_text) | |
| if parsed is None: | |
| logger.error("Model returned non-JSON output: %r", raw_text[:2000]) | |
| raise HTTPException( | |
| status_code=502, | |
| detail="Model returned invalid JSON.", | |
| ) | |
| try: | |
| return CompositionResponse.model_validate(parsed) | |
| except Exception as exc: | |
| logger.error( | |
| "Model output failed schema validation: %s raw=%r", | |
| exc, | |
| raw_text[:2000], | |
| ) | |
| raise HTTPException( | |
| status_code=502, | |
| detail="Model returned JSON with an invalid schema.", | |
| ) from exc | |
| # ============================================================ | |
| # Routes | |
| # ============================================================ | |
| def root(): | |
| return { | |
| "name": "MyVillage Activity Composition API", | |
| "status": "online", | |
| "model_repo": MODEL_REPO, | |
| "model_file": MODEL_FILENAME, | |
| "quantization": "Q8_0", | |
| "docs": "/docs", | |
| "health": "/health", | |
| "generate": "/generate", | |
| } | |
| def health(): | |
| if llm is None: | |
| raise HTTPException( | |
| status_code=503, | |
| detail="Model is not loaded yet.", | |
| ) | |
| return { | |
| "status": "ok", | |
| "model_loaded": True, | |
| "model_repo": MODEL_REPO, | |
| "model_file": MODEL_FILENAME, | |
| "quantization": "Q8_0", | |
| } | |
| def model_info(): | |
| return { | |
| "source_model": "mjpsm/activity-composition-model-400-qwen3.5-0.8b", | |
| "gguf_repo": MODEL_REPO, | |
| "gguf_file": MODEL_FILENAME, | |
| "quantization": "Q8_0", | |
| "runtime": "llama-cpp-python", | |
| "n_ctx": N_CTX, | |
| "n_threads": N_THREADS, | |
| "max_new_tokens": MAX_NEW_TOKENS, | |
| "temperature": 0.0, | |
| "repeat_penalty": REPEAT_PENALTY, | |
| "seed": SEED, | |
| } | |
| def generate(request: CompositionRequest): | |
| return generate_composition( | |
| request.model_dump(mode="json") | |
| ) | |