Download src/tools/spec_builder.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 5.53 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/tools/spec_builder.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/tools/spec_builder.py
-
curl -L -o spec_builder.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/tools/spec_builder.py
5.53 kB
| """spec_builder — derive a ToolSpec from a tool's pydantic InputModel (analytical tools round). | |
| The shipped analytics tools are pure compute functions (`src/tools/analytics/*.py`) | |
| registered as hand-written `ToolSpec` entries (`src/tools/registry.py`). As the new | |
| analytical tools land, each one declares a pydantic **InputModel** whose fields ARE | |
| its call arguments; this module turns that model — plus the three-part description | |
| the Planner reads — into the exact `ToolSpec` shape the registry already uses, so the | |
| model and the registry entry can never drift. | |
| Why not a base class (decided with the agent team): the tools STAY pure functions — | |
| the repo's deliberate "compute-only, easy to unit-test" design (see the STATUS note | |
| in `analytics/temporal.py`). This helper is the entire "framework": one function to | |
| build the registry spec from a model, one to compose the description string. A tool | |
| inherits from nothing. | |
| Contract kept byte-compatible with the hand-written specs (`src/tools/contracts.py`): | |
| input_schema = {"required": [...], "properties": {arg: {"type": ...}}} | |
| The planner validator (Check 8a) reads `required` (presence-only) and treats | |
| `properties` as prompt DOCUMENTATION, not a runtime type check (see the ToolSpec | |
| docstring). pydantic's `model_json_schema()` emits a richer schema (nested `$defs`, | |
| `anyOf` for optionals, per-field `title`); `input_schema_from_model` SIMPLIFIES it to | |
| the shape above — one primary JSON type per field, plus any `Field(description=...)` | |
| carried through so the per-input descriptions the tool team asked for reach the | |
| planner prompt. | |
| """ | |
| from __future__ import annotations | |
| from typing import Any | |
| from pydantic import BaseModel | |
| from src.tools.contracts import ToolSpec | |
| _NULL = "null" | |
| def _primary_type(prop: dict[str, Any]) -> str | None: | |
| """The single JSON type to advertise for one field. | |
| Handles the three shapes pydantic emits: a direct ``{"type": ...}``; an | |
| optional/union ``anyOf``/``oneOf`` (take the first non-null member); and a | |
| ``$ref`` to a nested model or enum (no primitive type — return None so the | |
| caller omits ``type`` rather than inventing one). | |
| """ | |
| t = prop.get("type") | |
| if isinstance(t, str): | |
| return t | |
| for union_key in ("anyOf", "oneOf"): | |
| for sub in prop.get(union_key, []): | |
| sub_t = sub.get("type") | |
| if isinstance(sub_t, str) and sub_t != _NULL: | |
| return sub_t | |
| return None | |
| def _simplify(prop: dict[str, Any]) -> dict[str, Any]: | |
| """One pydantic property schema -> the lightweight registry property dict.""" | |
| simplified: dict[str, Any] = {} | |
| primary = _primary_type(prop) | |
| if primary is not None: | |
| simplified["type"] = primary | |
| description = prop.get("description") | |
| if description: | |
| simplified["description"] = description | |
| return simplified | |
| def input_schema_from_model(model: type[BaseModel]) -> dict[str, Any]: | |
| """Build the registry ``input_schema`` ({required, properties}) from a model. | |
| ``required`` mirrors the model's no-default fields; ``properties`` is one | |
| simplified entry per field (see `_simplify`). Property ORDER follows the | |
| model's field declaration order, so keep ``data`` (the Pattern A placeholder) | |
| first in every InputModel for readable planner prompts. | |
| """ | |
| schema = model.model_json_schema() | |
| properties = schema.get("properties", {}) | |
| return { | |
| "required": list(schema.get("required", [])), | |
| "properties": {name: _simplify(prop) for name, prop in properties.items()}, | |
| } | |
| def spec_from_model( | |
| *, | |
| name: str, | |
| category: str, | |
| output_kind: str, | |
| description: str, | |
| input_model: type[BaseModel], | |
| phase: str = "P0", | |
| ) -> ToolSpec: | |
| """A `ToolSpec` for a functional analytics tool, with `input_schema` from `input_model`. | |
| Args: | |
| name: the `analyze_*` tool name (matches the invoker dispatch key). | |
| category: dotted registry category, e.g. ``"analytics.timeseries"``. | |
| output_kind: the `ToolOutput.kind` the compute fn produces. | |
| description: the prompt-style string the Planner reads — build it with | |
| `build_description` so every tool follows the same when/when-not/output | |
| shape. | |
| input_model: the tool's pydantic InputModel; its fields are the tool args. | |
| phase: rollout phase (P0/P1/P2), default "P0" to match existing specs. | |
| """ | |
| return ToolSpec( | |
| name=name, | |
| category=category, | |
| input_schema=input_schema_from_model(input_model), | |
| output_kind=output_kind, | |
| description=description, | |
| phase=phase, | |
| ) | |
| def build_description( | |
| *, | |
| summary: str, | |
| use_when: str, | |
| dont_use_when: list[str], | |
| output: str, | |
| examples: list[str] | None = None, | |
| ) -> str: | |
| """Compose a Planner-facing description in the house format (see visualization.py). | |
| Encodes the tool team's three required parts — WHEN to use, when NOT to use, | |
| and what the OUTPUT is — around a one-line summary, matching the existing | |
| `DESCRIPTION` constants so the planner prompt reads uniformly across old and | |
| new tools. | |
| """ | |
| lines = [f"Summary: {summary}", "", f"USE WHEN {use_when}", "", "DON'T USE WHEN:"] | |
| lines += [f" - {reason}" for reason in dont_use_when] | |
| lines += ["", f"OUTPUT: {output}"] | |
| if examples: # None or [] -> omit the section | |
| lines += ["", "Example questions:"] | |
| lines += [f" - {q}" for q in examples] | |
| return "\n".join(lines) | |