Download adam/experiment_tracker.py from SyntheticMDProductions/AI_Development_Automation_Manager: direct link, hf CLI and curl.
- Browser
- Download file 16.1 kB
-
https://huggingface.co/SyntheticMDProductions/AI_Development_Automation_Manager/resolve/main/adam/experiment_tracker.py
- Command line
-
hf download hf://SyntheticMDProductions/AI_Development_Automation_Manager/adam/experiment_tracker.py
-
curl -L -o experiment_tracker.py https://huggingface.co/SyntheticMDProductions/AI_Development_Automation_Manager/resolve/main/adam/experiment_tracker.py
16.1 kB
| from __future__ import annotations | |
| import json | |
| import re | |
| import sqlite3 | |
| from dataclasses import dataclass | |
| from datetime import datetime, timezone | |
| from pathlib import Path | |
| from typing import Any | |
| from adam.models import Job, JobStatus, SystemSnapshot | |
| def _utc_now() -> str: | |
| return datetime.now(timezone.utc).isoformat() | |
| def _json(value: Any) -> str: | |
| return json.dumps(value, sort_keys=True) | |
| def _safe_json(value: str, fallback: Any) -> Any: | |
| try: | |
| parsed = json.loads(value or "") | |
| except (TypeError, ValueError, json.JSONDecodeError): | |
| return fallback | |
| return parsed if isinstance(parsed, type(fallback)) else fallback | |
| def _safe_int(value: Any, default: int = 0) -> int: | |
| try: | |
| if isinstance(value, bool): | |
| return default | |
| return int(value) | |
| except (TypeError, ValueError): | |
| return default | |
| def _safe_float(value: Any, default: float = 0.0) -> float: | |
| try: | |
| if isinstance(value, bool): | |
| return default | |
| return float(value) | |
| except (TypeError, ValueError): | |
| return default | |
| def _loss_from_logs(logs: list[str]) -> float | None: | |
| for line in reversed(logs): | |
| match = re.search(r"\bloss(?:\s*[:=]\s*|\s+)(-?\d+(?:\.\d+)?(?:e[+-]?\d+)?)", line, re.I) | |
| if match: | |
| try: | |
| return float(match.group(1)) | |
| except ValueError: | |
| return None | |
| return None | |
| def _duration_seconds(job: Job) -> int: | |
| if not job.started_at: | |
| return 0 | |
| try: | |
| start = datetime.fromisoformat(job.started_at) | |
| end = datetime.fromisoformat(job.ended_at) if job.ended_at else datetime.now(timezone.utc) | |
| return max(0, int((end - start).total_seconds())) | |
| except ValueError: | |
| return 0 | |
| def _image_count(path: str) -> int: | |
| folder = Path(path).expanduser() | |
| if not folder.is_dir(): | |
| return 0 | |
| try: | |
| return sum( | |
| 1 for item in folder.rglob("*") | |
| if item.is_file() and item.suffix.casefold() in {".png", ".jpg", ".jpeg", ".webp", ".bmp"} | |
| ) | |
| except OSError: | |
| return 0 | |
| class ExperimentRun: | |
| id: str | |
| job_id: str | |
| timestamp: str | |
| model_architecture: str | |
| model_name: str | |
| trigger_word: str | |
| base_model: str | |
| dataset_path: str | |
| dataset_name: str | |
| dataset_item_count: int | |
| epochs: int | |
| batch_size: int | |
| learning_rate: float | |
| optimizer: str | |
| scheduler: str | |
| resolution: int | |
| seed: int | |
| status: str | |
| training_time_seconds: int | |
| final_loss: float | None | |
| output_folder: str | |
| checkpoint_paths: list[str] | |
| preview_images: list[str] | |
| peak_vram_gb: float | None | |
| hardware: dict[str, Any] | |
| settings: dict[str, Any] | |
| generation_settings: dict[str, Any] | |
| notes: str = "" | |
| quality_score: int | None = None | |
| def from_row(cls, row: sqlite3.Row) -> "ExperimentRun": | |
| payload = dict(row) | |
| for key in ("checkpoint_paths", "preview_images"): | |
| payload[key] = _safe_json(payload.get(key, "[]"), []) | |
| for key in ("hardware", "settings", "generation_settings"): | |
| payload[key] = _safe_json(payload.get(key, "{}"), {}) | |
| payload["quality_score"] = ( | |
| _safe_int(payload["quality_score"]) if payload.get("quality_score") is not None else None | |
| ) | |
| return cls(**payload) | |
| class ExperimentStore: | |
| def __init__(self, root: Path) -> None: | |
| self.root = root.resolve() | |
| self.path = self.root / "data" / "experiments.sqlite3" | |
| self.path.parent.mkdir(parents=True, exist_ok=True) | |
| self._init_db() | |
| def connect(self) -> sqlite3.Connection: | |
| connection = sqlite3.connect(self.path) | |
| connection.row_factory = sqlite3.Row | |
| return connection | |
| def _init_db(self) -> None: | |
| with self.connect() as db: | |
| db.execute( | |
| """ | |
| CREATE TABLE IF NOT EXISTS experiments ( | |
| id TEXT PRIMARY KEY, | |
| job_id TEXT UNIQUE NOT NULL, | |
| timestamp TEXT NOT NULL, | |
| model_architecture TEXT NOT NULL, | |
| model_name TEXT NOT NULL, | |
| trigger_word TEXT NOT NULL DEFAULT '', | |
| base_model TEXT NOT NULL, | |
| dataset_path TEXT NOT NULL, | |
| dataset_name TEXT NOT NULL, | |
| dataset_item_count INTEGER NOT NULL, | |
| epochs INTEGER NOT NULL, | |
| batch_size INTEGER NOT NULL, | |
| learning_rate REAL NOT NULL, | |
| optimizer TEXT NOT NULL, | |
| scheduler TEXT NOT NULL, | |
| resolution INTEGER NOT NULL, | |
| seed INTEGER NOT NULL, | |
| status TEXT NOT NULL, | |
| training_time_seconds INTEGER NOT NULL, | |
| final_loss REAL, | |
| output_folder TEXT NOT NULL, | |
| checkpoint_paths TEXT NOT NULL, | |
| preview_images TEXT NOT NULL, | |
| peak_vram_gb REAL, | |
| hardware TEXT NOT NULL, | |
| settings TEXT NOT NULL, | |
| generation_settings TEXT NOT NULL, | |
| notes TEXT NOT NULL DEFAULT '', | |
| quality_score INTEGER | |
| ) | |
| """ | |
| ) | |
| self._migrate_columns(db) | |
| def _migrate_columns(db: sqlite3.Connection) -> None: | |
| existing = {row["name"] for row in db.execute("PRAGMA table_info(experiments)").fetchall()} | |
| columns = { | |
| "id": "TEXT PRIMARY KEY", | |
| "job_id": "TEXT NOT NULL DEFAULT ''", | |
| "timestamp": "TEXT NOT NULL DEFAULT ''", | |
| "model_architecture": "TEXT NOT NULL DEFAULT ''", | |
| "model_name": "TEXT NOT NULL DEFAULT ''", | |
| "trigger_word": "TEXT NOT NULL DEFAULT ''", | |
| "base_model": "TEXT NOT NULL DEFAULT ''", | |
| "dataset_path": "TEXT NOT NULL DEFAULT ''", | |
| "dataset_name": "TEXT NOT NULL DEFAULT ''", | |
| "dataset_item_count": "INTEGER NOT NULL DEFAULT 0", | |
| "epochs": "INTEGER NOT NULL DEFAULT 0", | |
| "batch_size": "INTEGER NOT NULL DEFAULT 0", | |
| "learning_rate": "REAL NOT NULL DEFAULT 0", | |
| "optimizer": "TEXT NOT NULL DEFAULT ''", | |
| "scheduler": "TEXT NOT NULL DEFAULT ''", | |
| "resolution": "INTEGER NOT NULL DEFAULT 0", | |
| "seed": "INTEGER NOT NULL DEFAULT 0", | |
| "status": "TEXT NOT NULL DEFAULT ''", | |
| "training_time_seconds": "INTEGER NOT NULL DEFAULT 0", | |
| "final_loss": "REAL", | |
| "output_folder": "TEXT NOT NULL DEFAULT ''", | |
| "checkpoint_paths": "TEXT NOT NULL DEFAULT '[]'", | |
| "preview_images": "TEXT NOT NULL DEFAULT '[]'", | |
| "peak_vram_gb": "REAL", | |
| "hardware": "TEXT NOT NULL DEFAULT '{}'", | |
| "settings": "TEXT NOT NULL DEFAULT '{}'", | |
| "generation_settings": "TEXT NOT NULL DEFAULT '{}'", | |
| "notes": "TEXT NOT NULL DEFAULT ''", | |
| "quality_score": "INTEGER", | |
| } | |
| for name, definition in columns.items(): | |
| if name not in existing and name != "id": | |
| db.execute(f"ALTER TABLE experiments ADD COLUMN {name} {definition}") | |
| def record_job(self, job: Job, snapshot: SystemSnapshot | None = None) -> ExperimentRun | None: | |
| training_steps = [step for step in job.plan.steps if step.tool_id.endswith("_trainer")] | |
| if not training_steps: | |
| return None | |
| step = training_steps[-1] | |
| args = dict(step.arguments) | |
| architecture = step.tool_id.removesuffix("_trainer") | |
| dataset_path = str(args.get("dataset_dir", "")) | |
| output_folder = str(job.output_folder or args.get("output_dir", "")) | |
| preview_images = [job.preview_path] if job.preview_path else [] | |
| checkpoints = [] | |
| if output_folder: | |
| folder = Path(output_folder) | |
| if folder.is_dir(): | |
| try: | |
| checkpoints = [ | |
| str(path) | |
| for path in sorted(folder.rglob("*")) | |
| if path.is_file() and path.suffix.casefold() in {".safetensors", ".ckpt", ".pt", ".bin"} | |
| ][-10:] | |
| discovered_previews = [ | |
| str(path) | |
| for path in sorted(folder.rglob("*")) | |
| if path.is_file() | |
| and path.suffix.casefold() in {".png", ".jpg", ".jpeg", ".webp", ".bmp"} | |
| and any(token in path.name.casefold() for token in ("preview", "sample", "epoch")) | |
| ][-12:] | |
| preview_images = list(dict.fromkeys([*preview_images, *discovered_previews])) | |
| except OSError: | |
| checkpoints = [] | |
| hardware = {} | |
| peak_vram = None | |
| if snapshot is not None: | |
| hardware = { | |
| "gpu_name": snapshot.gpu_name, | |
| "gpu_percent": snapshot.gpu_percent, | |
| "vram_used_gb": snapshot.vram_used_gb, | |
| "vram_total_gb": snapshot.vram_total_gb, | |
| "memory_used_gb": snapshot.memory_used_gb, | |
| "memory_total_gb": snapshot.memory_total_gb, | |
| "cpu_percent": snapshot.cpu_percent, | |
| "gpu_temperature": snapshot.gpu_temperature, | |
| } | |
| peak_vram = snapshot.vram_used_gb or None | |
| run = ExperimentRun( | |
| id=f"EXP-{job.id}", | |
| job_id=job.id, | |
| timestamp=job.ended_at or _utc_now(), | |
| model_architecture=architecture, | |
| model_name=str(args.get("model_name", job.plan.project_name)), | |
| trigger_word=str(args.get("trigger_word", "")), | |
| base_model=str(args.get("base_model", args.get("base_model_path", ""))), | |
| dataset_path=dataset_path, | |
| dataset_name=Path(dataset_path).name if dataset_path else "", | |
| dataset_item_count=_image_count(dataset_path), | |
| epochs=_safe_int(args.get("epochs")), | |
| batch_size=_safe_int(args.get("batch_size")), | |
| learning_rate=_safe_float(args.get("learning_rate")), | |
| optimizer=str(args.get("optimizer", "")), | |
| scheduler=str(args.get("scheduler", args.get("sampler", ""))), | |
| resolution=_safe_int(args.get("resolution")), | |
| seed=_safe_int(args.get("seed", args.get("preview_seed", 0))), | |
| status=job.status.value, | |
| training_time_seconds=_duration_seconds(job), | |
| final_loss=_loss_from_logs(job.logs), | |
| output_folder=output_folder, | |
| checkpoint_paths=checkpoints, | |
| preview_images=preview_images, | |
| peak_vram_gb=peak_vram, | |
| hardware=hardware, | |
| settings=args, | |
| generation_settings={}, | |
| ) | |
| with self.connect() as db: | |
| db.execute( | |
| """ | |
| INSERT INTO experiments ( | |
| id, job_id, timestamp, model_architecture, model_name, trigger_word, base_model, | |
| dataset_path, dataset_name, dataset_item_count, epochs, batch_size, | |
| learning_rate, optimizer, scheduler, resolution, seed, status, | |
| training_time_seconds, final_loss, output_folder, checkpoint_paths, | |
| preview_images, peak_vram_gb, hardware, settings, generation_settings | |
| ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) | |
| ON CONFLICT(job_id) DO UPDATE SET | |
| timestamp=excluded.timestamp, | |
| trigger_word=excluded.trigger_word, | |
| status=excluded.status, | |
| training_time_seconds=excluded.training_time_seconds, | |
| final_loss=excluded.final_loss, | |
| output_folder=excluded.output_folder, | |
| checkpoint_paths=excluded.checkpoint_paths, | |
| preview_images=excluded.preview_images, | |
| peak_vram_gb=excluded.peak_vram_gb, | |
| hardware=excluded.hardware, | |
| settings=excluded.settings | |
| """, | |
| ( | |
| run.id, run.job_id, run.timestamp, run.model_architecture, run.model_name, | |
| run.trigger_word, run.base_model, run.dataset_path, run.dataset_name, run.dataset_item_count, | |
| run.epochs, run.batch_size, run.learning_rate, run.optimizer, run.scheduler, | |
| run.resolution, run.seed, run.status, run.training_time_seconds, run.final_loss, | |
| run.output_folder, _json(run.checkpoint_paths), _json(run.preview_images), | |
| run.peak_vram_gb, _json(run.hardware), _json(run.settings), | |
| _json(run.generation_settings), | |
| ), | |
| ) | |
| return run | |
| def list_runs(self, search: str = "", architecture: str = "", dataset: str = "", limit: int = 200) -> list[ExperimentRun]: | |
| clauses = [] | |
| params: list[Any] = [] | |
| if search: | |
| clauses.append("(id LIKE ? OR job_id LIKE ? OR model_name LIKE ? OR dataset_name LIKE ? OR dataset_path LIKE ? OR output_folder LIKE ? OR notes LIKE ? OR status LIKE ?)") | |
| term = f"%{search}%" | |
| params.extend([term, term, term, term, term, term, term, term]) | |
| if architecture: | |
| clauses.append("model_architecture = ?") | |
| params.append(architecture) | |
| if dataset: | |
| clauses.append("dataset_name LIKE ?") | |
| params.append(f"%{dataset}%") | |
| where = " WHERE " + " AND ".join(clauses) if clauses else "" | |
| with self.connect() as db: | |
| rows = db.execute( | |
| "SELECT * FROM experiments" + where + " ORDER BY timestamp DESC LIMIT ?", | |
| [*params, int(limit)], | |
| ).fetchall() | |
| return [ExperimentRun.from_row(row) for row in rows] | |
| def get(self, run_id: str) -> ExperimentRun | None: | |
| with self.connect() as db: | |
| row = db.execute("SELECT * FROM experiments WHERE id = ?", (run_id,)).fetchone() | |
| return ExperimentRun.from_row(row) if row else None | |
| def update_notes(self, run_id: str, notes: str, quality_score: int | None) -> None: | |
| with self.connect() as db: | |
| db.execute( | |
| "UPDATE experiments SET notes = ?, quality_score = ? WHERE id = ?", | |
| (notes, quality_score, run_id), | |
| ) | |
| def compare(self, run_ids: list[str]) -> list[dict[str, Any]]: | |
| runs = [run for run_id in run_ids if (run := self.get(run_id)) is not None] | |
| fields = [ | |
| "model_architecture", "epochs", "final_loss", "training_time_seconds", | |
| "resolution", "batch_size", "learning_rate", "scheduler", | |
| "peak_vram_gb", "dataset_name", "quality_score", | |
| ] | |
| rows = [] | |
| for field in fields: | |
| values = {run.id: getattr(run, field) for run in runs} | |
| comparable = {str(value) for value in values.values()} | |
| rows.append({"field": field, "changed": len(comparable) > 1, **values}) | |
| return rows | |
| def clone_request(self, run_id: str) -> str: | |
| run = self.get(run_id) | |
| if run is None: | |
| return "" | |
| options = { | |
| key: value | |
| for key, value in run.settings.items() | |
| if key not in {"dataset_dir", "model_name", "epochs", "output_dir", "resume_from"} | |
| } | |
| return ( | |
| f"From the {run.dataset_name or run.dataset_path} dataset, train a " | |
| f"{run.model_architecture.upper()} model for {run.epochs} epochs. " | |
| f"Name the model {run.model_name} Clone. " | |
| "[ADAM_TRAINING_OPTIONS:" + json.dumps(options, sort_keys=True) + "] " | |
| "[ADAM_TRAINER:" + run.model_architecture + "]" | |
| ) | |