Instructions to use Zipeng365/WISP with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Scikit-learn
How to use Zipeng365/WISP with Scikit-learn:
from huggingface_hub import hf_hub_download import joblib model = joblib.load( hf_hub_download("Zipeng365/WISP", "sklearn_model.joblib") ) # only load pickle files from sources you trust # read more about it here https://skops.readthedocs.io/en/stable/persistence.html - Notebooks
- Google Colab
- Kaggle
Download src/wisp/search/controller.py from Zipeng365/WISP: direct link, hf CLI and curl.
- Browser
- Download file 17.2 kB
-
https://huggingface.co/Zipeng365/WISP/resolve/main/src/wisp/search/controller.py
- Command line
-
hf download hf://Zipeng365/WISP/src/wisp/search/controller.py
-
curl -L -o controller.py https://huggingface.co/Zipeng365/WISP/resolve/main/src/wisp/search/controller.py
17.2 kB
| # Paper-scoped implementation; see SOURCE_PROVENANCE.json. | |
| from __future__ import annotations | |
| import hashlib | |
| import json | |
| import os | |
| import pickle | |
| import time | |
| from concurrent.futures import ThreadPoolExecutor | |
| from dataclasses import dataclass | |
| from pathlib import Path | |
| from typing import Any | |
| import numpy as np | |
| from wisp.core.data import HARData, concatenate_har | |
| from wisp.core.metrics import classification_report_dict | |
| from wisp.cpu.algorithms import build_cpu_estimator, estimate_pickle_size_mb | |
| from wisp.search.cascade import evaluate_candidate_cascade | |
| from wisp.search.dataset_fingerprint import DatasetFingerprint, compute_dataset_fingerprint | |
| from wisp.search.grammar import CandidateSpec, initial_candidates, mutate_candidate, normalize_spec | |
| from wisp.search.operator_program import TemporalProgram | |
| from wisp.search.posterior import MotifPosterior, island_key, sample_parent | |
| from wisp.search.profiles import SearchProfile, get_profile | |
| from wisp.utils.jsonl import append_jsonl, read_jsonl | |
| def _positive_int_env(name: str, default: int) -> int: | |
| try: | |
| return max(int(os.environ.get(name, default)), 1) | |
| except (TypeError, ValueError): | |
| return max(int(default), 1) | |
| def _stable_id(*, seed: int, generation: int, index: int, spec: CandidateSpec) -> str: | |
| payload = json.dumps(spec.to_dict(), sort_keys=True, separators=(',', ':'), default=str) | |
| digest = hashlib.sha1(f'{seed}:{generation}:{index}:{payload}'.encode()).hexdigest()[:12] | |
| return f'g{generation:03d}c{index:04d}_{digest}' | |
| def _spec_key(spec: CandidateSpec) -> str: | |
| return json.dumps(normalize_spec(spec).to_dict(), sort_keys=True, separators=(',', ':'), default=str) | |
| class WISPSearchController: | |
| train: HARData | |
| valid: HARData | |
| output: str | Path | |
| profile: str | SearchProfile = 'wisp_evolution' | |
| test: HARData | None = None | |
| population: int = 24 | |
| generations: int = 20 | |
| seed: int = 42 | |
| cost_weight: float = 0.01 | |
| robustness_weight: float = 0.1 | |
| candidate_workers: int | None = None | |
| n_jobs_per_candidate: int | None = None | |
| def __post_init__(self) -> None: | |
| if isinstance(self.profile, str): | |
| self.profile = get_profile(self.profile) | |
| if self.train.n_classes != self.valid.n_classes: | |
| raise ValueError('train and validation must share the same global label vocabulary') | |
| if self.test is not None and self.test.n_classes != self.train.n_classes: | |
| raise ValueError('test must share the same global label vocabulary') | |
| def _worker_count(self) -> int: | |
| if self.candidate_workers is not None: | |
| return max(int(self.candidate_workers), 1) | |
| return _positive_int_env('WISP_CANDIDATE_WORKERS', 1) | |
| def _per_candidate_jobs(self) -> int: | |
| if self.n_jobs_per_candidate is not None: | |
| return max(int(self.n_jobs_per_candidate), 1) | |
| explicit = os.environ.get('WISP_N_JOBS_PER_CANDIDATE') | |
| if explicit is not None: | |
| return _positive_int_env('WISP_N_JOBS_PER_CANDIDATE', 1) | |
| total = _positive_int_env('WISP_N_JOBS', 1) | |
| return max(total // self._worker_count(), 1) | |
| def _runtime_spec(self, spec: CandidateSpec) -> CandidateSpec: | |
| spec = normalize_spec(spec) | |
| if spec.route != 'cpu': | |
| raise ValueError('WISPSearchController is CPU-only; GPU search is not released') | |
| params = dict(spec.params) | |
| params.setdefault('n_jobs', self._per_candidate_jobs()) | |
| return CandidateSpec(name=spec.name, params=params, operators=list(spec.operators), route='cpu', parent_id=spec.parent_id, mutation=spec.mutation) | |
| def _constrain_spec(self, spec: CandidateSpec) -> CandidateSpec | None: | |
| spec = self._runtime_spec(spec) | |
| allowed = self.profile.allowed_cpu_families | |
| if allowed is not None and spec.name not in set(allowed): | |
| return None | |
| if self.profile.force_no_hmm: | |
| params = dict(spec.params) | |
| params['use_hmm'] = False | |
| operators = [op for op in spec.operators if op != 'hmm'] | |
| if 'argmax' not in operators: | |
| operators.append('argmax') | |
| spec = CandidateSpec(name=spec.name, params=params, operators=operators, route='cpu', parent_id=spec.parent_id, mutation=spec.mutation) | |
| if spec.mutation in set(self.profile.disabled_mutations): | |
| return None | |
| return spec | |
| def _evaluate_one(self, spec: CandidateSpec) -> tuple[str, dict[str, Any]]: | |
| try: | |
| cascade = evaluate_candidate_cascade(spec, self.train, self.valid, cost_weight=self.cost_weight, robustness_weight=self.robustness_weight if self.profile.use_stress else 0.0, max_stress_transforms=4 if self.profile.use_stress else 0) | |
| result = cascade.to_dict() | |
| predictive = float(result.get('cascade_score', result.get('score', -1000000000.0))) | |
| result['predictive_utility'] = predictive | |
| return ('ok' if cascade.accepted else 'rejected', result) | |
| except Exception as exc: | |
| return ('failed', {'predictive_utility': -1000000000.0, 'breeding_score': -1000000000.0, 'error': repr(exc), 'candidate': spec.to_dict()}) | |
| def _evaluate_many(self, specs: list[CandidateSpec]) -> list[tuple[str, dict[str, Any]]]: | |
| if not specs: | |
| return [] | |
| workers = min(self._worker_count(), len(specs)) | |
| if workers <= 1: | |
| return [self._evaluate_one(spec) for spec in specs] | |
| with ThreadPoolExecutor(max_workers=workers, thread_name_prefix='wisp') as pool: | |
| return list(pool.map(self._evaluate_one, specs)) | |
| def _initial_specs(self, posterior: MotifPosterior, fingerprint: DatasetFingerprint | None, family_transfer_bias: dict[str, float], rng: np.random.Generator, count: int) -> list[CandidateSpec]: | |
| specs = posterior.rank_initial_candidates(fingerprint=fingerprint, seed=self.seed, route='cpu', include_v2=self.profile.include_v2_families, motif_mode=self.profile.motif_mode, family_transfer_bias=family_transfer_bias) | |
| specs = [constrained for spec in specs if (constrained := self._constrain_spec(spec)) is not None] | |
| if not specs: | |
| raise RuntimeError(f'Profile {self.profile.name} removed every initial CPU family') | |
| base = list(specs) | |
| i = 0 | |
| while len(specs) < count: | |
| parent = base[i % len(base)] | |
| clone = CandidateSpec.from_dict(parent.to_dict()) | |
| clone.params = dict(clone.params) | |
| clone.params['seed'] = int(clone.params.get('seed', self.seed)) + 1009 * (i + 1) | |
| clone.mutation = 'initial_seed_variant' | |
| specs.append(clone) | |
| i += 1 | |
| return [spec for spec in specs[:count]] | |
| def _random_spec(self, rng: np.random.Generator, fingerprint: DatasetFingerprint | None) -> CandidateSpec: | |
| bases = initial_candidates(seed=int(rng.integers(0, 2 ** 31 - 1)), include_v2=self.profile.include_v2_families, route='cpu') | |
| spec = CandidateSpec.from_dict(bases[int(rng.integers(0, len(bases)))].to_dict()) | |
| depth = int(rng.integers(1, 5)) | |
| for _ in range(depth): | |
| spec = mutate_candidate(spec, rng=rng, mutation_bias=None, motif_weights=None if fingerprint is None else fingerprint.motif_weights()) | |
| spec.parent_id = None | |
| spec.mutation = f'random_depth_{depth}' | |
| constrained = self._constrain_spec(spec) | |
| if constrained is None: | |
| return self._random_spec(rng, fingerprint) | |
| return constrained | |
| def _record(self, *, db_path: Path, candidate_id: str, generation: int, spec: CandidateSpec, status: str, result: dict[str, Any], fingerprint: DatasetFingerprint | None, parent_id: str | None) -> dict[str, Any]: | |
| row = {'id': candidate_id, 'generation': int(generation), 'parent_id': parent_id, 'island': island_key(spec), 'status': status, 'profile': self.profile.to_dict(), 'dataset_fingerprint': None if fingerprint is None else fingerprint.to_dict(), 'motif_weights': None if fingerprint is None else fingerprint.motif_weights(), 'operator_certificate': TemporalProgram.from_candidate(spec).certificate(), 'result': result, 'spec': spec.to_dict()} | |
| append_jsonl(db_path, row) | |
| return row | |
| def _final_refit(self, best: dict[str, Any], out_dir: Path) -> dict[str, Any] | None: | |
| if self.test is None: | |
| return None | |
| spec = CandidateSpec.from_dict(best['spec']) | |
| model = build_cpu_estimator(spec.to_dict()) | |
| refit = concatenate_har(self.train, self.valid, source='train+validation') | |
| t0 = time.perf_counter() | |
| model.fit(refit.X, refit.y, groups=refit.subject, time_index=refit.time_index, n_classes=refit.n_classes) | |
| fit_s = time.perf_counter() - t0 | |
| t1 = time.perf_counter() | |
| pred = model.predict(self.test.X, groups=self.test.subject, time_index=self.test.time_index) | |
| pred_s = time.perf_counter() - t1 | |
| metrics = classification_report_dict(self.test.y, pred, fit_seconds=fit_s, predict_seconds=pred_s, model_size_mb=estimate_pickle_size_mb(model), label_names=self.test.label_names) | |
| np.savez_compressed(out_dir / 'locked_test_predictions.npz', y_true=self.test.y, y_pred=pred, subject=self.test.subject, time_index=self.test.time_index) | |
| with (out_dir / 'selected_model.pkl').open('wb') as f: | |
| pickle.dump(model, f) | |
| (out_dir / 'locked_test_metrics.json').write_text(json.dumps(metrics, indent=2, ensure_ascii=False, default=str), encoding='utf-8') | |
| return metrics | |
| def run(self) -> dict[str, Any]: | |
| out_dir = Path(self.output) | |
| out_dir.mkdir(parents=True, exist_ok=True) | |
| db_path = out_dir / 'program_database.jsonl' | |
| summary_path = out_dir / 'summary.json' | |
| config = {'profile': self.profile.to_dict(), 'population': int(self.population), 'generations': int(self.generations), 'seed': int(self.seed), 'cost_weight': float(self.cost_weight), 'robustness_weight': float(self.robustness_weight), 'candidate_workers': self._worker_count(), 'n_jobs_per_candidate': self._per_candidate_jobs()} | |
| if summary_path.exists(): | |
| old = json.loads(summary_path.read_text(encoding='utf-8')) | |
| if old.get('status') == 'completed' and old.get('search_config') == config: | |
| return old | |
| if db_path.exists() and db_path.stat().st_size: | |
| raise RuntimeError(f'Refusing to append to non-empty search database {db_path}') | |
| rng = np.random.default_rng(self.seed) | |
| fingerprint = compute_dataset_fingerprint(self.train, seed=self.seed) if self.profile.use_fingerprint else None | |
| family_transfer, mutation_transfer = ({}, {}) | |
| posterior = MotifPosterior() | |
| archive: list[CandidateSpec] = [] | |
| population: list[tuple[str, CandidateSpec, dict[str, Any]]] = [] | |
| rows: list[dict[str, Any]] = [] | |
| seen: set[str] = set() | |
| budget = int(self.profile.max_evaluations) | |
| initial_count = min(9, budget) | |
| seeds = self._initial_specs(posterior, fingerprint, family_transfer, rng, initial_count) | |
| seed_results = self._evaluate_many(seeds) | |
| for i, (spec, (status, result)) in enumerate(zip(seeds, seed_results)): | |
| novelty = 0.0 | |
| result['novelty_bonus'] = novelty | |
| result['breeding_score'] = float(result.get('predictive_utility', -1000000000.0)) + novelty | |
| cid = _stable_id(seed=self.seed, generation=0, index=i, spec=spec) | |
| row = self._record(db_path=db_path, candidate_id=cid, generation=0, spec=spec, status=status, result=result, fingerprint=fingerprint, parent_id=None) | |
| rows.append(row) | |
| seen.add(_spec_key(spec)) | |
| archive.append(spec) | |
| if status == 'ok': | |
| population.append((cid, spec, result)) | |
| if self.profile.use_posterior: | |
| posterior.update(spec, result, score_key='predictive_utility') | |
| if not population: | |
| raise RuntimeError('No accepted initial candidate') | |
| evaluated = len(seeds) | |
| generation = 0 | |
| while evaluated < budget and generation < int(self.generations): | |
| generation += 1 | |
| remaining = budget - evaluated | |
| proposal_count = min(max(self.population - max(2, self.population // 3), 1), remaining) | |
| proposals: list[tuple[str, str | None, CandidateSpec]] = [] | |
| attempts = 0 | |
| mutation_bias = posterior.mutation_bias(fingerprint, motif_mode=self.profile.motif_mode, transfer_bias=mutation_transfer) if self.profile.use_posterior else mutation_transfer if self.profile.transfer_prior else {} | |
| for disabled in self.profile.disabled_mutations: | |
| mutation_bias.pop(disabled, None) | |
| while len(proposals) < proposal_count: | |
| attempts += 1 | |
| if attempts > proposal_count * 100: | |
| raise RuntimeError('Unable to generate enough unique candidate specifications') | |
| if self.profile.search_strategy == 'random': | |
| parent_id = None | |
| child = self._random_spec(rng, fingerprint) | |
| else: | |
| parent_id, parent, parent_result = sample_parent(population, rng, use_islands=self.profile.use_islands, score_key='breeding_score') | |
| child = None | |
| if child is None: | |
| child = mutate_candidate(parent, rng=rng, parent_id=parent_id, mutation_bias=mutation_bias, motif_weights=None if fingerprint is None else fingerprint.motif_weights()) | |
| child = self._constrain_spec(child) | |
| if child is None: | |
| continue | |
| key = _spec_key(child) | |
| if key in seen: | |
| continue | |
| seen.add(key) | |
| index = evaluated + len(proposals) | |
| proposals.append((_stable_id(seed=self.seed, generation=generation, index=index, spec=child), parent_id, child)) | |
| evaluated_results = self._evaluate_many([spec for _, _, spec in proposals]) | |
| new_items: list[tuple[str, CandidateSpec, dict[str, Any]]] = [] | |
| for (cid, parent_id, spec), (status, result) in zip(proposals, evaluated_results): | |
| novelty = posterior.novelty_bonus(spec, archive) if self.profile.use_novelty and status == 'ok' else 0.0 | |
| result['novelty_bonus'] = float(novelty) | |
| result['breeding_score'] = float(result.get('predictive_utility', -1000000000.0)) + float(novelty) | |
| row = self._record(db_path=db_path, candidate_id=cid, generation=generation, spec=spec, status=status, result=result, fingerprint=fingerprint, parent_id=parent_id) | |
| rows.append(row) | |
| archive.append(spec) | |
| if status == 'ok': | |
| new_items.append((cid, spec, result)) | |
| if self.profile.use_posterior: | |
| posterior.update(spec, result, score_key='predictive_utility') | |
| evaluated += len(proposals) | |
| if self.profile.search_strategy == 'evolution': | |
| combined = population + new_items | |
| combined.sort(key=lambda item: float(item[2].get('breeding_score', -1000000000.0)), reverse=True) | |
| population = combined[:int(self.population)] | |
| else: | |
| population = (population + new_items)[-int(self.population):] | |
| if not population: | |
| raise RuntimeError(f'No accepted candidate after generation {generation}') | |
| ok_rows = [row for row in rows if row['status'] == 'ok'] | |
| if not ok_rows: | |
| raise RuntimeError('Search completed without an accepted candidate') | |
| final_key = self.profile.final_selection | |
| best = max(ok_rows, key=lambda row: float(row['result'].get(final_key, -1000000000.0))) | |
| best_breeding = max(ok_rows, key=lambda row: float(row['result'].get('breeding_score', -1000000000.0))) | |
| best_predictive = max(ok_rows, key=lambda row: float(row['result'].get('predictive_utility', -1000000000.0))) | |
| (out_dir / 'selected_candidate.json').write_text(json.dumps(best, indent=2, ensure_ascii=False, default=str), encoding='utf-8') | |
| test_metrics = self._final_refit(best, out_dir) | |
| summary = {'status': 'completed', 'version': 'wisp_search_controller', 'profile': self.profile.name, 'n_rows': len(rows), 'n_ok': len(ok_rows), 'n_evaluated': evaluated, 'best': best, 'best_predictive': best_predictive, 'best_breeding': best_breeding, 'locked_test_metrics': test_metrics, 'search_config': config, 'dataset_fingerprint': None if fingerprint is None else {**fingerprint.to_dict(), 'motifs': fingerprint.motifs(), 'motif_weights': fingerprint.motif_weights()}, 'posterior_summary': posterior.summary(), 'cross_task_family_bias': family_transfer, 'cross_task_mutation_bias': mutation_transfer} | |
| summary_path.write_text(json.dumps(summary, indent=2, ensure_ascii=False, default=str), encoding='utf-8') | |
| return summary | |
| __all__ = ['WISPSearchController'] | |