MHRN-Space / src /playground /closed_loop.py
ThomasHeisig's picture
Sync GitHub main af692537c27175c8410bf57a747be0caf23718e2 (part 8)
9d98328 verified
Raw History Blame Contribute Delete
50.8 kB
"""Closed-loop environment helpers for the non-canonical Playground.
The implementation is deliberately bounded and deterministic. It creates an
actual action -> consequence -> reward loop for Playground experiments, but it
does not make scientific, biological, cognitive, or generalization claims.
"""
from __future__ import annotations
import math
import random
from collections.abc import Mapping, Sequence
from typing import TYPE_CHECKING, cast
if TYPE_CHECKING:
from .models import PlaygroundConfig
CLOSED_LOOP_PRESETS: dict[str, dict[str, object]] = {
"pan_full_balanced": {
"label": "PAN · Vollprofil (Standard)",
"description": "Ausgewogener CPU-Startpunkt: 256 Neuronen, 2000 Ticks, PAN/STP/STDP, "
"Wachstum, Neural I/O und Strichmann. Getrennte Signalkanäle; kein "
"Nachweis global optimaler Werte.",
"hypothesis": "Explorative gemeinsame Ausführung der implementierten PAN-Mechanismen; "
"Güte und Laufzeit müssen je Aufgabe gemessen werden.",
"expected_success": None,
"required_features": [
"pan",
"action_loop",
"reward_modulated_stdp",
"sandbox",
"neural_io",
"growth",
],
"settings": {
"name": "pan_full_balanced",
"neuron_backend": "cpu",
"neuron_model": "pan_adex_5d",
"synapse_model": "pan_stp_stdp",
"plasticity_rule": "structural",
"topology": "mhrn_5d",
"stimulus": "oscillatory",
"readout": "population_vector",
"n_neurons": 256,
"edge_budget": 2048,
"ticks": 2000,
"seed": 12345,
"dimensions": 5,
"dt_ms": 1.0,
"weight": 4.0,
"weight_decay": 0.001,
"weight_max_clamp": 20.0,
"delay_ticks": 2,
"stimulus_current": 8.0,
"stimulus_rate_hz": 20.0,
"radius": 0.35,
"k_neighbors": 16,
"rewiring_probability": 0.15,
"modules": 2,
"ensemble_runs": 1,
"persist": True,
"pan_enabled": True,
"pan_dimensions": 5,
"pan_closed_loop": True,
"pan_feedback_gain": 1.5,
"pan_health_decay": 0.001,
"pan_apoptosis_threshold": 0.1,
"pan_aging_threshold": 0.3,
"pan_bias_current": 10.0,
"clock_mode": "dual",
"clock_base_hz": 100.0,
"clock_event_batch_ms": 10.0,
"execution_mode": "HYBRID_AUTO",
"execution_initial_mode": "EVENT_ONLY",
"execution_theta_high": 0.3,
"execution_theta_low": 0.05,
"execution_hysteresis": 0.02,
"execution_min_dwell": 100,
"execution_activity_window": 100,
"execution_transition_mode": "clean",
"execution_sync_on_switch": True,
"execution_log_transitions": True,
"execution_log_state_hash": True,
"growth_enabled": True,
"growth_neurogenesis": True,
"growth_synaptogenesis": True,
"growth_path_formation": True,
"growth_pruning": True,
"growth_activity_threshold": 0.03,
"growth_coactivation_threshold": 1,
"growth_information_threshold": 0.03,
"growth_prune_threshold": 0.05,
"growth_max_synapses_per_neuron": 32,
"growth_max_new_synapses_per_barrier": 8,
"cuda_budget_mb": 2048,
"offload_enabled": False,
"offload_snapshot_interval": 1000,
"hardware_profile_name": "reference_cpu",
"thalamic_gating_enabled": True,
"thalamic_relay_threshold": 0.0,
"thalamic_attention_gain": 1.15,
"thalamic_inhibition_gain": 0.5,
"cortical_layers_enabled": True,
"cortical_layer_count": 6,
"cortical_plasticity": True,
"cortical_learning_rate": 0.01,
"behavior_learning_enabled": True,
"behavior_action_count": 4,
"behavior_learning_rate": 0.2,
"behavior_epsilon": 0.2,
"behavior_target_action": 0,
"behavior_target_mode": "cycle",
"behavior_min_activity": 0.01,
"behavior_episode_ticks": 64,
"behavior_bias_current": 20.0,
"geometry_lambda_a": 0.5,
"geometry_lambda_b": 0.5,
"geometry_sigma": 0.1,
"geometry_p0": 0.3,
"geometry_mode": "mixed_additive",
"geometry_delay_velocity": 0.25,
"neural_io_enabled": True,
"neural_io_input_channels": 8,
"neural_io_output_channels": 16,
"neural_io_input_codec": "population_latency_v1",
"neural_io_output_decoder": "population_rate_v1",
"neural_io_input_payload": 0.5,
"neural_io_window_ticks": 16,
"neural_io_input_current": 25.0,
"neural_io_input_role": "GATEWAY_AFFERENT",
"neural_io_output_role": "GATEWAY_EFFERENT",
"neural_io_phase": "QUERY",
"neural_io_correlation_id": "auto",
"neural_io_modality": "digital",
"neural_io_source_id": "playground.input",
"input_topology": "channel_partitioned",
"input_channels": 8,
"input_channel_map": (),
"input_amplitude_per_channel": [8.0, 8.0, 8.0, 8.0, 4.0, 4.0, 4.0, 4.0],
"input_frequency_per_channel": [
20.0,
22.0,
24.0,
26.0,
30.0,
32.0,
34.0,
36.0,
],
"input_phase_per_channel": [0.0, 0.25, 0.5, 0.75, 0.0, 0.25, 0.5, 0.75],
"input_noise_sigma": 0.1,
"target_cue_channel": 0,
"reward_cue_channel": 5,
"action_feedback_channel": 6,
"posture_score_channel": 2,
"reward_event_channel": 3,
"posture_current_scale": 25.0,
"reward_event_scale": 25.0,
"posture_reward_enabled": True,
"posture_weight_upright": 0.4,
"posture_weight_height": 0.3,
"posture_weight_stability": 0.2,
"posture_weight_symmetry": 0.1,
"posture_target_height": 1.0,
"posture_tilt_max": 1.0,
"posture_velocity_max": 5.0,
"trigger_good_score": 0.85,
"trigger_good_duration": 10,
"trigger_good_reward": 1.0,
"trigger_warning_score": 0.4,
"trigger_warning_reward": -0.3,
"trigger_falling_rate": -0.05,
"trigger_falling_reward": -1.0,
"trigger_collapse_score": 0.1,
"trigger_collapse_reward": -5.0,
"trigger_recovery_bonus": 2.0,
"reward_continuous_alpha": 0.1,
"episode_termination_enabled": True,
"episode_max_ticks": 256,
"episode_reset_on_collapse": True,
"pan_feedback_delay": 0,
"pan_feedback_source": "population",
"pan_feedback_target": "all",
"pan_feedback_nonlinearity": "tanh",
"pan_feedback_threshold": 0.0,
"pan_feedback_saturation": 20.0,
"action_loop_enabled": True,
"action_loop_delay": 1,
"action_persistence": 1,
"action_to_input_map": "spatial",
"action_space_size": 4,
"action_coupling_strength": 2.0,
"action_noise": 0.05,
"freeze_actions": False,
"frozen_action_sequence": (),
"reward_signal_enabled": True,
"reward_magnitude": 5.0,
"reward_delay_ticks": 0,
"reward_shaping": "sparse",
"reward_baseline": 0.0,
"reward_decay": 0.0,
"reward_channel": 1,
"freeze_rewards": False,
"frozen_reward_sequence": (),
"parity_reference_source": "CPU_PYTHON_PLAYGROUND",
"parity_reference_commit": "",
"target_encoding": "one_hot",
"target_persistence": 16,
"target_cue_current": 30.0,
"target_shuffle": True,
"target_predictability": "deterministic",
"credit_window": 64,
"eligibility_trace_tau": 200.0,
"credit_assignment": "reward_modulated_stdp",
"td_lambda": 0.9,
"gamma_discount": 0.95,
"neuron_threshold_variance": 0.3,
"neuron_tau_m_variance": 0.2,
"inhibitory_fraction": 0.25,
"gaba_strength": 6.0,
"e_i_ratio": 3.0,
"delay_distribution": "lognormal",
"delay_mean_ticks": 3.0,
"refractory_variance": 0.3,
"adaptation_strength": 0.0,
"adaptation_tau": 200.0,
"oscillation_enabled": True,
"oscillation_frequency": 8.0,
"geometry_input_coupling": True,
"geometry_input_sigma": 0.2,
"sandbox_enabled": True,
"sandbox_physics": "stick_figure",
"sandbox_action_coupling": "direct",
"sandbox_sensor_noise": 0.05,
},
},
"open_loop_baseline": {
"description": "Kein Aktions-Loop, kein Ziel- oder Reward-Signal.",
"hypothesis": "Kontrolle: Verhalten bleibt ohne kausalen Loop am Zufallsniveau.",
"expected_success": 0.25,
"required_features": [],
"settings": {
"pan_feedback_gain": 0.05,
"action_loop_enabled": False,
"reward_signal_enabled": False,
"target_encoding": "none",
"input_topology": "uniform",
"inhibitory_fraction": 0.0,
"neuron_threshold_variance": 0.0,
"neuron_tau_m_variance": 0.0,
},
},
"strong_feedback_only": {
"description": "Starkes PAN-Feedback ohne Aktions- oder Reward-Loop.",
"hypothesis": "Prueft, ob starkes Feedback allein die Dynamik differenziert.",
"expected_success": None,
"required_features": ["pan_feedback"],
"settings": {
"pan_enabled": True,
"pan_feedback_gain": 2.0,
"pan_feedback_nonlinearity": "tanh",
"pan_feedback_saturation": 20.0,
"action_loop_enabled": False,
"reward_signal_enabled": False,
"target_encoding": "none",
"input_topology": "uniform",
"inhibitory_fraction": 0.0,
},
},
"differentiated_input_only": {
"description": "Acht getrennte Eingangskanaele ohne Aktions-Loop.",
"hypothesis": "Prueft Input-Differenzierung ohne geschlossene Kausalitaetskette.",
"expected_success": None,
"required_features": ["differentiated_input"],
"settings": {
"input_topology": "channel_partitioned",
"input_channels": 8,
"input_amplitude_per_channel": [8, 8, 8, 8, 4, 4, 4, 4],
"input_frequency_per_channel": [20, 22, 24, 26, 30, 32, 34, 36],
"input_phase_per_channel": [0, 0.25, 0.5, 0.75, 0, 0.25, 0.5, 0.75],
"pan_feedback_gain": 0.05,
"action_loop_enabled": False,
"reward_signal_enabled": False,
"target_encoding": "none",
"inhibitory_fraction": 0.0,
},
},
"minimal_closed_loop": {
"description": "Differenzierter Input + Ziel-Cue + Aktion + Reward.",
"hypothesis": "Prueft, ob ein minimal geschlossener Loop Lernen ueber Zufall erlaubt.",
"expected_success": 0.5,
"required_features": ["action_loop", "target_cue", "reward_channel"],
"settings": {
"neuron_model": "pan_adex_5d",
"pan_enabled": True,
"behavior_learning_enabled": True,
"input_topology": "channel_partitioned",
"input_channels": 8,
"target_encoding": "one_hot",
"target_cue_channel": 0,
"target_cue_current": 30.0,
"target_persistence": 16,
"action_loop_enabled": True,
"action_space_size": 4,
"action_to_input_map": "auto",
"action_coupling_strength": 2.0,
"pan_feedback_gain": 1.5,
"pan_feedback_nonlinearity": "tanh",
"reward_signal_enabled": True,
"reward_magnitude": 5.0,
"reward_channel": 1,
"inhibitory_fraction": 0.2,
"gaba_strength": 4.0,
"neuron_threshold_variance": 0.15,
"neuron_tau_m_variance": 0.1,
},
},
"credit_assignment": {
"description": "Minimaler Loop plus reward-modulated eligibility traces.",
"hypothesis": "Prueft, ob zeitliche Kreditzuweisung die Policy-Anpassung stabilisiert.",
"expected_success": None,
"required_features": ["action_loop", "reward_modulated_stdp"],
"settings": {
"closed_loop_preset": "minimal_closed_loop",
"credit_assignment": "reward_modulated_stdp",
"credit_window": 64,
"eligibility_trace_tau": 200.0,
"td_lambda": 0.9,
"gamma_discount": 0.95,
},
},
"heterogeneous_network": {
"description": "Minimaler Loop plus E/I- und Neuronen-Heterogenitaet.",
"hypothesis": "Prueft, ob Heterogenitaet Synchronie reduziert und Aktivitaet differenziert.",
"expected_success": None,
"required_features": ["heterogeneity"],
"settings": {
"closed_loop_preset": "minimal_closed_loop",
"inhibitory_fraction": 0.25,
"gaba_strength": 6.0,
"e_i_ratio": 3.0,
"neuron_threshold_variance": 0.3,
"neuron_tau_m_variance": 0.2,
"refractory_variance": 0.3,
"delay_distribution": "lognormal",
"delay_mean_ticks": 3.0,
"oscillation_enabled": True,
"oscillation_frequency": 8.0,
},
},
"spatial_embodiment": {
"description": "Heterogener Loop mit raeumlich gekoppeltem Input.",
"hypothesis": "Prueft, ob Geometrie-Input-Kopplung messbare Embodiment-Effekte erzeugt.",
"expected_success": None,
"required_features": ["spatial_input", "action_loop"],
"settings": {
"closed_loop_preset": "heterogeneous_network",
"input_topology": "spatial_gradient",
"geometry_input_coupling": True,
"geometry_input_sigma": 0.2,
"action_to_input_map": "spatial",
},
},
"full_embodiment": {
"description": "Raeumlicher Closed Loop plus Stick-Figure-Sandbox.",
"hypothesis": "Explorativer Gesamtsandkasten fuer sensorimotorische Kopplung.",
"expected_success": None,
"required_features": ["spatial_input", "action_loop", "sandbox"],
"settings": {
"closed_loop_preset": "spatial_embodiment",
"sandbox_enabled": True,
"sandbox_physics": "stick_figure",
"sandbox_action_coupling": "direct",
"sandbox_sensor_noise": 0.05,
},
},
}
CLOSED_LOOP_PRESETS.update(
{
"baseline_open_loop": {
"description": "Kontrolle ohne Aktions- und Reward-Loop.",
"hypothesis": "Ohne Closed Loop bleibt Verhalten am Zufallsniveau.",
"expected_success": 0.25,
"required_features": ["control"],
"settings": {
"action_loop_enabled": False,
"reward_signal_enabled": False,
"target_encoding": "none",
"pan_feedback_gain": 0.05,
"inhibitory_fraction": 0.0,
"input_topology": "uniform",
"plasticity_rule": "structural",
"weight": 4,
"ticks": 2000,
},
},
"baseline_heterogeneous": {
"description": "E/I-Balance und Neuronenheterogenität ohne Closed Loop.",
"hypothesis": "Heterogenität reduziert Synchronie, erzeugt aber allein kein Lernen.",
"expected_success": 0.25,
"required_features": ["heterogeneity"],
"settings": {
"closed_loop_preset": "baseline_open_loop",
"inhibitory_fraction": 0.25,
"gaba_strength": 6,
"e_i_ratio": 3,
"neuron_threshold_variance": 0.3,
"neuron_tau_m_variance": 0.2,
"delay_distribution": "lognormal",
"delay_mean_ticks": 3,
},
},
"fix_weight_explosion": {
"description": "Gewichtszerfall und hartes Maximum gegen Gewichts-Explosion.",
"hypothesis": "Decay und Clamp halten die Gewichte im deklarierten Korridor.",
"expected_success": None,
"required_features": ["weight_stabilization"],
"settings": {
"plasticity_rule": "structural",
"weight": 4,
"weight_decay": 0.01,
"weight_max_clamp": 10,
"ticks": 2000,
"inhibitory_fraction": 0.25,
"pan_feedback_gain": 1.5,
},
},
"fix_channel_separation": {
"description": "Getrennte Kanäle für Ziel, Reward und Aktion.",
"hypothesis": "Drei getrennte Signale bleiben im Netzwerk unterscheidbar.",
"expected_success": None,
"required_features": ["channel_separation"],
"settings": {
"target_cue_channel": 0,
"reward_cue_channel": 5,
"action_feedback_channel": 6,
"input_channels": 8,
"input_amplitude_per_channel": [8, 8, 8, 8, 8, 5, 5, 0],
},
},
"fix_weight_and_channels": {
"description": "Kombiniert Gewichts-Stabilisierung und Kanaltrennung.",
"hypothesis": "Ein stabiler, differenzierter Closed Loop wird möglich.",
"expected_success": None,
"required_features": ["weight_stabilization", "channel_separation"],
"settings": {
"closed_loop_preset": "fix_weight_explosion",
"target_cue_channel": 0,
"reward_cue_channel": 5,
"action_feedback_channel": 6,
"input_channels": 8,
"inhibitory_fraction": 0.25,
"pan_feedback_gain": 1.5,
},
},
"feedback_gain_sweep": {
"description": "Niedrigerer PAN-Feedback-Gain mit tanh-Sättigung.",
"hypothesis": "Ein Gain von 0.5 kann stabiler als 1.5 sein.",
"expected_success": None,
"required_features": ["pan_feedback"],
"settings": {
"pan_feedback_gain": 0.5,
"pan_feedback_nonlinearity": "tanh",
"pan_feedback_saturation": 20,
},
},
"input_differentiation": {
"description": "Per-Kanal-Amplituden, Frequenzen und Phasen.",
"hypothesis": "Differenzierter Input kann neuronale Spezialisierung erzeugen.",
"expected_success": None,
"required_features": ["differentiated_input"],
"settings": {
"input_topology": "channel_partitioned",
"input_channels": 8,
"input_amplitude_per_channel": [8, 8, 8, 8, 4, 4, 4, 4],
"input_frequency_per_channel": [20, 22, 24, 26, 30, 32, 34, 36],
"input_phase_per_channel": [0, 0.25, 0.5, 0.75, 0, 0.25, 0.5, 0.75],
"input_noise_sigma": 0.1,
},
},
"credit_assignment_trace": {
"description": "Reward-modulierte Eligibility-Traces.",
"hypothesis": "Zeitliche Kreditzuweisung stabilisiert Policy-Anpassung.",
"expected_success": None,
"required_features": ["reward_modulated_stdp"],
"settings": {
"credit_assignment": "reward_modulated_stdp",
"credit_window": 64,
"eligibility_trace_tau": 200,
"td_lambda": 0.9,
"gamma_discount": 0.95,
},
},
"reward_shaping_dense": {
"description": "Dichter, schwächerer Reward statt Sparse-Reward.",
"hypothesis": "Dense Reward erzeugt mehr Policy-Updates, eventuell mit weniger Stabilität.",
"expected_success": None,
"required_features": ["dense_reward"],
"settings": {
"reward_shaping": "dense",
"reward_magnitude": 2,
"reward_delay_ticks": 0,
},
},
"d1_minimal_closed_loop": {
"description": "Minimaler geschlossener Kreis mit Stabilisierung und Lernraten-Defaults.",
"hypothesis": "PAN kann mit einem minimalen Loop über Zufall lernen.",
"expected_success": 0.30,
"required_features": ["action_loop", "target_cue", "reward_channel"],
"settings": {
"closed_loop_preset": "minimal_closed_loop",
"action_loop_enabled": True,
"action_space_size": 4,
"target_encoding": "one_hot",
"target_cue_channel": 0,
"reward_signal_enabled": True,
"reward_channel": 5,
"posture_reward_enabled": True,
"posture_score_channel": 2,
"reward_event_channel": 3,
"pan_feedback_gain": 1.5,
"inhibitory_fraction": 0.25,
"weight_decay": 0.01,
"weight_max_clamp": 10,
"behavior_learning_rate": 0.2,
"behavior_epsilon": 0.2,
"ticks": 2000,
},
},
"d2_full_embodiment_v2": {
"description": "Vollständiger räumlicher Closed Loop mit allen Hebeln.",
"hypothesis": "Die kombinierte Kopplung erzeugt komplexe, aber prüfbare Dynamik.",
"expected_success": None,
"required_features": [
"spatial_input",
"action_loop",
"sandbox",
"reward_modulated_stdp",
],
"settings": {
"closed_loop_preset": "d1_minimal_closed_loop",
"input_topology": "spatial_gradient",
"geometry_input_coupling": True,
"geometry_input_sigma": 0.2,
"action_to_input_map": "spatial",
"action_coupling_strength": 2,
"credit_assignment": "reward_modulated_stdp",
"oscillation_enabled": True,
"oscillation_frequency": 8,
"sandbox_enabled": True,
"sandbox_physics": "stick_figure",
},
},
"d3_spatial_embodiment_only": {
"description": "Räumliche Kopplung ohne Ziel- oder Reward-Signal.",
"hypothesis": "Räumliche Kopplung erzeugt Reaktivität, aber keine Zielgerichtetheit.",
"expected_success": 0.25,
"required_features": ["spatial_input", "action_loop"],
"settings": {
"input_topology": "spatial_gradient",
"geometry_input_coupling": True,
"geometry_input_sigma": 0.2,
"action_to_input_map": "spatial",
"action_loop_enabled": True,
"reward_signal_enabled": False,
"target_encoding": "none",
},
},
"e1_diagnose_silence": {
"description": "Diagnostikprofil für niedrige Aktivität.",
"hypothesis": "Findet die Grenze zwischen Stille und Aktivität.",
"expected_success": None,
"required_features": ["diagnostic"],
"settings": {
"stimulus_current": 4,
"pan_bias_current": 5,
"weight": 4,
"inhibitory_fraction": 0.25,
},
},
"e2_diagnose_synchrony": {
"description": "Diagnostikprofil für E/I-Synchronie.",
"hypothesis": "Identifiziert die Inhibitionsschwelle der Synchronie.",
"expected_success": None,
"required_features": ["diagnostic"],
"settings": {
"inhibitory_fraction": 0.1,
"gaba_strength": 3,
"pan_feedback_gain": 0.5,
},
},
"e3_diagnose_context_policy": {
"description": "Policy-Diagnostik ohne Zielsignal.",
"hypothesis": "Prüft Policy-Verhalten ohne Kontext-Cue.",
"expected_success": 0.25,
"required_features": ["diagnostic"],
"settings": {
"target_encoding": "none",
"target_cue_channel": 0,
"behavior_target_mode": "fixed",
"behavior_target_action": 0,
},
},
"f1_robustness_seed_sweep": {
"description": "Reproduzierbarkeit mit Seed 42 und D1-Basis.",
"hypothesis": "Die Erfolgsrate bleibt über Seeds vergleichbar.",
"expected_success": None,
"required_features": ["seed_sweep"],
"settings": {"closed_loop_preset": "d1_minimal_closed_loop", "seed": 42},
},
"f2_robustness_scale_up": {
"description": "Skalierung auf 512 Neuronen.",
"hypothesis": "Mehr Neuronen verbessern Lernen nicht automatisch.",
"expected_success": None,
"required_features": ["scale_up"],
"settings": {
"closed_loop_preset": "d1_minimal_closed_loop",
"n_neurons": 512,
"edge_budget": 4096,
"k_neighbors": 16,
},
},
"f3_robustness_ablation": {
"description": "Ablationsbasis mit deaktiviertem Feedback.",
"hypothesis": "Ein kritischer Hebel fehlt und reduziert Lernen.",
"expected_success": 0.25,
"required_features": ["ablation"],
"settings": {
"closed_loop_preset": "d1_minimal_closed_loop",
"pan_feedback_gain": 0.05,
},
},
"g1_two_action_simple": {
"description": "Minimale Zwei-Aktionen-Lernaufgabe.",
"hypothesis": "Ein fixes Ziel mit zwei Aktionen sollte schnell lernbar sein.",
"expected_success": 0.9,
"required_features": ["sanity_check"],
"settings": {
"closed_loop_preset": "d1_minimal_closed_loop",
"behavior_action_count": 2,
"action_space_size": 2,
"behavior_target_mode": "fixed",
"behavior_target_action": 0,
"behavior_epsilon": 0.1,
"behavior_learning_rate": 0.3,
},
},
"g2_one_action_trivial": {
"description": "Trivialer Ein-Aktions-Sanity-Check.",
"hypothesis": "Die einzige mögliche Aktion muss ab Episode 1 erfolgreich sein.",
"expected_success": 1.0,
"required_features": ["sanity_check"],
"settings": {
"closed_loop_preset": "d1_minimal_closed_loop",
"behavior_action_count": 1,
"action_space_size": 1,
"behavior_target_mode": "fixed",
"behavior_target_action": 0,
"behavior_epsilon": 0.0,
},
},
"g3_gaba_sweep": {
"description": "E/I-Stärkeprofil für GABA-Sweep.",
"hypothesis": "Eine mittlere GABA-Stärke balanciert Aktivität und Synchronie.",
"expected_success": None,
"required_features": ["gaba_sweep"],
"settings": {"inhibitory_fraction": 0.25, "gaba_strength": 2},
},
"g4_feedback_nonlinearity": {
"description": "Vergleich der Feedback-Nichtlinearität.",
"hypothesis": "Die Form der Rückkopplung verändert Stabilität und Lernen.",
"expected_success": None,
"required_features": ["feedback_sweep"],
"settings": {"pan_feedback_nonlinearity": "linear"},
},
}
)
CLOSED_LOOP_PRESETS["pan_cuda_hybrid"] = {
"label": "PAN · CUDA-Hybrid",
"description": "PAN-Vollprofil mit CUDA-Membran, PAN-Zustand/Feedback, synaptischer Aussendung und Reward-Updates. Körper, Policy und RNG bleiben CPU; Delay-Queue und Neuron-Traces liegen auf der GPU. NVIDIA-Treiber und NVRTC erforderlich; kein Speedup-Versprechen.",
"hypothesis": "Explorativer hybrider CPU/GPU-Vergleich mit denselben PAN-Parametern und expliziter Komponentenanzeige.",
"expected_success": None,
"required_features": ["pan", "cuda_driver", "nvrtc", "sandbox", "neural_io"],
"settings": {
"closed_loop_preset": "pan_full_balanced",
"name": "pan_cuda_hybrid",
"neuron_backend": "cuda_pan",
},
}
def closed_loop_catalog() -> dict[str, object]:
"""Return UI-safe closed-loop capabilities and preset metadata."""
presets = []
for name, item in CLOSED_LOOP_PRESETS.items():
features = cast(list[str], item["required_features"])
settings = cast(dict[str, object], item["settings"])
presets.append(
{
"name": name,
"label": item.get("label", name),
"description": item["description"],
"hypothesis": item["hypothesis"],
"expected_success": item["expected_success"],
"required_features": list(features),
"settings": dict(settings),
}
)
return {
"classification": "PLAYGROUND_CLOSED_LOOP",
"scientific_evidence": False,
"runtime_status": "IMPLEMENTED_BOUNDED_REFERENCE",
"input_topologies": [
"uniform",
"channel_partitioned",
"spatial_gradient",
"random_per_neuron",
],
"feedback_sources": ["population", "layer", "subset", "hypervector"],
"feedback_targets": ["all", "layer", "random_subset"],
"feedback_nonlinearities": ["linear", "tanh", "sign", "clip"],
"target_encodings": ["none", "one_hot", "rate", "population_latency"],
"target_predictability": ["deterministic", "stochastic", "adversarial"],
"reward_shaping": ["sparse", "dense", "potential_based"],
"credit_assignment": ["none", "trace", "reward_modulated_stdp"],
"delay_distributions": ["fixed", "uniform", "lognormal", "gamma"],
"presets": presets,
"note": (
"Closed-loop behavior is Playground-only. Preset expectations are "
"hypotheses, not observed results or evidence."
),
}
def _expand(values: Sequence[float], count: int, default: float) -> list[float]:
clean = [float(value) for value in values]
if not clean:
return [default for _ in range(count)]
return [clean[index % len(clean)] for index in range(count)]
class ClosedLoopRuntime:
"""Deterministic channelized action/consequence/reward environment."""
def __init__(
self,
config: PlaygroundConfig,
coordinates: Sequence[Sequence[float]],
) -> None:
self.config = config
self.n_neurons = config.n_neurons
self.channels = config.input_channels
self.coordinates = coordinates
self.rng = random.Random(config.seed ^ 0xC105ED)
self.channel_map = self._build_channel_map()
self.amplitudes = _expand(
config.input_amplitude_per_channel,
self.channels,
0.0,
)
self.frequencies = _expand(
config.input_frequency_per_channel,
self.channels,
20.0,
)
self.phases = _expand(
config.input_phase_per_channel,
self.channels,
0.0,
)
self.action_history: list[int] = []
self.target_history: list[int] = []
self.reward_history: list[float] = []
self.successes = 0
self.pending_actions: list[tuple[int, int, int]] = []
self.pending_rewards: list[tuple[int, float, int, int]] = []
self.delivered_rewards: list[tuple[int, int, float]] = []
self.reward_trace = 0.0
self.last_action: int | None = None
self.observed_context: str | None = None
self.previous_action: int | None = None
self.target_permutation = list(range(config.action_space_size))
if config.target_shuffle:
self.rng.shuffle(self.target_permutation)
def _build_channel_map(self) -> list[list[int]]:
provided = self.config.input_channel_map
if provided:
return [list(group) for group in provided]
groups: list[list[int]] = [[] for _ in range(self.channels)]
if self.config.input_topology == "uniform":
all_neurons = list(range(self.n_neurons))
return [list(all_neurons) for _ in range(self.channels)]
if (
self.config.input_topology == "spatial_gradient"
or self.config.geometry_input_coupling
) and self.coordinates:
ordered = sorted(
range(self.n_neurons),
key=lambda index: (
float(self.coordinates[index][0])
if self.coordinates[index]
else 0.0
),
)
elif self.config.input_topology == "random_per_neuron":
ordered = list(range(self.n_neurons))
self.rng.shuffle(ordered)
else:
ordered = list(range(self.n_neurons))
for position, neuron_id in enumerate(ordered):
channel = min(
self.channels - 1,
(position * self.channels) // max(self.n_neurons, 1),
)
if self.config.input_topology == "random_per_neuron":
channel = self.rng.randrange(self.channels)
groups[channel].append(neuron_id)
return groups
def _target_for_episode(self, episode: int) -> int:
count = self.config.action_space_size
if self.config.behavior_target_mode == "fixed":
return self.config.behavior_target_action
mode = self.config.target_predictability
if mode == "stochastic":
target_rng = random.Random(self.config.seed ^ 0x7A267 ^ episode)
target = target_rng.randrange(count)
elif mode == "adversarial":
target = (
(self.last_action + 1) % count
if self.last_action is not None
else episode % count
)
else:
target = episode % count
if self.target_permutation:
target = self.target_permutation[target % len(self.target_permutation)]
return target
def current_target(self, tick: int) -> int:
episode_ticks = max(1, self.config.behavior_episode_ticks)
episode = tick // episode_ticks
return self._target_for_episode(episode)
def _target_channel_values(self, tick: int) -> list[float]:
values = [0.0 for _ in range(self.channels)]
if (
self.config.target_encoding == "none"
or self.config.target_cue_control == "absent"
):
return values
episode_ticks = max(1, self.config.behavior_episode_ticks)
phase_tick = tick % episode_ticks
if phase_tick >= self.config.target_persistence:
return values
target = self.current_target(tick)
if self.config.target_cue_control == "randomized":
# Separate RNG stream: changes the emitted cue, never the evaluator target.
cue_rng = random.Random(
self.config.seed ^ 0xC0E123 ^ (tick // episode_ticks)
)
target = cue_rng.randrange(self.config.action_space_size)
base = self.config.target_cue_channel % self.channels
if self.config.target_encoding == "one_hot":
channel = (base + target) % self.channels
values[channel] = self.config.target_cue_current
elif self.config.target_encoding == "rate":
values[base] = self.config.target_cue_current * (
(target + 1) / max(self.config.action_space_size, 1)
)
else:
latency = target % max(1, self.config.target_persistence)
if phase_tick == latency:
values[base] = self.config.target_cue_current
return values
def _action_vector(self, action: int) -> list[float]:
mapping = self.config.action_to_input_map
if not isinstance(mapping, str) and mapping:
rows = cast(Sequence[Sequence[float]], mapping)
row = rows[action % len(rows)]
return _expand(row, self.channels, 0.0)
values = [0.0 for _ in range(self.channels)]
base = self.config.action_feedback_channel % self.channels
if mapping == "spatial":
center = (action / max(self.config.action_space_size - 1, 1)) * max(
self.channels - 1, 1
)
sigma = max(self.config.geometry_input_sigma * self.channels, 0.5)
for channel in range(self.channels):
distance = (channel - center) / sigma
values[channel] = math.exp(-0.5 * distance * distance)
else:
values[(base + action) % self.channels] = 1.0
return values
def _channel_values(self, tick: int) -> list[float]:
time_s = tick * self.config.dt_ms / 1000.0
values = []
for channel in range(self.channels):
phase = self.phases[channel]
angle = 2.0 * math.pi * self.frequencies[channel] * time_s + phase
carrier = 0.5 + 0.5 * math.sin(angle)
values.append(self.amplitudes[channel] * carrier)
target_values = self._target_channel_values(tick)
# The policy only sees an emitted cue, never the evaluator's target.
# Remember a transient cue within its episode, and forget it at reset.
if tick % max(1, self.config.behavior_episode_ticks) == 0:
self.observed_context = None
if any(value != 0.0 for value in target_values):
phase = tick % max(1, self.config.behavior_episode_ticks)
signature = ",".join(format(value, ".12g") for value in target_values)
suffix = (
f":at:{phase}"
if self.config.target_encoding == "population_latency"
else ""
)
self.observed_context = (
f"cue:{self.config.target_encoding}:{signature}{suffix}"
)
for channel in range(self.channels):
values[channel] += target_values[channel]
active_actions: list[tuple[int, int, int]] = []
for start, end, action in self.pending_actions:
if start <= tick < end:
vector = self._action_vector(action)
for channel in range(self.channels):
noise = (
self.rng.gauss(0.0, self.config.action_noise)
if self.config.action_noise > 0.0
else 0.0
)
values[channel] += self.config.action_coupling_strength * (
vector[channel] + noise
)
if tick < end:
active_actions.append((start, end, action))
self.pending_actions = active_actions
remaining_rewards: list[tuple[int, float, int, int]] = []
for reward_tick, reward, action, target in self.pending_rewards:
if reward_tick <= tick:
self.reward_trace += reward
self.delivered_rewards.append((action, target, reward))
else:
remaining_rewards.append((reward_tick, reward, action, target))
self.pending_rewards = remaining_rewards
if self.config.reward_signal_enabled and abs(self.reward_trace) > 1e-12:
channel = self.config.reward_channel % self.channels
values[channel] += self.reward_trace
self.reward_trace *= self.config.reward_decay
if abs(self.reward_trace) < 1e-9:
self.reward_trace = 0.0
return values
def currents(self, tick: int) -> list[float]:
values = self._channel_values(tick)
currents = [0.0 for _ in range(self.n_neurons)]
counts = [0 for _ in range(self.n_neurons)]
for channel, neurons in enumerate(self.channel_map):
for neuron_id in neurons:
if 0 <= neuron_id < self.n_neurons:
currents[neuron_id] += values[channel]
counts[neuron_id] += 1
for neuron_id in range(self.n_neurons):
if counts[neuron_id] > 1:
currents[neuron_id] /= counts[neuron_id]
noise_sigma = self.config.input_noise_sigma
if self.config.sandbox_enabled:
noise_sigma = min(
1.0,
noise_sigma + self.config.sandbox_sensor_noise,
)
if noise_sigma > 0.0:
currents[neuron_id] += self.rng.gauss(0.0, noise_sigma)
return currents
def _reward(self, action: int, target: int) -> float:
magnitude = self.config.reward_magnitude
shaping = self.config.reward_shaping
match = action == target
if shaping == "dense":
distance = abs(action - target) / max(
self.config.action_space_size - 1,
1,
)
reward = magnitude * (1.0 - 2.0 * distance)
elif shaping == "potential_based":
current_distance = abs(action - target)
if self.previous_action is None:
previous_distance = self.config.action_space_size - 1
else:
previous_distance = abs(self.previous_action - target)
reward = (
magnitude
* (previous_distance - current_distance)
/ max(
self.config.action_space_size - 1,
1,
)
)
if match:
reward += magnitude
else:
reward = magnitude if match else 0.0
return reward - self.config.reward_baseline
def note_action(self, *, action: int, tick: int) -> tuple[int, float]:
target = self.current_target(tick)
episode_index = len(self.action_history)
reward = self._reward(action, target)
if self.config.freeze_rewards:
reward = self.config.frozen_reward_sequence[episode_index]
if action == target:
self.successes += 1
self.previous_action = self.last_action
self.last_action = action
self.action_history.append(action)
self.target_history.append(target)
self.reward_history.append(reward)
if self.config.action_loop_enabled:
start = tick + self.config.action_loop_delay
end = start + self.config.action_persistence
self.pending_actions.append((start, end, action))
if self.config.reward_signal_enabled:
reward_tick = tick + self.config.reward_delay_ticks
if self.config.reward_delay_ticks == 0:
self.reward_trace += reward
self.delivered_rewards.append((action, target, reward))
else:
self.pending_rewards.append((reward_tick, reward, action, target))
return target, reward
def consume_delivered_rewards(self) -> list[tuple[int, int, float]]:
rewards = list(self.delivered_rewards)
self.delivered_rewards.clear()
return rewards
def summary(self) -> dict[str, object]:
episodes = len(self.action_history)
if self.config.freeze_actions and self.config.freeze_rewards:
parity_mode = "FROZEN_ACTIONS_AND_REWARDS"
elif self.config.freeze_actions:
parity_mode = "FROZEN_ACTIONS"
elif self.config.freeze_rewards:
parity_mode = "FROZEN_REWARDS"
else:
parity_mode = "LIVE_CLOSED_LOOP"
return {
"classification": "PLAYGROUND_CLOSED_LOOP",
"scientific_evidence": False,
"preset": self.config.closed_loop_preset,
"input_topology": self.config.input_topology,
"input_channels": self.channels,
"channel_sizes": [len(group) for group in self.channel_map],
"action_loop_enabled": self.config.action_loop_enabled,
"action_space_size": self.config.action_space_size,
"freeze_actions": self.config.freeze_actions,
"freeze_rewards": self.config.freeze_rewards,
"parity_reference_source": self.config.parity_reference_source,
"parity_reference_commit": self.config.parity_reference_commit,
"target_encoding": self.config.target_encoding,
"policy_context_source": "emitted_target_cue_within_episode",
"observed_context": self.observed_context,
"reward_signal_enabled": self.config.reward_signal_enabled,
"reward_shaping": self.config.reward_shaping,
"credit_assignment": self.config.credit_assignment,
"episodes": episodes,
"successes": self.successes,
"success_fraction": self.successes / episodes if episodes else 0.0,
"action_history": list(self.action_history[-128:]),
"target_history": list(self.target_history[-128:]),
"reward_history": list(self.reward_history[-128:]),
"pending_action_effects": len(self.pending_actions),
"pending_rewards": len(self.pending_rewards),
"causal_chain": "target/input -> network -> action -> delayed input/reward",
"parity_mode": parity_mode,
}
def neuron_parameter_sets(
base: Mapping[str, float],
config: PlaygroundConfig,
) -> list[dict[str, float]]:
"""Create deterministic per-neuron threshold/tau variation."""
rng = random.Random(config.seed ^ 0x4E455552)
result: list[dict[str, float]] = []
for _ in range(config.n_neurons):
params = {key: float(value) for key, value in base.items()}
if "threshold" in params and config.neuron_threshold_variance > 0.0:
sigma = max(1.0, abs(params["threshold"]) * 0.1)
params["threshold"] += rng.gauss(
0.0,
sigma * config.neuron_threshold_variance,
)
if config.neuron_tau_m_variance > 0.0:
factor = max(
0.1,
1.0 + rng.gauss(0.0, config.neuron_tau_m_variance),
)
for key in ("tau_m_ms", "soma_tau_ms", "dendrite_tau_ms"):
if key in params:
params[key] = max(0.05, params[key] * factor)
result.append(params)
return result
def inhibitory_mask(
n_neurons: int,
fraction: float,
seed: int,
) -> list[bool]:
rng = random.Random(seed ^ 0x1A11B17)
order = list(range(n_neurons))
rng.shuffle(order)
count = int(round(n_neurons * fraction))
selected = set(order[:count])
return [index in selected for index in range(n_neurons)]
def sample_delay_ticks(
config: PlaygroundConfig,
rng: random.Random,
) -> int:
"""Sample one bounded synaptic delay according to the configured family."""
mode = config.delay_distribution
mean = max(1.0, config.delay_mean_ticks)
if mode == "uniform":
value = rng.uniform(1.0, max(1.0, 2.0 * mean - 1.0))
elif mode == "lognormal":
sigma = 0.5
mu = math.log(mean) - 0.5 * sigma * sigma
value = rng.lognormvariate(mu, sigma)
elif mode == "gamma":
value = rng.gammavariate(2.0, mean / 2.0)
else:
value = float(config.delay_ticks)
return max(1, min(64, int(round(value))))
class TemporalDynamics:
"""Small deterministic refractory/adaptation/oscillation reference."""
def __init__(self, config: PlaygroundConfig) -> None:
self.config = config
self.rng = random.Random(config.seed ^ 0x71AE)
self.refractory_until = [-1 for _ in range(config.n_neurons)]
self.adaptation = [0.0 for _ in range(config.n_neurons)]
self.refractory_ticks = []
for _ in range(config.n_neurons):
jitter = self.rng.gauss(0.0, config.refractory_variance)
self.refractory_ticks.append(max(1, int(round(1.0 + 2.0 * jitter))))
def begin_tick(self) -> None:
if self.config.adaptation_strength <= 0.0:
return
decay = math.exp(-self.config.dt_ms / max(self.config.adaptation_tau, 1e-6))
self.adaptation = [value * decay for value in self.adaptation]
def can_step(self, neuron_id: int, tick: int) -> bool:
return tick > self.refractory_until[neuron_id]
def current_adjustment(self, neuron_id: int, tick: int) -> float:
current = -self.config.adaptation_strength * self.adaptation[neuron_id]
if self.config.oscillation_enabled:
time_s = tick * self.config.dt_ms / 1000.0
current += math.sin(
2.0 * math.pi * self.config.oscillation_frequency * time_s
)
return current
def note_spikes(self, spiked_neurons: Sequence[int], tick: int) -> None:
for neuron_id in spiked_neurons:
self.refractory_until[neuron_id] = tick + self.refractory_ticks[neuron_id]
self.adaptation[neuron_id] += 1.0
def summary(self) -> dict[str, object]:
return {
"classification": "PLAYGROUND_TEMPORAL_DYNAMICS",
"scientific_evidence": False,
"refractory_variance": self.config.refractory_variance,
"adaptation_strength": self.config.adaptation_strength,
"adaptation_tau_ms": self.config.adaptation_tau,
"oscillation_enabled": self.config.oscillation_enabled,
"oscillation_frequency_hz": self.config.oscillation_frequency,
"mean_refractory_ticks": sum(self.refractory_ticks)
/ max(len(self.refractory_ticks), 1),
}