Download generator.py from RLTT/generator-v2: direct link, hf CLI and curl.
- Browser
- Download file 26.3 kB
-
https://huggingface.co/RLTT/generator-v2/resolve/main/generator.py
- Command line
-
hf download hf://RLTT/generator-v2/generator.py
-
curl -L -o generator.py https://huggingface.co/RLTT/generator-v2/resolve/main/generator.py
26.3 kB
| """generator_v2 β SARIMA-compositional backbone + chaotic-dynamical-systems and | |
| long-memory priors, for cascade. | |
| This is a meaningful capability upgrade over the ``sarima-compositional-v2`` | |
| prior. That generator spanned the *linear-stochastic* space very well (SARIMA | |
| with regime switching, stochastic volatility, fat tails, compositional | |
| components and warps). It structurally *could not* produce two whole classes of | |
| dynamics that real held-out series exhibit and that recent TSFM synthetic-prior | |
| research shows are high-value. v2 adds them as first-class cores: | |
| A. **Chaotic dynamical systems** (the DynaMix, NeurIPS'25 insight β a TSFM | |
| trained purely on a handful of chaotic attractors generalises zero-shot to | |
| real traffic/weather). We integrate continuous chaotic *flows* (Lorenz, | |
| RΓΆssler, Thomas, Chen, driven Van der Pol) with RK4, plus the Mackey-Glass | |
| delay system. These give deterministic-but-complex, broadband, long-range | |
| structure that no ARMA prior contains. Integration is a bounded scalar | |
| recursion capped at ~1.2k steps and resampled to the target length, so cost | |
| stays linear and small. | |
| B. **Long-memory / fractional processes.** SARIMA offers only I(0)/I(1)/I(2); | |
| nothing *between*. We add (i) power-law / colored-noise spectral synthesis | |
| (1/f^Ξ², Ξ²β[-1,3], via FFT) covering anti-persistent β pink β Brownian | |
| spectra, and (ii) ARFIMA(p,d,0) via truncated fractional differencing | |
| (Hosking coefficients), giving genuine long-range dependence (Hurst β 0.5). | |
| C. **Bilinear nonlinear-AR** (y_t = Ο y_{t-1} + b y_{t-1} e_{t-1} + ΞΈ e_{t-1} + | |
| e_t) β bursty multiplicative autocorrelation distinct from SETAR/GARCH. | |
| Cores are selected per series by ``core_weights`` (SARIMA stays dominant β it is | |
| the strongest single scorer). Every core then flows through the *same* | |
| compositional-component, nonlinear-warp, TSMixup and finalisation stack as | |
| before, so the new dynamics inherit all the enrichment breadth. | |
| Determinism: the corpus is a pure function of ``(seed, n_series)``. Every | |
| per-series sub-seed is derived from the master seed via | |
| ``np.random.SeedSequence``; no ``hash()``, wall-clock, or unseeded global RNG. | |
| The chaotic/bilinear scalar recursions are ordinary deterministic float | |
| arithmetic. Any core that diverges or returns non-finite falls back | |
| deterministically to the SARIMA core. | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import math | |
| from collections.abc import Iterator | |
| from pathlib import Path | |
| import numpy as np | |
| from scipy.signal import lfilter | |
| from cascade.interface import DataGenerator | |
| # Fixed stream id for the length-drawing RNG, kept distinct from any per-series | |
| # sub-seed so lengths are order-deterministic and independent of model draws. | |
| _LENGTH_STREAM_ID = 0x1E_2757 | |
| # Candidate seasonal periods (timesteps): intraday (4/24/48), business/weekly | |
| # (5/7/168), monthly (12/30) and longer rhythms. A period is used only when the | |
| # series is long enough for >= 3 full cycles. | |
| _SEASONAL_PERIODS: tuple[int, ...] = (4, 5, 7, 12, 24, 30, 48, 96, 144, 168, 336) | |
| # Continuous chaotic flows integrated by RK4. | |
| _CHAOTIC_FLOWS: tuple[str, ...] = ("lorenz", "rossler", "thomas", "chen", "vanderpol") | |
| class Generator(DataGenerator): | |
| """SARIMA-compositional + chaotic + long-memory corpus generator.""" | |
| def __init__(self, config_dir: str, *, seed: int) -> None: | |
| cfg_path = Path(config_dir) / "config.json" | |
| cfg = json.loads(cfg_path.read_text(encoding="utf-8")) if cfg_path.is_file() else {} | |
| self._seed = int(seed) | |
| self._min_len = int(cfg.get("min_length", 64)) | |
| self._max_len = int(cfg.get("max_length", 4096)) | |
| if not (1 <= self._min_len <= self._max_len): | |
| raise ValueError(f"invalid length band [{self._min_len}, {self._max_len}]") | |
| # SARIMA order caps (per-series orders drawn in [0, cap]). | |
| self._max_p = int(cfg.get("max_ar", 4)) | |
| self._max_q = int(cfg.get("max_ma", 4)) | |
| self._max_P = int(cfg.get("max_seasonal_ar", 2)) | |
| self._max_Q = int(cfg.get("max_seasonal_ma", 2)) | |
| # Integration-order weights (index i = probability mass for order i). | |
| self._d_weights = np.asarray(cfg.get("d_weights", [0.5, 0.38, 0.12]), dtype=np.float64) | |
| self._seasonal_d_weights = np.asarray( | |
| cfg.get("seasonal_d_weights", [0.7, 0.3]), dtype=np.float64 | |
| ) | |
| # Core-family selection weights (per series). | |
| cw = cfg.get("core_weights", {}) | |
| self._core_names = ("sarima", "chaotic", "longmem", "bilinear") | |
| self._core_w = np.asarray( | |
| [ | |
| float(cw.get("sarima", 0.60)), | |
| float(cw.get("chaotic", 0.16)), | |
| float(cw.get("longmem", 0.17)), | |
| float(cw.get("bilinear", 0.07)), | |
| ], | |
| dtype=np.float64, | |
| ) | |
| self._core_w = np.clip(self._core_w, 0.0, None) | |
| s = self._core_w.sum() | |
| self._core_w = self._core_w / s if s > 0 else np.array([1.0, 0.0, 0.0, 0.0]) | |
| # Chaotic integration budget (steps actually integrated before resampling). | |
| self._chaos_min_steps = int(cfg.get("chaos_min_steps", 512)) | |
| self._chaos_max_steps = int(cfg.get("chaos_max_steps", 1200)) | |
| # Enrichment probabilities. | |
| self._seasonal_prob = float(cfg.get("seasonal_prob", 0.5)) | |
| self._student_t_prob = float(cfg.get("student_t_prob", 0.35)) | |
| self._stoch_vol_prob = float(cfg.get("stoch_vol_prob", 0.4)) | |
| self._max_regimes = int(cfg.get("max_regimes", 3)) | |
| self._regime_prob = float(cfg.get("regime_prob", 0.5)) | |
| self._trend_prob = float(cfg.get("trend_prob", 0.5)) | |
| self._calendar_prob = float(cfg.get("calendar_prob", 0.45)) | |
| self._level_shift_prob = float(cfg.get("level_shift_prob", 0.35)) | |
| self._spike_prob = float(cfg.get("spike_prob", 0.3)) | |
| self._warp_prob = float(cfg.get("warp_prob", 0.45)) | |
| self._mixup_prob = float(cfg.get("mixup_prob", 0.25)) | |
| # Sanitisation knobs. | |
| self._max_abs = float(cfg.get("max_abs_value", 1.0e6)) | |
| self._standardize = bool(cfg.get("standardize", False)) | |
| def name(self) -> str: | |
| return "sarima-chaos-longmem-v3" | |
| # ββ deterministic sub-seeding ββββββββββββββββββββββββββββββββββββββββββββ | |
| def _series_rng(self, index: int) -> np.random.Generator: | |
| return np.random.default_rng(np.random.SeedSequence([self._seed, index])) | |
| # ββ stationary lag-polynomial construction βββββββββββββββββββββββββββββββ | |
| def _reflection_to_poly(rng: np.random.Generator, order: int, lo: float, hi: float) -> np.ndarray: | |
| """Reflection coefficients (PACF) in (lo, hi) β lag polynomial via | |
| Levinson-Durbin. Guarantees all roots outside the unit circle.""" | |
| if order <= 0: | |
| return np.array([1.0]) | |
| kappa = rng.uniform(lo, hi, size=order) | |
| phi = np.zeros(order, dtype=np.float64) | |
| for m in range(order): | |
| k = kappa[m] | |
| prev = phi[:m].copy() | |
| phi[m] = k | |
| if m > 0: | |
| phi[:m] = prev - k * prev[::-1] | |
| return np.concatenate(([1.0], -phi)) | |
| def _expand_seasonal(poly: np.ndarray, period: int) -> np.ndarray: | |
| """Lift a lag polynomial in ``L`` to one in ``L**period``.""" | |
| if poly.size <= 1 or period <= 1: | |
| return poly.copy() | |
| out = np.zeros((poly.size - 1) * period + 1, dtype=np.float64) | |
| out[0] = poly[0] | |
| for i in range(1, poly.size): | |
| out[i * period] = poly[i] | |
| return out | |
| def _choose_period(self, rng: np.random.Generator, length: int) -> int: | |
| if rng.random() >= self._seasonal_prob: | |
| return 1 | |
| candidates = [p for p in _SEASONAL_PERIODS if p * 3 <= length] | |
| return int(rng.choice(candidates)) if candidates else 1 | |
| def _weighted_order(rng: np.random.Generator, weights: np.ndarray) -> int: | |
| w = np.clip(weights, 0.0, None) | |
| total = w.sum() | |
| return int(rng.choice(len(w), p=w / total)) if total > 0.0 else 0 | |
| # ββ innovations: fat tails + stochastic volatility (vectorised) ββββββββββ | |
| def _draw_innovations(self, rng: np.random.Generator, n: int) -> np.ndarray: | |
| if rng.random() < self._student_t_prob: | |
| df = float(rng.uniform(3.0, 12.0)) | |
| eps = rng.standard_t(df, size=n) | |
| else: | |
| eps = rng.standard_normal(n) | |
| eps = eps * float(rng.uniform(0.3, 2.0)) | |
| # Stochastic volatility: multiply by exp(0.5 * AR(1) log-variance) β | |
| # a vectorised lfilter draw gives volatility clustering with no Python loop. | |
| if rng.random() < self._stoch_vol_prob: | |
| phi_v = float(rng.uniform(0.9, 0.995)) | |
| v_innov = rng.normal(0.0, float(rng.uniform(0.1, 0.4)), size=n) | |
| log_var = lfilter([1.0], [1.0, -phi_v], v_innov) | |
| log_var -= log_var.mean() | |
| eps = eps * np.exp(0.5 * np.clip(log_var, -6.0, 6.0)) | |
| return eps | |
| # ββ one stationary+integrated SARIMA segment βββββββββββββββββββββββββββββ | |
| def _sarima_segment(self, rng: np.random.Generator, length: int, period: int) -> np.ndarray: | |
| seasonal = period > 1 | |
| p = int(rng.integers(0, self._max_p + 1)) | |
| q = int(rng.integers(0, self._max_q + 1)) | |
| d = self._weighted_order(rng, self._d_weights) | |
| if seasonal: | |
| P = int(rng.integers(0, self._max_P + 1)) | |
| Q = int(rng.integers(0, self._max_Q + 1)) | |
| D = self._weighted_order(rng, self._seasonal_d_weights) | |
| else: | |
| P = Q = D = 0 | |
| if p == q == P == Q == d == D == 0: | |
| p = 1 if rng.random() < 0.5 else 0 | |
| q = 0 if p else 1 | |
| ar = self._reflection_to_poly(rng, p, -0.95, 0.95) | |
| ma = self._reflection_to_poly(rng, q, -0.9, 0.9) | |
| if seasonal: | |
| ar = np.convolve(ar, self._expand_seasonal(self._reflection_to_poly(rng, P, -0.95, 0.95), period)) | |
| ma = np.convolve(ma, self._expand_seasonal(self._reflection_to_poly(rng, Q, -0.9, 0.9), period)) | |
| memory = max(ar.size, ma.size, period * max(P, D, 1)) | |
| burn = int(min(2048, max(128, 4 * memory))) | |
| innov = self._draw_innovations(rng, length + burn) | |
| y = lfilter(ma, ar, innov)[burn:] | |
| for _ in range(d): | |
| y = np.cumsum(y - y.mean()) | |
| if seasonal and D > 0: | |
| seas_int = np.zeros(period + 1, dtype=np.float64) | |
| seas_int[0], seas_int[period] = 1.0, -1.0 | |
| for _ in range(D): | |
| y = lfilter([1.0], seas_int, y - y.mean()) | |
| return y | |
| # ββ regime-switching SARIMA core: stitch segments with level continuity ββ | |
| def _regime_core(self, rng: np.random.Generator, length: int) -> np.ndarray: | |
| period = self._choose_period(rng, length) | |
| n_regimes = 1 | |
| if rng.random() < self._regime_prob and length >= 96: | |
| n_regimes = int(rng.integers(2, self._max_regimes + 1)) | |
| if n_regimes == 1: | |
| return self._sarima_segment(rng, length, period) | |
| # Partition length into n_regimes contiguous segments (each >= 32). | |
| cuts = np.sort(rng.choice(np.arange(32, length - 32), size=n_regimes - 1, replace=False)) \ | |
| if length - 64 > n_regimes else np.array([], dtype=int) | |
| bounds = [0, *cuts.tolist(), length] | |
| segs, offset = [], 0.0 | |
| for a, b in zip(bounds[:-1], bounds[1:], strict=True): | |
| seg_len = b - a | |
| if seg_len <= 0: | |
| continue | |
| # Occasionally re-roll seasonality per regime for richer breaks. | |
| seg_period = period if rng.random() < 0.7 else self._choose_period(rng, seg_len) | |
| core = self._sarima_segment(rng, seg_len, seg_period) | |
| core = core - core[0] + offset | |
| segs.append(core) | |
| offset = core[-1] | |
| x = np.concatenate(segs) if segs else self._sarima_segment(rng, length, period) | |
| return x[:length] | |
| # ββ chaotic dynamical-systems core βββββββββββββββββββββββββββββββββββββββ | |
| def _rk4_step(deriv, s: tuple[float, ...], dt: float) -> tuple[float, ...]: | |
| k1 = deriv(s) | |
| s2 = tuple(a + 0.5 * dt * b for a, b in zip(s, k1)) | |
| k2 = deriv(s2) | |
| s3 = tuple(a + 0.5 * dt * b for a, b in zip(s, k2)) | |
| k3 = deriv(s3) | |
| s4 = tuple(a + dt * b for a, b in zip(s, k3)) | |
| k4 = deriv(s4) | |
| return tuple( | |
| a + (dt / 6.0) * (b + 2.0 * c + 2.0 * e + g) | |
| for a, b, c, e, g in zip(s, k1, k2, k3, k4) | |
| ) | |
| def _integrate(self, deriv, state, dt, burn, n, obs): | |
| s = state | |
| for _ in range(burn): | |
| s = self._rk4_step(deriv, s, dt) | |
| if not math.isfinite(s[obs]) or abs(s[obs]) > 1.0e8: | |
| return None | |
| out = np.empty(n, dtype=np.float64) | |
| for i in range(n): | |
| s = self._rk4_step(deriv, s, dt) | |
| v = s[obs] | |
| if not math.isfinite(v) or abs(v) > 1.0e8: | |
| return None | |
| out[i] = v | |
| return out | |
| def _chaotic_flow(self, rng: np.random.Generator, system: str, n: int): | |
| if system == "lorenz": | |
| sigma = 10.0 * rng.uniform(0.85, 1.15) | |
| rho = 28.0 * rng.uniform(0.9, 1.12) | |
| beta = (8.0 / 3.0) * rng.uniform(0.85, 1.15) | |
| dt = 0.01 * rng.uniform(0.7, 1.4) | |
| def f(s): | |
| x, y, z = s | |
| return (sigma * (y - x), x * (rho - z) - y, x * y - beta * z) | |
| state = (rng.uniform(-8.0, 8.0), rng.uniform(-8.0, 8.0), rng.uniform(5.0, 30.0)) | |
| obs = int(rng.integers(3)) | |
| elif system == "rossler": | |
| a = 0.2 * rng.uniform(0.7, 1.3) | |
| b = 0.2 * rng.uniform(0.7, 1.3) | |
| c = 5.7 * rng.uniform(0.85, 1.15) | |
| dt = 0.08 * rng.uniform(0.7, 1.3) | |
| def f(s): | |
| x, y, z = s | |
| return (-y - z, x + a * y, b + z * (x - c)) | |
| state = (rng.uniform(-5.0, 5.0), rng.uniform(-5.0, 5.0), rng.uniform(0.0, 5.0)) | |
| obs = int(rng.integers(3)) | |
| elif system == "thomas": | |
| bb = 0.19 * rng.uniform(0.75, 1.1) | |
| dt = 0.1 * rng.uniform(0.7, 1.3) | |
| def f(s): | |
| x, y, z = s | |
| return (math.sin(y) - bb * x, math.sin(z) - bb * y, math.sin(x) - bb * z) | |
| state = (rng.uniform(-2.0, 2.0), rng.uniform(-2.0, 2.0), rng.uniform(-2.0, 2.0)) | |
| obs = int(rng.integers(3)) | |
| elif system == "chen": | |
| a = 35.0 * rng.uniform(0.92, 1.08) | |
| b = 3.0 * rng.uniform(0.85, 1.15) | |
| c = 28.0 * rng.uniform(0.92, 1.08) | |
| dt = 0.004 * rng.uniform(0.7, 1.3) | |
| def f(s): | |
| x, y, z = s | |
| return (a * (y - x), (c - a) * x - x * z + c * y, x * y - b * z) | |
| state = (rng.uniform(-10.0, 10.0), rng.uniform(-10.0, 10.0), rng.uniform(10.0, 40.0)) | |
| obs = int(rng.integers(3)) | |
| else: # driven Van der Pol (phase carried as a third state) | |
| mu = rng.uniform(1.0, 5.0) | |
| amp = rng.uniform(0.0, 1.2) | |
| omega = rng.uniform(0.4, 1.6) | |
| dt = 0.05 * rng.uniform(0.7, 1.3) | |
| def f(s): | |
| x, v, ph = s | |
| return (v, mu * (1.0 - x * x) * v - x + amp * math.sin(ph), omega) | |
| state = (rng.uniform(-2.0, 2.0), rng.uniform(-2.0, 2.0), 0.0) | |
| obs = 0 | |
| burn = int(rng.integers(200, 800)) | |
| return self._integrate(f, state, dt, burn, n, obs) | |
| def _mackey_glass(self, rng: np.random.Generator, n: int): | |
| beta = rng.uniform(0.18, 0.28) | |
| gamma = rng.uniform(0.09, 0.11) | |
| tau = int(rng.integers(18, 31)) | |
| burn = int(rng.integers(200, 600)) | |
| total = tau + burn + n | |
| x = np.empty(total, dtype=np.float64) | |
| x[: tau + 1] = rng.uniform(0.5, 1.2, size=tau + 1) | |
| for t in range(tau, total - 1): | |
| xd = x[t - tau] | |
| nxt = x[t] + (beta * xd / (1.0 + xd**10) - gamma * x[t]) | |
| if not math.isfinite(nxt) or abs(nxt) > 1.0e8: | |
| return None | |
| x[t + 1] = nxt | |
| return x[tau + burn:] | |
| def _chaotic_core(self, rng: np.random.Generator, length: int): | |
| hi = min(self._chaos_max_steps, length) | |
| lo = min(self._chaos_min_steps, hi) | |
| n_int = int(rng.integers(lo, hi + 1)) if hi > lo else hi | |
| n_int = max(n_int, 16) | |
| if rng.random() < 0.2: | |
| traj = self._mackey_glass(rng, n_int) | |
| else: | |
| system = _CHAOTIC_FLOWS[int(rng.integers(len(_CHAOTIC_FLOWS)))] | |
| traj = self._chaotic_flow(rng, system, n_int) | |
| if traj is None or not np.isfinite(traj).all(): | |
| return None | |
| if n_int != length: | |
| xp = np.linspace(0.0, 1.0, n_int) | |
| xq = np.linspace(0.0, 1.0, length) | |
| traj = np.interp(xq, xp, traj) | |
| sd = float(traj.std()) | |
| if sd > 1e-12: | |
| traj = traj + rng.normal(0.0, 0.02 * sd, size=length) | |
| return traj | |
| # ββ long-memory / fractional core ββββββββββββββββββββββββββββββββββββββββ | |
| def _spectral_noise(self, rng: np.random.Generator, n: int) -> np.ndarray: | |
| """Power-law (1/f^Ξ²) colored noise via FFT synthesis. Ξ²β[-1,3] spans | |
| anti-persistent (blue) β pink β red/Brownian spectra.""" | |
| beta = rng.uniform(-1.0, 3.0) | |
| m = n // 2 + 1 | |
| f = np.arange(m, dtype=np.float64) | |
| f[0] = 1.0 | |
| amp = f ** (-beta / 2.0) | |
| amp[0] = 0.0 # zero DC β zero mean | |
| phases = rng.uniform(0.0, 2.0 * np.pi, size=m) | |
| spec = amp * (np.cos(phases) + 1j * np.sin(phases)) | |
| y = np.fft.irfft(spec, n=n) | |
| if rng.random() < 0.35: # occasional integration β fBm-like non-stationarity | |
| y = np.cumsum(y - y.mean()) | |
| return y | |
| def _arfima(self, rng: np.random.Generator, n: int) -> np.ndarray: | |
| """ARFIMA(p,d,0): apply truncated fractional-differencing (Hosking) | |
| coefficients (1-L)^{-d} to (optionally fat-tailed) innovations, with an | |
| optional short AR factor for combined short+long memory.""" | |
| d = rng.uniform(0.1, 0.45) | |
| K = int(min(n, 1500)) | |
| if K >= 2: | |
| k = np.arange(1, K, dtype=np.float64) | |
| psi = np.concatenate(([1.0], np.cumprod((k - 1.0 + d) / k))) | |
| else: | |
| psi = np.array([1.0]) | |
| eps = self._draw_innovations(rng, n) | |
| y = lfilter(psi, [1.0], eps) | |
| if rng.random() < 0.5: | |
| p = int(rng.integers(1, 3)) | |
| ar = self._reflection_to_poly(rng, p, -0.7, 0.7) | |
| y = lfilter([1.0], ar, y) | |
| return y | |
| def _long_memory_core(self, rng: np.random.Generator, length: int) -> np.ndarray: | |
| if rng.random() < 0.5: | |
| return self._spectral_noise(rng, length) | |
| return self._arfima(rng, length) | |
| # ββ bilinear nonlinear-AR core βββββββββββββββββββββββββββββββββββββββββββ | |
| def _bilinear_core(self, rng: np.random.Generator, length: int): | |
| phi = float(rng.uniform(-0.6, 0.6)) | |
| b = float(rng.uniform(-0.4, 0.4)) | |
| theta = float(rng.uniform(-0.5, 0.5)) | |
| eps = self._draw_innovations(rng, length) | |
| y = np.empty(length, dtype=np.float64) | |
| yp = 0.0 | |
| ep = 0.0 | |
| for t in range(length): | |
| e = float(eps[t]) | |
| val = phi * yp + b * yp * ep + theta * ep + e | |
| if not math.isfinite(val) or abs(val) > 1.0e8: | |
| return None | |
| y[t] = val | |
| yp = val | |
| ep = e | |
| return y | |
| # ββ core dispatcher ββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def _core(self, rng: np.random.Generator, length: int) -> np.ndarray: | |
| kind = self._core_names[int(rng.choice(len(self._core_names), p=self._core_w))] | |
| x = None | |
| if kind == "chaotic": | |
| x = self._chaotic_core(rng, length) | |
| elif kind == "longmem": | |
| x = self._long_memory_core(rng, length) | |
| elif kind == "bilinear": | |
| x = self._bilinear_core(rng, length) | |
| else: | |
| x = self._regime_core(rng, length) | |
| # Deterministic fallback for any divergent / malformed core. | |
| if x is None or x.size != length or not np.isfinite(x).all(): | |
| x = self._regime_core(rng, length) | |
| return x | |
| # ββ additive compositional components (scaled to the core) βββββββββββββββ | |
| def _add_components(self, rng: np.random.Generator, x: np.ndarray, length: int) -> np.ndarray: | |
| t = np.arange(length, dtype=np.float64) | |
| scale = x.std() | |
| if scale <= 1e-12: | |
| scale = 1.0 | |
| # Nested-calendar seasonality (short period nested in ~7x/~30x multiples). | |
| if rng.random() < self._calendar_prob: | |
| base_period = rng.uniform(4.0, 48.0) | |
| for ratio in (1.0, 7.0, 30.0): | |
| period = base_period * ratio | |
| if period >= length * 1.5: | |
| continue | |
| amp = scale * rng.uniform(0.2, 1.2) / ratio | |
| phase = rng.uniform(0.0, 2.0 * np.pi) | |
| x = x + amp * np.sin(2.0 * np.pi * t / period + phase) | |
| # Smooth deterministic trend (linear + occasional curvature). | |
| if rng.random() < self._trend_prob: | |
| u = t / max(1.0, length - 1) | |
| x = x + scale * rng.normal(0.0, 1.0) * u | |
| if rng.random() < 0.4: | |
| x = x + scale * rng.normal(0.0, 0.7) * (u - 0.5) ** 2 | |
| # Structural level shifts. | |
| if rng.random() < self._level_shift_prob: | |
| for _ in range(int(rng.integers(1, 4))): | |
| at = int(rng.integers(length // 10, max(length // 10 + 1, 9 * length // 10))) | |
| x = x.copy() | |
| x[at:] += scale * rng.normal(0.0, 1.0) | |
| # Sparse spikes / outliers. | |
| if rng.random() < self._spike_prob: | |
| n_spikes = int(rng.integers(1, max(2, length // 200 + 2))) | |
| idx = rng.integers(0, length, size=n_spikes) | |
| x = x.copy() | |
| x[idx] += scale * rng.normal(0.0, 4.0, size=n_spikes) | |
| return x | |
| # ββ optional invertible nonlinear warp β non-Gaussian / positive marginals β | |
| def _maybe_warp(self, rng: np.random.Generator, x: np.ndarray) -> np.ndarray: | |
| if rng.random() >= self._warp_prob: | |
| return x | |
| mu, sd = x.mean(), x.std() | |
| z = (x - mu) / sd if sd > 1e-12 else x - mu | |
| kind = rng.integers(0, 4) | |
| if kind == 0: # tail compression | |
| return np.arcsinh(rng.uniform(0.5, 3.0) * z) | |
| if kind == 1: # log-normal-like positivity / multiplicative | |
| return np.exp(np.clip(rng.uniform(0.2, 1.0) * z, -10.0, 10.0)) | |
| if kind == 2: # softplus positivity (demand/count-like) | |
| return np.log1p(np.exp(np.clip(rng.uniform(0.5, 1.5) * z, -20.0, 20.0))) | |
| gamma = rng.uniform(1.5, 3.0) # signed power (peaky) | |
| return np.sign(z) * np.abs(z) ** gamma | |
| # ββ full single draw βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def _draw_one(self, rng: np.random.Generator, length: int) -> np.ndarray: | |
| x = self._core(rng, length) | |
| x = self._add_components(rng, x, length) | |
| x = self._maybe_warp(rng, x) | |
| return x | |
| # ββ scaling + sanitisation βββββββββββββββββββββββββββββββββββββββββββββββ | |
| def _finalize(self, rng: np.random.Generator, y: np.ndarray, length: int) -> np.ndarray: | |
| x = np.asarray(y, dtype=np.float64).ravel() | |
| if x.size != length: | |
| if x.size > length: | |
| x = x[:length] | |
| else: | |
| x = np.concatenate([x, np.full(length - x.size, x[-1] if x.size else 0.0)]) | |
| if not np.isfinite(x).all(): | |
| x = np.nan_to_num(x, nan=0.0, posinf=self._max_abs, neginf=-self._max_abs) | |
| std = x.std() | |
| if std > 1e-12: | |
| x = x / std * float(rng.lognormal(mean=0.0, sigma=1.2)) | |
| x = x + rng.normal(0.0, 3.0) | |
| if self._standardize: | |
| std2 = x.std() | |
| if std2 > 1e-12: | |
| x = (x - x.mean()) / std2 | |
| np.clip(x, -self._max_abs, self._max_abs, out=x) | |
| if not np.isfinite(x).all(): | |
| x = rng.standard_normal(length) | |
| return np.ascontiguousarray(x, dtype=np.float64) | |
| # ββ entrypoint βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def generate(self, n_series: int) -> Iterator[np.ndarray]: | |
| if n_series <= 0: | |
| return | |
| len_rng = np.random.default_rng(np.random.SeedSequence([self._seed, _LENGTH_STREAM_ID])) | |
| for i in range(n_series): | |
| length = int(len_rng.integers(self._min_len, self._max_len + 1)) | |
| rng = self._series_rng(i + 1) | |
| x = self._draw_one(rng, length) | |
| # TSMixup: convex-combine two independent draws (Chronos augmentation). | |
| if rng.random() < self._mixup_prob: | |
| x2 = self._draw_one(rng, length) | |
| w = float(rng.uniform(0.2, 0.8)) | |
| sx, s2 = x.std() or 1.0, x2.std() or 1.0 | |
| x = w * (x / sx) + (1.0 - w) * (x2 / s2) | |
| yield self._finalize(rng, x, length) | |