Buckets:
| #!/usr/bin/env python3 | |
| """Generate exp6_numpy_drvhead.py from the byte-identical exp5 baseline. | |
| exp6 replaces FastF1's `add_driver_ahead()` (the ~130ms/lap hotspot in | |
| R.py/Rp.py's per-lap get_telemetry chain, ~75% of per-lap time) with an exact | |
| numpy re-implementation: | |
| * per-session precomputed arrays: per-driver laps (LapNumber, | |
| LapStartTime, Time) and session telemetry (SessionTime, Speed, Date) | |
| * per lap: the ~19 driver "distance since lap start" integrations are done | |
| with numpy cumsum over the exact same float64 sequence pandas computes, | |
| * the pandas "outer join on SessionTime" + per-row driver-ahead argmin | |
| matrix math is reproduced element-for-element. | |
| All arithmetic mirrors fastf1.core exactly (int64 ns for times, `/1e9` for | |
| seconds, `Speed / 3.6`, float64 cumsum), so the produced values are | |
| bit-identical. A fallback to the original `add_driver_ahead` runs on any | |
| unforeseen data shape, and `FF1_COMPARE_FAST=1` enables a live comparison | |
| (in-process assert) for validating runs. | |
| Byte-identity of the final JSON outputs is still verified independently with | |
| compare_outputs.py against the R.py reference outputs. | |
| """ | |
| import os | |
| HERE = os.path.dirname(os.path.abspath(__file__)) | |
| FAST_DRV = r''' | |
| # --------------------------------------------------------------------------- | |
| # Fast driver-ahead (exact numpy replica of fastf1 add_driver_ahead) | |
| # --------------------------------------------------------------------------- | |
| _FF1_COMPARE_FAST = os.environ.get("FF1_COMPARE_FAST", "") == "1" | |
| def _build_fast_drv_cache(session): | |
| """Precompute per-driver lookup arrays (called once per process).""" | |
| laps_tbl = session.laps | |
| cache = {} | |
| for drv in session.drivers: | |
| if drv not in session.car_data: | |
| continue | |
| lap_m = laps_tbl["DriverNumber"] == drv | |
| dl = laps_tbl[lap_m] | |
| if dl.empty: | |
| continue | |
| ln = np.asarray(dl["LapNumber"].to_numpy(), dtype=np.float64) | |
| lst = dl["LapStartTime"].to_numpy() | |
| lt = dl["Time"].to_numpy() | |
| lst_ns = np.asarray(lst.view("int64")).copy() | |
| lt_ns = np.asarray(lt.view("int64")).copy() | |
| lst_ns[pd.isna(lst)] = np.iinfo(np.int64).min | |
| lt_ns[pd.isna(lt)] = np.iinfo(np.int64).min | |
| order = np.argsort(ln, kind="stable") | |
| ln = ln[order] | |
| lst_ns = lst_ns[order] | |
| lt_ns = lt_ns[order] | |
| cd = session.car_data[drv] | |
| st = np.asarray(cd["SessionTime"].to_numpy().view("int64")).copy() | |
| sp = np.asarray(cd["Speed"].to_numpy()).copy() | |
| cache[drv] = (ln, lst_ns, lt_ns, st, sp) | |
| return cache | |
| _FAST_DCACHE = None | |
| def _fast_driver_ahead(car_data): | |
| """Exact replacement for `car_data.iloc[1:-1].add_driver_ahead()`. | |
| Returns a DataFrame with the five columns the upstream merge consumes | |
| (`DriverAhead`, `DistanceToDriverAhead`, `Date`, `Time`, `SessionTime`) | |
| whose values are byte-identical to FastF1's. | |
| """ | |
| global _FAST_DCACHE | |
| d = car_data.iloc[1:-1] | |
| def _fallback(): | |
| return ( | |
| d.add_driver_ahead() | |
| .loc[:, ("DriverAhead", "DistanceToDriverAhead", "Date", "Time", "SessionTime")] | |
| ) | |
| if len(d) < 2: | |
| return _fallback() | |
| session = getattr(d, "session", None) | |
| if session is None: | |
| return _fallback() | |
| key = id(session) | |
| if _FAST_DCACHE is None or _FAST_DCACHE[0] != key: | |
| _FAST_DCACHE = (key, _build_fast_drv_cache(session)) | |
| cache = _FAST_DCACHE[1] | |
| t_start = int(d["SessionTime"].iloc[0].value) | |
| t_end = int(d["SessionTime"].iloc[-1].value) | |
| own = str(d.driver) | |
| own_laps = cache.get(own) | |
| if own_laps is None: | |
| return _fallback() | |
| own_ln, own_lst_ns, own_lt_ns, _s, _p = own_laps | |
| i_fl = np.searchsorted(own_lst_ns, t_start, side="right") - 1 | |
| if i_fl < 0: | |
| return _fallback() | |
| first_lap_number = float(own_ln[i_fl]) | |
| combined = {} | |
| for drv in session.drivers: | |
| if drv not in cache: | |
| continue | |
| ln, lst_ns, lt_ns, st, sp = cache[drv] | |
| n_lap = len(ln) | |
| if n_lap == 0: | |
| continue | |
| # ---- lap_n_before (mirrors fastf1 semantics) ---- | |
| j_before = np.searchsorted(lst_ns, t_start, side="right") - 1 | |
| if j_before >= 0: | |
| lap_before = float(ln[j_before]) | |
| if lap_before < first_lap_number: | |
| lap_before += 1 | |
| else: | |
| lap_before = float(np.min(ln)) | |
| # ---- lap_n_after ---- | |
| j_after = np.searchsorted(lt_ns, t_end, side="left") | |
| if j_after < n_lap: | |
| lap_after = float(ln[j_after]) | |
| else: | |
| lap_after = float(np.max(ln)) | |
| # ---- relevant-laps selection with fastf1's pad loop ---- | |
| pad_before = 0 | |
| pad_after = 0 | |
| relevant = None | |
| while True: | |
| mask = (ln >= (lap_before - pad_before)) & (ln <= (lap_after + pad_after)) | |
| m_idx = np.where(mask)[0] | |
| if len(m_idx) == 0: | |
| relevant = None | |
| break | |
| if pad_before >= 1 or pad_after >= 1: | |
| relevant = (float(ln[m_idx[0]]), float(ln[m_idx[-1]])) | |
| _warn_once("pad rep") | |
| break | |
| if lst_ns[m_idx[-1]] == np.iinfo(np.int64).min: | |
| pad_before += 1 | |
| continue | |
| if lt_ns[m_idx[0]] == np.iinfo(np.int64).min: | |
| pad_after += 1 | |
| continue | |
| relevant = (float(ln[m_idx[0]]), float(ln[m_idx[-1]])) | |
| break | |
| if relevant is None: | |
| continue | |
| lap_lo, lap_hi = relevant | |
| i_before = int(np.where(ln == lap_lo)[0][0]) | |
| i_after = int(np.where(ln == lap_hi)[0][0]) | |
| win_start = lst_ns[i_before] | |
| win_end = lt_ns[i_after] | |
| lo_c = np.searchsorted(st, win_start, side="left") | |
| hi_c = np.searchsorted(st, win_end, side="right") | |
| n_win = hi_c - lo_c | |
| if n_win <= 0: | |
| continue | |
| # Mirror fastf1 calculate_differential_distance EXACTLY: | |
| # dt = Time.dt.total_seconds().diff() with Time = SessionTime - win_start | |
| # = (int64 ns shifted to lap start) / 1e9 then float diff | |
| # dt[0] = Time[0].total_seconds() = int64(ns_0 - win_start)/1e9 | |
| t_sec = (st[lo_c:hi_c] - win_start).astype(np.float64) / 1e9 | |
| dt = np.empty(n_win, dtype=np.float64) | |
| dt[0] = t_sec[0] | |
| if n_win > 1: | |
| dt[1:] = t_sec[1:] - t_sec[:-1] | |
| dist = np.cumsum(sp[lo_c:hi_c] / np.float64(3.6) * dt) | |
| lo2 = np.searchsorted(st[lo_c:hi_c], t_start, side="left") | |
| hi2 = np.searchsorted(st[lo_c:hi_c], t_end, side="right") | |
| if hi2 <= lo2: | |
| continue | |
| combined[drv] = (st[lo_c + lo2:lo_c + hi2].copy(), | |
| dist[lo2:hi2].copy()) | |
| keys = [drv for drv in session.drivers if drv in combined] | |
| if not keys or own not in combined: | |
| return _fallback() | |
| own_t = combined[own][0] | |
| if len(keys) == 1: | |
| uniq = own_t | |
| else: | |
| uniq = np.unique(np.concatenate([combined[k][0] for k in keys])) | |
| if uniq.shape != own_t.shape or not np.array_equal(uniq, own_t): | |
| return _fallback() | |
| n_u = len(uniq) | |
| mat = np.full((n_u, len(keys)), np.nan) | |
| for jj, k in enumerate(keys): | |
| tt, vv = combined[k] | |
| idx = np.searchsorted(uniq, tt) | |
| mat[idx, jj] = vv | |
| own_j = keys.index(own) | |
| own_col = mat[:, own_j] | |
| other_idx = [jj for jj in range(len(keys)) if jj != own_j] | |
| drv_map = [keys[jj] for jj in other_idx] | |
| other_dst = mat[:, other_idx].copy() | |
| # np.diff(other_dst, axis=0, prepend=other_dst[0].reshape(1,-1)) == 0 | |
| d_all = np.empty_like(other_dst) | |
| d_all[0] = 0 | |
| if other_dst.shape[0] > 1: | |
| d_all[1:] = other_dst[1:] - other_dst[:-1] | |
| other_dst[d_all == 0] = np.nan | |
| own2 = np.repeat(own_col.reshape((-1, 1)), other_dst.shape[1], axis=1) | |
| delta_dst = other_dst - own2 | |
| delta_dst[np.isnan(delta_dst)] = np.inf | |
| delta_dst[delta_dst < 0] = np.inf | |
| index_ahead = np.argmin(delta_dst, axis=1) | |
| da = np.array([drv_map[i] for i in index_ahead]) | |
| all_inf = np.all(delta_dst == np.inf, axis=1) | |
| da[all_inf] = "" | |
| dist_arr = delta_dst[np.arange(n_u), index_ahead] | |
| dist_arr[all_inf] = np.nan | |
| out = pd.DataFrame( | |
| { | |
| "DriverAhead": da, | |
| "DistanceToDriverAhead": dist_arr, | |
| "Date": d["Date"].to_numpy(), | |
| "Time": d["Time"].to_numpy(), | |
| "SessionTime": d["SessionTime"].to_numpy(), | |
| }, | |
| index=d.index, | |
| ) | |
| return out | |
| _warned_once = set() | |
| def _warn_once(msg): | |
| if msg not in _warned_once: | |
| _warned_once.add(msg) | |
| logger.warning(f"fast driver_ahead: {msg}") | |
| ''' | |
| FAST_DRV = FAST_DRV # used by generate() | |
| def generate(): | |
| with open(os.path.join(HERE, "exp5_saturate.py"), "r", encoding="utf-8") as f: | |
| src = f.read() | |
| anchor = "def _lap_telemetry_or_none(selected) -> Optional[pd.DataFrame]:" | |
| assert anchor in src | |
| old_tel = ''' # Same pipeline as fastf1.core.Laps.get_telemetry / Lap.get_telemetry. | |
| drv_ahead = ( | |
| car_data.iloc[1:-1] | |
| .add_driver_ahead() | |
| .loc[ | |
| :, | |
| ( | |
| "DriverAhead", | |
| "DistanceToDriverAhead", | |
| "Date", | |
| "Time", | |
| "SessionTime", | |
| ), | |
| ] | |
| )''' | |
| new_tel = ''' # Same pipeline as fastf1.core.Laps.get_telemetry / Lap.get_telemetry, | |
| # but with a numpy-accelerated add_driver_ahead. | |
| drv_ahead = _fast_driver_ahead(car_data) | |
| if not hasattr(drv_ahead, "fill_missing"): | |
| from fastf1.core import Telemetry | |
| drv_ahead = Telemetry( | |
| drv_ahead, session=car_data.session, driver=car_data.driver | |
| ) | |
| if _FF1_COMPARE_FAST and len(drv_ahead): | |
| _ref = ( | |
| car_data.iloc[1:-1] | |
| .add_driver_ahead() | |
| .loc[:, ("DriverAhead", "DistanceToDriverAhead", "Date", "Time", "SessionTime")] | |
| ) | |
| for _c in ("DriverAhead", "DistanceToDriverAhead"): | |
| _a = drv_ahead[_c].to_numpy() | |
| _b = _ref[_c].to_numpy() | |
| # equal_nan is only valid for float arrays (DriverAhead is | |
| # object-dtype strings) - compare accordingly | |
| if _a.shape != _b.shape or not np.array_equal( | |
| _a, _b, equal_nan=_a.dtype.kind in "fc" | |
| ): | |
| raise AssertionError(f"fast driver_ahead mismatch on {_c}")''' | |
| assert old_tel in src | |
| src = src.replace(old_tel, new_tel, 1) | |
| # insert the fast implementation before the class | |
| anchor2 = "class SeasonSessionExtractor:" | |
| assert anchor2 in src | |
| src = src.replace(anchor2, FAST_DRV + "\n\n\n" + anchor2, 1) | |
| dst = os.path.join(HERE, "exp6_numpy_drvhead.py") | |
| with open(dst, "w", encoding="utf-8") as f: | |
| f.write(src) | |
| print("wrote", dst) | |
| if __name__ == "__main__": | |
| generate() |
Xet Storage Details
- Size:
- 11.2 kB
- Xet hash:
- 4ff10c48e593d84fc539879b130672f61611697fa5823864b128282284e5ddac
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.