tracinginsights's picture
download
raw
11.2 kB
#!/usr/bin/env python3
"""Generate exp6_numpy_drvhead.py from the byte-identical exp5 baseline.
exp6 replaces FastF1's `add_driver_ahead()` (the ~130ms/lap hotspot in
R.py/Rp.py's per-lap get_telemetry chain, ~75% of per-lap time) with an exact
numpy re-implementation:
* per-session precomputed arrays: per-driver laps (LapNumber,
LapStartTime, Time) and session telemetry (SessionTime, Speed, Date)
* per lap: the ~19 driver "distance since lap start" integrations are done
with numpy cumsum over the exact same float64 sequence pandas computes,
* the pandas "outer join on SessionTime" + per-row driver-ahead argmin
matrix math is reproduced element-for-element.
All arithmetic mirrors fastf1.core exactly (int64 ns for times, `/1e9` for
seconds, `Speed / 3.6`, float64 cumsum), so the produced values are
bit-identical. A fallback to the original `add_driver_ahead` runs on any
unforeseen data shape, and `FF1_COMPARE_FAST=1` enables a live comparison
(in-process assert) for validating runs.
Byte-identity of the final JSON outputs is still verified independently with
compare_outputs.py against the R.py reference outputs.
"""
import os
HERE = os.path.dirname(os.path.abspath(__file__))
FAST_DRV = r'''
# ---------------------------------------------------------------------------
# Fast driver-ahead (exact numpy replica of fastf1 add_driver_ahead)
# ---------------------------------------------------------------------------
_FF1_COMPARE_FAST = os.environ.get("FF1_COMPARE_FAST", "") == "1"
def _build_fast_drv_cache(session):
"""Precompute per-driver lookup arrays (called once per process)."""
laps_tbl = session.laps
cache = {}
for drv in session.drivers:
if drv not in session.car_data:
continue
lap_m = laps_tbl["DriverNumber"] == drv
dl = laps_tbl[lap_m]
if dl.empty:
continue
ln = np.asarray(dl["LapNumber"].to_numpy(), dtype=np.float64)
lst = dl["LapStartTime"].to_numpy()
lt = dl["Time"].to_numpy()
lst_ns = np.asarray(lst.view("int64")).copy()
lt_ns = np.asarray(lt.view("int64")).copy()
lst_ns[pd.isna(lst)] = np.iinfo(np.int64).min
lt_ns[pd.isna(lt)] = np.iinfo(np.int64).min
order = np.argsort(ln, kind="stable")
ln = ln[order]
lst_ns = lst_ns[order]
lt_ns = lt_ns[order]
cd = session.car_data[drv]
st = np.asarray(cd["SessionTime"].to_numpy().view("int64")).copy()
sp = np.asarray(cd["Speed"].to_numpy()).copy()
cache[drv] = (ln, lst_ns, lt_ns, st, sp)
return cache
_FAST_DCACHE = None
def _fast_driver_ahead(car_data):
"""Exact replacement for `car_data.iloc[1:-1].add_driver_ahead()`.
Returns a DataFrame with the five columns the upstream merge consumes
(`DriverAhead`, `DistanceToDriverAhead`, `Date`, `Time`, `SessionTime`)
whose values are byte-identical to FastF1's.
"""
global _FAST_DCACHE
d = car_data.iloc[1:-1]
def _fallback():
return (
d.add_driver_ahead()
.loc[:, ("DriverAhead", "DistanceToDriverAhead", "Date", "Time", "SessionTime")]
)
if len(d) < 2:
return _fallback()
session = getattr(d, "session", None)
if session is None:
return _fallback()
key = id(session)
if _FAST_DCACHE is None or _FAST_DCACHE[0] != key:
_FAST_DCACHE = (key, _build_fast_drv_cache(session))
cache = _FAST_DCACHE[1]
t_start = int(d["SessionTime"].iloc[0].value)
t_end = int(d["SessionTime"].iloc[-1].value)
own = str(d.driver)
own_laps = cache.get(own)
if own_laps is None:
return _fallback()
own_ln, own_lst_ns, own_lt_ns, _s, _p = own_laps
i_fl = np.searchsorted(own_lst_ns, t_start, side="right") - 1
if i_fl < 0:
return _fallback()
first_lap_number = float(own_ln[i_fl])
combined = {}
for drv in session.drivers:
if drv not in cache:
continue
ln, lst_ns, lt_ns, st, sp = cache[drv]
n_lap = len(ln)
if n_lap == 0:
continue
# ---- lap_n_before (mirrors fastf1 semantics) ----
j_before = np.searchsorted(lst_ns, t_start, side="right") - 1
if j_before >= 0:
lap_before = float(ln[j_before])
if lap_before < first_lap_number:
lap_before += 1
else:
lap_before = float(np.min(ln))
# ---- lap_n_after ----
j_after = np.searchsorted(lt_ns, t_end, side="left")
if j_after < n_lap:
lap_after = float(ln[j_after])
else:
lap_after = float(np.max(ln))
# ---- relevant-laps selection with fastf1's pad loop ----
pad_before = 0
pad_after = 0
relevant = None
while True:
mask = (ln >= (lap_before - pad_before)) & (ln <= (lap_after + pad_after))
m_idx = np.where(mask)[0]
if len(m_idx) == 0:
relevant = None
break
if pad_before >= 1 or pad_after >= 1:
relevant = (float(ln[m_idx[0]]), float(ln[m_idx[-1]]))
_warn_once("pad rep")
break
if lst_ns[m_idx[-1]] == np.iinfo(np.int64).min:
pad_before += 1
continue
if lt_ns[m_idx[0]] == np.iinfo(np.int64).min:
pad_after += 1
continue
relevant = (float(ln[m_idx[0]]), float(ln[m_idx[-1]]))
break
if relevant is None:
continue
lap_lo, lap_hi = relevant
i_before = int(np.where(ln == lap_lo)[0][0])
i_after = int(np.where(ln == lap_hi)[0][0])
win_start = lst_ns[i_before]
win_end = lt_ns[i_after]
lo_c = np.searchsorted(st, win_start, side="left")
hi_c = np.searchsorted(st, win_end, side="right")
n_win = hi_c - lo_c
if n_win <= 0:
continue
# Mirror fastf1 calculate_differential_distance EXACTLY:
# dt = Time.dt.total_seconds().diff() with Time = SessionTime - win_start
# = (int64 ns shifted to lap start) / 1e9 then float diff
# dt[0] = Time[0].total_seconds() = int64(ns_0 - win_start)/1e9
t_sec = (st[lo_c:hi_c] - win_start).astype(np.float64) / 1e9
dt = np.empty(n_win, dtype=np.float64)
dt[0] = t_sec[0]
if n_win > 1:
dt[1:] = t_sec[1:] - t_sec[:-1]
dist = np.cumsum(sp[lo_c:hi_c] / np.float64(3.6) * dt)
lo2 = np.searchsorted(st[lo_c:hi_c], t_start, side="left")
hi2 = np.searchsorted(st[lo_c:hi_c], t_end, side="right")
if hi2 <= lo2:
continue
combined[drv] = (st[lo_c + lo2:lo_c + hi2].copy(),
dist[lo2:hi2].copy())
keys = [drv for drv in session.drivers if drv in combined]
if not keys or own not in combined:
return _fallback()
own_t = combined[own][0]
if len(keys) == 1:
uniq = own_t
else:
uniq = np.unique(np.concatenate([combined[k][0] for k in keys]))
if uniq.shape != own_t.shape or not np.array_equal(uniq, own_t):
return _fallback()
n_u = len(uniq)
mat = np.full((n_u, len(keys)), np.nan)
for jj, k in enumerate(keys):
tt, vv = combined[k]
idx = np.searchsorted(uniq, tt)
mat[idx, jj] = vv
own_j = keys.index(own)
own_col = mat[:, own_j]
other_idx = [jj for jj in range(len(keys)) if jj != own_j]
drv_map = [keys[jj] for jj in other_idx]
other_dst = mat[:, other_idx].copy()
# np.diff(other_dst, axis=0, prepend=other_dst[0].reshape(1,-1)) == 0
d_all = np.empty_like(other_dst)
d_all[0] = 0
if other_dst.shape[0] > 1:
d_all[1:] = other_dst[1:] - other_dst[:-1]
other_dst[d_all == 0] = np.nan
own2 = np.repeat(own_col.reshape((-1, 1)), other_dst.shape[1], axis=1)
delta_dst = other_dst - own2
delta_dst[np.isnan(delta_dst)] = np.inf
delta_dst[delta_dst < 0] = np.inf
index_ahead = np.argmin(delta_dst, axis=1)
da = np.array([drv_map[i] for i in index_ahead])
all_inf = np.all(delta_dst == np.inf, axis=1)
da[all_inf] = ""
dist_arr = delta_dst[np.arange(n_u), index_ahead]
dist_arr[all_inf] = np.nan
out = pd.DataFrame(
{
"DriverAhead": da,
"DistanceToDriverAhead": dist_arr,
"Date": d["Date"].to_numpy(),
"Time": d["Time"].to_numpy(),
"SessionTime": d["SessionTime"].to_numpy(),
},
index=d.index,
)
return out
_warned_once = set()
def _warn_once(msg):
if msg not in _warned_once:
_warned_once.add(msg)
logger.warning(f"fast driver_ahead: {msg}")
'''
FAST_DRV = FAST_DRV # used by generate()
def generate():
with open(os.path.join(HERE, "exp5_saturate.py"), "r", encoding="utf-8") as f:
src = f.read()
anchor = "def _lap_telemetry_or_none(selected) -> Optional[pd.DataFrame]:"
assert anchor in src
old_tel = ''' # Same pipeline as fastf1.core.Laps.get_telemetry / Lap.get_telemetry.
drv_ahead = (
car_data.iloc[1:-1]
.add_driver_ahead()
.loc[
:,
(
"DriverAhead",
"DistanceToDriverAhead",
"Date",
"Time",
"SessionTime",
),
]
)'''
new_tel = ''' # Same pipeline as fastf1.core.Laps.get_telemetry / Lap.get_telemetry,
# but with a numpy-accelerated add_driver_ahead.
drv_ahead = _fast_driver_ahead(car_data)
if not hasattr(drv_ahead, "fill_missing"):
from fastf1.core import Telemetry
drv_ahead = Telemetry(
drv_ahead, session=car_data.session, driver=car_data.driver
)
if _FF1_COMPARE_FAST and len(drv_ahead):
_ref = (
car_data.iloc[1:-1]
.add_driver_ahead()
.loc[:, ("DriverAhead", "DistanceToDriverAhead", "Date", "Time", "SessionTime")]
)
for _c in ("DriverAhead", "DistanceToDriverAhead"):
_a = drv_ahead[_c].to_numpy()
_b = _ref[_c].to_numpy()
# equal_nan is only valid for float arrays (DriverAhead is
# object-dtype strings) - compare accordingly
if _a.shape != _b.shape or not np.array_equal(
_a, _b, equal_nan=_a.dtype.kind in "fc"
):
raise AssertionError(f"fast driver_ahead mismatch on {_c}")'''
assert old_tel in src
src = src.replace(old_tel, new_tel, 1)
# insert the fast implementation before the class
anchor2 = "class SeasonSessionExtractor:"
assert anchor2 in src
src = src.replace(anchor2, FAST_DRV + "\n\n\n" + anchor2, 1)
dst = os.path.join(HERE, "exp6_numpy_drvhead.py")
with open(dst, "w", encoding="utf-8") as f:
f.write(src)
print("wrote", dst)
if __name__ == "__main__":
generate()

Xet Storage Details

Size:
11.2 kB
·
Xet hash:
4ff10c48e593d84fc539879b130672f61611697fa5823864b128282284e5ddac

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.