tracinginsights's picture
download
raw
19.5 kB
#!/usr/bin/env python3
"""Generate exp7_numpy_merge.py from the byte-identical exp6 baseline.
exp7 replaces the remaining per-lap pandas machinery in the get_telemetry
chain - `add_distance`/`add_relative_distance`, the two `merge_channels`
calls (car+driver-ahead and pos+car) and the `slice_by_lap(interpolate_edges)`
edge merge - with exact numpy/scipy replicas:
* `add_distance`: the pandas dt sequence is reproduced exactly, including
the scalar ``Timedelta.total_seconds()`` semantics for the first row
(``days*86400 + seconds + microseconds/1e6`` with floor-based components,
nanoseconds dropped) which differs from the vectorized ``dt.total_seconds``
used for the diff.
* `merge_channels(frequency='original')`: outer join on the int64-ns Date
union with per-channel interpolation dispatch matching pandas exactly:
- 'index' channels (Speed, RPM, Throttle, DistanceToDriverAhead):
np.interp on int64 ns dates (limit_direction='both', no preservation)
- 'quadratic' channels (Distance, RelativeDistance, X, Y, Z): scipy
interp1d(kind='quadratic', fill_value='extrapolate') with pandas'
leading-NaN preservation (limit_direction='forward')
- 'discrete' channels (nGear, Brake, DRS, DriverAhead, Status):
ffill().ffill().bfill()
- Source: fillna('interpolation')
* `slice_by_lap(interpolate_edges=True)`: the 2-row edges merge plus the
[start, end] mask and Time = SessionTime - start_time shift.
All arithmetic mirrors fastf1.core/pandas exactly, so the produced values are
bit-identical. `FF1_COMPARE_FAST=1` enables a live per-lap comparison against
the exp6 pandas path (in-process assert) for validating runs.
Byte-identity of the final JSON outputs is still verified independently with
compare_outputs.py against the R.py reference outputs.
"""
import os
HERE = os.path.dirname(os.path.abspath(__file__))
FAST_MERGE = r'''
# ---------------------------------------------------------------------------
# Fast merge/fill/slice (exact numpy/scipy replicas of the pandas chain)
# ---------------------------------------------------------------------------
from scipy.interpolate import interp1d as _scipy_interp1d
_TELEMETRY_CH = None
def _get_channel_registry():
global _TELEMETRY_CH
if _TELEMETRY_CH is None:
from fastf1.core import Telemetry
_TELEMETRY_CH = Telemetry._CHANNELS
return _TELEMETRY_CH
def _scalar_total_seconds(ns):
"""Scalar ``pd.Timedelta.total_seconds()``: days*86400 + seconds +
microseconds/1e6 with floor-based components (nanoseconds dropped)."""
days = ns // (86400 * 10**9)
rem = ns - days * (86400 * 10**9)
secs = (rem // 10**9) % 86400
us = (rem // 10**3) % 10**6
return float(days * 86400 + secs) + us / 1e6
def _fast_add_distance(car_data):
"""Replicate `car_data.add_distance().add_relative_distance()` exactly.
Returns (Distance, RelativeDistance) float64 arrays.
"""
Time = car_data["Time"].to_numpy()
ns = Time.view("int64")
ts = ns.astype(np.float64) / 1e9
dt = np.empty(len(ts), dtype=np.float64)
dt[0] = _scalar_total_seconds(int(ns[0]))
if len(ts) > 1:
dt[1:] = ts[1:] - ts[:-1]
ds = car_data["Speed"].to_numpy() / 3.6 * dt
dist = np.cumsum(ds)
rel = dist / dist[-1]
return dist, rel
def _ffill_f(arr):
out = arr.copy()
valid = ~np.isnan(out)
if not valid.any():
return out
last = np.full(len(out), -1, dtype=np.int64)
last[valid] = np.arange(len(out))[valid]
np.maximum.accumulate(last, out=last)
m = last >= 0
out[m] = out[last[m]]
return out
def _bfill_f(arr):
out = arr.copy()
valid = ~np.isnan(out)
if not valid.any():
return out
nxt = np.full(len(out), -1, dtype=np.int64)
cur = -1
for i in range(len(out) - 1, -1, -1):
if valid[i]:
cur = i
nxt[i] = cur
m = nxt >= 0
out[m] = out[nxt[m]]
return out
def _is_missing_obj(v):
return v is None or (isinstance(v, float) and np.isnan(v)) or (
isinstance(v, np.floating) and np.isnan(v)
)
def _ffill_obj(arr):
out = arr.copy()
valid = np.array([not _is_missing_obj(v) for v in out])
if not valid.any():
return out
last = np.full(len(out), -1, dtype=np.int64)
last[valid] = np.arange(len(out))[valid]
np.maximum.accumulate(last, out=last)
m = last >= 0
out[m] = out[last[m]]
return out
def _bfill_obj(arr):
out = arr.copy()
valid = np.array([not _is_missing_obj(v) for v in out])
if not valid.any():
return out
nxt = np.full(len(out), -1, dtype=np.int64)
cur = -1
for i in range(len(out) - 1, -1, -1):
if valid[i]:
cur = i
nxt[i] = cur
m = nxt >= 0
out[m] = out[nxt[m]]
return out
def _fill_discrete(values):
"""pandas `.ffill().ffill().bfill()` on a Series (float or object)."""
if values.dtype.kind == "O":
return _bfill_obj(_ffill_obj(values))
return _bfill_f(_ffill_f(values))
def _fill_channel(dates_i8, values, ch_name):
"""Replicate pandas' fill_missing dispatch for one channel.
dates_i8: int64 ns timestamps of the merged frame (sorted unique).
values: float64 or object array with NaN for missing.
"""
meta = _get_channel_registry().get(ch_name)
if meta is None:
return values
sig = meta["type"]
method = meta.get("method")
if sig == "discrete":
return _fill_discrete(values)
if ch_name == "Source":
out = values.copy()
if out.dtype.kind == "O":
m = np.array([_is_missing_obj(v) for v in out])
else:
m = np.isnan(out)
out[m] = "interpolation"
return out
if method in ("index", "values"):
if values.dtype.kind != "f":
return values
valid = ~np.isnan(values)
out = values.copy()
if valid.any() and not valid.all():
out[~valid] = np.interp(
dates_i8[~valid], dates_i8[valid], values[valid]
)
return out
if method == "quadratic":
if values.dtype.kind != "f":
return values
valid = ~np.isnan(values)
out = values.copy()
if valid.any() and not valid.all():
terp = _scipy_interp1d(
dates_i8[valid],
values[valid],
kind="quadratic",
fill_value="extrapolate",
bounds_error=False,
)
out[~valid] = terp(dates_i8[~valid])
first = int(np.argmax(valid))
out[:first] = np.nan # pandas limit_direction='forward'
return out
return values
def _merge_fill_cols(all_dates, frames, ch_names):
"""Outer-merge columns from frames and run fill_missing.
frames: list of (dates_i8, {col: values}) in merge priority order
(other first, then data) - later frames overwrite at shared dates,
mirroring `other.merge(data[...], how='outer')` + `update`.
"""
n = len(all_dates)
out = {}
for col in ch_names:
src = None
for _dates, arrays in frames:
if col in arrays:
src = arrays[col]
break
if src is None:
arr = np.full(n, np.nan)
elif src.dtype.kind == "O":
arr = np.full(n, np.nan, dtype=object)
else:
arr = np.full(n, np.nan)
for dates, arrays in frames:
if col in arrays:
idx = np.searchsorted(all_dates, dates)
arr[idx] = arrays[col]
out[col] = _fill_channel(all_dates, arr, col)
return out
def _lap_telemetry_sources(selected):
"""Return (pos_data, car_data) slices for the lap, or None when either
cannot be merged (FastF1's get_telemetry() would raise the same way)."""
pos_data = selected.get_pos_data(pad=1, pad_side="both")
if pos_data is None or pos_data.empty or "Date" not in pos_data.columns:
return None
car_data = selected.get_car_data(pad=1, pad_side="both")
if (
car_data is None
or car_data.empty
or "Date" not in car_data.columns
or len(car_data) < 3
):
return None
return pos_data, car_data
def _fast_lap_telemetry_or_none(selected, pos_data=None, car_data=None):
"""Build lap telemetry with numpy/scipy merge+fill+slice replicas.
pos_data/car_data (the get_pos_data/get_car_data(pad=1) slices) may be
passed in to avoid re-fetching them. Returns a DataFrame byte-identical
to the exp6 pandas path, or None when the pandas path would return None.
"""
if pos_data is None:
pos_data = selected.get_pos_data(pad=1, pad_side="both")
if pos_data is None or pos_data.empty or "Date" not in pos_data.columns:
return None
if car_data is None:
car_data = selected.get_car_data(pad=1, pad_side="both")
if (
car_data is None
or car_data.empty
or "Date" not in car_data.columns
or len(car_data) < 3
):
return None
# ---- merge1: car_data (data) + drv_ahead (other) -------------------
drv_ahead = _fast_driver_ahead(car_data)
if not hasattr(drv_ahead, "fill_missing"):
from fastf1.core import Telemetry
drv_ahead = Telemetry(
drv_ahead, session=car_data.session, driver=car_data.driver
)
dist, rel = _fast_add_distance(car_data)
car_dates = car_data["Date"].to_numpy().view("i8")
drv_dates = drv_ahead["Date"].to_numpy().view("i8")
m1_dates = np.unique(np.concatenate([drv_dates, car_dates]))
car_arrays = {c: car_data[c].to_numpy() for c in car_data.columns}
car_arrays["Distance"] = dist
car_arrays["RelativeDistance"] = rel
drv_arrays = {c: drv_ahead[c].to_numpy() for c in drv_ahead.columns}
m1_cols = [
"DriverAhead", "DistanceToDriverAhead", "Time", "SessionTime",
"RPM", "Speed", "nGear", "Throttle", "Brake", "DRS", "Source",
"Distance", "RelativeDistance",
]
m1 = _merge_fill_cols(
m1_dates, ((drv_dates, drv_arrays), (car_dates, car_arrays)), m1_cols
)
# ---- merge2: pos_data (data) + m1 (other) --------------------------
pos_dates = pos_data["Date"].to_numpy().view("i8")
m2_dates = np.unique(np.concatenate([m1_dates, pos_dates]))
pos_arrays = {c: pos_data[c].to_numpy() for c in pos_data.columns}
m2_cols = [
"DriverAhead", "DistanceToDriverAhead", "Time", "SessionTime",
"RPM", "Speed", "nGear", "Throttle", "Brake", "DRS", "Source",
"Distance", "RelativeDistance", "Status", "X", "Y", "Z",
]
m2 = _merge_fill_cols(
m2_dates, ((m1_dates, m1), (pos_dates, pos_arrays)), m2_cols
)
# ---- slice_by_lap(interpolate_edges=True) --------------------------
session = car_data.session
t0 = int(session.t0_date.value)
start = int(selected["LapStartTime"].iloc[0].value)
end = int(selected["Time"].iloc[0].value)
edge_dates = np.array([start + t0, end + t0], dtype=np.int64)
m3_dates = np.unique(np.concatenate([m2_dates, edge_dates]))
n3 = len(m3_dates)
m3 = {}
for col in m2_cols:
src = m2[col]
if src.dtype.kind == "O":
arr = np.full(n3, np.nan, dtype=object)
else:
arr = np.full(n3, np.nan)
idx = np.searchsorted(m3_dates, m2_dates)
arr[idx] = m2[col]
m3[col] = _fill_channel(m3_dates, arr, col)
st_ns = m3_dates - t0
time_ns = st_ns - start
sel = (st_ns >= start) & (st_ns <= end)
# build the output DataFrame with the same values/dtypes the pandas
# path produces (only the columns consumed downstream)
col_dtypes = {}
for col in m2_cols:
if col in car_data.columns:
col_dtypes[col] = car_data[col].dtype
elif col in pos_data.columns:
col_dtypes[col] = pos_data[col].dtype
elif col in drv_ahead.columns:
col_dtypes[col] = drv_ahead[col].dtype
else:
col_dtypes[col] = np.dtype("float64")
data = {}
data["Date"] = pd.to_datetime(m3_dates[sel])
data["Time"] = pd.to_timedelta(time_ns[sel])
for col in m2_cols:
if col in ("Time", "SessionTime"):
continue
v = m3[col][sel]
dt = col_dtypes[col]
if dt.kind == "O":
data[col] = np.asarray(v, dtype=object)
elif dt.kind == "b":
data[col] = v.astype(bool)
elif dt.kind in "iu":
data[col] = v.astype(np.int64)
else:
data[col] = v.astype(np.float64)
df = pd.DataFrame(data)
df["SessionTime"] = pd.to_timedelta(st_ns[sel])
return df
'''
def generate():
with open(os.path.join(HERE, "exp6_numpy_drvhead.py"), "r", encoding="utf-8") as f:
src = f.read()
# 1) insert the fast merge/fill/slice helpers before the class
anchor = "class SeasonSessionExtractor:"
assert anchor in src
src = src.replace(anchor, FAST_MERGE + "\n\n\n" + anchor, 1)
# 2) replace the per-lap telemetry builder with the fast replica that
# optionally asserts equality against the pandas path
old_tel = ''' # Same pipeline as fastf1.core.Laps.get_telemetry / Lap.get_telemetry,
# but with a numpy-accelerated add_driver_ahead.
drv_ahead = _fast_driver_ahead(car_data)
if not hasattr(drv_ahead, "fill_missing"):
from fastf1.core import Telemetry
drv_ahead = Telemetry(
drv_ahead, session=car_data.session, driver=car_data.driver
)
if _FF1_COMPARE_FAST and len(drv_ahead):
_ref = (
car_data.iloc[1:-1]
.add_driver_ahead()
.loc[:, ("DriverAhead", "DistanceToDriverAhead", "Date", "Time", "SessionTime")]
)
for _c in ("DriverAhead", "DistanceToDriverAhead"):
_a = drv_ahead[_c].to_numpy()
_b = _ref[_c].to_numpy()
# equal_nan is only valid for float arrays (DriverAhead is
# object-dtype strings) - compare accordingly
if _a.shape != _b.shape or not np.array_equal(
_a, _b, equal_nan=_a.dtype.kind in "fc"
):
raise AssertionError(f"fast driver_ahead mismatch on {_c}")
car_data = car_data.add_distance().add_relative_distance()
car_data = car_data.merge_channels(drv_ahead, frequency=None)
merged = pos_data.merge_channels(car_data, frequency=None)
return merged.slice_by_lap(selected, interpolate_edges=True)'''
new_tel = ''' # Same pipeline as fastf1.core.Laps.get_telemetry / Lap.get_telemetry,
# with numpy/scipy replicas for add_distance, both merge_channels
# calls and the slice edge-merge.
if _FF1_COMPARE_FAST:
from fastf1.core import Telemetry
_drv_ahead = _fast_driver_ahead(car_data)
if not hasattr(_drv_ahead, "fill_missing"):
_drv_ahead = Telemetry(
_drv_ahead, session=car_data.session, driver=car_data.driver
)
_ref = (
pos_data.merge_channels(
car_data.add_distance()
.add_relative_distance()
.merge_channels(_drv_ahead, frequency=None),
frequency=None,
).slice_by_lap(selected, interpolate_edges=True)
)
_fast = _fast_lap_telemetry_or_none(
selected, pos_data=pos_data, car_data=car_data
)
if _ref is None or _ref.empty:
if not (_fast is None or _fast.empty):
raise AssertionError("fast path produced data, pandas None")
return _fast
if _fast is None:
raise AssertionError("fast path None, pandas produced data")
if len(_ref) != len(_fast):
raise AssertionError(f"row count mismatch: {len(_ref)} vs {len(_fast)}")
for _c in _ref.columns:
_a = _ref[_c].to_numpy()
_b = _fast[_c].to_numpy()
if _a.dtype.kind == "O" or _b.dtype.kind == "O":
def _norm(x):
return None if _is_missing_obj(x) else x
if not all(_norm(x) == _norm(y) for x, y in zip(_a, _b)):
raise AssertionError(f"fast merge mismatch on {_c}")
elif not np.array_equal(_a, _b, equal_nan=True):
raise AssertionError(f"fast merge mismatch on {_c}")
return _fast
return _fast_lap_telemetry_or_none(
selected, pos_data=pos_data, car_data=car_data
)'''
assert old_tel in src, "exp6 _lap_telemetry_or_none body not found"
src = src.replace(old_tel, new_tel, 1)
# 2b) reuse the shared source pre-check inside _lap_telemetry_or_none.
old_pre = ''' pos_data = selected.get_pos_data(pad=1, pad_side="both")
if pos_data is None or pos_data.empty or "Date" not in pos_data.columns:
return None
car_data = selected.get_car_data(pad=1, pad_side="both")
if (
car_data is None
or car_data.empty
or "Date" not in car_data.columns
or len(car_data) < 3
):
return None
'''
new_pre = ''' sources = _lap_telemetry_sources(selected)
if sources is None:
return None
pos_data, car_data = sources
'''
assert old_pre in src, "exp6 _lap_telemetry_or_none pre-check not found"
src = src.replace(old_pre, new_pre, 1)
# 3) when the fast path cannot produce lap telemetry, distinguish a
# genuinely missing pos/car data slice (R.py's get_telemetry() raises
# the same way -> skip cheaply, keeping the pre-check fast) from an
# internal fast-path edge case on usable data (fall back to R.py's
# exact pandas `get_telemetry()` path so the parallel variant never
# generates fewer tel files than R.py).
old_skip = ''' telemetry = _lap_telemetry_or_none(selected)
if telemetry is None or telemetry.empty:
return False
'''
new_skip = ''' telemetry = _lap_telemetry_or_none(selected)
if telemetry is None or telemetry.empty:
# The fast path could not build telemetry. When pos/car data
# is genuinely missing, R.py's get_telemetry() raises the same
# way, so skip cheaply (no file is produced by either variant).
# When the data IS usable, the fast path hit an internal edge
# case: fall back to R.py's exact pandas path so Rf.py never
# generates fewer tel files than R.py.
if _lap_telemetry_sources(selected) is None:
return False
telemetry = selected.get_telemetry()
'''
assert old_skip in src, "exp7 _process_single_lap fast-path skip not found"
src = src.replace(old_skip, new_skip, 1)
dst = os.path.join(HERE, "exp7_numpy_merge.py")
with open(dst, "w", encoding="utf-8") as f:
f.write(src)
rf_dst = os.path.join(os.path.dirname(HERE), "Rf.py")
with open(rf_dst, "w", encoding="utf-8") as f:
f.write(src)
print("wrote", dst)
print("wrote", rf_dst)
if __name__ == "__main__":
generate()

Xet Storage Details

Size:
19.5 kB
·
Xet hash:
05f9f03c5a7342e692e40a8c3c34e9d372a35f997f19ef5718b7b321b07335ea

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.