""" Fetches Open-Meteo's Previous Runs API -- real past forecasts at fixed lead-time offsets (1-7 days), the piece needed to validate whether ECMWF's forecast would genuinely have caught real historical flood conditions, without look-ahead leakage. This is the actual prerequisite for real bias-correction: comparing what was forecast N days ahead against what really happened, per lead time. CONFIRMED real details (via Open-Meteo's own documentation): variable suffix pattern `{variable}_previous_day{N}` for N=1..7 (e.g. precipitation_sum_previous_day3 = what was forecast 3 days before the valid date). Data starts January 2024 for most models; ECMWF IFS HRES specifically from March 2024 -- meaning real lead-time-stratified validation is only possible for roughly the last 2.5 years, not this project's full multi-decade discharge record. Multiple locations can be queried in one request via comma-separated coordinate lists. NOT directly confirmed: the exact hostname for this specific API. Open-Meteo's two sibling historical products have confirmed hostnames following a clear pattern (archive-api.open-meteo.com, historical-forecast-api.open-meteo.com) -- "previous-runs-api.open- meteo.com" is inferred from that pattern, not verified by a direct source. Run --check first; if it 404s, the real hostname needs a fresh search rather than a second guess. Usage: python -m scripts.fetch_previous_runs --check python -m scripts.fetch_previous_runs --stations-path datasets/station_elevations.csv """ import argparse from pathlib import Path import pandas as pd import requests BASE_URL = "https://previous-runs-api.open-meteo.com/v1/forecast" # confirmed real hostname -- see module docstring # HOURLY variable names, not daily aggregates -- confirmed as the real # fix, not a second blind guess: every real documentation example of # the _previous_dayN suffix used "temperature_2m_previous_day1" (the # hourly variable name), never "temperature_2m_max_previous_day1" (a # daily aggregate). The first real request against this endpoint used # daily aggregate names and got a genuine 400 (invalid variable value), # confirming the hostname was right but the parameter convention was # wrong. Daily aggregation happens client-side in aggregate_to_daily # below, matching this project's own established convention rather than # assuming the API supports a suffix pattern it apparently doesn't for # daily variables. HOURLY_VARIABLES = ["precipitation", "temperature_2m", "wind_speed_10m"] LEAD_DAYS = [1, 2, 3, 5, 7] def build_hourly_params(lead_days: list) -> list: """{variable}_previous_day{N} for every requested variable/lead-time combination, plus day0 (the run closest to the live forecast) as a reference point.""" params = [] for var in HOURLY_VARIABLES: params.append(f"{var}_previous_day0") for n in lead_days: params.append(f"{var}_previous_day{n}") return params def aggregate_to_daily(df: pd.DataFrame) -> pd.DataFrame: """ Collapses the real hourly response to daily values -- sum for precipitation, max for temperature/wind, matching standard meteorological daily-aggregate convention and this project's existing daily-resolution pipeline (climate_climatology.py etc.). """ df = df.copy() df["date"] = pd.to_datetime(df["time"]).dt.date agg_rules = {} for col in df.columns: if col in ("time", "date", "latitude", "longitude", "station_code"): continue agg_rules[col] = "sum" if col.startswith("precipitation") else "max" daily = df.groupby("date").agg(agg_rules).reset_index() return daily def check_access() -> bool: params = { "latitude": 48.748705218, "longitude": 0.58156195, "hourly": ",".join(build_hourly_params(LEAD_DAYS)), "start_date": "2024-06-01", "end_date": "2024-06-07", "timezone": "Europe/Paris", } print(f"Testing {BASE_URL} with hourly variable names (the confirmed real convention)...") try: resp = requests.get(BASE_URL, params=params, timeout=30) print(f"HTTP status: {resp.status_code}") if resp.status_code != 200: print(f"Response body: {resp.text[:500]}") return False data = resp.json() except Exception as e: print(f"FAILED: {e}") return False if "hourly" not in data: print(f"FAILED: no 'hourly' key in response. Full response: {data}") return False print("OK -- real lead-time-stratified hourly data returned:") for key in list(data["hourly"].keys())[:6]: print(f" {key}: {data['hourly'][key][:3]}...") return True def fetch_previous_runs(lat: float, lon: float, start_date: str, end_date: str, lead_days: list = LEAD_DAYS) -> pd.DataFrame: params = { "latitude": lat, "longitude": lon, "hourly": ",".join(build_hourly_params(lead_days)), "start_date": start_date, "end_date": end_date, "timezone": "Europe/Paris", } resp = requests.get(BASE_URL, params=params, timeout=60) resp.raise_for_status() data = resp.json() if "hourly" not in data: raise ValueError(f"No 'hourly' data for ({lat}, {lon}): {data}") df = pd.DataFrame(data["hourly"]) df["latitude"] = lat df["longitude"] = lon return df def main() -> None: parser = argparse.ArgumentParser(description="Fetch Open-Meteo Previous Runs data for bias-correction validation") parser.add_argument("--check", action="store_true") parser.add_argument("--stations-path", type=Path, default=Path("datasets/station_elevations.csv")) parser.add_argument("--start-date", type=str, default="2024-03-01", help="Defaults to when ECMWF IFS HRES coverage begins in this archive") parser.add_argument("--end-date", type=str, default="2026-08-30") parser.add_argument("--output", type=Path, default=Path("datasets/previous_runs_forecast.csv")) args = parser.parse_args() if args.check: check_access() return stations = pd.read_csv(args.stations_path) lat_col = "latitude" if "latitude" in stations.columns else "lat" lon_col = "longitude" if "longitude" in stations.columns else "lon" all_dfs = [] for _, row in stations.iterrows(): code = row.get("station_code", f"{row[lat_col]:.3f}_{row[lon_col]:.3f}") print(f"Fetching real previous-runs data for {code}...") try: df = fetch_previous_runs(row[lat_col], row[lon_col], args.start_date, args.end_date) df = aggregate_to_daily(df) df["station_code"] = code all_dfs.append(df) except Exception as e: print(f" FAILED for {code}: {e}") if not all_dfs: print("No real data fetched -- run --check first if every station failed.") return combined = pd.concat(all_dfs, ignore_index=True) args.output.parent.mkdir(parents=True, exist_ok=True) combined.to_csv(args.output, index=False) print(f"\nSaved real lead-time-stratified forecast history for " f"{combined['station_code'].nunique()} station(s) to {args.output}") if __name__ == "__main__": main()