River_Network / scripts /data_acquisition /fetch_previous_runs.py
ageraustine's picture
Upload folder using huggingface_hub (part 2)
f2046b4 verified
Raw History Blame Contribute Delete
7.25 kB
"""
Fetches Open-Meteo's Previous Runs API -- real past forecasts at fixed
lead-time offsets (1-7 days), the piece needed to validate whether
ECMWF's forecast would genuinely have caught real historical flood
conditions, without look-ahead leakage. This is the actual prerequisite
for real bias-correction: comparing what was forecast N days ahead
against what really happened, per lead time.
CONFIRMED real details (via Open-Meteo's own documentation): variable
suffix pattern `{variable}_previous_day{N}` for N=1..7 (e.g.
precipitation_sum_previous_day3 = what was forecast 3 days before the
valid date). Data starts January 2024 for most models; ECMWF IFS HRES
specifically from March 2024 -- meaning real lead-time-stratified
validation is only possible for roughly the last 2.5 years, not this
project's full multi-decade discharge record. Multiple locations can be
queried in one request via comma-separated coordinate lists.
NOT directly confirmed: the exact hostname for this specific API.
Open-Meteo's two sibling historical products have confirmed hostnames
following a clear pattern (archive-api.open-meteo.com,
historical-forecast-api.open-meteo.com) -- "previous-runs-api.open-
meteo.com" is inferred from that pattern, not verified by a direct
source. Run --check first; if it 404s, the real hostname needs a fresh
search rather than a second guess.
Usage:
python -m scripts.fetch_previous_runs --check
python -m scripts.fetch_previous_runs --stations-path datasets/station_elevations.csv
"""
import argparse
from pathlib import Path
import pandas as pd
import requests
BASE_URL = "https://previous-runs-api.open-meteo.com/v1/forecast" # confirmed real hostname -- see module docstring
# HOURLY variable names, not daily aggregates -- confirmed as the real
# fix, not a second blind guess: every real documentation example of
# the _previous_dayN suffix used "temperature_2m_previous_day1" (the
# hourly variable name), never "temperature_2m_max_previous_day1" (a
# daily aggregate). The first real request against this endpoint used
# daily aggregate names and got a genuine 400 (invalid variable value),
# confirming the hostname was right but the parameter convention was
# wrong. Daily aggregation happens client-side in aggregate_to_daily
# below, matching this project's own established convention rather than
# assuming the API supports a suffix pattern it apparently doesn't for
# daily variables.
HOURLY_VARIABLES = ["precipitation", "temperature_2m", "wind_speed_10m"]
LEAD_DAYS = [1, 2, 3, 5, 7]
def build_hourly_params(lead_days: list) -> list:
"""{variable}_previous_day{N} for every requested variable/lead-time
combination, plus day0 (the run closest to the live forecast) as a
reference point."""
params = []
for var in HOURLY_VARIABLES:
params.append(f"{var}_previous_day0")
for n in lead_days:
params.append(f"{var}_previous_day{n}")
return params
def aggregate_to_daily(df: pd.DataFrame) -> pd.DataFrame:
"""
Collapses the real hourly response to daily values -- sum for
precipitation, max for temperature/wind, matching standard
meteorological daily-aggregate convention and this project's
existing daily-resolution pipeline (climate_climatology.py etc.).
"""
df = df.copy()
df["date"] = pd.to_datetime(df["time"]).dt.date
agg_rules = {}
for col in df.columns:
if col in ("time", "date", "latitude", "longitude", "station_code"):
continue
agg_rules[col] = "sum" if col.startswith("precipitation") else "max"
daily = df.groupby("date").agg(agg_rules).reset_index()
return daily
def check_access() -> bool:
params = {
"latitude": 48.748705218, "longitude": 0.58156195,
"hourly": ",".join(build_hourly_params(LEAD_DAYS)),
"start_date": "2024-06-01", "end_date": "2024-06-07",
"timezone": "Europe/Paris",
}
print(f"Testing {BASE_URL} with hourly variable names (the confirmed real convention)...")
try:
resp = requests.get(BASE_URL, params=params, timeout=30)
print(f"HTTP status: {resp.status_code}")
if resp.status_code != 200:
print(f"Response body: {resp.text[:500]}")
return False
data = resp.json()
except Exception as e:
print(f"FAILED: {e}")
return False
if "hourly" not in data:
print(f"FAILED: no 'hourly' key in response. Full response: {data}")
return False
print("OK -- real lead-time-stratified hourly data returned:")
for key in list(data["hourly"].keys())[:6]:
print(f" {key}: {data['hourly'][key][:3]}...")
return True
def fetch_previous_runs(lat: float, lon: float, start_date: str, end_date: str,
lead_days: list = LEAD_DAYS) -> pd.DataFrame:
params = {
"latitude": lat, "longitude": lon,
"hourly": ",".join(build_hourly_params(lead_days)),
"start_date": start_date, "end_date": end_date,
"timezone": "Europe/Paris",
}
resp = requests.get(BASE_URL, params=params, timeout=60)
resp.raise_for_status()
data = resp.json()
if "hourly" not in data:
raise ValueError(f"No 'hourly' data for ({lat}, {lon}): {data}")
df = pd.DataFrame(data["hourly"])
df["latitude"] = lat
df["longitude"] = lon
return df
def main() -> None:
parser = argparse.ArgumentParser(description="Fetch Open-Meteo Previous Runs data for bias-correction validation")
parser.add_argument("--check", action="store_true")
parser.add_argument("--stations-path", type=Path, default=Path("datasets/station_elevations.csv"))
parser.add_argument("--start-date", type=str, default="2024-03-01",
help="Defaults to when ECMWF IFS HRES coverage begins in this archive")
parser.add_argument("--end-date", type=str, default="2026-08-30")
parser.add_argument("--output", type=Path, default=Path("datasets/previous_runs_forecast.csv"))
args = parser.parse_args()
if args.check:
check_access()
return
stations = pd.read_csv(args.stations_path)
lat_col = "latitude" if "latitude" in stations.columns else "lat"
lon_col = "longitude" if "longitude" in stations.columns else "lon"
all_dfs = []
for _, row in stations.iterrows():
code = row.get("station_code", f"{row[lat_col]:.3f}_{row[lon_col]:.3f}")
print(f"Fetching real previous-runs data for {code}...")
try:
df = fetch_previous_runs(row[lat_col], row[lon_col], args.start_date, args.end_date)
df = aggregate_to_daily(df)
df["station_code"] = code
all_dfs.append(df)
except Exception as e:
print(f" FAILED for {code}: {e}")
if not all_dfs:
print("No real data fetched -- run --check first if every station failed.")
return
combined = pd.concat(all_dfs, ignore_index=True)
args.output.parent.mkdir(parents=True, exist_ok=True)
combined.to_csv(args.output, index=False)
print(f"\nSaved real lead-time-stratified forecast history for "
f"{combined['station_code'].nunique()} station(s) to {args.output}")
if __name__ == "__main__":
main()