Spaces:
Running on Zero
Running on Zero
File size: 7,253 Bytes
f2046b4 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 | """
Fetches Open-Meteo's Previous Runs API -- real past forecasts at fixed
lead-time offsets (1-7 days), the piece needed to validate whether
ECMWF's forecast would genuinely have caught real historical flood
conditions, without look-ahead leakage. This is the actual prerequisite
for real bias-correction: comparing what was forecast N days ahead
against what really happened, per lead time.
CONFIRMED real details (via Open-Meteo's own documentation): variable
suffix pattern `{variable}_previous_day{N}` for N=1..7 (e.g.
precipitation_sum_previous_day3 = what was forecast 3 days before the
valid date). Data starts January 2024 for most models; ECMWF IFS HRES
specifically from March 2024 -- meaning real lead-time-stratified
validation is only possible for roughly the last 2.5 years, not this
project's full multi-decade discharge record. Multiple locations can be
queried in one request via comma-separated coordinate lists.
NOT directly confirmed: the exact hostname for this specific API.
Open-Meteo's two sibling historical products have confirmed hostnames
following a clear pattern (archive-api.open-meteo.com,
historical-forecast-api.open-meteo.com) -- "previous-runs-api.open-
meteo.com" is inferred from that pattern, not verified by a direct
source. Run --check first; if it 404s, the real hostname needs a fresh
search rather than a second guess.
Usage:
python -m scripts.fetch_previous_runs --check
python -m scripts.fetch_previous_runs --stations-path datasets/station_elevations.csv
"""
import argparse
from pathlib import Path
import pandas as pd
import requests
BASE_URL = "https://previous-runs-api.open-meteo.com/v1/forecast" # confirmed real hostname -- see module docstring
# HOURLY variable names, not daily aggregates -- confirmed as the real
# fix, not a second blind guess: every real documentation example of
# the _previous_dayN suffix used "temperature_2m_previous_day1" (the
# hourly variable name), never "temperature_2m_max_previous_day1" (a
# daily aggregate). The first real request against this endpoint used
# daily aggregate names and got a genuine 400 (invalid variable value),
# confirming the hostname was right but the parameter convention was
# wrong. Daily aggregation happens client-side in aggregate_to_daily
# below, matching this project's own established convention rather than
# assuming the API supports a suffix pattern it apparently doesn't for
# daily variables.
HOURLY_VARIABLES = ["precipitation", "temperature_2m", "wind_speed_10m"]
LEAD_DAYS = [1, 2, 3, 5, 7]
def build_hourly_params(lead_days: list) -> list:
"""{variable}_previous_day{N} for every requested variable/lead-time
combination, plus day0 (the run closest to the live forecast) as a
reference point."""
params = []
for var in HOURLY_VARIABLES:
params.append(f"{var}_previous_day0")
for n in lead_days:
params.append(f"{var}_previous_day{n}")
return params
def aggregate_to_daily(df: pd.DataFrame) -> pd.DataFrame:
"""
Collapses the real hourly response to daily values -- sum for
precipitation, max for temperature/wind, matching standard
meteorological daily-aggregate convention and this project's
existing daily-resolution pipeline (climate_climatology.py etc.).
"""
df = df.copy()
df["date"] = pd.to_datetime(df["time"]).dt.date
agg_rules = {}
for col in df.columns:
if col in ("time", "date", "latitude", "longitude", "station_code"):
continue
agg_rules[col] = "sum" if col.startswith("precipitation") else "max"
daily = df.groupby("date").agg(agg_rules).reset_index()
return daily
def check_access() -> bool:
params = {
"latitude": 48.748705218, "longitude": 0.58156195,
"hourly": ",".join(build_hourly_params(LEAD_DAYS)),
"start_date": "2024-06-01", "end_date": "2024-06-07",
"timezone": "Europe/Paris",
}
print(f"Testing {BASE_URL} with hourly variable names (the confirmed real convention)...")
try:
resp = requests.get(BASE_URL, params=params, timeout=30)
print(f"HTTP status: {resp.status_code}")
if resp.status_code != 200:
print(f"Response body: {resp.text[:500]}")
return False
data = resp.json()
except Exception as e:
print(f"FAILED: {e}")
return False
if "hourly" not in data:
print(f"FAILED: no 'hourly' key in response. Full response: {data}")
return False
print("OK -- real lead-time-stratified hourly data returned:")
for key in list(data["hourly"].keys())[:6]:
print(f" {key}: {data['hourly'][key][:3]}...")
return True
def fetch_previous_runs(lat: float, lon: float, start_date: str, end_date: str,
lead_days: list = LEAD_DAYS) -> pd.DataFrame:
params = {
"latitude": lat, "longitude": lon,
"hourly": ",".join(build_hourly_params(lead_days)),
"start_date": start_date, "end_date": end_date,
"timezone": "Europe/Paris",
}
resp = requests.get(BASE_URL, params=params, timeout=60)
resp.raise_for_status()
data = resp.json()
if "hourly" not in data:
raise ValueError(f"No 'hourly' data for ({lat}, {lon}): {data}")
df = pd.DataFrame(data["hourly"])
df["latitude"] = lat
df["longitude"] = lon
return df
def main() -> None:
parser = argparse.ArgumentParser(description="Fetch Open-Meteo Previous Runs data for bias-correction validation")
parser.add_argument("--check", action="store_true")
parser.add_argument("--stations-path", type=Path, default=Path("datasets/station_elevations.csv"))
parser.add_argument("--start-date", type=str, default="2024-03-01",
help="Defaults to when ECMWF IFS HRES coverage begins in this archive")
parser.add_argument("--end-date", type=str, default="2026-08-30")
parser.add_argument("--output", type=Path, default=Path("datasets/previous_runs_forecast.csv"))
args = parser.parse_args()
if args.check:
check_access()
return
stations = pd.read_csv(args.stations_path)
lat_col = "latitude" if "latitude" in stations.columns else "lat"
lon_col = "longitude" if "longitude" in stations.columns else "lon"
all_dfs = []
for _, row in stations.iterrows():
code = row.get("station_code", f"{row[lat_col]:.3f}_{row[lon_col]:.3f}")
print(f"Fetching real previous-runs data for {code}...")
try:
df = fetch_previous_runs(row[lat_col], row[lon_col], args.start_date, args.end_date)
df = aggregate_to_daily(df)
df["station_code"] = code
all_dfs.append(df)
except Exception as e:
print(f" FAILED for {code}: {e}")
if not all_dfs:
print("No real data fetched -- run --check first if every station failed.")
return
combined = pd.concat(all_dfs, ignore_index=True)
args.output.parent.mkdir(parents=True, exist_ok=True)
combined.to_csv(args.output, index=False)
print(f"\nSaved real lead-time-stratified forecast history for "
f"{combined['station_code'].nunique()} station(s) to {args.output}")
if __name__ == "__main__":
main() |