File size: 7,253 Bytes
f2046b4
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
"""
Fetches Open-Meteo's Previous Runs API -- real past forecasts at fixed
lead-time offsets (1-7 days), the piece needed to validate whether
ECMWF's forecast would genuinely have caught real historical flood
conditions, without look-ahead leakage. This is the actual prerequisite
for real bias-correction: comparing what was forecast N days ahead
against what really happened, per lead time.

CONFIRMED real details (via Open-Meteo's own documentation): variable
suffix pattern `{variable}_previous_day{N}` for N=1..7 (e.g.
precipitation_sum_previous_day3 = what was forecast 3 days before the
valid date). Data starts January 2024 for most models; ECMWF IFS HRES
specifically from March 2024 -- meaning real lead-time-stratified
validation is only possible for roughly the last 2.5 years, not this
project's full multi-decade discharge record. Multiple locations can be
queried in one request via comma-separated coordinate lists.

NOT directly confirmed: the exact hostname for this specific API.
Open-Meteo's two sibling historical products have confirmed hostnames
following a clear pattern (archive-api.open-meteo.com,
historical-forecast-api.open-meteo.com) -- "previous-runs-api.open-
meteo.com" is inferred from that pattern, not verified by a direct
source. Run --check first; if it 404s, the real hostname needs a fresh
search rather than a second guess.

Usage:
    python -m scripts.fetch_previous_runs --check
    python -m scripts.fetch_previous_runs --stations-path datasets/station_elevations.csv
"""
import argparse
from pathlib import Path

import pandas as pd
import requests

BASE_URL = "https://previous-runs-api.open-meteo.com/v1/forecast"  # confirmed real hostname -- see module docstring

# HOURLY variable names, not daily aggregates -- confirmed as the real
# fix, not a second blind guess: every real documentation example of
# the _previous_dayN suffix used "temperature_2m_previous_day1" (the
# hourly variable name), never "temperature_2m_max_previous_day1" (a
# daily aggregate). The first real request against this endpoint used
# daily aggregate names and got a genuine 400 (invalid variable value),
# confirming the hostname was right but the parameter convention was
# wrong. Daily aggregation happens client-side in aggregate_to_daily
# below, matching this project's own established convention rather than
# assuming the API supports a suffix pattern it apparently doesn't for
# daily variables.
HOURLY_VARIABLES = ["precipitation", "temperature_2m", "wind_speed_10m"]
LEAD_DAYS = [1, 2, 3, 5, 7]


def build_hourly_params(lead_days: list) -> list:
    """{variable}_previous_day{N} for every requested variable/lead-time
    combination, plus day0 (the run closest to the live forecast) as a
    reference point."""
    params = []
    for var in HOURLY_VARIABLES:
        params.append(f"{var}_previous_day0")
        for n in lead_days:
            params.append(f"{var}_previous_day{n}")
    return params


def aggregate_to_daily(df: pd.DataFrame) -> pd.DataFrame:
    """
    Collapses the real hourly response to daily values -- sum for
    precipitation, max for temperature/wind, matching standard
    meteorological daily-aggregate convention and this project's
    existing daily-resolution pipeline (climate_climatology.py etc.).
    """
    df = df.copy()
    df["date"] = pd.to_datetime(df["time"]).dt.date
    agg_rules = {}
    for col in df.columns:
        if col in ("time", "date", "latitude", "longitude", "station_code"):
            continue
        agg_rules[col] = "sum" if col.startswith("precipitation") else "max"
    daily = df.groupby("date").agg(agg_rules).reset_index()
    return daily


def check_access() -> bool:
    params = {
        "latitude": 48.748705218, "longitude": 0.58156195,
        "hourly": ",".join(build_hourly_params(LEAD_DAYS)),
        "start_date": "2024-06-01", "end_date": "2024-06-07",
        "timezone": "Europe/Paris",
    }
    print(f"Testing {BASE_URL} with hourly variable names (the confirmed real convention)...")
    try:
        resp = requests.get(BASE_URL, params=params, timeout=30)
        print(f"HTTP status: {resp.status_code}")
        if resp.status_code != 200:
            print(f"Response body: {resp.text[:500]}")
            return False
        data = resp.json()
    except Exception as e:
        print(f"FAILED: {e}")
        return False

    if "hourly" not in data:
        print(f"FAILED: no 'hourly' key in response. Full response: {data}")
        return False

    print("OK -- real lead-time-stratified hourly data returned:")
    for key in list(data["hourly"].keys())[:6]:
        print(f"  {key}: {data['hourly'][key][:3]}...")
    return True


def fetch_previous_runs(lat: float, lon: float, start_date: str, end_date: str,
                         lead_days: list = LEAD_DAYS) -> pd.DataFrame:
    params = {
        "latitude": lat, "longitude": lon,
        "hourly": ",".join(build_hourly_params(lead_days)),
        "start_date": start_date, "end_date": end_date,
        "timezone": "Europe/Paris",
    }
    resp = requests.get(BASE_URL, params=params, timeout=60)
    resp.raise_for_status()
    data = resp.json()
    if "hourly" not in data:
        raise ValueError(f"No 'hourly' data for ({lat}, {lon}): {data}")

    df = pd.DataFrame(data["hourly"])
    df["latitude"] = lat
    df["longitude"] = lon
    return df



def main() -> None:
    parser = argparse.ArgumentParser(description="Fetch Open-Meteo Previous Runs data for bias-correction validation")
    parser.add_argument("--check", action="store_true")
    parser.add_argument("--stations-path", type=Path, default=Path("datasets/station_elevations.csv"))
    parser.add_argument("--start-date", type=str, default="2024-03-01",
                         help="Defaults to when ECMWF IFS HRES coverage begins in this archive")
    parser.add_argument("--end-date", type=str, default="2026-08-30")
    parser.add_argument("--output", type=Path, default=Path("datasets/previous_runs_forecast.csv"))
    args = parser.parse_args()

    if args.check:
        check_access()
        return

    stations = pd.read_csv(args.stations_path)
    lat_col = "latitude" if "latitude" in stations.columns else "lat"
    lon_col = "longitude" if "longitude" in stations.columns else "lon"

    all_dfs = []
    for _, row in stations.iterrows():
        code = row.get("station_code", f"{row[lat_col]:.3f}_{row[lon_col]:.3f}")
        print(f"Fetching real previous-runs data for {code}...")
        try:
            df = fetch_previous_runs(row[lat_col], row[lon_col], args.start_date, args.end_date)
            df = aggregate_to_daily(df)
            df["station_code"] = code
            all_dfs.append(df)
        except Exception as e:
            print(f"  FAILED for {code}: {e}")

    if not all_dfs:
        print("No real data fetched -- run --check first if every station failed.")
        return

    combined = pd.concat(all_dfs, ignore_index=True)
    args.output.parent.mkdir(parents=True, exist_ok=True)
    combined.to_csv(args.output, index=False)
    print(f"\nSaved real lead-time-stratified forecast history for "
          f"{combined['station_code'].nunique()} station(s) to {args.output}")


if __name__ == "__main__":
    main()