Spaces:
Running on Zero
Running on Zero
Download scripts/data_acquisition/fetch_previous_runs.py from ageraustine/River_Network: direct link, hf CLI and curl.
- Browser
- Download file 7.25 kB
-
https://huggingface.co/spaces/ageraustine/River_Network/resolve/main/scripts/data_acquisition/fetch_previous_runs.py
- Command line
-
hf download hf://spaces/ageraustine/River_Network/scripts/data_acquisition/fetch_previous_runs.py
-
curl -L -o fetch_previous_runs.py https://huggingface.co/spaces/ageraustine/River_Network/resolve/main/scripts/data_acquisition/fetch_previous_runs.py
7.25 kB
| """ | |
| Fetches Open-Meteo's Previous Runs API -- real past forecasts at fixed | |
| lead-time offsets (1-7 days), the piece needed to validate whether | |
| ECMWF's forecast would genuinely have caught real historical flood | |
| conditions, without look-ahead leakage. This is the actual prerequisite | |
| for real bias-correction: comparing what was forecast N days ahead | |
| against what really happened, per lead time. | |
| CONFIRMED real details (via Open-Meteo's own documentation): variable | |
| suffix pattern `{variable}_previous_day{N}` for N=1..7 (e.g. | |
| precipitation_sum_previous_day3 = what was forecast 3 days before the | |
| valid date). Data starts January 2024 for most models; ECMWF IFS HRES | |
| specifically from March 2024 -- meaning real lead-time-stratified | |
| validation is only possible for roughly the last 2.5 years, not this | |
| project's full multi-decade discharge record. Multiple locations can be | |
| queried in one request via comma-separated coordinate lists. | |
| NOT directly confirmed: the exact hostname for this specific API. | |
| Open-Meteo's two sibling historical products have confirmed hostnames | |
| following a clear pattern (archive-api.open-meteo.com, | |
| historical-forecast-api.open-meteo.com) -- "previous-runs-api.open- | |
| meteo.com" is inferred from that pattern, not verified by a direct | |
| source. Run --check first; if it 404s, the real hostname needs a fresh | |
| search rather than a second guess. | |
| Usage: | |
| python -m scripts.fetch_previous_runs --check | |
| python -m scripts.fetch_previous_runs --stations-path datasets/station_elevations.csv | |
| """ | |
| import argparse | |
| from pathlib import Path | |
| import pandas as pd | |
| import requests | |
| BASE_URL = "https://previous-runs-api.open-meteo.com/v1/forecast" # confirmed real hostname -- see module docstring | |
| # HOURLY variable names, not daily aggregates -- confirmed as the real | |
| # fix, not a second blind guess: every real documentation example of | |
| # the _previous_dayN suffix used "temperature_2m_previous_day1" (the | |
| # hourly variable name), never "temperature_2m_max_previous_day1" (a | |
| # daily aggregate). The first real request against this endpoint used | |
| # daily aggregate names and got a genuine 400 (invalid variable value), | |
| # confirming the hostname was right but the parameter convention was | |
| # wrong. Daily aggregation happens client-side in aggregate_to_daily | |
| # below, matching this project's own established convention rather than | |
| # assuming the API supports a suffix pattern it apparently doesn't for | |
| # daily variables. | |
| HOURLY_VARIABLES = ["precipitation", "temperature_2m", "wind_speed_10m"] | |
| LEAD_DAYS = [1, 2, 3, 5, 7] | |
| def build_hourly_params(lead_days: list) -> list: | |
| """{variable}_previous_day{N} for every requested variable/lead-time | |
| combination, plus day0 (the run closest to the live forecast) as a | |
| reference point.""" | |
| params = [] | |
| for var in HOURLY_VARIABLES: | |
| params.append(f"{var}_previous_day0") | |
| for n in lead_days: | |
| params.append(f"{var}_previous_day{n}") | |
| return params | |
| def aggregate_to_daily(df: pd.DataFrame) -> pd.DataFrame: | |
| """ | |
| Collapses the real hourly response to daily values -- sum for | |
| precipitation, max for temperature/wind, matching standard | |
| meteorological daily-aggregate convention and this project's | |
| existing daily-resolution pipeline (climate_climatology.py etc.). | |
| """ | |
| df = df.copy() | |
| df["date"] = pd.to_datetime(df["time"]).dt.date | |
| agg_rules = {} | |
| for col in df.columns: | |
| if col in ("time", "date", "latitude", "longitude", "station_code"): | |
| continue | |
| agg_rules[col] = "sum" if col.startswith("precipitation") else "max" | |
| daily = df.groupby("date").agg(agg_rules).reset_index() | |
| return daily | |
| def check_access() -> bool: | |
| params = { | |
| "latitude": 48.748705218, "longitude": 0.58156195, | |
| "hourly": ",".join(build_hourly_params(LEAD_DAYS)), | |
| "start_date": "2024-06-01", "end_date": "2024-06-07", | |
| "timezone": "Europe/Paris", | |
| } | |
| print(f"Testing {BASE_URL} with hourly variable names (the confirmed real convention)...") | |
| try: | |
| resp = requests.get(BASE_URL, params=params, timeout=30) | |
| print(f"HTTP status: {resp.status_code}") | |
| if resp.status_code != 200: | |
| print(f"Response body: {resp.text[:500]}") | |
| return False | |
| data = resp.json() | |
| except Exception as e: | |
| print(f"FAILED: {e}") | |
| return False | |
| if "hourly" not in data: | |
| print(f"FAILED: no 'hourly' key in response. Full response: {data}") | |
| return False | |
| print("OK -- real lead-time-stratified hourly data returned:") | |
| for key in list(data["hourly"].keys())[:6]: | |
| print(f" {key}: {data['hourly'][key][:3]}...") | |
| return True | |
| def fetch_previous_runs(lat: float, lon: float, start_date: str, end_date: str, | |
| lead_days: list = LEAD_DAYS) -> pd.DataFrame: | |
| params = { | |
| "latitude": lat, "longitude": lon, | |
| "hourly": ",".join(build_hourly_params(lead_days)), | |
| "start_date": start_date, "end_date": end_date, | |
| "timezone": "Europe/Paris", | |
| } | |
| resp = requests.get(BASE_URL, params=params, timeout=60) | |
| resp.raise_for_status() | |
| data = resp.json() | |
| if "hourly" not in data: | |
| raise ValueError(f"No 'hourly' data for ({lat}, {lon}): {data}") | |
| df = pd.DataFrame(data["hourly"]) | |
| df["latitude"] = lat | |
| df["longitude"] = lon | |
| return df | |
| def main() -> None: | |
| parser = argparse.ArgumentParser(description="Fetch Open-Meteo Previous Runs data for bias-correction validation") | |
| parser.add_argument("--check", action="store_true") | |
| parser.add_argument("--stations-path", type=Path, default=Path("datasets/station_elevations.csv")) | |
| parser.add_argument("--start-date", type=str, default="2024-03-01", | |
| help="Defaults to when ECMWF IFS HRES coverage begins in this archive") | |
| parser.add_argument("--end-date", type=str, default="2026-08-30") | |
| parser.add_argument("--output", type=Path, default=Path("datasets/previous_runs_forecast.csv")) | |
| args = parser.parse_args() | |
| if args.check: | |
| check_access() | |
| return | |
| stations = pd.read_csv(args.stations_path) | |
| lat_col = "latitude" if "latitude" in stations.columns else "lat" | |
| lon_col = "longitude" if "longitude" in stations.columns else "lon" | |
| all_dfs = [] | |
| for _, row in stations.iterrows(): | |
| code = row.get("station_code", f"{row[lat_col]:.3f}_{row[lon_col]:.3f}") | |
| print(f"Fetching real previous-runs data for {code}...") | |
| try: | |
| df = fetch_previous_runs(row[lat_col], row[lon_col], args.start_date, args.end_date) | |
| df = aggregate_to_daily(df) | |
| df["station_code"] = code | |
| all_dfs.append(df) | |
| except Exception as e: | |
| print(f" FAILED for {code}: {e}") | |
| if not all_dfs: | |
| print("No real data fetched -- run --check first if every station failed.") | |
| return | |
| combined = pd.concat(all_dfs, ignore_index=True) | |
| args.output.parent.mkdir(parents=True, exist_ok=True) | |
| combined.to_csv(args.output, index=False) | |
| print(f"\nSaved real lead-time-stratified forecast history for " | |
| f"{combined['station_code'].nunique()} station(s) to {args.output}") | |
| if __name__ == "__main__": | |
| main() |