Spaces:
Running
Running
Download scripts/data_processor.py from Jitendra12421/PREDICTIONSITE: direct link, hf CLI and curl.
- Browser
- Download file 1.62 kB
-
https://huggingface.co/spaces/Jitendra12421/PREDICTIONSITE/resolve/main/scripts/data_processor.py
- Command line
-
hf download hf://spaces/Jitendra12421/PREDICTIONSITE/scripts/data_processor.py
-
curl -L -o data_processor.py https://huggingface.co/spaces/Jitendra12421/PREDICTIONSITE/resolve/main/scripts/data_processor.py
1.62 kB
| import pandas as pd | |
| from pathlib import Path | |
| import tempfile | |
| import os | |
| import re | |
| def process_timeframe(base_parquet_path: Path, timeframe: str, format: str) -> str: | |
| """ | |
| Reads the base 1m parquet file, resamples to the requested timeframe, | |
| and saves as either parquet or csv in a temporary file. | |
| Returns the path to the temporary file. | |
| """ | |
| df = pd.read_parquet(base_parquet_path) | |
| # Convert 'date' to datetime if needed | |
| if not pd.api.types.is_datetime64_any_dtype(df['date']): | |
| df['date'] = pd.to_datetime(df['date']) | |
| df = df.set_index('date') | |
| # Map typical user timeframe inputs to pandas offset strings | |
| # '1m' -> '1min', '5m' -> '5min', '1h' -> '1h', '1d' -> '1D' | |
| pd_tf = timeframe.lower() | |
| pd_tf = re.sub(r'([0-9]+)m$', r'\1min', pd_tf) | |
| pd_tf = re.sub(r'([0-9]+)d$', r'\1D', pd_tf) | |
| pd_tf = re.sub(r'([0-9]+)h$', r'\1h', pd_tf) | |
| if pd_tf != '1min': | |
| agg_dict = { | |
| 'open': 'first', | |
| 'high': 'max', | |
| 'low': 'min', | |
| 'close': 'last', | |
| 'volume': 'sum' | |
| } | |
| df = df.resample(pd_tf).agg(agg_dict).dropna(subset=['close']) | |
| # Reset index so 'date' is a column again (usually expected in output) | |
| df = df.reset_index() | |
| # Save to temp file | |
| temp_dir = tempfile.gettempdir() | |
| filename = f"nifty50_{timeframe}.{format.lower()}" | |
| out_path = os.path.join(temp_dir, filename) | |
| if format.lower() == 'csv': | |
| df.to_csv(out_path, index=False) | |
| else: | |
| df.to_parquet(out_path, index=False) | |
| return out_path | |