Spaces:
Running
Running
File size: 1,622 Bytes
54c9d5f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 | import pandas as pd
from pathlib import Path
import tempfile
import os
import re
def process_timeframe(base_parquet_path: Path, timeframe: str, format: str) -> str:
"""
Reads the base 1m parquet file, resamples to the requested timeframe,
and saves as either parquet or csv in a temporary file.
Returns the path to the temporary file.
"""
df = pd.read_parquet(base_parquet_path)
# Convert 'date' to datetime if needed
if not pd.api.types.is_datetime64_any_dtype(df['date']):
df['date'] = pd.to_datetime(df['date'])
df = df.set_index('date')
# Map typical user timeframe inputs to pandas offset strings
# '1m' -> '1min', '5m' -> '5min', '1h' -> '1h', '1d' -> '1D'
pd_tf = timeframe.lower()
pd_tf = re.sub(r'([0-9]+)m$', r'\1min', pd_tf)
pd_tf = re.sub(r'([0-9]+)d$', r'\1D', pd_tf)
pd_tf = re.sub(r'([0-9]+)h$', r'\1h', pd_tf)
if pd_tf != '1min':
agg_dict = {
'open': 'first',
'high': 'max',
'low': 'min',
'close': 'last',
'volume': 'sum'
}
df = df.resample(pd_tf).agg(agg_dict).dropna(subset=['close'])
# Reset index so 'date' is a column again (usually expected in output)
df = df.reset_index()
# Save to temp file
temp_dir = tempfile.gettempdir()
filename = f"nifty50_{timeframe}.{format.lower()}"
out_path = os.path.join(temp_dir, filename)
if format.lower() == 'csv':
df.to_csv(out_path, index=False)
else:
df.to_parquet(out_path, index=False)
return out_path
|