File size: 1,622 Bytes
54c9d5f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
import pandas as pd
from pathlib import Path
import tempfile
import os
import re

def process_timeframe(base_parquet_path: Path, timeframe: str, format: str) -> str:
    """
    Reads the base 1m parquet file, resamples to the requested timeframe, 
    and saves as either parquet or csv in a temporary file.
    Returns the path to the temporary file.
    """
    df = pd.read_parquet(base_parquet_path)
    
    # Convert 'date' to datetime if needed
    if not pd.api.types.is_datetime64_any_dtype(df['date']):
        df['date'] = pd.to_datetime(df['date'])
        
    df = df.set_index('date')
    
    # Map typical user timeframe inputs to pandas offset strings
    # '1m' -> '1min', '5m' -> '5min', '1h' -> '1h', '1d' -> '1D'
    pd_tf = timeframe.lower()
    pd_tf = re.sub(r'([0-9]+)m$', r'\1min', pd_tf)
    pd_tf = re.sub(r'([0-9]+)d$', r'\1D', pd_tf)
    pd_tf = re.sub(r'([0-9]+)h$', r'\1h', pd_tf)
    
    if pd_tf != '1min':
        agg_dict = {
            'open': 'first',
            'high': 'max',
            'low': 'min',
            'close': 'last',
            'volume': 'sum'
        }
        df = df.resample(pd_tf).agg(agg_dict).dropna(subset=['close'])
    
    # Reset index so 'date' is a column again (usually expected in output)
    df = df.reset_index()
    
    # Save to temp file
    temp_dir = tempfile.gettempdir()
    filename = f"nifty50_{timeframe}.{format.lower()}"
    out_path = os.path.join(temp_dir, filename)
    
    if format.lower() == 'csv':
        df.to_csv(out_path, index=False)
    else:
        df.to_parquet(out_path, index=False)
        
    return out_path