Spaces:
Running
Running
File size: 6,722 Bytes
f5fb7f5 8b86b90 f5fb7f5 8b86b90 f5fb7f5 01fa959 f5fb7f5 8b86b90 f5fb7f5 01fa959 f5fb7f5 01fa959 f5fb7f5 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 | import os
import json
import math
import time
import requests
import pandas as pd
from datetime import datetime, date, timedelta
from zoneinfo import ZoneInfo
import pandas_market_calendars as mcal
import numpy as np
def _sanitize_value(v):
"""Replace NaN/Inf floats with 0 so JSON serialization doesn't break."""
if isinstance(v, float) and (math.isnan(v) or math.isinf(v)):
return 0.0
return v
def _sanitize_dict(d):
"""Recursively sanitize a dict of float values."""
return {k: _sanitize_dict(v) if isinstance(v, dict) else _sanitize_value(v) for k, v in d.items()}
IST = ZoneInfo("Asia/Kolkata")
BASE_DIR = os.path.dirname(__file__)
DATA_DIR = os.path.join(BASE_DIR, "data")
RULES_FILE = os.path.join(DATA_DIR, "t5_rules.json")
PREDICTIONS_FILE = os.path.join(BASE_DIR, "t5_predictions.json")
TICKERS = [
'ADANIENT', 'ADANIPORTS', 'APOLLOHOSP', 'ASIANPAINT', 'AXISBANK', 'BAJAJ-AUTO', 'BAJAJFINSV', 'BAJFINANCE',
'BHARTIARTL', 'BPCL', 'BRITANNIA', 'CIPLA', 'COALINDIA', 'DIVISLAB', 'DRREDDY', 'EICHERMOT', 'GRASIM',
'HCLTECH', 'HDFCBANK', 'HDFCLIFE', 'HEROMOTOCO', 'HINDALCO', 'HINDUNILVR', 'ICICIBANK', 'INDUSINDBK',
'INFY', 'ITC', 'JSWSTEEL', 'KOTAKBANK', 'LT', 'M&M', 'MARUTI', 'NESTLEIND', 'NTPC', 'ONGC', 'POWERGRID',
'RELIANCE', 'SBILIFE', 'SBIN', 'SUNPHARMA', 'TATACONSUM', 'TATAMOTORS', 'TATASTEEL', 'TCS', 'TECHM',
'TITAN', 'ULTRACEMCO', 'UPL', 'WIPRO'
]
def fetch_groww_history(ticker: str, start_ts: int, end_ts: int):
url = f"https://groww.in/v1/api/charting_service/v2/chart/exchange/NSE/segment/CASH/{ticker}?endTimeInMillis={end_ts}&intervalInMinutes=1&startTimeInMillis={start_ts}"
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)",
"Accept": "application/json"
}
try:
res = requests.get(url, headers=headers, timeout=10)
if res.status_code == 200:
data = res.json()
if data and 'candles' in data:
return data['candles']
return []
except Exception as e:
print(f"Error fetching {ticker}: {e}")
return []
def evaluate_rule(rule_str: str, features: dict) -> int:
if not rule_str:
return 0
# Convert 'AND' to python 'and'
py_rule = rule_str.replace("AND", "and")
try:
# features dict contains e.g. {'ret_5m': -0.01, 'gap': 0.005, ...}
# eval evaluates the boolean expression
result = eval(py_rule, {"__builtins__": None}, features)
return 1 if result else -1
except Exception as e:
print(f"Rule eval error: {e}")
return 0
def run_t5_pipeline():
print(f"[{datetime.now(IST)}] Starting T5 Engine Pipeline...")
if not os.path.exists(RULES_FILE):
print("T5 rules file not found!")
return {"status": "error", "reason": "Missing rules file"}
with open(RULES_FILE, "r") as f:
rules_list = json.load(f)
rules_dict = {item['Ticker']: item for item in rules_list if item['Rule']}
now = datetime.now(IST)
# Fetch data for the last 5 days to ensure we have yesterday and today
start_dt = now - timedelta(days=5)
start_ts = int(start_dt.timestamp() * 1000)
end_ts = int(now.timestamp() * 1000)
predictions = {}
for ticker in TICKERS:
candles = fetch_groww_history(ticker, start_ts, end_ts)
if not candles:
continue
# Format: [timestamp, open, high, low, close, volume]
df = pd.DataFrame(candles, columns=['timestamp', 'open', 'high', 'low', 'close', 'volume'])
df['date'] = pd.to_datetime(df['timestamp'], unit='s', utc=True).dt.tz_convert(IST)
df.set_index('date', inplace=True)
df.sort_index(inplace=True)
# Group by day
df['day'] = df.index.date
days = df['day'].unique()
if len(days) < 2:
print(f"{ticker}: Not enough days of data")
continue
# Today is the last day in the dataset
today_date = days[-1]
yesterday_date = days[-2]
# Yesterday's aggregation
yesterday_df = df[df['day'] == yesterday_date]
if yesterday_df.empty:
continue
prev_open = yesterday_df['open'].iloc[0]
prev_close = yesterday_df['close'].iloc[-1]
prev_vol = yesterday_df['volume'].sum()
# Today's first 5 mins aggregation (09:15 to 09:19 inclusive)
today_df = df[df['day'] == today_date]
first_5m_df = today_df.between_time('09:15', '09:19')
if first_5m_df.empty:
print(f"{ticker}: Missing first 5 mins data for today")
continue
open_5m = first_5m_df['open'].iloc[0]
high_5m = first_5m_df['high'].max()
low_5m = first_5m_df['low'].min()
close_5m = first_5m_df['close'].iloc[-1]
vol_5m = first_5m_df['volume'].sum()
# Calculate features
features = {}
features['ret_5m'] = (close_5m - open_5m) / open_5m if open_5m else 0
features['gap'] = (open_5m - prev_close) / prev_close if prev_close else 0
features['candle_shape'] = (close_5m - open_5m) / (high_5m - low_5m + 1e-9)
features['close_to_high'] = (close_5m - low_5m) / (high_5m - low_5m + 1e-9)
features['vol_5m_ratio'] = vol_5m / (prev_vol + 1e-9)
features['hl_spread'] = (high_5m - low_5m) / open_5m if open_5m else 0
features['prev_ret'] = (prev_close - prev_open) / prev_open if prev_open else 0
# Sanitize NaN/Inf values that break JSON serialization
features = _sanitize_dict(features)
# Evaluate rule
rule_item = rules_dict.get(ticker)
if rule_item:
pred = evaluate_rule(rule_item['Rule'], features)
predictions[ticker] = {
"prediction": "UP" if pred == 1 else ("DOWN" if pred == -1 else "FLAT"),
"features": features,
"rule_used": rule_item['Rule'],
"accuracy": round(rule_item.get('Test_Acc', 0.6432) * 100, 2)
}
time.sleep(0.5) # Rate limiting
output = {
"generated_at": now.isoformat(),
"date_target": str(now.date()),
"horizon": "Same day close > 09:19 close",
"predictions": predictions
}
with open(PREDICTIONS_FILE, "w") as f:
json.dump(output, f, indent=4)
print(f"[{datetime.now(IST)}] T5 Pipeline completed. {len(predictions)} predictions generated.")
return {"status": "success", "count": len(predictions)}
if __name__ == "__main__":
run_t5_pipeline()
|