multiticker / forecaster_engine.py
Jitendra12421's picture
Upload 6 files
69f5fb0 verified
Raw
History Blame
4.29 kB
import os
import json
import pandas as pd
import numpy as np
from datetime import datetime
DATA_FILE = os.path.join(os.path.dirname(__file__), "data", "nifty50_daily.parquet")
PREDICTIONS_FILE = os.path.join(os.path.dirname(__file__), "predictions.json")
def generate_predictions():
if not os.path.exists(DATA_FILE):
print(f"Data file missing: {DATA_FILE}")
return
df_all = pd.read_parquet(DATA_FILE)
tickers = df_all['ticker'].unique()
predictions = {}
forecast_date = None
for ticker in tickers:
df = df_all[df_all['ticker'] == ticker].copy()
df.sort_values('date', inplace=True)
df.set_index('date', inplace=True)
if len(df) < 130:
continue
forecast_date_ts = df.index[-1] + pd.Timedelta(days=1)
# Advance past weekends roughly for display
if forecast_date_ts.weekday() >= 5:
forecast_date_ts += pd.Timedelta(days=(7 - forecast_date_ts.weekday()))
if forecast_date is None:
forecast_date = forecast_date_ts.strftime('%Y-%m-%d')
daily_close = df['close']
# Features
delta = daily_close.diff()
gain = (delta.where(delta > 0, 0)).rolling(window=2).mean()
loss = (-delta.where(delta < 0, 0)).rolling(window=2).mean()
rs = gain / loss
rsi_2 = 100 - (100 / (1 + rs))
sma_3 = daily_close.rolling(window=3).mean()
dist_sma3 = daily_close / sma_3
sma_5 = daily_close.rolling(window=5).mean()
dist_sma5 = daily_close / sma_5
df_eval = pd.DataFrame({
'close': daily_close,
'rsi_2': rsi_2,
'dist_sma3': dist_sma3,
'dist_sma5': dist_sma5,
'1d_ret': daily_close.pct_change()
}).dropna()
# Target for historical testing
df_eval['actual_dir'] = np.where(df_eval['close'].shift(-1) > df_eval['close'], 1, -1)
# The last row is TODAY. We don't have tomorrow's close, so actual_dir is wrong for the last row.
# We test on the 120 days BEFORE today
test_set = df_eval.iloc[-121:-1]
today_data = df_eval.iloc[-1]
best_acc = 0
best_rule = None
# 1. RSI-2 threshold
for thresh in [10, 20, 30, 40, 50, 60, 70, 80, 90]:
for op in ['<', '>']:
sig = np.where(test_set['rsi_2'] < thresh if op == '<' else test_set['rsi_2'] > thresh, 1, -1)
acc = (sig == test_set['actual_dir']).mean()
if acc > best_acc:
best_acc = acc
best_rule = ('rsi_2', thresh, op)
# 2. SMA distance threshold
for feature in ['dist_sma3', 'dist_sma5']:
for thresh in [0.95, 0.98, 1.0, 1.02, 1.05]:
sig = np.where(test_set[feature] < thresh, 1, -1)
acc = (sig == test_set['actual_dir']).mean()
if acc > best_acc:
best_acc = acc
best_rule = (feature, thresh, '<')
# 3. 1d return
sig = np.where(test_set['1d_ret'] < 0, 1, -1)
acc = (sig == test_set['actual_dir']).mean()
if acc > best_acc:
best_acc = acc
best_rule = ('1d_ret', 0, '<')
feature, thresh, op = best_rule
val = today_data[feature]
if op == '<':
prediction = 1 if val < thresh else -1
else:
prediction = 1 if val > thresh else -1
predictions[ticker] = {
"prediction": "UP" if prediction == 1 else "DOWN",
"probability": round(best_acc * 100, 2),
"rule_used": f"{feature} {op} {thresh}"
}
output = {
"generated_at": datetime.now().isoformat(),
"forecast_date": forecast_date,
"predictions": predictions
}
with open(PREDICTIONS_FILE, "w") as f:
json.dump(output, f, indent=4)
print(f"Generated predictions for {forecast_date}. Saved to {PREDICTIONS_FILE}")
return output
if __name__ == "__main__":
generate_predictions()