Spaces:
Sleeping
Sleeping
File size: 14,897 Bytes
5ed2a0f 116b2cb d7cdf88 6590c42 116b2cb 6590c42 116b2cb d2145b9 116b2cb d2145b9 116b2cb d7cdf88 6590c42 5ed2a0f 6590c42 5ed2a0f 6590c42 116b2cb 5ed2a0f 116b2cb 5ed2a0f 6590c42 bf6578d 5ed2a0f bf6578d 5ed2a0f 116b2cb d2145b9 116b2cb 5ed2a0f d7cdf88 d2145b9 116b2cb d2145b9 5ed2a0f 6590c42 5ed2a0f 6590c42 116b2cb 5ed2a0f 116b2cb 5ed2a0f 92d5def 5ed2a0f d2145b9 5ed2a0f d2145b9 6590c42 d2145b9 116b2cb 5ed2a0f 116b2cb d2145b9 d7cdf88 116b2cb 92d5def 116b2cb d2145b9 bf6578d 5ed2a0f 116b2cb bf6578d d2145b9 15acb55 d2145b9 bf6578d d2145b9 15acb55 5ed2a0f bf6578d 116b2cb bf6578d 116b2cb 15acb55 d2145b9 15acb55 d2145b9 15acb55 bf6578d 116b2cb bf6578d 116b2cb 5ed2a0f 116b2cb 6590c42 d2145b9 116b2cb 6590c42 d2145b9 116b2cb d2145b9 6590c42 d2145b9 5ed2a0f d2145b9 116b2cb d2145b9 d7cdf88 5ed2a0f d2145b9 5ed2a0f d2145b9 92d5def d2145b9 116b2cb d2145b9 116b2cb 5ed2a0f 116b2cb d2145b9 5ed2a0f d2145b9 92d5def 116b2cb 5ed2a0f 116b2cb d2145b9 5ed2a0f 92d5def 5ed2a0f d7cdf88 5ed2a0f 116b2cb 5ed2a0f d2145b9 5ed2a0f d2145b9 116b2cb 5ed2a0f bf6578d 5ed2a0f d2145b9 bf6578d 5ed2a0f bf6578d d7cdf88 116b2cb 5ed2a0f d2145b9 d7cdf88 bf6578d 116b2cb 5ed2a0f 116b2cb 5ed2a0f 116b2cb d2145b9 116b2cb 6590c42 f733922 5ed2a0f 116b2cb d2145b9 5ed2a0f 92d5def 5ed2a0f 92d5def d2145b9 f733922 d2145b9 d7cdf88 5ed2a0f d2145b9 5ed2a0f d2145b9 116b2cb 5ed2a0f d2145b9 5ed2a0f d2145b9 116b2cb | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 | # app.py β DataSynthis_ML_JobTask
# Evaluate pretrained ARIMA + LSTM on a user-picked historical window inside 2020-10-03..2025-10-03
import os, math, json, pickle, traceback, warnings
from typing import Tuple
import numpy as np
import pandas as pd
import yfinance as yf
from datetime import datetime, date
from pandas.tseries.offsets import BDay
import gradio as gr
import matplotlib.pyplot as plt
from statsmodels.tsa.arima.model import ARIMA
from sklearn.metrics import mean_squared_error, mean_absolute_error, mean_absolute_percentage_error
# Silence pandas/np deprecation spam in logs
warnings.simplefilter("ignore", FutureWarning)
# -------------------------
# Historical universe bounds (DO NOT change)
# -------------------------
UNIVERSE_START = date(2020, 10, 3)
UNIVERSE_END = date(2025, 10, 3)
# -------------------------
# Artifact paths (must exist)
# -------------------------
ARIMA_PKL = "arima_model.pkl" # only to read (p,d,q) back
LSTM_KERAS = "lstm_model.keras" # Keras 3 native format
LSTM_INFO_1 = "lstm_info.json" # scaler/meta
LSTM_INFO_2 = "lstm_model/info.json"
REQUIRE_LSTM = True # enforce LSTM presence
# lazy globals
_LSTM_MODEL = None
_LSTM_META = None
POPULAR = ["AAPL","MSFT","GOOGL","AMZN","TSLA","NVDA","META","NFLX","IBM","ORCL"]
# =========================
# Small helpers
# =========================
def _as_scalar(x) -> float:
"""Robustly turn a pandas/np value into a plain float without FutureWarnings."""
arr = np.asarray(x).reshape(-1,)
return float(arr[0])
# =========================
# Data utilities
# =========================
def fetch_universe_data(ticker: str) -> pd.DataFrame:
"""Download daily close for the fixed universe, B-day freq, forward-fill."""
df = yf.download(
ticker,
start=UNIVERSE_START.isoformat(),
end=UNIVERSE_END.isoformat(),
interval="1d",
auto_adjust=True,
progress=False,
)
if df is None or df.empty:
raise ValueError(f"No data returned for '{ticker}' within the universe window.")
df = df[['Close']].copy()
if df.index.tz is not None:
df.index = df.index.tz_convert(None)
df = df.asfreq('B')
df['Close'] = df['Close'].ffill()
return df
def clip_to_universe(d: date) -> date:
return min(max(d, UNIVERSE_START), UNIVERSE_END)
def build_train_test_for_window(df: pd.DataFrame, start_sel: date, end_sel: date):
"""
Train = all history strictly before start_sel (business days).
Test = start_sel..end_sel inclusive (business days).
"""
start_sel = clip_to_universe(start_sel)
end_sel = clip_to_universe(end_sel)
if start_sel > end_sel:
raise ValueError("Selected start date must be <= end date.")
train_mask = df.index < pd.Timestamp(start_sel)
test_mask = (df.index >= pd.Timestamp(start_sel)) & (df.index <= pd.Timestamp(end_sel))
train_s = df.loc[train_mask, 'Close']
test_s = df.loc[test_mask, 'Close']
if len(test_s) < 5:
raise ValueError("Selected test window is too short. Pick a wider evaluation window.")
if len(train_s) < 40:
raise ValueError("Not enough training history before the selected start date.")
return train_s, test_s
# =========================
# ARIMA helpers
# =========================
def extract_order_from_pickle(default_order=(2,1,2)) -> Tuple[int, int, int]:
"""Read (p,d,q) from your saved ARIMA pickle if possible; fallback to default."""
try:
if os.path.exists(ARIMA_PKL):
with open(ARIMA_PKL, "rb") as f:
obj = pickle.load(f)
if hasattr(obj, "model") and hasattr(obj.model, "order"):
return tuple(int(x) for x in obj.model.order)
mo = getattr(obj, "model_orders", None)
if isinstance(mo, dict):
return (int(mo.get("ar", 0)), int(mo.get("diff", 0)), int(mo.get("ma", 0)))
except Exception as e:
print(f"[WARN] Could not read order from pickle: {e}")
return default_order
def arima_roll_eval(train_s: pd.Series, test_s: pd.Series, order=(2,1,2)):
"""Fit once on TRAIN, then roll 1-step across TEST, updating state each step."""
res = ARIMA(train_s, order=order,
enforce_stationarity=False, enforce_invertibility=False).fit()
preds, trues = [], []
for i in range(len(test_s)):
fcast = res.forecast(1)
preds.append(_as_scalar(fcast))
trues.append(_as_scalar(test_s.iloc[i]))
try:
res = res.append(test_s.iloc[i:i+1]) # update state with realized value
except Exception:
res = ARIMA(pd.concat([train_s, test_s.iloc[:i+1]]), order=order,
enforce_stationarity=False, enforce_invertibility=False).fit()
return np.array(trues), np.array(preds)
# =========================
# LSTM helpers
# =========================
def _maybe_load_lstm(required: bool = False) -> bool:
"""Load LSTM model + metadata with robust compatibility across TF/Keras builds."""
global _LSTM_MODEL, _LSTM_META
if _LSTM_MODEL is not None and _LSTM_META is not None:
return True
info_path = LSTM_INFO_1 if os.path.isfile(LSTM_INFO_1) else (LSTM_INFO_2 if os.path.isfile(LSTM_INFO_2) else None)
missing = []
if not os.path.isfile(LSTM_KERAS):
missing.append(LSTM_KERAS)
if info_path is None:
missing.append("lstm_info.json")
if missing:
msg = "LSTM artifacts missing: " + ", ".join(missing)
if required: raise RuntimeError(msg)
print("[WARN]", msg)
return False
try:
import tensorflow as tf
try:
_LSTM_MODEL = tf.keras.models.load_model(
LSTM_KERAS, compile=False, safe_mode=False,
custom_objects={"Orthogonal": tf.keras.initializers.Orthogonal},
)
except TypeError:
_LSTM_MODEL = tf.keras.models.load_model(
LSTM_KERAS, compile=False,
custom_objects={"Orthogonal": tf.keras.initializers.Orthogonal},
)
with open(info_path, "r") as f:
_LSTM_META = json.load(f)
for k in ("lookback", "train_min", "train_max"):
if k not in _LSTM_META:
raise RuntimeError(f"Missing key '{k}' in {info_path}")
if "feature_range" not in _LSTM_META:
_LSTM_META["feature_range"] = [0.0, 1.0]
return True
except Exception as e:
if required: raise RuntimeError(f"LSTM load failed: {e}")
print(f"[WARN] LSTM load failed: {e}")
_LSTM_MODEL, _LSTM_META = None, None
return False
def _minmax_scale_1d(x, vmin, vmax, fr_low, fr_high):
x = np.asarray(x, dtype="float32").reshape(-1,)
denom = (vmax - vmin) if vmax != vmin else 1.0
return (x - vmin) / denom * (fr_high - fr_low) + fr_low
def _minmax_unscale_1d(z, vmin, vmax, fr_low, fr_high):
z = np.asarray(z, dtype="float32").reshape(-1,)
return ((z - fr_low) / (fr_high - fr_low)) * (vmax - vmin) + vmin
def _make_lstm_input(last_close_series: pd.Series, lookback: int, meta: dict) -> np.ndarray:
closes = np.asarray(last_close_series.values, dtype="float32").reshape(-1,)
if len(closes) < lookback:
raise ValueError(f"Not enough history for LSTM: need {lookback}, have {len(closes)}.")
window = closes[-lookback:]
fr_low, fr_high = map(float, meta.get("feature_range", [0.0, 1.0]))
vmin, vmax = float(meta["train_min"]), float(meta["train_max"])
win_scaled = _minmax_scale_1d(window, vmin, vmax, fr_low, fr_high).reshape(lookback, 1)
return np.asarray(win_scaled, dtype="float32").reshape(1, lookback, 1)
def lstm_roll_eval(train_s: pd.Series, test_s: pd.Series):
"""Rolling 1-step evaluation across the TEST window."""
global _LSTM_MODEL, _LSTM_META
model, meta = _LSTM_MODEL, _LSTM_META
lookback = int(meta["lookback"])
fr_low, fr_high = map(float, meta.get("feature_range", [0.0, 1.0]))
vmin, vmax = float(meta["train_min"]), float(meta["train_max"])
full_hist = train_s.copy()
preds, trues = [], []
for i in range(len(test_s)):
x = _make_lstm_input(full_hist, lookback, meta)
yhat_scaled = model.predict(x, verbose=0).reshape(-1)[0]
yhat = _minmax_unscale_1d([yhat_scaled], vmin, vmax, fr_low, fr_high)[0]
preds.append(float(yhat))
trues.append(_as_scalar(test_s.iloc[i]))
# update history with the REAL observed value (fair evaluation)
full_hist = pd.concat([full_hist, test_s.iloc[i:i+1]])
return np.array(trues), np.array(preds)
# =========================
# Metrics & plotting
# =========================
def metrics_df(y_true, y_pred) -> pd.DataFrame:
rmse = math.sqrt(mean_squared_error(y_true, y_pred))
mae = mean_absolute_error(y_true, y_pred)
mape = mean_absolute_percentage_error(y_true, y_pred) * 100
return pd.DataFrame({"RMSE":[rmse], "MAE":[mae], "MAPE (%)":[mape]})
def plot_series(dates, actual, pred, title):
fig, ax = plt.subplots(figsize=(9, 4))
ax.plot(dates, actual, label="Actual", linewidth=2)
ax.plot(dates, pred, label="Prediction", linewidth=2)
ax.set_title(title)
ax.set_xlabel("Date"); ax.set_ylabel("USD ($)")
ax.grid(True, alpha=0.3); ax.legend()
fig.tight_layout()
return fig
# =========================
# Main callback
# =========================
def run_evaluation(ticker_choice: str, start_sel_txt: str, end_sel_txt: str):
"""
Evaluate pretrained ARIMA + LSTM on [start_sel, end_sel] inside the fixed universe.
Show metrics, two plots, and table (Date, Actual, ARIMA, LSTM).
"""
try:
ticker = ticker_choice.strip().upper()
if not ticker:
raise ValueError("Please choose a stock.")
# Parse user-entered dates (text boxes) -> YYYY-MM-DD
def parse_date(s: str) -> date:
return datetime.strptime(s.strip(), "%Y-%m-%d").date()
s_sel = clip_to_universe(parse_date(start_sel_txt))
e_sel = clip_to_universe(parse_date(end_sel_txt))
if s_sel > e_sel:
raise ValueError("Start date must be before or equal to end date.")
# Fetch universe data and build window train/test
df = fetch_universe_data(ticker)
train_s, test_s = build_train_test_for_window(df, s_sel, e_sel)
# Ensure LSTM can run (needs lookback history)
_maybe_load_lstm(required=REQUIRE_LSTM)
lookback = int(_LSTM_META["lookback"])
if len(train_s) < lookback:
raise ValueError(f"Not enough training history for LSTM lookback={lookback}. "
f"Pick an earlier start date.")
# ARIMA
order = extract_order_from_pickle(default_order=(2,1,2))
ar_true, ar_pred = arima_roll_eval(train_s, test_s, order=order)
ar_tbl = metrics_df(ar_true, ar_pred)
# LSTM
lstm_true, lstm_pred = lstm_roll_eval(train_s, test_s)
lstm_tbl = metrics_df(lstm_true, lstm_pred)
# Build aligned outputs
dates = pd.Index(test_s.index).astype(str).tolist()
actual = np.asarray(test_s.values).reshape(-1,)
arima_p = np.asarray(ar_pred).reshape(-1,)
lstm_p = np.asarray(lstm_pred).reshape(-1,)
table = pd.DataFrame({
"Date": dates,
"Actual ($)": np.round(actual, 2),
"ARIMA ($)": np.round(arima_p, 2),
"LSTM ($)": np.round(lstm_p, 2),
})
# Plots
fig_arima = plot_series(pd.to_datetime(dates), actual, arima_p,
f"ARIMA {order} β Actual vs Prediction")
fig_lstm = plot_series(pd.to_datetime(dates), actual, lstm_p,
f"LSTM (lookback={lookback}) β Actual vs Prediction")
# KPI Markdown (fixed format specifiers)
kpi = (
f"**Ticker:** {ticker} | **Window:** {s_sel} β {e_sel} \n"
f"**ARIMA {order}** β RMSE: {ar_tbl['RMSE'][0]:.2f} | MAE: {ar_tbl['MAE'][0]:.2f} | MAPE: {ar_tbl['MAPE (%)'][0]:.2f}% \n"
f"**LSTM (lookback={lookback})** β RMSE: {lstm_tbl['RMSE'][0]:.2f} | MAE: {lstm_tbl['MAE'][0]:.2f} | MAPE: {lstm_tbl['MAPE (%)'][0]:.2f}%"
)
return kpi, fig_arima, fig_lstm, table
except Exception as e:
traceback.print_exc()
return f"β {str(e)}", None, None, pd.DataFrame()
# =========================
# UI
# =========================
with gr.Blocks(title="DataSynthis_ML_JobTask", theme=gr.themes.Soft()) as demo:
gr.Markdown(
f"""
# DataSynthis_ML_JobTask: Stock Forcasting
**Universe:** {UNIVERSE_START} β {UNIVERSE_END}
Pick a stock and a date range **inside the universe**. Weβll evaluate both pretrained models using
**rolling 1-day-ahead** predictions across that range and show metrics, plots, and a comparison table.
"""
)
with gr.Row():
ticker_choice = gr.Dropdown(choices=POPULAR, value="AAPL",
label="Choose a stock")
with gr.Row():
start_sel = gr.Textbox(
label="Start date (YYYY-MM-DD, inside universe)",
value=str(UNIVERSE_START),
placeholder="YYYY-MM-DD"
)
end_sel = gr.Textbox(
label="End date (YYYY-MM-DD, inside universe)",
value=str(UNIVERSE_END),
placeholder="YYYY-MM-DD"
)
run_btn = gr.Button("Run Forcast", variant="primary")
gr.Markdown("### Results")
kpi = gr.Mardown if hasattr(gr, "Mardown") else gr.Markdown # just in case
kpi = gr.Markdown(label="Model scores (lower = better)", value="")
with gr.Row():
chart_arima = gr.Plot(label="ARIMA β Actual vs Predicted")
chart_lstm = gr.Plot(label="LSTM β Actual vs Predicted")
table = gr.Dataframe(label="Details", interactive=False)
run_btn.click(
run_evaluation,
inputs=[ticker_choice, start_sel, end_sel],
outputs=[kpi, chart_arima, chart_lstm, table]
)
with gr.Tab("Help"):
gr.Markdown(
"""
### What happens here?
- We fetch daily closing prices for your stock within the **fixed universe** (2020-10-03 β 2025-10-03).
- For your selected range, we train on all history **before** the start date, then predict **one day ahead** repeatedly
across the selected range (fair rolling evaluation).
- We do this for **both** models (ARIMA and LSTM) and show the metrics, plots, and a detailed table.
### Notes
- If the start date is too close to the universe start, LSTM may not have enough history for its lookback.
Pick an earlier start or reduce lookback when you train your LSTM.
"""
)
demo.launch() |