CashFlow / scanner /l2_factor.py
lhllamlam's picture
feat: wire 4 institutional flow factors (L2, options, ticks, intraday); stub data committed
6b66ac0 verified
Raw
History Blame Contribute Delete
4.06 kB
"""Level-2 large-resting-order factor.
Methodology (see design doc):
L2 = clip(+0.40 * depth_ratio
+ 0.30 * big_ratio
+ 0.30 * tanh(microprice_dev * 50),
-3, +3)
depth_ratio = dollar-depth on bid side / total top-5 dollar depth
big_ratio = count of "large" (>= BIG_SIZE) bids / total large orders
microprice = (best_bid*ask_sz + best_ask*bid_sz) / (bid_sz + ask_sz)
Spoofing mitigation:
- Only count orders with age_sec > SPOOF_AGE_THRESH (default 1.0s).
- If the book is "lying" (depth-heavy side has price action going the
other way) discount the factor.
Output is a single float, intended to be z-scored cross-sectionally
together with the other 8 factors in :mod:`scanner.scorer`.
"""
from __future__ import annotations
import math
from typing import Optional
from .factor_sources import get_data_source
# Tunable constants
BIG_SIZE = 10_000 # shares - threshold for "large" order
TOP_LEVELS = 5 # top N levels for depth
SPOOF_AGE_THRESH = 1.0 # seconds - orders younger than this ignored
MICROPRICE_SCALE = 50.0 # tanh slope
CLIP_RANGE = 3.0
def _book_lying(
depth_ratio: float, recent_return: float = 0.0
) -> bool:
"""Detect a "lying" book: heavy bid-side but price keeps falling (or
the symmetric case). ``recent_return`` is the 5-min return in decimal
(e.g. -0.005 = -0.5%).
"""
if depth_ratio > 0.6 and recent_return < -0.001:
return True
if depth_ratio < 0.4 and recent_return > 0.001:
return True
return False
def compute_l2_factor(
ticker: str,
recent_return: float = 0.0,
source=None,
) -> float:
"""Compute the Level-2 large-resting-order factor for ``ticker``.
Returns 0.0 if no book is available (caller should treat as missing,
not as a true neutral).
"""
if source is None:
source = get_data_source()
book = source.get_l2_snapshot(ticker)
if not book:
return 0.0
bids = book.get("bids") or []
asks = book.get("asks") or []
if not bids or not asks:
return 0.0
# Filter to stable (non-spoofed) orders
bids_stable = [b for b in bids if len(b) >= 4 and b[3] >= SPOOF_AGE_THRESH]
asks_stable = [a for a in asks if len(a) >= 4 and a[3] >= SPOOF_AGE_THRESH]
if not bids_stable or not asks_stable:
return 0.0
# 1. Depth ratio
depth_bid = sum(p * s for p, s, *_ in bids_stable[:TOP_LEVELS])
depth_ask = sum(p * s for p, s, *_ in asks_stable[:TOP_LEVELS])
total = depth_bid + depth_ask
if total <= 0:
return 0.0
depth_ratio = depth_bid / total
# 2. Large-order count ratio
big_bid = sum(1 for _, s, *_ in bids_stable if s >= BIG_SIZE)
big_ask = sum(1 for _, s, *_ in asks_stable if s >= BIG_SIZE)
big_total = big_bid + big_ask
big_ratio = (big_bid / big_total) if big_total > 0 else 0.5
# 3. Microprice deviation
best_bid, best_bid_sz = bids_stable[0][0], bids_stable[0][1]
best_ask, best_ask_sz = asks_stable[0][0], asks_stable[0][1]
microprice = (best_bid * best_ask_sz + best_ask * best_bid_sz) / (best_bid_sz + best_ask_sz)
mid = (best_bid + best_ask) / 2.0
if mid > 0:
microprice_dev = (microprice - mid) / mid
else:
microprice_dev = 0.0
raw = (
0.40 * (depth_ratio - 0.5) # center: 0 = neutral
+ 0.30 * (big_ratio - 0.5) # center: 0 = neutral
+ 0.30 * math.tanh(microprice_dev * MICROPRICE_SCALE) # already centered
)
# Spoofing discount
if _book_lying(depth_ratio, recent_return):
raw *= 0.3
return max(-CLIP_RANGE, min(CLIP_RANGE, raw))
def compute_l2_factors(
tickers: list[str],
returns: Optional[dict[str, float]] = None,
source=None,
) -> dict[str, float]:
"""Vectorised helper: returns ``{ticker: factor}`` for all tickers."""
returns = returns or {}
if source is None:
source = get_data_source()
return {t: compute_l2_factor(t, returns.get(t, 0.0), source) for t in tickers}