199 lines
12 KiB
Python
199 lines
12 KiB
Python
"""Deterministic price, breadth, relative-strength and market indicators."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import numpy as np
|
|
import pandas as pd
|
|
|
|
|
|
def price_features(bars: pd.DataFrame, *, slope_lookback: int = 5,
|
|
atr_period: int = 14, adr_period: int = 20) -> pd.DataFrame:
|
|
df = bars.copy().sort_index()
|
|
for column in ("open", "high", "low", "close", "volume"):
|
|
if column not in df:
|
|
raise ValueError(f"missing required column: {column}")
|
|
out = df.copy()
|
|
out["ema21_high"] = df["high"].ewm(span=21, adjust=False, min_periods=21).mean()
|
|
out["ema21_close"] = df["close"].ewm(span=21, adjust=False, min_periods=21).mean()
|
|
out["ema21_low"] = df["low"].ewm(span=21, adjust=False, min_periods=21).mean()
|
|
prior_close = df["close"].shift(1)
|
|
out["tr"] = pd.concat([df["high"] - df["low"], (df["high"] - prior_close).abs(),
|
|
(df["low"] - prior_close).abs()], axis=1).max(axis=1)
|
|
out["atr14"] = out["tr"].rolling(atr_period, min_periods=atr_period).mean()
|
|
day_range_pct = (df["high"] - df["low"]) / prior_close.replace(0, np.nan) * 100
|
|
out["adr20_pct"] = day_range_pct.rolling(adr_period, min_periods=adr_period).mean()
|
|
out["sma50"] = df["close"].rolling(50, min_periods=50).mean()
|
|
out["sma200"] = df["close"].rolling(200, min_periods=200).mean()
|
|
out["high_52w"] = df["high"].rolling(252, min_periods=252).max()
|
|
out["low_52w"] = df["low"].rolling(252, min_periods=252).min()
|
|
out["avg_volume20"] = df["volume"].rolling(20, min_periods=20).mean()
|
|
out["avg_value_turnover20"] = (df["close"] * df["volume"]).rolling(20, min_periods=20).mean()
|
|
out["extension_atr"] = (df["close"] - out["ema21_high"]) / out["atr14"].replace(0, np.nan)
|
|
d_high = out["ema21_high"].diff(slope_lookback)
|
|
d_close = out["ema21_close"].diff(slope_lookback)
|
|
d_low = out["ema21_low"].diff(slope_lookback)
|
|
out["trend_state"] = np.select(
|
|
[(d_high > 0) & (d_close > 0) & (d_low > 0), (d_high < 0) & (d_close < 0) & (d_low < 0)],
|
|
["UP", "DOWN"], default="NEUTRAL")
|
|
out["price_state"] = np.select(
|
|
[df["close"] > out["ema21_high"], df["close"] < out["ema21_low"]],
|
|
["ABOVE_STRUCTURE", "BELOW_STRUCTURE"], default="INSIDE_STRUCTURE")
|
|
weekly = df[["high", "low", "close"]].resample("W-FRI").agg(
|
|
{"high": "max", "low": "min", "close": "last"}).dropna()
|
|
for column in ("high", "close", "low"):
|
|
weekly[f"weekly_ema10_{column}"] = weekly[column].ewm(span=10, adjust=False, min_periods=10).mean()
|
|
wh, wc, wl = (weekly[f"weekly_ema10_{c}"].diff() for c in ("high", "close", "low"))
|
|
weekly["weekly_trend_state"] = np.select(
|
|
[(wh > 0) & (wc > 0) & (wl > 0), (wh < 0) & (wc < 0) & (wl < 0)],
|
|
["UP", "DOWN"], default="NEUTRAL")
|
|
weekly_fields = [f"weekly_ema10_{c}" for c in ("high", "close", "low")] + ["weekly_trend_state"]
|
|
out[weekly_fields] = weekly[weekly_fields].reindex(df.index, method="ffill")
|
|
out["volume_vs_avg20"] = df["volume"] / out["avg_volume20"].replace(0, np.nan)
|
|
return out
|
|
|
|
|
|
def relative_strength(prices: dict[str, pd.Series], benchmark: pd.Series,
|
|
weights: dict[str, float] | None = None,
|
|
set_benchmark: pd.Series | None = None,
|
|
high_prices: dict[str, pd.Series] | None = None,
|
|
members_by_date: dict[pd.Timestamp, list[str]] | None = None) -> pd.DataFrame:
|
|
"""Cross-sectional percentile rank from price returns; not proprietary TradersLab RS."""
|
|
weights = weights or {"1m": .20, "3m": .35, "12m": .30, "52w": .15}
|
|
if set(weights) != {"1m", "3m", "12m", "52w"} or not np.isclose(sum(weights.values()), 1.0):
|
|
raise ValueError("weights must include 1m/3m/12m/52w and sum to 1")
|
|
dates = benchmark.index
|
|
raw: dict[str, pd.DataFrame] = {}
|
|
for label, periods in {"1m": 21, "3m": 63, "12m": 252}.items():
|
|
returns = pd.DataFrame({symbol: series.reindex(dates).pct_change(periods) for symbol, series in prices.items()})
|
|
raw[label] = returns.sub(benchmark.reindex(dates).pct_change(periods), axis=0)
|
|
highs = high_prices or prices
|
|
raw["52w"] = pd.DataFrame({symbol: prices[symbol].reindex(dates) /
|
|
highs[symbol].reindex(dates).rolling(252, min_periods=252).max()
|
|
for symbol in prices})
|
|
if members_by_date is None:
|
|
pct = {key: value.rank(axis=1, pct=True) * 100 for key, value in raw.items()}
|
|
else:
|
|
pct = {}
|
|
for key, values in raw.items():
|
|
rankable = values.copy()
|
|
for date in rankable.index:
|
|
members = set(members_by_date.get(pd.Timestamp(date), []))
|
|
rankable.loc[date, ~rankable.columns.isin(members)] = np.nan
|
|
pct[key] = rankable.rank(axis=1, pct=True) * 100
|
|
score = sum(pct[key] * weight for key, weight in weights.items())
|
|
score.columns = [f"{symbol}_rs_percentile" for symbol in score.columns]
|
|
for key, frame in pct.items():
|
|
frame.columns = [f"{symbol}_rs_{key}" for symbol in frame.columns]
|
|
score = score.join(frame)
|
|
for label in ("1m", "3m", "12m"):
|
|
for symbol in prices:
|
|
score[f"{symbol}_rs_{label}_vs_set100ew"] = raw[label][symbol]
|
|
for symbol in prices:
|
|
score[f"{symbol}_52w_proximity_pct"] = raw["52w"][symbol] * 100
|
|
if set_benchmark is not None:
|
|
for label, periods in {"1m": 21, "3m": 63, "12m": 252}.items():
|
|
set_returns = pd.DataFrame({symbol: series.reindex(dates).pct_change(periods)
|
|
for symbol, series in prices.items()})
|
|
relative = set_returns.sub(set_benchmark.reindex(dates).pct_change(periods), axis=0)
|
|
for symbol in relative.columns:
|
|
score[f"{symbol}_rs_{label}_vs_set"] = relative[symbol]
|
|
return score
|
|
|
|
|
|
def breadth_series(closes: pd.DataFrame, members_by_date: dict[pd.Timestamp, list[str]]) -> pd.DataFrame:
|
|
changes = closes.pct_change(fill_method=None)
|
|
rows = []
|
|
for date in closes.index:
|
|
members = members_by_date.get(pd.Timestamp(date), [])
|
|
current = changes.loc[date, [symbol for symbol in members if symbol in changes.columns]].dropna()
|
|
advancing, declining = int((current > 0).sum()), int((current < 0).sum())
|
|
rows.append({"advancing": advancing, "declining": declining,
|
|
"unchanged": int((current == 0).sum()),
|
|
"net_advances": advancing - declining, "issues": int(len(current))})
|
|
return pd.DataFrame(rows, index=closes.index)
|
|
|
|
|
|
def mcclellan(breadth: pd.DataFrame, *, short_span: int = 19, long_span: int = 39,
|
|
z_window: int = 252) -> pd.DataFrame:
|
|
"""MCO=EMA19(net advances)-EMA39(net advances); MCSI=cumulative MCO."""
|
|
net = breadth["net_advances"].astype(float)
|
|
out = pd.DataFrame(index=breadth.index)
|
|
out["mco_ema_short"] = net.ewm(span=short_span, adjust=False, min_periods=short_span).mean()
|
|
out["mco_ema_long"] = net.ewm(span=long_span, adjust=False, min_periods=long_span).mean()
|
|
out["mco"] = out["mco_ema_short"] - out["mco_ema_long"]
|
|
minimum = min(z_window, 20)
|
|
out["mco_mean"] = out["mco"].rolling(z_window, min_periods=minimum).mean()
|
|
out["mco_std"] = out["mco"].rolling(z_window, min_periods=minimum).std(ddof=0)
|
|
out["mco_z"] = (out["mco"] - out["mco_mean"]) / out["mco_std"].replace(0, np.nan)
|
|
out["mcsi"] = out["mco"].fillna(0).cumsum()
|
|
out["mcsi_10dma"] = out["mcsi"].rolling(10, min_periods=10).mean()
|
|
prev, prev_ma = out["mcsi"].shift(1), out["mcsi_10dma"].shift(1)
|
|
rising, falling = out["mcsi"] > prev, out["mcsi"] < prev
|
|
reclaim = (out["mcsi"] > out["mcsi_10dma"]) & (prev <= prev_ma)
|
|
lose = (out["mcsi"] < out["mcsi_10dma"]) & (prev >= prev_ma)
|
|
out["mcsi_state"] = np.select(
|
|
[reclaim, lose, (out["mcsi"] > out["mcsi_10dma"]) & rising,
|
|
(out["mcsi"] < out["mcsi_10dma"]) & falling, rising, falling],
|
|
["RECLAIM_10DMA", "LOSE_10DMA", "UP", "DOWN", "CURL_UP", "CURL_DOWN"], default="DOWN")
|
|
return out
|
|
|
|
|
|
def equal_weight_index(closes: pd.DataFrame, members_by_date: dict[pd.Timestamp, list[str]],
|
|
base_value: float = 100.0) -> pd.Series:
|
|
"""Daily-rebalanced price-return proxy, equal-weighted over effective members."""
|
|
returns = closes.pct_change(fill_method=None)
|
|
daily_returns = []
|
|
counts = []
|
|
for date in closes.index:
|
|
members = [symbol for symbol in members_by_date.get(pd.Timestamp(date), []) if symbol in returns.columns]
|
|
daily = returns.loc[date, members].dropna() if members else pd.Series(dtype=float)
|
|
daily_returns.append(float(daily.mean()) if len(daily) else 0.0)
|
|
counts.append(len(daily))
|
|
result = pd.Series(base_value, index=closes.index, dtype=float) * (1 + pd.Series(daily_returns, index=closes.index)).cumprod()
|
|
result.name = "set100ew"
|
|
result.attrs["constituent_counts"] = counts
|
|
return result
|
|
|
|
|
|
def market_regimes(market: pd.DataFrame, *, overbought_extension_atr: float = 1.5,
|
|
repair_sigma: float = 1.0) -> pd.DataFrame:
|
|
"""Deterministic daily market state machine with reasons, no language model."""
|
|
out = market.copy()
|
|
regimes, reasons_all = [], []
|
|
previous = "CORRECTION"
|
|
for _, row in out.iterrows():
|
|
close, ema_high, ema_low = (row.get(key, np.nan) for key in ("close", "ema21_high", "ema21_low"))
|
|
sma50, ew_close, ew_ema = (row.get(key, np.nan) for key in ("sma50", "set100ew_close", "set100ew_ema21"))
|
|
trend, mcsi_state = row.get("trend_state", "NEUTRAL"), row.get("mcsi_state", "DOWN")
|
|
mco_z, net, extension = row.get("mco_z", np.nan), row.get("net_advances", 0), row.get("extension_atr", 0.0)
|
|
above_structure = pd.notna(close) and pd.notna(ema_high) and close > ema_high
|
|
below_structure = pd.notna(close) and pd.notna(ema_low) and close < ema_low
|
|
above_50 = pd.notna(close) and pd.notna(sma50) and close > sma50
|
|
ew_support = pd.notna(ew_close) and pd.notna(ew_ema) and ew_close > ew_ema
|
|
repaired = bool(above_structure and ew_support)
|
|
breakdown = bool(below_structure and trend == "DOWN" and mcsi_state in {"DOWN", "LOSE_10DMA"})
|
|
reasons: list[str] = []
|
|
if breakdown:
|
|
regime, reasons = "BREAKDOWN", ["SET below falling 21DMA structure", "MCSI confirms downside"]
|
|
elif pd.notna(extension) and extension > overbought_extension_atr and net > 0:
|
|
regime, reasons = "OVERBOUGHT", [f"SET extension exceeds {overbought_extension_atr:g} ATR", "breadth net advances positive"]
|
|
elif above_structure and above_50 and ew_support and trend == "UP" and mcsi_state in {"UP", "RECLAIM_10DMA"}:
|
|
regime, reasons = "CONFIRMED_UPTREND", ["SET above rising 21DMA and 50DMA", "SET100EW above 21DMA", "MCSI supportive"]
|
|
elif trend == "UP" and above_50 and not above_structure and ew_support:
|
|
regime, reasons = "UPTREND_PULLBACK", ["SET pullback holds 21DMA structure", "SET100EW remains above 21DMA"]
|
|
elif repaired and (mcsi_state in {"CURL_UP", "RECLAIM_10DMA", "UP"} or (pd.notna(mco_z) and mco_z < -repair_sigma)):
|
|
regime = "EARLY_UPTREND" if previous in {"REPAIR", "CORRECTION", "BREAKDOWN"} else "REPAIR"
|
|
reasons = ["SET and SET100EW reclaimed 21DMA", f"MCSI {mcsi_state.lower()}"]
|
|
elif (pd.notna(mco_z) and mco_z < -repair_sigma) or mcsi_state in {"CURL_UP", "RECLAIM_10DMA"}:
|
|
regime, reasons = "REPAIR", [f"MCO below -{repair_sigma:g} sigma" if pd.notna(mco_z) and mco_z < -repair_sigma else f"MCSI {mcsi_state.lower()}"]
|
|
else:
|
|
regime, reasons = "CORRECTION", ["SET/SET100EW confirmation absent"]
|
|
if net > 0:
|
|
reasons.append("breadth net advances positive")
|
|
regimes.append(regime)
|
|
reasons_all.append(reasons)
|
|
previous = regime
|
|
out["regime"], out["regime_reasons"] = regimes, reasons_all
|
|
out["new_risk_allowed"] = ~out["regime"].isin(["CORRECTION", "OVERBOUGHT", "BREAKDOWN"])
|
|
return out
|