f/fundlab/factors.py
Greg Pomerantz 9f666538c3 Add CASH axis: money-market/cash-equivalents as a return driver
A fund's yield is a near-constant in return space, so OLS puts it in
the intercept (alpha_ann), not in the shv/bil betas (which only
measure rate-CHANGE sensitivity). Cash funds therefore showed no
dominant driver.

- factors: cash_yield() (shv trailing-1y total return = local
  stand-in for the 13-wk T-bill / 0-3m Treasury index), AXES =
  DRIVERS + ['cash'] (alpha_ann / cash_yield, 1.0 = earns the cash
  rate); a display/clustering axis, not a regressor.
- cluster: cash column in the loading matrix; k-means distances use
  the cash-2x-emphasized matrix (a pure cash fund's level axis was
  otherwise swallowed by the low-exposure cloud); labels use raw
  values; label() cash-aware (0.25 threshold, 1.5x runner-up).
- app: cluster view over AXES with the emphasized distance.

The 331-fund 'no dominant driver' grab-bag now splits into
'cash (yield) +0.64' (n=187, MM + ultra-short) and 'cash (yield)
+1.26' (n=59 pure). 123/123 tests.
2026-08-30 13:27:41 -04:00

221 lines
8.6 KiB
Python

"""Factor screen v2: full OLS of every screened fund on the EXPANDED
sleeve set -> per-fund loading vectors.
This is the "benchmark-neutral" representation: each fund is expressed
as a vector of factor loadings (how much of its return each driver
explains) + an unexplained alpha. Clustering the loadings groups funds
by RETURN DRIVER (fundlab/cluster.py), which is what you want when
shopping for replacements - not a yes/no alpha verdict.
Deliberately OVERINCLUSIVE vs the v1 21-sleeve set: short/long rates,
IG/HY credit, munis, TIPS, preferreds, EM debt, style factors, all 8
major sectors, CTA, commodities. Missing sleeves (JNK, MBS, MUB, IJM,
HFR, XLI, XLC) - no clean long-history proxy in the local DB.
No portfolio-correlation screening anywhere in v2: corr_port is kept
as an information column (the user is considering replacing holdings,
so high-corr funds are REPLACEMENTS, not rejections).
Output: fundlab/factor_results.json (one entry per screened fund:
full + 5y loadings, R2, alpha, t, plus carried screen metadata).
"""
from __future__ import annotations
import json
import time
from functools import lru_cache
from pathlib import Path
import numpy as np
import pandas as pd
from fundlab import decompose
from fundlab.searchlist import BROAD_SLEEVES, PORTFOLIO
HERE = Path(__file__).parent
RESULTS = HERE / "factor_results.json"
SEARCH_ALL = HERE / "search_all.json"
SELECTED = HERE / "universe_cache" / "selected.json"
# expanded sleeve set: the v1 21 + rates/credit/style/sectors/CTA
EXTRA_SLEEVES = [
# short rates (carry axis, separate from T-bills)
"shy", # 1-3y Treasuries
# IG corporate (agg is ~half Treasuries; lqd isolates credit)
"lqd",
# high yield / credit spreads
"hyg", "pff", "emb",
# TIPS (real rates)
"tip",
# US style factors (qqq/ivv blend growth+value)
"vug", "vtv",
# sectors (1998+, overinclusive)
"xlk", "xlf", "xle", "xlv", "xlp", "xlu", "xly", "xlb",
# CTA / managed futures
"dbmf",
# broad commodities (gsg/djp/gld are single-commodity)
"dbb",
]
SLEEVES_V2 = list(dict.fromkeys(BROAD_SLEEVES + EXTRA_SLEEVES))
# regression/cluster basis: drop the >0.95 duplicates (2010+ correlations:
# vea~efa 0.995, qqq~vug 0.976, vug~xlk 0.955). bil/shv/shy stay - they
# finux (Fidelity Intl Bond) TERMINATED 2017 - it breaks the complete-
# case mask for every fund with post-2017 history (the v1 forward
# selector never hit this because sleeves were optional; full OLS needs
# a common sample). Residualize vblix on [ivv, tlt] (tlt~vblix 0.963)
# -> a PURE vol axis.
# bil dropped too: for near-cash funds, plain OLS expresses a tiny net
# short-rate exposure as huge offsetting shv/bil coefficients (shv +64 /
# bil -54) - shv (1-3m) alone carries the cash/ultra-short axis, shy
# keeps the 1-3y axis.
DROP_FROM_REGRESSION = {"vea", "vug", "finux", "bil"}
DRIVERS = [s for s in SLEEVES_V2 if s not in DROP_FROM_REGRESSION]
# clustering/display basis = the regression sleeves + a virtual CASH axis.
# A cash fund's yield is a near-constant in return space, so OLS puts it in
# the INTERCEPT (alpha_ann), not in any beta - BIL/SHV betas only measure
# the fund's response to rate CHANGES. The CASH axis makes the yield
# capture an explicit driver: alpha_ann normalized by the recent cash
# yield (shv trailing 1y total return = the local stand-in for the
# 13-week T-bill / 0-3m Treasury index, SIXM-style).
AXES = DRIVERS + ["cash"]
@lru_cache(maxsize=1)
def cash_yield() -> float:
"""Recent cash level: shv trailing 1y total return (fallback sgov,
then bil). ~3.7% in the current regime."""
for s in ("shv", "sgov", "bil"):
a = decompose.adj_close(s)
if a is None:
continue
year_ago = a[a.index <= a.index[-1] - pd.Timedelta(days=365)]
if len(year_ago) >= 1 and year_ago.iloc[-1] > 0:
return float(a.iloc[-1] / year_ago.iloc[-1] - 1.0)
return 0.03
def _resid(y: pd.Series, X: pd.DataFrame) -> pd.Series:
m = y.notna() & X.notna().all(axis=1)
Xa = np.column_stack([np.ones(m.sum())] + [X[c][m].to_numpy()
for c in X.columns])
beta, *_ = np.linalg.lstsq(Xa, y[m].to_numpy(), rcond=None)
full = np.column_stack([np.ones(len(y))] +
[np.where(m, X[c], np.nan) for c in X.columns])
r = y.to_numpy() - full @ beta
return pd.Series(np.where(m, r, np.nan), index=y.index, name=y.name)
def log(msg: str) -> None:
print(time.strftime("%H:%M:%S"), msg, flush=True)
@lru_cache(maxsize=1)
def _panel() -> pd.DataFrame:
log(f"building driver panel ({len(DRIVERS)} axes)")
p = decompose.returns_panel(SLEEVES_V2).sort_index()
p = p[~p.index.duplicated(keep="last")][DRIVERS]
# pure vol axis: vblix residualized on the core equity/duration axes
core = [c for c in ("ivv", "tlt") if c in p.columns]
if "vblix" in p.columns and core:
p["vblix"] = _resid(p["vblix"], p[core])
return p
# small ridge on the sleeve columns (NOT the intercept): without it,
# near-collinear rate sleeves (shv/bil/shy) leave a near-null direction
# in which cash-like funds get huge offsetting loadings (shv +64, bil -54)
# along an arbitrary axis - garbage for clustering.
RIDGE = 0.02
def _beta_window(y: pd.Series, X: pd.DataFrame, start: str) -> dict | None:
idx = y.index.intersection(X.index[X.index >= pd.to_datetime(start)])
y, X = y.loc[idx], X.loc[idx]
m = y.notna() & X.notna().all(axis=1)
y, X = y[m], X[m]
if len(y) < 250:
return None
cols = [X[c].to_numpy() for c in X.columns]
Xa = np.column_stack([np.ones(len(y))] + cols)
yv = y.to_numpy()
lam = np.eye(len(Xa.T)) * RIDGE
lam[0, 0] = 0.0 # don't shrink the intercept
beta, *_ = np.linalg.lstsq(Xa.T @ Xa + lam, Xa.T @ yv, rcond=None)
resid = yv - Xa @ beta
ssr = float(resid @ resid)
yss = float(((yv - yv.mean()) ** 2).sum())
r2 = 1.0 - ssr / yss if yss > 0 else 0.0
# t for the intercept (unshrunken OLS inference on the raw fit)
fit = decompose.ols(yv, Xa)
return {
"n": int(len(y)),
"betas": {c: float(beta[i + 1]) for i, c in enumerate(X.columns)},
"r2": float(r2),
"alpha_ann": float(fit["beta"][0]) * 252,
"alpha_t": float(fit["t"][0]),
}
return {
"n": int(len(y)),
"betas": {c: float(fit["beta"][i + 1]) for i, c in enumerate(X.columns)},
"r2": float(fit["r2"]),
"alpha_ann": float(fit["beta"][0]) * 252, # same as decompose()
"alpha_t": float(fit["t"][0]),
}
def factor_screen(sym: str, start: str = "1990-01-01") -> dict | None:
fund = decompose.adj_close(sym)
if fund is None or len(fund) < 250:
return None
y = fund.pct_change().dropna()
X = _panel().reindex(y.index)
# sleeves must have history for most of the fund's window
avail = [c for c in X.columns if X[c].notna().mean() > 0.3]
full = _beta_window(y, X[avail], start)
rec = _beta_window(y, X[avail], "2021-01-01")
return {"full": full, "rec5": rec}
def run() -> None:
def _ok(v: dict) -> bool:
f = v.get("full") or v.get("rec5")
return isinstance(f, dict) and "betas" in f
sel = json.loads(SELECTED.read_text())
search_all = (json.loads(SEARCH_ALL.read_text())
if SEARCH_ALL.exists() else {})
out = json.loads(RESULTS.read_text()) if RESULTS.exists() else {}
# resume: skip only GOOD entries (broken/None ones get redone when
# the model changes - this bit us: finux-era None entries were
# skipped forever after finux was dropped)
todo = [r for r in sel if not _ok(out.get(r["sym"], {}))]
log(f"factor screen: {len(sel)} funds, {len(todo)} to do")
for i, r in enumerate(todo):
sym = r["sym"]
try:
res = factor_screen(sym)
except Exception as e:
res = {"error": str(e)}
row = out.get(sym, {})
row.update({"name": r["name"], "local": r["local"],
"alpha_name": r["alpha_name"], **res})
# carry screen metadata (information columns, no screening)
sa = search_all.get(sym, {})
for k in ("corr_portfolio", "corr_benchmark", "verdict",
"alpha_ann_5y", "alpha_t_5y", "alpha_t_full",
"r2_5y", "max_dd", "alpha_pos_frac"):
if isinstance(sa.get(k), (int, float, str)):
row[k] = sa[k]
out[sym] = row
if (i + 1) % 200 == 0:
RESULTS.write_text(json.dumps(out, default=str))
log(f" {i + 1}/{len(todo)}")
RESULTS.write_text(json.dumps(out, default=str))
log(f"factor screen done: {len(out)} funds -> {RESULTS.name}")
if __name__ == "__main__":
run()