f/fundlab/report.py
Greg Pomerantz 2bbe4e58ec Style-tilt battery + commentary in the fund report pipeline
fundlab/styletilt.py: 22 style/asset sleeves regressed on excess-of-T-bill
returns (full history + 5y); BIC forward selection identifies the tilt
stack; residual-alpha verdict ('factor exposure, not skill' when t<1.75);
data-driven English commentary with sign-specific phrasing. Rendered as a
'Style tilts' block (factor table + prose) in both the app and the HTML
report. All 24 pre-built + 8 ad-hoc fund reports rebuilt.
2026-08-30 20:36:04 -04:00

836 lines
34 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Fund report: candidates + shortlist, self-contained HTML.
One section per fund:
- equity curve over the maximum available period (fund vs its fitted
reference mix vs IVV),
- performance: full history, calendar years, and the defined market
episodes (2022 bear, 2023 rate shock, 2024 vol spike, 2025 tariff
crash, 2026 Q1) - fund vs reference mix vs IVV, with the fund-vs-
reference gap (period alpha/timing),
- what drove it: the excess-of-T-bill sleeve loadings (full + 5y),
alpha + t, R^2, rolling drift, verdict,
- the reference mix explained: each picked sleeve with what it
actually exposes you to (not just the ticker),
- tax character + taxable/IRA placement,
- cluster peers: the 3-4 best funds of the same k=30 return-driver
cluster, compared, with this fund's advantages/disadvantages.
Run: .venv/bin/python -m fundlab.report -> reports/fund_report.html
"""
from __future__ import annotations
import html
import json
import time
from pathlib import Path
import numpy as np
import pandas as pd
from fundlab import cluster as _cl
from fundlab import decompose
from fundlab import factors
from fundlab.searchlist import BROAD_SLEEVES
HERE = Path(__file__).parent
REPORTS = HERE.parent / "reports"
REPORTS.mkdir(exist_ok=True)
OUT = REPORTS / "fund_report.html"
TAX = json.loads((HERE / "taxplan_results.json").read_text())
DD = json.loads((HERE / "drawdown_results.json").read_text())
FACTOR = json.loads((HERE / "factor_results.json").read_text())
SEARCH = json.loads((HERE / "search_all.json").read_text())
DECOMP = json.loads((HERE / "decompose_results.json").read_text())
FONDS = json.loads((HERE.parent / "funds.json").read_text())
KMEANS = json.loads((HERE / "cluster_kmeans.json").read_text())
# ------------------------------------------------------------------ sleeves
# what each sleeve actually EXPOSES you to (the report must explain the
# exposure, not just name the ticker)
SLEEVE_DESC = {
"qqq": ("US large growth (Nasdaq-100)",
"growth/tech-heavy US equities; high sensitivity to earnings "
"surprises and long-end rates (duration of growth cash flows)"),
"ivv": ("US large blend (S&P 500)",
"core US equity market; the default 'own the economy' exposure"),
"iwm": ("US small cap (Russell 2000)",
"small-cap cycle: domestic credit, margin pressure, IPO window"),
"vea": ("Intl developed ex-US (Vanguard)",
"developed-market equities outside the US (EU, Japan, UK); FX-"
"hedged-off, currency moves matter"),
"efa": ("Intl developed ex-US (MSCI EAFE)",
"developed-market equities outside the US; same exposure as VEA "
"via a different index provider"),
"vwo": ("Emerging-market equity",
"EM corporate profits + EM currency + China/FX flows; "
"high-vol, high-carry, dollar-sensitive"),
"vnq": ("US REITs",
"physical real estate: rents vs rates, leverage in the property "
"sector; equity-like income"),
"bil": ("1-3 month T-bills (cash)",
"the risk-free rate itself; ~zero volatility, pure carry"),
"shv": ("0-3 month T-bills (ultra-short cash)",
"the risk-free rate itself; netted out of the excess-return "
"regression, shown only as part of a fund's cash position"),
"shy": ("1-3 year Treasuries (short duration)",
"short duration: modest rate sensitivity, ~cash-plus-carry"),
"ief": ("7-10 year Treasuries (core duration)",
"the core rate bet: price moves when the Fed path changes"),
"tlt": ("20+ year Treasuries (long duration)",
"levered duration: big moves on rate expectations, steepener/"
"bull-steepener exposure"),
"vblix": ("VIX futures (pure vol axis)",
"crash insurance / short-vol funding; positive loading = "
"long-vol (rises in panic), negative = short-vol carry"),
"agg": ("Aggregate bonds (Treasuries + IG credit)",
"the core bond market: ~60% Treasuries, IG corporates, MBS; "
"moderate duration"),
"vweax": ("High-yield corporate bonds",
"credit spread cycle: HY junk yields, default risk in "
"recessions, strong carry in stable times"),
"hyg": ("High-yield corporate bonds (iShares)",
"same HY credit-spread exposure as VWEAX via a different fund"),
"vmbix": ("Agency RMBS (mortgage-backed)",
"mortgage credit + prepayment/extension risk; the refi cycle"),
"lqd": ("Investment-grade corporate bonds",
"IG credit spreads over Treasuries; milder default risk than HY"),
"finux": ("Intl bonds (Fidelity; TERMINATED 2017)",
"non-US credit/duration; data ends 2017 so recent windows "
"never use it"),
"pff": ("Preferred stocks",
"hybrid: fixed-rate equity-like instruments; rate + credit + "
"equity risk in one"),
"emb": ("Emerging-market debt",
"EM sovereign/corporate carry; EM currency + dollar cycle"),
"tip": ("TIPS (inflation-linked Treasuries)",
"breakeven inflation exposure: rises when inflation "
"expectations rise"),
"vtv": ("US value",
"cheap/asset-rich US names (financials, energy, cyclicals); "
"value-vs-growth style cycle"),
"dbmf": ("CTA / managed futures",
"trend-following across futures; long in crises, earns "
"un-correlated carry otherwise"),
"dbb": ("Broad commodities",
"commodity basket: inflation hedge + global growth proxy"),
"djp": ("Natural gas",
"a single volatile commodity: winter/hedging cycles"),
"gsg": ("Broad commodities (SPDR)",
"same commodity exposure as DBB via a different fund"),
"gld": ("Gold",
"crisis/inflation hedge; real-rate sensitive, no yield"),
"fxe": ("Long euros vs the dollar",
"EUR/USD: carries the euro interest-rate differential"),
"fxy": ("Long yen vs the dollar",
"USD/JPY: carries the Japan rate differential; carry-trade "
"crowding risk"),
"xlk": ("US tech sector", "the tech sector index"),
"xlf": ("US financials sector", "banks/insurance: rate + credit cycle"),
"xle": ("US energy sector", "oil prices + US drilling"),
"xlv": ("US healthcare sector", "defensive pharma/providers"),
"xlp": ("US staples sector", "defensive consumer"),
"xlu": ("US utilities sector", "bond-proxy utilities; rate sensitive"),
"xly": ("US consumer discretionary", "cyclicals: autos, retail, "
"leisure; earnings cycle"),
"xlb": ("US materials sector", "industrial materials: capex cycle"),
}
def sleeve_desc(sym: str) -> tuple[str, str]:
return SLEEVE_DESC.get(sym, (sym.upper(), "no description on file"))
# ------------------------------------------------------------------ data
def price(sym: str) -> pd.Series | None:
f = decompose.DATA / f"{sym.lower()}-history.csv"
if not f.exists():
return None
try:
s = (pd.read_csv(f, parse_dates=["Date"], index_col="Date")
["Adj Close"].dropna())
s = s[~s.index.duplicated(keep="last")].sort_index()
except Exception:
return None
return s if len(s) > 40 else None
EPISODES = [] # (label, start, end) from the drawdown engine
for _e in DD["episodes"]:
EPISODES.append((_e["label"], pd.Timestamp(_e["peak"]),
pd.Timestamp(_e["trough"])))
# k=30 cluster membership (sym -> cluster id/label)
MEMBER = {}
for _c, _v in KMEANS["clusters"].items():
for _s in _v["syms"]:
MEMBER[_s] = (_c, _v["label"], _v["n"])
def loading_matrix_cached():
return _cl.loading_matrix()
CENTROIDS = {} # sym -> (cluster, centroid) for NEW points
def _centroids(syms: list[str], V: np.ndarray) -> dict[int, np.ndarray]:
lab = _cl.kmeans(_cl.emphasized(V, _cl.AXES.index("cash")), 30)
out = {}
for c in range(30):
idx = [i for i, l in enumerate(lab) if l == c]
if idx:
out[c] = np.median(V[idx], 0)
return out
def cluster_of_row(row: np.ndarray) -> tuple[str, str]:
"""Assign a new loading row to the k=30 scheme (distance to cluster
medians in the emphasized space)."""
global CENTROIDS
if not CENTROIDS:
syms, V = loading_matrix_cached()
CENTROIDS = _centroids(syms, V)
r = row.copy()
r[_cl.AXES.index("cash")] *= _cl.CASH_EMPHASIS
best, bd = None, None
for c, cent in CENTROIDS.items():
d = float(np.sum((r - cent) ** 2))
if bd is None or d < bd:
best, bd = c, d
for c, v in KMEANS["clusters"].items():
if int(c) == best:
return (c, v["label"])
return (str(best), "?")
def factor_row(sym: str) -> dict:
"""Full + 5y excess-of-T-bill loadings for ANY fund (shortlist funds
are not in the 2,384 factor file)."""
if sym in FACTOR:
return FACTOR[sym]
r = factors.factor_screen(sym)
return r or {}
_FWD: dict = {}
def ref_components(sym: str, fr: dict, window: str = "rec5") -> list[dict]:
"""The fund's FORWARD-SELECTED reference (the mix its alpha is
measured against): BIC-gated selection from the 21 broad sleeves,
excess-of-T-bill. Recomputed (cached) because the screen stored only
the display strings. Falls back to the 34-sleeve OLS loadings when
the fund is not in the search universe."""
key = (sym, window)
if key in _FWD:
return _FWD[key]
out = None
try:
m = decompose.decompose(sym, start=decompose.RECENT_WINDOW
if window == "rec5" else None,
candidates={sym: BROAD_SLEEVES})
if m.get("components"):
out = [{"sym": c["sym"], "beta": c["beta"]}
for c in m["components"]]
except Exception:
out = None
if out is None:
comps = (fr.get(window) or fr.get("full") or {}).get("betas") or {}
out = [{"sym": s, "beta": b} for s, b in sorted(
comps.items(), key=lambda kv: -abs(kv[1])) if abs(b) >= 0.08]
_FWD[key] = out
return out
def search_row(sym: str) -> dict:
return SEARCH.get(sym, {})
def decomp_row(sym: str) -> dict:
return DECOMP.get(sym, {})
def tax_row(sym: str) -> dict:
up = sym.upper()
if up in TAX["shortlist"]:
return TAX["shortlist"][up]
for k in (sym, up, sym.lower()):
if k in TAX["candidates"]:
return TAX["candidates"][k]
return {}
def dd_row(sym: str) -> dict:
return DD["funds"].get(sym, {})
# ------------------------------------------------------------------ stats
def perf_stats(p: pd.Series, rf_annual: float = 0.037) -> dict:
r = p.pct_change().dropna()
years = (p.index[-1] - p.index[0]).days / 365.25
cagr = (p.iloc[-1] / p.iloc[0]) ** (1 / years) - 1 if years > 0.5 else np.nan
vol = r.std() * np.sqrt(252)
roll_max = p.cummax()
mdd = float(((p / roll_max) - 1).min())
sharpe = (cagr - rf_annual) / vol if vol > 0 else np.nan
return {"start": str(p.index[0].date()), "end": str(p.index[-1].date()),
"years": years, "cagr": cagr, "vol": vol, "mdd": mdd,
"sharpe": sharpe}
def window_ret(p: pd.Series, a, b) -> float | None:
s = p[(p.index >= a) & (p.index <= b)]
if len(s) < 2:
return None
return float(s.iloc[-1] / s.iloc[0] - 1)
def annual_table(p: pd.Series, n: int = 8) -> list[tuple[str, float]]:
yr = p.resample("YE").last().dropna()
out = []
for i in range(len(yr) - 1):
lab = str(yr.index[i].year)
if i == 0:
continue
out.append((lab, float(yr.iloc[i] / yr.iloc[i - 1] - 1)))
# partial current year from last full year-end
out.append(("YTD", float(p.iloc[-1] / yr.iloc[-1] - 1)))
return out[-(n + 1):]
def mix_series(refs: list[dict]) -> pd.Series | None:
"""Fitted reference mix: sum of beta * sleeve daily returns, cumulated
to a price path (rebased 100). Pure beta path - the fund minus this
is the alpha path."""
if not refs:
b = price("bil")
if b is None:
return None
return 100 * b / b.iloc[0]
idx = None
for c in refs:
p = price(c["sym"])
if p is None:
continue
r = c["beta"] * p.pct_change()
idx = r if idx is None else idx.combine(r, lambda x, y: x + y,
fill_value=0.0)
if idx is None:
return None
idx = idx.fillna(0.0)
path = (1 + idx).cumprod()
return 100 * path / path.iloc[0]
# ------------------------------------------------------------------ html
PLOTLY_JS = ""
def load_plotly():
global PLOTLY_JS
try:
import plotly
PLOTLY_JS = plotly.offline.get_plotlyjs()
except Exception:
PLOTLY_JS = None
def fig_html(fig, h: int = 420) -> str:
import plotly.io as pio
s = pio.to_html(fig, full_html=False, include_plotlyjs=False,
config={"displayModeBar": False}, div_id="")
return s.replace("<div>", f'<div style="height:{h}px">')
def equity_figure(sym: str, name: str, refs: list[dict]) -> str:
import plotly.graph_objects as go
p = price(sym)
if p is None:
return "<p>No local price history.</p>"
fig = go.Figure()
fig.add_trace(go.Scatter(x=p.index, y=100 * p / p.iloc[0],
name=sym.upper(), line=dict(width=2)))
m = mix_series(refs)
if m is not None:
fig.add_trace(go.Scatter(x=m.index, y=m, name="fitted reference",
line=dict(width=1.2, dash="dash")))
iv = price("ivv")
if iv is not None:
fig.add_trace(go.Scatter(
x=iv.index, y=100 * iv / iv.iloc[0], name="IVV (S&P 500)",
line=dict(width=1, dash="dot"), opacity=0.7))
fig.update_layout(height=430, margin=dict(l=10, r=10, t=30, b=10),
title=f"{html.escape(name)} - total return since "
f"{p.index[0].year} (rebased 100)",
legend=dict(orientation="h", y=1.08),
hovermode="x unified")
return fig_html(fig)
def fmt_pct(x, nd=1, sign=True):
if x is None or (isinstance(x, float) and np.isnan(x)):
return ""
s = f"{100 * x:+.{nd}f}%" if sign else f"{100 * x:.{nd}f}%"
return s
def fmt_r2(x):
return "" if x is None or (isinstance(x, float) and np.isnan(x)) \
else f"{x:.2f}"
def perf_table(p: pd.Series, refs: list[dict]) -> str:
m = mix_series(refs)
iv = price("ivv")
rows = []
def add(label, a, b, ann=False):
fr_ = window_ret(p, a, b)
mr = window_ret(m, a, b) if m is not None else None
ir = window_ret(iv, a, b) if iv is not None else None
gap = (fr_ - mr) if (fr_ is not None and mr is not None) else None
rows.append(f"<tr><td>{label}</td>"
f"<td>{fmt_pct(fr_)}</td><td>{fmt_pct(mr)}</td>"
f"<td>{fmt_pct(ir)}</td><td>{fmt_pct(gap)}</td></tr>")
st = perf_stats(p)
add("Full history", st["start"], st["end"])
add("Last 5y", "2021-01-01", st["end"])
add("Last 1y", "2025-09-01", st["end"])
for lab, a, b in EPISODES:
add(lab, a, b)
# calendar years (last 6)
yr = p.resample("YE").last().dropna()
years = [str(yr.index[i].year) for i in range(1, len(yr))][-6:]
for i, y in enumerate(years):
a = f"{y}-01-01"
b = f"{int(y) + 1}-01-01" if i < len(years) - 1 else st["end"]
add(y, a, b)
return ('<table><tr><th>period</th><th>fund</th><th>reference</th>'
'<th>IVV</th><th>fund ref</th></tr>' + "".join(rows)
+ '</table><p class="small">fund reference = period '
'alpha/timing (the part of that period the sleeve mix does '
'not explain). IVV shown for scale - for non-equity funds '
'the IVV column is only context.</p>')
def drivers_section(fr: dict, dr: dict, srow: dict) -> str:
out = []
if srow:
a5 = srow.get("alpha_ann_5y")
t5 = srow.get("alpha_t_5y")
r25 = srow.get("r2_5y")
if isinstance(r25, (int, float)) or isinstance(a5, (int, float)):
out.append(
f"<p><b>Reference model, last 5 years:</b> R² = "
f"{fmt_r2(r25 if isinstance(r25, (int, float)) else None)}, "
f"alpha = {fmt_pct(a5) if isinstance(a5, (int, float)) else ''}"
f"{' (t = ' + format(t5, '+.1f') + ')' if isinstance(t5, (int, float)) else ''} "
f"vs the fitted reference mix (next section).")
rf = (fr.get("full") or {}).get("alpha_ann")
tf = (fr.get("full") or {}).get("alpha_t")
r2f = srow.get("r2_full")
if isinstance(r2f, (int, float)) or isinstance(rf, (int, float)):
out.append(
f"<p><b>Reference model, full history:</b> R² = "
f"{fmt_r2(r2f if isinstance(r2f, (int, float)) else None)}, "
f"alpha = {fmt_pct(rf) if isinstance(rf, (int, float)) else ''}"
f"{' (t = ' + format(tf, '+.1f') + ')' if isinstance(tf, (int, float)) else ''}.</p>")
f = fr.get("rec5") or {}
betas = f.get("betas") or {}
top = sorted(betas.items(), key=lambda kv: -abs(kv[1]))[:6]
if top:
out.append(
"<p><b>Return-driver signature (34 sleeves, for clustering "
"context):</b> " + ", ".join(
f"{s} {b:+.2f}" for s, b in top) + " - net cash "
f"{1 - sum(betas.values()):+.2f}.</p>")
# drift + verdict from the curated decomposition, when available
if dr and "verdict" in dr:
out.append(f"<p><b>Decomposition verdict:</b> "
f"{html.escape(dr['verdict'])}</p>")
roll = dr.get("rolling") or {}
if roll.get("max_drift") is not None:
out.append(
f"<p>Weight stability: max 1y β-drift = "
f"{roll['max_drift']:.2f} (relative to full-sample β; "
f"0 = perfectly stable, &gt;1 = the weight is unstable).</p>")
note = dr.get("note")
if note:
out.append(f"<p class='small'>{html.escape(note)}</p>")
if srow:
v = str(srow.get("verdict", ""))
if v:
out.append(f"<p><b>Screen verdict:</b> {html.escape(v)}</p>")
return "".join(out)
def reference_section(refs: list[dict], refs_curve: list[dict],
sym: str) -> str:
weak = bool(refs) and not refs_curve
caveat = ""
if weak:
caveat = ("<p class='small'><b>Weak fit - read with care.</b> "
"These loadings are each individually significant but "
"collectively explain little of the excess return; the "
"economically meaningful picture is a mostly-cash "
"vehicle with idiosyncratic alpha (the performance "
"table anchors to cash, not to this mix).</p>")
if not refs:
srow = search_row(sym)
a = srow.get("alpha_ann_5y")
return caveat + ("<p><b>Reference: cash (the T-bill rate itself)."
"</b> No "
"sleeve passed the forward-selection gates, so the fund's "
"excess returns are not explained by any benchmark mix - "
f"its entire excess performance is idiosyncratic (5y alpha "
f"{fmt_pct(a) if isinstance(a, (int, float)) else ''}). "
"There is no meaningful 'beta' to this fund; it is a "
"standalone position.</p>")
sb = sum(c["beta"] for c in refs)
cash = 1 - sb
rows = []
for c in refs:
nm, desc = sleeve_desc(c["sym"])
rows.append(f"<tr><td><b>{c['sym'].upper()}</b> {c['beta']:+.2f}"
f"</td><td>{html.escape(nm)}</td>"
f"<td>{html.escape(desc)}</td></tr>")
cash_line = ""
if weak:
cash_line = caveat + cash_line
if cash > 0.05:
cash_line = (f"<p>The loadings sum to {sb:.2f}, i.e. the fund is "
f"~{cash:.0%} NET CASH (earns the T-bill rate; adds "
f"zero excess alpha).</p>")
elif cash < -0.05:
cash_line = (f"The loadings sum to {sb:.2f}, i.e. the fund is ~"
f"{-cash:.0%} NET LEVERED (borrows at ~the T-bill "
f"rate; that financing shows up as negative cash).</p>")
return ('<table><tr><th>loading</th><th>what it is</th>'
'<th>what it exposes you to</th></tr>' + "".join(rows)
+ "</table>" + cash_line
+ '<p class="small">The reference is NOT one index - it is '
'this fitted mix, rebuilt from the fund\'s own returns. '
'"Alpha" everywhere in this report means outperformance vs '
'this mix, in excess of the T-bill rate.</p>')
def tax_section(sym: str) -> str:
t = tax_row(sym)
if not t:
return "<p>No tax classification on file.</p>"
loc = t.get("location", "?")
basis = t.get("basis", "?")
conf = {"N-PORT": "from actual N-PORT holdings (high confidence)",
"sleeves": "from the return-sleeve mix (model, medium "
"confidence)"}.get(basis, basis)
notes = html.escape(t.get("notes") or "")
rec = {"TAXABLE": "Keep in the <b>taxable</b> account.",
"TAXABLE (munis)": "Keep in the <b>taxable</b> account - "
"tax-exempt interest is wasted in an IRA.",
"TAXABLE (defers to LTCG)": "Keep in the <b>taxable</b> "
"account - the income mostly "
"defers to the LTCG/ROC rate."}.get(
loc, f"Recommended account: <b>{html.escape(loc)}</b>.")
return (f"<p>Character score {t.get('score', '?')} "
f"({conf}). <b>Placement: {rec}</b></p>"
+ (f"<p class='small'>{notes}</p>" if notes else ""))
def cluster_section(sym: str, fr: dict, row: np.ndarray | None) -> str:
srow = search_row(sym)
if sym in MEMBER:
cid, clabel, cn = MEMBER[sym]
members = KMEANS["clusters"][cid]["syms"]
else:
if row is None:
return "<p>Not in the cluster scheme (no loading vector).</p>"
cid, clabel = cluster_of_row(row)
cn = KMEANS["clusters"].get(cid, {}).get("n", 0)
members = KMEANS["clusters"].get(cid, {}).get("syms", [])
peers = []
for s in members:
if s == sym:
continue
v = search_row(s)
if not isinstance(v.get("alpha_t_5y"), (int, float)):
continue
peers.append((v["alpha_t_5y"], v.get("r2_5y") or 0, s, v))
# best peers = highest POSITIVE alpha t (a fund with t = -5 is the
# cluster's worst, not a peer worth copying); top-4, positives first
peers.sort(key=lambda p: (p[0] > 0, p[0]), reverse=True)
pos = [p for p in peers if p[0] > 0]
peers = (pos + [p for p in peers if p[0] <= 0])[:4]
peers = [p[2] for p in peers]
if not peers:
return (f"<p>In cluster <b>{html.escape(clabel)}</b> (n={cn}) but "
"no peer with 5y alpha statistics.</p>")
rows = []
def statline(s: str) -> str:
p = price(s)
if p is None:
return ("", "", "", "", "", "n/a")
st = perf_stats(p)
v = search_row(s)
t5 = window_ret(p, "2021-01-01", st["end"])
r2 = v.get("r2_5y")
a5 = v.get("alpha_ann_5y")
tt = v.get("alpha_t_5y")
tax = (tax_row(s) or {}).get("location") or "n/a"
return (fmt_pct(t5), fmt_pct(st["cagr"]), fmt_pct(st["mdd"]),
fmt_r2(r2),
(f"{fmt_pct(a5)} (t={tt:+.1f})"
if isinstance(a5, (int, float)) else ""), tax)
p = price(sym)
st = perf_stats(p) if p is not None else {}
my = statline(sym)
rows.append(f"<tr class='me'><td><b>{sym.upper()} (this fund)</b></td>"
+ "".join(f"<td>{x}</td>" for x in my) + "</tr>")
for s in peers:
rows.append(f"<tr><td>{s.upper()}"
f"{html.escape((search_row(s) or {}).get('name', '')[:40])}"
f"</td>" + "".join(f"<td>{x}</td>" for x in statline(s))
+ "</tr>")
tbl = ('<table><tr><th>fund</th><th>5y</th><th>CAGR</th>'
'<th>maxDD</th><th>R² 5y</th><th>alpha 5y</th><th>tax</th></tr>'
+ "".join(rows) + "</table>")
# advantages / disadvantages: computed deltas vs the peer set
adv, dis = [], []
if p is not None:
t5 = window_ret(p, "2021-01-01", st["end"])
vals = {}
for s in peers:
pp = price(s)
if pp is None:
continue
ss = perf_stats(pp)
vals[s] = (window_ret(pp, "2021-01-01", ss["end"]), ss["mdd"],
ss["vol"])
if vals:
best_t5 = max(v[0] for v in vals.values() if v[0] is not None)
best_dd = max(v[1] for v in vals.values())
low_vol = min(v[2] for v in vals.values())
if t5 is not None and t5 >= best_t5 - 0.02:
adv.append("5y return at the top of the cluster")
elif t5 is not None and t5 < best_t5 - 0.10:
dis.append(f"5y return trails the best peer by "
f"{100 * (best_t5 - t5):.0f}pp")
if st["mdd"] > best_dd + 0.05:
dis.append(f"deeper drawdown than the calmest peer "
f"({fmt_pct(st['mdd'])} vs {fmt_pct(best_dd)})")
elif st["mdd"] < best_dd - 0.05:
adv.append("sharpest drawdown in the cluster")
if st["vol"] < low_vol - 0.02:
adv.append("lowest volatility in the cluster")
elif st["vol"] > low_vol + 0.05:
dis.append("meaningfully more volatile than the "
"calmest peer")
txt = ""
if adv:
txt += "<p><b>Advantages vs peers:</b> " + "; ".join(adv) + ".</p>"
if dis:
txt += "<p><b>Disadvantages vs peers:</b> " + "; ".join(dis) + ".</p>"
if not txt:
txt = ("<p>The fund sits in the middle of its cluster - no "
"decisive edge on return, drawdown or volatility vs the "
"peers; the choice among them should come down to alpha "
"quality (t-stat), tax fit and the conviction in the "
"strategy.</p>")
return (f"<p>Cluster: <b>{html.escape(clabel)}</b> "
f"(n={cn}, k=30 grouping by return-driver signature).</p>"
+ tbl + txt)
def fund_loading_row(fr: dict) -> np.ndarray:
f = fr.get("full") or fr.get("rec5") or {}
betas = f.get("betas") or {}
row = np.array([betas.get(s, 0.0) or 0.0 for s in factors.DRIVERS],
dtype=float)
return np.append(row, 1.0 - row.sum())
def strategy_block(sym: str) -> str:
f = FONDS.get(sym)
if not f:
return ""
out = ""
if f.get("strategy"):
out += f"<details><summary>Strategy (excerpt from the filing)</summary>"
f"<p class='small'>{html.escape(f['strategy'])}</p></details>"
return out
def style_block(sym: str) -> str:
from fundlab import styletilt
prof = styletilt.style_profile(sym)
if not prof or not prof.get("full"):
return ""
f5 = {x["sym"]: x for x in (prof.get("rec5") or {}).get("factors", [])}
rows = []
for x in prof["full"]["factors"]:
y5 = f5.get(x["sym"], {})
if abs(x["t"]) < 2 and abs(y5.get("t", 0) or 0) < 2:
continue
rows.append(
f"<tr><td>{html.escape(x['name'])}</td>"
f"<td>{x['beta']:+.2f}</td><td>{x['t']:+.1f}</td>"
f"<td>{y5['beta']:+.2f}</td><td>{y5['t']:+.1f}</td>"
f"<td>{'&#10003;' if x['sym'] in prof.get('identified', []) else ''}</td></tr>")
tbl = ('<table><tr><th>factor</th><th>β full</th><th>t full</th>'
'<th>β 5y</th><th>t 5y</th><th>identified</th></tr>'
+ "".join(rows) + "</table>")
para = "".join(f"<p>{html.escape(p)}</p>" for p in prof.get("commentary", []))
return ('<h4>Style tilts</h4>'
'<p class="small">22 style/asset sleeves regressed on the '
'excess returns of the fund (US funds proxy the global '
'factors); "identified" = survived the BIC forward-selection '
'gate on the full history.</p>' + tbl + para)
def narrative_block(sym: str) -> str:
from fundlab.narrative import narrate
return "".join(f"<p>{html.escape(par)}</p>" for par in narrate(sym))
# ------------------------------------------------------------------ build
def fund_section(sym: str, title: str, subtitle: str) -> str:
fr = factor_row(sym)
dr = decomp_row(sym)
srow = search_row(sym)
name = (srow.get("name") or (FONDS.get(sym) or {}).get("name")
or (fr or {}).get("name") or sym.upper())
refs = ref_components(sym, fr)
load = fund_loading_row(fr) if fr else None
# weak-fit (idiosyncratic) funds: the forward-selected mix is a
# statistically-thin spec combination (offsetting VIX/duration legs);
# anchoring the table to it misleads (a "reference" that loses 79% in
# 2022 for a fund that lost 2.4%). Below the fit gate the honest
# reference is CASH - the fund IS ~net-cash + idiosyncratic alpha.
r25 = srow.get("r2_5y") if isinstance(srow.get("r2_5y"),
(int, float)) else None
r2f = (fr.get("rec5") or fr.get("full") or {}).get("r2")
fit = r25 if r25 is not None else r2f
refs_curve = refs if (fit is not None and fit >= 0.5) else []
return f"""
<h3 id="{sym}">{html.escape(title)}
<span class="sub">{html.escape(subtitle)}</span></h3>
<div class="fund">{strategy_block(sym)}
<h4>Discussion</h4>
{narrative_block(sym)}
{equity_figure(sym, name, refs_curve)}
<h4>Performance</h4>
{perf_table(price(sym), refs_curve) if price(sym) is not None else '<p>no price data</p>'}
<h4>What drove the returns</h4>
{drivers_section(fr, dr, srow)}
{style_block(sym)}
<h4>The reference mix - and what it exposes you to</h4>
{reference_section(refs, refs_curve, sym)}
<h4>Tax character &amp; placement</h4>
{tax_section(sym)}
<h4>Peer comparison (same return-driver cluster)</h4>
{cluster_section(sym, fr, load)}
</div>"""
def main() -> None:
t0 = time.time()
load_plotly()
print("building report ...", flush=True)
cand = [v for v in SEARCH.values()
if isinstance(v, dict)
and str(v.get("verdict", "")).startswith("CANDIDATE")]
cand.sort(key=lambda v: -(v.get("alpha_t_5y")
if isinstance(v.get("alpha_t_5y"),
(int, float)) else -9))
# shortlist: the 16 actively-managed funds (3 are share classes of
# the same fund: pmfkx=pmaix, lcrix=lcorx, egrsx=eagmx)
from fundlab.decompose import ALIAS
from fundlab.nport import FUND_TOKENS
short = [s for s in FUND_TOKENS if s not in ALIAS]
toc = []
body = []
for i, v in enumerate(cand, 1):
s = v["sym"]
toc.append(f'<li><a href="#{s}">{s.upper()}</a> — '
f'{html.escape((v.get("name") or "")[:48])}</li>')
body.append(fund_section(
s, f"C{i:02d} · {s.upper()}",
f"{v.get('name', '')}{str(v.get('verdict', ''))[:60]}"))
for i, s in enumerate(short, 1):
nm = (FONDS.get(s) or {}).get("name", s.upper())
toc.append(f'<li><a href="#{s}">S{i:02d} · {s.upper()}</a> — '
f'{html.escape(nm[:48])}</li>')
body.append(fund_section(s, f"S{i:02d} · {s.upper()}", nm))
doc = f"""<!DOCTYPE html>
<html><head><meta charset="utf-8">
<title>Fund report - candidates &amp; shortlist</title>
<style>
body {{ font-family: -apple-system, 'Segoe UI', Helvetica, Arial, sans-serif;
margin: 0; color: #1a1a2e; background: #fafafa; }}
.wrap {{ display: flex; }}
nav {{ width: 270px; min-width: 270px; position: sticky; top: 0;
height: 100vh; overflow-y: auto; background: #fff;
border-right: 1px solid #ddd; padding: 16px; box-sizing: border-box; }}
main {{ flex: 1; padding: 24px 36px; max-width: 1150px; }}
h1 {{ font-size: 26px; }} h3 {{ border-top: 3px solid #1a1a2e;
padding-top: 18px; margin-top: 40px; }}
h3 .sub {{ font-weight: normal; font-size: 14px; color: #555;
display: block; margin-top: 4px; }}
h4 {{ margin: 18px 0 6px; font-size: 15px; }}
table {{ border-collapse: collapse; margin: 8px 0 14px; font-size: 13px; }}
th, td {{ border: 1px solid #ccc; padding: 4px 9px; text-align: left; }}
th {{ background: #eee; }}
tr.me td {{ background: #fff8e1; }}
.small {{ color: #555; font-size: 12.5px; }}
nav a {{ text-decoration: none; color: #1a1a2e; font-size: 13px;
display: block; padding: 3px 0; }}
nav a:hover {{ color: #b00020; }}
nav h2 {{ font-size: 14px; margin: 14px 0 6px; }}
details {{ margin: 6px 0; }} summary {{ cursor: pointer;
font-size: 13px; color: #333; }}
</style>
{"<script>" + PLOTLY_JS + "</script>" if PLOTLY_JS else
'<script src="https://cdn.plot.ly/plotly-2.35.2.min.js"></script>'}
</head><body><div class="wrap">
<nav>
<b>Return-driver report</b><br><span class="small">generated
{time.strftime('%Y-%m-%d %H:%M')}</span>
<h2>Candidates (alpha screen, excess of T-bill)</h2>
<ol>{''.join(toc[:len(cand)])}</ol>
<h2>Shortlist (your funds)</h2>
<ol start="{len(cand) + 1}">{''.join(toc[len(cand):])}</ol>
</nav>
<main>
<h1>Fund report - {len(cand)} alpha candidates &amp; {len(short)}
shortlist funds</h1>
<p class="small"><b>Method.</b> Every alpha in this report is computed
<b>in excess of the 3-month T-bill rate</b> (BIL total return as the
local risk-free series): the fund's daily total returns are regressed on
sleeve benchmarks that are netted against the same rate, so a cash
position contributes exactly zero. "Reference" is never one index - it
is each fund's <b>own fitted sleeve mix</b> (BIC forward selection on the
broad axes, or the full 34-sleeve OLS for the loadings). R² measures how
much of the excess return the mix explains; alpha is what is left.
"Net cash" = 1 (sum of loadings). Equity curves are total-return
(Adj Close) over the maximum local history, rebased to 100.</p>
{''.join(body)}
<p class="small" style="margin-top:40px">Generated by fundlab.report -
data as of {time.strftime('%Y-%m-%d')}. Local price histories may lag a
day or two. Cluster = k=30 k-means on the excess-return loading vectors
(34 sleeves + net-cash axis).</p>
</main></div></body></html>"""
OUT.write_text(doc)
print(f"wrote {OUT} in {time.time() - t0:.0f}s "
f"({len(doc) / 1e6:.1f} MB)")
if __name__ == "__main__":
main()