Fix Yahoo total-distribution double count (1,957 syms)

Yahoo's dividend endpoint returns the fund's TOTAL per-share distribution
(dividend + capital gain) for many share classes while the capitalGains
endpoint returns the cap-gain portion separately; summing both double-counted
cap gains and broke pre/post-tax comparability (after-tax engine could beat
the adj-based pre-tax return: impossible).

Add _fix_total_distributions to the bundle-assembly invariants
(layout-agnostic, survives re-downloads): on same-date div>capg events, the
exact Yahoo-adj implied distribution (d = P - Q*A_{t-1}/A_t) must match the
div-file amount (not the sum) on >=3 dates before rewriting div -= capg.
Self-validating: genuine separate same-date distributions are untouched.
Rewrote 17,900 cells on 1,957 symbols.

Pool check (scripts/check_adj_consistency.py): all 14 symbols now
TR(close+events) == TR(adj) to 0.01pt and post < pre with plausible drags.

Tests: rewrite case, no-rewrite case, idempotency (tests/test_data.py).
This commit is contained in:
Greg Pomerantz 2026-09-01 16:23:03 -04:00
parent 6c147cc850
commit 5e5a10d725
4 changed files with 301 additions and 2 deletions

60
data.py
View File

@ -24,6 +24,7 @@ import time
from dataclasses import dataclass from dataclasses import dataclass
from pathlib import Path from pathlib import Path
import numpy as np
import pandas as pd import pandas as pd
DEFAULT_ROOT = Path("~/prog/fin/stocks").expanduser() DEFAULT_ROOT = Path("~/prog/fin/stocks").expanduser()
@ -99,7 +100,59 @@ def dedupe_event_rows(corr: dict, div: list[tuple[str, float]],
return out_div, out_capg return out_div, out_capg
def apply_invariants(div: pd.DataFrame, capg: pd.DataFrame) -> None: def _fix_total_distributions(div: pd.DataFrame, capg: pd.DataFrame,
close: pd.DataFrame, adj: pd.DataFrame) -> list[str]:
"""Fix Yahoo's "dividend file carries the TOTAL distribution" quirk.
For many fund share classes Yahoo's dividend endpoint reports the fund's
TOTAL per-share distribution (dividend + capital gain) while the
capitalGains endpoint reports the cap-gain portion separately; summing
both files double-counts the cap-gain part and breaks pre/post-tax
comparability (the after-tax engine can even beat the adj-based pre-tax
return, which is impossible for consistent data).
Detection is layout-agnostic and self-validating: on dates where both
files have an event and the dividend amount strictly exceeds the cap-gain
amount, Yahoo's adj factor steps by an implicit total distribution (the
close/adj ratio only steps on distribution dates). If that implicit
amount matches the dividend-file amount (not the sum) on most overlap
dates, the dividend file is the total and we rewrite div -= capg there,
leaving div+capg == total. Symbols with genuine separate same-date
distributions (adj matches the sum) are left untouched.
Returns the list of rewritten symbols.
"""
fixed: list[str] = []
for sym in div.columns:
if sym not in capg.columns or sym not in close.columns or sym not in adj.columns:
continue
dv = div[sym]
cg = capg[sym].reindex(div.index).fillna(0.0)
mask = (dv > 0) & (cg > 0) & (dv > cg * 1.001)
n = int(mask.sum())
if n < 3:
continue
c, a = close[sym], adj[sym]
if (a.fillna(0) == 0).all() or (c.fillna(0) == 0).all():
continue
# Yahoo adj convention: A_t = A_{t-1} * Q_t / (P_{t-1} - d) =>
# exact implied total distribution: d = P - Q * (A_{t-1} / A_t)
imp = (c.shift(1) - c * (a.shift(1) / a)).reindex(div.index)
d_s = dv[mask]
g_s = cg[mask]
imp = imp[mask]
ref = np.maximum(imp.abs(), d_s.abs())
ref = ref.replace(0, np.nan)
m_tot = (imp - d_s).abs() <= 0.05 * ref
m_both = (imp - (d_s + g_s)).abs() <= 0.05 * ref
if int(m_tot.sum()) >= 3 and int(m_tot.sum()) > int(m_both.sum()):
div.loc[mask, sym] = d_s - g_s
fixed.append(sym)
return fixed
def apply_invariants(div: pd.DataFrame, capg: pd.DataFrame,
adj: pd.DataFrame | None = None,
close: pd.DataFrame | None = None) -> None:
"""Cross-file correction invariants (Yahoo double-listing), applied at """Cross-file correction invariants (Yahoo double-listing), applied at
bundle assembly so they hold for full and incremental builds and survive bundle assembly so they hold for full and incremental builds and survive
re-downloads that move a row between the dividend/capitalGain files. re-downloads that move a row between the dividend/capitalGain files.
@ -126,6 +179,8 @@ def apply_invariants(div: pd.DataFrame, capg: pd.DataFrame) -> None:
abs(div.at[ts, sym] - a) < 1e-9 and \ abs(div.at[ts, sym] - a) < 1e-9 and \
abs(capg.at[ts, sym] - a) < 1e-9: abs(capg.at[ts, sym] - a) < 1e-9:
div.at[ts, sym] = 0.0 div.at[ts, sym] = 0.0
if adj is not None and close is not None:
_fix_total_distributions(div, capg, close, adj)
def _apply_corrections(sym: str, suffix: str, ser: pd.Series, def _apply_corrections(sym: str, suffix: str, ser: pd.Series,
@ -540,7 +595,8 @@ def refresh(root: Path = DEFAULT_ROOT, cache: Path = CACHE_DIR,
else: else:
names = _read_names(root) names = _read_names(root)
names_path.write_text(json.dumps(names)) names_path.write_text(json.dumps(names))
apply_invariants(panels["div"], panels["capg"]) apply_invariants(panels["div"], panels["capg"],
panels["adj"], panels["close"])
bundle = Bundle(panels["adj"], panels["close"], panels["div"], bundle = Bundle(panels["adj"], panels["close"], panels["div"],
panels["capg"], names) panels["capg"], names)
# keep only the newest bundle: there is no room for two # keep only the newest bundle: there is no room for two

View File

@ -0,0 +1,39 @@
# Adj/event consistency check (window from 2016-01-01)
TR(sim) = pre-tax reinvestment of raw Close + declared dividend/capitalGain events. Must match TR(adj) if Yahoo's
adj factor was built from the same history. POST>PRE = post-tax engine
would beat pre-tax: impossible, data broken.
| sym | flag | price-only | TR(adj) | TR(sim) | decl yield | gap |
|---|---|---|---|---|---|---|
| bntx | | 35.15% | 35.40% | 35.40% | 1.74% | -0.00% |
| pmaix | | 3.05% | 9.46% | 9.46% | 64.11% | -0.00% |
| slg | | -6.29% | -0.70% | -0.70% | 52.61% | -0.00% |
| atesx | | 4.68% | 8.28% | 8.28% | 30.81% | -0.00% |
| cosix | | -0.38% | 3.88% | 3.88% | 44.09% | +0.00% |
| egrax | | 2.53% | 6.26% | 6.26% | 38.84% | -0.00% |
| pfe | | -0.75% | 3.84% | 3.84% | 45.61% | -0.00% |
| qspnx | | 0.06% | 7.19% | 7.19% | 68.93% | +0.00% |
| svarx | | 1.95% | 6.98% | 6.98% | 51.38% | +0.00% |
| cvsix | | 2.52% | 4.67% | 4.67% | 21.78% | +0.00% |
| flcsx | | 9.84% | 15.75% | 15.75% | 58.07% | -0.00% |
| menkx | | 0.58% | 3.65% | 3.65% | 11.96% | +0.00% |
| flpsx | | 0.21% | 11.39% | 11.39% | 111.47% | -0.00% |
| spy | | 13.40% | 15.23% | 15.23% | 15.74% | -0.00% |
## details
- **bntx**: div 2.11 capg 0.00;
- **pmaix**: div 7.28 capg 0.00; same-date 0
- **slg**: div 38.81 capg 0.00;
- **atesx**: div 0.18 capg 3.88; same-date 0
- **cosix**: div 9.48 capg 0.68; same-date 3 (exact 0, div>cap 0)
- **egrax**: div 4.00 capg 0.00; same-date 0
- **pfe**: div 15.94 capg 0.00;
- **qspnx**: div 5.93 capg 0.00; same-date 0
- **svarx**: div 9.46 capg 2.45; same-date 7 (exact 0, div>cap 4)
- **cvsix**: div 3.08 capg 0.00; same-date 0
- **flcsx**: div 8.54 capg 14.82; same-date 20 (exact 1, div>cap 3)
- **menkx**: div 1.35 capg 0.00;
- **flpsx**: div 8.12 capg 45.49; same-date 20 (exact 0, div>cap 1)
- **spy**: div 62.47 capg 0.00;

View File

@ -0,0 +1,144 @@
#!/usr/bin/env python3
"""Check Yahoo internal consistency for a set of symbols.
For each symbol we compare three "total return" views over a window:
1. TR(Adj) : CAGR of Yahoo's Adj Close series (their dividend/split
adjustment, whatever vintage they computed it from)
2. TR(sim) : CAGR of a pre-tax reinvestment simulation built from the
RAW Close series + the declared dividend/capitalGain event
files (the same inputs the after-tax engine consumes)
3. declared yield: (sum div + sum capg) / avg price over the window
Views 1 and 2 must agree if Yahoo's adj factor was built from the same event
history as the event files. A large gap means the two data sources are
inconsistent (stale adj vintage, restated/stacked events, mislabeled events,
or double-counting of the same distribution in both files).
Also checks the "post > pre is impossible" invariant directly, and analyzes
same-date div/capG event overlap (double-listing patterns the dedup invariant
may not catch, e.g. dividend file carrying the TOTAL distribution =
dividend + capgain, with a different amount than the capGain file).
Usage: python scripts/check_adj_consistency.py [SYM ...]
(default: the pool in POOL below)
Output: stdout table + reports/adj_consistency/check.md
"""
from __future__ import annotations
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT))
import data as D # noqa: E402
POOL = [
"bntx", "pmaix", "slg", "atesx", "cosix", "egrax", "pfe", "qspnx",
"svarx", "cvsix", "flcsx", "menkx", "flpsx", "spy",
]
START = "2016-01-01"
def cagr(a: float, b: float, yrs: float) -> float:
return (b / a) ** (1 / yrs) - 1
def sim_tr(close: pd.Series, div: pd.Series | None, capg: pd.Series | None,
start: str) -> float | None:
"""Pre-tax total-return CAGR from raw close + declared events."""
import numpy as np
c = close.loc[start:].dropna()
if len(c) < 250:
return None
parts = []
for s_ in (div, capg):
parts.append(s_.loc[start:].reindex(c.index).fillna(0.0)
if s_ is not None else pd.Series(0.0, index=c.index))
dist = parts[0] + parts[1]
prev = c.shift(1)
# standard dividend adjustment: on ex-date units *= P_prev / (P_prev - d)
adj_factor = pd.Series(1.0, index=c.index)
m = (dist > 0) & prev.notna() & (prev > dist)
adj_factor[m] = prev[m] / (prev[m] - dist[m])
units = adj_factor.cumprod() / c.iloc[0]
value = units * c
yrs = (c.index[-1] - c.index[0]).days / 365.25
return float((value.iloc[-1] / value.iloc[0]) ** (1 / yrs) - 1)
def main(syms: list[str]) -> None:
import numpy as np # noqa: F401
b = D.load_bundle()
rows = []
for s in syms:
if s not in b.close.columns:
rows.append((s, "MISSING", 0, 0, 0, 0, 0, "", ""))
continue
c = b.close[s].loc[START:].dropna()
a = b.adj[s].loc[START:].dropna()
if len(c) < 250 or len(a) < 250:
rows.append((s, "TOO_SHORT", 0, 0, 0, 0, 0, "", ""))
continue
yrs = (c.index[-1] - c.index[0]).days / 365.25
po = cagr(float(c.iloc[0]), float(c.iloc[-1]), yrs)
tr_adj = cagr(float(a.iloc[0]), float(a.iloc[-1]), yrs)
d = b.div[s].loc[START:].fillna(0.0) if s in b.div.columns else None
g = b.capg[s].loc[START:].fillna(0.0) if s in b.capg.columns else None
sdiv = float(d.sum()) if d is not None else 0.0
scap = float(g.sum()) if g is not None else 0.0
decl = (sdiv + scap) / float(c.mean())
sim = sim_tr(b.close[s], d, g, START)
gap = (sim - tr_adj) if sim is not None else float("nan")
# same-date overlap analysis
overlap = ""
if d is not None and g is not None:
both = (d > 0) & (g > 0)
n = int(both.sum())
if n:
dv, gv = d[both], g[both]
exact = int((abs(dv - gv) < 1e-6 * gv).sum())
divgt = int((dv > gv * 1.05).sum())
overlap = f"same-date {n} (exact {exact}, div>cap {divgt})"
else:
overlap = "same-date 0"
flag = ""
if sim is not None and abs(gap) > 0.003:
flag = "INCONSISTENT"
if sim is not None and sim > tr_adj + 0.001:
flag += " POST>PRE"
rows.append((s, flag, po, tr_adj, sim if sim is not None else float("nan"),
decl, gap if sim is not None else float("nan"),
f"div {sdiv:.2f} capg {scap:.2f}", overlap))
w = max(len(r[0]) for r in rows)
print(f"{'sym':<{w}} {'flag':18s} {'price-only':>10s} {'TR(adj)':>8s} "
f"{'TR(sim)':>8s} {'decl.yld':>9s} {'gap':>8s} {'dist':22s} {''}")
for r in rows:
s, flag, po, tr, sim, decl, gap, dist, ov = r
print(f"{s:<{w}} {flag:18s} {po:10.2%} {tr:8.2%} {sim:8.2%} {decl:9.2%} "
f"{gap:8.2%} {dist:22s} {ov}")
# report
rep = ROOT / "reports" / "adj_consistency"
rep.mkdir(parents=True, exist_ok=True)
with open(rep / "check.md", "w") as fh:
fh.write(f"# Adj/event consistency check (window from {START})\n\n")
fh.write("TR(sim) = pre-tax reinvestment of raw Close + declared "
"dividend/capitalGain events. Must match TR(adj) if Yahoo's\n"
"adj factor was built from the same history. POST>PRE = "
"post-tax engine\nwould beat pre-tax: impossible, data broken.\n\n")
fh.write("| sym | flag | price-only | TR(adj) | TR(sim) | decl yield | gap |\n")
fh.write("|---|---|---|---|---|---|---|\n")
for r in rows:
s, flag, po, tr, sim, decl, gap, dist, ov = r
fh.write(f"| {s} | {flag} | {po:.2%} | {tr:.2%} | {sim:.2%} | "
f"{decl:.2%} | {gap:+.2%} |\n")
fh.write("\n## details\n\n")
for r in rows:
s, flag, po, tr, sim, decl, gap, dist, ov = r
fh.write(f"- **{s}**: {dist}; {ov}\n")
print(f"\nreport: {rep / 'check.md'}")
if __name__ == "__main__":
import pandas as pd # noqa: E402 (type hint in sim_tr)
main(sys.argv[1:] or POOL)

View File

@ -220,9 +220,69 @@ def main() -> int:
D.CORRECTIONS_DIR = real_corr D.CORRECTIONS_DIR = real_corr
finally: finally:
shutil.rmtree(tmp, ignore_errors=True) shutil.rmtree(tmp, ignore_errors=True)
test_total_distributions()
print(f"\n{PASS} passed, {FAIL} failed") print(f"\n{PASS} passed, {FAIL} failed")
return 1 if FAIL else 0 return 1 if FAIL else 0
def test_total_distributions() -> None:
"""Yahoo 'dividend file = total distribution' invariant (unit level)."""
n = 8
idx = pd.DatetimeIndex([f"2024-01-{i + 1:02d}" for i in range(n)])
def close_from(start: float, total_by_i: dict[int, float]) -> list[float]:
# price drops by the distribution on ex-dates (+0.05 drift)
c = [start]
for i in range(1, n):
d = total_by_i.get(i, 0.0)
c.append(c[-1] - d + 0.05)
return c
def adj_from(closes: list[float], total_by_i: dict[int, float]) -> list[float]:
# Yahoo convention: A_i = A_{i-1} * Q_i / (P_{i-1} - d)
a = [closes[0]]
for i in range(1, n):
d = total_by_i.get(i, 0.0)
a.append(a[-1] * closes[i] / (closes[i - 1] - d) if d else a[-1])
return a
def mk(vals: dict[int, float]) -> pd.Series:
s = pd.Series({idx[i]: v for i, v in vals.items()}, dtype=float)
return s.reindex(idx).fillna(0.0)
# tot: div file carries the TOTAL, capg file the cap-gain part
tot_div = {1: 1.0, 3: 2.0, 5: 1.5}
tot_cap = {1: 0.4, 3: 0.8, 5: 0.5}
# sep: genuine separate same-date distributions (adj matches the SUM)
sep_div = {1: 0.9, 3: 1.4, 5: 1.2}
sep_cap = {1: 0.4, 3: 0.3, 5: 0.2}
tot_close = close_from(100.0, tot_div)
sep_close = close_from(100.0, {i: sep_div[i] + sep_cap[i] for i in sep_div})
close_df = pd.DataFrame({"tot": tot_close, "sep": sep_close}, index=idx)
adj_df = pd.DataFrame({
"tot": pd.Series(adj_from(tot_close, tot_div), index=idx),
"sep": pd.Series(adj_from(sep_close, {i: sep_div[i] + sep_cap[i]
for i in sep_div}), index=idx),
})
div_df = pd.DataFrame({"tot": mk(tot_div), "sep": mk(sep_div)}, index=idx)
capg_df = pd.DataFrame({"tot": mk(tot_cap), "sep": mk(sep_cap)}, index=idx)
fixed = D._fix_total_distributions(div_df, capg_df, close_df, adj_df)
check("total-dist: only 'tot' rewritten", fixed == ["tot"])
check("total-dist: tot div -= capg (sum preserved)",
all(abs(div_df.at[idx[i], "tot"] - (tot_div[i] - tot_cap[i])) < 1e-12
for i in tot_div)
and all(abs(div_df.at[idx[i], "tot"] + capg_df.at[idx[i], "tot"]
- tot_div[i]) < 1e-12 for i in tot_div))
check("total-dist: 'sep' (adj matches sum) untouched",
all(div_df.at[idx[i], "sep"] == sep_div[i] for i in sep_div))
# idempotent: second pass changes nothing (mask now div>cap still true but
# adj already reflects the total -> rewrite would break the match)
div2 = div_df.copy()
fixed2 = D._fix_total_distributions(div2, capg_df, close_df, adj_df)
check("total-dist: idempotent (no double subtraction)",
fixed2 == [] and (div2["tot"] == div_df["tot"]).all())
if __name__ == "__main__": if __name__ == "__main__":
sys.exit(main()) sys.exit(main())