"""CEF stage 2b: per-share financial highlights from the annual report.
For each CEF, fetch the recent N-CSR/N-CSRS submissions and parse the
per-share "Financial Highlights" table. CEF tables come in two
families (verified on EMO / TWN):
Less distributions to common shareholders from: (Franklin style)
Dividends ...
Return of capital ...
Total distributions to common shareholders ...
Less Distributions to Stockholders from: (classic CEF)
Net investment income ...
Net realized gains ...
Total distributions to stockholders ...
Gotchas handled:
- Blanks are dropped when the table flattens: a short row's values can
sit in ANY columns. EMO's ROC row (3.42, 0.93) belongs to columns
0 and 4, not 0 and 1; TWN's total-distributions row has 4 values
for 5 columns (a zero year dropped).
- The robust total per column comes from the NAV IDENTITY, not the
printed total row: dist[c] = NAV_beg[c] + ops[c] + captch[c]
- NAV_end[c]. The printed sub-rows (NII / gains / ROC) are then
placed combinatorially against those derived totals.
- The first "Net asset value, beginning of year" anchor in the doc is
the most recent 5-year table (a continuation table for older years
follows it). A "share" proximity guard skips dollar-basis mentions.
- Annual vs semi-annual: try the 3 most recent shareholder reports and
keep the one that parses with the most columns (annual = 5).
Also captured (optional lines): market value end of year (-> discount
series), leverage (debt + preferred vs net assets), tax-paying flag.
Run: python -m fundlab.cef_annual [SYM ...] (default: stage-2a
shortlist)
Output: fundlab/cef_annual.json (cached per fund in cef_cache/)
"""
from __future__ import annotations
import json
import re
from pathlib import Path
import numpy as np
from fundlab import edgar
HERE = Path(__file__).parent
CACHE = HERE / "cef_cache"
OUT = HERE / "cef_annual.json"
CHARACTER = HERE / "cef_character.json"
# distribution-character fractions for the placement score (1 = fully
# taxable-account favorable)
FRAC_NII = 0.10 # ordinary income (equity-fund NII is partly QD)
FRAC_GAINS = 0.70 # mix of STCG/LTCG - long-term equity funds skew LT
FRAC_ROC = 1.0 # defers to the investor's own LTCG on a >1y sale
FRAC_NII_MUNI = 1.0 # tax-exempt interest
NUMTOK = re.compile(r"\$?\(?\d[\d,]*(?:\.\d+)?\)?")
FOOTNOTE = re.compile(r"\([a-z0-9]{1,3}\)")
def _nums(tokens: list[str]) -> list[float | None]:
out = []
for t in tokens:
t = t.strip().rstrip(":")
if not re.fullmatch(r"\$?\(?\d[\d,]*(?:\.\d+)?\)?", t):
out.append(None)
continue
neg = t.startswith("(")
v = float(t.strip("$()").replace(",", ""))
out.append(-v if neg else v)
return out
def _row(block: str, label_rx: str, span: int = 700) -> list[float | None]:
"""Numeric row following the first label match in the block."""
rm = re.search(label_rx, block, re.I)
if not rm:
return []
tail = block[rm.end(): rm.end() + span]
tail = re.sub(r"\(\s+", "(", tail)
tail = re.sub(r"\s+\)", ")", tail)
vals: list[str] = []
for t in re.findall(r"\S+", tail):
if t == "$":
continue
if NUMTOK.fullmatch(t):
vals.append(t)
elif FOOTNOTE.fullmatch(t):
continue # footnote markers sit between values
elif vals:
break # first word after values = next label
return _nums(vals[:10])
def parse_highlights(t: str) -> dict | None:
"""Parse the per-share highlights table. Returns None on failure."""
# collapse whitespace: CEF HTML tables flatten with one
per line,
# splitting labels ("Net\nasset value, beginning of year")
t = re.sub(r"\s+", " ", t)
# anchor: first "Net asset value, beginning of year" with "share"
# nearby (skips dollar-basis statement mentions)
anchor = None
for m in re.finditer(r"net asset value(?: per (?:common )?share)?, beginning of (?:year|period)", t, re.I):
pre = t[max(0, m.start() - 400): m.start()]
if re.search(r"share", pre, re.I):
anchor = m
break
if not anchor:
return None
block = t[anchor.start(): anchor.start() + 6000]
nav_beg = _row(block, r"net asset value(?: per (?:common )?share)?, beginning of (?:year|period)")
nav_end = _row(block, r"net asset value(?: per (?:common )?share)?, end of (?:year|period)")
if len(nav_beg) < 3 or len(nav_end) < 3:
return None
ncols = len(nav_beg)
if len(nav_end) != ncols or ncols > 6:
return None
nii_ops = _row(block, r"Net investment (?:income|loss)\b")
gains_ops = _row(block, r"Net realized (?:and|&) (?:unrealized )?gain")
ops = _row(block,
r"total (?:income \(loss\) )?from (?:investment )?operations")
if not ops:
ops = [(_get(nii_ops, i) + _get(gains_ops, i)) for i in range(ncols)]
# CEFs adjust NAV for capital transactions (repurchases, rights
# offerings, tenders). Collect every variant row present, deduping a
# physical row that matches two labels (e.g. "Capital Share
# Transactions: Accretion (dilution) ...").
captch_rows = []
seen = []
for rx in (r"accretion (?:\(dilution\) )?to net asset value",
r"capital share transactions",
r"anti-dilutive impact of repurchase",
r"dilutive impact of rights offering",
r"anti-dilutive impact of tender offer"
r"repurchase of shares",):
row = _row(block, rx)
if row and any(len(row) == len(r2) and
all(abs((a or 0) - (b or 0)) < 0.005
for a, b in zip(row, r2)) for r2 in seen):
continue
seen.append(row)
captch_rows.append(row)
mkt_end = _row(block, r"market (?:value|price), end of year")
# distribution sub-rows (character). The block ends at NAV-end:
# footnotes below it also contain the word "dividends".
danchor = re.search(
r"distributions to (?:common )?shareholders from|"
r"distributions to stockholders from|"
r"distributions declared (?:to shareholders)? from|"
r"distributions from",
block, re.I)
dblock = block[danchor.start():] if danchor else block
em = re.search(r"net asset value(?: per (?:common )?share)?, end of (?:year|period)", dblock, re.I)
if em:
dblock = dblock[: em.start()]
# the TOTAL row label also contains "dividends" in some funds
# ("Total dividends and distributions to stockholders") - scope the
# sub-row search to before it
tm = re.search(r"total (?:dividends and )?distributions", dblock, re.I)
dwindow = dblock[: tm.start()] if tm else dblock
divs = (_row(dwindow, r"\bdividends\b")
or _row(dwindow, r"net investment income"))
gains = _row(dwindow, r"(?:net )?realized (?:capital )?gains?|"
r"capital gain distributions")
roc = _row(dwindow, r"return of capital")
# total distributions: the PRINTED row is authoritative when it has
# one value per column; otherwise derive from the NAV identity
# (dist = NAV_beg + ops + captch - NAV_end)
dist_row = _row(dblock, r"total (?:dividends and )?distributions")
complete = len(dist_row) == ncols
if complete:
dist_tot = list(dist_row)
# some funds print distributions positive, others negative
if sum(1 for v in dist_tot if v < 0) > ncols // 2:
dist_tot = [-v for v in dist_tot]
else:
dist_tot = None
# JOINT PLACEMENT. CEF tables drop zero columns, so a short row can
# sit in ANY subset of columns (TWN: gains in 4 of 5 cols, NII-divs
# in the other two; EMO: ROC in cols 0 and 4; KF: accretion in cols
# 1-4, not 0-3). One backtracking search places every short row
# (captch first - they affect the NAV identity - then the
# distribution rows) so the per-column arithmetic holds:
# printed total complete: divs+gains+roc == total
# printed total short: divs+gains+roc == NAV_beg+ops+captch-NAV_end
from itertools import combinations
dist_rows = (("divs", divs), ("gains", gains), ("roc", roc))
place = [(i, "c", row) for i, row in enumerate(captch_rows)
if 0 < len(row) < ncols]
place += [(i, "d", row) for i, (name, row) in enumerate(dist_rows)
if 0 < len(row) < ncols]
place.sort(key=lambda t: -len(t[2]))
base_d = [0.0] * ncols
for name, row in dist_rows:
if len(row) == ncols:
for c in range(ncols):
if row[c]:
base_d[c] += abs(row[c])
base_c = [0.0] * ncols
for row in captch_rows:
if len(row) == ncols:
for c in range(ncols):
if row[c]:
base_c[c] += row[c]
dsum = [0.0] * ncols
csum = [0.0] * ncols
assign = []
def rec(i):
if i == len(place):
for c in range(ncols):
if complete:
need = abs(dist_tot[c])
else:
need = abs(_get(nav_beg, c) + _get(ops, c) + base_c[c]
+ csum[c] - _get(nav_end, c))
if abs(base_d[c] + dsum[c] - need) > 0.02 * max(1.0, need) + 0.02:
return False
return True
idx, kind, vals = place[i]
nz = [v for v in vals if v]
for cols in combinations(range(ncols), len(nz)):
if kind == "d":
for c, v in zip(cols, nz):
dsum[c] += abs(v)
if complete and any(base_d[c] + dsum[c] > abs(dist_tot[c]) + 0.02
for c in cols):
for c, v in zip(cols, nz):
dsum[c] -= abs(v)
continue
else:
for c, v in zip(cols, nz):
csum[c] += v
if rec(i + 1):
assign.append((idx, kind, nz, cols))
return True
if kind == "d":
for c, v in zip(cols, nz):
dsum[c] -= abs(v)
else:
for c, v in zip(cols, nz):
csum[c] -= v
return False
if place and not rec(0):
# short rows that reconcile nowhere: mis-parsed table - reject
return None
def colvec(vals, cols):
vec = [0.0] * ncols
for ci, col in enumerate(cols):
vec[col] = vals[ci]
return vec
placed_d = {}
placed_c = {i: [0.0] * ncols for i in
(i for i, k, _ in place if k == "c")}
for idx, kind, nz, cols in assign:
if kind == "d":
placed_d[idx] = colvec(nz, cols)
else:
placed_c[idx] = colvec(nz, cols)
for i, (name, row) in enumerate(dist_rows):
if len(row) == ncols:
vec = list(row)
elif 0 < len(row) < ncols:
vec = placed_d.get(i, [0.0] * ncols)
else:
vec = [0.0] * ncols
if name == "divs":
divs = vec
elif name == "gains":
gains = vec
else:
roc = vec
captch = [base_c[c] for c in range(ncols)]
for i, vec in placed_c.items():
for c in range(ncols):
captch[c] += vec[c]
if not complete:
dist_tot = [(_get(nav_beg, c) + _get(ops, c) + captch[c]
- _get(nav_end, c)) for c in range(ncols)]
if sum(1 for v in dist_tot if v < 0) > ncols // 2:
dist_tot = [-v for v in dist_tot]
# sanity: components should (roughly) sum to the derived total
bad = sum(1 for c in range(ncols)
if abs(-(divs[c] + gains[c] + roc[c]) - dist_tot[c])
> 0.02 * max(1.0, abs(dist_tot[c])) + 0.02)
if bad > max(1, ncols // 2):
return None
out = {
"ncols": ncols,
"nav_beg": nav_beg, "nav_end": nav_end,
"mkt_end": mkt_end[:ncols] if mkt_end else [],
"dist_tot": dist_tot, "divs": divs, "gains": gains, "roc": roc,
"ops": ops[:ncols],
}
tot5 = sum(v for v in dist_tot if v > 0)
if tot5 > 0:
out["share_div"] = sum(-v for v in divs if v < 0) / tot5
out["share_gains"] = sum(-v for v in gains if v < 0) / tot5
out["share_roc"] = sum(-v for v in roc if v < 0) / tot5
return out
def _get(row: list, i: int) -> float:
if i < len(row) and row[i] is not None:
return row[i]
return 0.0
def _submissions(cik: int) -> dict | None:
try:
raw = edgar.sec_get(f"https://data.sec.gov/submissions/CIK{cik:010d}.json")
return json.loads(raw)
except Exception:
return None
def analyze(sym: str, cik: int, force: bool = False) -> dict:
cf = CACHE / f"{sym}.json"
if cf.exists() and not force:
return json.loads(cf.read_text())
d = _submissions(cik)
if not d:
return {"sym": sym, "error": "submissions fetch failed"}
r = d["filings"]["recent"]
forms = r["form"]
out: dict = {"sym": sym, "cik": cik, "name": d.get("name"),
"bdc": "10-K" in set(forms),
"n_tender": sum(1 for f in forms if f.startswith("N-23C")),
"ticker": (d.get("tickers") or [sym])[0]}
# try the 3 most recent shareholder reports, keep the best parse
cands = [(r["filingDate"][i], r["accessionNumber"][i])
for i in range(len(forms))
if forms[i] in ("N-CSR", "N-CSRS")][:3]
def _score(h):
# the 5-column annual table is the target; a 6-col semiannual
# (period + 5 FYs) is a fallback
return 2 if h["ncols"] == 5 else 1
best = None
for fd, acc in cands:
try:
raw = edgar.sec_get(
f"https://www.sec.gov/Archives/edgar/data/{cik}/{acc}.txt")
t = edgar.to_text(raw)
except Exception:
continue
h = parse_highlights(t)
if h and (best is None or _score(h) > _score(best[1])
or (_score(h) == _score(best[1])
and h["ncols"] > best[1]["ncols"])):
best = (fd, h, t)
if not best:
out["error"] = "no per-share table found in the 3 latest reports"
else:
fd, h, t = best
out["report_filed"] = fd
out.update(h)
# leverage + tax-paying flags (optional lines)
lm = re.search(
r"Loan and Debt Issuance Outstanding, End of Year \(000s\)\s*"
r"([\d,]+)", t)
pm = re.search(
r"Preferred Stock at Liquidation Value, End of Year \(000s\)\s*"
r"([\d,]+)", t)
nm = re.search(
r"Net assets applicable to common shareholders, end of year "
r"\(millions\)\s*\$?([\d,]+)", t)
if lm and nm:
debt = float(lm.group(1).replace(",", "")) * 1e3
pref = float(pm.group(1).replace(",", "")) * 1e3 if pm else 0.0
nav = float(nm.group(1).replace(",", "")) * 1e6
out["leverage_pct"] = round((debt + pref) / nav * 100, 1)
if re.search(r"income tax expenses?\s+[\d.]+\s*%", t):
out["tax_paying"] = True
CACHE.mkdir(exist_ok=True)
cf.write_text(json.dumps(out, default=str))
return out
def current_nav_discount(h: dict) -> float | None:
"""FY-end discount (market vs NAV) from the table; used as a
rough estimate of the current discount (no live NAV locally)."""
nav_end = h.get("nav_end") or []
mkt_end = h.get("mkt_end") or []
if not nav_end or not mkt_end or not nav_end[0]:
return None
return float(mkt_end[0] / nav_end[0] - 1)
def character_score(h: dict, name: str) -> float | None:
if "share_div" not in h:
return None
f_nii = FRAC_NII_MUNI if re.search(r"muni|tax[- ]?exempt", name, re.I) \
else FRAC_NII
return (h["share_div"] * f_nii + h["share_gains"] * FRAC_GAINS
+ h["share_roc"] * FRAC_ROC)
def run(syms: list[str] | None = None) -> dict:
uni = json.loads((HERE / "cef_universe.json").read_text())
if syms is None:
ch = json.loads(CHARACTER.read_text())
syms = [s.upper() for s in
sorted(ch, key=lambda s: -ch[s]["tax_arb"])]
res: dict = {}
for i, s in enumerate(syms, 1):
u = uni.get(s) or {}
cik = int(u.get("cik", 0))
a = analyze(s, cik)
a["char_actual"] = round(character_score(a, u.get("name", "")), 2) \
if "share_div" in a else None
if "nav_end" in a:
disc = current_nav_discount(a)
a["disc_now_approx"] = round(disc, 4) if disc is not None else None
res[s] = a
if "error" in a:
print(f"{i:2}/{len(syms)} {s:7} ERR: {a['error'][:40]}", flush=True)
else:
print(f"{i:2}/{len(syms)} {s:7} "
f"div {a.get('share_div',0):.0%} gain {a.get('share_gains',0):.0%} "
f"ROC {a.get('share_roc',0):.0%} char={a.get('char_actual')} "
f"lev={a.get('leverage_pct')}", flush=True)
OUT.write_text(json.dumps(res, indent=1, default=str))
print(f"wrote {OUT}")
return res
def _print(res: dict) -> None:
print(f"{'fund':7} {'name':38} {'fdiv':>5} {'fgain':>6} {'froc':>5}"
f" {'char':>5} {'disc':>7} {'lev%':>6} {'taxp':>4} {'tend':>5} "
f"{'report':>10}")
for s, v in res.items():
if "share_div" not in v:
print(f"{s:7} {v.get('name','')[:38]:38} -- "
f"{v.get('error', 'no data')}")
continue
disc = v.get("disc_now_approx")
print(f"{s:7} {v.get('name','')[:38]:38} "
f"{v['share_div']:5.0%} {v['share_gains']:6.0%} "
f"{v['share_roc']:5.0%} {v.get('char_actual') or 0:5.2f} "
f"{('' if disc is None else f'{disc:+.1%}'):>7} "
f"{str(v.get('leverage_pct', '')):>6} "
f"{'Y' if v.get('tax_paying') else '':>4} "
f"{v.get('n_tender', 0):>5} {v.get('report_filed',''):>10}")
if __name__ == "__main__":
import sys
args = [a.upper() for a in sys.argv[1:]]
r = run(args or None)
_print(r)
|