f/fundlab/cef_universe.py
Greg Pomerantz d1e85026bf CEF pass: universe (SEC report, 973 -> 295 listed) + stage 1 screen + stage 2a character
fundlab/cef_universe.py: SEC 'Closed-End Fund Information' report
(973 active CEFs) + company_tickers CIK join -> 295 listed common
classes (preferreds and 6(c)-converted companies flagged).  Supersedes
the N-2 full-index approach (N-2/A annual updates + 404ing index
paths).  Prices via goget.

fundlab/cef_screen.py: 290 screened - t5/t12, vol5, maxdd5, the 5
crash episodes, 12m payout proxy.  Energy/midstream CEFs (EMO +305%,
SRV +234%, NML +232%) top return AND crash resilience; Voya
Dividend-Premium series (IGD maxDD -16%); EM CEFs volatile + 14-24%
dist; long-dur munis -26..-42%.

fundlab/cef_character.py: 50-fund shortlist, 35-sleeve character +
crude tax_arb = character x (upside + 0.4 x vol).

Tests: 6 new cef checks (105 total).  RESEARCH.md: CEF form facts
(N-2ASR, N-PX, N-23C-3A, BDC caveats) + remaining stage 2b work.
2026-08-27 22:24:12 -04:00

160 lines
5.7 KiB
Python

"""CEF universe: the SEC's official "Closed-End Fund Information" report.
Primary source: https://www.sec.gov/files/investment/data/other/
closed-end-fund-information/closed-end-investment-company-<YYYY>.csv
(all ACTIVE 1940-Act closed-end companies: CIK, name, addresses,
filing status - 973 for 2026). This supersedes the N-2 full-index
approach (N-2/N-2-A filers are the CEFs that updated their
registration within the window - a biased subset; the report is the
whole active population).
Tickers: CIK join with the SEC company_tickers.json exchange map
(305 of 973 match - the rest are small/dark/OTC funds). Hyphenated
tickers (GAM-PB etc.) are PREFERRED series - dropped (their price
history is the preferred's near-flat NAV, useless for screening);
the common class is usually the fund's other listed security.
CEF notes for the rest of the pipeline:
- CEFs file NPORT-P (same format as open-end - xcheck.py works)
- shareholder reports are N-CSR/N-CSRS AND/OR N-2ASR (rocdetect must
try both)
- CEFs file N-PX proxies (open-end funds don't - a CEF signal)
- price = MARKET price (premium/discount to NAV is a free variable)
- Yahoo instrumentType is usually EQUITY - do NOT apply the open-end
MUTUALFUND filter
- BDCs are in the report too (name-flagged)
"""
from __future__ import annotations
import csv
import io
import json
import re
import urllib.request
from datetime import date, timedelta
from pathlib import Path
HERE = Path(__file__).parent
CACHE = HERE / "universe_cache"
CEF_CSV = CACHE / "cef_sec_2026.csv"
UNIVERSE = HERE / "cef_universe.json"
DATA = Path.home() / "prog/fin/stocks"
UA = {"User-Agent": "research test@example.com"}
MIN_DAYS = 1250 # >=5y of trading history (same bar as the 497 pass)
STALE_DAYS = 35 # last quote older than this = delisted
def _get(url: str) -> bytes:
req = urllib.request.Request(url, headers=UA)
return urllib.request.urlopen(req, timeout=120).read()
def fetch_cef_report(year: int = 2026) -> list[dict]:
"""The SEC active-CEF report (CIK, name, ...). Cached locally."""
if CEF_CSV.exists():
raw = CEF_CSV.read_bytes()
else:
CACHE.mkdir(exist_ok=True)
# 2024+ uses hyphenated filenames, earlier underscored
for pat in (f"closed-end-investment-company-{year}.csv",
f"closed-end_investment_company_{year}.csv",
"closed-end_investment_company.csv"):
url = ("https://www.sec.gov/files/investment/data/other/"
f"closed-end-fund-information/{pat}")
try:
raw = _get(url)
break
except Exception:
continue
else:
raise RuntimeError("could not fetch the SEC CEF report")
CEF_CSV.write_bytes(raw)
rows = list(csv.DictReader(io.StringIO(
raw.decode("utf-8-sig"))))
return [{"cik": r["CIK"].lstrip("0") or "0",
"name": r["Registrant_Name"].strip()} for r in rows]
def _company_tickers() -> dict:
"""cik(int) -> (ticker, title) from the SEC exchange-listed map."""
d = json.loads(_get("https://www.sec.gov/files/company_tickers.json"))
out: dict[int, tuple] = {}
for v in d.values():
cik = v.get("cik_str", v.get("cik"))
t = v.get("ticker", v.get("symbol"))
if cik is None or not t:
continue
out[int(cik)] = (t.upper(), v.get("title", ""))
return out
def _history(sym: str) -> tuple[int, str]:
"""(trading days, last date) from the local price file."""
p = DATA / f"{sym}-history.csv"
if not p.exists():
return 0, ""
last = ""
n = 0
with p.open() as f:
next(f, None) # header
for line in f:
n += 1
last = line.split(",")[0]
return n, last
def run() -> dict:
cefs = fetch_cef_report()
print(f"{len(cefs)} active CEFs in the SEC report", flush=True)
tick = _company_tickers()
# NOTE: do NOT exclude tickers present in the local DB - goget
# writes <sym>.json there, so after the first download every CEF
# would look "known". The SEC report already defines the set.
known = set()
for j in (Path(__file__).parent.parent / "funds.json",
HERE / "search_all.json", HERE / "search_external.json",
HERE / "xcheck_report.json"):
if j.exists():
d = json.loads(j.read_text())
if isinstance(d, dict):
known |= {k.upper() for k in d}
stale = (date.today() - timedelta(days=STALE_DAYS)).toordinal()
out: dict[str, dict] = {}
n_pref = n_unmatched = 0
for c in cefs:
hit = tick.get(int(c["cik"]))
if not hit:
n_unmatched += 1
continue
t, _yt = hit
if "-" in t: # preferred/series class
n_pref += 1
continue
if t in known:
continue
days, last = _history(t.lower())
live = False
try:
live = date.fromisoformat(last).toordinal() >= stale
except ValueError:
pass
out[t] = {"cik": c["cik"], "name": c["name"],
"bdc": bool(re.search(r"business development",
c["name"], re.I)),
"listed": days >= MIN_DAYS and live,
"days": days, "last": last, "known": False}
n_listed = sum(1 for v in out.values() if v["listed"])
n_bdc = sum(1 for v in out.values() if v["bdc"])
print(f"{len(out)} CEF tickers ({n_listed} live with >=5y history, "
f"{n_bdc} BDCs, {n_pref} preferred-series dropped, "
f"{n_unmatched} CIKs with no listed ticker)", flush=True)
UNIVERSE.write_text(json.dumps(out, indent=1))
print(f"wrote {UNIVERSE}")
return out
if __name__ == "__main__":
run()