f/fundlab/nport.py
Greg Pomerantz a09861f39f N-PORT cross-check: verify top candidates' actual holdings
fundlab/xcheck.py - for each screen candidate, resolve the fund's OWN
registrant CIK (browse-edgar; the 497-cover CIK is the family/trust),
get the exact series name for the ticker (the only reliable
disambiguator between sibling funds), walk the 4 most recent NPORT-P
filing dates, and parse holdings from the interactive NPORT XML
(primary_doc.xml at the accession root - NOT the XSL-rendered view the
submissions API points at). Exact seriesName match > best htm exhibit
parse. Buckets from the authoritative assetCat+issuerCat codes (ABS-O,
ABS-CBDO, DBT+UST/CORP/MUN/NUSS, LON, STIV, RA, EC+RF=fund, ...), not
position-name keywords. Resumable; raw filings cached under
nport_cache/raw/ (gitignored).

nport.py - _SECTION gains the "INVESTMENT PORTFOLIO (unaudited)"
variant (NPORT-EX Sch-F files); find_section/build gain a frac
token-tolerance param (Yahoo names drift from filing names); CMBS/ABS
bucket gains CLO/CBDO/DAC terms.

app Fund Lab - "N-PORT cross-check" expander: per-candidate table
(as-of, n, t5, top code-bucket, #1 position) + per-fund holdings
detail.

RESEARCH.md - cross-check verdicts. 21/22 resolved (qcmmrx is an MMF,
no holdings). The screen's top names are REAL:
- hmezx/mervx = genuine merger arb (equity in deal targets + escrow)
- egrix = 100% wrapper in one macro managed portfolio (underlying not
  NPORT-disclosed); etsix = fund of EV internal multi-strat accounts
- wmnux = discounted/zero-coupon corporate bonds + equity swaps (the
  "equity names" are bond issuers/swap underlyings)
- scfzx/rctix/aflix = securitized credit/CLO/distressed/levered loans
- hicox/fhmix/usmsx/btmix (munis), aguax/femdx (EM sovereign), anglx
  (agency MBS), lpxax (rotated out of prefs into bank/financial debt)
  = genuine missing-factor exposures the 35-sleeve model lacks
- fhcox/dultx/safex = short-duration carry (a short-duration sleeve
  would explain them)

tests/test_fundlab.py - test_xcheck (14 checks): parse_interactive,
code buckets, name-match normalization, series-name disambiguation.
Also: untrack fundlab/streamlit.log; gitignore raw/ + xcheck_run.log.

84 fundlab / 32 app / 14 data tests pass.
2026-08-27 12:43:17 -04:00

355 lines
14 KiB
Python

"""N-PORT schedule-of-investments parsing for the Fund Lab page.
The raw SOI HTML files live in ``nport_cache/<sym>.html`` (one per fund,
fetched from EDGAR; see ``nport_manifest.json`` for the source URL and
filing date of each). ``build()`` parses them into a compact JSON snapshot
(``nport_cache/<sym>.json``) that the Streamlit page renders.
Parsing is deliberately conservative: each fund family formats its SOI
differently (tables, div layouts, combined multi-fund files), so we work
on a whitespace-normalized text *stream* and extract what is reliably
present:
* the fund's own category/percentage lines (the fund's reported
composition — the star of the report),
* the as-of date and net assets,
* position rows carrying a dollar value (where the layout allows),
* a coarse keyword bucketing of those positions.
Anything we cannot parse is simply absent from the snapshot — the page
shows what we have and flags the rest.
"""
from __future__ import annotations
import bisect
import html as _html
import json
import re
from pathlib import Path
CACHE = Path(__file__).parent / "nport_cache"
MANIFEST = Path(__file__).parent / "nport_manifest.json"
_MONTHS = ("January|February|March|April|May|June|July|August|September|"
"October|November|December")
# "<Fund name> ... Schedule of Investments" (variants; NPORT-EX / Sch-F
# exhibit files use "INVESTMENT PORTFOLIO (unaudited)" instead)
_SECTION = re.compile(
r"(?:CONSOLIDATED\s+)?(?:SCHEDULE\s+OF\s+(?:PORTFOLIO\s+)?INVESTMENTS|"
r"PORTFOLIO\s+OF\s+INVESTMENTS|"
r"INVESTMENT\s+PORTFOLIO\s*\(?unaudited\)?)",
re.I)
# category: "NAME — 12.3%" / "NAME (12.3%)" / "NAME: 12.3%" / "NAME 12.3%"
# name: starts uppercase, letters and a few punctuation marks, 3..45 chars
# (the cap keeps merged table rows from spanning a whole line).
_CAT = re.compile(
r"(?<![A-Za-z0-9$])([A-Z][A-Za-z][A-Za-z0-9 ,/&'().\-]{1,43}?)"
r"\s*(?:[-–—:]|\(| )\s*(\d{1,3}\s?\.\d{1,2})\s*%")
def _trim_name(name: str) -> str:
"""Strip a mixed-case prefix (table header run-on) when the tail is an
all-caps category name: 'Maturity Fair Value CORPORATE BONDS' ->
'CORPORATE BONDS'. Title-case names are kept whole."""
parts = re.split(r"(?<=[a-z0-9\)-])\s+(?=[A-Z])", name)
if len(parts) > 1:
tail = parts[-1]
if tail.replace(" ", "").isupper() and len(tail) >= 4:
return tail
return name
_DOLLAR = re.compile(r"\$\s?(\d{1,3}(?:,\d{3})*(?:\.\d+)?|\d{2,}(?:\.\d+)?)")
# a line ending in a (comma-grouped) number: position value without "$"
_TRAILNUM = re.compile(r"([\d,]{4,}(?:\.\d+)?)\s*$")
_ASOF_LONG = re.compile(rf"\b((?:{_MONTHS}) \d{{1,2}}, \d{{4}})\b")
_ASOF_SHORT = re.compile(r"\b(\d{1,2}\.\d{1,2}\.\d{2,4})\b")
_NET = re.compile(r"net assets", re.I)
def rows(html: str) -> list[str]:
"""Row-aware HTML -> list of non-empty text lines (one per table row
/ block). Div-wrapped words in some filings stay on separate rows;
that is why most parsing is done on the joined *stream* instead."""
x = re.sub(r"(?i)</(tr|p|div|li|h[1-6])>|<br\s*/?>", "\n", html)
x = re.sub(r"<[^>]+>", " ", x)
x = _html.unescape(x).replace("\xa0", " ").replace("\r", " ")
out = []
for ln in x.split("\n"):
ln = re.sub(r"[ \t]+", " ", ln).strip()
if ln:
out.append(ln)
return out
def find_section(text: str, tokens: list[str],
frac: float = 1.0) -> tuple[int, int] | None:
"""Locate the fund's own SOI section (possibly in a multi-fund file).
``tokens`` are case-insensitive words that must appear in the section
header (a window around the "Schedule of Investments" phrase, which
contains or follows the fund name). The section extends over all
subsequent headers of the SAME fund (page-continuation footers) and
ends at the first header that no longer names the fund.
``frac`` relaxes the all-tokens gate: a header counts as the fund's
if >= frac of the tokens are present (Yahoo names often drift from
filing names - "Absolute Return Advantage Fund" vs "... Return
Fund" - and an all-tokens gate truncates the section at the first
continuation page).
"""
hits = list(_SECTION.finditer(text))
if not hits:
return None
def score(m: re.Match) -> int:
near = text[max(0, m.start() - 60): m.start()].lower()
before = text[max(0, m.start() - 150): m.start()].lower()
after = text[m.end(): m.end() + 80].lower()
s = sum(3 for t in tokens if t in near)
s += sum(1 for t in tokens
if t not in near and (t in before or t in after))
return s
def names_fund(m: re.Match) -> bool:
window = (text[max(0, m.start() - 150): m.start()]
+ text[m.end(): m.end() + 80]).lower()
if not tokens:
return False
hit = sum(1 for t in tokens if t in window)
return hit >= max(1, int(round(frac * len(tokens))))
best = None
for i, m in enumerate(hits):
s = score(m)
if s and (best is None or s > best[0]):
best = (s, i)
if best is None:
return None
start = max(0, hits[best[1]].start() - 200)
end = min(len(text), start + 400_000)
for j in range(best[1] + 1, len(hits)):
if names_fund(hits[j]):
continue
# cross-references never terminate, even mid-section
before = text[max(0, hits[j].start() - 30): hits[j].start()]
if before.strip().lower().endswith("see"):
continue
end = max(start, hits[j].start() - 150)
break
return start, end
def _to_float(s: str) -> float:
try:
return float(s.replace(",", ""))
except ValueError:
return 0.0
def parse_section(stream: str, start: int, end: int, rs: list[str],
row_starts: list[int]) -> dict:
seg = stream[start:end]
# the rows that fall inside the section (plus the partial first row,
# which may carry the section header)
i1 = bisect.bisect_right(row_starts, end)
i0 = bisect.bisect_left(row_starts, start)
if i0 and row_starts[i0] > start:
i0 -= 1
lines = [ln for ln in rs[i0:i1] if ln.strip()]
as_of = None
head = seg[:1500]
m = _ASOF_LONG.search(head) or _ASOF_SHORT.search(head)
if m:
as_of = m.group(1)
# -- categories (global, in order of appearance, deduplicated)
categories: list[dict] = []
seen: set[str] = set()
pos = 0
while len(categories) < 60:
cm = _CAT.search(seg, pos)
if cm is None:
break
name = _trim_name(cm.group(1).strip(" -–—:("))
# reject table-run-on junk: digits inside a category name (bond
# series, dates, page refs), stray cell markers ("a a a"), or
# column headers absorbed into the name. On a reject, resume from
# just after the name start so a clean match inside the same
# run-on is still found.
if (re.search(r"\d", name)
or re.search(r"\ba\s+a\b", name, re.I)
or re.search(r"\b(shares value|value shares|security shares|"
r"principal amount|fair value)\b", name, re.I)):
pos = cm.start(1) + 1
continue
key = name.lower()
if key not in seen:
seen.add(key)
categories.append({"name": name,
"pct": float(cm.group(2).replace(" ", ""))})
pos = cm.end()
# drop total/net-asset summary lines from the list
categories = [c for c in categories
if not re.match(r"^(total\s+)?(net\s+assets|investments)$",
c["name"], re.I)]
# -- positions: rows with a dollar value (last $ on the row) or,
# failing that, a trailing grouped number
positions: list[dict] = []
current_cat = ""
for ln in lines:
cm = _CAT.search(ln)
if cm and not _DOLLAR.search(ln):
current_cat = cm.group(1).strip()
dollars = _DOLLAR.findall(ln)
value = None
if dollars:
value = _to_float(dollars[-1])
elif _TRAILNUM.search(ln):
value = _to_float(_TRAILNUM.search(ln).group(1))
if value is None or value <= 0:
continue
if _NET.search(ln) or len(ln) < 8:
continue
if re.fullmatch(r"[\d,$.()\s]+", ln): # a bare number row
continue
positions.append({"text": re.sub(r"\s+", " ", ln)[:160],
"value": value, "cat": current_cat})
# -- net assets: last "net assets" mention with a nearby number
net_assets = None
for nm in list(_NET.finditer(seg))[-3:]:
window = seg[nm.end(): nm.end() + 120]
dm = _DOLLAR.search(window) or re.search(r"([\d,]{8,}(?:\.\d+)?)", window)
if dm:
net_assets = _to_float(dm.group(1))
break
return {"as_of": as_of, "categories": categories,
"positions": positions, "net_assets": net_assets}
# ------------------------------------------------------------------ buckets
_B_RE = [
("Futures", r"\b(futur|forward|swap)\b|\bOPEN (CURRENCY|FUTURES)\b|\bCURRENCY (EXCHANGE|FORWARD|OPTION)\b"),
("Commodities", r"\b(gold|silver|copper|crude|natural gas|oil|energy|agricultur|commodit|palladium|platinum|wheat|corn)\b"),
("Fund holdings", r"\b(fund|etf)\b, ?class |class [a-z0-9]{1,2}\b|\bexchange[- ]trad(?:ed)? (fund|funds|etf)|\b(open[- ]end|closed[- ]end) fund|\bmoney market fund\b|\bETF\b|\bFund\b(?!s of Investments)"),
("Agency MBS", r"\bmortgage\b|\bgnma\b|\bfhlmc\b|\bfnma\b|\bagency\b|\bmbs\b"),
("CMBS / ABS / CLO", r"\bcmo\b|commercial mortgage|collateralized (?:mortgage|loan)|\bclo\b|\bcbdo\b|\bdac\b|\btrust 20|abs (?:trust|securitization)|asset[- ]?backed"),
("High yield", r"high[- ]?yield"),
("US govt", r"treasur|us govt|u\.s\. government|t-bill"),
("IG credit / munis", r"corporate|sovereign|senior note|notes,|bonds,|munici|nonconvertible"),
("Cash & T-bills", r"^\s*cash\b|cash item|bank deposit|money market"),
("Equity (intl)", r"\b(foreign|non-?us|international|emerging|europe|japan|canada|\buk\b|german|french|austral|brazil|india|china|korea|switz|netherland|spain|italy)\b"),
("Equity (US)", r"\b(inc|corp|corporation|ltd|plc|group|llc|lp)\.?|common stock|equity|stock"),
]
_BUCKETS = [(name, re.compile(rx, re.I)) for name, rx in _B_RE]
def classify(text: str, category: str) -> str:
"""Keyword-bucket one position. Order matters: first match wins."""
ctx = f"{category} | {text}"
for name, rx in _BUCKETS:
if rx.search(ctx):
return name
return "Other"
def build(sym: str, fund_tokens: list[str], force: bool = False,
frac: float = 1.0) -> dict | None:
"""Parse the cached SOI for ``sym``; returns the snapshot (or None).
``fund_tokens`` identifies this fund's section inside multi-fund
family files (e.g. ["mortgage", "opportunities"]). ``frac`` is
passed to find_section (token-match tolerance).
"""
html_path = CACHE / f"{sym}.html"
json_path = CACHE / f"{sym}.json"
if not force and json_path.exists():
try:
return json.loads(json_path.read_text())
except Exception:
pass
if not html_path.exists():
return None
rs = rows(html_path.read_text())
stream = " ".join(rs)
row_starts: list[int] = []
pos = 0
for r in rs:
row_starts.append(pos)
pos += len(r) + 1
sec = find_section(stream, fund_tokens, frac=frac)
if sec is None:
return None
seg = parse_section(stream, *sec, rs, row_starts)
buckets: dict[str, float] = {}
for p in seg["positions"]:
b = classify(p["text"], p["cat"])
buckets[b] = buckets.get(b, 0.0) + p["value"]
total = sum(buckets.values()) or 1.0
snap = {
"sym": sym,
"as_of": seg["as_of"],
"net_assets": seg["net_assets"],
"n_positions": len(seg["positions"]),
"categories": seg["categories"],
"buckets": [{"name": k, "value": v, "pct": 100.0 * v / total}
for k, v in sorted(buckets.items(), key=lambda kv: -kv[1])],
"top": sorted(seg["positions"], key=lambda p: -p["value"])[:40],
}
json_path.write_text(json.dumps(snap, indent=1))
return snap
# ------------------------------------------------------------------ CLI
# fund-identifying tokens per ticker (used to find the fund's section in
# multi-fund family filings)
FUND_TOKENS = {
# atesx: no cached SOI (its family's current filings don't cover it);
# it still shows in the list with the manifest's note
"atesx": ["anchor", "risk", "equity"],
"atrfx": ["catalyst", "systematic"],
"cvsix": ["calamos", "market"],
"jlpsx": ["large cap core plus"],
"pmaix": ["victory", "pioneer"],
"pmfkx": ["victory", "pioneer"],
"pmorx": ["mortgage", "opportunities"],
"qspnx": ["style", "premia"],
"svarx": ["spectrum low volatility"],
"cosix": ["columbia strategic income"],
"mbxix": ["millburn", "hedge"],
"eagmx": ["global macro absolute return"],
"egrsx": ["global macro absolute return"],
"lcorx": ["leuthold core"],
"lcrix": ["leuthold core"],
"lamhx": ["dividend growth"],
}
def build_all(force: bool = False) -> dict:
"""Build/return snapshots for every fund with a cached SOI file."""
out = {}
for sym, tk in FUND_TOKENS.items():
s = build(sym, tk, force=force)
if s:
out[sym] = s
return out
def manifest() -> dict:
return json.loads(MANIFEST.read_text())
if __name__ == "__main__":
import sys
syms = sys.argv[1:] or list(FUND_TOKENS)
for s in syms:
r = build(s, FUND_TOKENS[s], force=True)
print(s, "->", "ok" if r else "FAILED")