From 6536c9903e24e908327a609401058933d53fdd9f Mon Sep 17 00:00:00 2001 From: Greg Pomerantz Date: Mon, 31 Aug 2026 14:41:03 -0400 Subject: [PATCH] Tier-3 verification: cross-check distributions against SEC filings scripts/verify_official.py locates each fund's latest N-CSR/N-CSRS/10-K/10-Q via EDGAR full-text search (full fund-name phrase first, then ticker + name words, then bare ticker), extracts the fund's Financial Highlights tables (both Vanguard-style and Victory-style layouts, calendar and non-calendar fiscal years, M/D/YY and month-name headers), matches the share class by per-share distribution series + NAV magnitude, and compares per period window against the local Yahoo CSVs (frozen copies for stale tickers). Per-symbol JSONs + SUMMARY.md land in reports/xcheck_official/; results are cached per fund. Run on the 29 curated funds: 8 ok (exact to 3dp, e.g. VTSAX 2021-2026H1), 1 mismatch (CVSIX FY2009: local 1.104 vs official 0.81 - the Yahoo 2008-12-18 row of 0.292 looks spurious), 3 weak-match (uncovered doc formats, e.g. Leuthold), 17 not-found (mostly ETF families whose 10-K layouts aren't covered yet). --- reports/xcheck_official/SUMMARY.md | 35 ++ reports/xcheck_official/agg.json | 6 + reports/xcheck_official/atesx.json | 98 +++++ reports/xcheck_official/atrfx.json | 6 + reports/xcheck_official/bnd.json | 89 ++++ reports/xcheck_official/cosix.json | 6 + reports/xcheck_official/cvsix.json | 77 ++++ reports/xcheck_official/eagmx.json | 6 + reports/xcheck_official/efa.json | 6 + reports/xcheck_official/egrsx.json | 6 + reports/xcheck_official/gld.json | 6 + reports/xcheck_official/iwm.json | 6 + reports/xcheck_official/jlpsx.json | 6 + reports/xcheck_official/lamhx.json | 6 + reports/xcheck_official/lcorx.json | 82 ++++ reports/xcheck_official/lcrix.json | 82 ++++ reports/xcheck_official/mbxix.json | 6 + reports/xcheck_official/pmaix.json | 89 ++++ reports/xcheck_official/pmfkx.json | 89 ++++ reports/xcheck_official/pmorx.json | 6 + reports/xcheck_official/qqq.json | 6 + reports/xcheck_official/qspnx.json | 6 + reports/xcheck_official/schd.json | 6 + reports/xcheck_official/shv.json | 6 + reports/xcheck_official/slv.json | 6 + reports/xcheck_official/svarx.json | 6 + reports/xcheck_official/vea.json | 88 ++++ reports/xcheck_official/vti.json | 89 ++++ reports/xcheck_official/vtsax.json | 89 ++++ reports/xcheck_official/vwo.json | 6 + scripts/verify_official.py | 651 +++++++++++++++++++++++++++++ 31 files changed, 1672 insertions(+) create mode 100644 reports/xcheck_official/SUMMARY.md create mode 100644 reports/xcheck_official/agg.json create mode 100644 reports/xcheck_official/atesx.json create mode 100644 reports/xcheck_official/atrfx.json create mode 100644 reports/xcheck_official/bnd.json create mode 100644 reports/xcheck_official/cosix.json create mode 100644 reports/xcheck_official/cvsix.json create mode 100644 reports/xcheck_official/eagmx.json create mode 100644 reports/xcheck_official/efa.json create mode 100644 reports/xcheck_official/egrsx.json create mode 100644 reports/xcheck_official/gld.json create mode 100644 reports/xcheck_official/iwm.json create mode 100644 reports/xcheck_official/jlpsx.json create mode 100644 reports/xcheck_official/lamhx.json create mode 100644 reports/xcheck_official/lcorx.json create mode 100644 reports/xcheck_official/lcrix.json create mode 100644 reports/xcheck_official/mbxix.json create mode 100644 reports/xcheck_official/pmaix.json create mode 100644 reports/xcheck_official/pmfkx.json create mode 100644 reports/xcheck_official/pmorx.json create mode 100644 reports/xcheck_official/qqq.json create mode 100644 reports/xcheck_official/qspnx.json create mode 100644 reports/xcheck_official/schd.json create mode 100644 reports/xcheck_official/shv.json create mode 100644 reports/xcheck_official/slv.json create mode 100644 reports/xcheck_official/svarx.json create mode 100644 reports/xcheck_official/vea.json create mode 100644 reports/xcheck_official/vti.json create mode 100644 reports/xcheck_official/vtsax.json create mode 100644 reports/xcheck_official/vwo.json create mode 100644 scripts/verify_official.py diff --git a/reports/xcheck_official/SUMMARY.md b/reports/xcheck_official/SUMMARY.md new file mode 100644 index 0000000..c133c3c --- /dev/null +++ b/reports/xcheck_official/SUMMARY.md @@ -0,0 +1,35 @@ +# Official-filing cross-check (SEC N-CSR/N-CSRS/10-K/10-Q) + +Verdicts: ok = all bounded fiscal periods agree; mismatch = class matched but some period differs (review the year, then correct via overrides/); weak-match = best class fit too poor to trust (format/fund mismatch); not-found = fund not located in any candidate filing. + +| Symbol | Verdict | Class | Periods (fy end, ~ = unbounded) | Detail | +|---|---|---|---|---| +| AGG | not-found | | | no candidate filing contained this fund (6 tried) | +| ATESX | weak-match | ? | 2010-08✗ 2015-08✗ 2024-02✗ 2023-08✗ 2022-08✗ 2021-08✗ 2019-08~ | | +| ATRFX | not-found | | | no candidate filing contained this fund (6 tried) | +| BND | ok | ETF Shares | 2025-12✓ 2024-12✓ 2023-12✓ 2022-12✓ 2021-12~ | | +| COSIX | not-found | | | no candidate filing contained this fund (4 tried) | +| CVSIX | mismatch | ? | 2011-10✓ 2010-10✓ 2009-10✗ 2008-10~ | | +| EAGMX | not-found | | | no candidate filing contained this fund (6 tried) | +| EFA | not-found | | | no candidate filing contained this fund (6 tried) | +| EGRSX | not-found | | | no candidate filing contained this fund (4 tried) | +| GLD | not-found | | | no candidate filing contained this fund (6 tried) | +| IWM | not-found | | | no candidate filing contained this fund (6 tried) | +| JLPSX | not-found | | | no candidate filing contained this fund (5 tried) | +| LAMHX | not-found | | | no candidate filing contained this fund (5 tried) | +| LCORX | weak-match | ? | 2023-09✗ 2022-09✗ 2021-09✓ 2020-09✗ 2019-09~ | | +| LCRIX | weak-match | ? | 2023-09✗ 2022-09✗ 2021-09✓ 2020-09✗ 2019-09~ | | +| MBXIX | not-found | | | no candidate filing contained this fund (6 tried) | +| PMAIX | ok | Class A | 2025-07✓ 2024-07✓ 2023-07✓ 2022-07✓ 2021-07~ | | +| PMFKX | ok | Class R6 | 2025-07✓ 2024-07✓ 2023-07✓ 2022-07✓ 2021-07~ | | +| PMORX | not-found | | | no candidate filing contained this fund (5 tried) | +| QQQ | not-found | | | no candidate filing contained this fund (6 tried) | +| QSPNX | not-found | | | no candidate filing contained this fund (6 tried) | +| SCHD | not-found | | | no candidate filing contained this fund (6 tried) | +| SHV | not-found | | | no candidate filing contained this fund (6 tried) | +| SLV | not-found | | | no candidate filing contained this fund (6 tried) | +| SVARX | not-found | | | no candidate filing contained this fund (5 tried) | +| VEA | ok | FTSE Developed Markets ETF Shares | 2025-12✓ 2024-12✓ 2023-12✓ 2022-12✓ 2021-12~ | | +| VTI | ok | ETF Shares | 2025-12✓ 2024-12✓ 2023-12✓ 2022-12✓ 2021-12~ | | +| VTSAX | ok | Admiral Shares | 2025-12✓ 2024-12✓ 2023-12✓ 2022-12✓ 2021-12~ | | +| VWO | not-found | | | no candidate filing contained this fund (6 tried) | diff --git a/reports/xcheck_official/agg.json b/reports/xcheck_official/agg.json new file mode 100644 index 0000000..7239533 --- /dev/null +++ b/reports/xcheck_official/agg.json @@ -0,0 +1,6 @@ +{ + "symbol": "agg", + "name": "iShares Core U.S. Aggregate Bond ETF", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (6 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/atesx.json b/reports/xcheck_official/atesx.json new file mode 100644 index 0000000..0228b96 --- /dev/null +++ b/reports/xcheck_official/atesx.json @@ -0,0 +1,98 @@ +{ + "symbol": "atesx", + "name": "Anchor Risk Mgd Equity Strategies Instl", + "filing": { + "accession": "0001580642-24-002611", + "doc": "anchor-semi_annual.htm", + "period_end": "2024-02-29", + "display": "Northern Lights Fund Trust IV (CIK 0001644419)", + "ciks": [ + "0001644419" + ], + "score": 9.239216 + }, + "how": "ticker", + "n_bytes": 769007, + "verdict": "weak-match", + "class": "?", + "mean_rel_err": 0.761, + "nav_check": null, + "periods": [ + { + "period": "current", + "end": "2020-08-31", + "official": 0.05, + "local": 5.868, + "diff": -5.818, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2010-08-31", + "official": 0.14, + "local": 0, + "diff": 0.14, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2015-08-31", + "official": 1.31, + "local": 0, + "diff": 1.31, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2024-02-29", + "official": 0.14, + "local": 0.108, + "diff": 0.032, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2023-08-31", + "official": 0.77, + "local": 1.967, + "diff": -1.197, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2022-08-31", + "official": 10.5, + "local": 0, + "diff": 10.5, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2021-08-31", + "official": 9.98, + "local": 2.708, + "diff": 7.272, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2019-08-31", + "official": 10.32, + "local": 3.16, + "diff": 7.16, + "ok": false, + "unbounded": true + } + ], + "other_classes": [ + "?", + "?" + ] +} \ No newline at end of file diff --git a/reports/xcheck_official/atrfx.json b/reports/xcheck_official/atrfx.json new file mode 100644 index 0000000..a3bb1e1 --- /dev/null +++ b/reports/xcheck_official/atrfx.json @@ -0,0 +1,6 @@ +{ + "symbol": "atrfx", + "name": "Catalyst Systematic Alpha I", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (6 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/bnd.json b/reports/xcheck_official/bnd.json new file mode 100644 index 0000000..88c8b9e --- /dev/null +++ b/reports/xcheck_official/bnd.json @@ -0,0 +1,89 @@ +{ + "symbol": "bnd", + "name": "Vanguard Total Bond Market Index Fund ETF Shares", + "filing": { + "accession": "0001104659-26-102243", + "doc": "tm2620528d3_ncsrs.htm", + "period_end": "2026-06-30", + "display": "VANGUARD BOND INDEX FUNDS (CIK 0000794105)", + "ciks": [ + "0000794105" + ], + "score": 1.4955673 + }, + "how": "ticker", + "n_bytes": 69871730, + "verdict": "ok", + "class": "ETF Shares", + "mean_rel_err": 0.0002, + "nav_check": { + "date": "2021-12-31", + "official": 84.77, + "local": 84.75, + "rel": 0.0002, + "ok": true + }, + "periods": [ + { + "period": "current", + "end": "2026-06-30", + "official": 1.212, + "local": 1.212, + "diff": 0.0, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2025-12-31", + "official": 2.857, + "local": 2.857, + "diff": 0.0, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2024-12-31", + "official": 2.638, + "local": 2.639, + "diff": -0.001, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2023-12-31", + "official": 2.269, + "local": 2.27, + "diff": -0.001, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2022-12-31", + "official": 1.867, + "local": 1.867, + "diff": 0.0, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2021-12-31", + "official": 1.798, + "local": 36.755, + "diff": -34.957, + "ok": false, + "unbounded": true + } + ], + "other_classes": [ + "Investor Shares", + "Admiral Shares", + "Institutional Shares", + "Institutional Plus Shares", + "Institutional Select Shares" + ] +} \ No newline at end of file diff --git a/reports/xcheck_official/cosix.json b/reports/xcheck_official/cosix.json new file mode 100644 index 0000000..1245757 --- /dev/null +++ b/reports/xcheck_official/cosix.json @@ -0,0 +1,6 @@ +{ + "symbol": "cosix", + "name": "Columbia Strategic Income A", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (4 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/cvsix.json b/reports/xcheck_official/cvsix.json new file mode 100644 index 0000000..c1f0145 --- /dev/null +++ b/reports/xcheck_official/cvsix.json @@ -0,0 +1,77 @@ +{ + "symbol": "cvsix", + "name": "Calamos Market Neutral Income A", + "filing": { + "accession": "0001193125-12-512223", + "doc": "d431187dncsr.htm", + "period_end": "2012-10-31", + "display": "CALAMOS INVESTMENT TRUST/IL (CIK 0000826732)", + "ciks": [ + "0000826732" + ], + "score": 27.455177 + }, + "how": "name:Calamos Market Neutral Income Fund (shared=4)", + "n_bytes": 7422370, + "verdict": "mismatch", + "class": "?", + "mean_rel_err": 0.0939, + "nav_check": { + "date": "2008-10-31", + "official": 10.97, + "local": 10.97, + "rel": 0.0, + "ok": true + }, + "periods": [ + { + "period": "current", + "end": "2012-10-31", + "official": 0.15, + "local": 0.153, + "diff": -0.003, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2011-10-31", + "official": 0.2, + "local": 0.2, + "diff": 0.0, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2010-10-31", + "official": 0.13, + "local": 0.128, + "diff": 0.002, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2009-10-31", + "official": 0.81, + "local": 1.104, + "diff": -0.294, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2008-10-31", + "official": 0.51, + "local": 14.311, + "diff": -13.801, + "ok": false, + "unbounded": true + } + ], + "other_classes": [ + "?", + "?" + ] +} \ No newline at end of file diff --git a/reports/xcheck_official/eagmx.json b/reports/xcheck_official/eagmx.json new file mode 100644 index 0000000..400d078 --- /dev/null +++ b/reports/xcheck_official/eagmx.json @@ -0,0 +1,6 @@ +{ + "symbol": "eagmx", + "name": "Eaton Vance Glbl Macr Absolute Return A", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (6 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/efa.json b/reports/xcheck_official/efa.json new file mode 100644 index 0000000..cf7678a --- /dev/null +++ b/reports/xcheck_official/efa.json @@ -0,0 +1,6 @@ +{ + "symbol": "efa", + "name": "iShares MSCI EAFE ETF", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (6 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/egrsx.json b/reports/xcheck_official/egrsx.json new file mode 100644 index 0000000..a9942ec --- /dev/null +++ b/reports/xcheck_official/egrsx.json @@ -0,0 +1,6 @@ +{ + "symbol": "egrsx", + "name": "Eaton Vance Glbl Macro Abs Ret Advtg R6", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (4 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/gld.json b/reports/xcheck_official/gld.json new file mode 100644 index 0000000..9656a7c --- /dev/null +++ b/reports/xcheck_official/gld.json @@ -0,0 +1,6 @@ +{ + "symbol": "gld", + "name": "SPDR Gold Shares", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (6 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/iwm.json b/reports/xcheck_official/iwm.json new file mode 100644 index 0000000..16f23ce --- /dev/null +++ b/reports/xcheck_official/iwm.json @@ -0,0 +1,6 @@ +{ + "symbol": "iwm", + "name": "iShares Russell 2000 ETF", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (6 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/jlpsx.json b/reports/xcheck_official/jlpsx.json new file mode 100644 index 0000000..b474931 --- /dev/null +++ b/reports/xcheck_official/jlpsx.json @@ -0,0 +1,6 @@ +{ + "symbol": "jlpsx", + "name": "JPMorgan US Large Cap Core Plus I", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (5 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/lamhx.json b/reports/xcheck_official/lamhx.json new file mode 100644 index 0000000..52fb85f --- /dev/null +++ b/reports/xcheck_official/lamhx.json @@ -0,0 +1,6 @@ +{ + "symbol": "lamhx", + "name": "Lord Abbett Dividend Growth R6", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (5 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/lcorx.json b/reports/xcheck_official/lcorx.json new file mode 100644 index 0000000..6298d2e --- /dev/null +++ b/reports/xcheck_official/lcorx.json @@ -0,0 +1,82 @@ +{ + "symbol": "lcorx", + "name": "Leuthold Core Investment Retail", + "filing": { + "accession": "0001999371-24-007094", + "doc": "lthd-ncsrs_033124.htm", + "period_end": "2024-03-31", + "display": "LEUTHOLD FUNDS INC (CIK 0001000351)", + "ciks": [ + "0001000351" + ], + "score": 39.260708 + }, + "how": "ticker", + "n_bytes": 6805383, + "verdict": "weak-match", + "class": "?", + "mean_rel_err": 0.5078, + "nav_check": null, + "periods": [ + { + "period": "current", + "end": "2024-03-31", + "official": 0.23, + "local": 1.99, + "diff": -1.76, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2023-09-30", + "official": 0.42, + "local": 3.064, + "diff": -2.644, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2022-09-30", + "official": 0.23, + "local": 2.3, + "diff": -2.07, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2021-09-30", + "official": 0.06, + "local": 0.05, + "diff": 0.01, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2020-09-30", + "official": 0.65, + "local": 0.584, + "diff": 0.066, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2019-09-30", + "official": 0.0, + "local": 31.111, + "diff": -31.111, + "ok": false, + "unbounded": true + } + ], + "other_classes": [ + "?", + "?", + "?", + "?" + ] +} \ No newline at end of file diff --git a/reports/xcheck_official/lcrix.json b/reports/xcheck_official/lcrix.json new file mode 100644 index 0000000..369b5d4 --- /dev/null +++ b/reports/xcheck_official/lcrix.json @@ -0,0 +1,82 @@ +{ + "symbol": "lcrix", + "name": "Leuthold Core Investment Institutional", + "filing": { + "accession": "0001999371-24-007094", + "doc": "lthd-ncsrs_033124.htm", + "period_end": "2024-03-31", + "display": "LEUTHOLD FUNDS INC (CIK 0001000351)", + "ciks": [ + "0001000351" + ], + "score": 39.260708 + }, + "how": "ticker", + "n_bytes": 6805383, + "verdict": "weak-match", + "class": "?", + "mean_rel_err": 0.5045, + "nav_check": null, + "periods": [ + { + "period": "current", + "end": "2024-03-31", + "official": 0.23, + "local": 1.989, + "diff": -1.759, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2023-09-30", + "official": 0.42, + "local": 3.078, + "diff": -2.658, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2022-09-30", + "official": 0.23, + "local": 2.3, + "diff": -2.07, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2021-09-30", + "official": 0.06, + "local": 0.05, + "diff": 0.01, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2020-09-30", + "official": 0.65, + "local": 0.593, + "diff": 0.057, + "ok": false, + "unbounded": false + }, + { + "period": "fy", + "end": "2019-09-30", + "official": 0.0, + "local": 21.246, + "diff": -21.246, + "ok": false, + "unbounded": true + } + ], + "other_classes": [ + "?", + "?", + "?", + "?" + ] +} \ No newline at end of file diff --git a/reports/xcheck_official/mbxix.json b/reports/xcheck_official/mbxix.json new file mode 100644 index 0000000..95ad699 --- /dev/null +++ b/reports/xcheck_official/mbxix.json @@ -0,0 +1,6 @@ +{ + "symbol": "mbxix", + "name": "Catalyst/Millburn Hedge Strategy I", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (6 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/pmaix.json b/reports/xcheck_official/pmaix.json new file mode 100644 index 0000000..d810907 --- /dev/null +++ b/reports/xcheck_official/pmaix.json @@ -0,0 +1,89 @@ +{ + "symbol": "pmaix", + "name": "Victory Pioneer Multi-Asset Income A", + "filing": { + "accession": "0001193125-26-146416", + "doc": "d126036dncsrs.htm", + "period_end": "2026-01-31", + "display": "Victory Portfolios IV (CIK 0002042316)", + "ciks": [ + "0002042316" + ], + "score": 34.46095 + }, + "how": "ticker", + "n_bytes": 5355742, + "verdict": "ok", + "class": "Class A", + "mean_rel_err": 0.0037, + "nav_check": { + "date": "2021-07-30", + "official": 11.67, + "local": 11.67, + "rel": 0.0, + "ok": true + }, + "periods": [ + { + "period": "current", + "end": "2026-01-31", + "official": 0.47, + "local": 0.457, + "diff": 0.013, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2025-07-31", + "official": 0.78, + "local": 0.779, + "diff": 0.001, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2024-07-31", + "official": 0.83, + "local": 0.828, + "diff": 0.002, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2023-07-31", + "official": 0.63, + "local": 0.636, + "diff": -0.006, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2022-07-31", + "official": 0.65, + "local": 0.651, + "diff": -0.001, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2021-07-31", + "official": 0.56, + "local": 6.413, + "diff": -5.853, + "ok": false, + "unbounded": true + } + ], + "other_classes": [ + "Class C", + "Class R6", + "Class Y", + "Class A", + "Class Y" + ] +} \ No newline at end of file diff --git a/reports/xcheck_official/pmfkx.json b/reports/xcheck_official/pmfkx.json new file mode 100644 index 0000000..01a2e58 --- /dev/null +++ b/reports/xcheck_official/pmfkx.json @@ -0,0 +1,89 @@ +{ + "symbol": "pmfkx", + "name": "Victory Pioneer Multi-Asset Income R6", + "filing": { + "accession": "0001193125-26-146416", + "doc": "d126036dncsrs.htm", + "period_end": "2026-01-31", + "display": "Victory Portfolios IV (CIK 0002042316)", + "ciks": [ + "0002042316" + ], + "score": 34.46095 + }, + "how": "ticker", + "n_bytes": 5355742, + "verdict": "ok", + "class": "Class R6", + "mean_rel_err": 0.0026, + "nav_check": { + "date": "2021-07-30", + "official": 12.02, + "local": 12.02, + "rel": 0.0, + "ok": true + }, + "periods": [ + { + "period": "current", + "end": "2026-01-31", + "official": 0.49, + "local": 0.487, + "diff": 0.003, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2025-07-31", + "official": 0.84, + "local": 0.838, + "diff": 0.002, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2024-07-31", + "official": 0.88, + "local": 0.882, + "diff": -0.002, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2023-07-31", + "official": 0.68, + "local": 0.684, + "diff": -0.004, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2022-07-31", + "official": 0.7, + "local": 0.7, + "diff": 0.0, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2021-07-31", + "official": 0.61, + "local": 4.571, + "diff": -3.961, + "ok": false, + "unbounded": true + } + ], + "other_classes": [ + "Class A", + "Class C", + "Class Y", + "Class A", + "Class Y" + ] +} \ No newline at end of file diff --git a/reports/xcheck_official/pmorx.json b/reports/xcheck_official/pmorx.json new file mode 100644 index 0000000..084dd32 --- /dev/null +++ b/reports/xcheck_official/pmorx.json @@ -0,0 +1,6 @@ +{ + "symbol": "pmorx", + "name": "Putnam Mortgage Opportunities A", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (5 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/qqq.json b/reports/xcheck_official/qqq.json new file mode 100644 index 0000000..5217e50 --- /dev/null +++ b/reports/xcheck_official/qqq.json @@ -0,0 +1,6 @@ +{ + "symbol": "qqq", + "name": "Invesco QQQ Trust", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (6 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/qspnx.json b/reports/xcheck_official/qspnx.json new file mode 100644 index 0000000..1e372e9 --- /dev/null +++ b/reports/xcheck_official/qspnx.json @@ -0,0 +1,6 @@ +{ + "symbol": "qspnx", + "name": "AQR Style Premia Alternative N", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (6 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/schd.json b/reports/xcheck_official/schd.json new file mode 100644 index 0000000..1ebeec9 --- /dev/null +++ b/reports/xcheck_official/schd.json @@ -0,0 +1,6 @@ +{ + "symbol": "schd", + "name": "Schwab U.S. Dividend Equity ETF", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (6 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/shv.json b/reports/xcheck_official/shv.json new file mode 100644 index 0000000..6fc2c29 --- /dev/null +++ b/reports/xcheck_official/shv.json @@ -0,0 +1,6 @@ +{ + "symbol": "shv", + "name": "iShares 0\u20131 Year Treasury Bond ETF", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (6 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/slv.json b/reports/xcheck_official/slv.json new file mode 100644 index 0000000..4ac49f4 --- /dev/null +++ b/reports/xcheck_official/slv.json @@ -0,0 +1,6 @@ +{ + "symbol": "slv", + "name": "iShares Silver Trust", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (6 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/svarx.json b/reports/xcheck_official/svarx.json new file mode 100644 index 0000000..c8f00a0 --- /dev/null +++ b/reports/xcheck_official/svarx.json @@ -0,0 +1,6 @@ +{ + "symbol": "svarx", + "name": "Spectrum Low Volatility Investor", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (5 tried)" +} \ No newline at end of file diff --git a/reports/xcheck_official/vea.json b/reports/xcheck_official/vea.json new file mode 100644 index 0000000..4c95492 --- /dev/null +++ b/reports/xcheck_official/vea.json @@ -0,0 +1,88 @@ +{ + "symbol": "vea", + "name": "Vanguard FTSE Developed Markets Index Fund ETF Shares", + "filing": { + "accession": "0001104659-26-102244", + "doc": "tm2620528d11_ncsrs.htm", + "period_end": "2026-06-30", + "display": "VANGUARD TAX-MANAGED FUNDS (CIK 0000923202)", + "ciks": [ + "0000923202" + ], + "score": 26.269861 + }, + "how": "ticker", + "n_bytes": 7676497, + "verdict": "ok", + "class": "FTSE Developed Markets ETF Shares", + "mean_rel_err": 0.0006, + "nav_check": { + "date": "2021-12-31", + "official": 51.14, + "local": 51.060001, + "rel": 0.0016, + "ok": true + }, + "periods": [ + { + "period": "current", + "end": "2026-06-30", + "official": 0.486, + "local": 0.486, + "diff": 0.0, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2025-12-31", + "official": 2.009, + "local": 2.01, + "diff": -0.001, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2024-12-31", + "official": 1.604, + "local": 1.605, + "diff": -0.001, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2023-12-31", + "official": 1.511, + "local": 1.512, + "diff": -0.001, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2022-12-31", + "official": 1.222, + "local": 1.223, + "diff": -0.001, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2021-12-31", + "official": 1.614, + "local": 16.87, + "diff": -15.256, + "ok": false, + "unbounded": true + } + ], + "other_classes": [ + "Investor Shares", + "Admiral Shares", + "Institutional Shares", + "Institutional Plus Shares" + ] +} \ No newline at end of file diff --git a/reports/xcheck_official/vti.json b/reports/xcheck_official/vti.json new file mode 100644 index 0000000..dfa8dfc --- /dev/null +++ b/reports/xcheck_official/vti.json @@ -0,0 +1,89 @@ +{ + "symbol": "vti", + "name": "Vanguard Morningstar Total Stock Market ETF", + "filing": { + "accession": "0001104659-26-102230", + "doc": "tm2620528d1_ncsrs.htm", + "period_end": "2026-06-30", + "display": "VANGUARD INDEX FUNDS (CIK 0000036405)", + "ciks": [ + "0000036405" + ], + "score": 24.098976 + }, + "how": "ticker", + "n_bytes": 23721562, + "verdict": "ok", + "class": "ETF Shares", + "mean_rel_err": 0.0002, + "nav_check": { + "date": "2021-12-31", + "official": 241.49, + "local": 241.440002, + "rel": 0.0002, + "ok": true + }, + "periods": [ + { + "period": "current", + "end": "2026-06-30", + "official": 2.042, + "local": 2.042, + "diff": 0.0, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2025-12-31", + "official": 3.757, + "local": 3.756, + "diff": 0.001, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2024-12-31", + "official": 3.674, + "local": 3.675, + "diff": -0.001, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2023-12-31", + "official": 3.413, + "local": 3.413, + "diff": 0.0, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2022-12-31", + "official": 3.183, + "local": 3.184, + "diff": -0.001, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2021-12-31", + "official": 2.93, + "local": 33.9345, + "diff": -31.0045, + "ok": false, + "unbounded": true + } + ], + "other_classes": [ + "Investor Shares", + "Admiral Shares", + "Institutional Shares", + "Institutional Plus Shares", + "Institutional Select Shares" + ] +} \ No newline at end of file diff --git a/reports/xcheck_official/vtsax.json b/reports/xcheck_official/vtsax.json new file mode 100644 index 0000000..45d5b7f --- /dev/null +++ b/reports/xcheck_official/vtsax.json @@ -0,0 +1,89 @@ +{ + "symbol": "vtsax", + "name": "Vanguard MStar Total Stk Mkt Idx Admiral", + "filing": { + "accession": "0001104659-26-102230", + "doc": "tm2620528d1_ncsrs.htm", + "period_end": "2026-06-30", + "display": "VANGUARD INDEX FUNDS (CIK 0000036405)", + "ciks": [ + "0000036405" + ], + "score": 2.8299427 + }, + "how": "ticker", + "n_bytes": 23721562, + "verdict": "ok", + "class": "Admiral Shares", + "mean_rel_err": 0.0, + "nav_check": { + "date": "2021-12-31", + "official": 117.56, + "local": 117.559998, + "rel": 0.0, + "ok": true + }, + "periods": [ + { + "period": "current", + "end": "2026-06-30", + "official": 0.985, + "local": 0.985, + "diff": 0.0, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2025-12-31", + "official": 1.814, + "local": 1.814, + "diff": 0.0, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2024-12-31", + "official": 1.776, + "local": 1.776, + "diff": 0.0, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2023-12-31", + "official": 1.651, + "local": 1.651, + "diff": 0.0, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2022-12-31", + "official": 1.54, + "local": 1.54, + "diff": 0.0, + "ok": true, + "unbounded": false + }, + { + "period": "fy", + "end": "2021-12-31", + "official": 1.415, + "local": 16.974, + "diff": -15.559, + "ok": false, + "unbounded": true + } + ], + "other_classes": [ + "Investor Shares", + "ETF Shares", + "Institutional Shares", + "Institutional Plus Shares", + "Institutional Select Shares" + ] +} \ No newline at end of file diff --git a/reports/xcheck_official/vwo.json b/reports/xcheck_official/vwo.json new file mode 100644 index 0000000..3ff179b --- /dev/null +++ b/reports/xcheck_official/vwo.json @@ -0,0 +1,6 @@ +{ + "symbol": "vwo", + "name": "Vanguard FTSE Emerging Markets Index Fund ETF Shares", + "verdict": "not-found", + "detail": "no candidate filing contained this fund (6 tried)" +} \ No newline at end of file diff --git a/scripts/verify_official.py b/scripts/verify_official.py new file mode 100644 index 0000000..2ebf959 --- /dev/null +++ b/scripts/verify_official.py @@ -0,0 +1,651 @@ +#!/usr/bin/env python3 +"""Tier-3 verification: cross-check local Yahoo distributions against +official SEC filings (N-CSR / N-CSRS / 10-K / 10-Q financial statements). + +For each symbol: + 1. EDGAR full-text search for the ticker restricted to annual/semi-annual + report forms; take the filing with the latest period end; + 2. fetch the document and locate the fund's Financial Highlights tables; + 3. parse every share-class block (per-share dividends from net investment + income, distributions from realized capital gains, NAV begin/end, + total return, for the current period + prior years); + 4. match the class to the ticker by the per-share distribution series + and by NAV magnitude vs the local year-end closes; + 5. compare official vs local (~/prog/fin/stocks CSVs, frozen copies for + stale tickers) per year and write a verdict. + +Verdicts: + ok best-matching class agrees on every bounded fiscal period + mismatch the class clearly matches but one or more periods differ + beyond tolerance -> real data finding; review the period + (per-date local rows vs the filing) and correct via + overrides/ (see reports/stale-funds.md conventions) + weak-match best class fit too poor to trust (uncovered document + format or wrong fund) -> manual review + not-found fund not located in any candidate filing + error fetch/parse failure (see the json for details) + +Notes: + - The oldest column has an unbounded local window (all earlier history) + and is displayed as ~ but never drives the verdict. + - Distributions are compared ex-date based, in the period window + (prev period end, period end]; this handles non-calendar fiscal years. + - ETF/annual-report formats are covered only partially (several ETF + families come back not-found); open-end N-CSR/N-CSRS formats of the + Vanguard/Victory style are the well-covered case. + +Usage: + python3 scripts/verify_official.py [SYM ...] # default: funds.json + python3 scripts/verify_official.py --stale # + all frozen tickers + python3 scripts/verify_official.py --json-only # skip SEC, re-emit report + +Output: reports/xcheck_official/{SYM}.json + SUMMARY.md +Politely rate-limited (~1 doc per second). Cached results are re-used +when the requested period is already covered (no re-fetch). +""" + +from __future__ import annotations + +import argparse +import json +import re +import sys +import time +from dataclasses import dataclass, field +from pathlib import Path +from typing import Optional + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +import fundlab.edgar as edgar # noqa: E402 (sec_get with UA + delay) + +import data as D # noqa: E402 + +ROOT = D.DEFAULT_ROOT +REPORT_DIR = Path(__file__).resolve().parent.parent / "reports" / "xcheck_official" +FUNDS = Path(__file__).resolve().parent.parent / "funds.json" + +EFTS_URL = "https://efts.sec.gov/LATEST/search-index?q={q}&forms=N-CSR,N-CSRS,10-K,10-Q" +FORMS = ("N-CSR", "N-CSRS", "10-K", "10-Q") +_STOP = {"fund", "funds", "inc", "ltd", "llc", "trust", "company", "the", + "and", "of", "for", "class", "series", "shares", "share", "in", + "on", "at", "to", "a", "an", "with", "portfolio", "portfolios", + "investment", "investments", "admiral", "investor", "investors", + "institutional", "etf", "plus", "select", "retail", "wholesale", + "group", "capital", "global", "us", "u.s."} + +TOL_REL = 0.015 # 1.5% per year +TOL_ABS = 0.015 # or $0.015/share (official figures are 3-dp) +NAV_TOL = 0.25 # class match: |local_close - official_nav| / nav + + +# ---------------------------------------------------------------- parsing +MONTHS = ("January|February|March|April|May|June|July|August|September|" + "October|November|December") + + +_MON = {m: f"{i + 1:02d}" for i, m in enumerate( + [x for x in MONTHS.split("|")])} + + +def _num(window: str) -> Optional[float]: + c = window.strip().replace("$", "").replace(",", "") + neg = c.startswith("(") and c.endswith(")") + c = c.strip("() ") + try: + v = float(c) + except ValueError: + return None + return -v if neg else v + + +# row labels found in fund Financial Highlights tables; used to bound +# each row's value window so one row's numbers never bleed into the next +_ROW_LBL = (r"Net Asset Value,\s*(?:Beginning|End) of Period" + r"|Net Investment Income" + r"|Capital Gain Distributions Received" + r"|Net Realized(?:\s+and\s+Unrealized)?" + r"|Total from Investment Operations" + r"|Total Distributions?" + r"|Dividends?\s+from\s+(?:Net\s+)?" + r"|Distributions?\s+from\s+(?:Net\s+)?" + r"|Total Return" + r"|Ratios/Supplemental") + + +def _decimals(text: str, start: int, end: int, n: int) -> Optional[list]: + """The n decimal numbers in `text[start:end]`, in order. + + Footnote markers (bare integers) and dashes are skipped because only + values with a decimal point are accepted; negative values may span + multiple cells ("(.961 | )"), which the pipe-stripping handles. + """ + w = re.sub(r"[|\s]+", "", text[start:end]) + w = w.replace("—", "-").replace(" ", "") + vals = [] + for m in re.finditer(r"\((\d+\.\d+|\.\d+)\)|(\d+\.\d+|\.\d+)", w): + v = float(m.group(1) or m.group(2)) + vals.append(-v if m.group(1) else v) + if len(vals) == n: + return vals + return None + + +_MON3 = {"jan": "01", "feb": "02", "mar": "03", "apr": "04", "may": "05", + "jun": "06", "jul": "07", "aug": "08", "sep": "09", "oct": "10", + "nov": "11", "dec": "12"} + + +def _parse_date(s: str) -> Optional[str]: + """'June 30, 2026' | '1/31/26' | 'Dec-31-25' -> ISO date.""" + s = re.sub(r"&#\w+;", " ", s).strip() + m = re.match(rf"((?:{MONTHS}))\.?\s+(\d{{1,2}}),?\s+(\d{{2,4}})", s, re.I) + if m: + y = m.group(3) if len(m.group(3)) == 4 else "20" + m.group(3) + return f"{y}-{_MON[m.group(1).title()]}-{int(m.group(2)):02d}" + m = re.match(r"(\d{1,2})[/-](\d{1,2})[/-](\d{2,4})", s) + if m: + y = m.group(3) if len(m.group(3)) == 4 else "20" + m.group(3) + if 1 <= int(m.group(1)) <= 12: + return f"{y}-{int(m.group(1)):02d}-{int(m.group(2)):02d}" + m = re.match(r"((?:" + "|".join(_MON3) + r"))\.?\s*[-/]\s*(\d{1,2})" + r"[-/]\s*(\d{2,4})", s, re.I) + if m: + y = m.group(3) if len(m.group(3)) == 4 else "20" + m.group(3) + return f"{y}-{_MON3[m.group(1).lower()]}-{int(m.group(2)):02d}" + return None + + +def _header_periods(text: str, i0: int, i_nav: int) -> list[dict]: + """Period labels from the header between the class label and the first + 'Net asset value, beginning' row. + + Handles both styles: + Six Months Ended | June 30, | 2026 | Year Ended December 31, | 2025 + | 2024 | ... (month-name dates + bare 4-digit years) + Six Months Ended 1/31/26 | Year Ended 7/31/25 | 7/31/24 | ... + (M/D/YY dates) + Bare years inherit the month-day of the preceding dated column. + """ + hdr = re.sub(r"&#\w+;", " ", text[i0:i_nav]) + hdr = re.sub(r"[|\s]+", " ", hdr).strip() + # columns are newest-first and dated/bare-year columns may interleave; + # scan left to right in a single pass, never sort + periods = [] + state = {"md": None} + + def _eat(m): + g = m.group(0) + iso = _parse_date(g) + if iso: + state["md"] = iso[5:] + periods.append({"end": iso, "year": iso[:4], "kind": "fy"}) + return " " + y = g.strip() + if not (2 <= len(y) <= 4) or not y.isdigit(): + return g + if len(y) == 2: + y = "20" + y + if not (1950 <= int(y) <= 2040): + return g + if not state["md"]: + state["md"] = "12-31" + periods.append({"end": f"{y}-{state['md']}", "year": y, "kind": "fy"}) + return " " + + hdr = re.sub(rf"(?:{MONTHS})\.?\s+\d{{1,2}},?\s+\d{{2,4}}" + rf"|\d{{1,2}}[/-]\d{{1,2}}[/-]\d{{2,4}}" + r"|\b(19\d{2}|20\d{2}|\d{2})\b", + _eat, hdr, flags=re.I) + # drop duplicate ends, keep first + seen, out = set(), [] + for p in periods: + if p["end"] in seen: + continue + seen.add(p["end"]) + out.append(p) + if len(out) >= 2 and out[0]["end"] >= out[1]["end"]: + out[0]["kind"] = "current" + return out + + +def parse_highlights(text: str) -> list[dict]: + """All share-class Financial Highlights blocks in `text` (pipe-delimited, + tag-stripped). A block starts at a 'Net asset value, beginning' row + (both Vanguard's 'Net Asset Value, Beginning of Period' and Victory's + 'Net asset value, beginning of period' style).""" + starts = [m.start() for m in + re.finditer(r"net\s+asset\s+value,\s*beginning", text, re.I)] + blocks = [] + for i, s in enumerate(starts): + end = starts[i + 1] if i + 1 < len(starts) else s + 25000 + pre = re.sub(r"&#\w+;", " ", text[max(0, s - 900):s]) + cells = [c.strip() for c in pre.split("|") if c.strip()] + # class label: nearest "... Shares" / "Class X" cell; the cells + # right before the row are often column years, not the label + label = "?" + for c in reversed(cells): + if re.fullmatch(r"[A-Za-z][A-Za-z0-9 \-]*\s+Shares", c) or \ + re.fullmatch(r"Class\s+[A-Z0-9]+", c): + label = re.sub(r"\s+", " ", c).strip() + break + # header window: back at most 2000 chars, but never across a + # "Table of Contents" page footer (its page numbers would be read + # as bare year columns) + h0 = max(0, s - 2000) + toc = text.rfind("Table of Contents", h0, s) + fh = text.rfind("Financial Highlights", h0, s) + h0 = max(h0, toc, fh) + periods = _header_periods(text, h0, s) + if len(periods) < 2: + continue + n = len(periods) + i_end = min(end, text.find("See accompanying", s) if + text.find("See accompanying", s) != -1 else end) + + def row(pattern: str) -> Optional[list]: + mm = re.search(pattern, text[s:i_end], re.I) + if not mm: + return None + i0, i1 = s + mm.end(), i_end + for b in [x.start() for x in re.finditer(_ROW_LBL, + text[s:i_end], re.I)]: + if b > mm.end(): + i1 = s + b + break + return _decimals(text, i0, i1, n) + + # distribution rows: prefer explicit "from net investment income" / + # "from realized capital gains" labels (Vanguard); otherwise the + # "Distributions to shareholders:" sub-table (Victory), where the + # NII row is the first "Net investment income" after that header + ni = row(r"dividends?\s+from\s+(?:net\s+)?investment\s+income") + cg = row(r"(?:distributions?|dividends?)\s+from\s+(?:net\s+)?realized\s+capital\s+gains?") + if ni is None: + dm = re.search(r"distributions?\s+to\s+shareholders", text[s:i_end], re.I) + if dm: + mm = re.search(r"net\s+investment\s+income", text[s + dm.end():i_end], re.I) + if mm: + i0 = s + dm.end() + mm.end() + ni = _decimals(text, i0, i_end, n) + tot = row(r"total\s+distributions?\b") + if ni is None and tot is None: + continue + if tot is None: + if ni is None or cg is None: + continue + tot = [round(abs(a) + abs(b), 4) for a, b in zip(ni, cg)] + # block values: this block's NAV rows + navb = row(r"net\s+asset\s+value,\s*beginning") + nave = row(r"net\s+asset\s+value,\s*end") + blocks.append({"class": label, "periods": periods, + "nii": [abs(v) for v in ni] if ni else None, + "capg": [abs(v) for v in cg] if cg else None, + "total": [abs(v) for v in tot], + "nav_begin": navb, "nav_end": nave}) + return blocks + + +def html_to_cells_text(html: bytes) -> str: + t = re.sub(rb"<\s*(br|/p|/div|/tr|/h[1-6])[^>]*>", b"|\n", html, flags=re.I) + t = re.sub(rb"<[^>]+>", b"|", t) + t = t.decode("utf-8", "replace") + t = re.sub(r" ?;", " ", t) + return t + + +# ---------------------------------------------------------------- EFTS +def _efits(q: str) -> list[dict]: + """EFTS hits -> [{accession, doc, period_end, display, ciks, score}].""" + import urllib.parse + d = json.loads(edgar.sec_get(EFTS_URL.format(q=urllib.parse.quote(q))) + .decode("utf-8", "replace")) + out, seen = [], set() + for h in d.get("hits", {}).get("hits", [])[:20]: + src = h.get("_source", {}) + acc, _, doc = h.get("_id", "").partition(":") + if acc in seen or not doc.lower().endswith((".htm", ".html")): + continue + seen.add(acc) + out.append({"accession": acc, "doc": doc, + "period_end": src.get("period_ending") or "", + "display": src.get("display_names", ["?"])[0], + "ciks": src.get("ciks", []), + "score": h.get("_score", 0)}) + out.sort(key=lambda f: f["period_end"], reverse=True) + return out + + +def find_filings(ticker: str, name: str, limit: int = 6) -> list[dict]: + """Candidate filings, best first. A bare ticker matches hundreds of + unrelated documents, so: (1) the full fund name as one phrase (it only + appears in the fund's own filings), (2) ticker + distinctive name + words, (3) the bare ticker.""" + def merge(f: list[dict]) -> None: + for x in f: + if x["accession"] not in {y["accession"] for y in out}: + out.append(x) + out: list[dict] = [] + words = [w for w in re.findall(r"[A-Za-z][A-Za-z\-'.]{2,}", name) + if w.lower() not in _STOP] + if len(words) >= 4: + merge(_efits('"' + name + '"')) + if len(out) < limit and words: + merge(_efits('"' + ticker + '" AND ' + + " AND ".join('"' + w + '"' for w in words[:3]))) + if len(out) < limit: + merge(_efits('"' + ticker + '"')) + return out[:limit] + + +def _title_runs(seg: str) -> list[str]: + return re.findall(r"[A-Z][A-Za-z0-9'\u2019\-]*" + r"(?:\s+[A-Z][A-Za-z0-9'\u2019\-]*)+", seg) + + +def fund_region(text: str, ticker: str, name: str) -> Optional[str]: + """The slice of the document holding this fund's highlights tables. + + The ticker appears only in the TOC/cover, not in the financial + statements. The TOC cell " Shares - TICKER" is preceded by the + official fund name cell; section headers in the statements drop + prefixes like "Vanguard ", so anchor on the name or a tail of it. + """ + occs = [m.start() for m in re.finditer(rf"(?= 3: + cands.append(prev) + break + cands.append(name) + for cand in dict.fromkeys(cands): + words = cand.split() + if len(words) < 4: + continue + for tail_n in range(len(words), 3, -1): + anchor = " ".join(words[-tail_n:]) + for m in re.finditer(re.escape(anchor), text, re.I): + fh = text.find("Financial Highlights", m.start(), m.start() + 4000) + if fh == -1: + continue + reg = text[fh:fh + 150000] + if parse_highlights(reg): + return reg + return None + + +def _wordset(s: str) -> set: + return set(re.findall(r"[a-z0-9]{3,}", s.lower())) - _STOP + + +def region_by_name(text: str, name: str, min_shared: int = 3): + """Fallback when the ticker is absent from the document (families that + don't publish distribution tickers): match the fund by name — the cell + immediately before each Financial Highlights heading — and keep the + best word-overlap candidate with a parseable region.""" + target = _wordset(name) + best = None + for m in re.finditer(r"Financial Highlights", text): + pre = re.sub(r"&#\w+;", " ", text[max(0, m.start() - 400):m.start()]) + pre = re.sub(r"[\s|]+", " ", pre).strip() + runs = re.findall(r"[A-Z][A-Za-z0-9'\u2019\-]*" + r"(?: [A-Z][A-Za-z0-9'\u2019\-]*){2,11}", pre) + if not runs: + continue + for cand in dict.fromkeys(runs): # any run, best overlap wins + shared = len(_wordset(cand) & target) + if shared < min_shared: + continue + reg = text[m.start():m.start() + 150000] + if not parse_highlights(reg): + continue + if best is None or shared > best[0]: + best = (shared, reg, cand) + return best + + +# ---------------------------------------------------------------- compare +def local_series(sym: str, root: Path, frozen: Path) -> dict: + """Local (date, amount) distributions and close series; frozen copies + shadow the data root.""" + def read(sfx: str): + p = frozen / f"{sym}-{sfx}.csv" + if not p.exists(): + p = root / f"{sym}-{sfx}.csv" + if not p.exists(): + return [] + out = [] + for line in p.read_text(errors="replace").splitlines()[1:]: + parts = line.split(",") + if len(parts) >= 2: + out.append((parts[0][:10], float(parts[1]))) + return out + dist = sorted(read("dividend") + read("capitalGain")) + p = frozen / f"{sym}-history.csv" + if not p.exists(): + p = root / f"{sym}-history.csv" + closes = [] + if p.exists(): + for line in p.read_text(errors="replace").splitlines()[1:]: + parts = line.split(",") + if len(parts) >= 5: + closes.append((parts[0][:10], float(parts[4]))) + return {"dist": dist, "close": closes} + + +def _window_sum(dist, lo, hi): + """Sum of distributions with lo < date <= hi (ISO date strings).""" + return round(sum(v for d, v in dist if (lo is None or d > lo) and d <= hi), + 4) + + +def compare(sym: str, blocks: list[dict], loc: dict) -> dict: + """Pick the best-matching class and compare period by period. + + Each official column covers (previous period end, this period end]; + local distributions are summed over the same window, so non-calendar + fiscal years compare correctly. + """ + def period_errs(b): + errs = [] + for i, p in enumerate(b["periods"]): + if p["kind"] != "fy": + continue + if i + 1 >= len(b["periods"]): + continue # oldest column: unbounded window, unverifiable + off = abs(b["total"][i]) + lo = b["periods"][i + 1]["end"] + lc = _window_sum(loc["dist"], lo, p["end"]) + errs.append(abs(off - lc) / max(off, lc, 1e-9)) + return errs + + best = None + for b in blocks: + errs = period_errs(b) + if not errs: + continue + me = sum(errs) / len(errs) + if best is None or me < best["err"]: + best = {"err": me, "block": b} + if best is None: + return {"verdict": "not-found", "classes": len(blocks), "detail": + "no class block with overlapping periods"} + b = best["block"] + rows = [] + mismatch = False + for i, p in enumerate(b["periods"]): + off = abs(b["total"][i]) + # columns are newest-first: the window lower bound is the NEXT + # (older) column's end; the oldest column is unbounded + lo = b["periods"][i + 1]["end"] if i + 1 < len(b["periods"]) else None + lc = _window_sum(loc["dist"], lo, p["end"]) + d = off - lc + ok = abs(d) <= TOL_ABS or abs(d) / max(off, 1e-9) <= TOL_REL + mismatch |= p["kind"] == "fy" and lo is not None and not ok + rows.append({"period": p["kind"], "end": p["end"], + "official": round(off, 4), "local": lc, + "diff": round(d, 4), "ok": ok, + "unbounded": lo is None}) + # NAV check: most recent period with an official ending NAV + nav = None + if b.get("nav_end") and loc["close"]: + closes = dict(loc["close"]) + for i in range(len(b["periods"]) - 1, -1, -1): + p = b["periods"][i] + prior = [d for d in closes if d <= p["end"]] + if not prior: + continue + o, l = abs(b["nav_end"][i]), closes[prior[-1]] + nav = {"date": prior[-1], "official": o, "local": l, + "rel": round(abs(o - l) / o, 4), + "ok": abs(o - l) / o <= NAV_TOL} + break + # verdict: a good class fit with a bad year is a real mismatch; a bad + # class fit overall means the document format (or fund) isn't matched + if best["err"] >= 0.30: + verdict = "weak-match" + else: + verdict = "mismatch" if mismatch else "ok" + return {"verdict": verdict, + "class": b["class"], "mean_rel_err": round(best["err"], 4), + "nav_check": nav, "periods": rows, + "other_classes": [x["class"] for x in blocks if x is not b]} + + +# ---------------------------------------------------------------- driver +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("syms", nargs="*") + ap.add_argument("--stale", action="store_true", + help="also verify all frozen (stale) tickers") + ap.add_argument("--json-only", action="store_true") + ap.add_argument("--limit", type=int, default=0) + args = ap.parse_args() + + syms = [s.lower() for s in args.syms] + if not syms and not args.stale: + syms = list(json.loads(FUNDS.read_text()).keys()) + if args.stale: + syms += sorted(f.stem for f in (D.FROZEN_DIR.glob("*.json"))) + syms = list(dict.fromkeys(syms)) + if args.limit: + syms = syms[:args.limit] + + REPORT_DIR.mkdir(parents=True, exist_ok=True) + summary = [] + t0 = time.monotonic() + for n, sym in enumerate(syms, 1): + out = REPORT_DIR / f"{sym}.json" + cached = None + if out.exists() and not args.json_only: + cached = json.loads(out.read_text()) + if not (cached.get("filing") or cached.get("period_end")): + cached = None # not-found from a prior run: retry + if cached: + summary.append(cached) + print(f"[{n}/{len(syms)}] {sym} cached {cached['verdict']}", flush=True) + continue + name = "" + meta_f = ROOT / f"{sym}.json" + if meta_f.exists(): + try: + name = json.loads(meta_f.read_text())["chart"]["result"][0] \ + .get("meta", {}).get("longName", "") + except Exception: + pass + if not name: + mf = D.FROZEN_DIR / f"{sym}.json" + if mf.exists(): + name = json.loads(mf.read_text()).get("longName", "") + try: + if args.json_only: + raise RuntimeError("json-only") + cands = find_filings(sym.upper(), name) + rec = {"symbol": sym, "name": name} + for filing in cands: + try: + doc = edgar.sec_get( + f"https://www.sec.gov/Archives/edgar/data/" + f"{int(filing['ciks'][0]):010d}/{filing['accession'].replace('-', '')}/" + f"{filing['doc']}") + except Exception: + continue + text = html_to_cells_text(doc) + region, how = None, None + if re.search(rf"(? {REPORT_DIR / 'SUMMARY.md'}") + return 0 + + +if __name__ == "__main__": + sys.exit(main())