fundlab: actively-managed fund support (16/16 tickers)
- ticker -> CIK scoped FTS (ciks filter) over the fund's own registrant, word queries shed Yahoo abbreviations shortest-first - fuzzy cover gate (SequenceMatcher, sliding window for wrapped names) + abbreviation table (Mgd/Glbl/Macr/Abs/Ret/Advtg/...) applied to both sides of the match - family-section matching inside multi-fund filings; similarity tiers: gate (0.85) returns immediately, floor (0.70) is a CIK-scoped fallback ranked against sibling funds' documents (best sim, then newest) - objective phrasings: 'seeks to ...', 'seeks investment results ...', 'The Fund's investment objective is ...', '(the Fund) investment objective is ...' (boilerplate 'is not fundamental' rejected); TOC headings skipped by trying all heads - sec_get retries on mid-stream connection drops
This commit is contained in:
parent
e9f8dfc462
commit
5c6d95a8b4
339
fundlab/edgar.py
339
fundlab/edgar.py
|
|
@ -10,6 +10,7 @@ well under 10 requests/second (we insert REQUEST_DELAY between calls).
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import http.client
|
||||||
import json
|
import json
|
||||||
import re
|
import re
|
||||||
import time
|
import time
|
||||||
|
|
@ -63,7 +64,8 @@ def sec_get(url: str, timeout: int = 120, retries: int = 3) -> bytes:
|
||||||
time.sleep(2.0 * (attempt + 1))
|
time.sleep(2.0 * (attempt + 1))
|
||||||
continue
|
continue
|
||||||
raise
|
raise
|
||||||
except (urllib.error.URLError, TimeoutError) as e:
|
except (urllib.error.URLError, TimeoutError,
|
||||||
|
http.client.HTTPException) as e:
|
||||||
if attempt + 1 < retries:
|
if attempt + 1 < retries:
|
||||||
time.sleep(2.0 * (attempt + 1))
|
time.sleep(2.0 * (attempt + 1))
|
||||||
continue
|
continue
|
||||||
|
|
@ -122,16 +124,17 @@ def cik_recent_filings(cik: str, types: str, count: int = 6) -> list[dict]:
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
def fts_search(query: str, forms: str | None = None, size: int = 20) -> list[dict]:
|
def fts_search(query: str, forms: str | None = None, size: int = 20,
|
||||||
|
ciks: str | None = None) -> list[dict]:
|
||||||
"""EDGAR full-text search.
|
"""EDGAR full-text search.
|
||||||
|
|
||||||
Hits are sorted by relevance, then date: a fund's own prospectus ranks
|
Hits are sorted by relevance, then date. `ciks` restricts to one
|
||||||
above documents that merely mention the ticker (funds investing in it,
|
registrant — combined with the cover gates this is how a fund's own
|
||||||
family fund lists). Relevance alone is not enough — section attribution
|
filing is found inside a fund family's filings.
|
||||||
in extract_objective is the real guard — but ordering by it means the
|
|
||||||
first match is usually the right fund's.
|
|
||||||
"""
|
"""
|
||||||
params = {"q": query, "forms": forms or "", "size": str(size)}
|
params = {"q": query, "forms": forms or "", "size": str(size)}
|
||||||
|
if ciks:
|
||||||
|
params["ciks"] = ciks
|
||||||
url = f"https://efts.sec.gov/LATEST/search-index?{urllib.parse.urlencode(params)}"
|
url = f"https://efts.sec.gov/LATEST/search-index?{urllib.parse.urlencode(params)}"
|
||||||
d = json.loads(sec_get(url, timeout=60))
|
d = json.loads(sec_get(url, timeout=60))
|
||||||
hits = d.get("hits", {}).get("hits", [])
|
hits = d.get("hits", {}).get("hits", [])
|
||||||
|
|
@ -163,6 +166,18 @@ _SEEK = (r"((?:the|its) (?:fund|trust|etf|portfolio)\s*\)?[^\.\n]{0,120}?"
|
||||||
r"seeks (?:to |investment results |investment objective)?"
|
r"seeks (?:to |investment results |investment objective)?"
|
||||||
r"[^.]{10,400}\.)")
|
r"[^.]{10,400}\.)")
|
||||||
|
|
||||||
|
# alternate phrasing: "The Fund's investment objective is ... ." — the
|
||||||
|
# subject sits BEFORE the 'investment objective' anchor
|
||||||
|
_SEEK2 = (r"((?:the|its) (?:fund|trust|etf|portfolio)'?s?\s*"
|
||||||
|
r"investment objective is [^.]{5,300}\.?)")
|
||||||
|
# boilerplate to reject: "...investment objective is not fundamental ..."
|
||||||
|
_SEEK2_BAD = re.compile(
|
||||||
|
r"(?i)^(?:the|its) (?:fund|trust|etf|portfolio)'?s?\s*"
|
||||||
|
r"investment objective is not\b")
|
||||||
|
|
||||||
|
# yet another phrasing: "... (the Fund) investment objective is ... ."
|
||||||
|
_SEEK3 = r"((?:\(\s*the fund\s*\)?\s+investment objective is [^.]{5,300}\.))"
|
||||||
|
|
||||||
|
|
||||||
def _name_re(name: str) -> str:
|
def _name_re(name: str) -> str:
|
||||||
"""Pattern for a fund name; to_text() turns 'U.S.' into 'US', so the
|
"""Pattern for a fund name; to_text() turns 'U.S.' into 'US', so the
|
||||||
|
|
@ -207,6 +222,37 @@ def _on_cover(text: str, name: str | None, ticker: str | None,
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _objective_at(text: str, p: int) -> str | None:
|
||||||
|
"""Objective sentence at/below the heading at p (any phrasing)."""
|
||||||
|
m = re.search(_SEEK, text[p: p + 2500], re.I)
|
||||||
|
if m:
|
||||||
|
return _clean(m.group(1))
|
||||||
|
for pat in (_SEEK2, _SEEK3):
|
||||||
|
m = re.search(pat, text[max(0, p - 200): p + 2500], re.I)
|
||||||
|
if m and not _SEEK2_BAD.match(m.group(1)):
|
||||||
|
return _clean(m.group(1))
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _first_objective(text: str) -> str | None:
|
||||||
|
"""Objective sentence after an 'Investment Objective' heading (TOC
|
||||||
|
entries come first and have no sentence — try them all), or the first
|
||||||
|
'seeks ...' sentence early in the document."""
|
||||||
|
heads = [m.start() for m in re.finditer(r"investment objective", text, re.I)]
|
||||||
|
for p in heads[:25]:
|
||||||
|
obj = _objective_at(text, p)
|
||||||
|
if obj:
|
||||||
|
return obj
|
||||||
|
m = re.search(_SEEK, text[:15000], re.I)
|
||||||
|
if m:
|
||||||
|
return _clean(m.group(1))
|
||||||
|
for pat in (_SEEK2, _SEEK3):
|
||||||
|
m = re.search(pat, text[:15000], re.I)
|
||||||
|
if m and not _SEEK2_BAD.match(m.group(1)):
|
||||||
|
return _clean(m.group(1))
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
def extract_objective(text: str, name: str, ticker: str | None = None,
|
def extract_objective(text: str, name: str, ticker: str | None = None,
|
||||||
gate: int = COVER_GATE) -> str | None:
|
gate: int = COVER_GATE) -> str | None:
|
||||||
"""Extract the fund's investment objective sentence from a document
|
"""Extract the fund's investment objective sentence from a document
|
||||||
|
|
@ -217,16 +263,138 @@ def extract_objective(text: str, name: str, ticker: str | None = None,
|
||||||
"""
|
"""
|
||||||
if not _on_cover(text, name, ticker, gate):
|
if not _on_cover(text, name, ticker, gate):
|
||||||
return None
|
return None
|
||||||
|
return _first_objective(text)
|
||||||
|
|
||||||
|
|
||||||
|
# share-class / distribution words that Yahoo appends to fund names but
|
||||||
|
# that never appear in the fund's registered name
|
||||||
|
_STOP_WORDS = {"instl", "institutional", "investor", "retail", "admiral",
|
||||||
|
"shares", "share", "class", "fund", "r6", "r-6",
|
||||||
|
"r4", "r-4", "r3", "a", "b", "c", "i", "n", "z", "y"}
|
||||||
|
|
||||||
|
# Yahoo abbreviates fund names; map the common abbreviations back to the
|
||||||
|
# registered spelling so fuzzy matching sees the same words on both sides
|
||||||
|
_ABBR = {"mgd": "managed", "glbl": "global", "macr": "macro",
|
||||||
|
"abs": "absolute", "ret": "return", "advtg": "advantaged",
|
||||||
|
"strtgy": "strategy", "invst": "investment", "portf": "portfolio",
|
||||||
|
"opps": "opportunities", "mkt": "market", "eqy": "equity",
|
||||||
|
"incm": "income", "dist": "distribution", "div": "dividend",
|
||||||
|
"ltd": "limited", "intl": "international"}
|
||||||
|
|
||||||
|
FUZZY_GATE = 0.85 # name similarity needed to accept a document
|
||||||
|
FUZZY_FLOOR = 0.70 # weaker matches are still usable when RANKED against
|
||||||
|
# sibling funds' documents (CIK-scoped passes)
|
||||||
|
|
||||||
|
|
||||||
|
def _norm_name(name: str) -> str:
|
||||||
|
"""Lowercase, drop share-class words, expand Yahoo abbreviations — the
|
||||||
|
form used for fuzzy name matching (applied to names AND to the document
|
||||||
|
regions they are matched against)."""
|
||||||
|
out = []
|
||||||
|
for w in name.split():
|
||||||
|
w = w.lower().strip(".,'")
|
||||||
|
if w in _STOP_WORDS:
|
||||||
|
continue
|
||||||
|
out.append(_ABBR.get(w, w))
|
||||||
|
return " ".join(out)
|
||||||
|
|
||||||
|
|
||||||
|
def _sim(a: str, b: str) -> float:
|
||||||
|
from difflib import SequenceMatcher
|
||||||
|
return SequenceMatcher(None, a, b).ratio()
|
||||||
|
|
||||||
|
|
||||||
|
def _fuzzy_contains(region: str, target: str, threshold: float) -> float:
|
||||||
|
"""Best similarity of the target against the region, sliding a
|
||||||
|
target-length window (names wrap across lines in converted HTML, so
|
||||||
|
line-based matching is not enough). Both sides are normalized."""
|
||||||
|
region = _norm_name(re.sub(r"\s+", " ", region))
|
||||||
|
best = 0.0
|
||||||
|
for i in range(0, max(1, len(region) - len(target) + 1), 8):
|
||||||
|
s = _sim(target, region[i:i + len(target) + 6])
|
||||||
|
if s > best:
|
||||||
|
best = s
|
||||||
|
if best >= 1.0:
|
||||||
|
break
|
||||||
|
return best
|
||||||
|
|
||||||
|
|
||||||
|
def _fuzzy_cover(text: str, name: str, limit: int = 1500,
|
||||||
|
threshold: float = FUZZY_GATE) -> float:
|
||||||
|
"""Fuzzy cover gate: the fund's REGISTERED name (which differs from the
|
||||||
|
Yahoo name by abbreviations: Mgd/Managed, Glbl/Global, Macr/Macro) in
|
||||||
|
title position on the cover page. Returns the best similarity."""
|
||||||
|
target = _norm_name(name)
|
||||||
|
if len(target) < 8:
|
||||||
|
return 0.0
|
||||||
|
return _fuzzy_contains(text[:limit], target, threshold)
|
||||||
|
|
||||||
|
|
||||||
|
def extract_objective_family(text: str, name: str,
|
||||||
|
threshold: float = FUZZY_GATE
|
||||||
|
) -> tuple[float, str] | None:
|
||||||
|
"""Find the fund's section inside a FAMILY filing (one prospectus
|
||||||
|
covering many funds) and extract its objective.
|
||||||
|
|
||||||
|
The fund's own section header (name above the 'Fund Summary' / 'Investment
|
||||||
|
Objective' box) is fuzzy-matched against the fund name — robust to the
|
||||||
|
abbreviations Yahoo uses. Used only on documents of the fund's OWN
|
||||||
|
registrant (ticker -> CIK), where a name match is meaningful.
|
||||||
|
Returns (similarity, objective) or None.
|
||||||
|
"""
|
||||||
|
target = _norm_name(name)
|
||||||
|
if len(target) < 8:
|
||||||
|
return None
|
||||||
heads = [m.start() for m in re.finditer(r"investment objective", text, re.I)]
|
heads = [m.start() for m in re.finditer(r"investment objective", text, re.I)]
|
||||||
if heads:
|
for p in heads:
|
||||||
m = re.search(_SEEK, text[heads[0]: heads[0] + 2500], re.I)
|
s = _fuzzy_contains(text[max(0, p - 500):p], target, threshold)
|
||||||
if m:
|
if s >= threshold:
|
||||||
return _clean(m.group(1))
|
obj = _objective_at(text, p)
|
||||||
m = re.search(_SEEK, text[:15000], re.I)
|
if obj:
|
||||||
return _clean(m.group(1)) if m else None
|
return s, obj
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _eval_doc(text: str, name: str, ticker: str | None = None) -> tuple[float, str]:
|
||||||
|
"""Run all document-level strategies in order: exact cover gate, fuzzy
|
||||||
|
cover gate (registered name differs by abbreviations), family section
|
||||||
|
(large multi-fund filings). Returns (similarity, objective) or None."""
|
||||||
|
if _on_cover(text, name, ticker, COVER_GATE):
|
||||||
|
obj = _first_objective(text)
|
||||||
|
if obj:
|
||||||
|
return 1.0, obj
|
||||||
|
sim = _fuzzy_cover(text, name)
|
||||||
|
if sim >= FUZZY_GATE:
|
||||||
|
obj = _first_objective(text)
|
||||||
|
if obj:
|
||||||
|
return sim, obj
|
||||||
|
if len(text) > 100_000:
|
||||||
|
res = extract_objective_family(text, name)
|
||||||
|
if res:
|
||||||
|
return res
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _name_queries(name: str) -> list[str]:
|
||||||
|
"""FTS queries for a fund name: all significant words, then dropping
|
||||||
|
the SHORTEST words first (Yahoo abbreviations like Mgd/Glbl/Macr are
|
||||||
|
short and break quoted-phrase / word queries)."""
|
||||||
|
words = [w.strip(".,'") for w in name.split()
|
||||||
|
if w.lower().strip(".,'") not in _STOP_WORDS]
|
||||||
|
words = [w for w in words if len(w) > 1]
|
||||||
|
out = []
|
||||||
|
while len(words) >= 3:
|
||||||
|
out.append(" ".join(f'"{w}"' for w in words) + ' "seeks"')
|
||||||
|
words = sorted(words, key=len, reverse=True)[:-1] # drop shortest
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
def _clean(s: str) -> str:
|
def _clean(s: str) -> str:
|
||||||
|
# "(the Fund) investment objective is ..." -> "The Fund's ..."
|
||||||
|
s = re.sub(r"^(?:\(\s*)?((?:the|its) (?:fund|trust|etf|portfolio))"
|
||||||
|
r"\s*\)?\s+investment objective is",
|
||||||
|
r"\1's investment objective is", s, flags=re.I)
|
||||||
|
s = re.sub(r"^\(\s*", "", s)
|
||||||
# captures like "(the Fund) seeks to ..." leave a stray ')'
|
# captures like "(the Fund) seeks to ..." leave a stray ')'
|
||||||
s = re.sub(r"^\)\s*", "", s)
|
s = re.sub(r"^\)\s*", "", s)
|
||||||
s = re.sub(r"^((?:the|its) (?:fund|trust|etf|portfolio)"
|
s = re.sub(r"^((?:the|its) (?:fund|trust|etf|portfolio)"
|
||||||
|
|
@ -280,13 +448,19 @@ def accession_docs(cik: str, accession: str) -> list[str]:
|
||||||
|
|
||||||
|
|
||||||
def _try_docs(hits: list[dict], name: str, ticker: str | None,
|
def _try_docs(hits: list[dict], name: str, ticker: str | None,
|
||||||
max_docs: int, gate: int) -> dict | None:
|
max_docs: int, seen: set, floor: float = FUZZY_GATE) -> dict | None:
|
||||||
"""Fetch FTS hits (relevance order) and return the first document that
|
"""Fetch FTS hits (relevance order) and return the BEST document from
|
||||||
passes the cover gate and yields an objective sentence."""
|
which an objective can be extracted (see _eval_doc). `floor` is the
|
||||||
|
minimum name similarity: CIK-scoped passes can accept weaker matches
|
||||||
|
(sibling funds' documents rank below the right one), other passes
|
||||||
|
require the full gate."""
|
||||||
|
best: tuple[float, dict] | None = None
|
||||||
fetched = 0
|
fetched = 0
|
||||||
for h in hits:
|
for h in hits:
|
||||||
if fetched >= max_docs:
|
if fetched >= max_docs:
|
||||||
break
|
break
|
||||||
|
if (h["cik"], h["accession"], h["filename"]) in seen:
|
||||||
|
continue
|
||||||
# annual reports: the fund's part may be a sibling document of the
|
# annual reports: the fund's part may be a sibling document of the
|
||||||
# same filing, so enumerate the accession's documents
|
# same filing, so enumerate the accession's documents
|
||||||
names = [h["filename"]]
|
names = [h["filename"]]
|
||||||
|
|
@ -300,72 +474,125 @@ def _try_docs(hits: list[dict], name: str, ticker: str | None,
|
||||||
raw = sec_get(doc_url(h["cik"], h["accession"], fn))
|
raw = sec_get(doc_url(h["cik"], h["accession"], fn))
|
||||||
except urllib.error.HTTPError:
|
except urllib.error.HTTPError:
|
||||||
continue
|
continue
|
||||||
|
seen.add((h["cik"], h["accession"], fn))
|
||||||
fetched += 1
|
fetched += 1
|
||||||
if len(raw) > MAX_DOC_BYTES:
|
if len(raw) > MAX_DOC_BYTES:
|
||||||
continue
|
continue
|
||||||
obj = extract_objective(to_text(raw), name, ticker, gate)
|
res = _eval_doc(to_text(raw), name, ticker)
|
||||||
if obj:
|
if res and res[0] >= floor and (best is None or res[0] > best[0]):
|
||||||
return {
|
best = (res[0], {
|
||||||
"objective": obj,
|
"objective": res[1],
|
||||||
"category": classify_category(obj, name),
|
"category": classify_category(res[1], name),
|
||||||
"form": h["form"],
|
"form": h["form"],
|
||||||
"file_date": h["file_date"],
|
"file_date": h["file_date"],
|
||||||
"url": doc_url(h["cik"], h["accession"], fn),
|
"url": doc_url(h["cik"], h["accession"], fn),
|
||||||
}
|
})
|
||||||
return None
|
return best
|
||||||
|
|
||||||
|
|
||||||
def fetch_fund(fund_name: str, ticker: str | None = None,
|
def fetch_fund(fund_name: str, ticker: str | None = None,
|
||||||
max_docs: int = 8) -> dict | None:
|
max_docs: int = 8) -> dict | None:
|
||||||
"""Full pipeline for one fund. Returns metadata dict or None.
|
"""Full pipeline for one fund. Returns metadata dict or None.
|
||||||
|
|
||||||
Passes, most direct first:
|
Every candidate document is run through _eval_doc (exact cover gate,
|
||||||
0. ticker -> registrant CIK (standalone trusts like Invesco QQQ
|
then fuzzy cover gate for registered names that differ from the Yahoo
|
||||||
Trust / iShares Silver Trust): their most recent prospectus
|
name by abbreviations, then family-section matching). Passes, most
|
||||||
filings, cover-gated;
|
direct first:
|
||||||
1. fund-name search over prospectus forms — name in title position
|
0. ticker -> registrant CIK (the fund's own filer): its recent
|
||||||
or (TICKER) on the cover (COVER_GATE);
|
prospectus filings, then a CIK-scoped full-text search over the
|
||||||
2. fund-name search over annual reports — name in title position
|
fund name's words (shortest dropped first, to shed Yahoo
|
||||||
(TITLE_GATE);
|
abbreviations); the fuzzy gates are safe because the CIK is the
|
||||||
3. ticker search over prospectus forms — (TICKER) in title position
|
fund's own registrant;
|
||||||
(TITLE_GATE); catches funds whose registered name changed over
|
1. fund-name search over prospectus forms;
|
||||||
time.
|
2. fund-name search over annual reports;
|
||||||
|
3. fund-name WORDS over prospectus forms (abbreviation-tolerant);
|
||||||
|
4. ticker search over prospectus forms.
|
||||||
Returns None rather than a wrong fund's objective.
|
Returns None rather than a wrong fund's objective.
|
||||||
"""
|
"""
|
||||||
tk = ticker.upper() if ticker else None
|
tk = ticker.upper() if ticker else None
|
||||||
|
seen: set = set()
|
||||||
|
fallback: tuple[float, dict] | None = None
|
||||||
|
|
||||||
|
def _better(a: tuple[float, dict] | None,
|
||||||
|
b: tuple[float, dict] | None) -> tuple[float, dict] | None:
|
||||||
|
"""Best of two (similarity, metadata): higher sim, then newer date."""
|
||||||
|
if a is None:
|
||||||
|
return b
|
||||||
|
if b is None:
|
||||||
|
return a
|
||||||
|
ka = (a[0], a[1].get("file_date", ""))
|
||||||
|
kb = (b[0], b[1].get("file_date", ""))
|
||||||
|
return a if ka >= kb else b
|
||||||
|
|
||||||
|
def _try(query: str, forms: str, ciks: str | None = None,
|
||||||
|
floor: float = FUZZY_GATE) -> dict | None:
|
||||||
|
res = _try_docs(fts_search(query, forms, size=100, ciks=ciks),
|
||||||
|
fund_name, tk, max_docs, seen, floor=floor)
|
||||||
|
return res[1] if res else None
|
||||||
|
|
||||||
|
# 0) the fund's own registrant — documents here may carry WEAKER name
|
||||||
|
# matches (FUZZY_FLOOR): the right fund's document ranks above its
|
||||||
|
# siblings', so the best match across the CIK's documents wins
|
||||||
if tk:
|
if tk:
|
||||||
comp = ticker_to_company(tk)
|
comp = ticker_to_company(tk)
|
||||||
if comp:
|
if comp:
|
||||||
for fl in cik_recent_filings(comp[0], PROSPECTUS_FORMS, count=15):
|
cik = comp[0]
|
||||||
if not fl["doc"]:
|
gate_cik: tuple[float, dict] | None = None
|
||||||
|
floor_cik: tuple[float, dict] | None = None
|
||||||
|
|
||||||
|
def _collect(res: tuple[float, dict] | None) -> None:
|
||||||
|
nonlocal gate_cik, floor_cik
|
||||||
|
if not res or res[0] < FUZZY_FLOOR:
|
||||||
|
return
|
||||||
|
if res[0] >= FUZZY_GATE:
|
||||||
|
gate_cik = _better(gate_cik, res)
|
||||||
|
else:
|
||||||
|
floor_cik = _better(floor_cik, res)
|
||||||
|
|
||||||
|
for fl in cik_recent_filings(cik, PROSPECTUS_FORMS, count=15):
|
||||||
|
if not fl["doc"] or (cik, fl["accession"], fl["doc"]) in seen:
|
||||||
continue
|
continue
|
||||||
try:
|
try:
|
||||||
raw = sec_get(doc_url(comp[0], fl["accession"], fl["doc"]))
|
raw = sec_get(doc_url(cik, fl["accession"], fl["doc"]))
|
||||||
except urllib.error.HTTPError:
|
except urllib.error.HTTPError:
|
||||||
continue
|
continue
|
||||||
|
seen.add((cik, fl["accession"], fl["doc"]))
|
||||||
if len(raw) > MAX_DOC_BYTES:
|
if len(raw) > MAX_DOC_BYTES:
|
||||||
continue
|
continue
|
||||||
obj = extract_objective(to_text(raw), fund_name, tk, COVER_GATE)
|
res = _eval_doc(to_text(raw), fund_name, tk)
|
||||||
if obj:
|
if res:
|
||||||
return {
|
_collect((res[0], {
|
||||||
"objective": obj,
|
"objective": res[1],
|
||||||
"category": classify_category(obj, fund_name),
|
"category": classify_category(res[1], fund_name),
|
||||||
"form": fl["form"],
|
"form": fl["form"],
|
||||||
"file_date": fl["filed"],
|
"file_date": fl["filed"],
|
||||||
"url": doc_url(comp[0], fl["accession"], fl["doc"]),
|
"url": doc_url(cik, fl["accession"], fl["doc"]),
|
||||||
}
|
}))
|
||||||
q = f'"{fund_name}" "seeks"'
|
for q in _name_queries(fund_name):
|
||||||
res = _try_docs(fts_search(q, PROSPECTUS_FORMS, size=100),
|
_collect(_try_docs(fts_search(q, PROSPECTUS_FORMS, size=100,
|
||||||
fund_name, tk, max_docs, COVER_GATE)
|
ciks=cik),
|
||||||
if res:
|
fund_name, tk, max_docs, seen,
|
||||||
return res
|
floor=FUZZY_FLOOR))
|
||||||
res = _try_docs(fts_search(q, ANNUAL_FORMS, size=100),
|
if gate_cik:
|
||||||
fund_name, tk, max_docs, TITLE_GATE)
|
return gate_cik[1]
|
||||||
if res:
|
# floor-tier (weaker name match) is a FALLBACK: the ordinary
|
||||||
return res
|
# passes below may find a stronger document
|
||||||
if tk:
|
fallback = _better(fallback, floor_cik)
|
||||||
res = _try_docs(fts_search(f'"{tk}" "seeks"', PROSPECTUS_FORMS, size=100),
|
# 1) exact fund name
|
||||||
fund_name, tk, max_docs, TITLE_GATE)
|
for forms in (PROSPECTUS_FORMS, ANNUAL_FORMS):
|
||||||
|
res = _try(f'"{fund_name}" "seeks"', forms)
|
||||||
if res:
|
if res:
|
||||||
return res
|
return res
|
||||||
|
# 3) fund name words (abbreviation-tolerant)
|
||||||
|
for q in _name_queries(fund_name):
|
||||||
|
res = _try(q, PROSPECTUS_FORMS)
|
||||||
|
if res:
|
||||||
|
return res
|
||||||
|
# 4) ticker
|
||||||
|
if tk:
|
||||||
|
res = _try(f'"{tk}" "seeks"', PROSPECTUS_FORMS)
|
||||||
|
if res:
|
||||||
|
return res
|
||||||
|
if fallback:
|
||||||
|
return fallback[1]
|
||||||
return None
|
return None
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue
Block a user