Stock-Tool/dist/stock_screener.py

2754 lines
113 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""
Stock Evaluation Screener
=========================
Evaluates every common stock listed on NYSE, NASDAQ, and AMEX (~6 500+ tickers)
using a multi-factor composite scoring formula designed to act as a quantitative
analyst. This covers all major US indices: S&P 500/400/600, Russell 1000/2000/3000,
DJIA, NYSE Composite, NASDAQ Composite, and every sector/thematic index derived
from US-listed equities.
FORMULA (Composite Score):
Score = 0.18*Value + 0.18*Growth + 0.14*Momentum + 0.11*Quality
+ 0.08*Profitability + 0.13*Sentiment + 0.10*Analyst + 0.08*Risk
Each component is scored 0100 based on absolute benchmarks for that metric.
Scores are INDEPENDENT — each stock is evaluated on its own merits, not
relative to other stocks in the dataset.
Value — how attractively priced (P/E, P/B, EV/EBITDA vs benchmarks)
Growth — revenue, earnings, and EPS trajectory
Momentum — price return across 1M/3M/6M/12M windows
Quality — financial strength (ROE, ROA, debt, liquidity)
Profitability — margins and free cash flow yield
Sentiment — recent news tone via VADER NLP
Analyst — consensus rating and upside to price target
Risk — short interest, volatility, and beta stability
"""
import argparse
import html
import random
import re
import sys
import time
import warnings
from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import datetime, timedelta, timezone
import numpy as np
import pandas as pd
import requests
import yfinance as yf
warnings.filterwarnings("ignore")
# ---------------------------------------------------------------------------
# VADER sentiment — optional but strongly recommended
# pip install vaderSentiment
# ---------------------------------------------------------------------------
try:
from vaderSentiment.vaderSentiment import SentimentIntensityAnalyzer
_VADER = SentimentIntensityAnalyzer()
except ImportError:
_VADER = None
# ---------------------------------------------------------------------------
# Configuration
# ---------------------------------------------------------------------------
WEIGHTS = {
"value": 0.18,
"growth": 0.16,
"momentum": 0.13,
"quality": 0.14,
"profitability": 0.13,
"sentiment": 0.10,
"analyst": 0.10,
"risk": 0.06,
}
# Human-readable investment strategy presets.
# Each preset is a complete weight dict that must sum to 1.0.
STRATEGY_PRESETS = {
"Balanced (Default)": {
"value": 0.18, "growth": 0.16, "momentum": 0.13, "quality": 0.14,
"profitability": 0.13, "sentiment": 0.10, "analyst": 0.10, "risk": 0.06,
},
"High Growth Companies": {
"value": 0.08, "growth": 0.28, "momentum": 0.25, "quality": 0.08,
"profitability": 0.08, "sentiment": 0.12, "analyst": 0.07, "risk": 0.04,
},
"Conservative": {
"value": 0.20, "growth": 0.12, "momentum": 0.06, "quality": 0.27,
"profitability": 0.20, "sentiment": 0.04, "analyst": 0.08, "risk": 0.03,
},
"Recent Uptrends": {
"value": 0.05, "growth": 0.12, "momentum": 0.40, "quality": 0.05,
"profitability": 0.05, "sentiment": 0.18, "analyst": 0.10, "risk": 0.05,
},
"Buy and Hold Long Term": {
"value": 0.30, "growth": 0.19, "momentum": 0.05, "quality": 0.20,
"profitability": 0.18, "sentiment": 0.03, "analyst": 0.05, "risk": 0.00,
},
"Low Risk, High Dividend Yield": {
"value": 0.22, "growth": 0.08, "momentum": 0.05, "quality": 0.22,
"profitability": 0.30, "sentiment": 0.05, "analyst": 0.05, "risk": 0.03,
},
"High Risk High Reward": {
"value": 0.10, "growth": 0.00, "momentum": 0.00, "quality": 0.15,
"profitability": 0.00, "sentiment": 0.05, "analyst": 0.30, "risk": 0.05,
"reverse_momentum": 0.35,
},
}
HIST_BATCH = 100 # tickers per yf.download() batch
HIST_PAUSE = 3.0 # seconds between history batches
INFO_BATCH = 20 # tickers per info-fetch batch
INFO_PAUSE = 4.0 # seconds between info batches
BATCH_SIZE = INFO_BATCH
BATCH_PAUSE = INFO_PAUSE
# ---------------------------------------------------------------------------
# Ticker Fetching
# ---------------------------------------------------------------------------
# Valid common-stock ticker: 1-5 uppercase letters, optionally followed by
# a dash and 1-2 letters for share classes (e.g. BRK-A, BRK-B)
_TICKER_RE = re.compile(r'^[A-Z]{1,5}(-[A-Z]{1,2})?$')
def _clean_ticker(symbol: str) -> str | None:
"""Normalize a raw symbol and return it if it looks like a common stock."""
t = symbol.strip().upper().replace(".", "-")
return t if _TICKER_RE.match(t) else None
_NASDAQ100_FALLBACK = [
"AAPL", "MSFT", "NVDA", "AMZN", "META", "GOOGL", "GOOG", "TSLA", "AVGO", "COST",
"NFLX", "TMUS", "ASML", "AMD", "PEP", "LIN", "CSCO", "ADBE", "QCOM", "INTU",
"TXN", "AMAT", "AMGN", "CMCSA", "ISRG", "MU", "LRCX", "MRVL", "KLAC", "REGN",
"PANW", "CRWD", "SNPS", "CDNS", "ABNB", "ORLY", "MNST", "NXPI", "FTNT", "MAR",
"AEP", "PYPL", "MCHP", "PAYX", "ADSK", "CTAS", "IDXX", "ROST", "KDP", "FAST",
"ODFL", "DXCM", "VRSK", "CPRT", "BKR", "GEHC", "ON", "EXC", "LULU", "ZS",
"TEAM", "PCAR", "WDAY", "CTSH", "BIIB", "GILD", "CEG", "TTD", "MRNA", "EA",
"EBAY", "ILMN", "INTC", "SMCI", "ARM", "FANG", "WBD", "DLTR", "SIRI", "GFS",
"HON", "MSTR", "DASH", "RBLX", "COIN", "DDOG", "ZM", "OKTA", "SNOW", "NET",
"HUBS", "BILL", "CFLT", "MDB", "GTLB", "U", "APP", "PINS", "RIVN", "UBER",
]
def get_nasdaq100_tickers() -> list[str]:
"""Live NASDAQ-100 from Wikipedia; falls back to a hardcoded list."""
tickers = _fetch_wikipedia_index(
"https://en.wikipedia.org/wiki/Nasdaq-100", "NASDAQ 100")
if tickers:
return tickers
print(" [WARN] Using hardcoded NASDAQ-100 fallback list")
tickers = list(dict.fromkeys(_NASDAQ100_FALLBACK))
print(f" NASDAQ 100 : {len(tickers):>5} tickers (fallback)")
return tickers
_NASDAQ_HEADERS = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
"Accept": "application/json, text/plain, */*",
"Referer": "https://www.nasdaq.com/",
}
def _fetch_exchange(exchange: str) -> list[str]:
"""
Pull every listed stock for one exchange from the NASDAQ screener API.
Covers NYSE (~2 800 tickers), NASDAQ (~3 500), and AMEX (~300).
Together these three exchanges list every component of every major US index:
S&P 500/400/600, Russell 1000/2000/3000, DJIA, NYSE Composite, etc.
"""
url = (
"https://api.nasdaq.com/api/screener/stocks"
f"?tableonly=true&limit=10000&exchange={exchange}&download=true"
)
try:
resp = requests.get(url, headers=_NASDAQ_HEADERS, timeout=30)
resp.raise_for_status()
rows = resp.json()["data"]["table"]["rows"]
tickers = [_clean_ticker(r.get("symbol", "")) for r in rows]
tickers = [t for t in tickers if t]
print(f" {exchange:<8}: {len(tickers):>5} tickers")
return tickers
except Exception as e:
print(f" [WARN] Could not fetch {exchange} tickers: {e}")
return []
def _fetch_nasdaqtrader_listings(exchange_filter: str | None = None) -> list[str]:
"""
Download the official NASDAQ Trader exchange listing flat files.
These are publicly accessible static files updated daily — no API key,
no bot detection, no rate limiting.
exchange_filter: None = all exchanges, or "NASDAQ", "NYSE", "AMEX"
nasdaqlisted.txt — every NASDAQ-listed stock
cols: Symbol | Name | MarketCat | TestIssue | FinancialStatus | LotSize | ETF | NextShares
otherlisted.txt — every NYSE / AMEX / Arca / BATS / IEX listed stock
cols: ACTSymbol | Name | Exchange | CQSSymbol | ETF | LotSize | TestIssue | NASDAQSymbol
Exchange codes: N=NYSE A=AMEX/NYSE-American P=NYSE-Arca Z=BATS V=IEX
"""
from io import StringIO
base = "https://www.nasdaqtrader.com/dynamic/SymDir/"
tickers = []
# ── nasdaqlisted.txt ──────────────────────────────────────────────────
if exchange_filter in (None, "NASDAQ"):
try:
resp = requests.get(base + "nasdaqlisted.txt",
headers=_HTTP_HEADERS, timeout=20)
resp.raise_for_status()
count = 0
for line in resp.text.splitlines()[1:]:
if line.startswith("File Creation"):
break
parts = line.split("|")
if len(parts) < 7:
continue
test_issue = parts[3].strip() # col 3 = TestIssue
is_etf = parts[6].strip() # col 6 = ETF
if test_issue == "Y" or is_etf == "Y":
continue
t = _clean_ticker(parts[0].strip())
if t:
tickers.append(t)
count += 1
print(f" NASDAQ listed : {count:>5} tickers")
except Exception as e:
print(f" [WARN] nasdaqlisted.txt failed: {e}")
# ── otherlisted.txt ───────────────────────────────────────────────────
if exchange_filter in (None, "NYSE", "AMEX"):
# Map friendly name → exchange code(s) in the file
_code_map = {
"NYSE": {"N"},
"AMEX": {"A"},
None: {"N", "A", "P", "Z", "V"},
}
allowed_codes = _code_map.get(exchange_filter, {"N", "A", "P", "Z", "V"})
try:
resp = requests.get(base + "otherlisted.txt",
headers=_HTTP_HEADERS, timeout=20)
resp.raise_for_status()
count = 0
for line in resp.text.splitlines()[1:]:
if line.startswith("File Creation"):
break
parts = line.split("|")
if len(parts) < 7:
continue
exchange = parts[2].strip() # col 2 = Exchange code
is_etf = parts[4].strip() # col 4 = ETF
test_issue = parts[6].strip() # col 6 = TestIssue
if test_issue == "Y" or is_etf == "Y":
continue
if exchange not in allowed_codes:
continue
t = _clean_ticker(parts[0].strip())
if t:
tickers.append(t)
count += 1
lbl = exchange_filter or "other exchanges"
print(f" {lbl:<15}: {count:>5} tickers")
except Exception as e:
print(f" [WARN] otherlisted.txt failed: {e}")
return tickers
def _fetch_otc_tickers() -> list[str]:
"""
Fetch OTC-traded US company tickers from SEC EDGAR's exchange-tagged file.
company_tickers_exchange.json tags every SEC-registered company with its
exchange (Nasdaq, NYSE, OTC, etc.) — no bot detection, same source we
already trust for the fallback ticker list.
Filters to entries where exchange == "OTC", giving OTCQX / OTCQB /
Pink Sheet stocks not covered by the NASDAQ Trader exchange files.
"""
try:
resp = requests.get(
"https://www.sec.gov/files/company_tickers_exchange.json",
headers={"User-Agent": "stockscreener contact@example.com"},
timeout=20,
)
resp.raise_for_status()
payload = resp.json()
# Format: {"fields": ["cik","name","ticker","exchange"], "data": [[...], ...]}
fields = payload.get("fields", [])
rows = payload.get("data", [])
try:
exch_idx = fields.index("exchange")
ticker_idx = fields.index("ticker")
except ValueError:
print(" [WARN] SEC EDGAR exchange file format changed — no OTC tickers")
return []
tickers = [
_clean_ticker(str(row[ticker_idx]))
for row in rows
if str(row[exch_idx]).upper() == "OTC"
]
tickers = [t for t in tickers if t]
print(f" OTC (SEC EDGAR): {len(tickers):>5} tickers")
return tickers
except Exception as e:
print(f" [WARN] SEC EDGAR OTC fetch failed: {e}")
return []
def _fetch_sec_all_tickers() -> list[str]:
"""
Pull every US-listed company ticker from the SEC EDGAR company tickers file.
This is a publicly accessible, bot-friendly endpoint (~10 000+ tickers).
Used as the fallback when NASDAQ Trader files are unavailable.
"""
try:
resp = requests.get(
"https://www.sec.gov/files/company_tickers.json",
headers={"User-Agent": "stockscreener contact@example.com"},
timeout=20,
)
resp.raise_for_status()
data = resp.json()
tickers = [_clean_ticker(v.get("ticker", "")) for v in data.values()]
tickers = [t for t in tickers if t]
print(f" SEC EDGAR : {len(tickers):>5} tickers")
return tickers
except Exception as e:
print(f" [WARN] SEC EDGAR ticker fetch failed: {e}")
return []
_HTTP_HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
def _fetch_sp500() -> list[str]:
# Primary: Wikipedia — reflects additions/removals as soon as the page updates
tickers = _fetch_wikipedia_index(
"https://en.wikipedia.org/wiki/List_of_S%26P_500_companies", "S&P 500")
if tickers:
return tickers
# Fallback: community-maintained GitHub CSV
try:
from io import StringIO
url = ("https://raw.githubusercontent.com/datasets/"
"s-and-p-500-companies/main/data/constituents.csv")
resp = requests.get(url, headers=_HTTP_HEADERS, timeout=20)
resp.raise_for_status()
tickers = [_clean_ticker(t)
for t in pd.read_csv(StringIO(resp.text))["Symbol"]]
tickers = [t for t in tickers if t]
print(f" S&P 500 : {len(tickers):>5} tickers (GitHub fallback)")
return tickers
except Exception as e:
print(f" [WARN] S&P 500 GitHub fallback failed: {e}")
return []
def _fetch_wikipedia_index(url: str, label: str, min_count: int = 10) -> list[str]:
"""
Pull index constituents from a Wikipedia list-of-companies page.
Strips footnote markers (e.g. 'AAPL[a]') and skips tables that are
too small to be the real index table (avoids false positives).
"""
try:
from io import StringIO
resp = requests.get(url, headers=_HTTP_HEADERS, timeout=20)
resp.raise_for_status()
tables = pd.read_html(StringIO(resp.text))
for df in tables:
for col in ("Symbol", "Ticker", "Ticker symbol"):
if col in df.columns:
# Strip footnote brackets like [a], [1], [note 1]
cleaned = (df[col].astype(str)
.str.replace(r'\[.*?\]', '', regex=True)
.str.strip())
tickers = [_clean_ticker(t) for t in cleaned]
tickers = [t for t in tickers if t]
if len(tickers) < min_count:
continue # skip unrelated small tables
print(f" {label:<12}: {len(tickers):>5} tickers")
return tickers
except Exception as e:
print(f" [WARN] {label} failed: {e}")
return []
# iShares Russell ETF holdings CSVs — publicly downloadable
_RUSSELL_ETFS = {
"Russell 1000": ("239707", "ishares-russell-1000-etf", "IWB"),
"Russell 2000": ("239659", "ishares-russell-2000-etf", "IWM"),
"Russell 3000": ("239714", "ishares-russell-3000-etf", "IWV"),
}
def _fetch_russell(label: str) -> list[str]:
product_id, slug, etf = _RUSSELL_ETFS[label]
url = (
f"https://www.ishares.com/us/products/{product_id}/{slug}"
f"/1467271812596.ajax?fileType=csv&fileName={etf}_holdings&dataType=fund"
)
try:
from io import StringIO
resp = requests.get(url, headers=_HTTP_HEADERS, timeout=40)
resp.raise_for_status()
lines = resp.text.splitlines()
# iShares CSVs have a few metadata lines before the real header
header_idx = next(
(i for i, ln in enumerate(lines)
if ln.startswith("Ticker,") or ",Ticker," in ln or ln.startswith("Name,")),
None,
)
if header_idx is None:
raise ValueError("Header row not found in iShares CSV")
df = pd.read_csv(StringIO("\n".join(lines[header_idx:])))
col = next((c for c in df.columns if c.strip() == "Ticker"), None)
if not col:
raise ValueError("'Ticker' column missing")
tickers = [_clean_ticker(str(t)) for t in df[col]]
tickers = [t for t in tickers if t and t.upper() not in ("-", "NAN", "CASH")]
print(f" {label:<12}: {len(tickers):>5} tickers")
return tickers
except Exception as e:
print(f" [WARN] {label} iShares CSV failed: {e}")
return []
_DJIA_FALLBACK = [
"AAPL", "AMGN", "AMZN", "AXP", "BA", "CAT", "CRM", "CSCO", "CVX", "DIS",
"DOW", "GS", "HD", "HON", "IBM", "JNJ", "JPM", "KO", "MCD", "MMM",
"MRK", "MSFT", "NKE", "PG", "SHW", "TRV", "UNH", "V", "VZ", "WMT",
]
def get_djia_tickers() -> list[str]:
"""Live DJIA components from Wikipedia; falls back to a hardcoded list."""
tickers = _fetch_wikipedia_index(
"https://en.wikipedia.org/wiki/Dow_Jones_Industrial_Average", "DJIA")
if tickers:
return tickers
print(" [WARN] Using hardcoded DJIA fallback list")
tickers = list(dict.fromkeys(_DJIA_FALLBACK))
print(f" DJIA : {len(tickers):>5} tickers (fallback)")
return tickers
# Human-readable index names exposed to the GUI
INDEX_CHOICES = [
"All",
"S&P 500", "S&P 400", "S&P 600", "S&P 1500",
"Russell 1000", "Russell 2000", "Russell 3000",
"NASDAQ 100", "NASDAQ", "NYSE", "AMEX",
"DJIA",
]
def collect_tickers(index: str = "All") -> list[str]:
"""
Fetch the constituent tickers for the requested index.
"All" — every common stock on all US exchanges via NASDAQ Trader files
"All + OTC" — exchange-listed + OTC/Pink Sheet stocks
"S&P 500/400/600/1500" — via Wikipedia / GitHub CSV
"Russell 1000/2000/3000" — via iShares ETF holdings CSVs
"NASDAQ 100" — live Wikipedia list with hardcoded fallback
"NASDAQ/NYSE/AMEX" — per-exchange listings via NASDAQ Trader files
"DJIA" — 30 components from Wikipedia with hardcoded fallback
Fallback chain for "All" and single-exchange lookups:
NASDAQ Trader files → NASDAQ screener API → SEC EDGAR → Russell 3000 → S&P 500
"""
print(f"Fetching tickers for index: {index} ...")
sp400_url = "https://en.wikipedia.org/wiki/List_of_S%26P_400_companies"
sp600_url = "https://en.wikipedia.org/wiki/List_of_S%26P_600_companies"
def _all_exchange_tickers() -> list[str]:
"""All exchange-listed US common stocks with full fallback chain."""
# 1. NASDAQ Trader official listing files (most reliable — static, no bot detection)
t = _fetch_nasdaqtrader_listings(exchange_filter=None)
if t:
return t
# 2. NASDAQ screener API (often blocked but try anyway)
print(" [WARN] NASDAQ Trader files failed — trying NASDAQ screener API")
t = []
for exch in ("NYSE", "NASDAQ", "AMEX"):
t.extend(_fetch_exchange(exch))
if t:
return t
# 3. SEC EDGAR company registry (~10 000+ tickers)
print(" [WARN] NASDAQ API failed — trying SEC EDGAR")
t = _fetch_sec_all_tickers()
if t:
return t
# 4. Russell 3000
print(" [WARN] SEC EDGAR failed — trying Russell 3000")
t = _fetch_russell("Russell 3000")
if t:
return t
# 5. S&P 1500
print(" [WARN] Russell 3000 failed — trying S&P 1500")
t = (_fetch_sp500()
+ _fetch_wikipedia_index(sp400_url, "S&P 400")
+ _fetch_wikipedia_index(sp600_url, "S&P 600"))
if t:
return t
# 6. S&P 500 last resort
print(" [WARN] All broad sources failed — falling back to S&P 500")
return _fetch_sp500()
if index == "All":
tickers = _all_exchange_tickers()
elif index == "S&P 500":
tickers = _fetch_sp500()
elif index == "S&P 400":
tickers = _fetch_wikipedia_index(sp400_url, "S&P 400")
elif index == "S&P 600":
tickers = _fetch_wikipedia_index(sp600_url, "S&P 600")
elif index == "S&P 1500":
tickers = (_fetch_sp500()
+ _fetch_wikipedia_index(sp400_url, "S&P 400")
+ _fetch_wikipedia_index(sp600_url, "S&P 600"))
elif index in ("Russell 1000", "Russell 2000", "Russell 3000"):
tickers = _fetch_russell(index)
if not tickers:
print(f" [WARN] iShares CSV failed — using full exchange listing for {index}")
tickers = _all_exchange_tickers()
elif index == "NASDAQ 100":
tickers = get_nasdaq100_tickers()
elif index == "NASDAQ":
tickers = _fetch_nasdaqtrader_listings(exchange_filter="NASDAQ")
if not tickers:
tickers = _fetch_exchange("NASDAQ")
elif index == "NYSE":
tickers = _fetch_nasdaqtrader_listings(exchange_filter="NYSE")
if not tickers:
tickers = _fetch_exchange("NYSE")
elif index == "AMEX":
tickers = _fetch_nasdaqtrader_listings(exchange_filter="AMEX")
if not tickers:
tickers = _fetch_exchange("AMEX")
elif index == "DJIA":
tickers = get_djia_tickers()
else:
print(f" [WARN] Unknown index '{index}' — defaulting to All")
tickers = _all_exchange_tickers()
combined = list(dict.fromkeys(tickers)) # deduplicate, preserve order
print(f" Total : {len(combined):>5} unique tickers\n")
return combined
# ---------------------------------------------------------------------------
# News Sentiment (multi-source VADER)
# ---------------------------------------------------------------------------
# Sources pulled in order:
# 1. Yahoo Finance — via yfinance Ticker.news (built-in, no key)
# 2. Google News RSS — public RSS feed, no API key required
# ---------------------------------------------------------------------------
def _extract_yfinance_news(ticker_obj) -> list[tuple[str, float, str, str]]:
"""Return (title, timestamp, url, summary) tuples from yfinance Ticker.news."""
results = []
try:
news = ticker_obj.news or []
now = time.time()
for item in news[:30]:
if not isinstance(item, dict):
continue
title = ts = url = summary = None
content = item.get("content", {})
if isinstance(content, dict) and content.get("title"):
title = content["title"]
pub = content.get("pubDate", "")
try:
ts = datetime.fromisoformat(
pub.replace("Z", "+00:00")).timestamp()
except Exception:
ts = now
# Try several URL fields used across yfinance versions
canon = content.get("canonicalUrl") or {}
url = (canon.get("url") if isinstance(canon, dict) else None) \
or content.get("url") or content.get("clickThroughUrl", {}).get("url") \
or item.get("link")
summary = content.get("summary") or content.get("description") or ""
elif item.get("title"):
title = item["title"]
ts = item.get("providerPublishTime", now)
url = item.get("link") or item.get("url") or ""
summary = item.get("summary") or item.get("description") or ""
if title:
clean = re.sub(r'\s+', ' ',
html.unescape(re.sub(r'<[^>]+>', ' ', summary or ""))).strip()
results.append((title, float(ts or now), url or "", clean))
except Exception:
pass
return results
def _extract_google_news_rss(ticker: str, company_name: str = "") -> list[tuple[str, float, str, str]]:
"""
Fetch headlines from Google News RSS for the given ticker.
Returns (title, timestamp, url, summary) tuples.
No API key required.
"""
import re as _re
import xml.etree.ElementTree as ET
import urllib.parse
results = []
try:
query = f"{ticker} stock"
if company_name:
query += f" {company_name}"
encoded = urllib.parse.quote(query)
rss_url = (
f"https://news.google.com/rss/search"
f"?q={encoded}&hl=en-US&gl=US&ceid=US:en"
)
headers = {"User-Agent": "Mozilla/5.0 (compatible; StockScreener/1.0)"}
resp = requests.get(rss_url, headers=headers, timeout=8)
resp.raise_for_status()
root = ET.fromstring(resp.content)
now = time.time()
for item in root.findall(".//item")[:30]:
title_el = item.find("title")
pub_el = item.find("pubDate")
desc_el = item.find("description")
if title_el is None or not title_el.text:
continue
title = title_el.text.strip()
ts = now
if pub_el is not None and pub_el.text:
try:
from email.utils import parsedate_to_datetime
ts = parsedate_to_datetime(pub_el.text).timestamp()
except Exception:
pass
# guid is more reliably a plain URL in Google News RSS than <link>
guid_el = item.find("guid")
link = (guid_el.text.strip() if guid_el is not None and guid_el.text else "") \
or (item.findtext("link") or "")
# Strip HTML tags and unescape entities from description snippet
raw_desc = desc_el.text if desc_el is not None and desc_el.text else ""
summary = _re.sub(r'\s+', ' ',
html.unescape(_re.sub(r"<[^>]+>", " ", raw_desc))).strip()
results.append((title, ts, link, summary))
except Exception:
pass
return results
def get_news_headlines(ticker_obj, ticker: str = "", company_name: str = "") \
-> tuple:
"""
Aggregate headlines from Yahoo Finance and Google News RSS and compute
recency-weighted VADER sentiment.
Returns (aggregate_score, top_headlines) where:
- aggregate_score is float in [-1.0, +1.0] or None
- top_headlines is a list of (title, vader_score, age_days) tuples,
sorted by impact (abs(score) × recency weight), up to 8 entries.
"""
if _VADER is None:
return None, []
all_items: list[tuple[str, float, str, str]] = []
all_items.extend(_extract_yfinance_news(ticker_obj))
if ticker:
all_items.extend(_extract_google_news_rss(ticker, company_name))
if not all_items:
return None, []
# Deduplicate by normalised title
seen: set[str] = set()
unique: list[tuple[str, float, str, str]] = []
for title, ts, url, summary in all_items:
key = title.lower().strip()[:80]
if key not in seen:
seen.add(key)
unique.append((title, ts, url, summary))
now = time.time()
total_w = total_s = 0.0
# (title, score, age_days, weight, url, summary)
scored: list[tuple[str, float, float, float, str, str]] = []
for title, ts, url, summary in unique:
age_days = max(0.0, (now - ts) / 86400)
if age_days <= 7:
weight = 1.0
else:
weight = max(0.05, 1.0 - (age_days - 7) / 23.0)
score = _VADER.polarity_scores(title)["compound"]
total_s += score * weight
total_w += weight
scored.append((title, score, age_days, weight, url, summary))
aggregate = total_s / total_w if total_w else None
# Top headlines by impact: abs(score) × recency weight
scored.sort(key=lambda x: abs(x[1]) * x[3], reverse=True)
top = [(title, vader_score, age_days, url, summary)
for title, vader_score, age_days, _, url, summary in scored[:8]]
return aggregate, top
def _news_sentiment(ticker_obj, ticker: str = "", company_name: str = "") -> float | None:
"""Backward-compatible wrapper — returns only the aggregate score."""
score, _ = get_news_headlines(ticker_obj, ticker=ticker, company_name=company_name)
return score
# ---------------------------------------------------------------------------
# EDGAR Bulk Fundamentals
# ---------------------------------------------------------------------------
# Fetches financial statement data for all US public companies from the
# SEC EDGAR XBRL frames API. One HTTP call per financial concept returns
# values for ALL filers simultaneously — ~50 calls cover every metric
# needed for the scoring model in roughly 2 minutes.
#
# Coverage: ~8590% of exchange-listed US stocks. Tickers with no EDGAR
# coverage still appear in results and score on price/momentum/analyst
# data only; their fundamental fields default to NaN.
# ---------------------------------------------------------------------------
_EDGAR_FRAMES = "https://data.sec.gov/api/xbrl/frames"
_EDGAR_TICKERS = "https://www.sec.gov/files/company_tickers.json"
_EDGAR_UA = {"User-Agent": "StockScreener contact@investmenttool.com"}
_FRAME_CACHE: dict = {} # in-memory cache so the same frame is never fetched twice
def _get_recent_quarters(n: int = 9) -> list[tuple[int, int]]:
"""
Return the last n (year, quarter) tuples, most-recent first, using a
45-day filing lag so we only request periods most companies have filed.
"""
from datetime import date, timedelta as _td
ref = date.today() - _td(days=45)
y, q = ref.year, (ref.month - 1) // 3 + 1
out = []
for _ in range(n):
out.append((y, q))
q -= 1
if q == 0:
q, y = 4, y - 1
return out
def _fetch_one_frame(concept: str, unit: str, period: str) -> dict[int, float]:
"""
Fetch all companies' values for one EDGAR XBRL concept/period.
Returns {cik (int): value (float)}. Results are cached in _FRAME_CACHE
so the same concept/period is never downloaded twice in one run.
"""
key = (concept, unit, period)
if key in _FRAME_CACHE:
return _FRAME_CACHE[key]
url = f"{_EDGAR_FRAMES}/us-gaap/{concept}/{unit}/{period}.json"
try:
resp = requests.get(url, headers=_EDGAR_UA, timeout=20)
if resp.status_code == 404:
_FRAME_CACHE[key] = {}
return {}
resp.raise_for_status()
result: dict[int, float] = {}
for row in resp.json().get("data", []):
# The EDGAR frames API returns each row as a dict with keys:
# accn, cik, entityName, loc, start, end, val
# (Not as a list — handle both formats for safety.)
try:
if isinstance(row, dict):
cik = int(row["cik"])
val = float(row["val"])
else:
# Legacy/alternative format: [accn, cik, entityName, loc, end, val]
if len(row) < 6:
continue
cik = int(row[1])
val = float(row[5])
result[cik] = val # last entry wins (restatements)
except (KeyError, TypeError, ValueError):
continue
_FRAME_CACHE[key] = result
time.sleep(0.15) # stay comfortably under SEC's 10 req/sec limit
return result
except Exception:
# Do not cache transient failures (network errors, timeouts, etc.)
# Only 404s (already handled above) should produce a permanent empty cache entry.
return {}
def _sum_frames(concepts: list[str], unit: str,
periods: list[str]) -> dict[int, float]:
"""
Sum values across periods (TTM) trying each concept in order.
The primary concept is tried first; gaps are filled by fallback concepts.
"""
ttm: dict[int, dict] = {} # {cik: {period: value}}
for concept in concepts:
for period in periods:
for cik, val in _fetch_one_frame(concept, unit, period).items():
ttm.setdefault(cik, {})
if period not in ttm[cik]:
ttm[cik][period] = val
return {cik: sum(pv.values()) for cik, pv in ttm.items() if len(pv) >= 4}
def _best_frame(concepts: list[str], unit: str,
periods: list[str]) -> dict[int, float]:
"""
Return the most-recent available value per CIK (for balance-sheet items).
Iterates periods most-recent first; first value found for each CIK wins.
"""
best: dict[int, float] = {}
for concept in concepts:
for period in periods:
for cik, val in _fetch_one_frame(concept, unit, period).items():
if cik not in best:
best[cik] = val
return best
def _build_cik_maps() -> tuple[dict[int, str], dict[int, str], dict[str, int]]:
"""
Fetch SEC company_tickers.json and return three dicts:
cik_to_ticker — {cik_int: "AAPL"} (one representative ticker per CIK)
cik_to_name — {cik_int: "Apple Inc."}
ticker_to_cik — {ticker: cik_int} (ALL tickers, handles dual-class shares)
ticker_to_cik covers every row in the SEC file so dual-class stocks
(GOOGL/GOOG, BRK-A/BRK-B, etc.) each map to the same CIK and therefore
both receive EDGAR fundamental data.
"""
try:
resp = requests.get(_EDGAR_TICKERS, headers=_EDGAR_UA, timeout=20)
resp.raise_for_status()
data = resp.json()
cik_to_ticker: dict[int, str] = {}
cik_to_name: dict[int, str] = {}
ticker_to_cik: dict[str, int] = {}
for v in data.values():
cik = int(v["cik_str"])
ticker = v["ticker"].upper()
title = v.get("title", "")
ticker_to_cik[ticker] = cik # every ticker gets its own entry
if cik not in cik_to_ticker:
cik_to_ticker[cik] = ticker # first seen wins for repr
cik_to_name[cik] = title
return cik_to_ticker, cik_to_name, ticker_to_cik
except Exception as e:
print(f" [WARN] CIK map unavailable: {e}")
return {}, {}, {}
def _fetch_edgar_bulk(close_map: dict[str, pd.Series]) -> dict[str, dict]:
"""
Fetch fundamental data for all tickers from the EDGAR XBRL frames API.
Returns {ticker: {field: value}} with keys matching the rest of the
scoring pipeline. Takes ~2 minutes for the full US universe.
Fields returned (NaN-safe — missing values simply absent from the dict):
market_cap, pe_trailing, pb_ratio, ev_ebitda, eps_trailing,
revenue, revenue_growth, earnings_growth, roe, roa,
debt_to_equity, total_debt, total_cash, book_value, current_ratio,
profit_margin, operating_margin, fcf_yield, name
"""
print("Phase 2A: Fetching EDGAR bulk fundamentals ...")
cik_to_ticker, cik_to_name, ticker_to_cik = _build_cik_maps()
if not ticker_to_cik:
print(" [WARN] CIK map unavailable — EDGAR bulk skipped")
return {}
quarters = _get_recent_quarters(9)
ttm_periods = [f"CY{y}Q{q}" for y, q in quarters[:4]]
prev_periods= [f"CY{y}Q{q}" for y, q in quarters[4:8]]
bs_periods = [f"CY{y}Q{q}I" for y, q in quarters[:3]]
print(f" TTM : {ttm_periods}")
print(f" BS : {bs_periods[0]} (+{len(bs_periods)-1} fallback periods)")
# ── Balance sheet (most recent available quarter) ──────────────────────
assets = _best_frame(["Assets"], "USD", bs_periods)
equity = _best_frame(["StockholdersEquity",
"StockholdersEquityIncludingPortionAttributable"
"ToNoncontrollingInterest",
"CommonStockholdersEquity"], "USD", bs_periods)
cur_assets = _best_frame(["AssetsCurrent"], "USD", bs_periods)
cur_liab = _best_frame(["LiabilitiesCurrent"], "USD", bs_periods)
cash = _best_frame(["CashAndCashEquivalentsAtCarryingValue",
"CashCashEquivalentsAndShortTermInvestments"], "USD", bs_periods)
lt_debt = _best_frame(["LongTermDebt", "LongTermDebtNoncurrent",
"LongTermDebtAndCapitalLeaseObligations",
"LongTermDebtAndFinanceLeaseLiabilities",
"FinanceLeaseLiabilityNoncurrent"], "USD", bs_periods)
shares = _best_frame(["CommonStockSharesOutstanding"], "shares", bs_periods)
# ── Income statement TTM ────────────────────────────────────────────────
revenue = _sum_frames(["Revenues",
"RevenueFromContractWithCustomer"
"ExcludingAssessedTax",
"SalesRevenueNet"], "USD", ttm_periods)
net_income = _sum_frames(["NetIncomeLoss"], "USD", ttm_periods)
op_income = _sum_frames(["OperatingIncomeLoss"], "USD", ttm_periods)
dna = _sum_frames(["DepreciationDepletionAndAmortization",
"DepreciationAndAmortization"], "USD", ttm_periods)
# ── Cash flow TTM ───────────────────────────────────────────────────────
op_cf = _sum_frames(["NetCashProvidedByUsedInOperatingActivities"], "USD", ttm_periods)
capex = _sum_frames(["PaymentsToAcquirePropertyPlantAndEquipment"], "USD", ttm_periods)
# ── Prior-year TTM for growth rates ────────────────────────────────────
rev_prev = _sum_frames(["Revenues",
"RevenueFromContractWithCustomer"
"ExcludingAssessedTax",
"SalesRevenueNet"], "USD", prev_periods)
ni_prev = _sum_frames(["NetIncomeLoss"], "USD", prev_periods)
print(f" Revenue data : {len(revenue)} companies")
# ── Assemble per-ticker records ─────────────────────────────────────────
result: dict[str, dict] = {}
for ticker, close in close_map.items():
cik = ticker_to_cik.get(ticker)
if cik is None:
continue # no SEC filing found for this ticker
price = float(close.iloc[-1])
rev = revenue.get(cik); ni = net_income.get(cik)
op_i = op_income.get(cik); d_a = dna.get(cik)
ocf = op_cf.get(cik); cx = capex.get(cik)
tot_a = assets.get(cik); eq = equity.get(cik)
ca = cur_assets.get(cik); cl = cur_liab.get(cik)
csh = cash.get(cik); ltd = lt_debt.get(cik)
sh = shares.get(cik)
r_p = rev_prev.get(cik); n_p = ni_prev.get(cik)
mkt_cap = (price * sh) if sh and sh > 0 else None
eps_trail = (ni / sh) if ni is not None and sh and sh > 0 else None
pe_trail = (price / eps_trail) if eps_trail and eps_trail > 0 else None
pb = (price / (eq / sh)) if eq and sh and sh > 0 and eq > 0 else None
ebitda = ((op_i + d_a) if op_i is not None and d_a is not None else op_i)
ev = ((mkt_cap + (ltd or 0) - (csh or 0)) if mkt_cap is not None else None)
ev_ebitda = (ev / ebitda) if ev and ebitda and ebitda > 0 else None
ev_rev = (ev / rev) if ev and rev and rev > 0 else None
roe = (ni / eq) if ni is not None and eq and eq > 0 else None
roa = (ni / tot_a) if ni is not None and tot_a and tot_a > 0 else None
de = (ltd / eq) if ltd is not None and eq and eq > 0 else None
curr_r = (ca / cl) if ca and cl and cl > 0 else None
pm = (ni / rev) if ni is not None and rev and rev > 0 else None
om = (op_i / rev) if op_i is not None and rev and rev > 0 else None
fcf = ((ocf - cx) if ocf is not None and cx is not None else ocf)
fcf_yield = (fcf / mkt_cap) if fcf is not None and mkt_cap and mkt_cap > 0 else None
rev_g = ((rev - r_p) / abs(r_p) if rev is not None and r_p and r_p != 0 else None)
ni_g = ((ni - n_p) / abs(n_p) if ni is not None and n_p and n_p != 0 else None)
result[ticker] = {
"name": cik_to_name.get(cik, ticker),
"sector": "N/A",
"industry": "N/A",
"market_cap": mkt_cap,
"shares_outstanding": sh,
"pe_trailing": pe_trail,
"pb_ratio": pb,
"ev_ebitda": ev_ebitda,
"ev_revenue": ev_rev,
"eps_trailing": eps_trail,
"revenue": rev,
"revenue_growth": rev_g,
"earnings_growth": ni_g,
"roe": roe,
"roa": roa,
"debt_to_equity": de,
"total_debt": ltd,
"total_cash": csh,
"book_value": (eq / sh) if eq and sh and sh > 0 else None,
"current_ratio": curr_r,
"profit_margin": pm,
"operating_margin": om,
"fcf_yield": fcf_yield,
}
print(f" EDGAR matched : {len(result)} tickers\n")
return result
# ---------------------------------------------------------------------------
# Analyst + Fundamentals Cache (Verimund API)
# ---------------------------------------------------------------------------
try:
from license_check import _API_URL as _API_URL, get_api_headers as _get_api_headers
except ImportError:
_API_URL = None
_get_api_headers = lambda: {}
def _load_analyst_cache() -> dict[str, dict]:
"""
Load analyst consensus and news sentiment from the Verimund API cache.
Returns {ticker: {field: value}} or {} on any failure.
Warns (but still proceeds) if cache data is more than 48 hours old.
"""
if not _API_URL:
print(" [WARN] API config unavailable — analyst cache skipped")
return {}
try:
resp = requests.get(
f"{_API_URL}/cache/analyst",
headers=_get_api_headers(),
timeout=30,
)
resp.raise_for_status()
rows = resp.json().get("data", [])
if not rows:
print(" [WARN] Analyst cache is empty — analyst scores will be neutral")
return {}
valid_ts = [r["updated_at"] for r in rows if r.get("updated_at")]
if valid_ts:
age_h = (
datetime.now(timezone.utc)
- datetime.fromisoformat(max(valid_ts).replace("Z", "+00:00"))
).total_seconds() / 3600
if age_h > 48:
print(f" [WARN] Analyst cache is {age_h:.0f}h old "
"(refresh runs tonight)")
else:
print(f" Analyst cache : {len(rows)} tickers "
f"(updated {age_h:.1f}h ago)")
return {r["ticker"]: r for r in rows}
except Exception as e:
print(f" [WARN] Could not load analyst cache: {e}")
return {}
def _load_fundamentals_cache() -> dict[str, dict]:
"""
Load pre-computed fundamental metrics from the Verimund API cache.
Returns {ticker: {field: value}} or {} on any failure.
"""
if not _API_URL:
print(" [WARN] API config unavailable — fundamentals cache skipped")
return {}
try:
resp = requests.get(
f"{_API_URL}/cache/fundamentals",
headers=_get_api_headers(),
timeout=30,
)
resp.raise_for_status()
rows = resp.json().get("data", [])
if not rows:
print(" [WARN] Fundamentals cache is empty — falling back to live EDGAR")
return {}
valid_ts = [r["updated_at"] for r in rows if r.get("updated_at")]
if valid_ts:
age_h = (
datetime.now(timezone.utc)
- datetime.fromisoformat(max(valid_ts).replace("Z", "+00:00"))
).total_seconds() / 3600
if age_h > 48:
print(f" [WARN] Fundamentals cache is {age_h:.0f}h old "
"(refresh runs tonight)")
else:
print(f" Fundamentals cache: {len(rows)} tickers "
f"(updated {age_h:.1f}h ago)")
return {r["ticker"]: r for r in rows}
except Exception as e:
print(f" [WARN] Could not load fundamentals cache: {e}")
return {}
# ---------------------------------------------------------------------------
# FINRA Short Interest
# ---------------------------------------------------------------------------
def _fetch_finra_short() -> dict[str, float]:
"""
Download the FINRA RegSHO daily short-volume file.
Returns {ticker: short_volume_ratio} where ratio = short_vol / total_vol.
This is a short-volume ratio, not short % of float, but it is a reliable
intraday bearish-pressure signal and a suitable proxy for the Risk score.
Files are published each trading day at:
https://cdn.finra.org/equity/regsho/daily/
"""
from datetime import date as _date, timedelta as _td
for days_back in range(1, 6):
d = _date.today() - _td(days=days_back)
if d.weekday() >= 5: # skip Saturday / Sunday
continue
url = (
"https://cdn.finra.org/equity/regsho/daily/"
f"CNMSshvol{d.strftime('%Y%m%d')}.txt"
)
try:
resp = requests.get(url, headers=_HTTP_HEADERS, timeout=15)
if resp.status_code == 404:
continue
resp.raise_for_status()
result: dict[str, float] = {}
for line in resp.text.splitlines()[1:]:
parts = line.split("|")
if len(parts) < 4:
continue
try:
sv = float(parts[1]); tv = float(parts[3])
if tv > 0:
result[parts[0].strip()] = sv / tv
except (ValueError, IndexError):
continue
if result:
print(f" FINRA short : {len(result)} tickers ({d})")
return result
except Exception:
continue
print(" [WARN] FINRA short interest unavailable")
return {}
# ---------------------------------------------------------------------------
# Beta Calculation
# ---------------------------------------------------------------------------
def _calc_beta_map(close_map: dict[str, pd.Series],
start_date: str) -> dict[str, float]:
"""
Calculate 1-year beta vs SPY for every ticker in close_map.
Downloads SPY history once, then computes cov(stock, SPY) / var(SPY)
in pandas — pure vectorised math, takes ~1030 seconds for 6,500 tickers.
"""
try:
spy_raw = yf.download("SPY", start=start_date,
auto_adjust=True, progress=False)
if spy_raw is None or spy_raw.empty:
return {}
spy_col = next(
(c for c in ("Close", "Adj Close") if c in spy_raw.columns), None
)
if not spy_col:
return {}
spy_close = spy_raw[spy_col]
# Newer yfinance returns MultiIndex columns even for single tickers —
# squeeze a single-column DataFrame to a Series.
if isinstance(spy_close, pd.DataFrame):
spy_close = spy_close.squeeze(axis=1)
spy_ret = spy_close.dropna().pct_change().dropna()
beta_map: dict[str, float] = {}
for ticker, close in close_map.items():
try:
s_ret = close.pct_change().dropna()
common = s_ret.index.intersection(spy_ret.index)
if len(common) < 50:
continue
m = spy_ret.loc[common]
var_m = float(m.var())
if var_m > 0:
beta_map[ticker] = float(s_ret.loc[common].cov(m) / var_m)
except Exception:
continue
print(f" Beta : {len(beta_map)} tickers calculated")
return beta_map
except Exception as e:
print(f" [WARN] Beta calculation failed: {e}")
return {}
# ---------------------------------------------------------------------------
# Data Fetching
# ---------------------------------------------------------------------------
# Sentinel so the rest of the file still compiles if something references these
_yf_session = None
_yf_crumb = None
def _download_history_batch(tickers: list[str], start: str) -> dict[str, pd.Series]:
"""
Download price history for a batch of tickers via yf.download().
yfinance manages cookies and crumb internally — no custom auth needed.
Returns {ticker: Close Series} for tickers with ≥ 20 trading days.
"""
if not tickers:
return {}
result: dict[str, pd.Series] = {}
try:
# yf.download handles auth/crumb internally
raw = yf.download(
tickers if len(tickers) > 1 else tickers[0],
start=start,
auto_adjust=True,
progress=False,
)
if raw is None or raw.empty:
return result
if len(tickers) == 1:
# Single ticker — newer yfinance returns MultiIndex columns even for
# a single ticker, so raw["Close"] may be a single-column DataFrame.
# Squeeze it to a Series so downstream float(close.iloc[-1]) works.
t = tickers[0]
col = next((c for c in ("Close", "Adj Close") if c in raw.columns), None)
if col:
s = raw[col]
if isinstance(s, pd.DataFrame):
s = s.squeeze(axis=1) # DataFrame → Series
s = s.dropna()
if len(s) >= 20:
result[t] = s
else:
# Multiple tickers → MultiIndex columns
cols = raw.columns
if isinstance(cols, pd.MultiIndex):
lvl0 = cols.get_level_values(0)
if "Close" in lvl0:
close_df = raw["Close"]
for t in tickers:
if t in close_df.columns:
s = close_df[t].dropna()
if len(s) >= 20:
result[t] = s
else:
# (ticker, field) layout used by some yfinance versions
for t in tickers:
if t in lvl0:
try:
sub = raw[t]
col = next((c for c in ("Close", "Adj Close")
if c in sub.columns), None)
if col:
s = sub[col].dropna()
if len(s) >= 20:
result[t] = s
except (KeyError, TypeError):
pass
except Exception as e:
print(f"\n [WARN] yf.download batch failed: {e}")
return result
def _fetch_fundamentals(
ticker: str,
close: pd.Series,
sector_filter: tuple | None = None,
mktcap_range: tuple | None = None,
) -> dict | None:
"""
Fetch fundamental/analyst/sentiment data for one ticker via yf.Ticker.
yfinance handles all Yahoo Finance auth internally.
Price history is supplied from Phase 1 so we only need the info call.
sector_filter: (column, value) tuple — e.g. ("sector", "Technology") or
("industry", "Aerospace & Defense"). None = no filter.
mktcap_range: (min, max) in dollars, either bound may be None.
"""
# Small random delay to spread concurrent worker requests
time.sleep(random.uniform(0.0, 1.0))
try:
stock = yf.Ticker(ticker)
info = stock.info
if not info or not isinstance(info, dict) or len(info) < 3:
return None
# ── Screen-time sector filter ──────────────────────────────────────
if sector_filter is not None:
col, val = sector_filter
actual = info.get("sector") if col == "sector" else info.get("industry")
if actual != val:
return None
# ── Screen-time market-cap filter ──────────────────────────────────
if mktcap_range is not None:
try:
mkt = float(info.get("marketCap") or 0)
except (TypeError, ValueError):
mkt = 0.0
lo, hi = mktcap_range
if lo is not None and mkt < lo:
return None
if hi is not None and mkt >= hi:
return None
price_now = float(close.iloc[-1])
def pct_return(days_back: int) -> float | None:
if len(close) <= days_back:
return None
idx = max(0, len(close) - days_back)
past = float(close.iloc[idx])
return (price_now - past) / past if past > 0 else None
returns_30d = close.iloc[-30:].pct_change().dropna()
volatility = float(returns_30d.std() * (252 ** 0.5)) if len(returns_30d) >= 5 else None
try:
rec_mean = float(info.get("recommendationMean")) if info.get("recommendationMean") is not None else None
except (TypeError, ValueError):
rec_mean = None
analyst_norm = (5.0 - rec_mean) / 4.0 if rec_mean is not None and 1 <= rec_mean <= 5 else None
try:
target_price = float(info.get("targetMeanPrice")) if info.get("targetMeanPrice") is not None else None
except (TypeError, ValueError):
target_price = None
analyst_upside = (target_price / price_now - 1.0
if target_price is not None and price_now > 0 else None)
try:
fcf = float(info.get("freeCashflow")) if info.get("freeCashflow") is not None else None
except (TypeError, ValueError):
fcf = None
try:
mkt = float(info.get("marketCap")) if info.get("marketCap") is not None else None
except (TypeError, ValueError):
mkt = None
fcf_yield = (fcf / mkt) if fcf is not None and mkt is not None and mkt > 0 else None
company_name = info.get("longName", "")
news_sent, news_headlines = get_news_headlines(stock, ticker=ticker, company_name=company_name)
return {
"ticker": ticker,
"name": info.get("longName", ticker),
"sector": info.get("sector", "N/A"),
"industry": info.get("industry", "N/A"),
"price": price_now,
"pe_forward": info.get("forwardPE"),
"pe_trailing": info.get("trailingPE"),
"pb_ratio": info.get("priceToBook"),
"ev_ebitda": info.get("enterpriseToEbitda"),
"revenue_growth": info.get("revenueGrowth"),
"earnings_growth": info.get("earningsGrowth"),
"eps_forward": info.get("forwardEps"),
"eps_trailing": info.get("trailingEps"),
"ret_1m": pct_return(21),
"ret_3m": pct_return(63),
"ret_6m": pct_return(126),
"ret_12m": pct_return(252),
"roe": info.get("returnOnEquity"),
"roa": info.get("returnOnAssets"),
"debt_to_equity": (info["debtToEquity"] / 100.0) if info.get("debtToEquity") is not None else None,
"total_debt": info.get("totalDebt"),
"total_cash": info.get("totalCash"),
"book_value": info.get("bookValue"),
"current_ratio": info.get("currentRatio"),
"profit_margin": info.get("profitMargins"),
"operating_margin": info.get("operatingMargins"),
"fcf_yield": fcf_yield,
"news_sentiment": news_sent,
"news_headlines": news_headlines,
"analyst_norm": analyst_norm,
"analyst_upside": analyst_upside,
"analyst_count": info.get("numberOfAnalystOpinions"),
"analyst_target": target_price,
"recommendation": info.get("recommendationKey", "N/A"),
"short_percent": info.get("shortPercentOfFloat"),
"short_ratio": info.get("shortRatio"),
"beta": info.get("beta"),
"volatility_30d": volatility,
"market_cap": info.get("marketCap"),
"52w_high": info.get("fiftyTwoWeekHigh"),
"52w_low": info.get("fiftyTwoWeekLow"),
"dividend_yield": info.get("dividendYield"),
"revenue": info.get("totalRevenue"),
"business_summary": info.get("longBusinessSummary", "") or "",
}
except Exception:
return None
def fetch_all(
tickers: list[str],
max_workers: int = 5,
screen_filters: dict | None = None,
) -> pd.DataFrame:
"""
Multi-source data fetch replacing the old per-ticker yf.Ticker.info loop.
Phase 1 — yf.download() batch price history (unchanged, ~7 min)
Phase 2A — EDGAR XBRL frames: bulk fundamentals for all tickers (~2 min)
Phase 2B — Supabase analyst_cache: analyst consensus + news sentiment (~2s)
Phase 2C — FINRA RegSHO daily file: short-volume ratios (~5s)
Phase 2D — Beta calculated locally vs SPY from price history (~20s)
Phase 3 — Assemble one record per ticker, apply any active filters
Tickers with no EDGAR coverage (typically micro/nano-cap OTC stocks) still
appear in results — they score on price momentum, risk, and analyst data;
fundamental fields default to NaN and receive the 50-point neutral score.
screen_filters keys (all optional):
sector_filter (col, val) tuple | None
mktcap_range (lo, hi) tuple | None
"""
filters = screen_filters or {}
total = len(tickers)
start_date = (datetime.today() - timedelta(days=380)).strftime("%Y-%m-%d")
start_time = time.time()
# ── Phase 1: batch price history ──────────────────────────────────────
print(f"Phase 1: downloading price history for {total} tickers ...")
close_map: dict[str, pd.Series] = {}
for i in range(0, total, HIST_BATCH):
batch = tickers[i: i + HIST_BATCH]
closes = _download_history_batch(batch, start_date)
close_map.update(closes)
scanned = min(i + HIST_BATCH, total)
print(
f"\r {scanned}/{total} scanned valid={len(close_map)}"
f" ({scanned / total * 100:.0f}%) ",
end="", flush=True,
)
if i + HIST_BATCH < total:
time.sleep(HIST_PAUSE)
valid = list(close_map.keys())
print(f"\n Phase 1 done — {len(valid)}/{total} tickers have price history.\n")
if not valid:
print(" [WARN] yf.download returned no data; trying individual history calls ...")
for t in tickers[:500]:
try:
hist = yf.Ticker(t).history(start=start_date, auto_adjust=True)
if not hist.empty and len(hist) >= 20:
col = "Close" if "Close" in hist.columns else "Adj Close"
if col in hist.columns:
s = hist[col].dropna()
if len(s) >= 20:
close_map[t] = s
except Exception:
pass
time.sleep(random.uniform(0.3, 0.7))
valid = list(close_map.keys())
if not valid:
print("ERROR: No tickers returned valid price history.")
return pd.DataFrame()
print(f" Fallback found {len(valid)} tickers with history.\n")
# ── Phase 2A: Fundamentals from Supabase cache (nightly EDGAR+FINRA) ──
print("Phase 2A: Loading fundamentals cache ...")
edgar_data = _load_fundamentals_cache()
if not edgar_data:
# Cache empty (first run before nightly job) — fall back to live EDGAR
print(" Falling back to live EDGAR bulk fetch ...")
edgar_data = _fetch_edgar_bulk(close_map)
# ── Phase 2B: Analyst + news cache from Supabase ──────────────────────
print("Phase 2B: Loading analyst cache ...")
analyst_cache = _load_analyst_cache()
# ── Phase 2C: FINRA short-volume ratios ───────────────────────────────
# Primary: pull from fundamentals_cache (populated by nightly FINRA fetch).
# Fallback: live FINRA fetch when cache coverage < 60% of screened tickers.
# The two sources are always merged so no ticker misses due to cache gaps.
finra_short: dict[str, float] = {
t: d["short_percent"]
for t, d in edgar_data.items()
if d.get("short_percent") is not None
}
coverage = len(finra_short) / max(len(valid), 1)
print(f"Phase 2C: FINRA short interest loaded from cache ({len(finra_short)} tickers, {coverage:.0%} coverage)")
# ── Phase 2D: Beta vs SPY ─────────────────────────────────────────────
print("Phase 2D: Calculating beta ...")
beta_map = _calc_beta_map(close_map, start_date)
# ── Phase 3: assemble per-ticker records ──────────────────────────────
sector_filter = filters.get("sector_filter")
mktcap_range = filters.get("mktcap_range")
results: list[dict] = []
filter_labels = []
if sector_filter:
filter_labels.append(f"sector={sector_filter[1]}")
if mktcap_range:
lo, hi = mktcap_range
filter_labels.append(
f"mktcap={'$'+str(int(lo//1e9))+'B+' if lo else ''}"
f"{'$'+str(int(hi//1e9))+'B' if hi else ''}"
)
filter_note = f" [{', '.join(filter_labels)}]" if filter_labels else ""
print(f"\nPhase 3: assembling {len(valid)} records{filter_note} ...")
for ticker in valid:
close = close_map[ticker]
edgar = edgar_data.get(ticker, {})
analyst = analyst_cache.get(ticker, {})
price_now = float(close.iloc[-1])
def pct_return(days_back: int) -> float | None:
if len(close) <= days_back:
return None
idx = max(0, len(close) - days_back)
past = float(close.iloc[idx])
return (price_now - past) / past if past > 0 else None
returns_30d = close.iloc[-30:].pct_change().dropna()
volatility = (float(returns_30d.std() * (252 ** 0.5))
if len(returns_30d) >= 5 else None)
# Live market cap: price_now × shares_outstanding beats stale cached value
sh_out = edgar.get("shares_outstanding")
mkt_cap = (price_now * sh_out) if sh_out else edgar.get("market_cap")
# Track data source for N/A display in GUI
_data_source = "cache" if edgar else "none"
# EV/Revenue for growth/early-stage value scoring
ev_rev = edgar.get("ev_revenue")
# ── Filters ───────────────────────────────────────────────────────
if sector_filter is not None:
col, val = sector_filter
field = "sector" if col == "sector" else "industry"
actual = analyst.get(field) or edgar.get(field, "N/A")
if actual != val:
continue
if mktcap_range is not None:
lo, hi = mktcap_range
mkt = float(mkt_cap or 0)
if lo is not None and mkt < lo:
continue
if hi is not None and mkt >= hi:
continue
# ── Analyst fields ────────────────────────────────────────────────
analyst_target = analyst.get("analyst_target")
analyst_upside = analyst.get("analyst_upside")
if analyst_upside is None and analyst_target and price_now > 0:
try:
analyst_upside = float(analyst_target) / price_now - 1.0
except (TypeError, ValueError):
analyst_upside = None
results.append({
"ticker": ticker,
"name": analyst.get("name") or edgar.get("name") or ticker,
"sector": analyst.get("sector") or edgar.get("sector", "N/A"),
"industry": analyst.get("industry") or edgar.get("industry", "N/A"),
"price": price_now,
"_data_source": _data_source,
# Valuation
"pe_forward": analyst.get("pe_forward"),
"pe_trailing": edgar.get("pe_trailing"),
"pb_ratio": edgar.get("pb_ratio"),
"ev_ebitda": edgar.get("ev_ebitda"),
"ev_revenue": ev_rev,
# Growth
"revenue_growth": edgar.get("revenue_growth"),
"earnings_growth": edgar.get("earnings_growth"),
"eps_forward": analyst.get("eps_forward"),
"eps_trailing": edgar.get("eps_trailing"),
# Momentum (price-derived)
"ret_1m": pct_return(21),
"ret_3m": pct_return(63),
"ret_6m": pct_return(126),
"ret_12m": pct_return(252),
# Quality
"roe": edgar.get("roe"),
"roa": edgar.get("roa"),
"debt_to_equity": edgar.get("debt_to_equity"),
"total_debt": edgar.get("total_debt"),
"total_cash": edgar.get("total_cash"),
"book_value": edgar.get("book_value"),
"current_ratio": edgar.get("current_ratio"),
# Profitability
"profit_margin": edgar.get("profit_margin"),
"operating_margin": edgar.get("operating_margin"),
"fcf_yield": edgar.get("fcf_yield"),
# Sentiment (from nightly cache)
"news_sentiment": analyst.get("news_sentiment"),
"news_headlines": [],
# Analyst (from nightly cache)
"analyst_norm": analyst.get("analyst_norm"),
"analyst_upside": analyst_upside,
"analyst_count": analyst.get("analyst_count"),
"analyst_target": analyst_target,
"recommendation": analyst.get("recommendation", "N/A"),
# Bug 10 fix: use only shortPercentOfFloat (0.01.0 float fraction) from yfinance.
# FINRA short-volume ratio (sv/total_vol) is on a different scale (typically 0.30.6)
# and is incompatible with the short_pts breakpoints which expect % of float.
# Do NOT fall back to finra_short here; leave as None when yfinance data is absent.
"short_percent": analyst.get("short_percent"),
"short_ratio": analyst.get("short_ratio"),
"piotroski_score": edgar.get("piotroski_score"),
"accruals_ratio": edgar.get("accruals_ratio"),
"news_headline_count": analyst.get("news_headline_count"),
"beta": beta_map.get(ticker),
"volatility_30d": volatility,
# Market data
"market_cap": mkt_cap,
"52w_high": float(close.max()),
"52w_low": float(close.min()),
"dividend_yield": edgar.get("dividend_yield"),
"revenue": edgar.get("revenue"),
"business_summary": "",
})
print(f"\nDone. Assembled {len(results)} stocks in"
f" {time.time() - start_time:.0f}s.\n")
df = pd.DataFrame(results)
if not df.empty:
df = df.drop_duplicates(subset=["ticker"], keep="first").reset_index(drop=True)
return df
# ---------------------------------------------------------------------------
# Sector Stats Loader
# ---------------------------------------------------------------------------
def _load_sector_stats() -> dict[str, dict]:
"""
Load pre-computed sector statistics from the Verimund API.
Returns {sector: {metric: (median, MAD)}} or {} on failure.
Used by z-score scoring to normalize metrics relative to sector peers.
"""
if not _API_URL:
return {}
try:
resp = requests.get(
f"{_API_URL}/cache/sector",
headers=_get_api_headers(),
timeout=15,
)
resp.raise_for_status()
rows = resp.json().get("data", [])
result: dict[str, dict] = {}
for row in rows:
sector = row.get("sector")
if not sector:
continue
stats: dict[str, tuple] = {}
for key, val in row.items():
if key.endswith("_med"):
metric = key[:-4]
mad_key = f"{metric}_mad"
mad = row.get(mad_key)
if val is not None and mad is not None and mad > 0:
stats[metric] = (float(val), float(mad))
result[sector] = stats
print(f" Sector stats loaded: {len(result)} sectors")
return result
except Exception as e:
print(f" [WARN] Sector stats unavailable: {e}")
return {}
# ---------------------------------------------------------------------------
# Scoring
# ---------------------------------------------------------------------------
def _abs_score(v, breakpoints: list, default: float = 50.0) -> float:
"""
Piecewise-linear interpolation of a scalar value v against a list of
(metric_value, score) breakpoints sorted by metric_value ascending.
Returns `default` when v is NaN/None.
Example: breakpoints=[(0,0),(10,50),(20,100)]
v=5 → 25.0
v=15 → 75.0
v=25 → 100.0 (clamped)
"""
if not breakpoints:
return default
if v is None:
return default
try:
if np.isnan(v):
return default
except (TypeError, ValueError):
return default
pts = breakpoints
if v <= pts[0][0]:
return float(pts[0][1])
if v >= pts[-1][0]:
return float(pts[-1][1])
for i in range(len(pts) - 1):
x0, y0 = pts[i]
x1, y1 = pts[i + 1]
if x0 <= v <= x1:
t = (v - x0) / (x1 - x0)
return float(y0 + t * (y1 - y0))
return default
def _score_series(series: pd.Series, breakpoints: list, default: float = 50.0) -> pd.Series:
"""Apply _abs_score element-wise to a Series."""
return series.apply(lambda v: _abs_score(v, breakpoints, default))
def _z_score(value, metric: str, sector: str, sector_stats: dict,
default: float = 50.0) -> float:
"""
Normalize value relative to sector peers using median and MAD, then map
the resulting z-score to a 0100 scale via piecewise-linear interpolation.
Falls back to default (50 = neutral) when value or stats are unavailable.
Z-score curve: z ≤ -2.5 → 3, z=0 → 50, z ≥ +2.5 → 97
"""
if value is None:
return default
try:
if np.isnan(float(value)):
return default
except (TypeError, ValueError):
return default
stats = sector_stats.get(sector, {})
pair = stats.get(metric)
if pair is None:
# Try market-wide fallback
pair = sector_stats.get("__market__", {}).get(metric)
if pair is None:
return default
med, mad = pair
if mad <= 0:
return default
z = (float(value) - med) / mad
z = max(-3.0, min(3.0, z)) # winsorize
z_pts = [(-2.5, 3), (-1.5, 18), (-0.5, 38), (0.0, 50),
(0.5, 62), (1.5, 82), (2.5, 97)]
return _abs_score(z, z_pts, default)
def _z_series(series: pd.Series, metric: str, sectors: pd.Series,
sector_stats: dict, default: float = 50.0) -> pd.Series:
"""Apply _z_score element-wise using per-row sector labels."""
return pd.Series(
[_z_score(v, metric, s, sector_stats, default)
for v, s in zip(series, sectors)],
index=series.index,
)
def _classify_profile(row: pd.Series) -> dict:
"""
Classify a stock into one or more company profiles with blend weights.
Returns {profile_name: weight} summing to 1.0.
Supports soft blending for ambiguous cases.
"""
sector = str(row.get("sector") or "").lower()
rev_g = row.get("revenue_growth")
pm = row.get("profit_margin")
de = row.get("debt_to_equity")
revenue = row.get("revenue")
mktcap = row.get("market_cap")
earn_g = row.get("earnings_growth")
# Financial: banks, insurance, credit — D/E irrelevant, use ROA/P/B
fin_sectors = ("financial", "bank", "insurance", "credit")
if any(s in sector for s in fin_sectors):
return {"financial": 1.0}
# Capital intensive: utilities, energy, industrials with meaningful leverage
cap_sectors = ("utilities", "energy", "industrials", "basic materials")
if any(s in sector for s in cap_sectors) and de is not None and de > 0.8:
return {"capital_intensive": 1.0}
# Score membership in remaining profiles
scores = {}
# Early stage: very small revenue or deeply negative margins
es_score = 0.0
if revenue is not None and revenue < 30_000_000:
es_score += 0.6
if pm is not None and pm < -0.30 and mktcap is not None and mktcap < 1_000_000_000:
es_score += 0.4
if es_score > 0:
scores["early_stage"] = min(es_score, 1.0)
# High growth: fast revenue expansion, possibly thin margins
hg_score = 0.0
if rev_g is not None and rev_g > 0.25:
hg_score += 0.7
elif rev_g is not None and rev_g > 0.15 and pm is not None and pm < 0.05:
hg_score += 0.5
if hg_score > 0:
scores["high_growth"] = min(hg_score, 1.0)
# Turnaround: strong earnings recovery from a loss position
if earn_g is not None and earn_g > 0.40 and pm is not None and pm < 0.0:
scores["turnaround"] = 0.8
# Mature: default for profitable, slower-growth companies
if not scores:
return {"mature": 1.0}
# Normalize to sum = 1.0
total = sum(scores.values())
return {k: v / total for k, v in scores.items()}
# Profile-specific sub-component weights for each scoring dimension.
# Structure: {profile: {dimension: {metric: weight}}}
_PROFILE_WEIGHTS: dict[str, dict[str, dict[str, float]]] = {
"mature": {
"value": {"pe": 0.35, "pb": 0.20, "ev_ebitda": 0.45},
"growth": {"revenue_growth": 0.30, "earnings_growth": 0.45, "eps_growth": 0.25},
"quality": {"roa": 0.40, "roe": 0.20, "debt_to_equity": 0.30, "current_ratio": 0.10, "piotroski": 0.20},
"profitability": {"fcf_yield": 0.40, "operating_margin": 0.38, "profit_margin": 0.22, "accruals": 0.15},
},
"high_growth": {
"value": {"pe": 0.00, "pb": 0.20, "ev_ebitda": 0.35, "ev_revenue": 0.45},
"growth": {"revenue_growth": 0.50, "earnings_growth": 0.30, "eps_growth": 0.20},
"quality": {"roa": 0.35, "roe": 0.15, "debt_to_equity": 0.15, "current_ratio": 0.35, "piotroski": 0.20},
"profitability": {"fcf_yield": 0.15, "operating_margin": 0.50, "profit_margin": 0.35, "accruals": 0.15},
},
"financial": {
"value": {"pe": 0.35, "pb": 0.50, "ev_ebitda": 0.15},
"growth": {"revenue_growth": 0.30, "earnings_growth": 0.50, "eps_growth": 0.20},
"quality": {"roa": 0.50, "roe": 0.30, "debt_to_equity": 0.00, "current_ratio": 0.20, "piotroski": 0.15},
"profitability": {"fcf_yield": 0.30, "operating_margin": 0.00, "profit_margin": 0.70, "accruals": 0.15},
},
"capital_intensive": {
"value": {"pe": 0.15, "pb": 0.20, "ev_ebitda": 0.65},
"growth": {"revenue_growth": 0.35, "earnings_growth": 0.40, "eps_growth": 0.25},
"quality": {"roa": 0.30, "roe": 0.15, "debt_to_equity": 0.40, "current_ratio": 0.15, "piotroski": 0.15},
"profitability": {"fcf_yield": 0.35, "operating_margin": 0.45, "profit_margin": 0.20, "accruals": 0.15},
},
"early_stage": {
"value": {"pe": 0.00, "pb": 0.15, "ev_ebitda": 0.00, "ev_revenue": 0.85},
"growth": {"revenue_growth": 0.70, "earnings_growth": 0.20, "eps_growth": 0.10},
"quality": {"roa": 0.20, "roe": 0.00, "debt_to_equity": 0.00, "current_ratio": 0.80, "piotroski": 0.10},
# revenue_growth_proxy removed — revenue growth already carries 70% of the growth score
"profitability": {"fcf_yield": 0.00, "operating_margin": 1.00, "profit_margin": 0.00,
"revenue_growth_proxy": 0.00},
},
"turnaround": {
"value": {"pe": 0.00, "pb": 0.30, "ev_ebitda": 0.70},
"growth": {"revenue_growth": 0.15, "earnings_growth": 0.80, "eps_growth": 0.05},
"quality": {"roa": 0.35, "roe": 0.15, "debt_to_equity": 0.35, "current_ratio": 0.15, "piotroski": 0.20},
"profitability": {"fcf_yield": 0.25, "operating_margin": 0.45, "profit_margin": 0.30, "accruals": 0.15},
},
}
def _profile_score_row(row: pd.Series, dimension: str,
metric_scores: dict[str, float]) -> float:
"""
Compute a single dimension score for one row using profile-blended weights.
metric_scores: {metric_name: 0-100 score} — pre-computed for this row.
"""
profile_blend = _classify_profile(row)
total = 0.0
for profile, profile_weight in profile_blend.items():
pw = _PROFILE_WEIGHTS.get(profile, _PROFILE_WEIGHTS["mature"])
dim_weights = pw.get(dimension, {})
# Normalize weights to only available metrics
available = {m: w for m, w in dim_weights.items()
if m in metric_scores and w > 0}
if not available:
total += profile_weight * 50.0
continue
w_sum = sum(available.values())
dim_score = sum(metric_scores[m] * (w / w_sum) for m, w in available.items())
total += profile_weight * dim_score
return total
def compute_value_score(df: pd.DataFrame,
sector_stats: dict | None = None) -> pd.Series:
"""
Value Score — sector-relative z-score for each valuation metric, then
profile-blended sub-weights applied per row.
"""
ss = sector_stats or {}
sectors = df.get("sector", pd.Series(["N/A"] * len(df), index=df.index))
# Z-score forward and trailing PE against their own sector distributions, then combine
# Lower valuation ratios are better — invert z-scores so cheap stocks rank high
fwd_s = 100.0 - _z_series(df["pe_forward"].where(df["pe_forward"] > 0), "pe_forward", sectors, ss)
trl_s = 100.0 - _z_series(df["pe_trailing"].where(df["pe_trailing"] > 0), "pe_trailing", sectors, ss)
pe_s = fwd_s.combine_first(trl_s)
pb_s = 100.0 - _z_series(df["pb_ratio"], "pb_ratio", sectors, ss)
ev_s = 100.0 - _z_series(df["ev_ebitda"], "ev_ebitda", sectors, ss)
evr_s = 100.0 - _z_series(df.get("ev_revenue", pd.Series(
[None]*len(df), index=df.index)),
"ev_revenue", sectors, ss)
scores = []
for i, row in df.iterrows():
ms = {
"pe": float(pe_s.loc[i]),
"pb": float(pb_s.loc[i]),
"ev_ebitda": float(ev_s.loc[i]),
"ev_revenue": float(evr_s.loc[i]),
}
scores.append(_profile_score_row(row, "value", ms))
return pd.Series(scores, index=df.index)
def compute_growth_score(df: pd.DataFrame,
sector_stats: dict | None = None) -> pd.Series:
"""
Growth Score — sector-relative z-scores for revenue growth, earnings growth,
and EPS growth rate (clamped to ±150% to avoid breakeven-crossing distortion).
"""
ss = sector_stats or {}
sectors = df.get("sector", pd.Series(["N/A"] * len(df), index=df.index))
# EPS growth rate: (forward - trailing) / abs(trailing), clamped ±1.5
eps_t = pd.to_numeric(df["eps_trailing"], errors="coerce")
eps_f = pd.to_numeric(df["eps_forward"], errors="coerce")
eps_g = ((eps_f - eps_t) / eps_t.abs()).clip(-1.5, 1.5).where(eps_t.abs() > 1e-9)
rev_s = _z_series(df["revenue_growth"], "revenue_growth", sectors, ss)
ear_s = _z_series(df["earnings_growth"], "earnings_growth", sectors, ss)
eps_s = _z_series(eps_g, "eps_growth", sectors, ss)
scores = []
for i, row in df.iterrows():
ms = {
"revenue_growth": float(rev_s.loc[i]),
"earnings_growth": float(ear_s.loc[i]),
"eps_growth": float(eps_s.loc[i]),
}
scores.append(_profile_score_row(row, "growth", ms))
return pd.Series(scores, index=df.index)
def compute_momentum_score(df: pd.DataFrame) -> pd.Series:
"""
Momentum Score — absolute breakpoints (price returns are inherently comparable).
1M weight reduced to 5% to minimise short-term reversal noise.
"""
r1_pts = [(-0.15, 5), (-0.05, 25), (0.0, 45), (0.05, 60), (0.10, 75), (0.20, 90), (0.30, 100)]
r3_pts = [(-0.20, 5), (-0.05, 25), (0.0, 42), (0.08, 58), (0.15, 72), (0.25, 88), (0.40, 100)]
r6_pts = [(-0.25, 5), (-0.05, 22), (0.0, 38), (0.10, 55), (0.20, 70), (0.35, 87), (0.55, 100)]
r12_pts = [(-0.30, 5), (-0.05, 20), (0.0, 35), (0.12, 52), (0.25, 68), (0.40, 85), (0.65, 100)]
return (
_score_series(df["ret_1m"], r1_pts) * 0.05 +
_score_series(df["ret_3m"], r3_pts) * 0.25 +
_score_series(df["ret_6m"], r6_pts) * 0.45 +
_score_series(df["ret_12m"], r12_pts) * 0.25
)
def compute_reverse_momentum_score(df: pd.DataFrame) -> pd.Series:
"""
Reverse Momentum Score — rewards stocks that have fallen the most recently.
Used by the High Risk High Reward strategy preset.
"""
r1_pts = [(-0.30, 100), (-0.20, 88), (-0.10, 75), (-0.05, 62), (0.0, 48), (0.05, 32), (0.10, 18), (0.20, 8), (0.30, 3)]
r3_pts = [(-0.40, 100), (-0.25, 88), (-0.15, 75), (-0.05, 60), (0.0, 45), (0.08, 30), (0.20, 15), (0.35, 5)]
r6_pts = [(-0.55, 100), (-0.35, 88), (-0.20, 72), (-0.05, 55), (0.0, 42), (0.10, 28), (0.25, 15), (0.40, 5)]
r12_pts = [(-0.65, 100), (-0.40, 85), (-0.25, 68), (-0.05, 52), (0.0, 38), (0.12, 22), (0.30, 10), (0.50, 3)]
return (
_score_series(df["ret_1m"], r1_pts) * 0.35 +
_score_series(df["ret_3m"], r3_pts) * 0.35 +
_score_series(df["ret_6m"], r6_pts) * 0.20 +
_score_series(df["ret_12m"], r12_pts) * 0.10
)
def compute_quality_score(df: pd.DataFrame,
sector_stats: dict | None = None) -> pd.Series:
"""
Quality Score — sector-relative z-scores with profile-blended sub-weights.
ROA promoted to primary metric; ROE demoted to reduce leverage-gaming bias.
"""
ss = sector_stats or {}
sectors = df.get("sector", pd.Series(["N/A"] * len(df), index=df.index))
roa_s = _z_series(df["roa"], "roa", sectors, ss)
roe_s = _z_series(df["roe"], "roe", sectors, ss)
cr_s = _z_series(df["current_ratio"], "current_ratio", sectors, ss)
# Negative D/E means negative equity (insolvency risk) — treat as extremely high leverage
de_raw = df["debt_to_equity"].where(df["debt_to_equity"] >= 0, 999.0)
# D/E z-score is inverted — lower D/E is better, so flip the z-score
de_s = 100.0 - _z_series(de_raw, "debt_to_equity", sectors, ss)
# Piotroski partial F-Score (0-4) → 0-100 via absolute breakpoints
p_raw = pd.to_numeric(df.get("piotroski_score",
pd.Series([None] * len(df), index=df.index)), errors="coerce")
p_pts = [(0, 5), (1, 30), (2, 50), (3, 72), (4, 95)]
piotroski_s = _score_series(p_raw, p_pts)
scores = []
for i, row in df.iterrows():
ms = {
"roa": float(roa_s.loc[i]),
"roe": float(roe_s.loc[i]),
"debt_to_equity": float(de_s.loc[i]),
"current_ratio": float(cr_s.loc[i]),
"piotroski": float(piotroski_s.loc[i]),
}
scores.append(_profile_score_row(row, "quality", ms))
return pd.Series(scores, index=df.index)
def compute_profitability_score(df: pd.DataFrame,
sector_stats: dict | None = None) -> pd.Series:
"""
Profitability Score — sector-relative z-scores. FCF yield promoted to 40%
for mature companies. Early stage profile substitutes revenue growth.
"""
ss = sector_stats or {}
sectors = df.get("sector", pd.Series(["N/A"] * len(df), index=df.index))
fcf_s = _z_series(df["fcf_yield"], "fcf_yield", sectors, ss)
om_s = _z_series(df["operating_margin"], "operating_margin", sectors, ss)
pm_s = _z_series(df["profit_margin"], "profit_margin", sectors, ss)
rev_s = _z_series(df["revenue_growth"], "revenue_growth", sectors, ss)
# Sloan accruals ratio: more negative = earnings are cash-backed = better quality
acc_raw = pd.to_numeric(df.get("accruals_ratio",
pd.Series([None] * len(df), index=df.index)), errors="coerce")
acc_pts = [(-0.25, 95), (-0.10, 78), (-0.03, 62), (0.0, 50),
(0.03, 38), (0.10, 22), (0.25, 5)]
accruals_s = _score_series(acc_raw, acc_pts)
scores = []
for i, row in df.iterrows():
ms = {
"fcf_yield": float(fcf_s.loc[i]),
"operating_margin": float(om_s.loc[i]),
"profit_margin": float(pm_s.loc[i]),
"revenue_growth_proxy": float(rev_s.loc[i]), # used by early_stage
"accruals": float(accruals_s.loc[i]),
}
scores.append(_profile_score_row(row, "profitability", ms))
return pd.Series(scores, index=df.index)
def compute_sentiment_score(df: pd.DataFrame) -> pd.Series:
"""
News Sentiment Score — VADER compound score mapped to 0100, confidence-weighted
by headline count. Fewer headlines blend toward neutral (50) to avoid overreacting
to a single article.
"""
sent_pts = [(-1.0, 0), (-0.40, 20), (-0.15, 36), (0.0, 50),
(0.15, 64), (0.40, 80), (1.0, 100)]
base = _score_series(df["news_sentiment"], sent_pts)
# Derive headline count from stored field or from news_headlines list
if "news_headline_count" in df.columns:
count = pd.to_numeric(df["news_headline_count"], errors="coerce").fillna(0)
elif "news_headlines" in df.columns:
count = df["news_headlines"].apply(
lambda h: len(h) if isinstance(h, list) else 0
).astype(float)
else:
return base
# 5+ headlines = full confidence; 0 headlines = 5% confidence (nearly neutral)
confidence = (count.clip(upper=5) / 5.0).clip(lower=0.05)
return base * confidence + 50.0 * (1.0 - confidence)
def compute_analyst_score(df: pd.DataFrame,
sector_stats: dict | None = None) -> pd.Series:
"""
Analyst Score — consensus (absolute) + upside (sector z-score) + count (absolute).
Upside top breakpoints compressed to dampen analyst optimism bias.
"""
ss = sector_stats or {}
sectors = df.get("sector", pd.Series(["N/A"] * len(df), index=df.index))
norm_pts = [(0.0, 0), (0.25, 25), (0.50, 50), (0.65, 65),
(0.75, 78), (0.875, 90), (1.0, 100)]
count_pts = [(0, 0), (1, 20), (3, 40), (5, 55), (10, 70), (20, 85), (30, 100)]
# Upside: sector z-score with dampened ceiling for extreme claims
upside_pts = [(-0.25, 3), (-0.08, 20), (0.0, 35), (0.05, 48),
(0.12, 62), (0.25, 80), (0.40, 88), (0.60, 92)]
norm_s = _score_series(df["analyst_norm"], norm_pts)
count_s = _score_series(df["analyst_count"], count_pts)
upside_s = _z_series(df["analyst_upside"], "analyst_upside", sectors, ss)
# Fall back to absolute breakpoints when sector stats unavailable
upside_abs = _score_series(df["analyst_upside"], upside_pts)
has_stats = sectors.map(lambda s: s in ss and "analyst_upside" in ss.get(s, {}))
upside_s = upside_s.where(has_stats, upside_abs)
return (
norm_s * 0.43 +
upside_s * 0.42 +
count_s * 0.15
)
def compute_risk_score(df: pd.DataFrame) -> pd.Series:
"""
Risk Score — higher score = lower risk.
Dynamic short interest weighting: contributes 35% when data is present,
drops out (weight redistributed to volatility/beta) when absent.
Beta: asymmetric scoring peaking at beta 0.2-0.4 (genuine defensive).
"""
vol_pts = [(0.05, 100), (0.10, 88), (0.15, 78), (0.25, 62),
(0.35, 45), (0.50, 25), (0.70, 8), (1.0, 3)]
beta_pts = [(0.0, 88), (0.2, 96), (0.4, 90), (0.7, 78), (0.85, 70),
(1.0, 62), (1.3, 45), (1.6, 28), (2.0, 12), (3.0, 3)]
short_pts= [(0.0, 100), (0.03, 88), (0.05, 75), (0.10, 55),
(0.15, 35), (0.20, 18), (0.30, 5)]
beta_clipped = df["beta"].fillna(1.0).clip(lower=0.0)
vol_s = _score_series(df["volatility_30d"], vol_pts)
beta_s = _score_series(beta_clipped, beta_pts)
short_s = _score_series(df["short_percent"], short_pts)
has_short = df["short_percent"].notna()
# Dynamic weighting
score_with = short_s * 0.35 + vol_s * 0.40 + beta_s * 0.25
score_without = vol_s * 0.65 + beta_s * 0.35
return score_with.where(has_short, score_without)
def score_stocks(df: pd.DataFrame, weights: dict | None = None,
sector_stats: dict | None = None) -> pd.DataFrame:
df = df.copy()
df = df.drop_duplicates(subset=["ticker"], keep="first").reset_index(drop=True)
# Load sector stats if not provided (allows GUI re-scoring to pass cached stats)
ss = sector_stats if sector_stats is not None else _load_sector_stats()
_NUMERIC_COLS = [
"pe_forward", "pe_trailing", "pb_ratio", "ev_ebitda", "ev_revenue",
"revenue_growth", "earnings_growth", "eps_forward", "eps_trailing",
"ret_1m", "ret_3m", "ret_6m", "ret_12m",
"roe", "roa", "debt_to_equity", "current_ratio",
"profit_margin", "operating_margin", "fcf_yield",
"news_sentiment", "news_headline_count", "analyst_norm", "analyst_upside", "analyst_count",
"short_percent", "short_ratio", "volatility_30d", "beta", "price", "market_cap",
"piotroski_score", "accruals_ratio",
]
for col in _NUMERIC_COLS:
if col in df.columns:
df[col] = pd.to_numeric(df[col], errors="coerce")
df["score_value"] = compute_value_score(df, ss)
df["score_growth"] = compute_growth_score(df, ss)
df["score_momentum"] = compute_momentum_score(df)
df["score_reverse_momentum"] = compute_reverse_momentum_score(df)
df["score_quality"] = compute_quality_score(df, ss)
df["score_profitability"] = compute_profitability_score(df, ss)
df["score_sentiment"] = compute_sentiment_score(df)
df["score_analyst"] = compute_analyst_score(df, ss)
df["score_risk"] = compute_risk_score(df)
w = weights if weights is not None else WEIGHTS
df["composite_score"] = sum(
df[f"score_{k}"] * v for k, v in w.items()
)
return df.sort_values("composite_score", ascending=False).reset_index(drop=True)
# ---------------------------------------------------------------------------
# Display
# ---------------------------------------------------------------------------
def fmt_pct(val) -> str:
if val is None or (isinstance(val, float) and np.isnan(val)):
return "N/A"
return f"{val*100:.1f}%"
def fmt_float(val, decimals=2) -> str:
if val is None or (isinstance(val, float) and np.isnan(val)):
return "N/A"
return f"{val:.{decimals}f}"
def fmt_mcap(val) -> str:
if val is None or (isinstance(val, float) and np.isnan(val)):
return "N/A"
if val >= 1e12: return f"${val/1e12:.1f}T"
if val >= 1e9: return f"${val/1e9:.1f}B"
if val >= 1e6: return f"${val/1e6:.1f}M"
return f"${val:.0f}"
def print_results(df: pd.DataFrame, top_n: int = 50) -> None:
top = df.head(top_n).copy()
top.index = range(1, len(top) + 1)
W = 140
print("=" * W)
print(f" TOP {top_n} STOCKS — {datetime.today().strftime('%Y-%m-%d %H:%M')}")
print("=" * W)
print(
f"{'#':>3} {'Ticker':<7} {'Name':<26} {'Sector':<20} "
f"{'Score':>6} {'Val':>5} {'Grw':>5} {'Mom':>5} {'Qlt':>5} "
f"{'Pft':>5} {'Sent':>5} {'Anlst':>6} {'Risk':>5} "
f"{'P/E':>6} {'RevGrw':>7} {'ROE':>6} {'MktCap':>8}"
)
print("-" * W)
for rank, row in top.iterrows():
print(
f"{rank:>3} {row['ticker']:<7} {str(row['name'])[:25]:<26} "
f"{str(row['sector'])[:19]:<20} "
f"{row['composite_score']:>6.1f} "
f"{row['score_value']:>5.1f} {row['score_growth']:>5.1f} "
f"{row['score_momentum']:>5.1f} {row['score_quality']:>5.1f} "
f"{row['score_profitability']:>5.1f} {row['score_sentiment']:>5.1f} "
f"{row['score_analyst']:>6.1f} {row['score_risk']:>5.1f} "
f"{fmt_float(row.get('pe_forward') or row.get('pe_trailing')):>6} "
f"{fmt_pct(row.get('revenue_growth')):>7} "
f"{fmt_pct(row.get('roe')):>6} "
f"{fmt_mcap(row.get('market_cap')):>8}"
)
print("=" * W)
w = WEIGHTS
print(
f"\nFormula: {w['value']*100:.0f}%*Val + {w['growth']*100:.0f}%*Grw + "
f"{w['momentum']*100:.0f}%*Mom + {w['quality']*100:.0f}%*Qlt + "
f"{w['profitability']*100:.0f}%*Pft + {w['sentiment']*100:.0f}%*Sent + "
f"{w['analyst']*100:.0f}%*Anlst + {w['risk']*100:.0f}%*Risk\n"
)
def save_csv(df: pd.DataFrame, top_n: int) -> str:
cols = [
"ticker", "name", "sector", "price", "market_cap",
"composite_score", "score_value", "score_growth", "score_momentum",
"score_quality", "score_profitability", "score_sentiment",
"score_analyst", "score_risk",
"pe_forward", "pe_trailing", "pb_ratio", "ev_ebitda",
"revenue_growth", "earnings_growth",
"ret_1m", "ret_3m", "ret_6m", "ret_12m",
"roe", "roa", "debt_to_equity", "current_ratio",
"profit_margin", "operating_margin", "fcf_yield",
"news_sentiment", "analyst_norm", "analyst_upside",
"analyst_count", "analyst_target", "recommendation",
"short_percent", "beta", "volatility_30d",
]
out_cols = [c for c in cols if c in df.columns]
filename = f"top{top_n}_stocks_{datetime.today().strftime('%Y%m%d')}.csv"
df.head(top_n)[out_cols].to_csv(filename, index=False)
return filename
# ---------------------------------------------------------------------------
# Entry Point
# ---------------------------------------------------------------------------
def parse_args():
p = argparse.ArgumentParser(description="Multi-factor stock screener")
p.add_argument("--top", type=int, default=50)
p.add_argument("--workers", type=int, default=15)
p.add_argument("--output", choices=["console", "csv", "both"], default="console")
p.add_argument("--index", default="All", choices=INDEX_CHOICES,
help="Index universe to screen (default: All)")
return p.parse_args()
def main():
args = parse_args()
print("\n" + "=" * 60)
print(" MULTI-FACTOR STOCK SCREENER (8-Factor Analyst Model)")
print("=" * 60 + "\n")
if _VADER is None:
print(" [INFO] vaderSentiment not installed — sentiment scores will be neutral.")
print(" Run: pip install vaderSentiment\n")
tickers = collect_tickers(index=args.index)
if not tickers:
print("ERROR: Could not retrieve any ticker lists.")
sys.exit(1)
df_raw = fetch_all(tickers, max_workers=args.workers)
if df_raw.empty:
print("ERROR: No data returned.")
sys.exit(1)
print("Computing scores...")
df_scored = score_stocks(df_raw)
print(f"Scored {len(df_scored)} stocks.\n")
if args.output in ("console", "both"):
print_results(df_scored, top_n=args.top)
if args.output in ("csv", "both"):
print(f"Results saved to: {save_csv(df_scored, top_n=args.top)}")
# ---------------------------------------------------------------------------
# Insider Trading — SEC EDGAR Form 4
# ---------------------------------------------------------------------------
_FORM4_TX_CODES: dict[str, str] = {
"P": "Buy", "S": "Sale",
"A": "Award", "D": "Disposition",
"F": "Tax Withholding","G": "Gift",
"M": "Option Exercise","X": "Option Exercise",
"O": "Option Exercise","J": "Other",
"I": "Plan Trade", "C": "Conversion",
}
def _parse_form4_xml(xml_bytes: bytes, filing_url: str, filed_at: str) -> list[dict]:
"""Parse a Form 4 XML document and return a list of transaction dicts."""
import xml.etree.ElementTree as ET
try:
root = ET.fromstring(xml_bytes)
except ET.ParseError:
return []
def _val(node, *tags):
"""Return stripped text from the first matching .//tag/value or .//tag."""
for tag in tags:
n = node.find(f".//{tag}/value")
if n is not None and n.text:
return n.text.strip()
n = node.find(f".//{tag}")
if n is not None and n.text:
return n.text.strip()
return ""
ticker = _val(root, "issuerTradingSymbol")
company = _val(root, "issuerName")
insider = _val(root, "rptOwnerName")
period = _val(root, "periodOfReport")
title = ""
rel = root.find(".//reportingOwnerRelationship")
if rel is not None:
ot = _val(rel, "officerTitle")
if ot:
title = ot
elif _val(rel, "isDirector") in ("1", "true"):
title = "Director"
elif _val(rel, "isTenPercentOwner") in ("1", "true"):
title = "10% Owner"
elif _val(rel, "isOfficer") in ("1", "true"):
title = "Officer"
transactions: list[dict] = []
for tx in root.findall(".//nonDerivativeTransaction"):
tx_date = _val(tx, "transactionDate") or period
tx_code = _val(tx, "transactionCode")
tx_type = _FORM4_TX_CODES.get(tx_code, tx_code or "Other")
try: shares = float(_val(tx, "transactionShares"))
except: shares = None
try: price = float(_val(tx, "transactionPricePerShare"))
except: price = None
try: owned = float(_val(tx, "sharesOwnedFollowingTransaction"))
except: owned = None
value = (shares * price) if (shares and price) else None
transactions.append({
"filed_date": (filed_at or "")[:10] or tx_date,
"tx_date": tx_date,
"ticker": ticker.upper() if ticker else "",
"company": company,
"insider": insider,
"title": title,
"transaction_type": tx_type,
"transaction_code": tx_code,
"shares": shares,
"price": price,
"value": value,
"owned_after": owned,
"url": filing_url,
})
return transactions
def fetch_insider_trades(days_back: int = 3, progress_cb=None) -> list[dict]:
"""
Fetch ALL Form 4 insider transactions from SEC EDGAR for the given period.
Paginates the EDGAR EFTS search API to collect every filing in the date
range, then parallel-fetches each Form 4 XML for full transaction details.
A shared rate-limiter keeps requests under EDGAR's 10 req/s limit.
days_back : number of calendar days back from today to search.
progress_cb : optional callable(done_count, total_count).
Returns a list of transaction dicts sorted by tx_date descending.
"""
import time
import threading
from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import date, timedelta
_UA = {"User-Agent": "UltimateInvestmentTool contact@investmenttool.com"}
# ── Rate limiter: max ~9 requests/sec across all threads ──────────────
_rate_lock = threading.Lock()
_last_req = [0.0]
MIN_INTERVAL = 0.12 # seconds between requests → ~8 req/s, safely under 10
def _get(url, **kwargs):
with _rate_lock:
wait = MIN_INTERVAL - (time.monotonic() - _last_req[0])
if wait > 0:
time.sleep(wait)
_last_req[0] = time.monotonic()
return requests.get(url, **kwargs)
end_dt = date.today()
start_dt = end_dt - timedelta(days=max(days_back, 1))
# ── Step 1: paginate EDGAR EFTS to collect all filing hits ────────────
PAGE_SIZE = 100
all_hits: list[dict] = []
from_offset = 0
while True:
try:
resp = _get(
"https://efts.sec.gov/LATEST/search-index",
params={
"forms": "4",
"dateRange": "custom",
"startdt": start_dt.strftime("%Y-%m-%d"),
"enddt": end_dt.strftime("%Y-%m-%d"),
"from": from_offset,
"hits": PAGE_SIZE,
},
headers=_UA,
timeout=15,
)
resp.raise_for_status()
except Exception:
break
page_hits = resp.json().get("hits", {}).get("hits", [])
all_hits.extend(page_hits)
if len(page_hits) < PAGE_SIZE:
break # reached the last page
from_offset += PAGE_SIZE
if not all_hits:
return []
# ── Step 2: fetch + parse each Form 4 XML (rate-limited, parallel) ────
def _fetch_one(hit: dict) -> list[dict]:
try:
src = hit.get("_source", {})
filed = src.get("file_date", "") or src.get("filed_at", "")
# Accession number from _source.adsh (not _id which has ":filename" appended)
acc_no = src.get("adsh", "")
if not acc_no:
return []
# XML filename is embedded in _id as "{adsh}:{filename}"
xml_filename = hit["_id"].split(":")[-1]
# CIK: use ciks[0] from _source — NOT the accession prefix (which is the filing agent)
ciks = src.get("ciks", [])
if not ciks:
return []
cik_int = int(ciks[0].lstrip("0") or "0")
if cik_int == 0:
return []
acc_nd = acc_no.replace("-", "")
base = f"https://www.sec.gov/Archives/edgar/data/{cik_int}/{acc_nd}"
filing_url = f"{base}/{acc_no}-index.htm"
xml_url = f"{base}/{xml_filename}"
xml_r = _get(xml_url, headers=_UA, timeout=8)
xml_r.raise_for_status()
return _parse_form4_xml(xml_r.content, filing_url, filed)
except Exception:
return []
total = len(all_hits)
done_n = 0
results: list[dict] = []
with ThreadPoolExecutor(max_workers=5) as ex:
futs = {ex.submit(_fetch_one, h): h for h in all_hits}
for fut in as_completed(futs):
done_n += 1
results.extend(fut.result() or [])
if progress_cb:
try:
progress_cb(done_n, total)
except Exception:
pass
results.sort(
key=lambda x: (x.get("tx_date") or "", x.get("filed_date") or ""),
reverse=True,
)
return results
# ---------------------------------------------------------------------------
# Hedge Fund Holdings — SEC EDGAR 13F-HR Filings
# ---------------------------------------------------------------------------
def _parse_13f_xml(xml_bytes: bytes) -> list[dict]:
"""
Parse a 13F information table XML document.
Handles any namespace variant used by EDGAR filers.
Returns a list of holding dicts with keys:
company, class_, cusip, value (USD), shares, shr_type, put_call
"""
import xml.etree.ElementTree as ET
import re as _re
# SEC serves 13F XML wrapped in an HTML document (Content-Type: text/html).
# Strip the HTML wrapper and extract just the <informationTable> block.
if xml_bytes[:500].upper().find(b'<!DOCTYPE HTML') != -1:
m = _re.search(rb'(<informationTable[\s\S]*?</informationTable>)',
xml_bytes, _re.IGNORECASE)
if m:
xml_bytes = m.group(1)
try:
root = ET.fromstring(xml_bytes)
except ET.ParseError:
return []
# Strip XML namespace prefix so tag matching is namespace-agnostic
def _bare(tag: str) -> str:
return tag.split("}")[-1] if "}" in tag else tag
def _find(node, bare_tag):
for child in node:
if _bare(child.tag) == bare_tag:
return child
return None
def _text(node, bare_tag) -> str:
n = _find(node, bare_tag)
return n.text.strip() if n is not None and n.text else ""
# Root is <informationTable>; entries are <infoTable>
info_tables = [c for c in root if _bare(c.tag) == "infoTable"]
holdings: list[dict] = []
for entry in info_tables:
company = _text(entry, "nameOfIssuer")
class_ = _text(entry, "titleOfClass")
cusip = _text(entry, "cusip")
put_call = _text(entry, "putCall")
try:
value_k = float(_text(entry, "value"))
except (ValueError, TypeError):
value_k = None
shr_node = _find(entry, "shrsOrPrnAmt")
shares, shr_type = None, "SH"
if shr_node is not None:
try:
shares = float(_text(shr_node, "sshPrnamt"))
except (ValueError, TypeError):
pass
t = _text(shr_node, "sshPrnamtType")
if t:
shr_type = t
holdings.append({
"company": company,
"class_": class_,
"cusip": cusip,
"value": value_k * 1000.0 if value_k is not None else None,
"shares": shares,
"shr_type": shr_type,
"put_call": put_call,
})
return holdings
def fetch_hedge_fund_filings(
days_back: int = 90,
max_filers: int = 25,
progress_cb = None,
) -> list[dict]:
"""
Fetch individual stock holdings from recent SEC 13F-HR filings.
For each unique fund that filed a 13F in the given period (up to
max_filers), fetches the information-table XML and parses every holding.
Returns a flat list of holding dicts (one row per stock per fund).
days_back : calendar days back from today to search for filings.
max_filers : hard cap on the number of unique filers to process.
progress_cb : optional callable(done_count, total_count).
"""
import threading as _th
_UA = {"User-Agent": "UltimateInvestmentTool contact@investmenttool.com"}
# Rate-limiter: stay safely under EDGAR's 10 req/s cap
_rate_lock = _th.Lock()
_last_req = [0.0]
MIN_INTERVAL = 0.12
def _get(url, **kwargs):
with _rate_lock:
wait = MIN_INTERVAL - (time.monotonic() - _last_req[0])
if wait > 0:
time.sleep(wait)
_last_req[0] = time.monotonic()
return requests.get(url, headers=_UA, timeout=15, **kwargs)
from datetime import date, timedelta as _td
end_dt = date.today()
start_dt = end_dt - _td(days=max(days_back, 1))
# ── Step 1: Collect unique 13F-HR filings from EFTS ──────────────────
PAGE_SIZE = 100
seen_adsh : set = set()
unique_hits : list = []
from_offset = 0
while len(unique_hits) < max_filers:
try:
resp = _get(
"https://efts.sec.gov/LATEST/search-index",
params={
"q": "",
"forms": "13F-HR",
"dateRange": "custom",
"startdt": start_dt.strftime("%Y-%m-%d"),
"enddt": end_dt.strftime("%Y-%m-%d"),
"from": from_offset,
"hits": PAGE_SIZE,
},
)
resp.raise_for_status()
except Exception as _e:
print(f" [HF] EFTS search failed: {_e}")
break
page_hits = resp.json().get("hits", {}).get("hits", [])
for hit in page_hits:
adsh = hit.get("_source", {}).get("adsh", "")
if adsh and adsh not in seen_adsh:
seen_adsh.add(adsh)
unique_hits.append(hit)
if len(unique_hits) >= max_filers:
break
if len(page_hits) < PAGE_SIZE:
break
from_offset += PAGE_SIZE
if not unique_hits:
return []
# ── Step 2: Fetch each filing's info-table XML and parse holdings ─────
def _fetch_one(hit: dict) -> list[dict]:
try:
src = hit.get("_source", {})
acc_no = src.get("adsh", "")
ciks = src.get("ciks", [])
display = src.get("display_names") or src.get("entity_name") or []
fund_name = display[0].split(" (CIK")[0].strip() if display else "Unknown Fund"
filed = (src.get("file_date") or "")[:10]
period = (src.get("period_ending") or src.get("period_of_report") or "")[:10]
if not acc_no or not ciks:
return []
cik_int = int(ciks[0].lstrip("0") or "0")
if cik_int == 0:
return []
acc_nd = acc_no.replace("-", "")
base = (f"https://www.sec.gov/Archives/edgar/data"
f"/{cik_int}/{acc_nd}")
filing_url = f"{base}/{acc_no}-index.htm"
# Fetch the filing index page and locate the INFORMATION TABLE doc.
# Parse the document table properly rather than a fragile backwards search.
idx_r = _get(f"{base}/{acc_no}-index.htm")
if idx_r.status_code != 200:
print(f" [HF] {fund_name}: index fetch failed HTTP {idx_r.status_code}")
return []
idx_html = idx_r.text
# Walk every table row; find the one whose type cell is INFORMATION TABLE
info_href = None
for row_m in re.finditer(r'<tr[^>]*>(.*?)</tr>', idx_html,
re.IGNORECASE | re.DOTALL):
row_text = row_m.group(1)
if "INFORMATION TABLE" in row_text.upper():
href_m = re.search(r'href="(/Archives/[^"]+)"',
row_text, re.IGNORECASE)
if href_m:
info_href = href_m.group(1)
break
if not info_href:
print(f" [HF] {fund_name}: INFORMATION TABLE row not found in index")
return []
xml_r = _get("https://www.sec.gov" + info_href)
if xml_r.status_code != 200:
print(f" [HF] {fund_name}: XML fetch failed HTTP {xml_r.status_code}")
return []
raw_holdings = _parse_13f_xml(xml_r.content)
if not raw_holdings:
print(f" [HF] {fund_name}: XML parsed but 0 holdings returned")
return []
results: list[dict] = []
for h in raw_holdings:
results.append({
**h,
"fund_name": fund_name,
"filed_date": filed,
"period": period,
"url": filing_url,
})
return results
except Exception as exc:
print(f" [HF] {fund_name}: unexpected error — {exc}")
return []
total = len(unique_hits)
done_n = 0
results: list[dict] = []
with ThreadPoolExecutor(max_workers=3) as ex:
futs = {ex.submit(_fetch_one, h): h for h in unique_hits}
for fut in as_completed(futs):
done_n += 1
results.extend(fut.result() or [])
if progress_cb:
try:
progress_cb(done_n, total)
except Exception:
pass
results.sort(
key=lambda x: (x.get("filed_date") or "", x.get("value") or 0.0),
reverse=True,
)
return results
if __name__ == "__main__":
main()