Stock-Tool/dist/stock_screener.py

2087 lines
87 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""
Stock Evaluation Screener
=========================
Evaluates every common stock listed on NYSE, NASDAQ, and AMEX (~6 500+ tickers)
using a multi-factor composite scoring formula designed to act as a quantitative
analyst. This covers all major US indices: S&P 500/400/600, Russell 1000/2000/3000,
DJIA, NYSE Composite, NASDAQ Composite, and every sector/thematic index derived
from US-listed equities.
FORMULA (Composite Score):
Score = 0.18*Value + 0.18*Growth + 0.14*Momentum + 0.11*Quality
+ 0.08*Profitability + 0.13*Sentiment + 0.10*Analyst + 0.08*Risk
Each component is scored 0100 based on absolute benchmarks for that metric.
Scores are INDEPENDENT — each stock is evaluated on its own merits, not
relative to other stocks in the dataset.
Value — how attractively priced (P/E, P/B, EV/EBITDA vs benchmarks)
Growth — revenue, earnings, and EPS trajectory
Momentum — price return across 1M/3M/6M/12M windows
Quality — financial strength (ROE, ROA, debt, liquidity)
Profitability — margins and free cash flow yield
Sentiment — recent news tone via VADER NLP
Analyst — consensus rating and upside to price target
Risk — short interest, volatility, and beta stability
"""
import argparse
import html
import random
import re
import sys
import time
import warnings
from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import datetime, timedelta, timezone
import numpy as np
import pandas as pd
import requests
import yfinance as yf
warnings.filterwarnings("ignore")
# ---------------------------------------------------------------------------
# VADER sentiment — optional but strongly recommended
# pip install vaderSentiment
# ---------------------------------------------------------------------------
try:
from vaderSentiment.vaderSentiment import SentimentIntensityAnalyzer
_VADER = SentimentIntensityAnalyzer()
except ImportError:
_VADER = None
# ---------------------------------------------------------------------------
# Configuration
# ---------------------------------------------------------------------------
WEIGHTS = {
"value": 0.18,
"growth": 0.18,
"momentum": 0.14,
"quality": 0.11,
"profitability": 0.08,
"sentiment": 0.13,
"analyst": 0.10,
"risk": 0.08,
}
# Human-readable investment strategy presets.
# Each preset is a complete weight dict that must sum to 1.0.
STRATEGY_PRESETS = {
"Balanced (Default)": {
"value": 0.18, "growth": 0.18, "momentum": 0.14, "quality": 0.11,
"profitability": 0.08, "sentiment": 0.13, "analyst": 0.10, "risk": 0.08,
},
"High Growth Companies": {
"value": 0.08, "growth": 0.28, "momentum": 0.25, "quality": 0.08,
"profitability": 0.08, "sentiment": 0.12, "analyst": 0.07, "risk": 0.04,
},
"Conservative": {
"value": 0.20, "growth": 0.12, "momentum": 0.06, "quality": 0.25,
"profitability": 0.18, "sentiment": 0.06, "analyst": 0.08, "risk": 0.05,
},
"Recent Uptrends": {
"value": 0.05, "growth": 0.12, "momentum": 0.40, "quality": 0.05,
"profitability": 0.05, "sentiment": 0.18, "analyst": 0.10, "risk": 0.05,
},
"Buy and Hold Long Term": {
"value": 0.30, "growth": 0.22, "momentum": 0.05, "quality": 0.20,
"profitability": 0.15, "sentiment": 0.03, "analyst": 0.05, "risk": 0.00,
},
"Low Risk, High Dividend Yield": {
"value": 0.22, "growth": 0.10, "momentum": 0.05, "quality": 0.22,
"profitability": 0.28, "sentiment": 0.05, "analyst": 0.05, "risk": 0.03,
},
"High Risk High Reward": {
"value": 0.10, "growth": 0.00, "momentum": 0.00, "quality": 0.15,
"profitability": 0.00, "sentiment": 0.05, "analyst": 0.30, "risk": 0.05,
"reverse_momentum": 0.35,
},
}
HIST_BATCH = 100 # tickers per yf.download() batch
HIST_PAUSE = 3.0 # seconds between history batches
INFO_BATCH = 20 # tickers per info-fetch batch
INFO_PAUSE = 4.0 # seconds between info batches
BATCH_SIZE = INFO_BATCH
BATCH_PAUSE = INFO_PAUSE
# ---------------------------------------------------------------------------
# Ticker Fetching
# ---------------------------------------------------------------------------
# Valid common-stock ticker: 1-5 uppercase letters, optionally followed by
# a dash and 1-2 letters for share classes (e.g. BRK-A, BRK-B)
_TICKER_RE = re.compile(r'^[A-Z]{1,5}(-[A-Z]{1,2})?$')
def _clean_ticker(symbol: str) -> str | None:
"""Normalize a raw symbol and return it if it looks like a common stock."""
t = symbol.strip().upper().replace(".", "-")
return t if _TICKER_RE.match(t) else None
_NASDAQ100_FALLBACK = [
"AAPL", "MSFT", "NVDA", "AMZN", "META", "GOOGL", "GOOG", "TSLA", "AVGO", "COST",
"NFLX", "TMUS", "ASML", "AMD", "PEP", "LIN", "CSCO", "ADBE", "QCOM", "INTU",
"TXN", "AMAT", "AMGN", "CMCSA", "ISRG", "MU", "LRCX", "MRVL", "KLAC", "REGN",
"PANW", "CRWD", "SNPS", "CDNS", "ABNB", "ORLY", "MNST", "NXPI", "FTNT", "MAR",
"AEP", "PYPL", "MCHP", "PAYX", "ADSK", "CTAS", "IDXX", "ROST", "KDP", "FAST",
"ODFL", "DXCM", "VRSK", "CPRT", "BKR", "GEHC", "ON", "EXC", "LULU", "ZS",
"TEAM", "PCAR", "WDAY", "CTSH", "BIIB", "GILD", "CEG", "TTD", "MRNA", "EA",
"EBAY", "ILMN", "INTC", "SMCI", "ARM", "FANG", "WBD", "DLTR", "SIRI", "GFS",
"HON", "MSTR", "DASH", "RBLX", "COIN", "DDOG", "ZM", "OKTA", "SNOW", "NET",
"HUBS", "BILL", "CFLT", "MDB", "GTLB", "U", "APP", "PINS", "RIVN", "UBER",
]
def get_nasdaq100_tickers() -> list[str]:
"""Live NASDAQ-100 from Wikipedia; falls back to a hardcoded list."""
tickers = _fetch_wikipedia_index(
"https://en.wikipedia.org/wiki/Nasdaq-100", "NASDAQ 100")
if tickers:
return tickers
print(" [WARN] Using hardcoded NASDAQ-100 fallback list")
tickers = list(dict.fromkeys(_NASDAQ100_FALLBACK))
print(f" NASDAQ 100 : {len(tickers):>5} tickers (fallback)")
return tickers
_NASDAQ_HEADERS = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
"Accept": "application/json, text/plain, */*",
"Referer": "https://www.nasdaq.com/",
}
def _fetch_exchange(exchange: str) -> list[str]:
"""
Pull every listed stock for one exchange from the NASDAQ screener API.
Covers NYSE (~2 800 tickers), NASDAQ (~3 500), and AMEX (~300).
Together these three exchanges list every component of every major US index:
S&P 500/400/600, Russell 1000/2000/3000, DJIA, NYSE Composite, etc.
"""
url = (
"https://api.nasdaq.com/api/screener/stocks"
f"?tableonly=true&limit=10000&exchange={exchange}&download=true"
)
try:
resp = requests.get(url, headers=_NASDAQ_HEADERS, timeout=30)
resp.raise_for_status()
rows = resp.json()["data"]["table"]["rows"]
tickers = [_clean_ticker(r.get("symbol", "")) for r in rows]
tickers = [t for t in tickers if t]
print(f" {exchange:<8}: {len(tickers):>5} tickers")
return tickers
except Exception as e:
print(f" [WARN] Could not fetch {exchange} tickers: {e}")
return []
def _fetch_nasdaqtrader_listings(exchange_filter: str | None = None) -> list[str]:
"""
Download the official NASDAQ Trader exchange listing flat files.
These are publicly accessible static files updated daily — no API key,
no bot detection, no rate limiting.
exchange_filter: None = all exchanges, or "NASDAQ", "NYSE", "AMEX"
nasdaqlisted.txt — every NASDAQ-listed stock
cols: Symbol | Name | MarketCat | TestIssue | FinancialStatus | LotSize | ETF | NextShares
otherlisted.txt — every NYSE / AMEX / Arca / BATS / IEX listed stock
cols: ACTSymbol | Name | Exchange | CQSSymbol | ETF | LotSize | TestIssue | NASDAQSymbol
Exchange codes: N=NYSE A=AMEX/NYSE-American P=NYSE-Arca Z=BATS V=IEX
"""
from io import StringIO
base = "https://www.nasdaqtrader.com/dynamic/SymDir/"
tickers = []
# ── nasdaqlisted.txt ──────────────────────────────────────────────────
if exchange_filter in (None, "NASDAQ"):
try:
resp = requests.get(base + "nasdaqlisted.txt",
headers=_HTTP_HEADERS, timeout=20)
resp.raise_for_status()
count = 0
for line in resp.text.splitlines()[1:]:
if line.startswith("File Creation"):
break
parts = line.split("|")
if len(parts) < 7:
continue
test_issue = parts[3].strip() # col 3 = TestIssue
is_etf = parts[6].strip() # col 6 = ETF
if test_issue == "Y" or is_etf == "Y":
continue
t = _clean_ticker(parts[0].strip())
if t:
tickers.append(t)
count += 1
print(f" NASDAQ listed : {count:>5} tickers")
except Exception as e:
print(f" [WARN] nasdaqlisted.txt failed: {e}")
# ── otherlisted.txt ───────────────────────────────────────────────────
if exchange_filter in (None, "NYSE", "AMEX"):
# Map friendly name → exchange code(s) in the file
_code_map = {
"NYSE": {"N"},
"AMEX": {"A"},
None: {"N", "A", "P", "Z", "V"},
}
allowed_codes = _code_map.get(exchange_filter, {"N", "A", "P", "Z", "V"})
try:
resp = requests.get(base + "otherlisted.txt",
headers=_HTTP_HEADERS, timeout=20)
resp.raise_for_status()
count = 0
for line in resp.text.splitlines()[1:]:
if line.startswith("File Creation"):
break
parts = line.split("|")
if len(parts) < 7:
continue
exchange = parts[2].strip() # col 2 = Exchange code
is_etf = parts[4].strip() # col 4 = ETF
test_issue = parts[6].strip() # col 6 = TestIssue
if test_issue == "Y" or is_etf == "Y":
continue
if exchange not in allowed_codes:
continue
t = _clean_ticker(parts[0].strip())
if t:
tickers.append(t)
count += 1
lbl = exchange_filter or "other exchanges"
print(f" {lbl:<15}: {count:>5} tickers")
except Exception as e:
print(f" [WARN] otherlisted.txt failed: {e}")
return tickers
def _fetch_otc_tickers() -> list[str]:
"""
Fetch OTC-traded US company tickers from SEC EDGAR's exchange-tagged file.
company_tickers_exchange.json tags every SEC-registered company with its
exchange (Nasdaq, NYSE, OTC, etc.) — no bot detection, same source we
already trust for the fallback ticker list.
Filters to entries where exchange == "OTC", giving OTCQX / OTCQB /
Pink Sheet stocks not covered by the NASDAQ Trader exchange files.
"""
try:
resp = requests.get(
"https://www.sec.gov/files/company_tickers_exchange.json",
headers={"User-Agent": "stockscreener contact@example.com"},
timeout=20,
)
resp.raise_for_status()
payload = resp.json()
# Format: {"fields": ["cik","name","ticker","exchange"], "data": [[...], ...]}
fields = payload.get("fields", [])
rows = payload.get("data", [])
try:
exch_idx = fields.index("exchange")
ticker_idx = fields.index("ticker")
except ValueError:
print(" [WARN] SEC EDGAR exchange file format changed — no OTC tickers")
return []
tickers = [
_clean_ticker(str(row[ticker_idx]))
for row in rows
if str(row[exch_idx]).upper() == "OTC"
]
tickers = [t for t in tickers if t]
print(f" OTC (SEC EDGAR): {len(tickers):>5} tickers")
return tickers
except Exception as e:
print(f" [WARN] SEC EDGAR OTC fetch failed: {e}")
return []
def _fetch_sec_all_tickers() -> list[str]:
"""
Pull every US-listed company ticker from the SEC EDGAR company tickers file.
This is a publicly accessible, bot-friendly endpoint (~10 000+ tickers).
Used as the fallback when NASDAQ Trader files are unavailable.
"""
try:
resp = requests.get(
"https://www.sec.gov/files/company_tickers.json",
headers={"User-Agent": "stockscreener contact@example.com"},
timeout=20,
)
resp.raise_for_status()
data = resp.json()
tickers = [_clean_ticker(v.get("ticker", "")) for v in data.values()]
tickers = [t for t in tickers if t]
print(f" SEC EDGAR : {len(tickers):>5} tickers")
return tickers
except Exception as e:
print(f" [WARN] SEC EDGAR ticker fetch failed: {e}")
return []
_HTTP_HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
def _fetch_sp500() -> list[str]:
# Primary: Wikipedia — reflects additions/removals as soon as the page updates
tickers = _fetch_wikipedia_index(
"https://en.wikipedia.org/wiki/List_of_S%26P_500_companies", "S&P 500")
if tickers:
return tickers
# Fallback: community-maintained GitHub CSV
try:
from io import StringIO
url = ("https://raw.githubusercontent.com/datasets/"
"s-and-p-500-companies/main/data/constituents.csv")
resp = requests.get(url, headers=_HTTP_HEADERS, timeout=20)
resp.raise_for_status()
tickers = [_clean_ticker(t)
for t in pd.read_csv(StringIO(resp.text))["Symbol"]]
tickers = [t for t in tickers if t]
print(f" S&P 500 : {len(tickers):>5} tickers (GitHub fallback)")
return tickers
except Exception as e:
print(f" [WARN] S&P 500 GitHub fallback failed: {e}")
return []
def _fetch_wikipedia_index(url: str, label: str, min_count: int = 10) -> list[str]:
"""
Pull index constituents from a Wikipedia list-of-companies page.
Strips footnote markers (e.g. 'AAPL[a]') and skips tables that are
too small to be the real index table (avoids false positives).
"""
try:
from io import StringIO
resp = requests.get(url, headers=_HTTP_HEADERS, timeout=20)
resp.raise_for_status()
tables = pd.read_html(StringIO(resp.text))
for df in tables:
for col in ("Symbol", "Ticker", "Ticker symbol"):
if col in df.columns:
# Strip footnote brackets like [a], [1], [note 1]
cleaned = (df[col].astype(str)
.str.replace(r'\[.*?\]', '', regex=True)
.str.strip())
tickers = [_clean_ticker(t) for t in cleaned]
tickers = [t for t in tickers if t]
if len(tickers) < min_count:
continue # skip unrelated small tables
print(f" {label:<12}: {len(tickers):>5} tickers")
return tickers
except Exception as e:
print(f" [WARN] {label} failed: {e}")
return []
# iShares Russell ETF holdings CSVs — publicly downloadable
_RUSSELL_ETFS = {
"Russell 1000": ("239707", "ishares-russell-1000-etf", "IWB"),
"Russell 2000": ("239659", "ishares-russell-2000-etf", "IWM"),
"Russell 3000": ("239714", "ishares-russell-3000-etf", "IWV"),
}
def _fetch_russell(label: str) -> list[str]:
product_id, slug, etf = _RUSSELL_ETFS[label]
url = (
f"https://www.ishares.com/us/products/{product_id}/{slug}"
f"/1467271812596.ajax?fileType=csv&fileName={etf}_holdings&dataType=fund"
)
try:
from io import StringIO
resp = requests.get(url, headers=_HTTP_HEADERS, timeout=40)
resp.raise_for_status()
lines = resp.text.splitlines()
# iShares CSVs have a few metadata lines before the real header
header_idx = next(
(i for i, ln in enumerate(lines)
if ln.startswith("Ticker,") or ",Ticker," in ln or ln.startswith("Name,")),
None,
)
if header_idx is None:
raise ValueError("Header row not found in iShares CSV")
df = pd.read_csv(StringIO("\n".join(lines[header_idx:])))
col = next((c for c in df.columns if c.strip() == "Ticker"), None)
if not col:
raise ValueError("'Ticker' column missing")
tickers = [_clean_ticker(str(t)) for t in df[col]]
tickers = [t for t in tickers if t and t.upper() not in ("-", "NAN", "CASH")]
print(f" {label:<12}: {len(tickers):>5} tickers")
return tickers
except Exception as e:
print(f" [WARN] {label} iShares CSV failed: {e}")
return []
_DJIA_FALLBACK = [
"AAPL", "AMGN", "AMZN", "AXP", "BA", "CAT", "CRM", "CSCO", "CVX", "DIS",
"DOW", "GS", "HD", "HON", "IBM", "JNJ", "JPM", "KO", "MCD", "MMM",
"MRK", "MSFT", "NKE", "PG", "SHW", "TRV", "UNH", "V", "VZ", "WMT",
]
def get_djia_tickers() -> list[str]:
"""Live DJIA components from Wikipedia; falls back to a hardcoded list."""
tickers = _fetch_wikipedia_index(
"https://en.wikipedia.org/wiki/Dow_Jones_Industrial_Average", "DJIA")
if tickers:
return tickers
print(" [WARN] Using hardcoded DJIA fallback list")
tickers = list(dict.fromkeys(_DJIA_FALLBACK))
print(f" DJIA : {len(tickers):>5} tickers (fallback)")
return tickers
# Human-readable index names exposed to the GUI
INDEX_CHOICES = [
"All",
"All + OTC",
"S&P 500", "S&P 400", "S&P 600", "S&P 1500",
"Russell 1000", "Russell 2000", "Russell 3000",
"NASDAQ 100", "NASDAQ", "NYSE", "AMEX",
"DJIA",
]
def collect_tickers(index: str = "All") -> list[str]:
"""
Fetch the constituent tickers for the requested index.
"All" — every common stock on all US exchanges via NASDAQ Trader files
"All + OTC" — exchange-listed + OTC/Pink Sheet stocks
"S&P 500/400/600/1500" — via Wikipedia / GitHub CSV
"Russell 1000/2000/3000" — via iShares ETF holdings CSVs
"NASDAQ 100" — live Wikipedia list with hardcoded fallback
"NASDAQ/NYSE/AMEX" — per-exchange listings via NASDAQ Trader files
"DJIA" — 30 components from Wikipedia with hardcoded fallback
Fallback chain for "All" and single-exchange lookups:
NASDAQ Trader files → NASDAQ screener API → SEC EDGAR → Russell 3000 → S&P 500
"""
print(f"Fetching tickers for index: {index} ...")
sp400_url = "https://en.wikipedia.org/wiki/List_of_S%26P_400_companies"
sp600_url = "https://en.wikipedia.org/wiki/List_of_S%26P_600_companies"
def _all_exchange_tickers() -> list[str]:
"""All exchange-listed US common stocks with full fallback chain."""
# 1. NASDAQ Trader official listing files (most reliable — static, no bot detection)
t = _fetch_nasdaqtrader_listings(exchange_filter=None)
if t:
return t
# 2. NASDAQ screener API (often blocked but try anyway)
print(" [WARN] NASDAQ Trader files failed — trying NASDAQ screener API")
t = []
for exch in ("NYSE", "NASDAQ", "AMEX"):
t.extend(_fetch_exchange(exch))
if t:
return t
# 3. SEC EDGAR company registry (~10 000+ tickers)
print(" [WARN] NASDAQ API failed — trying SEC EDGAR")
t = _fetch_sec_all_tickers()
if t:
return t
# 4. Russell 3000
print(" [WARN] SEC EDGAR failed — trying Russell 3000")
t = _fetch_russell("Russell 3000")
if t:
return t
# 5. S&P 1500
print(" [WARN] Russell 3000 failed — trying S&P 1500")
t = (_fetch_sp500()
+ _fetch_wikipedia_index(sp400_url, "S&P 400")
+ _fetch_wikipedia_index(sp600_url, "S&P 600"))
if t:
return t
# 6. S&P 500 last resort
print(" [WARN] All broad sources failed — falling back to S&P 500")
return _fetch_sp500()
if index == "All":
tickers = _all_exchange_tickers()
elif index == "All + OTC":
tickers = _all_exchange_tickers()
otc = _fetch_otc_tickers()
if otc:
tickers = tickers + otc
else:
print(" [WARN] OTC Markets unavailable — showing exchange-listed only")
elif index == "S&P 500":
tickers = _fetch_sp500()
elif index == "S&P 400":
tickers = _fetch_wikipedia_index(sp400_url, "S&P 400")
elif index == "S&P 600":
tickers = _fetch_wikipedia_index(sp600_url, "S&P 600")
elif index == "S&P 1500":
tickers = (_fetch_sp500()
+ _fetch_wikipedia_index(sp400_url, "S&P 400")
+ _fetch_wikipedia_index(sp600_url, "S&P 600"))
elif index in ("Russell 1000", "Russell 2000", "Russell 3000"):
tickers = _fetch_russell(index)
if not tickers:
print(f" [WARN] iShares CSV failed — using full exchange listing for {index}")
tickers = _all_exchange_tickers()
elif index == "NASDAQ 100":
tickers = get_nasdaq100_tickers()
elif index == "NASDAQ":
tickers = _fetch_nasdaqtrader_listings(exchange_filter="NASDAQ")
if not tickers:
tickers = _fetch_exchange("NASDAQ")
elif index == "NYSE":
tickers = _fetch_nasdaqtrader_listings(exchange_filter="NYSE")
if not tickers:
tickers = _fetch_exchange("NYSE")
elif index == "AMEX":
tickers = _fetch_nasdaqtrader_listings(exchange_filter="AMEX")
if not tickers:
tickers = _fetch_exchange("AMEX")
elif index == "DJIA":
tickers = get_djia_tickers()
else:
print(f" [WARN] Unknown index '{index}' — defaulting to All")
tickers = _all_exchange_tickers()
combined = list(dict.fromkeys(tickers)) # deduplicate, preserve order
print(f" Total : {len(combined):>5} unique tickers\n")
return combined
# ---------------------------------------------------------------------------
# News Sentiment (multi-source VADER)
# ---------------------------------------------------------------------------
# Sources pulled in order:
# 1. Yahoo Finance — via yfinance Ticker.news (built-in, no key)
# 2. Google News RSS — public RSS feed, no API key required
# ---------------------------------------------------------------------------
def _extract_yfinance_news(ticker_obj) -> list[tuple[str, float, str, str]]:
"""Return (title, timestamp, url, summary) tuples from yfinance Ticker.news."""
results = []
try:
news = ticker_obj.news or []
now = time.time()
for item in news[:30]:
if not isinstance(item, dict):
continue
title = ts = url = summary = None
content = item.get("content", {})
if isinstance(content, dict) and content.get("title"):
title = content["title"]
pub = content.get("pubDate", "")
try:
ts = datetime.fromisoformat(
pub.replace("Z", "+00:00")).timestamp()
except Exception:
ts = now
# Try several URL fields used across yfinance versions
canon = content.get("canonicalUrl") or {}
url = (canon.get("url") if isinstance(canon, dict) else None) \
or content.get("url") or content.get("clickThroughUrl", {}).get("url") \
or item.get("link")
summary = content.get("summary") or content.get("description") or ""
elif item.get("title"):
title = item["title"]
ts = item.get("providerPublishTime", now)
url = item.get("link") or item.get("url") or ""
summary = item.get("summary") or item.get("description") or ""
if title:
clean = re.sub(r'\s+', ' ',
html.unescape(re.sub(r'<[^>]+>', ' ', summary or ""))).strip()
results.append((title, float(ts or now), url or "", clean))
except Exception:
pass
return results
def _extract_google_news_rss(ticker: str, company_name: str = "") -> list[tuple[str, float, str, str]]:
"""
Fetch headlines from Google News RSS for the given ticker.
Returns (title, timestamp, url, summary) tuples.
No API key required.
"""
import re as _re
import xml.etree.ElementTree as ET
import urllib.parse
results = []
try:
query = f"{ticker} stock"
if company_name:
query += f" {company_name}"
encoded = urllib.parse.quote(query)
rss_url = (
f"https://news.google.com/rss/search"
f"?q={encoded}&hl=en-US&gl=US&ceid=US:en"
)
headers = {"User-Agent": "Mozilla/5.0 (compatible; StockScreener/1.0)"}
resp = requests.get(rss_url, headers=headers, timeout=8)
resp.raise_for_status()
root = ET.fromstring(resp.content)
now = time.time()
for item in root.findall(".//item")[:30]:
title_el = item.find("title")
pub_el = item.find("pubDate")
desc_el = item.find("description")
if title_el is None or not title_el.text:
continue
title = title_el.text.strip()
ts = now
if pub_el is not None and pub_el.text:
try:
from email.utils import parsedate_to_datetime
ts = parsedate_to_datetime(pub_el.text).timestamp()
except Exception:
pass
# guid is more reliably a plain URL in Google News RSS than <link>
guid_el = item.find("guid")
link = (guid_el.text.strip() if guid_el is not None and guid_el.text else "") \
or (item.findtext("link") or "")
# Strip HTML tags and unescape entities from description snippet
raw_desc = desc_el.text if desc_el is not None and desc_el.text else ""
summary = _re.sub(r'\s+', ' ',
html.unescape(_re.sub(r"<[^>]+>", " ", raw_desc))).strip()
results.append((title, ts, link, summary))
except Exception:
pass
return results
def get_news_headlines(ticker_obj, ticker: str = "", company_name: str = "") \
-> tuple:
"""
Aggregate headlines from Yahoo Finance and Google News RSS and compute
recency-weighted VADER sentiment.
Returns (aggregate_score, top_headlines) where:
- aggregate_score is float in [-1.0, +1.0] or None
- top_headlines is a list of (title, vader_score, age_days) tuples,
sorted by impact (abs(score) × recency weight), up to 8 entries.
"""
if _VADER is None:
return None, []
all_items: list[tuple[str, float, str, str]] = []
all_items.extend(_extract_yfinance_news(ticker_obj))
if ticker:
all_items.extend(_extract_google_news_rss(ticker, company_name))
if not all_items:
return None, []
# Deduplicate by normalised title
seen: set[str] = set()
unique: list[tuple[str, float, str, str]] = []
for title, ts, url, summary in all_items:
key = title.lower().strip()[:80]
if key not in seen:
seen.add(key)
unique.append((title, ts, url, summary))
now = time.time()
total_w = total_s = 0.0
# (title, score, age_days, weight, url, summary)
scored: list[tuple[str, float, float, float, str, str]] = []
for title, ts, url, summary in unique:
age_days = max(0.0, (now - ts) / 86400)
if age_days <= 7:
weight = 1.0
else:
weight = max(0.05, 1.0 - (age_days - 7) / 23.0)
score = _VADER.polarity_scores(title)["compound"]
total_s += score * weight
total_w += weight
scored.append((title, score, age_days, weight, url, summary))
aggregate = total_s / total_w if total_w else None
# Top headlines by impact: abs(score) × recency weight
scored.sort(key=lambda x: abs(x[1]) * x[3], reverse=True)
top = [(title, vader_score, age_days, url, summary)
for title, vader_score, age_days, _, url, summary in scored[:8]]
return aggregate, top
def _news_sentiment(ticker_obj, ticker: str = "", company_name: str = "") -> float | None:
"""Backward-compatible wrapper — returns only the aggregate score."""
score, _ = get_news_headlines(ticker_obj, ticker=ticker, company_name=company_name)
return score
# ---------------------------------------------------------------------------
# EDGAR Bulk Fundamentals
# ---------------------------------------------------------------------------
# Fetches financial statement data for all US public companies from the
# SEC EDGAR XBRL frames API. One HTTP call per financial concept returns
# values for ALL filers simultaneously — ~50 calls cover every metric
# needed for the scoring model in roughly 2 minutes.
#
# Coverage: ~8590% of exchange-listed US stocks. Tickers with no EDGAR
# coverage still appear in results and score on price/momentum/analyst
# data only; their fundamental fields default to NaN.
# ---------------------------------------------------------------------------
_EDGAR_FRAMES = "https://data.sec.gov/api/xbrl/frames"
_EDGAR_TICKERS = "https://www.sec.gov/files/company_tickers.json"
_EDGAR_UA = {"User-Agent": "StockScreener contact@investmenttool.com"}
_FRAME_CACHE: dict = {} # in-memory cache so the same frame is never fetched twice
def _get_recent_quarters(n: int = 9) -> list[tuple[int, int]]:
"""
Return the last n (year, quarter) tuples, most-recent first, using a
45-day filing lag so we only request periods most companies have filed.
"""
from datetime import date, timedelta as _td
ref = date.today() - _td(days=45)
y, q = ref.year, (ref.month - 1) // 3 + 1
out = []
for _ in range(n):
out.append((y, q))
q -= 1
if q == 0:
q, y = 4, y - 1
return out
def _fetch_one_frame(concept: str, unit: str, period: str) -> dict[int, float]:
"""
Fetch all companies' values for one EDGAR XBRL concept/period.
Returns {cik (int): value (float)}. Results are cached in _FRAME_CACHE
so the same concept/period is never downloaded twice in one run.
"""
key = (concept, unit, period)
if key in _FRAME_CACHE:
return _FRAME_CACHE[key]
url = f"{_EDGAR_FRAMES}/us-gaap/{concept}/{unit}/{period}.json"
try:
resp = requests.get(url, headers=_EDGAR_UA, timeout=20)
if resp.status_code == 404:
_FRAME_CACHE[key] = {}
return {}
resp.raise_for_status()
result: dict[int, float] = {}
for row in resp.json().get("data", []):
# The EDGAR frames API returns each row as a dict with keys:
# accn, cik, entityName, loc, start, end, val
# (Not as a list — handle both formats for safety.)
try:
if isinstance(row, dict):
cik = int(row["cik"])
val = float(row["val"])
else:
# Legacy/alternative format: [accn, cik, entityName, loc, end, val]
if len(row) < 6:
continue
cik = int(row[1])
val = float(row[5])
result[cik] = val # last entry wins (restatements)
except (KeyError, TypeError, ValueError):
continue
_FRAME_CACHE[key] = result
time.sleep(0.15) # stay comfortably under SEC's 10 req/sec limit
return result
except Exception:
_FRAME_CACHE[key] = {}
return {}
def _sum_frames(concepts: list[str], unit: str,
periods: list[str]) -> dict[int, float]:
"""
Sum values across periods (TTM) trying each concept in order.
The primary concept is tried first; gaps are filled by fallback concepts.
"""
ttm: dict[int, dict] = {} # {cik: {period: value}}
for concept in concepts:
for period in periods:
for cik, val in _fetch_one_frame(concept, unit, period).items():
ttm.setdefault(cik, {})
if period not in ttm[cik]:
ttm[cik][period] = val
return {cik: sum(pv.values()) for cik, pv in ttm.items()}
def _best_frame(concepts: list[str], unit: str,
periods: list[str]) -> dict[int, float]:
"""
Return the most-recent available value per CIK (for balance-sheet items).
Iterates periods most-recent first; first value found for each CIK wins.
"""
best: dict[int, float] = {}
for concept in concepts:
for period in periods:
for cik, val in _fetch_one_frame(concept, unit, period).items():
if cik not in best:
best[cik] = val
return best
def _build_cik_maps() -> tuple[dict[int, str], dict[int, str]]:
"""
Fetch SEC company_tickers.json and return:
cik_to_ticker — {cik_int: "AAPL"}
cik_to_name — {cik_int: "Apple Inc."}
"""
try:
resp = requests.get(_EDGAR_TICKERS, headers=_EDGAR_UA, timeout=20)
resp.raise_for_status()
data = resp.json()
return (
{int(v["cik_str"]): v["ticker"].upper() for v in data.values()},
{int(v["cik_str"]): v["title"] for v in data.values()},
)
except Exception as e:
print(f" [WARN] CIK map unavailable: {e}")
return {}, {}
def _fetch_edgar_bulk(close_map: dict[str, pd.Series]) -> dict[str, dict]:
"""
Fetch fundamental data for all tickers from the EDGAR XBRL frames API.
Returns {ticker: {field: value}} with keys matching the rest of the
scoring pipeline. Takes ~2 minutes for the full US universe.
Fields returned (NaN-safe — missing values simply absent from the dict):
market_cap, pe_trailing, pb_ratio, ev_ebitda, eps_trailing,
revenue, revenue_growth, earnings_growth, roe, roa,
debt_to_equity, total_debt, total_cash, book_value, current_ratio,
profit_margin, operating_margin, fcf_yield, name
"""
print("Phase 2A: Fetching EDGAR bulk fundamentals ...")
cik_to_ticker, cik_to_name = _build_cik_maps()
if not cik_to_ticker:
print(" [WARN] CIK map unavailable — EDGAR bulk skipped")
return {}
quarters = _get_recent_quarters(9)
ttm_periods = [f"CY{y}Q{q}" for y, q in quarters[:4]]
prev_periods= [f"CY{y}Q{q}" for y, q in quarters[4:8]]
bs_periods = [f"CY{y}Q{q}I" for y, q in quarters[:3]]
print(f" TTM : {ttm_periods}")
print(f" BS : {bs_periods[0]} (+{len(bs_periods)-1} fallback periods)")
# ── Balance sheet (most recent available quarter) ──────────────────────
assets = _best_frame(["Assets"], "USD", bs_periods)
equity = _best_frame(["StockholdersEquity",
"StockholdersEquityIncludingPortionAttributable"
"ToNoncontrollingInterest"], "USD", bs_periods)
cur_assets = _best_frame(["AssetsCurrent"], "USD", bs_periods)
cur_liab = _best_frame(["LiabilitiesCurrent"], "USD", bs_periods)
cash = _best_frame(["CashAndCashEquivalentsAtCarryingValue",
"CashCashEquivalentsAndShortTermInvestments"], "USD", bs_periods)
lt_debt = _best_frame(["LongTermDebt", "LongTermDebtNoncurrent",
"LongTermDebtAndCapitalLeaseObligations"], "USD", bs_periods)
shares = _best_frame(["CommonStockSharesOutstanding"], "shares", bs_periods)
# ── Income statement TTM ────────────────────────────────────────────────
revenue = _sum_frames(["Revenues",
"RevenueFromContractWithCustomer"
"ExcludingAssessedTax",
"SalesRevenueNet"], "USD", ttm_periods)
net_income = _sum_frames(["NetIncomeLoss"], "USD", ttm_periods)
op_income = _sum_frames(["OperatingIncomeLoss"], "USD", ttm_periods)
dna = _sum_frames(["DepreciationDepletionAndAmortization",
"DepreciationAndAmortization"], "USD", ttm_periods)
# ── Cash flow TTM ───────────────────────────────────────────────────────
op_cf = _sum_frames(["NetCashProvidedByUsedInOperatingActivities"], "USD", ttm_periods)
capex = _sum_frames(["PaymentsToAcquirePropertyPlantAndEquipment"], "USD", ttm_periods)
# ── Prior-year TTM for growth rates ────────────────────────────────────
rev_prev = _sum_frames(["Revenues",
"RevenueFromContractWithCustomer"
"ExcludingAssessedTax",
"SalesRevenueNet"], "USD", prev_periods)
ni_prev = _sum_frames(["NetIncomeLoss"], "USD", prev_periods)
print(f" Revenue data : {len(revenue)} companies")
# ── Assemble per-ticker records ─────────────────────────────────────────
valid_set = set(close_map.keys())
result: dict[str, dict] = {}
for cik, ticker in cik_to_ticker.items():
if ticker not in valid_set:
continue
price = float(close_map[ticker].iloc[-1])
rev = revenue.get(cik); ni = net_income.get(cik)
op_i = op_income.get(cik); d_a = dna.get(cik)
ocf = op_cf.get(cik); cx = capex.get(cik)
tot_a = assets.get(cik); eq = equity.get(cik)
ca = cur_assets.get(cik); cl = cur_liab.get(cik)
csh = cash.get(cik); ltd = lt_debt.get(cik)
sh = shares.get(cik)
r_p = rev_prev.get(cik); n_p = ni_prev.get(cik)
mkt_cap = (price * sh) if sh and sh > 0 else None
eps_trail = (ni / sh) if ni is not None and sh and sh > 0 else None
pe_trail = (price / eps_trail) if eps_trail and eps_trail > 0 else None
pb = (price / (eq / sh)) if eq and sh and sh > 0 and eq > 0 else None
ebitda = ((op_i + d_a) if op_i is not None and d_a is not None else op_i)
ev = ((mkt_cap + (ltd or 0) - (csh or 0)) if mkt_cap is not None else None)
ev_ebitda = (ev / ebitda) if ev and ebitda and ebitda > 0 else None
roe = (ni / eq) if ni is not None and eq and eq > 0 else None
roa = (ni / tot_a) if ni is not None and tot_a and tot_a > 0 else None
de = (ltd / eq) if ltd is not None and eq and eq > 0 else None
curr_r = (ca / cl) if ca and cl and cl > 0 else None
pm = (ni / rev) if ni is not None and rev and rev > 0 else None
om = (op_i / rev) if op_i is not None and rev and rev > 0 else None
fcf = ((ocf - cx) if ocf is not None and cx is not None else ocf)
fcf_yield = (fcf / mkt_cap) if fcf is not None and mkt_cap and mkt_cap > 0 else None
rev_g = ((rev - r_p) / abs(r_p) if rev is not None and r_p and r_p != 0 else None)
ni_g = ((ni - n_p) / abs(n_p) if ni is not None and n_p and n_p != 0 else None)
result[ticker] = {
"name": cik_to_name.get(cik, ticker),
"sector": "N/A",
"industry": "N/A",
"market_cap": mkt_cap,
"pe_trailing": pe_trail,
"pb_ratio": pb,
"ev_ebitda": ev_ebitda,
"eps_trailing": eps_trail,
"revenue": rev,
"revenue_growth": rev_g,
"earnings_growth": ni_g,
"roe": roe,
"roa": roa,
"debt_to_equity": de,
"total_debt": ltd,
"total_cash": csh,
"book_value": eq,
"current_ratio": curr_r,
"profit_margin": pm,
"operating_margin": om,
"fcf_yield": fcf_yield,
}
print(f" EDGAR matched : {len(result)} tickers\n")
return result
# ---------------------------------------------------------------------------
# Analyst + News Cache (Supabase)
# ---------------------------------------------------------------------------
try:
from license_check import _SUPABASE_URL as _SB_URL, _SUPABASE_ANON as _SB_ANON
except ImportError:
_SB_URL = _SB_ANON = None
def _load_analyst_cache() -> dict[str, dict]:
"""
Load analyst consensus and news sentiment from the Supabase cache table.
Returns {ticker: {field: value}} or {} on any failure.
Warns (but still proceeds) if cache data is more than 48 hours old.
"""
if not _SB_URL or not _SB_ANON:
print(" [WARN] Supabase config unavailable — analyst cache skipped")
return {}
try:
resp = requests.get(
f"{_SB_URL}/rest/v1/analyst_cache"
"?select=ticker,pe_forward,eps_forward,analyst_norm,analyst_upside,"
"analyst_count,analyst_target,recommendation,news_sentiment,"
"sector,industry,updated_at"
"&limit=20000",
headers={"apikey": _SB_ANON, "Content-Type": "application/json"},
timeout=20,
)
resp.raise_for_status()
rows = resp.json()
if not rows:
print(" [WARN] Analyst cache is empty — analyst scores will be neutral")
return {}
valid_ts = [r["updated_at"] for r in rows if r.get("updated_at")]
if valid_ts:
age_h = (
datetime.now(timezone.utc)
- datetime.fromisoformat(max(valid_ts).replace("Z", "+00:00"))
).total_seconds() / 3600
if age_h > 48:
print(f" [WARN] Analyst cache is {age_h:.0f}h old "
"(refresh runs tonight)")
else:
print(f" Analyst cache : {len(rows)} tickers "
f"(updated {age_h:.1f}h ago)")
return {r["ticker"]: r for r in rows}
except Exception as e:
print(f" [WARN] Could not load analyst cache: {e}")
return {}
# ---------------------------------------------------------------------------
# FINRA Short Interest
# ---------------------------------------------------------------------------
def _fetch_finra_short() -> dict[str, float]:
"""
Download the FINRA RegSHO daily short-volume file.
Returns {ticker: short_volume_ratio} where ratio = short_vol / total_vol.
This is a short-volume ratio, not short % of float, but it is a reliable
intraday bearish-pressure signal and a suitable proxy for the Risk score.
Files are published each trading day at:
https://cdn.finra.org/equity/regsho/daily/
"""
from datetime import date as _date, timedelta as _td
for days_back in range(1, 6):
d = _date.today() - _td(days=days_back)
if d.weekday() >= 5: # skip Saturday / Sunday
continue
url = (
"https://cdn.finra.org/equity/regsho/daily/"
f"CNMSshvol{d.strftime('%Y%m%d')}.txt"
)
try:
resp = requests.get(url, headers=_HTTP_HEADERS, timeout=15)
if resp.status_code == 404:
continue
resp.raise_for_status()
result: dict[str, float] = {}
for line in resp.text.splitlines()[1:]:
parts = line.split("|")
if len(parts) < 4:
continue
try:
sv = float(parts[1]); tv = float(parts[3])
if tv > 0:
result[parts[0].strip()] = sv / tv
except (ValueError, IndexError):
continue
if result:
print(f" FINRA short : {len(result)} tickers ({d})")
return result
except Exception:
continue
print(" [WARN] FINRA short interest unavailable")
return {}
# ---------------------------------------------------------------------------
# Beta Calculation
# ---------------------------------------------------------------------------
def _calc_beta_map(close_map: dict[str, pd.Series],
start_date: str) -> dict[str, float]:
"""
Calculate 1-year beta vs SPY for every ticker in close_map.
Downloads SPY history once, then computes cov(stock, SPY) / var(SPY)
in pandas — pure vectorised math, takes ~1030 seconds for 6,500 tickers.
"""
try:
spy_raw = yf.download("SPY", start=start_date,
auto_adjust=True, progress=False)
if spy_raw is None or spy_raw.empty:
return {}
spy_col = next(
(c for c in ("Close", "Adj Close") if c in spy_raw.columns), None
)
if not spy_col:
return {}
spy_close = spy_raw[spy_col]
# Newer yfinance returns MultiIndex columns even for single tickers —
# squeeze a single-column DataFrame to a Series.
if isinstance(spy_close, pd.DataFrame):
spy_close = spy_close.squeeze(axis=1)
spy_ret = spy_close.dropna().pct_change().dropna()
beta_map: dict[str, float] = {}
for ticker, close in close_map.items():
try:
s_ret = close.pct_change().dropna()
common = s_ret.index.intersection(spy_ret.index)
if len(common) < 50:
continue
m = spy_ret.loc[common]
var_m = float(m.var())
if var_m > 0:
beta_map[ticker] = float(s_ret.loc[common].cov(m) / var_m)
except Exception:
continue
print(f" Beta : {len(beta_map)} tickers calculated")
return beta_map
except Exception as e:
print(f" [WARN] Beta calculation failed: {e}")
return {}
# ---------------------------------------------------------------------------
# Data Fetching
# ---------------------------------------------------------------------------
# Sentinel so the rest of the file still compiles if something references these
_yf_session = None
_yf_crumb = None
def _download_history_batch(tickers: list[str], start: str) -> dict[str, pd.Series]:
"""
Download price history for a batch of tickers via yf.download().
yfinance manages cookies and crumb internally — no custom auth needed.
Returns {ticker: Close Series} for tickers with ≥ 20 trading days.
"""
if not tickers:
return {}
result: dict[str, pd.Series] = {}
try:
# yf.download handles auth/crumb internally
raw = yf.download(
tickers if len(tickers) > 1 else tickers[0],
start=start,
auto_adjust=True,
progress=False,
)
if raw is None or raw.empty:
return result
if len(tickers) == 1:
# Single ticker — newer yfinance returns MultiIndex columns even for
# a single ticker, so raw["Close"] may be a single-column DataFrame.
# Squeeze it to a Series so downstream float(close.iloc[-1]) works.
t = tickers[0]
col = next((c for c in ("Close", "Adj Close") if c in raw.columns), None)
if col:
s = raw[col]
if isinstance(s, pd.DataFrame):
s = s.squeeze(axis=1) # DataFrame → Series
s = s.dropna()
if len(s) >= 20:
result[t] = s
else:
# Multiple tickers → MultiIndex columns
cols = raw.columns
if isinstance(cols, pd.MultiIndex):
lvl0 = cols.get_level_values(0)
if "Close" in lvl0:
close_df = raw["Close"]
for t in tickers:
if t in close_df.columns:
s = close_df[t].dropna()
if len(s) >= 20:
result[t] = s
else:
# (ticker, field) layout used by some yfinance versions
for t in tickers:
if t in lvl0:
try:
sub = raw[t]
col = next((c for c in ("Close", "Adj Close")
if c in sub.columns), None)
if col:
s = sub[col].dropna()
if len(s) >= 20:
result[t] = s
except (KeyError, TypeError):
pass
except Exception as e:
print(f"\n [WARN] yf.download batch failed: {e}")
return result
def _fetch_fundamentals(
ticker: str,
close: pd.Series,
sector_filter: tuple | None = None,
mktcap_range: tuple | None = None,
) -> dict | None:
"""
Fetch fundamental/analyst/sentiment data for one ticker via yf.Ticker.
yfinance handles all Yahoo Finance auth internally.
Price history is supplied from Phase 1 so we only need the info call.
sector_filter: (column, value) tuple — e.g. ("sector", "Technology") or
("industry", "Aerospace & Defense"). None = no filter.
mktcap_range: (min, max) in dollars, either bound may be None.
"""
# Small random delay to spread concurrent worker requests
time.sleep(random.uniform(0.0, 1.0))
try:
stock = yf.Ticker(ticker)
info = stock.info
if not info or not isinstance(info, dict) or len(info) < 3:
return None
# ── Screen-time sector filter ──────────────────────────────────────
if sector_filter is not None:
col, val = sector_filter
actual = info.get("sector") if col == "sector" else info.get("industry")
if actual != val:
return None
# ── Screen-time market-cap filter ──────────────────────────────────
if mktcap_range is not None:
try:
mkt = float(info.get("marketCap") or 0)
except (TypeError, ValueError):
mkt = 0.0
lo, hi = mktcap_range
if lo is not None and mkt < lo:
return None
if hi is not None and mkt >= hi:
return None
price_now = float(close.iloc[-1])
def pct_return(days_back: int) -> float | None:
idx = max(0, len(close) - days_back)
past = float(close.iloc[idx])
return (price_now - past) / past if past > 0 else None
returns_30d = close.iloc[-30:].pct_change().dropna()
volatility = float(returns_30d.std() * (252 ** 0.5)) if len(returns_30d) >= 5 else None
try:
rec_mean = float(info.get("recommendationMean")) if info.get("recommendationMean") is not None else None
except (TypeError, ValueError):
rec_mean = None
analyst_norm = (5.0 - rec_mean) / 4.0 if rec_mean is not None and 1 <= rec_mean <= 5 else None
try:
target_price = float(info.get("targetMeanPrice")) if info.get("targetMeanPrice") is not None else None
except (TypeError, ValueError):
target_price = None
analyst_upside = (target_price / price_now - 1.0
if target_price is not None and price_now > 0 else None)
try:
fcf = float(info.get("freeCashflow")) if info.get("freeCashflow") is not None else None
except (TypeError, ValueError):
fcf = None
try:
mkt = float(info.get("marketCap")) if info.get("marketCap") is not None else None
except (TypeError, ValueError):
mkt = None
fcf_yield = (fcf / mkt) if fcf is not None and mkt is not None and mkt > 0 else None
company_name = info.get("longName", "")
news_sent, news_headlines = get_news_headlines(stock, ticker=ticker, company_name=company_name)
return {
"ticker": ticker,
"name": info.get("longName", ticker),
"sector": info.get("sector", "N/A"),
"industry": info.get("industry", "N/A"),
"price": price_now,
"pe_forward": info.get("forwardPE"),
"pe_trailing": info.get("trailingPE"),
"pb_ratio": info.get("priceToBook"),
"ev_ebitda": info.get("enterpriseToEbitda"),
"revenue_growth": info.get("revenueGrowth"),
"earnings_growth": info.get("earningsGrowth"),
"eps_forward": info.get("forwardEps"),
"eps_trailing": info.get("trailingEps"),
"ret_1m": pct_return(21),
"ret_3m": pct_return(63),
"ret_6m": pct_return(126),
"ret_12m": pct_return(252),
"roe": info.get("returnOnEquity"),
"roa": info.get("returnOnAssets"),
"debt_to_equity": info.get("debtToEquity"),
"total_debt": info.get("totalDebt"),
"total_cash": info.get("totalCash"),
"book_value": info.get("bookValue"),
"current_ratio": info.get("currentRatio"),
"profit_margin": info.get("profitMargins"),
"operating_margin": info.get("operatingMargins"),
"fcf_yield": fcf_yield,
"news_sentiment": news_sent,
"news_headlines": news_headlines,
"analyst_norm": analyst_norm,
"analyst_upside": analyst_upside,
"analyst_count": info.get("numberOfAnalystOpinions"),
"analyst_target": target_price,
"recommendation": info.get("recommendationKey", "N/A"),
"short_percent": info.get("shortPercentOfFloat"),
"beta": info.get("beta"),
"volatility_30d": volatility,
"market_cap": info.get("marketCap"),
"52w_high": info.get("fiftyTwoWeekHigh"),
"52w_low": info.get("fiftyTwoWeekLow"),
"dividend_yield": info.get("dividendYield"),
"revenue": info.get("totalRevenue"),
"business_summary": info.get("longBusinessSummary", "") or "",
}
except Exception:
return None
def fetch_all(
tickers: list[str],
max_workers: int = 5,
screen_filters: dict | None = None,
) -> pd.DataFrame:
"""
Multi-source data fetch replacing the old per-ticker yf.Ticker.info loop.
Phase 1 — yf.download() batch price history (unchanged, ~7 min)
Phase 2A — EDGAR XBRL frames: bulk fundamentals for all tickers (~2 min)
Phase 2B — Supabase analyst_cache: analyst consensus + news sentiment (~2s)
Phase 2C — FINRA RegSHO daily file: short-volume ratios (~5s)
Phase 2D — Beta calculated locally vs SPY from price history (~20s)
Phase 3 — Assemble one record per ticker, apply any active filters
Tickers with no EDGAR coverage (typically micro/nano-cap OTC stocks) still
appear in results — they score on price momentum, risk, and analyst data;
fundamental fields default to NaN and receive the 50-point neutral score.
screen_filters keys (all optional):
sector_filter (col, val) tuple | None
mktcap_range (lo, hi) tuple | None
"""
filters = screen_filters or {}
total = len(tickers)
start_date = (datetime.today() - timedelta(days=380)).strftime("%Y-%m-%d")
start_time = time.time()
# ── Phase 1: batch price history ──────────────────────────────────────
print(f"Phase 1: downloading price history for {total} tickers ...")
close_map: dict[str, pd.Series] = {}
for i in range(0, total, HIST_BATCH):
batch = tickers[i: i + HIST_BATCH]
closes = _download_history_batch(batch, start_date)
close_map.update(closes)
scanned = min(i + HIST_BATCH, total)
print(
f"\r {scanned}/{total} scanned valid={len(close_map)}"
f" ({scanned / total * 100:.0f}%) ",
end="", flush=True,
)
if i + HIST_BATCH < total:
time.sleep(HIST_PAUSE)
valid = list(close_map.keys())
print(f"\n Phase 1 done — {len(valid)}/{total} tickers have price history.\n")
if not valid:
print(" [WARN] yf.download returned no data; trying individual history calls ...")
for t in tickers[:500]:
try:
hist = yf.Ticker(t).history(start=start_date, auto_adjust=True)
if not hist.empty and len(hist) >= 20:
col = "Close" if "Close" in hist.columns else "Adj Close"
if col in hist.columns:
s = hist[col].dropna()
if len(s) >= 20:
close_map[t] = s
except Exception:
pass
time.sleep(random.uniform(0.3, 0.7))
valid = list(close_map.keys())
if not valid:
print("ERROR: No tickers returned valid price history.")
return pd.DataFrame()
print(f" Fallback found {len(valid)} tickers with history.\n")
# ── Phase 2A: EDGAR bulk fundamentals ─────────────────────────────────
edgar_data = _fetch_edgar_bulk(close_map)
# ── Phase 2B: Analyst + news cache from Supabase ──────────────────────
print("Phase 2B: Loading analyst cache ...")
analyst_cache = _load_analyst_cache()
# ── Phase 2C: FINRA short-volume ratios ───────────────────────────────
print("Phase 2C: Fetching FINRA short interest ...")
finra_short = _fetch_finra_short()
# ── Phase 2D: Beta vs SPY ─────────────────────────────────────────────
print("Phase 2D: Calculating beta ...")
beta_map = _calc_beta_map(close_map, start_date)
# ── Phase 3: assemble per-ticker records ──────────────────────────────
sector_filter = filters.get("sector_filter")
mktcap_range = filters.get("mktcap_range")
results: list[dict] = []
filter_labels = []
if sector_filter:
filter_labels.append(f"sector={sector_filter[1]}")
if mktcap_range:
lo, hi = mktcap_range
filter_labels.append(
f"mktcap={'$'+str(int(lo//1e9))+'B+' if lo else ''}"
f"{'$'+str(int(hi//1e9))+'B' if hi else ''}"
)
filter_note = f" [{', '.join(filter_labels)}]" if filter_labels else ""
print(f"\nPhase 3: assembling {len(valid)} records{filter_note} ...")
for ticker in valid:
close = close_map[ticker]
edgar = edgar_data.get(ticker, {})
analyst = analyst_cache.get(ticker, {})
price_now = float(close.iloc[-1])
def pct_return(days_back: int) -> float | None:
idx = max(0, len(close) - days_back)
past = float(close.iloc[idx])
return (price_now - past) / past if past > 0 else None
returns_30d = close.iloc[-30:].pct_change().dropna()
volatility = (float(returns_30d.std() * (252 ** 0.5))
if len(returns_30d) >= 5 else None)
mkt_cap = edgar.get("market_cap")
# ── Filters ───────────────────────────────────────────────────────
if sector_filter is not None:
col, val = sector_filter
field = "sector" if col == "sector" else "industry"
actual = analyst.get(field) or edgar.get(field, "N/A")
if actual != val:
continue
if mktcap_range is not None:
lo, hi = mktcap_range
mkt = float(mkt_cap or 0)
if lo is not None and mkt < lo:
continue
if hi is not None and mkt >= hi:
continue
# ── Analyst fields ────────────────────────────────────────────────
analyst_target = analyst.get("analyst_target")
analyst_upside = analyst.get("analyst_upside")
if analyst_upside is None and analyst_target and price_now > 0:
try:
analyst_upside = float(analyst_target) / price_now - 1.0
except (TypeError, ValueError):
analyst_upside = None
results.append({
"ticker": ticker,
"name": edgar.get("name", ticker),
"sector": analyst.get("sector") or edgar.get("sector", "N/A"),
"industry": analyst.get("industry") or edgar.get("industry", "N/A"),
"price": price_now,
# Valuation
"pe_forward": analyst.get("pe_forward"),
"pe_trailing": edgar.get("pe_trailing"),
"pb_ratio": edgar.get("pb_ratio"),
"ev_ebitda": edgar.get("ev_ebitda"),
# Growth
"revenue_growth": edgar.get("revenue_growth"),
"earnings_growth": edgar.get("earnings_growth"),
"eps_forward": analyst.get("eps_forward"),
"eps_trailing": edgar.get("eps_trailing"),
# Momentum (price-derived)
"ret_1m": pct_return(21),
"ret_3m": pct_return(63),
"ret_6m": pct_return(126),
"ret_12m": pct_return(252),
# Quality
"roe": edgar.get("roe"),
"roa": edgar.get("roa"),
"debt_to_equity": edgar.get("debt_to_equity"),
"total_debt": edgar.get("total_debt"),
"total_cash": edgar.get("total_cash"),
"book_value": edgar.get("book_value"),
"current_ratio": edgar.get("current_ratio"),
# Profitability
"profit_margin": edgar.get("profit_margin"),
"operating_margin": edgar.get("operating_margin"),
"fcf_yield": edgar.get("fcf_yield"),
# Sentiment (from nightly cache)
"news_sentiment": analyst.get("news_sentiment"),
"news_headlines": [],
# Analyst (from nightly cache)
"analyst_norm": analyst.get("analyst_norm"),
"analyst_upside": analyst_upside,
"analyst_count": analyst.get("analyst_count"),
"analyst_target": analyst_target,
"recommendation": analyst.get("recommendation", "N/A"),
# Risk
"short_percent": finra_short.get(ticker),
"beta": beta_map.get(ticker),
"volatility_30d": volatility,
# Market data
"market_cap": mkt_cap,
"52w_high": float(close.max()),
"52w_low": float(close.min()),
"dividend_yield": None,
"revenue": edgar.get("revenue"),
"business_summary": "",
})
print(f"\nDone. Assembled {len(results)} stocks in"
f" {time.time() - start_time:.0f}s.\n")
df = pd.DataFrame(results)
if not df.empty:
df = df.drop_duplicates(subset=["ticker"], keep="first").reset_index(drop=True)
return df
# ---------------------------------------------------------------------------
# Scoring
# ---------------------------------------------------------------------------
def _abs_score(v, breakpoints: list, default: float = 50.0) -> float:
"""
Piecewise-linear interpolation of a scalar value v against a list of
(metric_value, score) breakpoints sorted by metric_value ascending.
Returns `default` when v is NaN/None.
Example: breakpoints=[(0,0),(10,50),(20,100)]
v=5 → 25.0
v=15 → 75.0
v=25 → 100.0 (clamped)
"""
if not breakpoints:
return default
if v is None:
return default
try:
if np.isnan(v):
return default
except (TypeError, ValueError):
return default
pts = breakpoints
if v <= pts[0][0]:
return float(pts[0][1])
if v >= pts[-1][0]:
return float(pts[-1][1])
for i in range(len(pts) - 1):
x0, y0 = pts[i]
x1, y1 = pts[i + 1]
if x0 <= v <= x1:
t = (v - x0) / (x1 - x0)
return float(y0 + t * (y1 - y0))
return default
def _score_series(series: pd.Series, breakpoints: list, default: float = 50.0) -> pd.Series:
"""Apply _abs_score element-wise to a Series."""
return series.apply(lambda v: _abs_score(v, breakpoints, default))
def compute_value_score(df: pd.DataFrame) -> pd.Series:
"""
Value Score — how attractively priced is the stock on absolute terms.
Lower multiples = better value = higher score.
P/E: <10 excellent, 15 good, 25 fair, 40+ expensive
P/B: <1 excellent, 2 good, 4 fair, 8+ expensive
EV/EBITDA: <6 excellent, 10 good, 18 fair, 30+ expensive
"""
pe = df["pe_forward"].combine_first(df["pe_trailing"]).where(lambda x: x > 0)
pe_pts = [(5, 100), (10, 85), (15, 70), (20, 55), (25, 40), (35, 20), (50, 5)]
pb_pts = [(0.5, 100), (1.0, 85), (2.0, 70), (3.0, 55), (4.0, 40), (6.0, 20), (10.0, 5)]
ev_pts = [(4, 100), (6, 85), (10, 70), (14, 55), (18, 40), (25, 20), (35, 5)]
return (
_score_series(pe, pe_pts) * 0.40 +
_score_series(df["pb_ratio"], pb_pts) * 0.30 +
_score_series(df["ev_ebitda"], ev_pts) * 0.30
)
def compute_growth_score(df: pd.DataFrame) -> pd.Series:
"""
Growth Score — how fast is the business expanding.
Revenue/earnings growth as decimals (0.10 = 10%).
Negative growth penalised; hyper-growth (>50%) excellent.
"""
eps_delta = df["eps_forward"] - df["eps_trailing"]
rev_pts = [(-0.10, 5), (0.0, 20), (0.05, 40), (0.10, 55), (0.20, 70), (0.30, 85), (0.50, 100)]
ear_pts = [(-0.15, 5), (0.0, 20), (0.05, 38), (0.10, 55), (0.20, 70), (0.35, 85), (0.55, 100)]
eps_pts = [(-2.0, 5), (0.0, 30), (0.25, 50), (0.50, 65), (1.0, 80), (2.0, 92), (3.0, 100)]
return (
_score_series(df["revenue_growth"], rev_pts) * 0.40 +
_score_series(df["earnings_growth"], ear_pts) * 0.40 +
_score_series(eps_delta, eps_pts) * 0.20
)
def compute_momentum_score(df: pd.DataFrame) -> pd.Series:
"""
Momentum Score — recent price performance vs absolute thresholds.
Returns as decimals (0.10 = +10%). Negative = penalty, >30% = top.
"""
r1_pts = [(-0.15, 5), (-0.05, 25), (0.0, 45), (0.05, 60), (0.10, 75), (0.20, 90), (0.30, 100)]
r3_pts = [(-0.20, 5), (-0.05, 25), (0.0, 42), (0.08, 58), (0.15, 72), (0.25, 88), (0.40, 100)]
r6_pts = [(-0.25, 5), (-0.05, 22), (0.0, 38), (0.10, 55), (0.20, 70), (0.35, 87), (0.55, 100)]
r12_pts = [(-0.30, 5), (-0.05, 20), (0.0, 35), (0.12, 52), (0.25, 68), (0.40, 85), (0.65, 100)]
return (
_score_series(df["ret_1m"], r1_pts) * 0.10 +
_score_series(df["ret_3m"], r3_pts) * 0.25 +
_score_series(df["ret_6m"], r6_pts) * 0.40 +
_score_series(df["ret_12m"], r12_pts) * 0.25
)
def compute_reverse_momentum_score(df: pd.DataFrame) -> pd.Series:
"""
Reverse Momentum Score — rewards stocks that have fallen the most recently.
Inverts the standard momentum scale: large negative returns → high score.
Short-term windows weighted more heavily to capture recent drawdowns.
-30% in a month → 100; flat → ~50; +30% in a month → 3.
"""
r1_pts = [(-0.30, 100), (-0.20, 88), (-0.10, 75), (-0.05, 62), (0.0, 48), (0.05, 32), (0.10, 18), (0.20, 8), (0.30, 3)]
r3_pts = [(-0.40, 100), (-0.25, 88), (-0.15, 75), (-0.05, 60), (0.0, 45), (0.08, 30), (0.20, 15), (0.35, 5)]
r6_pts = [(-0.55, 100), (-0.35, 88), (-0.20, 72), (-0.05, 55), (0.0, 42), (0.10, 28), (0.25, 15), (0.40, 5)]
r12_pts = [(-0.65, 100), (-0.40, 85), (-0.25, 68), (-0.05, 52), (0.0, 38), (0.12, 22), (0.30, 10), (0.50, 3)]
return (
_score_series(df["ret_1m"], r1_pts) * 0.35 +
_score_series(df["ret_3m"], r3_pts) * 0.35 +
_score_series(df["ret_6m"], r6_pts) * 0.20 +
_score_series(df["ret_12m"], r12_pts) * 0.10
)
def compute_quality_score(df: pd.DataFrame) -> pd.Series:
"""
Quality Score — balance-sheet and capital efficiency.
ROE: as decimal, >0.20 (20%) is strong; <0 penalised.
ROA: >0.10 (10%) excellent.
D/E: lower is safer; >2.0 high risk.
Current ratio: >2 comfortable; <1 stressed.
"""
roe_pts = [(-0.10, 5), (0.0, 20), (0.05, 40), (0.10, 55), (0.15, 68), (0.20, 80), (0.30, 92), (0.40, 100)]
roa_pts = [(-0.05, 5), (0.0, 20), (0.03, 40), (0.06, 55), (0.10, 70), (0.15, 85), (0.20, 100)]
de_pts = [(0.0, 100), (0.25, 90), (0.50, 78), (1.0, 62), (1.5, 48), (2.0, 32), (3.0, 15), (5.0, 5)]
cr_pts = [(0.5, 5), (0.8, 20), (1.0, 38), (1.5, 60), (2.0, 80), (2.5, 90), (3.0, 100)]
return (
_score_series(df["roe"], roe_pts) * 0.35 +
_score_series(df["roa"], roa_pts) * 0.30 +
_score_series(df["debt_to_equity"], de_pts) * 0.20 +
_score_series(df["current_ratio"], cr_pts) * 0.15
)
def compute_profitability_score(df: pd.DataFrame) -> pd.Series:
"""
Profitability Score — how much of revenue becomes profit.
All margins as decimals. Negative margins penalised.
FCF yield: FCF / market cap. >5% excellent.
"""
pm_pts = [(-0.10, 5), (0.0, 18), (0.03, 35), (0.08, 52), (0.15, 68), (0.25, 84), (0.35, 100)]
om_pts = [(-0.10, 5), (0.0, 18), (0.05, 38), (0.12, 55), (0.20, 70), (0.30, 86), (0.40, 100)]
fcf_pts = [(-0.05, 5), (0.0, 20), (0.01, 35), (0.03, 50), (0.05, 65), (0.08, 82), (0.12, 100)]
return (
_score_series(df["profit_margin"], pm_pts) * 0.35 +
_score_series(df["operating_margin"], om_pts) * 0.35 +
_score_series(df["fcf_yield"], fcf_pts) * 0.30
)
def compute_sentiment_score(df: pd.DataFrame) -> pd.Series:
"""
News Sentiment Score — VADER compound score mapped from [-1, +1] to [0, 100].
Neutral (0) → 50; strongly positive (+0.5) → 75; strongly negative (-0.5) → 25.
"""
sent_pts = [(-1.0, 0), (-0.5, 25), (-0.2, 38), (0.0, 50), (0.2, 62), (0.5, 75), (1.0, 100)]
return _score_series(df["news_sentiment"], sent_pts)
def compute_analyst_score(df: pd.DataFrame) -> pd.Series:
"""
Analyst Score — consensus rating + price target upside.
analyst_norm: 0=Strong Sell → 1=Strong Buy (absolute mapping to 0100).
analyst_upside: expected % gain (decimal). >20% = very bullish.
analyst_count: more coverage = higher confidence (logarithmic).
"""
norm_pts = [(0.0, 0), (0.25, 25), (0.50, 50), (0.65, 65), (0.75, 78), (0.875, 90), (1.0, 100)]
upside_pts = [(-0.20, 5), (-0.05, 20), (0.0, 35), (0.05, 48), (0.10, 60), (0.20, 78), (0.35, 95), (0.50, 100)]
count_pts = [(0, 0), (1, 20), (3, 40), (5, 55), (10, 70), (20, 85), (30, 100)]
return (
_score_series(df["analyst_norm"], norm_pts) * 0.50 +
_score_series(df["analyst_upside"], upside_pts) * 0.35 +
_score_series(df["analyst_count"], count_pts) * 0.15
)
def compute_risk_score(df: pd.DataFrame) -> pd.Series:
"""
Risk Score — higher score means LOWER risk (more attractive).
Short interest %: >15% very bearish signal; <3% low risk.
Volatility 30d (annualised): >60% high risk; <15% stable.
Beta deviation from 1.0: extreme values (very high or negative) add tail risk.
"""
short_pts = [(0.0, 100), (0.03, 88), (0.05, 75), (0.10, 55), (0.15, 35), (0.20, 18), (0.30, 5)]
vol_pts = [(0.05, 100), (0.10, 88), (0.15, 78), (0.25, 62), (0.35, 45), (0.50, 25), (0.70, 8), (1.0, 3)]
beta_dev = (df["beta"].fillna(1.0) - 1.0).abs()
bdev_pts = [(0.0, 100), (0.20, 85), (0.40, 70), (0.60, 55), (0.80, 40), (1.0, 25), (1.50, 10), (2.0, 5)]
return (
_score_series(df["short_percent"], short_pts) * 0.50 +
_score_series(df["volatility_30d"], vol_pts) * 0.30 +
_score_series(beta_dev, bdev_pts) * 0.20
)
def score_stocks(df: pd.DataFrame, weights: dict | None = None) -> pd.DataFrame:
df = df.copy()
df = df.drop_duplicates(subset=["ticker"], keep="first").reset_index(drop=True)
# Force every metric column to float before scoring.
# yfinance sometimes returns strings ("Infinity", "N/A") for PE/EPS fields
# on edge-case securities (REITs, ADRs, CLEFs, foreign listings) — common
# in large indices like NYSE. Without coercion, comparisons like `x > 0`
# inside compute_value_score raise TypeError on those rows.
_NUMERIC_COLS = [
"pe_forward", "pe_trailing", "pb_ratio", "ev_ebitda",
"revenue_growth", "earnings_growth", "eps_forward", "eps_trailing",
"ret_1m", "ret_3m", "ret_6m", "ret_12m",
"roe", "roa", "debt_to_equity", "current_ratio",
"profit_margin", "operating_margin", "fcf_yield",
"news_sentiment", "analyst_norm", "analyst_upside", "analyst_count",
"short_percent", "volatility_30d", "beta", "price", "market_cap",
]
for col in _NUMERIC_COLS:
if col in df.columns:
df[col] = pd.to_numeric(df[col], errors="coerce")
df["score_value"] = compute_value_score(df)
df["score_growth"] = compute_growth_score(df)
df["score_momentum"] = compute_momentum_score(df)
df["score_reverse_momentum"] = compute_reverse_momentum_score(df)
df["score_quality"] = compute_quality_score(df)
df["score_profitability"] = compute_profitability_score(df)
df["score_sentiment"] = compute_sentiment_score(df)
df["score_analyst"] = compute_analyst_score(df)
df["score_risk"] = compute_risk_score(df)
w = weights if weights is not None else WEIGHTS
df["composite_score"] = sum(
df[f"score_{k}"] * v for k, v in w.items()
)
return df.sort_values("composite_score", ascending=False).reset_index(drop=True)
# ---------------------------------------------------------------------------
# Display
# ---------------------------------------------------------------------------
def fmt_pct(val) -> str:
if val is None or (isinstance(val, float) and np.isnan(val)):
return "N/A"
return f"{val*100:.1f}%"
def fmt_float(val, decimals=2) -> str:
if val is None or (isinstance(val, float) and np.isnan(val)):
return "N/A"
return f"{val:.{decimals}f}"
def fmt_mcap(val) -> str:
if val is None or (isinstance(val, float) and np.isnan(val)):
return "N/A"
if val >= 1e12: return f"${val/1e12:.1f}T"
if val >= 1e9: return f"${val/1e9:.1f}B"
if val >= 1e6: return f"${val/1e6:.1f}M"
return f"${val:.0f}"
def print_results(df: pd.DataFrame, top_n: int = 50) -> None:
top = df.head(top_n).copy()
top.index = range(1, len(top) + 1)
W = 140
print("=" * W)
print(f" TOP {top_n} STOCKS — {datetime.today().strftime('%Y-%m-%d %H:%M')}")
print("=" * W)
print(
f"{'#':>3} {'Ticker':<7} {'Name':<26} {'Sector':<20} "
f"{'Score':>6} {'Val':>5} {'Grw':>5} {'Mom':>5} {'Qlt':>5} "
f"{'Pft':>5} {'Sent':>5} {'Anlst':>6} {'Risk':>5} "
f"{'P/E':>6} {'RevGrw':>7} {'ROE':>6} {'MktCap':>8}"
)
print("-" * W)
for rank, row in top.iterrows():
print(
f"{rank:>3} {row['ticker']:<7} {str(row['name'])[:25]:<26} "
f"{str(row['sector'])[:19]:<20} "
f"{row['composite_score']:>6.1f} "
f"{row['score_value']:>5.1f} {row['score_growth']:>5.1f} "
f"{row['score_momentum']:>5.1f} {row['score_quality']:>5.1f} "
f"{row['score_profitability']:>5.1f} {row['score_sentiment']:>5.1f} "
f"{row['score_analyst']:>6.1f} {row['score_risk']:>5.1f} "
f"{fmt_float(row.get('pe_forward') or row.get('pe_trailing')):>6} "
f"{fmt_pct(row.get('revenue_growth')):>7} "
f"{fmt_pct(row.get('roe')):>6} "
f"{fmt_mcap(row.get('market_cap')):>8}"
)
print("=" * W)
w = WEIGHTS
print(
f"\nFormula: {w['value']*100:.0f}%*Val + {w['growth']*100:.0f}%*Grw + "
f"{w['momentum']*100:.0f}%*Mom + {w['quality']*100:.0f}%*Qlt + "
f"{w['profitability']*100:.0f}%*Pft + {w['sentiment']*100:.0f}%*Sent + "
f"{w['analyst']*100:.0f}%*Anlst + {w['risk']*100:.0f}%*Risk\n"
)
def save_csv(df: pd.DataFrame, top_n: int) -> str:
cols = [
"ticker", "name", "sector", "price", "market_cap",
"composite_score", "score_value", "score_growth", "score_momentum",
"score_quality", "score_profitability", "score_sentiment",
"score_analyst", "score_risk",
"pe_forward", "pe_trailing", "pb_ratio", "ev_ebitda",
"revenue_growth", "earnings_growth",
"ret_1m", "ret_3m", "ret_6m", "ret_12m",
"roe", "roa", "debt_to_equity", "current_ratio",
"profit_margin", "operating_margin", "fcf_yield",
"news_sentiment", "analyst_norm", "analyst_upside",
"analyst_count", "analyst_target", "recommendation",
"short_percent", "beta", "volatility_30d",
]
out_cols = [c for c in cols if c in df.columns]
filename = f"top{top_n}_stocks_{datetime.today().strftime('%Y%m%d')}.csv"
df.head(top_n)[out_cols].to_csv(filename, index=False)
return filename
# ---------------------------------------------------------------------------
# Entry Point
# ---------------------------------------------------------------------------
def parse_args():
p = argparse.ArgumentParser(description="Multi-factor stock screener")
p.add_argument("--top", type=int, default=50)
p.add_argument("--workers", type=int, default=15)
p.add_argument("--output", choices=["console", "csv", "both"], default="console")
p.add_argument("--index", default="All", choices=INDEX_CHOICES,
help="Index universe to screen (default: All)")
return p.parse_args()
def main():
args = parse_args()
print("\n" + "=" * 60)
print(" MULTI-FACTOR STOCK SCREENER (8-Factor Analyst Model)")
print("=" * 60 + "\n")
if _VADER is None:
print(" [INFO] vaderSentiment not installed — sentiment scores will be neutral.")
print(" Run: pip install vaderSentiment\n")
tickers = collect_tickers(index=args.index)
if not tickers:
print("ERROR: Could not retrieve any ticker lists.")
sys.exit(1)
df_raw = fetch_all(tickers, max_workers=args.workers)
if df_raw.empty:
print("ERROR: No data returned.")
sys.exit(1)
print("Computing scores...")
df_scored = score_stocks(df_raw)
print(f"Scored {len(df_scored)} stocks.\n")
if args.output in ("console", "both"):
print_results(df_scored, top_n=args.top)
if args.output in ("csv", "both"):
print(f"Results saved to: {save_csv(df_scored, top_n=args.top)}")
# ---------------------------------------------------------------------------
# Insider Trading — SEC EDGAR Form 4
# ---------------------------------------------------------------------------
_FORM4_TX_CODES: dict[str, str] = {
"P": "Buy", "S": "Sale",
"A": "Award", "D": "Disposition",
"F": "Tax Withholding","G": "Gift",
"M": "Option Exercise","X": "Option Exercise",
"O": "Option Exercise","J": "Other",
"I": "Plan Trade", "C": "Conversion",
}
def _parse_form4_xml(xml_bytes: bytes, filing_url: str, filed_at: str) -> list[dict]:
"""Parse a Form 4 XML document and return a list of transaction dicts."""
import xml.etree.ElementTree as ET
try:
root = ET.fromstring(xml_bytes)
except ET.ParseError:
return []
def _val(node, *tags):
"""Return stripped text from the first matching .//tag/value or .//tag."""
for tag in tags:
n = node.find(f".//{tag}/value")
if n is not None and n.text:
return n.text.strip()
n = node.find(f".//{tag}")
if n is not None and n.text:
return n.text.strip()
return ""
ticker = _val(root, "issuerTradingSymbol")
company = _val(root, "issuerName")
insider = _val(root, "rptOwnerName")
period = _val(root, "periodOfReport")
title = ""
rel = root.find(".//reportingOwnerRelationship")
if rel is not None:
ot = _val(rel, "officerTitle")
if ot:
title = ot
elif _val(rel, "isDirector") in ("1", "true"):
title = "Director"
elif _val(rel, "isTenPercentOwner") in ("1", "true"):
title = "10% Owner"
elif _val(rel, "isOfficer") in ("1", "true"):
title = "Officer"
transactions: list[dict] = []
for tx in root.findall(".//nonDerivativeTransaction"):
tx_date = _val(tx, "transactionDate") or period
tx_code = _val(tx, "transactionCode")
tx_type = _FORM4_TX_CODES.get(tx_code, tx_code or "Other")
try: shares = float(_val(tx, "transactionShares"))
except: shares = None
try: price = float(_val(tx, "transactionPricePerShare"))
except: price = None
try: owned = float(_val(tx, "sharesOwnedFollowingTransaction"))
except: owned = None
value = (shares * price) if (shares and price) else None
transactions.append({
"filed_date": (filed_at or "")[:10] or tx_date,
"tx_date": tx_date,
"ticker": ticker.upper() if ticker else "",
"company": company,
"insider": insider,
"title": title,
"transaction_type": tx_type,
"transaction_code": tx_code,
"shares": shares,
"price": price,
"value": value,
"owned_after": owned,
"url": filing_url,
})
return transactions
def fetch_insider_trades(days_back: int = 3, max_results: int = 60,
progress_cb=None) -> list[dict]:
"""
Fetch recent Form 4 insider transactions from SEC EDGAR.
Uses the EDGAR EFTS search API to list recent filings, then
parallel-fetches each Form 4 XML for full transaction details.
days_back : number of calendar days back from today to search.
max_results : cap on number of filings fetched (each filing may
contain multiple individual transactions).
progress_cb : optional callable(done_count, total_count).
Returns a list of transaction dicts sorted by tx_date descending.
"""
from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import date, timedelta
_UA = {"User-Agent": "UltimateInvestmentTool contact@investmenttool.com"}
end_dt = date.today()
start_dt = end_dt - timedelta(days=max(days_back, 1))
# ── Step 1: list recent Form 4 filings from EDGAR EFTS ────────────────
try:
resp = requests.get(
"https://efts.sec.gov/LATEST/search-index",
params={
"forms": "4",
"dateRange": "custom",
"startdt": start_dt.strftime("%Y-%m-%d"),
"enddt": end_dt.strftime("%Y-%m-%d"),
},
headers=_UA,
timeout=12,
)
resp.raise_for_status()
hits = resp.json().get("hits", {}).get("hits", [])[:max_results]
except Exception:
return []
if not hits:
return []
# ── Step 2: fetch + parse each Form 4 XML (parallel) ──────────────────
def _fetch_one(hit: dict) -> list[dict]:
try:
src = hit.get("_source", {})
filed = src.get("file_date", "") or src.get("filed_at", "")
# Accession number from _source.adsh (not _id which has ":filename" appended)
acc_no = src.get("adsh", "")
if not acc_no:
return []
# XML filename is embedded in _id as "{adsh}:{filename}"
xml_filename = hit["_id"].split(":")[-1]
# CIK: use ciks[0] from _source — NOT the accession prefix (which is the filing agent)
ciks = src.get("ciks", [])
if not ciks:
return []
cik_int = int(ciks[0].lstrip("0") or "0")
if cik_int == 0:
return []
acc_nd = acc_no.replace("-", "")
base = f"https://www.sec.gov/Archives/edgar/data/{cik_int}/{acc_nd}"
filing_url = f"{base}/{acc_no}-index.htm"
xml_url = f"{base}/{xml_filename}"
xml_r = requests.get(xml_url, headers=_UA, timeout=8)
xml_r.raise_for_status()
return _parse_form4_xml(xml_r.content, filing_url, filed)
except Exception:
return []
total = len(hits)
done_n = 0
results: list[dict] = []
with ThreadPoolExecutor(max_workers=6) as ex:
futs = {ex.submit(_fetch_one, h): h for h in hits}
for fut in as_completed(futs):
done_n += 1
results.extend(fut.result() or [])
if progress_cb:
try:
progress_cb(done_n, total)
except Exception:
pass
results.sort(
key=lambda x: (x.get("tx_date") or "", x.get("filed_date") or ""),
reverse=True,
)
return results
if __name__ == "__main__":
main()