Hybrid architecture: Item 7 only to Gemini, yfinance for metrics; HTML cleansing; README and find_toc script

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
shawnkim1997
2026-02-13 17:43:26 +00:00
co-authored by Cursor
parent 522bf4dc75
commit 8819d5bcfa
4 changed files with 324 additions and 212 deletions
+157 -195
View File
@@ -1,8 +1,9 @@
"""
10-K Financial Analyzer (Google Gemini 1.5 Flash)
- Download 10-K from SEC EDGAR and extract text
- Analysis using Item 7 (MD&A) and Item 8 (Financial Statements) via Gemini 1.5 Flash (generous free tier, large context)
- CFA-style summary, key metrics table, and CFA Investment Report section
10-K Financial Analyzer (Google Gemini) — Hybrid Architecture
- Download 10-K from SEC EDGAR; extract Item 7 (MD&A) only for AI.
- Quantitative: financial metrics (Revenue, Net Income, Operating Cash Flow) from yfinance.
- Qualitative: Item 7 only to Gemini for strategic direction, risks, and sentiment analysis.
- HTML cleansing before sending text to LLM to minimise tokens.
- All content in British English.
"""
@@ -18,20 +19,19 @@ import streamlit as st
import pandas as pd
from bs4 import BeautifulSoup
# Load .env if python-dotenv is available
try:
from dotenv import load_dotenv
load_dotenv()
except ImportError:
pass
def get_edgar_downloader():
from sec_edgar_downloader import Downloader
return Downloader
def extract_text_from_html(html_path: Path) -> str:
"""Extract plain text from an HTML file."""
try:
with open(html_path, "r", encoding="utf-8", errors="replace") as f:
soup = BeautifulSoup(f.read(), "lxml")
@@ -44,7 +44,6 @@ def extract_text_from_html(html_path: Path) -> str:
def extract_text_from_file(file_path: Path) -> str:
"""Extract text by file extension (HTML or TXT)."""
suf = file_path.suffix.lower()
if suf in (".htm", ".html"):
return extract_text_from_html(file_path)
@@ -57,7 +56,6 @@ def extract_text_from_file(file_path: Path) -> str:
return ""
# ---------- Selective Section Extraction (pre-filter: only Item 7 & 8, no PART I / ITEM 16) ----------
ITEM7_PATTERNS = [
r"Item\s+7\s*[.:]\s*Management['\u2019]s\s+Discussion\s+and\s+Analysis",
r"ITEM\s+7\s*[.:]\s*Management['\u2019]s\s+Discussion",
@@ -71,7 +69,6 @@ ITEM8_PATTERNS = [
def _find_section_start(text: str, patterns: list, item_num: int) -> int:
"""Return start index of first matching pattern, or -1."""
for pat in patterns:
m = re.search(pat, text, re.IGNORECASE)
if m:
@@ -81,13 +78,11 @@ def _find_section_start(text: str, patterns: list, item_num: int) -> int:
def prefilter_after_item7(full_text: str) -> str:
"""Drop PART I, ITEM 16; keep only from Item 7 onward to reduce noise and token use."""
start = _find_section_start(full_text, ITEM7_PATTERNS, 7)
return full_text[start:] if start >= 0 else full_text
def find_item_section(text: str, item_num: int, title_keywords: list) -> str:
"""Extract only Item N section (regex-based). Used for Item 7 (MD&A) and Item 8 (Financial Statements)."""
patterns = ITEM7_PATTERNS if item_num == 7 else ITEM8_PATTERNS
start = _find_section_start(text, patterns, item_num)
if start == -1:
@@ -99,7 +94,6 @@ def find_item_section(text: str, item_num: int, title_keywords: list) -> str:
if not match:
return ""
start = match.start()
# End at next "Item N" (next major section)
next_item = re.search(r"\n\s*Item\s+\d+\s+", text[start + 100 :], re.IGNORECASE)
if next_item:
end = start + 100 + next_item.start()
@@ -109,14 +103,10 @@ def find_item_section(text: str, item_num: int, title_keywords: list) -> str:
def smart_chunk(section: str, max_chars: int = 30000, head_ratio: float = 0.5) -> str:
"""
If section exceeds max_chars, keep head and tail (quantitative data often at start/end).
Reduces tokens while preserving high-signal content.
"""
if len(section) <= max_chars:
return section
head_size = int(max_chars * head_ratio)
tail_size = max_chars - head_size - 100 # reserve for separator
tail_size = max_chars - head_size - 100
return (
section[:head_size]
+ "\n\n[ ... middle omitted to stay within token limit ... ]\n\n"
@@ -124,8 +114,40 @@ def smart_chunk(section: str, max_chars: int = 30000, head_ratio: float = 0.5) -
)
def clean_text_for_llm(text: str) -> str:
"""
Token-compression cleansing before sending to LLM: strip HTML remnants,
collapse whitespace, remove page numbers and excessive special characters.
"""
if not text or not text.strip():
return ""
# Remove any remaining HTML tags (safe on plain text)
text = re.sub(r"<[^>]+>", " ", text)
# Collapse multiple spaces to one
text = re.sub(r"[ \t]+", " ", text)
# Normalise line endings and collapse many blank lines to at most two newlines
text = re.sub(r"\r\n?", "\n", text)
text = re.sub(r"\n{3,}", "\n\n", text)
lines = []
for line in text.split("\n"):
line = line.strip()
# Drop lines that are only digits (page numbers) or only punctuation/dashes
if not line:
lines.append("")
continue
if re.fullmatch(r"\d+", line) or re.fullmatch(r"[\.\-\s\-]+", line):
continue
# Short boilerplate lines (e.g. "Page 1 of 2") — optional: drop very short lines that look like page refs
if re.match(r"^(page\s+\d+|\d+)\s*$", line, re.IGNORECASE) and len(line) < 20:
continue
lines.append(line)
# Rejoin and collapse again
result = "\n".join(lines)
result = re.sub(r"\n{3,}", "\n\n", result)
return result.strip()
def find_downloaded_10k_path(download_root: Path, ticker: str) -> Optional[Path]:
"""Return the path to the latest 10-K folder for the given ticker under download_root."""
ticker_upper = ticker.upper()
for base in (download_root / "sec-edgar-filings", download_root):
path_10k = base / ticker_upper / "10-K"
@@ -148,7 +170,6 @@ def find_downloaded_10k_path(download_root: Path, ticker: str) -> Optional[Path]
def get_main_10k_text(filing_dir: Path) -> str:
"""Find the main document (HTML/TXT) in the 10-K folder and return its full text."""
all_text = []
for ext in ("*.htm", "*.html", "*.txt"):
for path in filing_dir.rglob(ext):
@@ -164,15 +185,12 @@ def get_main_10k_text(filing_dir: Path) -> str:
return main_text
# ---------- Gemini 1.5 Flash: stable, generous free tier, good for large 10-K text ----------
GEMINI_MODEL = "gemini-2.0-flash"
# Wait 1 minute before retry when rate limited (free tier resets after a short period)
RATE_LIMIT_WAIT_SEC = 60
DELAY_BETWEEN_CALLS_SEC = 8
def get_gemini_model(api_key: str):
"""Return configured Gemini Flash model (generous free tier for large documents)."""
import google.generativeai as genai
genai.configure(api_key=api_key)
return genai.GenerativeModel(GEMINI_MODEL)
@@ -189,7 +207,6 @@ def _is_rate_limit_error(e: Exception) -> bool:
def _generate_with_retry(model, content, generation_config, max_retries: int = 3):
"""Call model.generate_content with retry on 429 (wait then retry up to max_retries times)."""
last_err = None
for attempt in range(max_retries + 1):
try:
@@ -203,62 +220,99 @@ def _generate_with_retry(model, content, generation_config, max_retries: int = 3
raise last_err
def get_ai_summary_and_report(
api_key: str,
full_text: str,
item7_text: str,
item8_text: str,
ticker: str,
) -> tuple[str, str]:
def get_metrics_from_yfinance(ticker: str) -> pd.DataFrame:
"""
Produce detailed analysis and CFA report. Only Item 7 and Item 8 are sent (pre-filtered).
Sections are smart-chunked (head + tail) when long to keep token use low and avoid 429.
Quantitative data: fetch Revenue, Net Income, Operating Cash Flow from yfinance
(no LLM; fast and accurate). Returns a DataFrame suitable for Streamlit display.
"""
try:
import yfinance as yf
except ImportError:
return pd.DataFrame()
try:
t = yf.Ticker(ticker.upper())
financials = t.financials # annual income statement
cashflow = t.cashflow # annual cash flow
if financials is None or financials.empty:
return pd.DataFrame()
# Prefer common index names (yfinance varies by region)
rev_row = None
for name in ("Total Revenue", "Revenue", "Net Revenue", "Operating Revenue"):
if name in financials.index:
rev_row = financials.loc[name]
break
ni_row = None
for name in ("Net Income", "Net Income Common Stockholders", "Net Income Including Noncontrolling Interests"):
if name in financials.index:
ni_row = financials.loc[name]
break
ocf_row = None
if cashflow is not None and not cashflow.empty:
for name in ("Operating Cash Flow", "Cash From Operating Activities", "Cash From Operations"):
if name in cashflow.index:
ocf_row = cashflow.loc[name]
break
# Align by date (columns are often datetime)
dates = financials.columns.tolist()
if not dates:
return pd.DataFrame()
# Sort descending (most recent first) and take up to 5 years
dates = sorted(dates, reverse=True)[:5]
cashflow_cols = list(cashflow.columns) if cashflow is not None and not cashflow.empty else []
data = {}
for d in dates:
yr = d.year if hasattr(d, "year") else int(str(d)[:4])
rev_val = (rev_row[d] / 1e6) if rev_row is not None and d in rev_row.index else None
ni_val = (ni_row[d] / 1e6) if ni_row is not None and d in ni_row.index else None
ocf_val = None
if ocf_row is not None:
if d in ocf_row.index:
ocf_val = ocf_row[d] / 1e6
else:
for c in cashflow_cols:
cy = c.year if hasattr(c, "year") else int(str(c)[:4])
if cy == yr:
ocf_val = ocf_row[c] / 1e6
break
data[yr] = {"Revenue": rev_val, "Net Income": ni_val, "Operating Cash Flow": ocf_val}
df = pd.DataFrame(data).T
df.index.name = "Fiscal Year"
df = df.astype(float).round(2)
return df
except Exception:
return pd.DataFrame()
def get_ai_summary_and_report(api_key: str, item7_text: str, ticker: str) -> tuple[str, str]:
"""
Qualitative only: send Item 7 (MD&A) to Gemini. Focus on strategic direction,
market risks, and sentiment—not on summarising financial statement numbers.
"""
model = get_gemini_model(api_key)
item7_text = clean_text_for_llm(item7_text)
item7_text = smart_chunk(item7_text, max_chars=20000)
# Smart chunking: when over limit, keep head + tail (figures often at start/end)
max_chars_per_section = 30000
item7_text = smart_chunk(item7_text, max_chars=max_chars_per_section)
item8_text = smart_chunk(item8_text, max_chars=max_chars_per_section)
user_prompt = f"""You are a CFA charterholder and senior equity analyst. Use British English.
user_prompt = f"""You are a CFA charterholder and senior equity analyst. Use British English. Omit unnecessary qualifiers and filler; focus on figures, risks, and material facts.
The text below is Item 7 (Management's Discussion and Analysis) only from the 10-K for company ticker: {ticker}. Do NOT ask for financial statements or numbers—this is a qualitative analysis.
Analyse the following 10-K excerpts for company ticker: {ticker}. The text below contains ONLY Item 7 (MD&A) and Item 8 (Financial Statements)—other sections have been pre-filtered out.
Your task:
1. **Strategic direction**: How does management describe its strategy, priorities, and capital allocation? What are the main growth drivers or initiatives?
2. **Market and business risks**: What material risks (competitive, regulatory, operational, macro) does management emphasise? Be specific and cite the wording where relevant.
3. **Tone (Sentiment)**: Overall, is the tone of MD&A more positive, cautious, or negative? Highlight 23 phrases or themes that support your view.
Use the provided text to produce a thorough, evidence-based analysis. Cite specific numbers and risk disclosures where relevant.
Then write a "CFA INVESTMENT REPORT" section with:
- **Executive Summary**: 23 sentences on the company's narrative and management's message.
- **Investment Thesis**: Key strengths and catalysts from the discussion.
- **Key Risks to the Thesis**: Main downside risks from the text.
- **Conclusion**: Balanced wrap-up.
First, write a "DETAILED ANALYSIS" section with exactly three paragraphs (use subheadings):
1. **Financial Health**: Liquidity (current ratio, cash position, credit facilities), leverage (debt/equity, interest coverage), capital structure, and any covenant or refinancing risks. Cite figures from the statements.
2. **Profitability**: Revenue and earnings trends, margins (gross, operating, net), earnings quality (e.g. non-GAAP adjustments, one-time items), and sustainability of earnings. Use numbers from the 10-K.
3. **Key Risks**: Material risk factors from MD&A and notes (market, credit, operational, legal, ESG if material). Be specific; quote or paraphrase the filing.
Keep the entire response in British English. Use clear section headers. Do not invent figures—only refer to what is in the text."""
Then, write a "CFA INVESTMENT REPORT" section in the style of a formal sell-side or buy-side investment memo. Include these subsections with clear headings:
- **Executive Summary**: 23 sentences on the company's position and your high-level view.
- **Investment Thesis**: Why an investor might consider this company (strengths, catalysts). Be specific.
- **Valuation Considerations**: What to watch (multiples, growth, margins, capital allocation). No exact price target required.
- **Key Risks to the Thesis**: Main downside risks that could invalidate the thesis.
- **Conclusion**: One short paragraph with a balanced wrap-up (e.g. Hold/Overweight/Underweight context and what would change your view).
Keep the entire response in British English. Use clear section headers (e.g. ## or **) and professional language."""
full_content = f"""--- Item 7. Management's Discussion and Analysis (full or extended excerpt) ---
{item7_text}
--- Item 8. Financial Statements and Notes (full or extended excerpt) ---
{item8_text}
---
{user_prompt}"""
full_content = f"""--- Item 7. Management's Discussion and Analysis (MD&A) ---\n\n{item7_text}\n\n---\n\n{user_prompt}"""
try:
response = _generate_with_retry(
model,
full_content,
{"temperature": 0.3, "max_output_tokens": 8192},
)
response = _generate_with_retry(model, full_content, {"temperature": 0.3, "max_output_tokens": 8192})
except Exception as api_err:
if _is_rate_limit_error(api_err):
raise RuntimeError("Rate limit exceeded. Please try again in a few minutes.") from api_err
@@ -268,69 +322,18 @@ Keep the entire response in British English. Use clear section headers (e.g. ##
return "No analysis generated.", "No report generated."
text = response.text.strip()
# Split into "DETAILED ANALYSIS" and "CFA INVESTMENT REPORT" if the model used those headers
detailed = ""
report = ""
if "CFA INVESTMENT REPORT" in text.upper() or "CFA Investment Report" in text:
detailed, report = text, ""
if "CFA INVESTMENT REPORT" in text.upper():
parts = re.split(r"\n\s*(?:CFA INVESTMENT REPORT|CFA Investment Report)\s*\n", text, maxsplit=1, flags=re.IGNORECASE)
detailed = (parts[0].replace("DETAILED ANALYSIS", "").strip() if parts else "").strip() or text
report = parts[1].strip() if len(parts) > 1 else ""
if not detailed:
detailed = text
else:
detailed = text
report = "(CFA Investment Report section not clearly separated; full analysis above.)"
return detailed, report
def get_metrics_table_from_ai(api_key: str, item8_text: str, ticker: str) -> pd.DataFrame:
"""Extract Revenue, Net Income, Operating Cash Flow from Item 8 only (pre-filtered)."""
model = get_gemini_model(api_key)
excerpt = smart_chunk(item8_text, max_chars=25000)
prompt = f"""You are a financial analyst. From the 10-K Item 8 excerpt below for company {ticker}, extract the following for the most recent 35 fiscal years. Focus only on figures; omit filler text.
- Revenue (or Net sales)
- Net Income (or Net earnings attributable to common shareholders)
- Cash flows from operating activities (Operating Cash Flow)
Reply with ONLY a single JSON object, no other text. Use fiscal years as keys (e.g. "2023", "2022", "2021").
Format:
{{"Revenue": {{"2023": 123.45, "2022": 100.0}}, "Net Income": {{"2023": 20.0, "2022": 18.0}}, "Operating Cash Flow": {{"2023": 25.0, "2022": 22.0}}}}
Use numbers in millions (e.g. 394328 for $394,328 million). If a value is not found, use null.
Item 8 excerpt:
{excerpt}"""
try:
response = _generate_with_retry(
model,
prompt,
{"temperature": 0.1, "max_output_tokens": 1024},
)
except Exception as api_err:
if _is_rate_limit_error(api_err):
raise RuntimeError("Rate limit exceeded. Please try again in a few minutes.") from api_err
raise
if not response or not response.text:
return pd.DataFrame()
text = response.text.strip()
json_match = re.search(r"\{[\s\S]*\}", text)
if not json_match:
return pd.DataFrame()
try:
data = json.loads(json_match.group())
return pd.DataFrame(data)
except Exception:
return pd.DataFrame()
def download_and_extract_sections(ticker: str, email: str) -> tuple[str, str, str]:
"""Download 10-K, pre-filter, extract Item 7 & 8 only. Returns (full_text, item7, item8)."""
Downloader = get_edgar_downloader()
with tempfile.TemporaryDirectory() as tmpdir:
download_root = Path(tmpdir)
@@ -342,57 +345,37 @@ def download_and_extract_sections(ticker: str, email: str) -> tuple[str, str, st
full_text = get_main_10k_text(filing_dir)
if not full_text:
raise ValueError("Could not extract text from the 10-K.")
text_from_item7 = prefilter_after_item7(full_text)
item7 = find_item_section(text_from_item7, 7, ["Management's Discussion", "MD&A", "Analysis"])
item8 = find_item_section(text_from_item7, 8, ["Financial Statements", "Consolidated"])
if not item7:
item7 = smart_chunk(text_from_item7[:120000], max_chars=30000)
item7 = smart_chunk(text_from_item7[:120000], max_chars=20000)
if not item8:
remainder = text_from_item7[100000:220000] if len(text_from_item7) > 100000 else text_from_item7
item8 = smart_chunk(remainder, max_chars=30000)
item8 = smart_chunk(remainder, max_chars=20000)
return full_text, item7, item8
def run_analysis(ticker: str, api_key: str, email: str, analysis_only: bool = False) -> tuple[str, str, str, pd.DataFrame]:
"""Download 10-K, extract Item 7/8, call Gemini; return summary, report, full_text, metrics table."""
full_text, item7, item8 = download_and_extract_sections(ticker, email)
detailed_summary, cfa_report = get_ai_summary_and_report(api_key, full_text, item7, item8, ticker)
full_text, item7, _ = download_and_extract_sections(ticker, email)
detailed_summary, cfa_report = get_ai_summary_and_report(api_key, item7, ticker)
if analysis_only:
df_metrics = pd.DataFrame()
else:
time.sleep(DELAY_BETWEEN_CALLS_SEC)
df_metrics = get_metrics_table_from_ai(api_key, item8, ticker)
df_metrics = get_metrics_from_yfinance(ticker)
return detailed_summary, cfa_report, full_text, df_metrics
# ---------- Streamlit UI ----------
st.set_page_config(page_title="10-K Financial Analyzer", layout="wide")
st.title("10-K Financial Analyzer")
st.caption("Download 10-K from SEC EDGAR; view detailed analysis and a CFA-style investment report. Powered by Google Gemini.")
st.caption("Hybrid: 10-K Item 7 (MD&A) → Gemini for sentiment & risks; financial metrics from yfinance. British English.")
with st.sidebar:
st.header("Settings")
google_api_key = st.text_input(
"Google API Key (Gemini)",
type="password",
value=os.environ.get("GOOGLE_API_KEY", ""),
help="Obtain from https://aistudio.google.com/apikey (Google AI Studio).",
)
email = st.text_input(
"SEC EDGAR Email Address",
value=os.environ.get("SEC_EDGAR_EMAIL", ""),
help="Required for SEC programmatic download policy compliance.",
)
analysis_only = st.checkbox(
"Analysis only (1 API call)",
value=False,
help="Skip metrics table to use only 1 API call. Turn on if you often hit rate limits.",
)
google_api_key = st.text_input("Google API Key (Gemini)", type="password", value=os.environ.get("GOOGLE_API_KEY", ""), help="Obtain from https://aistudio.google.com/apikey")
email = st.text_input("SEC EDGAR Email Address", value=os.environ.get("SEC_EDGAR_EMAIL", ""), help="Required for SEC programmatic download policy compliance.")
analysis_only = st.checkbox("Analysis only (1 API call)", value=False, help="Skip metrics table to use only 1 API call.")
st.session_state["google_api_key"] = google_api_key
st.session_state["email"] = email
st.session_state["analysis_only"] = analysis_only
@@ -401,7 +384,6 @@ ticker = st.text_input("Stock Ticker (e.g. AAPL, MSFT)", value="AAPL", max_chars
if not ticker:
st.info("Enter a ticker and click 'Run Analysis', or pick one from the S&P 500 list below.")
# S&P 500 sample: (Company name, Ticker) shown at bottom
SP500_SAMPLE = [
("Apple Inc.", "AAPL"), ("Microsoft Corporation", "MSFT"), ("Amazon.com Inc.", "AMZN"),
("NVIDIA Corporation", "NVDA"), ("Alphabet Inc. (Google)", "GOOGL"), ("Meta Platforms Inc. (Facebook)", "META"),
@@ -409,30 +391,13 @@ SP500_SAMPLE = [
("Visa Inc.", "V"), ("UnitedHealth Group Inc.", "UNH"), ("Procter & Gamble Co.", "PG"),
("Exxon Mobil Corporation", "XOM"), ("Johnson & Johnson", "JNJ"), ("Mastercard Inc.", "MA"),
("Chevron Corporation", "CVX"), ("Home Depot Inc.", "HD"), ("Merck & Co. Inc.", "MRK"),
("AbbVie Inc.", "ABBV"), ("Costco Wholesale Corporation", "COST"),
("PepsiCo Inc.", "PEP"), ("Coca-Cola Company", "KO"), ("Pfizer Inc.", "PFE"),
("Walmart Inc.", "WMT"), ("Netflix Inc.", "NFLX"), ("Adobe Inc.", "ADBE"),
("Salesforce Inc.", "CRM"), ("Comcast Corporation", "CMCSA"), ("Cisco Systems Inc.", "CSCO"),
("AbbVie Inc.", "ABBV"), ("Costco Wholesale Corporation", "COST"), ("PepsiCo Inc.", "PEP"),
("Coca-Cola Company", "KO"), ("Pfizer Inc.", "PFE"), ("Walmart Inc.", "WMT"), ("Netflix Inc.", "NFLX"),
("Adobe Inc.", "ADBE"), ("Salesforce Inc.", "CRM"), ("Comcast Corporation", "CMCSA"), ("Cisco Systems Inc.", "CSCO"),
("Oracle Corporation", "ORCL"), ("Intel Corporation", "INTC"), ("American Express Company", "AXP"),
("Bank of America Corp.", "BAC"), ("Wells Fargo & Company", "WFC"), ("Verizon Communications Inc.", "VZ"),
("AT&T Inc.", "T"), ("Disney (Walt Disney Co.)", "DIS"), ("Nike Inc.", "NKE"),
("McDonald's Corporation", "MCD"), ("Starbucks Corporation", "SBUX"), ("Goldman Sachs Group Inc.", "GS"),
("Morgan Stanley", "MS"), ("Boeing Company", "BA"), ("Caterpillar Inc.", "CAT"),
("3M Company", "MMM"), ("Honeywell International Inc.", "HON"), ("IBM (International Business Machines)", "IBM"),
("Qualcomm Inc.", "QCOM"), ("Texas Instruments Inc.", "TXN"), ("Amgen Inc.", "AMGN"),
("Gilead Sciences Inc.", "GILD"), ("Bristol-Myers Squibb Company", "BMY"), ("Eli Lilly and Company", "LLY"),
("Union Pacific Corporation", "UNP"), ("Lockheed Martin Corporation", "LMT"), ("Raytheon Technologies Corp.", "RTX"),
("Target Corporation", "TGT"), ("Lowe's Companies Inc.", "LOW"), ("Booking Holdings Inc.", "BKNG"),
("PayPal Holdings Inc.", "PYPL"), ("Broadcom Inc.", "AVGO"), ("Schlumberger Ltd.", "SLB"),
("ConocoPhillips", "COP"), ("Phillips 66", "PSX"),
("Ford Motor Company", "F"), ("General Motors Company", "GM"), ("General Electric Company", "GE"),
("FedEx Corporation", "FDX"), ("United Parcel Service Inc.", "UPS"), ("Delta Air Lines Inc.", "DAL"),
("American Airlines Group Inc.", "AAL"), ("Southwest Airlines Co.", "LUV"),
("Abbott Laboratories", "ABT"), ("Thermo Fisher Scientific Inc.", "TMO"), ("Danaher Corporation", "DHR"),
("Accenture plc", "ACN"), ("Intuit Inc.", "INTU"),
("ServiceNow Inc.", "NOW"), ("Workday Inc.", "WDAY"), ("Snowflake Inc.", "SNOW"),
("Zoom Video Communications Inc.", "ZM"), ("Spotify Technology S.A.", "SPOT"), ("Uber Technologies Inc.", "UBER"),
("Airbnb Inc.", "ABNB"), ("Moderna Inc.", "MRNA"), ("Regeneron Pharmaceuticals Inc.", "REGN"),
("AT&T Inc.", "T"), ("Disney (Walt Disney Co.)", "DIS"), ("Nike Inc.", "NKE"), ("McDonald's Corporation", "MCD"),
("Starbucks Corporation", "SBUX"), ("Goldman Sachs Group Inc.", "GS"), ("Morgan Stanley", "MS"),
]
st.caption("Select a ticker above or choose from the list below.")
@@ -444,37 +409,35 @@ if st.button("Run Analysis"):
api_key = st.session_state.get("google_api_key", "")
email = st.session_state.get("email", "")
if not api_key:
st.error("Please enter your Google API Key (Gemini) in Settings. You may also set GOOGLE_API_KEY in a .env file.")
st.error("Please enter your Google API Key (Gemini) in Settings.")
st.stop()
if not email:
st.error("Please enter your SEC EDGAR email address in Settings.")
st.stop()
analysis_only = st.session_state.get("analysis_only", False)
try:
with st.spinner("Step 1/2: Downloading 10-K and extracting Item 7 & 8 (selective sections only)..."):
full_text, item7, item8 = download_and_extract_sections(ticker, email)
with st.spinner("Step 2/2: Running Gemini analysis (typically 3090s; if rate limited, we wait 60s then retry)..."):
detailed_summary, cfa_report = get_ai_summary_and_report(api_key, full_text, item7, item8, ticker)
with st.spinner("Step 1/2: Downloading 10-K and extracting Item 7 (MD&A)..."):
full_text, item7, _ = download_and_extract_sections(ticker, email)
with st.spinner("Step 2/2: Running Gemini (qualitative analysis) and fetching financial metrics..."):
detailed_summary, cfa_report = get_ai_summary_and_report(api_key, item7, ticker)
if analysis_only:
df_metrics = pd.DataFrame()
else:
time.sleep(DELAY_BETWEEN_CALLS_SEC)
df_metrics = get_metrics_table_from_ai(api_key, item8, ticker)
df_metrics = get_metrics_from_yfinance(ticker)
st.success("Analysis complete.")
st.subheader("Detailed Analysis (Financial Health, Profitability, Key Risks)")
st.subheader("Detailed Analysis (Strategy, Risks, Sentiment — from Item 7 MD&A)")
st.markdown(detailed_summary)
st.subheader("CFA Investment Report")
st.markdown(cfa_report)
st.subheader("Key Financial Metrics (Revenue, Net Income, Operating Cash Flow)")
st.subheader("Key Financial Metrics (Revenue, Net Income, Operating Cash Flow) — from yfinance")
if not df_metrics.empty:
st.dataframe(df_metrics, use_container_width=True)
st.caption("Values in millions (USD). Source: yfinance.")
elif analysis_only:
st.info("Metrics skipped (Analysis only mode). Turn off 'Analysis only' in Settings to fetch metrics.")
st.info("Metrics skipped (Analysis only mode).")
else:
st.info("No metrics extracted. Check the full Item 8 text.")
st.info("No metrics available for this ticker from yfinance.")
with st.expander("View excerpt of extracted 10-K text"):
st.text(full_text[:15000] + ("..." if len(full_text) > 15000 else ""))
@@ -485,29 +448,28 @@ if st.button("Run Analysis"):
except RuntimeError as e:
st.error(str(e))
if analysis_only:
st.warning("You already have **Analysis only** on (1 API call). The limit is on Google's side — wait **25 minutes** without clicking, then press Run Analysis again.")
st.warning("You already have Analysis only on. Wait 25 minutes, then try again.")
else:
st.info("Wait 25 minutes, then try again. Or enable 'Analysis only (1 API call)' in Settings to reduce usage.")
st.info("Wait 25 minutes, or enable Analysis only (1 API call) in Settings.")
except Exception as e:
err_msg = str(e).lower()
if "429" in err_msg or ("resource" in err_msg and "exhausted" in err_msg):
st.error("Rate limit exceeded. Please try again in a few minutes.")
if analysis_only:
st.warning("You already have **Analysis only** on. Google's free tier limit is reached — wait **25 minutes**, then press Run Analysis again (no need to change settings).")
else:
st.info("Wait 25 minutes, then retry. Or enable **Analysis only (1 API call)** in the sidebar.")
st.info("Wait 25 minutes, or enable **Analysis only (1 API call)** in the sidebar.")
elif "404" in err_msg or "not found" in err_msg:
st.error("The selected model is not available. Please try again later or check Google AI Studio for available models.")
st.error("The selected model is not available. Check Google AI Studio for available models.")
elif "timeout" in err_msg or "retryerror" in err_msg or "600" in err_msg:
st.error("Request timed out. The API took too long to respond.")
st.info("Try again, or enable **Analysis only (1 API call)** to send less data.")
else:
st.error("An error occurred. Please try again later.")
st.caption("If the problem persists, check your API key and internet connection.")
with st.expander("Error details (for troubleshooting)"):
st.code(repr(e), language="text")
st.caption("Share this with support if the issue continues.")
st.divider()
st.subheader("S&P 500 companies (sample) — Company name & Ticker")
st.caption("Click a row to copy the ticker, or type it in the box above.")
st.caption("Type a ticker from the list into the box above.")
df_sp = pd.DataFrame(SP500_SAMPLE, columns=["Company name", "Ticker"])
with st.expander("Show list", expanded=True):
st.dataframe(df_sp, use_container_width=True, hide_index=True)