"""SEC EDGAR 10-K download, parsing, section extraction, and caching. Handles the full pipeline from downloading a 10-K filing via ``sec_edgar_downloader`` through HTML stripping to isolating individual Item sections (1A, 3, 7, 8, 9A) and persisting the cleaned text to a local JSON cache under ``data/``. """ import json import logging import re import tempfile from pathlib import Path from typing import Any, Dict, List, Optional import httpx from bs4 import BeautifulSoup from bs4.element import Comment, Tag from server.services.text_chunker import clean_text_for_llm, smart_chunk logger = logging.getLogger(__name__) # --------------------------------------------------------------------------- # Paths # --------------------------------------------------------------------------- _DATA_DIR: Path = Path(__file__).resolve().parents[3] / "data" # --------------------------------------------------------------------------- # SEC EDGAR Filing URL Resolver # --------------------------------------------------------------------------- _CIK_CACHE: Dict[str, int] = {} _SEC_HEADERS = {"User-Agent": "ATLAS-Terminal admin@atlas.local"} def _resolve_cik(ticker: str) -> Optional[int]: """Resolve ticker → CIK via SEC's company_tickers.json.""" t = ticker.upper().strip() if t in _CIK_CACHE: return _CIK_CACHE[t] try: resp = httpx.get( "https://www.sec.gov/files/company_tickers.json", headers=_SEC_HEADERS, timeout=15, ) resp.raise_for_status() data = resp.json() for entry in data.values(): tk = entry.get("ticker", "") cik = entry.get("cik_str") if tk: _CIK_CACHE[tk.upper()] = int(cik) return _CIK_CACHE.get(t) except Exception: logger.warning("Failed to resolve CIK for %s", t) return None def get_sec_filing_url(ticker: str) -> Optional[str]: """Return the URL of the latest 10-K filing document on SEC EDGAR.""" cik = _resolve_cik(ticker) if cik is None: return None cik_padded = str(cik).zfill(10) try: resp = httpx.get( f"https://data.sec.gov/submissions/CIK{cik_padded}.json", headers=_SEC_HEADERS, timeout=15, ) resp.raise_for_status() data: Dict[str, Any] = resp.json() recent = data.get("filings", {}).get("recent", {}) forms = recent.get("form", []) accessions = recent.get("accessionNumber", []) docs = recent.get("primaryDocument", []) for i, form in enumerate(forms): if form in ("10-K", "10-K/A"): acc_no_dash = accessions[i].replace("-", "") return ( f"https://www.sec.gov/Archives/edgar/data" f"/{cik_padded}/{acc_no_dash}/{docs[i]}" ) except Exception: logger.warning("Failed to get filing URL for %s", ticker) return None # --------------------------------------------------------------------------- # Section-header regex patterns # --------------------------------------------------------------------------- ITEM1A_PATTERNS: List[str] = [ r"Item\s+1A\s*[.:]\s*Risk\s+Factors", r"ITEM\s+1A\s*[.:]\s*Risk\s+Factors", ] ITEM7_PATTERNS: List[str] = [ r"Item\s+7\s*[.:]\s*Management['\u2019]s\s+Discussion\s+and\s+Analysis", r"ITEM\s+7\s*[.:]\s*Management['\u2019]s\s+Discussion", r"Item\s+7\s*[.:]\s*[\w\s]+MD&A", ] ITEM8_PATTERNS: List[str] = [ r"Item\s+8\s*[.:]\s*Financial\s+Statements", r"ITEM\s+8\s*[.:]\s*Financial\s+Statements", ] ITEM3_PATTERNS: List[str] = [ r"Item\s+3\s*[.:]\s*Legal\s+Proceedings", r"ITEM\s+3\s*[.:]\s*Legal\s+Proceedings", ] ITEM9A_PATTERNS: List[str] = [ r"Item\s+9A\s*[.:]\s*Controls\s+and\s+Procedures", r"Item\s+9A\s*[.:]\s*Internal\s+Control", r"ITEM\s+9A\s*[.:]\s*Controls", ] # --------------------------------------------------------------------------- # HTML helpers # --------------------------------------------------------------------------- def _slice_html_items_1a_to_9a(raw_html: str) -> str: """Fast string-level slice: keep only Item 1A through end of Item 9A.""" if not raw_html or len(raw_html) < 5000: return raw_html start = -1 for needle in ("Item 1A", "ITEM 1A", "Item 1a"): i = raw_html.find(needle) if i != -1 and (start == -1 or i < start): start = i if start == -1: m = re.search(r"Item\s+1A\s", raw_html, re.IGNORECASE) start = m.start() if m else 0 else: start = max(0, start - 200) search_region = raw_html[start:] end_match = re.search( r"Item\s+10\s|Item\s+12\s|Part\s+III\b|PART\s+III\b", search_region, re.IGNORECASE, ) end = start + end_match.start() if end_match else len(raw_html) end = min(end, start + 8_000_000) return raw_html[start:end] def _extract_text_from_html_string(html_str: str) -> str: """Parse an HTML string and return plain text (tables/scripts removed).""" if not html_str or not html_str.strip(): return "" try: soup = BeautifulSoup(html_str, "lxml") except Exception: soup = BeautifulSoup(html_str, "html.parser") for tag in soup.find_all(["table", "img", "svg", "style", "script"]): tag.decompose() return soup.get_text(separator="\n", strip=True) def extract_text_from_html(html_path: Path) -> str: """Read an HTML file, slice to Items 1A-9A, and return plain text.""" try: with open(html_path, "r", encoding="utf-8", errors="replace") as f: raw = f.read() except Exception: with open(html_path, "r", encoding="latin-1", errors="replace") as f: raw = f.read() chunk = _slice_html_items_1a_to_9a(raw) return _extract_text_from_html_string(chunk) def extract_text_from_file(file_path: Path) -> str: """Extract plain text from an HTML or TXT file.""" suf = file_path.suffix.lower() if suf in (".htm", ".html"): return extract_text_from_html(file_path) if suf == ".txt": with open(file_path, "r", encoding="utf-8", errors="replace") as f: text = f.read() text = re.sub(r"<[^>]+>", " ", text) text = re.sub(r"\s+", " ", text) return text return "" # --------------------------------------------------------------------------- # Section finders # --------------------------------------------------------------------------- def _find_section_start(text: str, patterns: List[str], item_num: int) -> int: """Return character offset where *item_num* section begins, or -1.""" for pat in patterns: m = re.search(pat, text, re.IGNORECASE) if m: return m.start() m = re.search(r"\bItem\s+" + str(item_num) + r"\b", text, re.IGNORECASE) return m.start() if m else -1 def find_item_section_generic( text: str, patterns: List[str], item_num: int, title_keywords: List[str], max_chars: int = 120_000, ) -> str: """Extract a single Item section from full 10-K text.""" start = _find_section_start(text, patterns, item_num) if start == -1: pattern = re.compile( r"\bItem\s+" + str(item_num) + r"\b[.\s]*[^\n]*(" + "|".join(re.escape(k) for k in title_keywords) + r")?", re.IGNORECASE, ) match = pattern.search(text) if not match: return "" start = match.start() next_item = re.search(r"\n\s*Item\s+\d+[A-Z]?\s+", text[start + 100:], re.IGNORECASE) end = start + 100 + next_item.start() if next_item else min(start + max_chars, len(text)) return text[start:end].strip() def _extract_item_from_full( text: str, patterns: List[str], item_num: int, keywords: List[str], max_chars: int = 60_000, ) -> str: """Extract one item section from full 10-K text.""" start = _find_section_start(text, patterns, item_num) if start < 0: pat = re.compile( r"\bItem\s+" + str(item_num) + r"[A-Z]?\b[.\s]*[^\n]*", re.IGNORECASE, ) match = pat.search(text) start = match.start() if match else -1 if start < 0: return "" next_item = re.search(r"\n\s*Item\s+\d+[A-Z]?\s+", text[start + 100:], re.IGNORECASE) end = start + 100 + next_item.start() if next_item else min(start + max_chars, len(text)) return text[start:end].strip() # --------------------------------------------------------------------------- # Filing directory helpers # --------------------------------------------------------------------------- def _get_edgar_downloader() -> type: """Lazy import of ``sec_edgar_downloader.Downloader``.""" from sec_edgar_downloader import Downloader return Downloader def find_downloaded_10k_path(download_root: Path, ticker: str) -> Optional[Path]: """Locate the most recent 10-K filing directory on disk.""" ticker_upper = ticker.upper() for base in (download_root / "sec-edgar-filings", download_root): path_10k = base / ticker_upper / "10-K" if path_10k.exists(): subdirs = sorted( [d for d in path_10k.iterdir() if d.is_dir()], key=lambda x: x.name, reverse=True, ) if subdirs: return subdirs[0] for base in (download_root / "sec-edgar-filings", download_root): if not base.exists(): continue for company_dir in base.iterdir(): if not company_dir.is_dir(): continue path_10k = company_dir / "10-K" if path_10k.exists(): subdirs = sorted( [d for d in path_10k.iterdir() if d.is_dir()], key=lambda x: x.name, reverse=True, ) if subdirs: return subdirs[0] return None def find_all_10k_filing_dirs(download_root: Path, ticker: str) -> List[Path]: """Return all 10-K filing directories sorted newest-first.""" ticker_upper = ticker.upper() for base in (download_root / "sec-edgar-filings", download_root): path_10k = base / ticker_upper / "10-K" if path_10k.exists(): return sorted( [d for d in path_10k.iterdir() if d.is_dir()], key=lambda x: x.name, reverse=True, ) return [] def get_main_10k_text(filing_dir: Path) -> str: """Return the longest extracted text from all files in *filing_dir*.""" all_text: List[tuple] = [] for ext in ("*.htm", "*.html", "*.txt"): for path in filing_dir.rglob(ext): try: t = extract_text_from_file(path) if len(t) > 1000: all_text.append((path, t)) except Exception: continue if not all_text: return "" _, main_text = max(all_text, key=lambda x: len(x[1])) return main_text def get_main_10k_html_path(filing_dir: Path) -> Optional[Path]: """Pick the largest ``.htm`` / ``.html`` file (primary 10-K document).""" best: Optional[Path] = None best_size = 0 for path in filing_dir.rglob("*.htm*"): try: sz = path.stat().st_size if sz > best_size: best_size = sz best = path except OSError: continue return best def read_main_10k_html_raw(filing_dir: Path) -> str: """Read raw HTML from the main filing document.""" path = get_main_10k_html_path(filing_dir) if not path: return "" for enc in ("utf-8", "latin-1"): try: return path.read_text(encoding=enc, errors="replace") except Exception: continue return "" def _strip_scripts_keep_html(html: str) -> str: """Remove ``