"""Webpage content fetching with caching, PDF extraction, and summarization helpers.""" import copy import io import json import os import re import logging from datetime import datetime, timedelta from typing import List from urllib.parse import urljoin, urlsplit, quote import httpx from bs4 import BeautifulSoup from src.constants import WEB_FETCH_SOFT_MAX_BYTES, WEB_FETCH_HARD_MAX_BYTES, WEB_FETCH_USER_AGENT from src import outbound_fetch as _outbound_fetch from .analytics import RateLimitError, error_logger from .cache import ( CONTENT_CACHE_DIR, content_cache_index, generate_cache_key, cleanup_cache, ) logger = logging.getLogger(__name__) def _is_private_address(addr): return _outbound_fetch._is_private_address(addr) def _resolve_hostname_ips(hostname): return _outbound_fetch._resolve_hostname_ips(hostname) def _public_http_url(url): return _outbound_fetch._public_http_url(url, resolver=_resolve_hostname_ips) def _resolve_public_ips(url): return _outbound_fetch._resolve_public_ips(url, resolver=_resolve_hostname_ips) _PinnedBackend = _outbound_fetch._PinnedBackend _PinnedTransport = _outbound_fetch._PinnedTransport BodyTooLargeError = _outbound_fetch.BodyTooLargeError _CappedFetch = _outbound_fetch._CappedFetch def _get_public_url(url, headers, timeout, max_redirects=5, max_bytes=None): return _outbound_fetch._get_public_url( url, headers=headers, timeout=timeout, max_redirects=max_redirects, max_bytes=max_bytes, resolve_public_ips=_resolve_public_ips, transport_factory=_PinnedTransport, ) # PDF extraction (optional dependency) try: from pdfminer.high_level import extract_text as pdf_extract_text except ImportError: pdf_extract_text = None # type: ignore try: from pypdf import PdfReader except ImportError: PdfReader = None # type: ignore def _extract_pdf_text(pdf_bytes: bytes, url: str = "") -> str: """Extract PDF text with available permissive dependencies.""" # Prefer pypdf's layout mode. Plain text extraction and pdfminer often # collapse table columns into an ambiguous number stream, which makes a # correct source passage easy for the model to misread. if PdfReader is not None: try: reader = PdfReader(io.BytesIO(pdf_bytes)) pages: List[str] = [] for idx, page in enumerate(reader.pages): try: try: page_text = page.extract_text(extraction_mode="layout") or "" except TypeError: page_text = page.extract_text() or "" except Exception as e: logger.warning(f"pypdf extraction failed for {url} page {idx + 1}: {e}") page_text = "" if page_text.strip(): pages.append(f"[Page {idx + 1}]\n{page_text.strip()}") if pages: return "\n\n".join(pages) except Exception as e: logger.warning(f"pypdf extraction failed for {url}: {e}") if pdf_extract_text is not None: try: text = pdf_extract_text(io.BytesIO(pdf_bytes)) or "" if text.strip(): return text except Exception as e: logger.warning(f"pdfminer extraction failed for {url}: {e}") if PdfReader is None and pdf_extract_text is None: logger.error("No PDF text extractor installed; install pdfminer.six or pypdf.") return "" # ---------------------------------------------------------------------- # HTML extraction helpers # ---------------------------------------------------------------------- def _extract_meta(soup: BeautifulSoup) -> dict: """Pull meta description and keywords if present.""" description = "" keywords = "" desc_tag = soup.find("meta", attrs={"name": re.compile("description", re.I)}) if desc_tag and desc_tag.get("content"): description = desc_tag["content"].strip() kw_tag = soup.find("meta", attrs={"name": re.compile("keywords", re.I)}) if kw_tag and kw_tag.get("content"): keywords = kw_tag["content"].strip() return {"description": description, "keywords": keywords} def _extract_og_image(soup: BeautifulSoup) -> str: """Extract the best representative image URL from meta tags. Only returns absolute http(s) URLs -- skips relative paths and data URIs. """ candidates = [] for prop in ("og:image", "og:image:url", "og:image:secure_url"): tag = soup.find("meta", attrs={"property": prop}) if tag and tag.get("content", "").strip(): candidates.append(tag["content"].strip()) tag = soup.find("meta", attrs={"name": "twitter:image"}) if tag and tag.get("content", "").strip(): candidates.append(tag["content"].strip()) tag = soup.find("meta", attrs={"name": "thumbnail"}) if tag and tag.get("content", "").strip(): candidates.append(tag["content"].strip()) for url in candidates: if url.startswith(("https://", "http://")) and not url.endswith((".svg", ".ico")): return url return "" def _linked_text(area, base_url: str) -> str: """Preserve observed anchor destinations and block order without fetching links.""" area = copy.copy(area) for anchor in area.find_all('a', href=True): label = ' '.join(anchor.get_text(' ', strip=True).split()) href = str(anchor.get('href') or '').strip() if not label or not href or href.startswith('#'): continue target = urljoin(base_url, href) try: parsed = urlsplit(target) if parsed.scheme not in {'http', 'https'} or not parsed.hostname or parsed.username or parsed.password: continue except ValueError: continue label = re.sub(r'([\\\[\]])', r'\\\1', label) target = quote(target, safe=":/?#[]@!$&'()*+,;=%~_-.") anchor.replace_with(f'[{label}](<{target}>)') for block in area.find_all(['p', 'li', 'tr', 'h1', 'h2', 'h3', 'h4', 'article', 'br']): block.insert_before('\n') block.insert_after('\n') return '\n'.join(' '.join(line.split()) for line in area.get_text(' ', strip=False).splitlines() if line.strip()) def _page_entries(areas, base_url: str) -> list[dict]: """Recognize repeated listing structures, retaining DOM order, not popularity.""" entries = [] seen = set() for area in areas: nodes = ([area] if area.name == 'article' else []) + area.find_all(['li', 'article', 'tr']) for node in nodes: anchor = None if node.name == 'tr': cells = node.find_all(['td', 'th'], recursive=False) if cells and re.fullmatch(r'\d+[.)]?', cells[0].get_text(strip=True)): anchor = next((a for a in node.find_all('a', href=True) if a.get_text(strip=True)), None) elif node.name == 'article': heading = node.find(['h1', 'h2', 'h3', 'h4']) anchor = heading.find('a', href=True) if heading else None elif node.parent and node.parent.name == 'ol': anchor = node.find('a', href=True) if not anchor: continue title = ' '.join(anchor.get_text(' ', strip=True).split()) href = str(anchor.get('href') or '').strip() if not title or not href or href.startswith('#'): continue url = urljoin(base_url, href) try: parsed = urlsplit(url) if parsed.scheme not in {'http', 'https'} or not parsed.hostname or parsed.username or parsed.password: continue except ValueError: continue if (title, url) not in seen: seen.add((title, url)) entries.append({'title': title, 'url': url}) if len(entries) == 100: return entries return entries if len(entries) >= 2 else [] def _extract_lists(soup: BeautifulSoup) -> List[List[str]]: """Return a list of lists, each inner list representing a