"""Fetch and parse ADE PublishDetails pages.""" from __future__ import annotations import re from typing import Any from urllib.parse import urljoin from bs4 import BeautifulSoup from sfda_parser.sfda_scraper.detail_pdf import parse_affected_device_list from sfda_parser.sfda_scraper.http_client import ADE_BASE, create_session, request_with_retry DOWNLOAD_PATTERN = re.compile( r'formaction=["\'](?P/BaseService/DownloadDocument\?download=[^"\']+)["\']', re.IGNORECASE, ) TOKEN_PATTERN = re.compile( r'name=["\']__RequestVerificationToken["\'][^>]*value=["\']([^"\']+)["\']', re.IGNORECASE, ) PUBLISH_ID_PATTERN = re.compile(r"/Fsca/PublishDetails/(\d+)", re.IGNORECASE) HTML_FIELD_MAP = { "reference_number": "ReferenceNumber", "manufacturer": "ManufacturerName", "product_trade_name": "ProductTradeName", "model": "ModelNbr", "authorized_representative": "AuthorizedRepresentative", "problem_reason": "ProblemReason", "action": "Action", } class NoPdfDownloadLinkError(ValueError): """ADE PublishDetails page has no DownloadDocument link.""" def _field_value(soup: BeautifulSoup, field_name: str) -> str: el = soup.find(attrs={"name": field_name}) if not el: return "" if el.name == "textarea": return el.get_text(strip=True) return (el.get("value") or "").strip() def _extract_download_path(html: str) -> str | None: match = DOWNLOAD_PATTERN.search(html) if match: return match.group("path") soup = BeautifulSoup(html, "html.parser") button = soup.select_one("button[formaction*='DownloadDocument']") if button and button.get("formaction"): return button["formaction"] return None def _extract_verification_token(html: str) -> str | None: match = TOKEN_PATTERN.search(html) if match: return match.group(1) soup = BeautifulSoup(html, "html.parser") inp = soup.find("input", {"name": "__RequestVerificationToken"}) if inp and inp.get("value"): return inp["value"] return None def parse_publish_details_html(html: str, detail_url: str = "") -> dict[str, Any]: soup = BeautifulSoup(html, "html.parser") publish_id = "" if detail_url: match = PUBLISH_ID_PATTERN.search(detail_url) if match: publish_id = match.group(1) download_path = _extract_download_path(html) data: dict[str, Any] = { "detail_url": detail_url, "publish_id": publish_id, "pdf_download_url": urljoin(ADE_BASE, download_path) if download_path else "", } for key, field_name in HTML_FIELD_MAP.items(): data[key] = _field_value(soup, field_name) return data def fetch_publish_details_html(detail_url: str, session=None) -> dict[str, Any]: session = session or create_session() response = request_with_retry(session, "GET", detail_url, timeout=60) return parse_publish_details_html(response.text, detail_url=detail_url) def fetch_publish_details( detail_url: str, session=None, include_device_list_from_pdf: bool = False, ) -> dict[str, Any]: """ADE page fields; optionally parse Material/GTIN/Batch table from attached PDF.""" session = session or create_session() result = fetch_publish_details_html(detail_url, session=session) if include_device_list_from_pdf: try: pdf_bytes = download_detail_pdf(detail_url, session=session) devices = parse_affected_device_list(pdf_bytes) result["device_list_from_pdf"] = [d.as_dict() for d in devices] result["device_list_from_pdf_count"] = len(devices) except Exception as exc: result["device_list_from_pdf"] = [] result["device_list_from_pdf_count"] = 0 result["pdf_parse_error"] = str(exc) else: result["device_list_from_pdf"] = [] result["device_list_from_pdf_count"] = 0 return result def resolve_detail_url_for_ncmdr_ref(ncmdr_ref: str, session=None) -> str | None: """Resolve ADE URL from recent weekly PDFs when not stored on SFDA Entries.""" from sfda_parser.sfda_scraper.weekly_list import fetch_recent_weekly_alerts from sfda_parser.sfda_scraper.weekly_pdf import parse_safety_alerts_from_pdf session = session or create_session() target = (ncmdr_ref or "").strip().upper() for weekly in fetch_recent_weekly_alerts(session=session): pdf_response = session.get(weekly.pdf_url, timeout=120) pdf_response.raise_for_status() for alert in parse_safety_alerts_from_pdf(pdf_response.content): if alert.ncmdr_ref.strip().upper() == target: return alert.detail_url return None def _download_detail_pdf_from_html( html: str, detail_url: str, session ) -> bytes: download_path = _extract_download_path(html) token = _extract_verification_token(html) if not download_path: raise NoPdfDownloadLinkError(f"No PDF download link on ADE page: {detail_url}") download_url = urljoin(ADE_BASE, download_path) headers = {"Referer": detail_url} data = {} if token: data["__RequestVerificationToken"] = token pdf_response = request_with_retry( session, "POST", download_url, data=data, headers=headers, timeout=120, ) content_type = (pdf_response.headers.get("Content-Type") or "").lower() if "pdf" not in content_type and not pdf_response.content.startswith(b"%PDF"): raise ValueError(f"ADE download did not return a PDF for {detail_url}") return pdf_response.content def fetch_ade_detail( detail_url: str, session=None ) -> tuple[bytes | None, dict[str, Any]]: """Fetch ADE detail page; download PDF when available. Returns (pdf_bytes or None, publish_details). PDF is None when the ADE page has no DownloadDocument link. """ session = session or create_session() response = request_with_retry(session, "GET", detail_url, timeout=60) html = response.text publish_details = parse_publish_details_html(html, detail_url=detail_url) try: pdf_bytes = _download_detail_pdf_from_html(html, detail_url, session) except NoPdfDownloadLinkError: return None, publish_details return pdf_bytes, publish_details def download_detail_pdf(detail_url: str, session=None) -> bytes: pdf_bytes, _publish_details = fetch_ade_detail(detail_url, session=session) if pdf_bytes is None: raise NoPdfDownloadLinkError(f"No PDF download link on ADE page: {detail_url}") return pdf_bytes