"""NASA fetcher โ€” public-domain imagery/video. JSON API: https://images-api.nasa.gov/search (keyless). Each search item links an asset-collection JSON (the item's `href`) listing the concrete file URLs; the chosen media file is resolved from there. NASA media is public domain, but some items embed third-party content โ€” flagged in notes for review (ยง6.4). """ from __future__ import annotations from tools.ingest.base import Candidate SEARCH = "https://images-api.nasa.gov/search" _THIRD_PARTY_CAVEAT = "[NASA: may embed third-party content โ€” verify before use]" _VIDEO_EXTS = (".mp4", ".mov", ".m4v", ".webm") _AUDIO_EXTS = (".mp3", ".m4a", ".wav", ".flac", ".ogg") def _ext_from_url(url: str) -> str: tail = url.rsplit("/", 1)[-1].split("?", 1)[0] return tail.rsplit(".", 1)[-1].lower() if "." in tail else "" def _pick_asset(assets, media_type: str) -> str: if not isinstance(assets, list): return "" prefs = _AUDIO_EXTS if media_type == "audio" else _VIDEO_EXTS for ext in prefs: for url in assets: if isinstance(url, str) and url.lower().endswith(ext): return url for url in assets: if isinstance(url, str) and not url.lower().endswith((".jpg", ".png", ".json")): return url return "" class NasaFetcher: archive = "nasa" def __init__(self, client): self.client = client def _candidate(self, item) -> Candidate: data0 = (item.get("data") or [{}])[0] nasa_id = data0.get("nasa_id", "") title = data0.get("title", "") media_type = data0.get("media_type", "") description = (data0.get("description", "") or "").strip() description = f"{description} {_THIRD_PARTY_CAVEAT}".strip() href = item.get("href", "") assets = self.client.get_json(href) if href else [] media_url = _pick_asset(assets, media_type) return Candidate( source_archive=self.archive, source_url=f"https://images.nasa.gov/details-{nasa_id}", media_url=media_url, title=title, license="public_domain", attribution="", suggested_id=f"nasa-{nasa_id}", media_ext=_ext_from_url(media_url), description=description, ) def search(self, query: str, *, limit: int) -> list[Candidate]: url = f"{SEARCH}?q={query}&media_type=video" data = self.client.get_json(url) items = (data.get("collection", {}).get("items") or [])[:limit] return [self._candidate(i) for i in items] def resolve(self, identifier: str) -> Candidate: url = f"{SEARCH}?nasa_id={identifier}" data = self.client.get_json(url) items = data.get("collection", {}).get("items") or [] if not items: raise ValueError(f"nasa: no item for {identifier!r}") return self._candidate(items[0])