From cecc5a0f61d1b95ce0fd5a76c4be3de6a6d68750 Mon Sep 17 00:00:00 2001 From: Ben Stull Date: Thu, 4 Jun 2026 06:32:18 -0700 Subject: [PATCH] feat: LibriVox/NASA/Internet Archive fetchers (+ deferred stubs) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Per sub-project-2 plan Task 9 / spec §6.4. Three keyless first-ship fetchers parse documented JSON APIs via an injected HttpClient; license/attribution go through tools.licensing; NASA third-party + IA no-license cases are flagged in notes for review. musopen/fma/freesound are explicit deferred stubs raising NotImplementedError; freesound documents FREESOUND_API_TOKEN (secret, env-only). Co-Authored-By: Claude Opus 4.8 (1M context) --- tests/test_fetchers.py | 181 +++++++++++++++++++++++++++++++ tools/ingest/fma.py | 26 +++++ tools/ingest/freesound.py | 34 ++++++ tools/ingest/internet_archive.py | 91 ++++++++++++++++ tools/ingest/librivox.py | 71 ++++++++++++ tools/ingest/musopen.py | 25 +++++ tools/ingest/nasa.py | 79 ++++++++++++++ 7 files changed, 507 insertions(+) create mode 100644 tests/test_fetchers.py create mode 100644 tools/ingest/fma.py create mode 100644 tools/ingest/freesound.py create mode 100644 tools/ingest/internet_archive.py create mode 100644 tools/ingest/librivox.py create mode 100644 tools/ingest/musopen.py create mode 100644 tools/ingest/nasa.py diff --git a/tests/test_fetchers.py b/tests/test_fetchers.py new file mode 100644 index 0000000..51ebb0b --- /dev/null +++ b/tests/test_fetchers.py @@ -0,0 +1,181 @@ +import json + +import pytest + +from tools.http import HttpClient +from tools.ingest.internet_archive import InternetArchiveFetcher +from tools.ingest.librivox import LibriVoxFetcher +from tools.ingest.nasa import NasaFetcher + + +class _Resp: + def __init__(self, data: bytes): + self._data = data + + def read(self): + return self._data + + def __enter__(self): + return self + + def __exit__(self, *a): + return False + + +def fake_opener(mapping): + """Dispatch a urllib Request to a canned payload by URL substring.""" + + def opener(req, timeout=None): + url = req.full_url + for key, payload in mapping.items(): + if key in url: + if isinstance(payload, (bytes, bytearray)): + return _Resp(bytes(payload)) + return _Resp(json.dumps(payload).encode("utf-8")) + raise AssertionError(f"unexpected url: {url}") + + return opener + + +def test_librivox_fetcher(): + payload = { + "books": [ + { + "id": "123", + "title": "Meditations", + "url_librivox": "https://librivox.org/meditations/", + "url_zip_file": "https://archive.org/download/meditations/meditations_mp3.zip", + "authors": [{"first_name": "Marcus", "last_name": "Aurelius"}], + "description": "Stoic philosophy.", + } + ] + } + client = HttpClient(opener=fake_opener({"librivox.org/api": payload})) + cands = LibriVoxFetcher(client).search("medit", limit=5) + c = cands[0] + assert c.source_archive == "librivox" + assert c.license == "public_domain" and c.attribution == "" + assert c.suggested_id == "librivox-meditations" + assert c.media_url.endswith(".zip") + assert c.media_ext == "zip" + assert "Aurelius" in c.description + + +def test_nasa_fetcher(): + search_payload = { + "collection": { + "items": [ + { + "data": [ + { + "nasa_id": "as08-14-2383", + "title": "Earthrise", + "description": "View of Earth from the Moon", + "media_type": "video", + } + ], + "href": "https://images-assets.nasa.gov/video/as08-14-2383/collection.json", + } + ] + } + } + asset_payload = [ + "https://images-assets.nasa.gov/video/as08-14-2383/as08-14-2383~orig.mp4", + "https://images-assets.nasa.gov/video/as08-14-2383/as08-14-2383~thumb.jpg", + ] + client = HttpClient( + opener=fake_opener( + { + "images-api.nasa.gov/search": search_payload, + "collection.json": asset_payload, + } + ) + ) + c = NasaFetcher(client).search("earth", limit=3)[0] + assert c.suggested_id == "nasa-as08-14-2383" + assert c.license == "public_domain" and c.attribution == "" + assert c.media_url.endswith("orig.mp4") + assert c.media_ext == "mp4" + assert "Earth" in c.description + + +def test_internet_archive_cc_by(): + meta_payload = { + "metadata": { + "identifier": "earthrise", + "title": "Earthrise", + "licenseurl": "http://creativecommons.org/licenses/by/4.0/", + "creator": "NASA", + "description": "Apollo 8 footage.", + }, + "files": [ + {"name": "earthrise.mp4", "format": "h.264", "source": "original"}, + {"name": "earthrise.png", "format": "PNG", "source": "derivative"}, + ], + "server": "ia800100.us.archive.org", + "dir": "/12/items/earthrise", + } + client = HttpClient(opener=fake_opener({"archive.org/metadata/earthrise": meta_payload})) + c = InternetArchiveFetcher(client).resolve("earthrise") + assert c.suggested_id == "ia-earthrise" + assert c.license == "cc_by" and "NASA" in c.attribution + assert c.media_url == "https://ia800100.us.archive.org/12/items/earthrise/earthrise.mp4" + assert c.media_ext == "mp4" + assert c.source_url == "https://archive.org/details/earthrise" + + +def test_internet_archive_no_license_assumed_public_domain_and_flagged(): + meta_payload = { + "metadata": {"identifier": "oldfilm", "title": "Old Film"}, + "files": [{"name": "oldfilm.mp4", "source": "original"}], + "server": "ia.example.org", + "dir": "/x/items/oldfilm", + } + client = HttpClient(opener=fake_opener({"archive.org/metadata/oldfilm": meta_payload})) + c = InternetArchiveFetcher(client).resolve("oldfilm") + assert c.license == "public_domain" + assert "verify" in c.description.lower() + + +def test_internet_archive_search_resolves_each_hit(): + search_payload = {"response": {"docs": [{"identifier": "earthrise"}]}} + meta_payload = { + "metadata": { + "identifier": "earthrise", + "title": "Earthrise", + "licenseurl": "https://creativecommons.org/publicdomain/mark/1.0/", + }, + "files": [{"name": "earthrise.mp4", "source": "original"}], + "server": "ia.example.org", + "dir": "/x", + } + client = HttpClient( + opener=fake_opener( + { + "advancedsearch.php": search_payload, + "archive.org/metadata/earthrise": meta_payload, + } + ) + ) + cands = InternetArchiveFetcher(client).search("earthrise", limit=5) + assert [c.suggested_id for c in cands] == ["ia-earthrise"] + assert cands[0].license == "public_domain" + + +def test_deferred_stubs_raise_not_implemented(): + from tools.ingest.fma import FmaFetcher + from tools.ingest.freesound import FreesoundFetcher + from tools.ingest.musopen import MusopenFetcher + + with pytest.raises(NotImplementedError): + MusopenFetcher(None).search("x", limit=1) + with pytest.raises(NotImplementedError): + FmaFetcher(None).search("x", limit=1) + with pytest.raises(NotImplementedError): + FreesoundFetcher(None).search("x", limit=1) + + +def test_fetchers_expose_archive_label(): + assert LibriVoxFetcher(None).archive == "librivox" + assert NasaFetcher(None).archive == "nasa" + assert InternetArchiveFetcher(None).archive == "internet_archive" diff --git a/tools/ingest/fma.py b/tools/ingest/fma.py new file mode 100644 index 0000000..e74553a --- /dev/null +++ b/tools/ingest/fma.py @@ -0,0 +1,26 @@ +"""Free Music Archive fetcher — DEFERRED (see spec §6.4). + +CC-licensed music (cc_by / cc_by_nc / cc0). Highest API-stability risk: the +public FMA API has been deprecated/changed, so a real implementation resolves +from a track URL + page metadata. The Fetcher seam is wired; it raises until +implemented. +""" + +from __future__ import annotations + +from tools.ingest.base import Candidate + +_DEFERRED = "deferred — see spec §6.4 (FMA public API deprecated/changed)" + + +class FmaFetcher: + archive = "fma" + + def __init__(self, client): + self.client = client + + def search(self, query: str, *, limit: int) -> list[Candidate]: + raise NotImplementedError(_DEFERRED) + + def resolve(self, identifier: str) -> Candidate: + raise NotImplementedError(_DEFERRED) diff --git a/tools/ingest/freesound.py b/tools/ingest/freesound.py new file mode 100644 index 0000000..d5c7b74 --- /dev/null +++ b/tools/ingest/freesound.py @@ -0,0 +1,34 @@ +"""Freesound fetcher — DEFERRED (see spec §6.4). + +CC-licensed sound effects / field recordings (cc0 / cc_by / cc_by_nc) via API +v2 (https://freesound.org/apiv2). Requires an API token supplied through the +FREESOUND_API_TOKEN environment variable (or the macOS Keychain). That token is +a SECRET: it is read from the environment only and must NEVER be written into a +record, a log line, or a transcript (wgl hard secrets rule, spec §9). + +The Fetcher seam is wired; it raises until implemented. +""" + +from __future__ import annotations + +import os + +from tools.ingest.base import Candidate + +_DEFERRED = "deferred — see spec §6.4 (Freesound requires FREESOUND_API_TOKEN)" +TOKEN_ENV = "FREESOUND_API_TOKEN" + + +class FreesoundFetcher: + archive = "freesound" + + def __init__(self, client, token: str | None = None): + self.client = client + # Read the secret from the environment only; never log or store it. + self._token = token or os.environ.get(TOKEN_ENV) + + def search(self, query: str, *, limit: int) -> list[Candidate]: + raise NotImplementedError(_DEFERRED) + + def resolve(self, identifier: str) -> Candidate: + raise NotImplementedError(_DEFERRED) diff --git a/tools/ingest/internet_archive.py b/tools/ingest/internet_archive.py new file mode 100644 index 0000000..e77bbc0 --- /dev/null +++ b/tools/ingest/internet_archive.py @@ -0,0 +1,91 @@ +"""Internet Archive fetcher (incl. Prelinger). + +Metadata API: https://archive.org/metadata/ (keyless); search via +advancedsearch.php. License is read from `licenseurl`/`rights`/`license` and +normalized through tools.licensing. When no explicit license is present the +item is assumed public_domain and flagged in notes for review (§6.4); an +explicitly non-free license (e.g. All Rights Reserved) raises and is not +ingested. +""" + +from __future__ import annotations + +from tools.ingest.base import Candidate +from tools.licensing import normalize_license + +META = "https://archive.org/metadata/" +SEARCH = "https://archive.org/advancedsearch.php" +_NO_LICENSE_CAVEAT = "[IA: no explicit license — assumed public_domain, verify before use]" +_MEDIA_EXTS = ( + ".mp4", ".mov", ".m4v", ".ogv", ".webm", + ".mp3", ".m4a", ".wav", ".flac", ".ogg", +) + + +def _ext_from_name(name: str) -> str: + return name.rsplit(".", 1)[-1].lower() if "." in name else "" + + +def _pick_file(files): + originals = [f for f in files if f.get("source") == "original"] + for pool in (originals, files): + for f in pool: + if str(f.get("name", "")).lower().endswith(_MEDIA_EXTS): + return f + return None + + +class InternetArchiveFetcher: + archive = "internet_archive" + + def __init__(self, client): + self.client = client + + def resolve(self, identifier: str) -> Candidate: + data = self.client.get_json(f"{META}{identifier}") + meta = data.get("metadata", {}) + files = data.get("files", []) or [] + server = data.get("server", "") + directory = data.get("dir", "") + + chosen = _pick_file(files) + media_url = "" + media_ext = "" + if chosen and server: + name = chosen["name"] + media_url = f"https://{server}{directory}/{name}" + media_ext = _ext_from_name(name) + + raw_license = ( + meta.get("licenseurl") or meta.get("rights") or meta.get("license") or "" + ) + creator = meta.get("creator", "") or "" + if isinstance(creator, list): + creator = ", ".join(str(c) for c in creator) + description = (meta.get("description", "") or "").strip() + + ambiguous = (not raw_license) or ("no known copyright" in raw_license.lower()) + if raw_license: + lic, attr = normalize_license(raw_license, creator=creator) + else: + lic, attr = "public_domain", "" + if ambiguous: + description = f"{description} {_NO_LICENSE_CAVEAT}".strip() + + return Candidate( + source_archive=self.archive, + source_url=f"https://archive.org/details/{identifier}", + media_url=media_url, + title=meta.get("title", identifier), + license=lic, + attribution=attr, + suggested_id=f"ia-{identifier}", + media_ext=media_ext, + description=description, + ) + + def search(self, query: str, *, limit: int) -> list[Candidate]: + url = f"{SEARCH}?q={query}&fl[]=identifier&rows={limit}&output=json" + data = self.client.get_json(url) + docs = (data.get("response", {}).get("docs") or [])[:limit] + return [self.resolve(doc["identifier"]) for doc in docs] diff --git a/tools/ingest/librivox.py b/tools/ingest/librivox.py new file mode 100644 index 0000000..7009023 --- /dev/null +++ b/tools/ingest/librivox.py @@ -0,0 +1,71 @@ +"""LibriVox fetcher — public-domain audiobook recordings. + +JSON API: https://librivox.org/api/feed/audiobooks (keyless). LibriVox is +public domain by charter; the reader/author is credited in notes but +attribution is not required. Section-level (per-track) file resolution is a +future refinement; this resolves the audiobook's zip as the media target. +""" + +from __future__ import annotations + +import re + +from tools.ingest.base import Candidate +from tools.licensing import librivox_license + +BASE = "https://librivox.org/api/feed/audiobooks" + + +def _slug(text: str) -> str: + s = re.sub(r"[^a-z0-9]+", "-", (text or "").lower()).strip("-") + return s or "untitled" + + +def _ext_from_url(url: str, default: str) -> str: + tail = url.rsplit("/", 1)[-1] + return tail.rsplit(".", 1)[-1].lower() if "." in tail else default + + +class LibriVoxFetcher: + archive = "librivox" + + def __init__(self, client): + self.client = client + + def _candidate(self, book) -> Candidate: + title = book.get("title", "") + authors = book.get("authors") or [] + credit = ", ".join( + f"{a.get('first_name', '')} {a.get('last_name', '')}".strip() + for a in authors + ).strip(", ") + media_url = book.get("url_zip_file", "") + description = (book.get("description", "") or "").strip() + if credit: + description = (description + f" (author: {credit})").strip() + lic, attr = librivox_license() + return Candidate( + source_archive=self.archive, + source_url=book.get("url_librivox", ""), + media_url=media_url, + title=title, + license=lic, + attribution=attr, + suggested_id=f"librivox-{_slug(title)}", + media_ext=_ext_from_url(media_url, "zip"), + description=description, + ) + + def search(self, query: str, *, limit: int) -> list[Candidate]: + url = f"{BASE}/title/^{query}?format=json&limit={limit}" + data = self.client.get_json(url) + books = (data.get("books") or [])[:limit] + return [self._candidate(b) for b in books] + + def resolve(self, identifier: str) -> Candidate: + url = f"{BASE}/id/{identifier}?format=json" + data = self.client.get_json(url) + books = data.get("books") or [] + if not books: + raise ValueError(f"librivox: no audiobook for {identifier!r}") + return self._candidate(books[0]) diff --git a/tools/ingest/musopen.py b/tools/ingest/musopen.py new file mode 100644 index 0000000..afe3e47 --- /dev/null +++ b/tools/ingest/musopen.py @@ -0,0 +1,25 @@ +"""Musopen fetcher — DEFERRED (see spec §6.4). + +Public-domain / CC classical music. Access has historically been gated and may +require an API key (a secret — never logged, see spec §9). The Fetcher seam is +wired so enabling it later is purely additive; it raises until implemented. +""" + +from __future__ import annotations + +from tools.ingest.base import Candidate + +_DEFERRED = "deferred — see spec §6.4 (Musopen access may require an API key)" + + +class MusopenFetcher: + archive = "musopen" + + def __init__(self, client): + self.client = client + + def search(self, query: str, *, limit: int) -> list[Candidate]: + raise NotImplementedError(_DEFERRED) + + def resolve(self, identifier: str) -> Candidate: + raise NotImplementedError(_DEFERRED) diff --git a/tools/ingest/nasa.py b/tools/ingest/nasa.py new file mode 100644 index 0000000..a71023c --- /dev/null +++ b/tools/ingest/nasa.py @@ -0,0 +1,79 @@ +"""NASA fetcher — public-domain imagery/video. + +JSON API: https://images-api.nasa.gov/search (keyless). Each search item links +an asset-collection JSON (the item's `href`) listing the concrete file URLs; +the chosen media file is resolved from there. NASA media is public domain, but +some items embed third-party content — flagged in notes for review (§6.4). +""" + +from __future__ import annotations + +from tools.ingest.base import Candidate + +SEARCH = "https://images-api.nasa.gov/search" +_THIRD_PARTY_CAVEAT = "[NASA: may embed third-party content — verify before use]" +_VIDEO_EXTS = (".mp4", ".mov", ".m4v", ".webm") +_AUDIO_EXTS = (".mp3", ".m4a", ".wav", ".flac", ".ogg") + + +def _ext_from_url(url: str) -> str: + tail = url.rsplit("/", 1)[-1].split("?", 1)[0] + return tail.rsplit(".", 1)[-1].lower() if "." in tail else "" + + +def _pick_asset(assets, media_type: str) -> str: + if not isinstance(assets, list): + return "" + prefs = _AUDIO_EXTS if media_type == "audio" else _VIDEO_EXTS + for ext in prefs: + for url in assets: + if isinstance(url, str) and url.lower().endswith(ext): + return url + for url in assets: + if isinstance(url, str) and not url.lower().endswith((".jpg", ".png", ".json")): + return url + return "" + + +class NasaFetcher: + archive = "nasa" + + def __init__(self, client): + self.client = client + + def _candidate(self, item) -> Candidate: + data0 = (item.get("data") or [{}])[0] + nasa_id = data0.get("nasa_id", "") + title = data0.get("title", "") + media_type = data0.get("media_type", "") + description = (data0.get("description", "") or "").strip() + description = f"{description} {_THIRD_PARTY_CAVEAT}".strip() + + href = item.get("href", "") + assets = self.client.get_json(href) if href else [] + media_url = _pick_asset(assets, media_type) + return Candidate( + source_archive=self.archive, + source_url=f"https://images.nasa.gov/details-{nasa_id}", + media_url=media_url, + title=title, + license="public_domain", + attribution="", + suggested_id=f"nasa-{nasa_id}", + media_ext=_ext_from_url(media_url), + description=description, + ) + + def search(self, query: str, *, limit: int) -> list[Candidate]: + url = f"{SEARCH}?q={query}&media_type=video" + data = self.client.get_json(url) + items = (data.get("collection", {}).get("items") or [])[:limit] + return [self._candidate(i) for i in items] + + def resolve(self, identifier: str) -> Candidate: + url = f"{SEARCH}?nasa_id={identifier}" + data = self.client.get_json(url) + items = data.get("collection", {}).get("items") or [] + if not items: + raise ValueError(f"nasa: no item for {identifier!r}") + return self._candidate(items[0])