diff --git a/docs/source/api.rst b/docs/source/api.rst index e7067feb5..3c11d036f 100644 --- a/docs/source/api.rst +++ b/docs/source/api.rst @@ -121,6 +121,7 @@ Built-in Implementations implementations.git.GitFileSystem implementations.github.GithubFileSystem implementations.http.HTTPFileSystem + implementations.ia.InternetArchiveFileSystem implementations.jupyter.JupyterFileSystem implementations.libarchive.LibArchiveFileSystem implementations.local.LocalFileSystem @@ -175,6 +176,9 @@ Built-in Implementations .. autoclass:: fsspec.implementations.http.HTTPFileSystem :members: __init__ +.. autoclass:: fsspec.implementations.ia.InternetArchiveFileSystem + :members: __init__ + .. autoclass:: fsspec.implementations.jupyter.JupyterFileSystem :members: __init__ diff --git a/docs/source/changelog.rst b/docs/source/changelog.rst index 3f78f3659..2e0da5fb7 100644 --- a/docs/source/changelog.rst +++ b/docs/source/changelog.rst @@ -6,6 +6,9 @@ Dev Enhancements +- ``ia://`` filesystem for files in Internet Archive items: an + ``HTTPFileSystem`` reading from ``archive.org/download`` with credentials from + ``ia configure``'s ``ia.ini`` - Allow instance-local memory stores with ``global_store=False`` while preserving normal instance caching and the default shared store (#1904) diff --git a/fsspec/implementations/ia.py b/fsspec/implementations/ia.py new file mode 100644 index 000000000..d07c1613f --- /dev/null +++ b/fsspec/implementations/ia.py @@ -0,0 +1,222 @@ +"""Files in Internet Archive items, ``ia:///``.""" + +import configparser +import os +from dataclasses import dataclass + +import aiohttp +from aiohttp import hdrs + +from ..utils import stringify_path +from .http import HTTPFileSystem + +DOWNLOAD_URL = "https://archive.org/download/" +AUTH_DOMAIN = ".archive.org" +# The environment variable names the ``internetarchive`` package uses. +ENV_ACCESS_KEY = "IA_ACCESS_KEY_ID" +ENV_SECRET_KEY = "IA_SECRET_ACCESS_KEY" +ENV_CONFIG_FILE = "IA_CONFIG_FILE" + + +def ia_config_path(): + """The ``ia.ini`` written by ``ia configure``, or None. + + Searched the way the ``internetarchive`` package searches: ``$IA_CONFIG_FILE``, + ``$XDG_CONFIG_HOME/internetarchive/ia.ini`` (``XDG_CONFIG_HOME`` defaulting to + ``~/.config``), ``~/.config/ia.ini``, ``~/.ia``. The first existing file wins. + """ + home = os.path.expanduser("~") + xdg = os.environ.get("XDG_CONFIG_HOME") or os.path.join(home, ".config") + candidates = [ + os.environ.get(ENV_CONFIG_FILE), + os.path.join(xdg, "internetarchive", "ia.ini"), + os.path.join(home, ".config", "ia.ini"), + os.path.join(home, ".ia"), + ] + return next((p for p in candidates if p and os.path.isfile(p)), None) + + +@dataclass(frozen=True) +class IACredentials: + """The IA-S3 key pair that logs a request in at archive.org; None means anonymous.""" + + access_key: "str | None" = None + secret_key: "str | None" = None + config_file: "str | None" = None # where they were read from + + @property + def anonymous(self): + return not self.access_key + + @property + def authorization(self): + """The ``Authorization`` header value, or None when anonymous.""" + if self.anonymous: + return None + return f"LOW {self.access_key}:{self.secret_key}" + + +def load_ia_credentials(config_file=None): + """Credentials from ``ia.ini`` and the environment; anonymous when there are none. + + ``[s3] access/secret`` are read from ``config_file`` (default: + :func:`ia_config_path`); the ``[cookies]`` section ``ia configure`` also writes is + ignored, as the keys log a request in on their own. ``IA_ACCESS_KEY_ID`` / + ``IA_SECRET_ACCESS_KEY`` override the file's keys and must be set together. A + missing file or section is not an error: public items need nothing. + """ + path = config_file or ia_config_path() + access_key = secret_key = None + if path and os.path.isfile(path): + # RawConfigParser: the file's cookie values contain '%' (URL-encoded + # emails), which the default interpolation would reject. + parser = configparser.RawConfigParser() + parser.read(path, encoding="utf-8") + access_key = parser.get("s3", "access", fallback="").strip() or None + secret_key = parser.get("s3", "secret", fallback="").strip() or None + + env_access = os.environ.get(ENV_ACCESS_KEY) + env_secret = os.environ.get(ENV_SECRET_KEY) + if bool(env_access) != bool(env_secret): + raise ValueError(f"{ENV_ACCESS_KEY} and {ENV_SECRET_KEY} must be set together") + if env_access: + access_key, secret_key = env_access, env_secret + if not (access_key and secret_key): + access_key = secret_key = None + found = path if path and os.path.isfile(path) else None + return IACredentials(access_key, secret_key, found) + + +def in_domain(host, domain): + """Whether ``host`` is ``domain`` (leading dot optional) or a subdomain of it.""" + if not host: + return False + domain = domain.lower().lstrip(".") + host = host.lower().rstrip(".") + return host == domain or host.endswith("." + domain) + + +def authorized_request_class(authorization, domain, base=aiohttp.ClientRequest): + """A ``ClientRequest`` that sends ``authorization`` to every host in ``domain``. + + aiohttp drops ``Authorization`` when a redirect changes origin, and every + archive.org download is such a redirect (to a data node). It builds a new + request from the session's ``request_class`` for each hop, so setting the header + here puts it back on the data-node hop; testing the host keeps it from leaving + ``domain`` should a data node redirect elsewhere. + """ + + class AuthorizedRequest(base): + def update_headers(self, headers): + super().update_headers(headers) + if in_domain(self.url.host, domain): + self.headers[hdrs.AUTHORIZATION] = authorization + + return AuthorizedRequest + + +class InternetArchiveFileSystem(HTTPFileSystem): + """Files in Internet Archive items, addressed as ``ia:///``. + + Every path is read from ``https://archive.org/download//``, + which redirects to a data node that honours HTTP Range requests, so this is + ``HTTPFileSystem`` with a path mapping and archive.org credentials. (IA's + S3-like API at ``s3.us.archive.org`` is not used: it ignores Range headers and + redirects GET to plain-http data nodes, so block reads would fetch whole files.) + + Public items need no credentials. Restricted items need the account's IA-S3 + keys, as written by ``ia configure`` from the ``internetarchive`` package to + ``ia.ini``; that file is found the way the package finds it (``$IA_CONFIG_FILE``, + ``$XDG_CONFIG_HOME/internetarchive/ia.ini``, ``~/.config/ia.ini``, ``~/.ia``). + Explicit arguments win over the file. + + Parameters + ---------- + access_key, secret_key: str, optional + IA S3 keys, sent as ``Authorization: LOW :`` to every + archive.org host, including the data node the download URL redirects to + (aiohttp drops the header on a cross-origin redirect; the session's request + class puts it back, and only for ``.archive.org`` hosts). Both or neither; + ``IA_ACCESS_KEY_ID`` / ``IA_SECRET_ACCESS_KEY`` in the environment override + ``ia.ini``. Pass empty strings to stay anonymous despite an ``ia.ini``. + config_file: str, optional + Path to an ``ia.ini`` to read credentials from instead of the default lookup. + kwargs: + Passed to ``HTTPFileSystem``. + """ + + protocol = "ia" + download_url = DOWNLOAD_URL + auth_domain = AUTH_DOMAIN + + def __init__(self, access_key=None, secret_key=None, config_file=None, **kwargs): + if access_key is None and secret_key is None: + found = load_ia_credentials(config_file) + access_key, secret_key = found.access_key, found.secret_key + elif bool(access_key) != bool(secret_key): + raise ValueError("access_key and secret_key must be given together") + credentials = IACredentials(access_key or None, secret_key or None) + self.access_key = credentials.access_key + self.authorization = credentials.authorization + if self.authorization: + client_kwargs = dict(kwargs.pop("client_kwargs", None) or {}) + client_kwargs["request_class"] = authorized_request_class( + self.authorization, + self.auth_domain, + client_kwargs.get("request_class", aiohttp.ClientRequest), + ) + kwargs["client_kwargs"] = client_kwargs + super().__init__(**kwargs) + + @property + def fsid(self): + return "ia" + + @classmethod + def _strip_protocol(cls, path): + """``ia://item/file`` (or bare ``item/file``) becomes the download URL. + + A URL passes through unchanged, so the mapping is idempotent. + """ + if isinstance(path, list): + return [cls._strip_protocol(p) for p in path] + path = stringify_path(path) + if path.startswith("ia://"): + path = path[5:] + elif "://" in path: + return path + return cls.download_url + path.lstrip("/") + + def unstrip_protocol(self, name): + if name.startswith(self.download_url): + return "ia://" + name[len(self.download_url) :] + if name.startswith("ia://"): + return name + return "ia://" + name.lstrip("/") + + # archive.org answers a restricted item with 403 (401 for a bad LOW key). + # HTTPFileSystem reports a failed HEAD+GET as FileNotFoundError and any other + # status through raise_for_status(); both become PermissionError so callers can + # tell "no such file" from "log in". The two methods callers reach with an + # ia:// path directly (open() strips before _open) map it here too. + async def _info(self, url, **kwargs): + url = self._strip_protocol(url) + try: + return await super()._info(url, **kwargs) + except FileNotFoundError as exc: + cause = exc.__cause__ + if isinstance(cause, aiohttp.ClientResponseError) and cause.status in ( + 401, + 403, + ): + raise PermissionError(url) from cause + raise + + async def _cat_file(self, url, start=None, end=None, **kwargs): + url = self._strip_protocol(url) + return await super()._cat_file(url, start=start, end=end, **kwargs) + + def _raise_not_found_for_status(self, response, url): + if response.status in (401, 403): + raise PermissionError(url) + super()._raise_not_found_for_status(response, url) diff --git a/fsspec/implementations/tests/test_ia.py b/fsspec/implementations/tests/test_ia.py new file mode 100644 index 000000000..6d6c2837c --- /dev/null +++ b/fsspec/implementations/tests/test_ia.py @@ -0,0 +1,304 @@ +"""ia:///: the Internet Archive filesystem. + +Offline: the URI mapping, credential loading from ``ia.ini`` and the environment, reads +through the local HTTP test server, and the ``Authorization`` header's scoping, including a +stand-in archive.org whose download URL redirects to a data node on another origin. One +opt-in network test (``IA_NETWORK_TESTS=1``; not ``FSSPEC_IA_*``, which fsspec.config would +turn into a constructor argument) reads the first bytes of a public item. +""" + +import os +import threading +from http.server import BaseHTTPRequestHandler, HTTPServer + +import pytest + +import fsspec +from fsspec.implementations.ia import ( + InternetArchiveFileSystem, + ia_config_path, + in_domain, + load_ia_credentials, +) +from fsspec.tests.conftest import data, server # noqa: F401 + +pytest.importorskip("aiohttp") + +# What ``ia configure`` writes: the [cookies] section is ignored (and must not trip the +# parser with its '%'). +INI = """[s3] +access = AKIA-TEST +secret = s3cr3t +[cookies] +logged-in-user = someone%40example.org; expires=Sat, 28-Aug-2027 19:39:52 GMT; path=/; domain=.archive.org +logged-in-sig = 1756000000-abcdef0123456789; expires=Sat, 28-Aug-2027 19:39:52 GMT; path=/; domain=.archive.org +[general] +screenname = Some One +""" + +PUBLIC_ITEM = ( + "ia://EOT24PRE-20240926175758-crawl808/EOT24PRE-20240926175758-00032.warc.gz" +) + + +@pytest.fixture(autouse=True) +def isolated_credentials(tmp_path, monkeypatch): + """Never read the developer's real ia.ini; never reuse a cached instance, which + would carry the previous test's credentials.""" + # os.path.expanduser("~") reads HOME on POSIX and USERPROFILE on Windows, which + # never consults HOME; both have to move for the lookup to leave the real one alone. + monkeypatch.setenv("HOME", str(tmp_path / "home")) + monkeypatch.setenv("USERPROFILE", str(tmp_path / "home")) + for name in ( + "IA_CONFIG_FILE", + "XDG_CONFIG_HOME", + "IA_ACCESS_KEY_ID", + "IA_SECRET_ACCESS_KEY", + ): + monkeypatch.delenv(name, raising=False) + InternetArchiveFileSystem.clear_instance_cache() + yield + InternetArchiveFileSystem.clear_instance_cache() + + +@pytest.fixture +def ini(tmp_path): + path = tmp_path / "ia.ini" + path.write_text(INI, encoding="utf-8") + return str(path) + + +class _CrossOriginHandler(BaseHTTPRequestHandler): + """``/download/`` redirects to ``/items/`` on the data node, a second + server on another port and so another origin, which is where aiohttp drops the + ``Authorization`` header. The item route serves ``files`` by Range and, when + ``required`` is set, only to a request carrying that ``Authorization`` value.""" + + files = {} + required = None + hits = [] + data_node = None # "http://host:port", set by the fixture + + def log_message(self, *args): + pass + + def do_GET(self): + self.hits.append((self.path, dict(self.headers))) + if self.path.startswith("/download/"): + self.send_response(302) + self.send_header("Location", f"{self.data_node}/items/{self.path[10:]}") + self.send_header("Content-Length", "0") + self.end_headers() + return + body = self.files.get(self.path) + if body is None: + self.send_error(404) + return + if self.required and self.headers.get("Authorization") != self.required: + self.send_error(403) + return + status, start, end = 200, 0, len(body) - 1 + if "Range" in self.headers: + first, _, last = self.headers["Range"][len("bytes=") :].partition("-") + start, end = int(first), (min(int(last), end) if last else end) + status = 206 + chunk = body[start : end + 1] + self.send_response(status) + self.send_header("Content-Length", str(len(chunk))) + self.send_header("Accept-Ranges", "bytes") + if status == 206: + self.send_header("Content-Range", f"bytes {start}-{end}/{len(body)}") + self.end_headers() + if self.command == "GET": + self.wfile.write(chunk) + + do_HEAD = do_GET + + +@pytest.fixture +def ia_server(monkeypatch): + """A stand-in archive.org on ``localhost``: the download server redirects to a data + node on a second port. Both are in the (patched) auth domain ``localhost``; set + ``handler.data_node`` to a ``127.0.0.1`` URL to move the node out of it.""" + handler = type( + "Handler", (_CrossOriginHandler,), {"files": {}, "required": None, "hits": []} + ) + servers = [HTTPServer(("127.0.0.1", 0), handler) for _ in range(2)] + for httpd in servers: + threading.Thread(target=httpd.serve_forever, daemon=True).start() + front, node = servers + handler.data_node = f"http://localhost:{node.server_port}" + handler.node_port = node.server_port + monkeypatch.setattr( + InternetArchiveFileSystem, + "download_url", + f"http://localhost:{front.server_port}/download/", + ) + monkeypatch.setattr(InternetArchiveFileSystem, "auth_domain", "localhost") + try: + yield handler + finally: + for httpd in servers: + httpd.shutdown() + httpd.server_close() + + +def test_uri_maps_to_download_url_and_back(): + url = "https://archive.org/download/some-item/some-file.warc.gz" + assert ( + InternetArchiveFileSystem._strip_protocol("ia://some-item/some-file.warc.gz") + == url + ) + assert ( + InternetArchiveFileSystem._strip_protocol("some-item/some-file.warc.gz") == url + ) + assert InternetArchiveFileSystem._strip_protocol(url) == url # idempotent + fs = InternetArchiveFileSystem() + assert fs.unstrip_protocol(url) == "ia://some-item/some-file.warc.gz" + assert fs.unstrip_protocol("some-item/x") == "ia://some-item/x" + + +def test_registered(): + fs, path = fsspec.core.url_to_fs("ia://some-item/some-file") + assert isinstance(fs, InternetArchiveFileSystem) + assert path == "https://archive.org/download/some-item/some-file" + assert fs.fsid == "ia" + + +def test_no_config_means_anonymous(): + assert ia_config_path() is None + credentials = load_ia_credentials() + assert credentials.anonymous and credentials.config_file is None + assert credentials.authorization is None + fs = InternetArchiveFileSystem() + assert fs.access_key is None and fs.authorization is None + assert "request_class" not in fs.client_kwargs + + +def test_ini_is_parsed(ini, monkeypatch): + monkeypatch.setenv("IA_CONFIG_FILE", ini) + assert ia_config_path() == ini + credentials = load_ia_credentials() + assert (credentials.access_key, credentials.secret_key) == ("AKIA-TEST", "s3cr3t") + assert credentials.config_file == ini + assert not hasattr(credentials, "cookies") + fs = InternetArchiveFileSystem() + assert fs.authorization == "LOW AKIA-TEST:s3cr3t" + assert "Authorization" not in fs.kwargs.get("headers", {}) # scoped, not global + + +def test_ini_lookup_order(tmp_path, monkeypatch): + home = tmp_path / "home" + xdg_default = home / ".config" / "internetarchive" / "ia.ini" + dot_ia = home / ".ia" + for path in (xdg_default, dot_ia): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(INI) + assert ia_config_path() == str(xdg_default) + xdg_default.unlink() + assert ia_config_path() == str(dot_ia) + xdg = tmp_path / "xdg" / "internetarchive" + xdg.mkdir(parents=True) + (xdg / "ia.ini").write_text(INI) + monkeypatch.setenv("XDG_CONFIG_HOME", str(tmp_path / "xdg")) + assert ia_config_path() == str(xdg / "ia.ini") + monkeypatch.setenv("IA_CONFIG_FILE", str(dot_ia)) + assert ia_config_path() == str(dot_ia) + + +def test_environment_keys_override_the_file_and_come_in_pairs(ini, monkeypatch): + monkeypatch.setenv("IA_ACCESS_KEY_ID", "ENV-KEY") + monkeypatch.setenv("IA_SECRET_ACCESS_KEY", "ENV-SECRET") + credentials = load_ia_credentials(ini) + assert (credentials.access_key, credentials.secret_key) == ("ENV-KEY", "ENV-SECRET") + monkeypatch.delenv("IA_SECRET_ACCESS_KEY") + with pytest.raises(ValueError, match="must be set together"): + load_ia_credentials(ini) + + +def test_explicit_arguments_win_over_the_file(ini, monkeypatch): + monkeypatch.setenv("IA_CONFIG_FILE", ini) + fs = InternetArchiveFileSystem(access_key="explicit", secret_key="s") + assert fs.authorization == "LOW explicit:s" + anonymous = InternetArchiveFileSystem(access_key="", secret_key="") + assert anonymous.authorization is None + with pytest.raises(ValueError, match="must be given together"): + InternetArchiveFileSystem(access_key="only-one") + + +def test_authorization_is_scoped_to_archive_org(): + """The header goes to archive.org and its data nodes, and nowhere else.""" + for host in ( + "archive.org", + "dn721904.ca.archive.org", + "ARCHIVE.ORG", + "s3.us.archive.org.", + ): + assert in_domain(host, ".archive.org") + for host in ("archive.example.com", "notarchive.org", "example.org", "", None): + assert not in_domain(host, ".archive.org") + + +def test_reads_go_through_the_download_url(server, monkeypatch): + monkeypatch.setattr(InternetArchiveFileSystem, "download_url", server.address + "/") + fs = InternetArchiveFileSystem(headers={"give_length": "true", "use_206": "true"}) + assert fs.cat("ia://index/realfile") == data + with fs.open("ia://index/realfile", "rb") as f: + assert f.read() == data + assert fs.cat_file("ia://index/realfile", start=1, end=10) == data[1:10] + assert fs.info("ia://index/realfile")["size"] == len(data) + + +def test_refusal_is_a_permission_error(server, monkeypatch): + monkeypatch.setattr(InternetArchiveFileSystem, "download_url", server.address + "/") + fs = InternetArchiveFileSystem() + with pytest.raises(PermissionError): + fs.cat("ia://unauthorized") + with pytest.raises(PermissionError): + fs.open("ia://unauthorized", "rb") + with pytest.raises(FileNotFoundError): + fs.cat("ia://index/missing") + + +def test_authorization_survives_the_cross_origin_redirect(ia_server): + """The keys must reach the data node, which sits behind a cross-origin redirect where + aiohttp has already dropped the Authorization header; the request class re-adds it.""" + ia_server.files["/items/restricted/file"] = data + ia_server.required = "LOW k:s" + + anonymous = InternetArchiveFileSystem() + with pytest.raises(PermissionError): + anonymous.cat("ia://restricted/file") + with pytest.raises(PermissionError): + anonymous.open("ia://restricted/file", "rb") + + fs = InternetArchiveFileSystem(access_key="k", secret_key="s") + ia_server.hits.clear() + assert fs.cat("ia://restricted/file") == data + assert fs.cat_file("ia://restricted/file", start=2, end=5) == data[2:5] + with fs.open("ia://restricted/file", "rb", block_size=4) as f: + assert f.read() == data + hops = {p.split("/")[1] for p, _ in ia_server.hits} + assert hops == {"download", "items"} + assert all(h.get("Authorization") == "LOW k:s" for _, h in ia_server.hits) + assert not any("Cookie" in h for _, h in ia_server.hits) + + +def test_authorization_does_not_leave_the_domain(ia_server): + """A redirect to a host outside ``auth_domain`` gets no Authorization header.""" + ia_server.files["/items/public/file"] = data + ia_server.data_node = f"http://127.0.0.1:{ia_server.node_port}" # not `localhost` + + fs = InternetArchiveFileSystem(access_key="k", secret_key="s") + assert fs.cat("ia://public/file") == data + first_hop = next(h for p, h in ia_server.hits if p.startswith("/download/")) + assert first_hop.get("Authorization") == "LOW k:s" + data_hop = next(h for p, h in ia_server.hits if p.startswith("/items/")) + assert "Authorization" not in data_hop + + +@pytest.mark.skipif(not os.environ.get("IA_NETWORK_TESTS"), reason="needs archive.org") +def test_public_item_over_the_network(): + fs = InternetArchiveFileSystem(access_key="", secret_key="") + assert fs.cat_file(PUBLIC_ITEM, start=0, end=2) == b"\x1f\x8b" # gzip magic + assert fs.size(PUBLIC_ITEM) > 1_000_000_000 diff --git a/fsspec/registry.py b/fsspec/registry.py index 305d1e890..53f21d0ab 100644 --- a/fsspec/registry.py +++ b/fsspec/registry.py @@ -161,6 +161,10 @@ def register_implementation(name, cls, clobber=False, errtxt=None): "class": "fsspec.implementations.http.HTTPFileSystem", "err": 'HTTPFileSystem requires "requests" and "aiohttp" to be installed', }, + "ia": { + "class": "fsspec.implementations.ia.InternetArchiveFileSystem", + "err": 'InternetArchiveFileSystem requires "requests" and "aiohttp" to be installed', + }, "jlab": { "class": "fsspec.implementations.jupyter.JupyterFileSystem", "err": "Jupyter FS requires requests to be installed",