From b006d0541c3cbe7ef0070daf7da6eb9d39e07ded Mon Sep 17 00:00:00 2001 From: Cody Baker Date: Mon, 28 Sep 2026 14:43:14 +0000 Subject: [PATCH 01/12] Add content_fingerprint hook to memoize_path Some resources can vouch for their content better than a stat() of a path ever could, e.g. objects carrying a content digest such as a git-annex key, which need not be paths at all (and so are never cached today: realpath() raises and memoize_path falls through to a plain call). memoize_path() now takes an optional content_fingerprint callable. Called with the value of the first argument, it returns a picklable fingerprint of that value's content, or None. A value with a fingerprint is cached under ("content", fingerprint) plus the cache's tokens instead of under its path and stat(); the value itself stays excluded from joblib's key, so it need not be picklable. The "modified just now" window does not apply, since the content cannot change without its fingerprint changing. Values without a fingerprint are handled exactly as before. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Qn5WBSiQgoZoL4fytF6nEr --- README.rst | 17 +++++++ src/fscacher/cache.py | 57 +++++++++++++++++---- src/fscacher/tests/test_cache.py | 86 ++++++++++++++++++++++++++++++++ 3 files changed, 150 insertions(+), 10 deletions(-) diff --git a/README.rst b/README.rst index 405cadd..cbb3d7f 100644 --- a/README.rst +++ b/README.rst @@ -48,6 +48,23 @@ of ``path``, the cache is ignored. be a sequence of names of arguments of the decorated function that will be ignored for caching purposes. +``memoize_path()`` also optionally takes a ``content_fingerprint`` callable, +for resources that can vouch for their content better than a ``stat()`` can +(e.g., objects carrying a content digest, which need not be paths at all). It +is called with the value of the first argument and returns either a +fingerprint of that value's content (any picklable object, such as a digest +string) or ``None``. A value with a fingerprint is cached under it (plus the +cache's tokens) instead of under its path, so any two values with equal +fingerprints share cached results; values without one are handled as usual: + +.. code:: python + + @cache.memoize_path( + content_fingerprint=lambda x: x.digest if isinstance(x, Blob) else None + ) + def foo(path_or_blob, ...): + ... + Caches are stored on-disk and thus persist between Python runs. To clear a given ``PersistentCache`` and erase its data store, call the ``clear()`` method. diff --git a/src/fscacher/cache.py b/src/fscacher/cache.py index b0a1758..02a5a01 100644 --- a/src/fscacher/cache.py +++ b/src/fscacher/cache.py @@ -84,9 +84,32 @@ def memoize(self, f=None, *, exclude_kwargs=None): return f return self._memory.cache(f, ignore=exclude_kwargs) - def memoize_path(self, f=None, *, exclude_kwargs=None): + def memoize_path(self, f=None, *, exclude_kwargs=None, content_fingerprint=None): + """ + Memoize a function whose first argument is a path, keyed on a + fingerprint of the file or directory at that path + + Parameters + ---------- + exclude_kwargs: list of str, optional + Names of arguments of the decorated function to ignore for caching + purposes + content_fingerprint: callable, optional + Called with the value of the first argument; it may return a + fingerprint of the *content* that value refers to (e.g., a content + digest), or `None` if it has none. A value with a fingerprint is + cached under it (and the cache's tokens) instead of under a path and + its ``stat()``, so it need not be a path at all: two values with equal + fingerprints share their cached results. The fingerprint must be + picklable, and equal fingerprints must imply identical content. + Values without one are handled as without ``content_fingerprint``. + """ if f is None: - return partial(self.memoize_path, exclude_kwargs=exclude_kwargs) + return partial( + self.memoize_path, + exclude_kwargs=exclude_kwargs, + content_fingerprint=content_fingerprint, + ) if self._ignore_cache: return f @@ -121,6 +144,14 @@ def fingerprinted(path, *args, **kwargs): + (list(exclude_kwargs) if exclude_kwargs is not None else []), ) + def call_fingerprinted(key, args, kwargs): + # inject the fingerprint (and tokens) into the signature + kwargs_ = kwargs.copy() + kwargs_[fingerprint_kwarg] = key + ( + tuple(self._tokens) if self._tokens else () + ) + return fingerprinted(*args, **kwargs_) + @wraps(f) def fingerprinter(*args, **kwargs): # we need to dereference symlinks and use that path in the function @@ -128,6 +159,19 @@ def fingerprinter(*args, **kwargs): bound = sig.bind(*args, **kwargs) bound.apply_defaults() path_orig = bound.arguments[path_arg] + if content_fingerprint is not None: + cfprint = content_fingerprint(path_orig) + if cfprint is not None: + lgr.debug( + "Calling memoized version of %s for content fingerprint %r", + f, + cfprint, + ) + # No modified_in_window() check: content cannot change + # without its fingerprint changing + ret = call_fingerprinted(("content", cfprint), args, kwargs) + lgr.log(1, "Returning value %r", ret) + return ret try: path = op.realpath(path_orig) except TypeError: @@ -154,14 +198,7 @@ def fingerprinter(*args, **kwargs): ret = f(*args, **kwargs) else: lgr.debug("Calling memoized version of %s for %s", f, path) - # If there is a fingerprint -- inject it into the signature - kwargs_ = kwargs.copy() - kwargs_[fingerprint_kwarg] = ( - (path,) - + fprint.to_tuple() - + (tuple(self._tokens) if self._tokens else ()) - ) - ret = fingerprinted(*args, **kwargs_) + ret = call_fingerprinted((path,) + fprint.to_tuple(), args, kwargs) lgr.log(1, "Returning value %r", ret) return ret diff --git a/src/fscacher/tests/test_cache.py b/src/fscacher/tests/test_cache.py index 13e585e..5a3fedb 100644 --- a/src/fscacher/tests/test_cache.py +++ b/src/fscacher/tests/test_cache.py @@ -8,6 +8,7 @@ import subprocess import sys import time +from typing import Optional import pytest from .. import PersistentCache from ..cache import DirFingerprint, FileFingerprint @@ -558,3 +559,88 @@ def memoread_extra(path, arg, kwarg=None, extra=None): (path, 1, "quux", "bar"), (path, 1, None, "foo"), ] + + +@dataclass +class Blob: + """A non-path resource that may know a fingerprint of its content""" + + content: str + fingerprint: Optional[str] = None + + +def blob_fingerprint(x): + return x.fingerprint if isinstance(x, Blob) else None + + +def test_memoize_path_content_fingerprint(cache, tmp_path): + calls = [] + + @cache.memoize_path(content_fingerprint=blob_fingerprint) + def memoread(src, arg=0, kwarg=None): + calls.append((src, arg, kwarg)) + if isinstance(src, Blob): + return f"{src.content}:{arg}:{kwarg}" + with open(src) as f: + return f"{f.read()}:{arg}:{kwarg}" + + # A value without a fingerprint is not cached + assert memoread(Blob("content")) == "content:0:None" + assert memoread(Blob("content")) == "content:0:None" + assert len(calls) == 2 + + # A value with one is cached by it: a twin is served from the cache, + # however its other arguments are passed + assert memoread(Blob("content", "A"), 1) == "content:1:None" + assert len(calls) == 3 + assert memoread(Blob("never read", "A"), 1) == "content:1:None" + assert memoread(Blob("never read", "A"), arg=1) == "content:1:None" + assert memoread(src=Blob("never read", "A"), arg=1) == "content:1:None" + assert len(calls) == 3 + + # The fingerprint and the other arguments are part of the key + assert memoread(Blob("other", "B"), 1) == "other:1:None" + assert memoread(Blob("content", "A"), 1, kwarg="q") == "content:1:q" + assert len(calls) == 5 + + # Paths are still fingerprinted by stat(), whatever the content + path = tmp_path / "file.dat" + path.write_text("content") + time.sleep(cache._min_dtime * 1.1) + assert memoread(path, 1) == "content:1:None" + assert memoread(path, 1) == "content:1:None" + assert len(calls) == 6 + + +def test_memoize_path_content_fingerprint_tokens(tmp_path_factory): + calls = [] + + def memoread(src): + calls.append(src) + return src.content + + path = tmp_path_factory.mktemp("cache") + c1 = PersistentCache(path=path, tokens=["1"]) + c2 = PersistentCache(path=path, tokens=["2"]) + m1 = c1.memoize_path(memoread, content_fingerprint=blob_fingerprint) + m2 = c2.memoize_path(memoread, content_fingerprint=blob_fingerprint) + assert m1(Blob("content", "A")) == "content" + assert m1(Blob("never read", "A")) == "content" + assert len(calls) == 1 + assert m2(Blob("content", "A")) == "content" + assert len(calls) == 2 + + +def test_memoize_path_content_fingerprint_ignored(monkeypatch, tmp_path): + monkeypatch.setenv("FSCACHER_CACHE", "ignore") + cache = PersistentCache(path=tmp_path) + calls = [] + + @cache.memoize_path(content_fingerprint=blob_fingerprint) + def memoread(src): + calls.append(src) + return src.content + + assert memoread(Blob("content", "A")) == "content" + assert memoread(Blob("content", "A")) == "content" + assert len(calls) == 2 From c43232e14636f73c026c7284eb1c4def618d8d8a Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 22:41:08 +0000 Subject: [PATCH 02/12] Rework into custom_fingerprint, an alternative to stat(), plus a git-annex key one Per review: - Rename `content_fingerprint` to `custom_fingerprint`: what a fingerprint vouches for is up to the callable. - Make it a drop-in alternative to the stat()-based fingerprint rather than a separate code path: `_get_fingerprint()` returns either a `CustomFingerprint` or a `PathFingerprint` (the stat()-based one with its path, keyed exactly as before), and the rest of `memoize_path` (no fingerprint -> direct call, modified-recently window, key injection) is shared. A custom fingerprint is never "modified recently": it is up to the callable to change it when the result may change. - Also consult the callable for each entry met while fingerprinting a directory, so that e.g. a tree with some annexed files dropped can still be fingerprinted (such entries do not feed the modified-recently window). - Document the callable's contract: it runs on every call, so it must be cheap and return None (fall back to stat()) rather than raise for what it does not recognize; results are shared by equal fingerprints, so include the path unless the result does not depend on it. - Add `fscacher.annex_key_fingerprint`, a real such callable: it reads the link of a *locked* annexed file (no git-annex call) and returns its key if the backend hashes the content (SHA*, SHA3_*, SKEIN*, BLAKE2*, MD5, with or without E); unlocked files (whose key goes stale when edited, as on adjusted branches/Windows), WORM, URL/VURL and external keys fall back to stat(). It pairs the key with the path by default (`pair_with_path`), as results may depend on the path or extension. Results for a locked file survive its content being dropped, which is documented and tested. - Tests: the helper on a fake annex layout (accepted and rejected backends, pairing, non-annexed and non-path values, dropped content), memoize_path end to end on it (twins across paths and extensions with and without pairing, dropped content, stat() fallbacks, a directory with dropped files), and against real git-annex when available (locked twins, unlocked and edited, dropped, WORM). Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Qn5WBSiQgoZoL4fytF6nEr --- README.rst | 34 ++-- src/fscacher/__init__.py | 3 +- src/fscacher/annex.py | 49 ++++++ src/fscacher/cache.py | 168 ++++++++++++------- src/fscacher/tests/test_annex.py | 277 +++++++++++++++++++++++++++++++ src/fscacher/tests/test_cache.py | 23 ++- 6 files changed, 473 insertions(+), 81 deletions(-) create mode 100644 src/fscacher/annex.py create mode 100644 src/fscacher/tests/test_annex.py diff --git a/README.rst b/README.rst index cbb3d7f..ee342fa 100644 --- a/README.rst +++ b/README.rst @@ -48,21 +48,35 @@ of ``path``, the cache is ignored. be a sequence of names of arguments of the decorated function that will be ignored for caching purposes. -``memoize_path()`` also optionally takes a ``content_fingerprint`` callable, -for resources that can vouch for their content better than a ``stat()`` can -(e.g., objects carrying a content digest, which need not be paths at all). It -is called with the value of the first argument and returns either a -fingerprint of that value's content (any picklable object, such as a digest -string) or ``None``. A value with a fingerprint is cached under it (plus the -cache's tokens) instead of under its path, so any two values with equal -fingerprints share cached results; values without one are handled as usual: +``memoize_path()`` also optionally takes a ``custom_fingerprint`` callable, an +alternative to the built-in ``stat()``-based fingerprint. It is called with the +value of the first argument (and with the path of each entry met while +fingerprinting a directory) and returns either a fingerprint (a picklable value +with a stable ``repr()``, such as a string or a tuple of strings) or ``None`` to +fall back to ``stat()``. It runs on every call, so it must be cheap and return +``None`` rather than raise for anything it does not recognize. Results are +shared between all values with equal fingerprints, and a custom fingerprint is +trusted to change whenever the result may change (there is no "modified just +now" window for it); the value itself need not even be a path. + +``fscacher.annex_key_fingerprint`` is such a callable for git-annex'ed files: it +fingerprints a *locked* file by its key if the key's backend hashes the content +(``SHA*``, ``SHA3_*``, ``SKEIN*``, ``BLAKE2*``, ``MD5``, with or without the +``E`` suffix), and leaves anything else (unlocked files, ``WORM`` or ``URL`` +keys, ...) to ``stat()``. Results then survive the file being moved, or its +content dropped. By default, the key is paired with the file's path; pass +``pair_with_path=False`` to share results between all files with the same +content, e.g. across clones, if the result does not depend on the path: .. code:: python + from functools import partial + from fscacher import PersistentCache, annex_key_fingerprint + @cache.memoize_path( - content_fingerprint=lambda x: x.digest if isinstance(x, Blob) else None + custom_fingerprint=partial(annex_key_fingerprint, pair_with_path=False) ) - def foo(path_or_blob, ...): + def foo(path, ...): ... Caches are stored on-disk and thus persist between Python runs. To clear a diff --git a/src/fscacher/__init__.py b/src/fscacher/__init__.py index 340e8cd..f024125 100644 --- a/src/fscacher/__init__.py +++ b/src/fscacher/__init__.py @@ -5,6 +5,7 @@ """ from ._version import get_versions +from .annex import annex_key_fingerprint from .cache import PersistentCache __version__ = get_versions()["version"] @@ -13,4 +14,4 @@ __license__ = "MIT" __url__ = "https://github.com/con/fscacher" -__all__ = ["PersistentCache"] +__all__ = ["PersistentCache", "annex_key_fingerprint"] diff --git a/src/fscacher/annex.py b/src/fscacher/annex.py new file mode 100644 index 0000000..55d5661 --- /dev/null +++ b/src/fscacher/annex.py @@ -0,0 +1,49 @@ +"""Fingerprinting of git-annex'ed files by their keys""" + +import os +import os.path as op +import re + +#: Backends whose keys are hashes of the content, and thus pin it (with or +#: without the ``E`` suffix, which adds the file extension to the key). Others +#: do not: ``WORM`` keys are made of size, mtime and file name, ``URL`` and +#: ``VURL`` keys of a URL whose content may change, and external ``X*`` +#: backends are unknown. +CONTENT_HASH_BACKEND_RE = re.compile( + r"(SHA(1|224|256|384|512)|SHA3_(224|256|384|512)|SKEIN(256|512)" + r"|BLAKE2(B|BP|S|SP)\d+|MD5)E?" +) + + +def annex_key_fingerprint(path, *, pair_with_path=True): + """ + Fingerprint a locked git-annex'ed file by its key, for use as the + ``custom_fingerprint`` of `PersistentCache.memoize_path` + + Returns the key if ``path`` is a symbolic link into a git-annex object + store (``.../annex/objects/.../KEY/KEY``) and the key's backend hashes the + content (see `CONTENT_HASH_BACKEND_RE`); otherwise returns `None`, so that + the file is fingerprinted by ``stat()`` as usual. That is the case of an + unlocked file (including files on an adjusted branch, as on Windows), + whose key is not updated when the file is modified. Only the link is read, + so this is cheap, and a file whose content is not present (e.g., dropped) + is still fingerprinted: cached results are then returned for it. + + With ``pair_with_path`` (the default), the fingerprint is the pair of the + absolute path (as given, not dereferenced) and the key, so results are only + shared by files at the same path, as needed when they depend on the path + (e.g., on the extension or on neighboring files). Without it, results are + shared by all files with the same key, e.g., across clones of a dataset. + """ + try: + path = os.fsdecode(os.fspath(path)) + target = os.fsdecode(os.readlink(path)) + except (TypeError, ValueError, OSError): + return None + parts = target.replace("\\", "/").split("/") + if len(parts) < 6 or parts[-6:-4] != ["annex", "objects"] or parts[-1] != parts[-2]: + return None + key = parts[-1] + if not CONTENT_HASH_BACKEND_RE.fullmatch(key.split("-", 1)[0]): + return None + return (op.abspath(path), key) if pair_with_path else key diff --git a/src/fscacher/cache.py b/src/fscacher/cache.py index 02a5a01..498a161 100644 --- a/src/fscacher/cache.py +++ b/src/fscacher/cache.py @@ -84,7 +84,7 @@ def memoize(self, f=None, *, exclude_kwargs=None): return f return self._memory.cache(f, ignore=exclude_kwargs) - def memoize_path(self, f=None, *, exclude_kwargs=None, content_fingerprint=None): + def memoize_path(self, f=None, *, exclude_kwargs=None, custom_fingerprint=None): """ Memoize a function whose first argument is a path, keyed on a fingerprint of the file or directory at that path @@ -94,21 +94,29 @@ def memoize_path(self, f=None, *, exclude_kwargs=None, content_fingerprint=None) exclude_kwargs: list of str, optional Names of arguments of the decorated function to ignore for caching purposes - content_fingerprint: callable, optional - Called with the value of the first argument; it may return a - fingerprint of the *content* that value refers to (e.g., a content - digest), or `None` if it has none. A value with a fingerprint is - cached under it (and the cache's tokens) instead of under a path and - its ``stat()``, so it need not be a path at all: two values with equal - fingerprints share their cached results. The fingerprint must be - picklable, and equal fingerprints must imply identical content. - Values without one are handled as without ``content_fingerprint``. + custom_fingerprint: callable, optional + An alternative to the built-in ``stat()``-based fingerprint, e.g. + `fscacher.annex.annex_key_fingerprint`. It is called with the value + of the first argument, and with the path of each entry met while + fingerprinting a directory, and returns either a fingerprint of it or + `None` to fall back to ``stat()``. It is called on every call of the + decorated function, so it must be cheap, and it must return `None` + rather than raise for anything it does not recognize (including + plain paths, unless it fingerprints them). + + A fingerprint must change whenever the result may change (there is + no "modified just now" window for it), and must be picklable and have + a stable ``repr()`` (e.g., a string or a tuple of strings). Results + are shared between all values with equal fingerprints, wherever they + are: include the path in the fingerprint unless the result does not + depend on it. As the value itself is not part of the cache key, it + need not be a path at all. """ if f is None: return partial( self.memoize_path, exclude_kwargs=exclude_kwargs, - content_fingerprint=content_fingerprint, + custom_fingerprint=custom_fingerprint, ) if self._ignore_cache: return f @@ -144,14 +152,6 @@ def fingerprinted(path, *args, **kwargs): + (list(exclude_kwargs) if exclude_kwargs is not None else []), ) - def call_fingerprinted(key, args, kwargs): - # inject the fingerprint (and tokens) into the signature - kwargs_ = kwargs.copy() - kwargs_[fingerprint_kwarg] = key + ( - tuple(self._tokens) if self._tokens else () - ) - return fingerprinted(*args, **kwargs_) - @wraps(f) def fingerprinter(*args, **kwargs): # we need to dereference symlinks and use that path in the function @@ -159,52 +159,57 @@ def fingerprinter(*args, **kwargs): bound = sig.bind(*args, **kwargs) bound.apply_defaults() path_orig = bound.arguments[path_arg] - if content_fingerprint is not None: - cfprint = content_fingerprint(path_orig) - if cfprint is not None: - lgr.debug( - "Calling memoized version of %s for content fingerprint %r", - f, - cfprint, - ) - # No modified_in_window() check: content cannot change - # without its fingerprint changing - ret = call_fingerprinted(("content", cfprint), args, kwargs) - lgr.log(1, "Returning value %r", ret) - return ret - try: - path = op.realpath(path_orig) - except TypeError: + fprint = self._get_fingerprint(path_orig, custom_fingerprint) + if fprint is None: lgr.debug( - "Calling %s directly since argument is not a path-like object", f + "Calling %s directly since no fingerprint for %r", f, path_orig ) - return f(*args, **kwargs) - if path != path_orig: - lgr.log(5, "Dereferenced %r into %r", path_orig, path) - if op.isdir(path): - fprint = self._get_dir_fingerprint(path) - else: - fprint = self._get_file_fingerprint(path) - if fprint is None: - lgr.debug("Calling %s directly since no fingerprint for %r", f, path) - # just call the function -- we have no fingerprint, - # probably does not exist or permissions are wrong + # just call the function -- we have no fingerprint, probably + # not a path, does not exist, or permissions are wrong ret = f(*args, **kwargs) # We should still pass through if file was modified just now, # since that could mask out quick modifications. # Target use cases will not be like that. elif fprint.modified_in_window(self._min_dtime): - lgr.debug("Calling %s directly since too short for %r", f, path) + lgr.debug("Calling %s directly since too short for %r", f, path_orig) ret = f(*args, **kwargs) else: - lgr.debug("Calling memoized version of %s for %s", f, path) - ret = call_fingerprinted((path,) + fprint.to_tuple(), args, kwargs) + lgr.debug("Calling memoized version of %s for %r", f, path_orig) + # If there is a fingerprint -- inject it into the signature + kwargs_ = kwargs.copy() + kwargs_[fingerprint_kwarg] = fprint.to_tuple() + ( + tuple(self._tokens) if self._tokens else () + ) + ret = fingerprinted(*args, **kwargs_) lgr.log(1, "Returning value %r", ret) return ret # and we memoize actually that function return fingerprinter + def _get_fingerprint(self, value, custom_fingerprint=None): + """ + Fingerprint of ``value``: the custom one if any, else one of the file + or directory at its dereferenced path, or `None` + """ + if custom_fingerprint is not None: + custom = custom_fingerprint(value) + if custom is not None: + lgr.log(5, "Custom fingerprint for %r: %r", value, custom) + return CustomFingerprint(custom) + try: + path = op.realpath(value) + except TypeError: + lgr.debug("Cannot fingerprint %r: not a path-like object", value) + return None + if path != value: + lgr.log(5, "Dereferenced %r into %r", value, path) + if op.isdir(path): + fprint = self._get_dir_fingerprint(path, custom_fingerprint) + else: + fprint = self._get_file_fingerprint(path) + return None if fprint is None else PathFingerprint(path, fprint) + @staticmethod def _get_file_fingerprint(path): """Simplistic generic file fingerprinting based on ctime, mtime, and size""" @@ -219,7 +224,7 @@ def _get_file_fingerprint(path): lgr.debug(f"Cannot fingerprint {path}: {exc}") @staticmethod - def _get_dir_fingerprint(path): + def _get_dir_fingerprint(path, custom_fingerprint=None): fprint = DirFingerprint() dirqueue = deque([path]) try: @@ -227,7 +232,14 @@ def _get_dir_fingerprint(path): d = dirqueue.popleft() with os.scandir(d) as entries: for e in entries: - if e.is_dir(follow_symlinks=True): + custom = ( + custom_fingerprint(e.path) + if custom_fingerprint is not None + else None + ) + if custom is not None: + fprint.add_custom(e.path, custom) + elif e.is_dir(follow_symlinks=True): dirqueue.append(e.path) else: s = e.stat(follow_symlinks=True) @@ -251,22 +263,56 @@ def to_tuple(self): return tuple(self) +class PathFingerprint: + """A fingerprint of a file or directory, together with its path""" + + def __init__(self, path, fprint): + self.path = path + self.fprint = fprint + + def modified_in_window(self, min_dtime): + return self.fprint.modified_in_window(min_dtime) + + def to_tuple(self): + return (self.path,) + self.fprint.to_tuple() + + +class CustomFingerprint: + """A fingerprint returned by a ``custom_fingerprint`` callable""" + + def __init__(self, value): + self.value = value + + def modified_in_window(self, min_dtime): # noqa: U100 + # it is up to the callable to change the fingerprint on modification + return False + + def to_tuple(self): + # a path key starts with an absolute path, never with this marker + return ("custom", self.value) + + class DirFingerprint: def __init__(self): self.last_modified = None self.hash = None def add_file(self, path, fprint: FileFingerprint): - fprint_hash = md5( - ascii((str(path), fprint.to_tuple())).encode("us-ascii") - ).digest() - if self.hash is None: - self.hash = fprint_hash + self._add_hash( + md5(ascii((str(path), fprint.to_tuple())).encode("us-ascii")).digest() + ) + if self.last_modified is None or self.last_modified < fprint.mtime_ns: self.last_modified = fprint.mtime_ns - else: - self.hash = xor_bytes(self.hash, fprint_hash) - if self.last_modified < fprint.mtime_ns: - self.last_modified = fprint.mtime_ns + + def add_custom(self, path, value): + self._add_hash( + md5(ascii((str(path), ("custom", value))).encode("us-ascii")).digest() + ) + + def _add_hash(self, fprint_hash): + self.hash = ( + fprint_hash if self.hash is None else xor_bytes(self.hash, fprint_hash) + ) def modified_in_window(self, min_dtime): if self.last_modified is None: diff --git a/src/fscacher/tests/test_annex.py b/src/fscacher/tests/test_annex.py new file mode 100644 index 0000000..162647c --- /dev/null +++ b/src/fscacher/tests/test_annex.py @@ -0,0 +1,277 @@ +from functools import partial +import os +import os.path as op +from pathlib import Path +import shutil +import subprocess +import time +import pytest +from .. import PersistentCache, annex_key_fingerprint + +KEY = "SHA256E-s7--0123456789abcdef.dat" + + +@pytest.fixture(scope="function") +def cache(tmp_path_factory): + return PersistentCache(path=tmp_path_factory.mktemp("cache")) + + +def annex_link(repo: Path, relpath: str, key: str, content="content") -> Path: + """ + Create a locked annexed file in a fake annex layout; with ``content=None``, + its content is missing (as if dropped) + """ + obj = repo / ".git" / "annex" / "objects" / "Xx" / "Yy" / key / key + obj.parent.mkdir(parents=True, exist_ok=True) + if content is not None: + obj.write_text(content) + link = repo / relpath + link.parent.mkdir(parents=True, exist_ok=True) + try: + os.symlink(op.relpath(obj, link.parent), link) + except OSError: + pytest.skip("symlinks are not supported here") + return link + + +def drop(link: Path) -> None: + os.unlink(op.realpath(link)) + + +@pytest.mark.parametrize( + "key", + [ + KEY, + "SHA256-s7--0123456789abcdef", + "SHA1-s7--0123", + "SHA512E-s7--0123.nwb", + "SHA3_256E-s7--0123.dat", + "SKEIN512-s7--0123", + "BLAKE2B256E-s7--0123.dat", + "BLAKE2SP224-s7--0123", + "MD5E-s7--0123.dat", + ], +) +def test_annex_key_fingerprint_content_hash_backends(tmp_path, key): + link = annex_link(tmp_path, "sub/file.dat", key) + assert annex_key_fingerprint(link) == (str(link), key) + assert annex_key_fingerprint(str(link)) == (str(link), key) + assert annex_key_fingerprint(os.fsencode(link)) == (str(link), key) + assert annex_key_fingerprint(link, pair_with_path=False) == key + # Relative paths are paired as absolute ones + assert annex_key_fingerprint(op.relpath(link)) == (str(link), key) + + +@pytest.mark.parametrize( + "key", + [ + "WORM-s7-m1700000000--file.dat", + "URL--https&c%%example.com%file.dat", + "VURL--https&c%%example.com%file.dat", + "XFOO-s7--0123", + "SHA256Z-s7--0123", + ], +) +def test_annex_key_fingerprint_other_backends(tmp_path, key): + assert annex_key_fingerprint(annex_link(tmp_path, "file.dat", key)) is None + + +def test_annex_key_fingerprint_not_annexed(tmp_path): + regular = tmp_path / "regular.dat" + regular.write_text("content") + other = tmp_path / "other.dat" + try: + # a link to a file named like a key, but not in an annex object store + (tmp_path / KEY).mkdir() + (tmp_path / KEY / KEY).write_text("content") + os.symlink(op.join(KEY, KEY), other) + except OSError: + pytest.skip("symlinks are not supported here") + for value in [regular, other, tmp_path, tmp_path / "missing", 42, None]: + assert annex_key_fingerprint(value) is None + + +def test_annex_key_fingerprint_dropped(tmp_path): + link = annex_link(tmp_path, "file.dat", KEY, content=None) + assert not op.exists(link) + assert annex_key_fingerprint(link) == (str(link), KEY) + + +def make_reader(cache, calls, **kwargs): + @cache.memoize_path(**kwargs) + def read(path): + calls.append(str(path)) + with open(path) as f: + return f"{op.basename(path)}:{f.read()}" + + return read + + +@pytest.mark.parametrize("pair_with_path", [True, False]) +def test_memoize_path_annex_twins(cache, tmp_path, pair_with_path): + calls = [] + read = make_reader( + cache, + calls, + custom_fingerprint=partial( + annex_key_fingerprint, pair_with_path=pair_with_path + ), + ) + # Twins: same key at different paths, with different extensions + a = annex_link(tmp_path, "a.dat", KEY) + b = annex_link(tmp_path, "sub/b.nwb", KEY) + assert read(a) == "a.dat:content" + assert read(a) == "a.dat:content" + assert calls == [str(a)] + if pair_with_path: + # each path has its own entry, as the result may depend on it + assert read(b) == "b.nwb:content" + assert read(b) == "b.nwb:content" + assert calls == [str(a), str(b)] + else: + # the result is shared, even though here it depends on the path + assert read(b) == "a.dat:content" + assert calls == [str(a)] + + +def test_memoize_path_annex_dropped(cache, tmp_path): + calls = [] + read = make_reader(cache, calls, custom_fingerprint=annex_key_fingerprint) + link = annex_link(tmp_path, "file.dat", KEY) + assert read(link) == "file.dat:content" + drop(link) + # The result cached while the content was present is still returned + assert read(link) == "file.dat:content" + assert calls == [str(link)] + # A file whose content was never present is read (and fails) every time + missing = annex_link(tmp_path, "missing.dat", "SHA256E-s3--fedcba.dat", None) + for _ in range(2): + with pytest.raises(FileNotFoundError): + read(missing) + assert calls == [str(link), str(missing), str(missing)] + + +def test_memoize_path_annex_fallback_to_stat(cache, tmp_path): + calls = [] + read = make_reader(cache, calls, custom_fingerprint=annex_key_fingerprint) + # A WORM key does not vouch for the content: stat() is used, which cannot + # fingerprint a dropped file, so the function is called every time + worm = annex_link(tmp_path, "worm.dat", "WORM-s7-m1--worm.dat", None) + for _ in range(2): + with pytest.raises(FileNotFoundError): + read(worm) + assert calls == [str(worm)] * 2 + # An unlocked file (a regular file) is fingerprinted by stat(), so its + # modifications are noticed + calls.clear() + unlocked = tmp_path / "unlocked.dat" + unlocked.write_text("content") + time.sleep(cache._min_dtime * 1.1) + assert read(unlocked) == "unlocked.dat:content" + assert read(unlocked) == "unlocked.dat:content" + time.sleep(cache._min_dtime * 1.1) + unlocked.write_text("edited") + time.sleep(cache._min_dtime * 1.1) + assert read(unlocked) == "unlocked.dat:edited" + assert calls == [str(unlocked)] * 2 + + +def test_memoize_path_annex_directory(cache, tmp_path): + calls = [] + + def listdir(path): + calls.append(str(path)) + return sorted( + (p.relative_to(path).as_posix(), op.exists(p)) + for p in Path(path).rglob("*") + if not p.is_dir() or p.is_symlink() + ) + + with_keys = cache.memoize_path(listdir, custom_fingerprint=annex_key_fingerprint) + with_stat = cache.memoize_path(listdir) + ds = tmp_path / "ds" + # the object store lives outside of the directory, as for a .zarr in a + # dataset + for name in ["a.dat", "sub/b.dat"]: + annex_link(tmp_path, f"ds/{name}", f"SHA256E-s7--{name[-5]}.dat") + drop(ds / "sub" / "b.dat") + (ds / "regular.txt").write_text("regular") + time.sleep(cache._min_dtime * 1.1) + expected = [("a.dat", True), ("regular.txt", True), ("sub/b.dat", False)] + # With stat() alone, the dropped file prevents fingerprinting the directory + assert with_stat(ds) == with_stat(ds) == expected + assert calls == [str(ds)] * 2 + # With keys, the directory is fingerprinted and its listing cached + calls.clear() + assert with_keys(ds) == with_keys(ds) == expected + assert calls == [str(ds)] + # ... until a file changes: an annexed one gets another key ... + os.unlink(ds / "a.dat") + annex_link(tmp_path, "ds/a.dat", "SHA256E-s6--other.dat") + assert with_keys(ds) == expected + assert len(calls) == 2 + # ... or a regular file is modified + time.sleep(cache._min_dtime * 1.1) + (ds / "regular.txt").write_text("modified") + time.sleep(cache._min_dtime * 1.1) + assert with_keys(ds) == with_keys(ds) == expected + assert len(calls) == 3 + + +@pytest.mark.skipif(shutil.which("git-annex") is None, reason="git annex required") +def test_memoize_path_git_annex(cache, tmp_path, monkeypatch): + for var in ["GIT_AUTHOR_NAME", "GIT_COMMITTER_NAME"]: + monkeypatch.setenv(var, "Test") + for var in ["GIT_AUTHOR_EMAIL", "GIT_COMMITTER_EMAIL"]: + monkeypatch.setenv(var, "test@example.com") + + def git(*args): + subprocess.run(["git", *args], cwd=tmp_path, check=True) + + git("init", "-q") + git("annex", "init", "-q") + git("config", "annex.backend", "SHA256E") + (tmp_path / "a.dat").write_text("content") + (tmp_path / "sub").mkdir() + (tmp_path / "sub" / "b.dat").write_text("content") + (tmp_path / "c.dat").write_text("other") + (tmp_path / "w.dat").write_text("worm") + git("annex", "add", "-q", "a.dat", "sub", "c.dat") + git("-c", "annex.backend=WORM", "annex", "add", "-q", "w.dat") + git("commit", "-q", "-m", "Add files") + a, b, c, w = (tmp_path / p for p in ["a.dat", "sub/b.dat", "c.dat", "w.dat"]) + if not op.islink(a): + pytest.skip("git-annex does not lock files here (adjusted branch?)") + + key = annex_key_fingerprint(a, pair_with_path=False) + assert key.startswith("SHA256E-s7--") + assert annex_key_fingerprint(b, pair_with_path=False) == key + assert annex_key_fingerprint(w) is None + + calls = [] + read = make_reader( + cache, + calls, + custom_fingerprint=partial(annex_key_fingerprint, pair_with_path=False), + ) + assert read(a) == "a.dat:content" + assert read(b) == "a.dat:content" + assert calls == [str(a)] + + # Unlocked, a file is fingerprinted by stat(), and edits are noticed + assert read(c) == "c.dat:other" + git("annex", "unlock", "-q", "c.dat") + assert annex_key_fingerprint(c) is None + time.sleep(cache._min_dtime * 1.1) + assert read(c) == "c.dat:other" + time.sleep(cache._min_dtime * 1.1) + c.write_text("edited") + time.sleep(cache._min_dtime * 1.1) + assert read(c) == "c.dat:edited" + assert calls == [str(a), str(c), str(c), str(c)] + + # Dropped, a locked file's cached result is still returned + git("annex", "drop", "-q", "--force", "a.dat") + assert not op.exists(a) + assert read(a) == "a.dat:content" + assert len(calls) == 4 diff --git a/src/fscacher/tests/test_cache.py b/src/fscacher/tests/test_cache.py index 5a3fedb..ac7d3b6 100644 --- a/src/fscacher/tests/test_cache.py +++ b/src/fscacher/tests/test_cache.py @@ -573,10 +573,10 @@ def blob_fingerprint(x): return x.fingerprint if isinstance(x, Blob) else None -def test_memoize_path_content_fingerprint(cache, tmp_path): +def test_memoize_path_custom_fingerprint(cache, tmp_path): calls = [] - @cache.memoize_path(content_fingerprint=blob_fingerprint) + @cache.memoize_path(custom_fingerprint=blob_fingerprint) def memoread(src, arg=0, kwarg=None): calls.append((src, arg, kwarg)) if isinstance(src, Blob): @@ -584,7 +584,7 @@ def memoread(src, arg=0, kwarg=None): with open(src) as f: return f"{f.read()}:{arg}:{kwarg}" - # A value without a fingerprint is not cached + # A value without a fingerprint (nor a path) is not cached assert memoread(Blob("content")) == "content:0:None" assert memoread(Blob("content")) == "content:0:None" assert len(calls) == 2 @@ -603,16 +603,21 @@ def memoread(src, arg=0, kwarg=None): assert memoread(Blob("content", "A"), 1, kwarg="q") == "content:1:q" assert len(calls) == 5 - # Paths are still fingerprinted by stat(), whatever the content + # Paths, for which the callable returns None, are fingerprinted by stat() path = tmp_path / "file.dat" path.write_text("content") time.sleep(cache._min_dtime * 1.1) assert memoread(path, 1) == "content:1:None" assert memoread(path, 1) == "content:1:None" assert len(calls) == 6 + time.sleep(cache._min_dtime * 1.1) + path.write_text("changed") + time.sleep(cache._min_dtime * 1.1) + assert memoread(path, 1) == "changed:1:None" + assert len(calls) == 7 -def test_memoize_path_content_fingerprint_tokens(tmp_path_factory): +def test_memoize_path_custom_fingerprint_tokens(tmp_path_factory): calls = [] def memoread(src): @@ -622,8 +627,8 @@ def memoread(src): path = tmp_path_factory.mktemp("cache") c1 = PersistentCache(path=path, tokens=["1"]) c2 = PersistentCache(path=path, tokens=["2"]) - m1 = c1.memoize_path(memoread, content_fingerprint=blob_fingerprint) - m2 = c2.memoize_path(memoread, content_fingerprint=blob_fingerprint) + m1 = c1.memoize_path(memoread, custom_fingerprint=blob_fingerprint) + m2 = c2.memoize_path(memoread, custom_fingerprint=blob_fingerprint) assert m1(Blob("content", "A")) == "content" assert m1(Blob("never read", "A")) == "content" assert len(calls) == 1 @@ -631,12 +636,12 @@ def memoread(src): assert len(calls) == 2 -def test_memoize_path_content_fingerprint_ignored(monkeypatch, tmp_path): +def test_memoize_path_custom_fingerprint_ignored(monkeypatch, tmp_path): monkeypatch.setenv("FSCACHER_CACHE", "ignore") cache = PersistentCache(path=tmp_path) calls = [] - @cache.memoize_path(content_fingerprint=blob_fingerprint) + @cache.memoize_path(custom_fingerprint=blob_fingerprint) def memoread(src): calls.append(src) return src.content From 7dd8e7dd009121f71f745c8055005fe581de6521 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 23:03:33 +0000 Subject: [PATCH 03/12] test_annex: do not compute a relative path across drives On Windows CI the temporary directory is on C: while the working directory is on D:, so os.path.relpath() raised. Test relative paths by changing into the temporary directory instead. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Qn5WBSiQgoZoL4fytF6nEr --- src/fscacher/tests/test_annex.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/src/fscacher/tests/test_annex.py b/src/fscacher/tests/test_annex.py index 162647c..863a21e 100644 --- a/src/fscacher/tests/test_annex.py +++ b/src/fscacher/tests/test_annex.py @@ -52,14 +52,15 @@ def drop(link: Path) -> None: "MD5E-s7--0123.dat", ], ) -def test_annex_key_fingerprint_content_hash_backends(tmp_path, key): +def test_annex_key_fingerprint_content_hash_backends(tmp_path, monkeypatch, key): link = annex_link(tmp_path, "sub/file.dat", key) assert annex_key_fingerprint(link) == (str(link), key) assert annex_key_fingerprint(str(link)) == (str(link), key) assert annex_key_fingerprint(os.fsencode(link)) == (str(link), key) assert annex_key_fingerprint(link, pair_with_path=False) == key # Relative paths are paired as absolute ones - assert annex_key_fingerprint(op.relpath(link)) == (str(link), key) + monkeypatch.chdir(tmp_path) + assert annex_key_fingerprint(op.join("sub", "file.dat")) == (str(link), key) @pytest.mark.parametrize( From 302f55db11cb9183e09c0782ffbae675d7d6b406 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 00:40:21 +0000 Subject: [PATCH 04/12] Deny Claude posting comments/reviews on GitHub PRs and issues Add project-level .claude/settings.json denying the GitHub MCP tools that post, edit, or resolve comments and reviews on pull requests and issues. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Qn5WBSiQgoZoL4fytF6nEr --- .claude/settings.json | 13 +++++++++++++ 1 file changed, 13 insertions(+) create mode 100644 .claude/settings.json diff --git a/.claude/settings.json b/.claude/settings.json new file mode 100644 index 0000000..724c239 --- /dev/null +++ b/.claude/settings.json @@ -0,0 +1,13 @@ +{ + "permissions": { + "deny": [ + "mcp__github__add_issue_comment", + "mcp__github__update_issue_comment", + "mcp__github__add_reply_to_pull_request_comment", + "mcp__github__add_comment_to_pending_review", + "mcp__github__pull_request_review_write", + "mcp__github__resolve_review_thread", + "mcp__github__unresolve_review_thread" + ] + } +} From a5a181a2bfc789789d582e4e4646e3f3e8cec8be Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 00:53:57 +0000 Subject: [PATCH 05/12] Document pair_with_path of annex_key_fingerprint as a parameter Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Qn5WBSiQgoZoL4fytF6nEr --- src/fscacher/annex.py | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/src/fscacher/annex.py b/src/fscacher/annex.py index 55d5661..f7591c8 100644 --- a/src/fscacher/annex.py +++ b/src/fscacher/annex.py @@ -29,11 +29,14 @@ def annex_key_fingerprint(path, *, pair_with_path=True): so this is cheap, and a file whose content is not present (e.g., dropped) is still fingerprinted: cached results are then returned for it. - With ``pair_with_path`` (the default), the fingerprint is the pair of the - absolute path (as given, not dereferenced) and the key, so results are only - shared by files at the same path, as needed when they depend on the path - (e.g., on the extension or on neighboring files). Without it, results are - shared by all files with the same key, e.g., across clones of a dataset. + Parameters + ---------- + pair_with_path: bool, optional + If true (the default), the fingerprint is the pair of the absolute path + (as given, not dereferenced) and the key, so results are only shared by + files at the same path, as needed when they depend on the path (e.g., on + the extension or on neighboring files). If false, results are shared by + all files with the same key, e.g., across clones of a dataset. """ try: path = os.fsdecode(os.fspath(path)) From e30360535683796f931d4a384286906008cd4d97 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 00:54:28 +0000 Subject: [PATCH 06/12] Trim custom_fingerprint docstring in memoize_path Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Qn5WBSiQgoZoL4fytF6nEr --- src/fscacher/cache.py | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/src/fscacher/cache.py b/src/fscacher/cache.py index 498a161..c9538f5 100644 --- a/src/fscacher/cache.py +++ b/src/fscacher/cache.py @@ -99,10 +99,7 @@ def memoize_path(self, f=None, *, exclude_kwargs=None, custom_fingerprint=None): `fscacher.annex.annex_key_fingerprint`. It is called with the value of the first argument, and with the path of each entry met while fingerprinting a directory, and returns either a fingerprint of it or - `None` to fall back to ``stat()``. It is called on every call of the - decorated function, so it must be cheap, and it must return `None` - rather than raise for anything it does not recognize (including - plain paths, unless it fingerprints them). + `None` to fall back to ``stat()``. A fingerprint must change whenever the result may change (there is no "modified just now" window for it), and must be picklable and have From ad4f67b5798a416aa0bb59e356488892fcbb3769 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 00:54:58 +0000 Subject: [PATCH 07/12] Drop fingerprint contract paragraph from memoize_path docstring Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Qn5WBSiQgoZoL4fytF6nEr --- src/fscacher/cache.py | 8 -------- 1 file changed, 8 deletions(-) diff --git a/src/fscacher/cache.py b/src/fscacher/cache.py index c9538f5..30a287a 100644 --- a/src/fscacher/cache.py +++ b/src/fscacher/cache.py @@ -100,14 +100,6 @@ def memoize_path(self, f=None, *, exclude_kwargs=None, custom_fingerprint=None): of the first argument, and with the path of each entry met while fingerprinting a directory, and returns either a fingerprint of it or `None` to fall back to ``stat()``. - - A fingerprint must change whenever the result may change (there is - no "modified just now" window for it), and must be picklable and have - a stable ``repr()`` (e.g., a string or a tuple of strings). Results - are shared between all values with equal fingerprints, wherever they - are: include the path in the fingerprint unless the result does not - depend on it. As the value itself is not part of the cache key, it - need not be a path at all. """ if f is None: return partial( From 37ce6fd07297fec40e64854181920e323c1dca01 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 00:57:19 +0000 Subject: [PATCH 08/12] Pass os.DirEntry to custom_fingerprint during directory walks; trim README Passing the DirEntry lets annex_key_fingerprint skip regular files using the file type from the directory listing, without a readlink() system call, so walking a large tree of mostly regular files costs no more than before. Also trim the README section on custom_fingerprint. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Qn5WBSiQgoZoL4fytF6nEr --- README.rst | 30 ++++++------------------------ src/fscacher/annex.py | 3 +++ src/fscacher/cache.py | 6 +++--- src/fscacher/tests/test_annex.py | 18 ++++++++++++++++++ 4 files changed, 30 insertions(+), 27 deletions(-) diff --git a/README.rst b/README.rst index ee342fa..5e07861 100644 --- a/README.rst +++ b/README.rst @@ -48,34 +48,16 @@ of ``path``, the cache is ignored. be a sequence of names of arguments of the decorated function that will be ignored for caching purposes. -``memoize_path()`` also optionally takes a ``custom_fingerprint`` callable, an -alternative to the built-in ``stat()``-based fingerprint. It is called with the -value of the first argument (and with the path of each entry met while -fingerprinting a directory) and returns either a fingerprint (a picklable value -with a stable ``repr()``, such as a string or a tuple of strings) or ``None`` to -fall back to ``stat()``. It runs on every call, so it must be cheap and return -``None`` rather than raise for anything it does not recognize. Results are -shared between all values with equal fingerprints, and a custom fingerprint is -trusted to change whenever the result may change (there is no "modified just -now" window for it); the value itself need not even be a path. - -``fscacher.annex_key_fingerprint`` is such a callable for git-annex'ed files: it -fingerprints a *locked* file by its key if the key's backend hashes the content -(``SHA*``, ``SHA3_*``, ``SKEIN*``, ``BLAKE2*``, ``MD5``, with or without the -``E`` suffix), and leaves anything else (unlocked files, ``WORM`` or ``URL`` -keys, ...) to ``stat()``. Results then survive the file being moved, or its -content dropped. By default, the key is paired with the file's path; pass -``pair_with_path=False`` to share results between all files with the same -content, e.g. across clones, if the result does not depend on the path: +``memoize_path()`` can also take a ``custom_fingerprint`` callable to use +instead of ``stat()``; it returns ``None`` to fall back to ``stat()``. For +example, ``fscacher.annex_key_fingerprint`` fingerprints locked git-annex'ed +files by their keys, so results survive their content being dropped: .. code:: python - from functools import partial - from fscacher import PersistentCache, annex_key_fingerprint + from fscacher import annex_key_fingerprint - @cache.memoize_path( - custom_fingerprint=partial(annex_key_fingerprint, pair_with_path=False) - ) + @cache.memoize_path(custom_fingerprint=annex_key_fingerprint) def foo(path, ...): ... diff --git a/src/fscacher/annex.py b/src/fscacher/annex.py index f7591c8..f271454 100644 --- a/src/fscacher/annex.py +++ b/src/fscacher/annex.py @@ -38,6 +38,9 @@ def annex_key_fingerprint(path, *, pair_with_path=True): the extension or on neighboring files). If false, results are shared by all files with the same key, e.g., across clones of a dataset. """ + if isinstance(path, os.DirEntry) and not path.is_symlink(): + # known from the directory listing, without a system call + return None try: path = os.fsdecode(os.fspath(path)) target = os.fsdecode(os.readlink(path)) diff --git a/src/fscacher/cache.py b/src/fscacher/cache.py index 30a287a..0c64291 100644 --- a/src/fscacher/cache.py +++ b/src/fscacher/cache.py @@ -97,8 +97,8 @@ def memoize_path(self, f=None, *, exclude_kwargs=None, custom_fingerprint=None): custom_fingerprint: callable, optional An alternative to the built-in ``stat()``-based fingerprint, e.g. `fscacher.annex.annex_key_fingerprint`. It is called with the value - of the first argument, and with the path of each entry met while - fingerprinting a directory, and returns either a fingerprint of it or + of the first argument, and with the `os.DirEntry` of each entry met + while fingerprinting a directory, and returns either a fingerprint or `None` to fall back to ``stat()``. """ if f is None: @@ -222,7 +222,7 @@ def _get_dir_fingerprint(path, custom_fingerprint=None): with os.scandir(d) as entries: for e in entries: custom = ( - custom_fingerprint(e.path) + custom_fingerprint(e) if custom_fingerprint is not None else None ) diff --git a/src/fscacher/tests/test_annex.py b/src/fscacher/tests/test_annex.py index 863a21e..aac8c0b 100644 --- a/src/fscacher/tests/test_annex.py +++ b/src/fscacher/tests/test_annex.py @@ -98,6 +98,24 @@ def test_annex_key_fingerprint_dropped(tmp_path): assert annex_key_fingerprint(link) == (str(link), KEY) +def test_annex_key_fingerprint_dir_entries(tmp_path, monkeypatch): + link = annex_link(tmp_path, "ds/file.dat", KEY) + (tmp_path / "ds" / "regular.dat").write_text("content") + readlinks = [] + readlink = os.readlink + + def spy(path): + readlinks.append(os.fspath(path)) + return readlink(path) + + monkeypatch.setattr(os, "readlink", spy) + with os.scandir(tmp_path / "ds") as entries: + fprints = {e.name: annex_key_fingerprint(e) for e in entries} + assert fprints == {"file.dat": (str(link), KEY), "regular.dat": None} + # Only the symlink is read: a regular file is recognized from the listing + assert readlinks == [str(link)] + + def make_reader(cache, calls, **kwargs): @cache.memoize_path(**kwargs) def read(path): From e0eca3e0c498cb87d649ac77f0e5d3e7def77014 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 00:58:27 +0000 Subject: [PATCH 09/12] Add design notes for custom fingerprints and git-annex keys Record the design decisions trimmed from the docstrings and README: the custom_fingerprint contract, which annex files are fingerprinted by key and why, pair_with_path, dropped content, and directory walks. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Qn5WBSiQgoZoL4fytF6nEr --- docs/design/git-annex-content.md | 124 +++++++++++++++++++++++++++++++ 1 file changed, 124 insertions(+) create mode 100644 docs/design/git-annex-content.md diff --git a/docs/design/git-annex-content.md b/docs/design/git-annex-content.md new file mode 100644 index 0000000..e069097 --- /dev/null +++ b/docs/design/git-annex-content.md @@ -0,0 +1,124 @@ +# Custom fingerprints and git-annex content + +Design notes for `memoize_path(custom_fingerprint=...)` and +`fscacher.annex_key_fingerprint`, introduced in +[#113](https://github.com/con/fscacher/pull/113). The docstrings and README +only say what these are; this records why they work the way they do. + +## Motivation + +`memoize_path` keys its cache on a `stat()` of the path argument (mtime, +ctime, size, inode). Some resources can vouch for their content better: + +- A *locked* git-annex'ed file is a symlink whose target names a key; for + content-hash backends, the key pins the content. It also survives the file + being moved, re-cloned, or its content dropped. +- Non-path objects (e.g. dandi-cli's `Readable`s, which may stream remote + content) have no `stat()` at all, but may know a fingerprint of their + content. + +## `custom_fingerprint`: an alternative to `stat()`, not a second code path + +The callable replaces only the `stat()`-based fingerprint; everything else +(falling back to a direct call when there is no fingerprint, injecting the +fingerprint into the cache key, tokens) is shared. `_get_fingerprint()` +returns either a `CustomFingerprint` or a `PathFingerprint`, or `None`, and the +rest of `memoize_path` does not care which. `PathFingerprint` keys exactly as +before, so existing caches stay valid. + +The contract for a callable: + +- **It returns `None` to fall back to `stat()`**, for anything it does not + recognize, including plain paths unless it fingerprints them. It must not + raise for such values. +- **It runs on every call** of the decorated function (and per entry of a + directory, see below), so it must be cheap. `annex_key_fingerprint` only + reads the symlink; it never runs git-annex. +- **Equal fingerprints share results, wherever they are.** The value of the + path argument is excluded from the cache key, so a fingerprint must include + the path unless the result does not depend on it. For the same reason the + value need not be a path at all. +- **Fingerprints must be picklable with a stable `repr()`**, as joblib hashes + them into the key (e.g. a string or a tuple of strings). +- **There is no "modified just now" window.** With `stat()`, a file modified + within `_min_dtime` is not cached, since a quick later modification might + not change the fingerprint. A custom fingerprint is trusted to change + whenever the result may change; the callable is responsible for that. This + is only justified for fingerprints that pin the content, like hash-based + annex keys. + +## Which git-annex files are fingerprinted by key + +Only **locked** files: a symlink whose target looks like +`.../annex/objects/*/*/KEY/KEY`. The link is read with `readlink()` only. + +Only **content-hash backends**: `SHA1`, `SHA224`/`256`/`384`/`512`, +`SHA3_224`/`256`/`384`/`512`, `SKEIN256`/`512`, `BLAKE2B*`/`BP*`/`S*`/`SP*`, +and `MD5`, each with or without the `E` suffix (which adds the extension). +Everything else falls back to `stat()`: + +- `WORM` keys are made of size, mtime and file name, so they do not pin the + content. +- `URL` and `VURL` keys name a URL whose content may change. +- External `X*` backends are unknown. + +**Unlocked files** also fall back to `stat()`: an unlocked file is a regular +file whose key is not updated until it is re-added, so an edit would not +change the key. This includes every file on an *adjusted unlocked branch*, +which git-annex uses on filesystems without usable symlinks or permissions, +notably native Windows. There, `annex_key_fingerprint` is a no-op. + +WSL behaves like Linux on its own filesystem (e.g. under `~`). On a Windows +drive under `/mnt/c`, git-annex is likely to treat the filesystem as crippled +and use an adjusted branch, so the same `stat()` fallback applies. Neither is +tested in CI. + +## `pair_with_path` + +By default the fingerprint is `(absolute path, key)`: the path as given, not +dereferenced (dereferencing would give the object path, identical for all +files with the key). Results are then only shared by files at the same path, +which is needed whenever the result depends on the path, e.g. on the extension +(the key's extension may differ from the file's) or on neighboring files. + +With `pair_with_path=False`, the fingerprint is the key alone, and results are +shared by all files with the same content: twins in a dataset, and the same +file across clones. Only use it for results that depend on the content only. + +## Dropped content + +With `stat()`, a locked file whose content was dropped is a broken symlink: it +cannot be fingerprinted, so the function is called directly every time. + +With the key, it is still fingerprinted, so a result cached while the content +was present is returned after `git annex drop`. This is useful (e.g. metadata +of files no longer present locally) and deliberate. A file whose content was +never present has no cached result, so the function is called and fails as +before. + +## Directories + +`memoize_path` fingerprints a directory argument by walking it and combining +the `stat()` of every entry (this predates custom fingerprints). With `stat()` +alone, a single dropped annexed file (a broken symlink) makes the whole +directory unfingerprintable, and the mtimes of annexed files feed the +"modified just now" window. + +The walk now also consults the callable for each entry and, if it returns a +fingerprint, uses it for that entry instead of `stat()` (and without feeding +the window). A tree with dropped annexed files, such as a `.zarr` in a +dandiset, can then be cached. + +The callable is given the `os.DirEntry` of each entry rather than its path. A +`DirEntry` is path-like, so callables written for paths keep working, and it +lets `annex_key_fingerprint` skip regular files using the file type from the +directory listing, without a `readlink()` system call. The cost of the walk is +then unchanged for regular files, and for locked annexed files a `readlink()` +replaces a `stat()` that would follow the link. + +**Open question.** The walk is still O(number of entries) on every call, as +before. For very large trees (Zarrs with millions of chunks) a cheaper +fingerprint would be needed, e.g. the git tree hash of a committed directory, +and it is not yet decided whether custom fingerprints should take part in +directory walks at all in the first release, or whether directories should be +left to `stat()` until that is designed. From 6468a256005c5ec229e5f7c7fbd6fd25e737a725 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 01:36:36 +0000 Subject: [PATCH 10/12] Remove project-level Claude settings Keep that configuration personal for now, as asked in review. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Qn5WBSiQgoZoL4fytF6nEr --- .claude/settings.json | 13 ------------- 1 file changed, 13 deletions(-) delete mode 100644 .claude/settings.json diff --git a/.claude/settings.json b/.claude/settings.json deleted file mode 100644 index 724c239..0000000 --- a/.claude/settings.json +++ /dev/null @@ -1,13 +0,0 @@ -{ - "permissions": { - "deny": [ - "mcp__github__add_issue_comment", - "mcp__github__update_issue_comment", - "mcp__github__add_reply_to_pull_request_comment", - "mcp__github__add_comment_to_pending_review", - "mcp__github__pull_request_review_write", - "mcp__github__resolve_review_thread", - "mcp__github__unresolve_review_thread" - ] - } -} From 0ae8e87d9b688704d1e12aa35c4a74897c10d717 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 16:15:12 +0000 Subject: [PATCH 11/12] Only consult custom_fingerprint for the top-level argument The directory walk is the stat() code path, which is only taken once custom_fingerprint returned None for the directory, so it no longer consults the callable per entry: a directory is fingerprinted as before, unless the callable fingerprints the whole tree itself. Per-entry fingerprints (e.g. for Zarrs with dropped annexed chunks) are left as an open question in the design notes. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Qn5WBSiQgoZoL4fytF6nEr --- docs/design/git-annex-content.md | 38 ++++++++++------------ src/fscacher/annex.py | 3 -- src/fscacher/cache.py | 43 +++++++++--------------- src/fscacher/tests/test_annex.py | 56 ++++++++++---------------------- 4 files changed, 51 insertions(+), 89 deletions(-) diff --git a/docs/design/git-annex-content.md b/docs/design/git-annex-content.md index e069097..1eb4403 100644 --- a/docs/design/git-annex-content.md +++ b/docs/design/git-annex-content.md @@ -31,8 +31,7 @@ The contract for a callable: - **It returns `None` to fall back to `stat()`**, for anything it does not recognize, including plain paths unless it fingerprints them. It must not raise for such values. -- **It runs on every call** of the decorated function (and per entry of a - directory, see below), so it must be cheap. `annex_key_fingerprint` only +- **It runs on every call** of the decorated function, so it must be cheap. `annex_key_fingerprint` only reads the symlink; it never runs git-annex. - **Equal fingerprints share results, wherever they are.** The value of the path argument is excluded from the cache key, so a fingerprint must include @@ -104,21 +103,20 @@ alone, a single dropped annexed file (a broken symlink) makes the whole directory unfingerprintable, and the mtimes of annexed files feed the "modified just now" window. -The walk now also consults the callable for each entry and, if it returns a -fingerprint, uses it for that entry instead of `stat()` (and without feeding -the window). A tree with dropped annexed files, such as a `.zarr` in a -dandiset, can then be cached. - -The callable is given the `os.DirEntry` of each entry rather than its path. A -`DirEntry` is path-like, so callables written for paths keep working, and it -lets `annex_key_fingerprint` skip regular files using the file type from the -directory listing, without a `readlink()` system call. The cost of the walk is -then unchanged for regular files, and for locked annexed files a `readlink()` -replaces a `stat()` that would follow the link. - -**Open question.** The walk is still O(number of entries) on every call, as -before. For very large trees (Zarrs with millions of chunks) a cheaper -fingerprint would be needed, e.g. the git tree hash of a committed directory, -and it is not yet decided whether custom fingerprints should take part in -directory walks at all in the first release, or whether directories should be -left to `stat()` until that is designed. +**The callable only sees the top-level argument.** For a directory, it may +return a fingerprint of the whole tree itself; if it returns `None` (as +`annex_key_fingerprint` does), the directory is walked and `stat()`-ed exactly +as before, dropped files included. This keeps a single call site and a single +kind of argument, so the callable stays a plain alternative to `stat()`. + +An earlier iteration also consulted the callable for each entry met during the +walk (passing the `os.DirEntry`, so that `annex_key_fingerprint` could skip +regular files without a system call), so that trees with dropped annexed files +could be cached. It was dropped: it made the callable part of the `stat()` +code path through a second, implicit contract, and it is better designed +together with the question below. + +**Open question.** The walk is O(number of entries) on every call. For very +large trees (Zarrs with millions of chunks, possibly with dropped chunks) a +cheaper fingerprint of the whole tree would be needed, e.g. the git tree hash +of a committed directory, or per-entry fingerprints within the walk as above. diff --git a/src/fscacher/annex.py b/src/fscacher/annex.py index f271454..f7591c8 100644 --- a/src/fscacher/annex.py +++ b/src/fscacher/annex.py @@ -38,9 +38,6 @@ def annex_key_fingerprint(path, *, pair_with_path=True): the extension or on neighboring files). If false, results are shared by all files with the same key, e.g., across clones of a dataset. """ - if isinstance(path, os.DirEntry) and not path.is_symlink(): - # known from the directory listing, without a system call - return None try: path = os.fsdecode(os.fspath(path)) target = os.fsdecode(os.readlink(path)) diff --git a/src/fscacher/cache.py b/src/fscacher/cache.py index 0c64291..3739565 100644 --- a/src/fscacher/cache.py +++ b/src/fscacher/cache.py @@ -97,9 +97,10 @@ def memoize_path(self, f=None, *, exclude_kwargs=None, custom_fingerprint=None): custom_fingerprint: callable, optional An alternative to the built-in ``stat()``-based fingerprint, e.g. `fscacher.annex.annex_key_fingerprint`. It is called with the value - of the first argument, and with the `os.DirEntry` of each entry met - while fingerprinting a directory, and returns either a fingerprint or - `None` to fall back to ``stat()``. + of the first argument only, and returns either a fingerprint or + `None` to fall back to ``stat()``. For a directory, it may return a + fingerprint of the whole tree, or `None` to fingerprint it by + ``stat()``-ing each file as usual. """ if f is None: return partial( @@ -194,7 +195,7 @@ def _get_fingerprint(self, value, custom_fingerprint=None): if path != value: lgr.log(5, "Dereferenced %r into %r", value, path) if op.isdir(path): - fprint = self._get_dir_fingerprint(path, custom_fingerprint) + fprint = self._get_dir_fingerprint(path) else: fprint = self._get_file_fingerprint(path) return None if fprint is None else PathFingerprint(path, fprint) @@ -213,7 +214,7 @@ def _get_file_fingerprint(path): lgr.debug(f"Cannot fingerprint {path}: {exc}") @staticmethod - def _get_dir_fingerprint(path, custom_fingerprint=None): + def _get_dir_fingerprint(path): fprint = DirFingerprint() dirqueue = deque([path]) try: @@ -221,14 +222,7 @@ def _get_dir_fingerprint(path, custom_fingerprint=None): d = dirqueue.popleft() with os.scandir(d) as entries: for e in entries: - custom = ( - custom_fingerprint(e) - if custom_fingerprint is not None - else None - ) - if custom is not None: - fprint.add_custom(e.path, custom) - elif e.is_dir(follow_symlinks=True): + if e.is_dir(follow_symlinks=True): dirqueue.append(e.path) else: s = e.stat(follow_symlinks=True) @@ -287,21 +281,16 @@ def __init__(self): self.hash = None def add_file(self, path, fprint: FileFingerprint): - self._add_hash( - md5(ascii((str(path), fprint.to_tuple())).encode("us-ascii")).digest() - ) - if self.last_modified is None or self.last_modified < fprint.mtime_ns: + fprint_hash = md5( + ascii((str(path), fprint.to_tuple())).encode("us-ascii") + ).digest() + if self.hash is None: + self.hash = fprint_hash self.last_modified = fprint.mtime_ns - - def add_custom(self, path, value): - self._add_hash( - md5(ascii((str(path), ("custom", value))).encode("us-ascii")).digest() - ) - - def _add_hash(self, fprint_hash): - self.hash = ( - fprint_hash if self.hash is None else xor_bytes(self.hash, fprint_hash) - ) + else: + self.hash = xor_bytes(self.hash, fprint_hash) + if self.last_modified < fprint.mtime_ns: + self.last_modified = fprint.mtime_ns def modified_in_window(self, min_dtime): if self.last_modified is None: diff --git a/src/fscacher/tests/test_annex.py b/src/fscacher/tests/test_annex.py index aac8c0b..1dd2622 100644 --- a/src/fscacher/tests/test_annex.py +++ b/src/fscacher/tests/test_annex.py @@ -98,24 +98,6 @@ def test_annex_key_fingerprint_dropped(tmp_path): assert annex_key_fingerprint(link) == (str(link), KEY) -def test_annex_key_fingerprint_dir_entries(tmp_path, monkeypatch): - link = annex_link(tmp_path, "ds/file.dat", KEY) - (tmp_path / "ds" / "regular.dat").write_text("content") - readlinks = [] - readlink = os.readlink - - def spy(path): - readlinks.append(os.fspath(path)) - return readlink(path) - - monkeypatch.setattr(os, "readlink", spy) - with os.scandir(tmp_path / "ds") as entries: - fprints = {e.name: annex_key_fingerprint(e) for e in entries} - assert fprints == {"file.dat": (str(link), KEY), "regular.dat": None} - # Only the symlink is read: a regular file is recognized from the listing - assert readlinks == [str(link)] - - def make_reader(cache, calls, **kwargs): @cache.memoize_path(**kwargs) def read(path): @@ -196,8 +178,15 @@ def test_memoize_path_annex_fallback_to_stat(cache, tmp_path): def test_memoize_path_annex_directory(cache, tmp_path): + seen = [] + + def fingerprint(path): + seen.append(path) + return annex_key_fingerprint(path) + calls = [] + @cache.memoize_path(custom_fingerprint=fingerprint) def listdir(path): calls.append(str(path)) return sorted( @@ -206,35 +195,24 @@ def listdir(path): if not p.is_dir() or p.is_symlink() ) - with_keys = cache.memoize_path(listdir, custom_fingerprint=annex_key_fingerprint) - with_stat = cache.memoize_path(listdir) ds = tmp_path / "ds" # the object store lives outside of the directory, as for a .zarr in a # dataset for name in ["a.dat", "sub/b.dat"]: annex_link(tmp_path, f"ds/{name}", f"SHA256E-s7--{name[-5]}.dat") - drop(ds / "sub" / "b.dat") (ds / "regular.txt").write_text("regular") time.sleep(cache._min_dtime * 1.1) - expected = [("a.dat", True), ("regular.txt", True), ("sub/b.dat", False)] - # With stat() alone, the dropped file prevents fingerprinting the directory - assert with_stat(ds) == with_stat(ds) == expected - assert calls == [str(ds)] * 2 - # With keys, the directory is fingerprinted and its listing cached - calls.clear() - assert with_keys(ds) == with_keys(ds) == expected + # The callable only sees the top-level argument, for which it returns None, + # so the directory is fingerprinted by stat()-ing each file, as without it + expected = [("a.dat", True), ("regular.txt", True), ("sub/b.dat", True)] + assert listdir(ds) == listdir(ds) == expected assert calls == [str(ds)] - # ... until a file changes: an annexed one gets another key ... - os.unlink(ds / "a.dat") - annex_link(tmp_path, "ds/a.dat", "SHA256E-s6--other.dat") - assert with_keys(ds) == expected - assert len(calls) == 2 - # ... or a regular file is modified - time.sleep(cache._min_dtime * 1.1) - (ds / "regular.txt").write_text("modified") - time.sleep(cache._min_dtime * 1.1) - assert with_keys(ds) == with_keys(ds) == expected - assert len(calls) == 3 + assert seen == [ds, ds] + # ... and thus a dropped file prevents fingerprinting the directory + drop(ds / "sub" / "b.dat") + expected[-1] = ("sub/b.dat", False) + assert listdir(ds) == listdir(ds) == expected + assert calls == [str(ds)] * 3 @pytest.mark.skipif(shutil.which("git-annex") is None, reason="git annex required") From 4f89cd7479aaa66ff92b42879906b67528359c9a Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 17:44:55 +0000 Subject: [PATCH 12/12] Bring design notes in line with the code Clarify that a custom fingerprint must include the path itself (as annex_key_fingerprint does by default), that exceptions from the callable propagate, and that the paired path is not dereferenced. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Qn5WBSiQgoZoL4fytF6nEr --- docs/design/git-annex-content.md | 25 ++++++++++++++++++------- 1 file changed, 18 insertions(+), 7 deletions(-) diff --git a/docs/design/git-annex-content.md b/docs/design/git-annex-content.md index 1eb4403..dbb2252 100644 --- a/docs/design/git-annex-content.md +++ b/docs/design/git-annex-content.md @@ -30,13 +30,20 @@ The contract for a callable: - **It returns `None` to fall back to `stat()`**, for anything it does not recognize, including plain paths unless it fingerprints them. It must not - raise for such values. -- **It runs on every call** of the decorated function, so it must be cheap. `annex_key_fingerprint` only - reads the symlink; it never runs git-annex. -- **Equal fingerprints share results, wherever they are.** The value of the - path argument is excluded from the cache key, so a fingerprint must include - the path unless the result does not depend on it. For the same reason the - value need not be a path at all. + raise for such values: an exception propagates, so the decorated call + fails. This is deliberate, as an exception there most likely is a bug in + the callable, which should not silently disable caching. (A way to say "do + not cache", e.g. a named `fscacher.NO_CACHE` constant, can be added if ever + needed.) +- **It runs on every call** of the decorated function, so it must be cheap. + `annex_key_fingerprint` only reads the symlink; it never runs git-annex. +- **Equal fingerprints share results, wherever they are.** The path argument + itself is not part of the cache key: in `stat()` mode the (dereferenced) path + is part of the fingerprint, but a custom fingerprint is used as is. So a + callable must include the path in its fingerprint unless the result does not + depend on it, as `annex_key_fingerprint` does by default (see + `pair_with_path` below). For the same reason the value need not be a path + at all. - **Fingerprints must be picklable with a stable `repr()`**, as joblib hashes them into the key (e.g. a string or a tuple of strings). - **There is no "modified just now" window.** With `stat()`, a file modified @@ -79,6 +86,10 @@ dereferenced (dereferencing would give the object path, identical for all files with the key). Results are then only shared by files at the same path, which is needed whenever the result depends on the path, e.g. on the extension (the key's extension may differ from the file's) or on neighboring files. +The path is made absolute with `os.path.abspath`, which does not resolve +symlinks (neither the file's nor its parent directories'), so the same file +reached through different paths gets separate entries; that errs on the safe +side. With `pair_with_path=False`, the fingerprint is the key alone, and results are shared by all files with the same content: twins in a dataset, and the same