Skip to content
Draft
9 changes: 7 additions & 2 deletions dandi/cli/cmd_validate.py
Original file line number Diff line number Diff line change
Expand Up @@ -170,8 +170,13 @@ def _filter_results(
"in a datalad dataset without fetched data). 'error' (default) emits a "
"concise error per file, 'skip' skips each such file with a warning, "
"'only-non-data' skips content-dependent validators but still validates "
"path layout.",
type=click.Choice(["error", "only-non-data", "skip"], case_sensitive=True),
"path layout, 'stream' streams the content of annexed files with "
"datalad-fuse so that content-dependent validators run without the files "
"having to be downloaded (requires git-annex and `pip install "
"'dandi[datalad]'`).",
type=click.Choice(
["error", "only-non-data", "skip", "stream"], case_sensitive=True
),
default="error",
)
@click.option(
Expand Down
43 changes: 42 additions & 1 deletion dandi/cli/tests/test_cmd_validate.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
import json
from pathlib import Path
from typing import cast
import shutil
from typing import Any, cast

from click.testing import CliRunner
import pytest
Expand Down Expand Up @@ -998,6 +999,46 @@ def test_validate_missing_file_content_no_broken_symlinks(tmp_path: Path) -> Non
assert "FILE_CONTENT_MISSING" not in r_skip.output


@pytest.mark.ai_generated
def test_validate_missing_file_content_stream(
tmp_path: Path, simple3_nwb: Path, range_http_server: tuple[Any, str]
) -> None:
"""--missing-file-content=stream validates annexed content with datalad-fuse."""
from ...tests.fixtures import make_git_annex_dandiset
from ...tests.skip import skipif

skipif.no_git_annex()
pytest.importorskip("datalad_fuse.fsspec")
_, base_url = range_http_server
shutil.copy(simple3_nwb, tmp_path / "served" / "content.nwb")
ds = tmp_path / "ds"
make_git_annex_dandiset(
ds, {"sub-001/sub-001.nwb": (simple3_nwb, f"{base_url}/content.nwb")}
)
out = tmp_path / "out.jsonl"
r = CliRunner().invoke(
validate,
[
"--missing-file-content",
"stream",
"--min-severity",
"INFO",
"-f",
"json_lines",
"-o",
str(out),
str(ds),
],
)
assert "Traceback" not in r.output
ids = [rec.id for rec in load_validation_jsonl([str(out)])]
assert "DANDI.FILE_CONTENT_STREAMED" in ids
assert "DANDI.FILE_CONTENT_MISSING" not in ids
# simple3_nwb lacks a subject_id, which only content-based checks notice
assert "NWBI.check_subject_id_exists" in ids
assert r.exit_code == 1


@pytest.mark.ai_generated
def test_validate_broken_symlink_path_argument(tmp_path: Path) -> None:
"""A broken symlink can be given directly as a path to validate.
Expand Down
82 changes: 73 additions & 9 deletions dandi/files/bases.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@
from pathlib import Path
import re
from threading import Lock
import traceback
from typing import IO, Any, Generic
from xml.etree.ElementTree import fromstring

Expand All @@ -32,7 +33,7 @@
set_asset_schema_key,
)
from dandi.metadata.core import get_default_metadata
from dandi.misctypes import DUMMY_DANDI_ETAG, Digest, LocalReadableFile, P
from dandi.misctypes import DUMMY_DANDI_ETAG, Digest, LocalReadableFile, P, Readable
from dandi.utils import post_upload_size_check, pre_upload_size_check, yaml_load
from dandi.validate._types import (
ORIGIN_INTERNAL_DANDI,
Expand Down Expand Up @@ -320,12 +321,22 @@ class LocalFileAsset(LocalAsset):
an asset of a Dandiset
"""

#: An alternative source for the asset's content, used instead of the file
#: at `filepath` when set. This is used to stream the content of an
#: annexed file (e.g., in a DataLad dataset) that is not present locally;
#: see `dandi.support.datalad_fuse`.
content_source: Readable | None = None

def _content(self) -> Path | Readable:
"""The source to read the asset's content from"""
return self.content_source if self.content_source is not None else self.filepath

def get_metadata(
self,
digest: Digest | None = None,
ignore_errors: bool = True,
) -> BareAsset:
metadata = get_default_metadata(self.filepath, digest=digest)
metadata = get_default_metadata(self._content(), digest=digest)
metadata.path = self.path
return metadata

Expand Down Expand Up @@ -515,7 +526,7 @@ def get_metadata(
from dandi.metadata.nwb import nwb2asset

try:
metadata = nwb2asset(self.filepath, digest=digest)
metadata = nwb2asset(self._content(), digest=digest)
except Exception as e:
lgr.warning(
"Failed to extract NWB metadata from %s: %s: %s",
Expand All @@ -524,7 +535,7 @@ def get_metadata(
str(e),
)
if ignore_errors:
metadata = get_default_metadata(self.filepath, digest=digest)
metadata = get_default_metadata(self._content(), digest=digest)
else:
raise
metadata.path = self.path
Expand Down Expand Up @@ -555,12 +566,16 @@ def get_validation_errors(
pass
else:
# Avoid heavy import by importing within function:
from nwbinspector import Importance, inspect_nwbfile, load_config
from nwbinspector import Importance, load_config

# Avoid heavy import by importing within function:
from dandi.pynwb_utils import validate as pynwb_validate

errors.extend(pynwb_validate(self.filepath, devel_debug=devel_debug))
errors.extend(
pynwb_validate(
self.filepath, devel_debug=devel_debug, readable=self.content_source
)
)
if schema_version is not None:
errors.extend(
super().get_validation_errors(
Expand All @@ -576,9 +591,7 @@ def get_validation_errors(
validator_version=str(_get_nwb_inspector_version()),
)

for error in inspect_nwbfile(
nwbfile_path=self.filepath,
skip_validate=True,
for error in self._inspect_nwbfile(
config=load_config(filepath_or_keyword="dandi"),
importance_threshold=Importance.BEST_PRACTICE_VIOLATION,
):
Expand Down Expand Up @@ -623,6 +636,57 @@ def get_validation_errors(
)
return errors

def _inspect_nwbfile(self, **kwargs: Any) -> Iterator[Any]:
"""
Yield the messages from inspecting the file with nwbinspector, reading
the content from `content_source` if it is set (in which case what
`nwbinspector.inspect_nwbfile()` does for a local file is replicated
here for the streamed content)
"""
# Avoid heavy import by importing within function:
from nwbinspector import (
Importance,
InspectorMessage,
inspect_nwbfile,
inspect_nwbfile_object,
)

if self.content_source is None:
yield from inspect_nwbfile(
nwbfile_path=self.filepath, skip_validate=True, **kwargs
)
return

# Avoid heavy import by importing within function:
import h5py
from pynwb import NWBHDF5IO

from dandi.pynwb_utils import open_readable

try:
with open_readable(self.content_source) as fp, h5py.File(
fp, "r"
) as h5, NWBHDF5IO(file=h5, mode="r", load_namespaces=True) as io:
nwbfile = io.read()
for message in inspect_nwbfile_object(nwbfile_object=nwbfile, **kwargs):
if message is not None:
message.file_path = str(self.filepath)
yield message
except Exception as e:
# Report the failure the same way inspect_nwbfile() does for a
# local file that PyNWB cannot read
yield InspectorMessage(
message=traceback.format_exc(),
importance=Importance.ERROR,
check_function_name=(
f"During io.read(), an error occurred: "
f"{type(e).__module__}.{type(e).__name__}. "
"This indicates that PyNWB was unable to read the file. "
"See the traceback message for more details."
),
file_path=str(self.filepath),
)


class VideoAsset(LocalFileAsset):
pass
Expand Down
2 changes: 1 addition & 1 deletion dandi/files/bids.py
Original file line number Diff line number Diff line change
Expand Up @@ -250,7 +250,7 @@ def get_metadata(
) -> BareAsset:
metadata = self.bids_dataset_description.get_asset_metadata(self)
start_time = end_time = datetime.now().astimezone()
add_common_metadata(metadata, self.filepath, start_time, end_time, digest)
add_common_metadata(metadata, self._content(), start_time, end_time, digest)
metadata.path = self.path
return metadata

Expand Down
51 changes: 44 additions & 7 deletions dandi/pynwb_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -611,8 +611,9 @@ def rename_if_matched(_name: str, obj: Any) -> None:
f.visititems(rename_if_matched)


@validate_cache.memoize_path
def validate(path: str | Path, devel_debug: bool = False) -> list[ValidationResult]:
def validate(
path: str | Path, devel_debug: bool = False, *, readable: Readable | None = None
) -> list[ValidationResult]:
"""Run validation on a file and return errors

In case of an exception being thrown, an error message added to the
Expand All @@ -621,8 +622,38 @@ def validate(path: str | Path, devel_debug: bool = False) -> list[ValidationResu
Parameters
----------
path: str or Path
Path of the file, as reported in the returned results
devel_debug: bool, optional
Whether to re-raise exceptions instead of reporting them as errors
readable: Readable, optional
If given, the file's content is read from this `Readable` instead of from
``path`` (e.g., to stream the content of an annexed file which is not
present locally); the results are then cached only if it is an
`AnnexedReadableFile` (see `annex_fingerprint`).
"""
source: str | Readable = readable if readable is not None else str(path)
# fscacher is untyped, so the memoized _validate returns Any for mypy
return cast(
"list[ValidationResult]",
_validate(source, str(path), devel_debug=devel_debug),
)


@validate_cache.memoize_path(custom_fingerprint=annex_fingerprint)
def _validate(
source: str | Readable, path: str, devel_debug: bool = False
) -> list[ValidationResult]:
"""`validate` proper, with the content source as first argument for caching

Parameters
----------
source: str or Readable
What to read the file's content from: its path or a `Readable`
path: str
Path of the file, as reported in the returned results
devel_debug: bool
Whether to re-raise exceptions instead of reporting them as errors
"""
path = str(path) # Might come in as pathlib's PATH
errors: list[ValidationResult] = []

# To overcome
Expand All @@ -634,7 +665,7 @@ def validate(path: str | Path, devel_debug: bool = False) -> list[ValidationResu
)
version = None
try:
version = get_nwb_version(path, sanitize=False)
version = get_nwb_version(source, sanitize=False)
except Exception:
# we just will not remove any errors, it is required so should be some
pass
Expand Down Expand Up @@ -680,9 +711,15 @@ def validate(path: str | Path, devel_debug: bool = False) -> list[ValidationResu
)

try:
# Validates against the namespaces cached in the file; pynwb falls back
# to its own namespaces if the file has none cached
error_outputs = pynwb.validate(path=path)
# Either way, validates against the namespaces cached in the file; pynwb
# falls back to its own namespaces if the file has none cached
if isinstance(source, Readable):
with open_readable(source) as fp, h5py.File(fp, "r") as h5, NWBHDF5IO(
file=h5, mode="r", load_namespaces=True
) as reader:
error_outputs = pynwb.validate(io=reader)
else:
error_outputs = pynwb.validate(path=source)
except Exception as exc:
if devel_debug:
raise
Expand Down
Loading
Loading