Coverage for python/lsst/images/ndf/_input_archive.py: 82%
321 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-09-29 10:33 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-09-29 10:33 +0000
1# This file is part of lsst-images.
2#
3# Developed for the LSST Data Management System.
4# This product includes software developed by the LSST Project
5# (https://www.lsst.org).
6# See the COPYRIGHT file at the top-level directory of this distribution
7# for details of code ownership.
8#
9# Use of this source code is governed by a 3-clause BSD-style
10# license that can be found in the LICENSE file.
12from __future__ import annotations
14__all__ = ("NdfInputArchive", "read_starlink")
16import logging
17from collections.abc import Callable, Iterator
18from contextlib import contextmanager
19from types import EllipsisType
20from typing import IO, Any, Self
22import astropy.io.fits
23import astropy.table
24import astropy.units as u
25import h5py
26import numpy as np
28from lsst.resources import ResourcePath, ResourcePathExpression
30from .._geom import Box
31from .._image import Image
32from .._mask import Mask, MaskPlane, MaskSchema
33from .._masked_image import MaskedImage
34from .._transforms import FrameSet, SkyProjection
35from .._transforms import _ast as astshim
36from .._transforms._frames import GeneralFrame
37from ..fits._common import FitsOpaqueMetadata
38from ..serialization import (
39 ArchiveInfo,
40 ArchiveReadError,
41 ArchiveTree,
42 ArrayReferenceModel,
43 InlineArrayModel,
44 InputArchive,
45 TableModel,
46 no_header_updates,
47 parameterize_tree,
48 tree_class_for_info,
49)
50from ..serialization._backends import _is_binary_stream
51from ..serialization._common import _ARCHIVE_READ_CONTEXT, _check_format_version
52from . import _hds
53from ._common import NdfPointerModel
54from ._model import HdsPrimitive, NdfDocument
56_LOG = logging.getLogger(__name__)
58_NDF_FORMAT_VERSION = 1
59"""Container layout version this release of `NdfInputArchive` understands."""
61_LSST_EXTENSION_PREFIXES = ("/MORE/LSST", "/LSST")
62"""Fixed locations of the LSST extension structure within an NDF.
64Only these top-level locations are searched; nested pointer trees can
65therefore never be mistaken for the main one.
66"""
69def _get_lsst_primitive(
70 get_primitive: Callable[[str], HdsPrimitive | None], name: str
71) -> HdsPrimitive | None:
72 """Find a named primitive in the LSST extension structure.
74 Parameters
75 ----------
76 get_primitive
77 Callable returning the primitive at an absolute path, or `None` if
78 there is no primitive there.
79 name
80 Component name within the LSST extension, e.g. ``DATA_MODEL``.
81 """
82 for prefix in _LSST_EXTENSION_PREFIXES:
83 if (primitive := get_primitive(f"{prefix}/{name}")) is not None:
84 return primitive
85 return None
88def _read_format_version(get_primitive: Callable[[str], HdsPrimitive | None]) -> int | None:
89 """Read the container-layout ``FORMAT_VERSION`` from the LSST extension.
91 Returns
92 -------
93 `int` or `None`
94 The container layout version, or `None` if the file has no
95 ``FORMAT_VERSION``. `NdfOutputArchive` always writes one, so absence
96 means either no LSST extension at all -- a Starlink NDF, which has no
97 layout of ours to version and is read by `read_starlink` -- or a
98 damaged extension. Callers decide which of those they are looking at;
99 neither may be answered by guessing a version.
100 """
101 primitive = _get_lsst_primitive(get_primitive, "FORMAT_VERSION")
102 if primitive is None:
103 return None
104 # The writer emits the version as a 0-d int32 numpy array; .item()
105 # unwraps to a Python int.
106 return int(primitive.read_array().item())
109def _read_archive_info(get_primitive: Callable[[str], HdsPrimitive | None], source: str) -> ArchiveInfo:
110 """Read the schema URL and format version from the LSST extension.
112 Parameters
113 ----------
114 get_primitive
115 Callable returning the primitive at an absolute path, or `None` if
116 there is no primitive there.
117 source
118 Description of the file being read, used in error messages.
119 """
120 schema_url: str | None = None
121 if (data_model := _get_lsst_primitive(get_primitive, "DATA_MODEL")) is not None: 121 ↛ 124line 121 didn't jump to line 124 because the condition on line 121 was always true
122 lines = data_model.read_char_array()
123 schema_url = lines[0].strip() if lines else None
124 if not schema_url: 124 ↛ 125line 124 didn't jump to line 125 because the condition on line 124 was never true
125 raise ArchiveReadError(
126 f"Could not read the schema of {source} from /MORE/LSST/DATA_MODEL or /LSST/DATA_MODEL."
127 )
128 # DATA_MODEL was found, so this is one of our files and carries the layout
129 # stamp its writer emits alongside it.
130 if (format_version := _read_format_version(get_primitive)) is None:
131 raise ArchiveReadError(
132 f"Could not read the container format version of {source} from "
133 "/MORE/LSST/FORMAT_VERSION or /LSST/FORMAT_VERSION."
134 )
135 return ArchiveInfo.from_schema_url(schema_url, format_version=format_version)
138class NdfInputArchive(InputArchive[NdfPointerModel]):
139 """Reads HDS-on-HDF5 NDF files written by `NdfOutputArchive`.
141 Instances should only be constructed via the :meth:`open` context
142 manager.
144 Parameters
145 ----------
146 file
147 Open `h5py.File` handle. Owned by the caller of :meth:`open`;
148 the archive does not close it.
149 """
151 def __init__(self, file: h5py.File) -> None:
152 self._file = file
153 self._document = NdfDocument.from_hdf5(file)
154 self._opaque_metadata = FitsOpaqueMetadata()
155 self._deserialized_pointer_cache: dict[str, Any] = {}
156 self._frame_set_cache: dict[str, FrameSet] = {}
157 self._read_opaque_fits_metadata()
158 self._check_format_version()
160 @classmethod
161 def get_basic_info(cls, path: ResourcePathExpression) -> ArchiveInfo:
162 """Read the schema URL from the ``DATA_MODEL`` scalar and the
163 ``FORMAT_VERSION`` primitive without deserializing pixel data.
165 Reading the datasets directly from the HDF5 file avoids building
166 the full internal model, which would eagerly read the
167 (potentially large) JSON tree.
169 Parameters
170 ----------
171 path
172 Path to the archive to read.
173 """
174 ospath = ResourcePath(path).ospath
176 with h5py.File(ospath, "r") as handle:
178 def get_primitive(component_path: str) -> HdsPrimitive | None:
179 node = handle.get(component_path)
180 return HdsPrimitive.from_hdf5(node) if isinstance(node, h5py.Dataset) else None
182 return _read_archive_info(get_primitive, repr(path))
184 @classmethod
185 @contextmanager
186 def open_tree(
187 cls,
188 path: ResourcePathExpression | IO[bytes],
189 *,
190 partial: bool = True,
191 **backend_kwargs: Any,
192 ) -> Iterator[tuple[Self, ArchiveTree, ArchiveInfo]]:
193 """Open the NDF file and yield ``(archive, tree, info)``.
195 The schema is read from the open document's ``DATA_MODEL`` rather than
196 a separate `get_basic_info` open. Requires the symmetric LSST JSON
197 tree; ``partial`` is accepted but not meaningful, since h5py reads
198 lazily regardless.
200 Parameters
201 ----------
202 path
203 The file resource to open, or a seekable binary stream
204 containing the file's content.
205 partial
206 Accepted for interface compatibility but not meaningful; h5py
207 reads lazily regardless.
208 **backend_kwargs
209 Backend-specific options; none are currently used.
210 """
211 with cls.open(path) as archive:
212 if archive._get_main_json_path() is None: 212 ↛ 213line 212 didn't jump to line 213 because the condition on line 212 was never true
213 raise ArchiveReadError(
214 f"{path!r} has no LSST JSON tree; only the symmetric read path is supported."
215 )
216 info = archive.info
217 tree_cls = tree_class_for_info(info, path)
218 parameterized = parameterize_tree(tree_cls, NdfPointerModel)
219 tree = archive.get_tree(parameterized)
220 yield archive, tree, info
222 @classmethod
223 @contextmanager
224 def open(cls, path: ResourcePathExpression | IO[bytes]) -> Iterator[Self]:
225 """Open an NDF file for reading and yield an `NdfInputArchive`.
227 Remote ResourcePaths are materialised locally first; fsspec-direct
228 h5py reads are a deferred follow-up.
230 Parameters
231 ----------
232 path
233 Path to the NDF file to open, or a seekable binary stream
234 containing the file's content.
235 """
236 if _is_binary_stream(path):
237 with h5py.File(path, "r") as f:
238 yield cls(f)
239 return
240 rp = ResourcePath(path)
241 with rp.as_local() as local:
242 with h5py.File(local.ospath, "r") as f:
243 yield cls(f)
245 def get_tree[T: ArchiveTree](self, model_type: type[T]) -> T:
246 """Read and validate the main Pydantic tree at ``/MORE/LSST/JSON``.
248 Parameters
249 ----------
250 model_type
251 Archive tree model type to validate the JSON tree against.
252 """
253 json_path = self._get_main_json_path()
254 if json_path is None:
255 raise ArchiveReadError(
256 "File has no /MORE/LSST/JSON tree; this is either a "
257 "Starlink-only NDF (use ndf.read_starlink() for auto-detect) or "
258 "the file was written by an unrelated tool."
259 )
260 json_text = _read_json_record(self._get_primitive(json_path), json_path)
261 return model_type.model_validate_json(json_text, context=_ARCHIVE_READ_CONTEXT)
263 def deserialize_pointer[U: ArchiveTree, V](
264 self,
265 pointer: NdfPointerModel,
266 model_type: type[U],
267 deserializer: Callable[[U, InputArchive[NdfPointerModel]], V],
268 ) -> V:
269 # Cache by pointer.path so repeated dereferences reuse the same
270 # deserialised result and don't re-run the deserializer.
271 if (cached := self._deserialized_pointer_cache.get(pointer.path)) is not None:
272 return cached
273 if not self._has_model_path(pointer.path): 273 ↛ 274line 273 didn't jump to line 274 because the condition on line 273 was never true
274 raise ArchiveReadError(f"Pointer reference {pointer.path!r} not found in NDF file.")
275 primitive = self._get_primitive(pointer.path)
276 json_text = _read_json_record(primitive, pointer.path)
277 model = model_type.model_validate_json(json_text, context=_ARCHIVE_READ_CONTEXT)
278 result = deserializer(model, self)
279 self._deserialized_pointer_cache[pointer.path] = result
280 if isinstance(result, FrameSet):
281 self._frame_set_cache[pointer.path] = result
282 return result
284 def get_frame_set(self, pointer: NdfPointerModel) -> FrameSet:
285 try:
286 return self._frame_set_cache[pointer.path]
287 except KeyError:
288 raise AssertionError(
289 f"Frame set at {pointer.path!r} must be deserialised via "
290 f"deserialize_pointer before any dependent transform can be."
291 ) from None
293 def get_array(
294 self,
295 model: ArrayReferenceModel | InlineArrayModel,
296 *,
297 slices: tuple[slice, ...] | EllipsisType = ...,
298 strip_header: Callable[[astropy.io.fits.Header], None] = no_header_updates,
299 ) -> np.ndarray:
300 if isinstance(model, InlineArrayModel):
301 data: np.ndarray = np.array(model.data, dtype=model.datatype.to_numpy())
302 return data if slices is ... else data[slices]
303 if not isinstance(model.source, str) or not model.source.startswith("ndf:"):
304 raise ArchiveReadError(
305 f"NdfInputArchive cannot resolve array source {model.source!r}; "
306 f"expected an 'ndf:<HDF5-path>' reference."
307 )
308 path = model.source[len("ndf:") :]
309 if not self._has_model_path(path): 309 ↛ 310line 309 didn't jump to line 310 because the condition on line 309 was never true
310 raise ArchiveReadError(f"Array reference {path!r} not in file.")
311 primitive = self._get_primitive(path)
312 # h5py supports lazy slicing via dataset[slices].
313 if isinstance(primitive.data, h5py.Dataset): 313 ↛ 315line 313 didn't jump to line 315 because the condition on line 313 was always true
314 return primitive.data[()] if slices is ... else primitive.data[slices]
315 data = primitive.read_array()
316 return data if slices is ... else data[slices]
318 def get_table(
319 self,
320 model: TableModel,
321 strip_header: Callable[[astropy.io.fits.Header], None] = no_header_updates,
322 ) -> astropy.table.Table:
323 result = astropy.table.Table(meta=model.meta)
324 for column_model in model.columns:
325 if isinstance(column_model.data, InlineArrayModel): 325 ↛ 326line 325 didn't jump to line 326 because the condition on line 325 was never true
326 data: Any = column_model.data.data
327 else:
328 data = self.get_array(column_model.data, strip_header=strip_header)
329 result[column_model.name] = astropy.table.Column(
330 data,
331 name=column_model.name,
332 dtype=column_model.data.datatype.to_numpy(),
333 unit=column_model.unit,
334 description=column_model.description,
335 meta=column_model.meta,
336 )
337 return result
339 def get_structured_array(
340 self,
341 model: TableModel,
342 strip_header: Callable[[astropy.io.fits.Header], None] = no_header_updates,
343 ) -> np.ndarray:
344 return self.get_table(model, strip_header).as_array()
346 def _read_opaque_fits_metadata(self) -> None:
347 if not self._has_model_path("/MORE/FITS"):
348 return
349 cards = self._get_primitive("/MORE/FITS").read_char_array()
350 # FITS Header.fromstring expects fixed-width 80-char cards
351 # concatenated; pad each card defensively so readers tolerate
352 # files written with shorter widths.
353 header = astropy.io.fits.Header.fromstring("".join(c.ljust(80) for c in cards))
354 self._opaque_metadata.add_header(header, name="", ver=1)
356 def get_opaque_metadata(self) -> FitsOpaqueMetadata:
357 return self._opaque_metadata
359 @property
360 def info(self) -> ArchiveInfo:
361 """Schema/format info read from the open document's ``DATA_MODEL``
362 (`.serialization.ArchiveInfo`).
363 """
364 return _read_archive_info(self._get_optional_primitive, repr(self._file.filename))
366 def _get_main_json_path(self) -> str | None:
367 """Return the path of the main LSST JSON tree, if present."""
368 for prefix in _LSST_EXTENSION_PREFIXES:
369 path = f"{prefix}/JSON"
370 if self._has_model_path(path):
371 return path
372 return None
374 def _check_format_version(self) -> None:
375 """Read FORMAT_VERSION from the NDF top-level structure and check it.
377 A file carrying an LSST JSON tree was written by this package and so
378 must carry the layout stamp too; a missing stamp there is a damaged
379 file, not a legacy one. A Starlink NDF has no JSON tree and no layout
380 of ours to version, so there is nothing to check -- those are read by
381 `read_starlink`, which opens the archive through this same class.
383 DATA_MODEL is informational only on read; the JSON tree's
384 ``schema_version`` / ``min_read_version`` drive data-model
385 compatibility.
386 """
387 on_disk = _read_format_version(self._get_optional_primitive)
388 if on_disk is None:
389 if self._get_main_json_path() is not None:
390 raise ArchiveReadError(
391 f"{self._file.filename!r} has an LSST JSON tree but no container format version "
392 "at /MORE/LSST/FORMAT_VERSION or /LSST/FORMAT_VERSION."
393 )
394 return
395 _check_format_version("ndf", on_disk, _NDF_FORMAT_VERSION)
397 def _has_model_path(self, path: str) -> bool:
398 """Return `True` if a path exists in the NDF document model."""
399 try:
400 self._document.get(path)
401 except KeyError:
402 return False
403 return True
405 def _get_primitive(self, path: str) -> HdsPrimitive:
406 """Return a primitive component from the NDF document model."""
407 node = self._document.get(path)
408 if not isinstance(node, HdsPrimitive): 408 ↛ 409line 408 didn't jump to line 409 because the condition on line 408 was never true
409 raise ArchiveReadError(f"NDF reference {path!r} is not a primitive dataset.")
410 return node
412 def _get_optional_primitive(self, path: str) -> HdsPrimitive | None:
413 """Return a primitive from the document model, or `None` if there is
414 no primitive at ``path``.
415 """
416 try:
417 node = self._document.get(path)
418 except KeyError:
419 return None
420 return node if isinstance(node, HdsPrimitive) else None
423def read_starlink[T: Any](cls_: type[T], path: ResourcePathExpression) -> T:
424 """Reconstruct an `~lsst.images.Image` or `~lsst.images.MaskedImage`
425 from a schema-less Starlink NDF.
427 Files written by this package carry a ``/MORE/LSST/JSON`` tree and are
428 read through the generic `lsst.images.serialization.read_archive` /
429 `lsst.images.serialization.open_archive`. A Starlink-produced NDF has
430 no such tree and therefore no schema, so it cannot go through that
431 path; this function auto-detects a minimal recognised-component set
432 (``DATA_ARRAY``, ``VARIANCE``, ``QUALITY``, ``MORE.FITS``) instead.
433 ``WCS`` is reconstructed when possible; other components are
434 logged-and-dropped.
436 Parameters
437 ----------
438 cls_
439 Expected return type; `~lsst.images.Image` and
440 `~lsst.images.MaskedImage` are the only types the auto-detect path
441 can produce.
442 path
443 File path or `lsst.resources.ResourcePathExpression`.
445 Returns
446 -------
447 object
448 The deserialized ``cls`` instance.
450 Raises
451 ------
452 ArchiveReadError
453 If the file has an LSST JSON tree (use the generic ``read_archive``
454 instead) or no recognised ``DATA_ARRAY`` component.
455 """
456 with NdfInputArchive.open(path) as archive:
457 if archive._get_main_json_path() is not None: 457 ↛ 458line 457 didn't jump to line 458 because the condition on line 457 was never true
458 raise ArchiveReadError(
459 f"{path!r} has an LSST JSON tree; read it with serialization.read_archive()/open_archive()."
460 )
461 return _read_auto_detect(cls_, archive)
464def _read_auto_detect[T: Any](cls: type[T], archive: NdfInputArchive) -> T:
465 """Reconstruct an `Image` (or `MaskedImage`) from a Starlink NDF.
467 Recognised components: ``DATA_ARRAY`` (in either simple or complex
468 form), ``VARIANCE``, ``QUALITY``, ``MORE.FITS``. Other components
469 (``WCS``, ``HISTORY``, ``AXIS``, ``LABEL``, custom ``MORE.*``,
470 ``_LOGICAL`` primitives) are warned-and-dropped.
471 """
472 f = archive._file
473 ndf_group = _locate_ndf_root(f)
475 # DATA_ARRAY is required.
476 if "DATA_ARRAY" not in ndf_group:
477 raise ArchiveReadError(f"Auto-detect read of {f.filename!r}: no DATA_ARRAY component.")
478 data_arr, bbox = _read_data_array_with_bbox(ndf_group["DATA_ARRAY"])
480 # VARIANCE / QUALITY are optional.
481 variance_arr: np.ndarray | None = None
482 variance_bbox: Any | None = None
483 if "VARIANCE" in ndf_group:
484 variance_arr, variance_bbox = _read_data_array_with_bbox(ndf_group["VARIANCE"])
485 quality_arr: np.ndarray | None = None
486 quality_bbox: Any | None = None
487 quality_badbits = 255
488 if "QUALITY" in ndf_group and isinstance(ndf_group["QUALITY"], h5py.Group):
489 q = ndf_group["QUALITY"]
490 quality_badbits = _read_quality_badbits(q)
491 if "QUALITY" in q and isinstance(q["QUALITY"], h5py.Dataset): 491 ↛ 492line 491 didn't jump to line 492 because the condition on line 491 was never true
492 quality_arr = _validate_quality_array(_hds.read_array(q["QUALITY"]))
493 quality_bbox = _make_bbox(x_min=0, y_min=0, array=quality_arr)
494 elif "QUALITY" in q and isinstance(q["QUALITY"], h5py.Group): 494 ↛ 498line 494 didn't jump to line 498 because the condition on line 494 was always true
495 quality_arr, quality_bbox = _read_data_array_with_bbox(q["QUALITY"])
496 quality_arr = _validate_quality_array(quality_arr)
498 sky_projection: SkyProjection | None = None
499 if "WCS" in ndf_group:
500 try:
501 wcs_group = ndf_group["WCS"]
502 if isinstance(wcs_group, h5py.Group) and "DATA" in wcs_group: 502 ↛ 520line 502 didn't jump to line 520 because the condition on line 502 was always true
503 wcs_lines = _hds.read_char_array(wcs_group["DATA"])
504 wcs_text = _hds.decode_ndf_ast_data(wcs_lines)
505 ast_obj = astshim.Object.fromString(wcs_text)
506 if isinstance(ast_obj, astshim.FrameSet): 506 ↛ 520line 506 didn't jump to line 520 because the condition on line 506 was always true
507 pixel_frame = GeneralFrame(unit=u.pix)
508 sky_projection = SkyProjection.from_ast_frame_set(
509 ast_obj,
510 pixel_frame,
511 pixel_bounds=bbox,
512 )
513 except Exception:
514 _LOG.warning(
515 "Could not reconstruct Projection from WCS in %s; dropping.",
516 f.filename,
517 exc_info=True,
518 )
520 unit = _read_ndf_units(ndf_group)
522 # Anything unrecognised: warn-and-drop.
523 recognised = {
524 "DATA_ARRAY",
525 "VARIANCE",
526 "QUALITY",
527 "WCS",
528 "MORE",
529 "TITLE",
530 "LABEL",
531 "UNITS",
532 "HISTORY",
533 "AXIS",
534 }
535 for name in ndf_group:
536 if name not in recognised: 536 ↛ 537line 536 didn't jump to line 537 because the condition on line 536 was never true
537 _LOG.warning(
538 "Ignoring unrecognised NDF component %s/%s during auto-detect read.",
539 ndf_group.name,
540 name,
541 )
543 # Build the requested in-memory object. Any NDF can be read as an Image;
544 # MaskedImage construction uses whatever VARIANCE/QUALITY are present and
545 # lets the MaskedImage constructor provide defaults for missing planes.
546 image = Image(data_arr, bbox=bbox, unit=unit, sky_projection=sky_projection)
547 obj: Any
548 if cls is Image:
549 obj = image
550 elif issubclass(cls, MaskedImage):
551 if quality_arr is not None:
552 schema = _make_quality_mask_schema(quality_badbits)
553 mask = Mask(quality_arr[:, :, np.newaxis], schema=schema, bbox=quality_bbox)
554 else:
555 schema = MaskSchema([MaskPlane(name="BAD", description="Bad pixel.")])
556 mask = None
557 variance = Image(variance_arr, bbox=variance_bbox) if variance_arr is not None else None
558 obj = cls(
559 image=image,
560 mask=mask,
561 mask_schema=schema if mask is None else None,
562 variance=variance,
563 )
564 else:
565 raise ArchiveReadError(
566 f"Auto-detect can produce Image or MaskedImage, but caller asked for {cls.__name__}."
567 )
568 obj._opaque_metadata = archive.get_opaque_metadata()
569 return obj
572def _read_ndf_units(ndf_group: h5py.Group) -> u.UnitBase | None:
573 """Read the NDF UNITS component, if present."""
574 if "UNITS" not in ndf_group or not isinstance(ndf_group["UNITS"], h5py.Dataset):
575 return None
576 dataset = ndf_group["UNITS"]
577 if dataset.dtype.kind != "S": 577 ↛ 578line 577 didn't jump to line 578 because the condition on line 577 was never true
578 _LOG.warning("Ignoring non-character NDF UNITS component in %s.", ndf_group.name)
579 return None
580 if dataset.ndim == 0: 580 ↛ 588line 580 didn't jump to line 588 because the condition on line 580 was always true
581 raw = dataset[()]
582 if isinstance(raw, np.bytes_): 582 ↛ 584line 582 didn't jump to line 584 because the condition on line 582 was always true
583 raw = bytes(raw)
584 if not isinstance(raw, bytes): 584 ↛ 585line 584 didn't jump to line 585 because the condition on line 584 was never true
585 return None
586 units_text = raw.decode("ascii").rstrip(" ")
587 else:
588 records = _hds.read_char_array(dataset)
589 units_text = records[0] if records else ""
590 if not units_text: 590 ↛ 591line 590 didn't jump to line 591 because the condition on line 590 was never true
591 return None
592 for kwargs in ({"format": "fits"}, {}): 592 ↛ 597line 592 didn't jump to line 597 because the loop on line 592 didn't complete
593 try:
594 return u.Unit(units_text, **kwargs)
595 except ValueError:
596 continue
597 _LOG.warning("Could not parse NDF UNITS value %r in %s.", units_text, ndf_group.name)
598 return None
601def _read_quality_badbits(quality_group: h5py.Group) -> int:
602 """Read the scalar NDF QUALITY.BADBITS value."""
603 badbits = quality_group.get("BADBITS")
604 if not isinstance(badbits, h5py.Dataset): 604 ↛ 605line 604 didn't jump to line 605 because the condition on line 604 was never true
605 return 255
606 value = np.asarray(_hds.read_array(badbits)).reshape(-1)
607 if value.size == 0: 607 ↛ 608line 607 didn't jump to line 608 because the condition on line 607 was never true
608 return 255
609 return int(value[0])
612def _validate_quality_array(quality: np.ndarray) -> np.ndarray:
613 """Return an NDF QUALITY array as a `numpy.uint8` mask plane."""
614 if quality.dtype != np.dtype(np.uint8): 614 ↛ 615line 614 didn't jump to line 615 because the condition on line 614 was never true
615 raise ArchiveReadError(f"NDF QUALITY array has dtype {quality.dtype}; expected uint8.")
616 return quality
619def _make_quality_mask_schema(badbits: int) -> MaskSchema:
620 """Create a fallback `MaskSchema` for an unnamed 8-bit QUALITY array."""
621 planes = []
622 for bit in range(8):
623 mask = 1 << bit
624 description = f"NDF QUALITY bit {bit}."
625 if badbits & mask:
626 description += " Selected by BADBITS."
627 planes.append(MaskPlane(name=f"MASK{bit}", description=description))
628 return MaskSchema(planes, dtype=np.uint8)
631def _locate_ndf_root(f: h5py.File) -> h5py.Group:
632 """Return the group representing the top-level NDF.
634 Most files have the NDF at the root group itself. A few wrap it
635 in a single-child container at the root; we accept that shape
636 too. Anything more elaborate raises.
637 """
638 root_class = f["/"].attrs.get(_hds.ATTR_CLASS)
639 if isinstance(root_class, bytes):
640 root_class = root_class.decode("ascii")
641 if root_class == "NDF": 641 ↛ 644line 641 didn't jump to line 644 because the condition on line 641 was always true
642 return f["/"]
643 # Maybe a one-level container.
644 candidates = []
645 for name, child in f["/"].items():
646 if isinstance(child, h5py.Group):
647 cls_attr = child.attrs.get(_hds.ATTR_CLASS)
648 if isinstance(cls_attr, bytes):
649 cls_attr = cls_attr.decode("ascii")
650 if cls_attr == "NDF":
651 candidates.append(name)
652 if len(candidates) == 1:
653 return f[candidates[0]]
654 raise ArchiveReadError(
655 f"Could not locate top-level NDF in {f.filename!r}; "
656 f"expected the root group or a single NDF-typed child."
657 )
660def _read_data_array_with_bbox(
661 obj: h5py.Group | h5py.Dataset,
662) -> tuple[np.ndarray, Any]:
663 """Read a DATA_ARRAY component in either simple or complex form.
665 The complex form (what our writer always produces) is an HDS
666 ARRAY structure (h5py group with CLASS="ARRAY") containing
667 ``DATA`` and ``ORIGIN`` primitives. The simple form is a bare
668 primitive dataset.
670 Returns
671 -------
672 array, bbox : tuple
673 ``array`` is the C-order numpy data (shape ``(height, width)``
674 for 2D images). ``bbox`` is constructed from the ORIGIN if
675 present, else from a default origin of (0, 0).
676 """
677 if isinstance(obj, h5py.Dataset): 677 ↛ 679line 677 didn't jump to line 679 because the condition on line 677 was never true
678 # Simple form.
679 array = _hds.read_array(obj)
680 bbox = _make_bbox(x_min=0, y_min=0, array=array)
681 return array, bbox
682 # Complex form: an HDS structure with DATA + ORIGIN.
683 data = _hds.read_array(obj["DATA"])
684 if "ORIGIN" in obj:
685 origin = _hds.read_array(obj["ORIGIN"])
686 bbox = _make_bbox(x_min=int(origin[0]), y_min=int(origin[1]), array=data)
687 else:
688 bbox = _make_bbox(x_min=0, y_min=0, array=data)
689 return data, bbox
692def _read_json_record(primitive: HdsPrimitive, path: str) -> str:
693 """Read a JSON document stored as a single _CHAR*N record.
695 Our writer always emits JSON trees as a single-element character
696 array sized to the document. Joining multiple records would lose
697 trailing whitespace inside JSON string values, since
698 `read_char_array` strips trailing spaces per record.
699 """
700 records = primitive.read_char_array()
701 if len(records) != 1: 701 ↛ 702line 701 didn't jump to line 702 because the condition on line 701 was never true
702 raise ArchiveReadError(f"Expected a single _CHAR*N record at {path!r}, got {len(records)}.")
703 return records[0]
706def _make_bbox(*, x_min: int, y_min: int, array: np.ndarray) -> Any:
707 """Build an lsst.images.Box for a 2D image array.
709 The array is C-order ``(height, width)``. NDF stores ``ORIGIN``
710 in Fortran axis order ``(x_min, y_min)``.
711 """
712 if array.ndim != 2: 712 ↛ 713line 712 didn't jump to line 713 because the condition on line 712 was never true
713 raise ArchiveReadError(f"Auto-detect read only supports 2D arrays, got ndim={array.ndim}.")
714 # Box.from_shape takes (height, width) and start=(y_start, x_start).
715 return Box.from_shape(array.shape, start=(y_min, x_min))