Coverage for python/lsst/images/ndf/_input_archive.py: 82%

321 statements  

« prev     ^ index     » next       coverage.py v7.16.0, created at 2026-09-02 09:46 +0000

1# This file is part of lsst-images. 

2# 

3# Developed for the LSST Data Management System. 

4# This product includes software developed by the LSST Project 

5# (https://www.lsst.org). 

6# See the COPYRIGHT file at the top-level directory of this distribution 

7# for details of code ownership. 

8# 

9# Use of this source code is governed by a 3-clause BSD-style 

10# license that can be found in the LICENSE file. 

11 

12from __future__ import annotations 

13 

14__all__ = ("NdfInputArchive", "read_starlink") 

15 

16import logging 

17from collections.abc import Callable, Iterator 

18from contextlib import contextmanager 

19from types import EllipsisType 

20from typing import IO, Any, Self 

21 

22import astropy.io.fits 

23import astropy.table 

24import astropy.units as u 

25import h5py 

26import numpy as np 

27 

28from lsst.resources import ResourcePath, ResourcePathExpression 

29 

30from .._geom import Box 

31from .._image import Image 

32from .._mask import Mask, MaskPlane, MaskSchema 

33from .._masked_image import MaskedImage 

34from .._transforms import FrameSet, SkyProjection 

35from .._transforms import _ast as astshim 

36from .._transforms._frames import GeneralFrame 

37from ..fits._common import FitsOpaqueMetadata 

38from ..serialization import ( 

39 ArchiveInfo, 

40 ArchiveReadError, 

41 ArchiveTree, 

42 ArrayReferenceModel, 

43 InlineArrayModel, 

44 InputArchive, 

45 TableModel, 

46 no_header_updates, 

47 parameterize_tree, 

48 tree_class_for_info, 

49) 

50from ..serialization._backends import _is_binary_stream 

51from ..serialization._common import _ARCHIVE_READ_CONTEXT, _check_format_version 

52from . import _hds 

53from ._common import NdfPointerModel 

54from ._model import HdsPrimitive, NdfDocument 

55 

56_LOG = logging.getLogger(__name__) 

57 

58_NDF_FORMAT_VERSION = 1 

59"""Container layout version this release of `NdfInputArchive` understands.""" 

60 

61_LSST_EXTENSION_PREFIXES = ("/MORE/LSST", "/LSST") 

62"""Fixed locations of the LSST extension structure within an NDF. 

63 

64Only these top-level locations are searched; nested pointer trees can 

65therefore never be mistaken for the main one. 

66""" 

67 

68 

69def _get_lsst_primitive( 

70 get_primitive: Callable[[str], HdsPrimitive | None], name: str 

71) -> HdsPrimitive | None: 

72 """Find a named primitive in the LSST extension structure. 

73 

74 Parameters 

75 ---------- 

76 get_primitive 

77 Callable returning the primitive at an absolute path, or `None` if 

78 there is no primitive there. 

79 name 

80 Component name within the LSST extension, e.g. ``DATA_MODEL``. 

81 """ 

82 for prefix in _LSST_EXTENSION_PREFIXES: 

83 if (primitive := get_primitive(f"{prefix}/{name}")) is not None: 

84 return primitive 

85 return None 

86 

87 

88def _read_format_version(get_primitive: Callable[[str], HdsPrimitive | None]) -> int | None: 

89 """Read the container-layout ``FORMAT_VERSION`` from the LSST extension. 

90 

91 Returns 

92 ------- 

93 `int` or `None` 

94 The container layout version, or `None` if the file has no 

95 ``FORMAT_VERSION``. `NdfOutputArchive` always writes one, so absence 

96 means either no LSST extension at all -- a Starlink NDF, which has no 

97 layout of ours to version and is read by `read_starlink` -- or a 

98 damaged extension. Callers decide which of those they are looking at; 

99 neither may be answered by guessing a version. 

100 """ 

101 primitive = _get_lsst_primitive(get_primitive, "FORMAT_VERSION") 

102 if primitive is None: 

103 return None 

104 # The writer emits the version as a 0-d int32 numpy array; .item() 

105 # unwraps to a Python int. 

106 return int(primitive.read_array().item()) 

107 

108 

109def _read_archive_info(get_primitive: Callable[[str], HdsPrimitive | None], source: str) -> ArchiveInfo: 

110 """Read the schema URL and format version from the LSST extension. 

111 

112 Parameters 

113 ---------- 

114 get_primitive 

115 Callable returning the primitive at an absolute path, or `None` if 

116 there is no primitive there. 

117 source 

118 Description of the file being read, used in error messages. 

119 """ 

120 schema_url: str | None = None 

121 if (data_model := _get_lsst_primitive(get_primitive, "DATA_MODEL")) is not None: 121 ↛ 124line 121 didn't jump to line 124 because the condition on line 121 was always true

122 lines = data_model.read_char_array() 

123 schema_url = lines[0].strip() if lines else None 

124 if not schema_url: 124 ↛ 125line 124 didn't jump to line 125 because the condition on line 124 was never true

125 raise ArchiveReadError( 

126 f"Could not read the schema of {source} from /MORE/LSST/DATA_MODEL or /LSST/DATA_MODEL." 

127 ) 

128 # DATA_MODEL was found, so this is one of our files and carries the layout 

129 # stamp its writer emits alongside it. 

130 if (format_version := _read_format_version(get_primitive)) is None: 

131 raise ArchiveReadError( 

132 f"Could not read the container format version of {source} from " 

133 "/MORE/LSST/FORMAT_VERSION or /LSST/FORMAT_VERSION." 

134 ) 

135 return ArchiveInfo.from_schema_url(schema_url, format_version=format_version) 

136 

137 

138class NdfInputArchive(InputArchive[NdfPointerModel]): 

139 """Reads HDS-on-HDF5 NDF files written by `NdfOutputArchive`. 

140 

141 Instances should only be constructed via the :meth:`open` context 

142 manager. 

143 

144 Parameters 

145 ---------- 

146 file 

147 Open `h5py.File` handle. Owned by the caller of :meth:`open`; 

148 the archive does not close it. 

149 """ 

150 

151 def __init__(self, file: h5py.File) -> None: 

152 self._file = file 

153 self._document = NdfDocument.from_hdf5(file) 

154 self._opaque_metadata = FitsOpaqueMetadata() 

155 self._deserialized_pointer_cache: dict[str, Any] = {} 

156 self._frame_set_cache: dict[str, FrameSet] = {} 

157 self._read_opaque_fits_metadata() 

158 self._check_format_version() 

159 

160 @classmethod 

161 def get_basic_info(cls, path: ResourcePathExpression) -> ArchiveInfo: 

162 """Read the schema URL from the ``DATA_MODEL`` scalar and the 

163 ``FORMAT_VERSION`` primitive without deserializing pixel data. 

164 

165 Reading the datasets directly from the HDF5 file avoids building 

166 the full internal model, which would eagerly read the 

167 (potentially large) JSON tree. 

168 

169 Parameters 

170 ---------- 

171 path 

172 Path to the archive to read. 

173 """ 

174 ospath = ResourcePath(path).ospath 

175 

176 with h5py.File(ospath, "r") as handle: 

177 

178 def get_primitive(component_path: str) -> HdsPrimitive | None: 

179 node = handle.get(component_path) 

180 return HdsPrimitive.from_hdf5(node) if isinstance(node, h5py.Dataset) else None 

181 

182 return _read_archive_info(get_primitive, repr(path)) 

183 

184 @classmethod 

185 @contextmanager 

186 def open_tree( 

187 cls, 

188 path: ResourcePathExpression | IO[bytes], 

189 *, 

190 partial: bool = True, 

191 **backend_kwargs: Any, 

192 ) -> Iterator[tuple[Self, ArchiveTree, ArchiveInfo]]: 

193 """Open the NDF file and yield ``(archive, tree, info)``. 

194 

195 The schema is read from the open document's ``DATA_MODEL`` rather than 

196 a separate `get_basic_info` open. Requires the symmetric LSST JSON 

197 tree; ``partial`` is accepted but not meaningful, since h5py reads 

198 lazily regardless. 

199 

200 Parameters 

201 ---------- 

202 path 

203 The file resource to open, or a seekable binary stream 

204 containing the file's content. 

205 partial 

206 Accepted for interface compatibility but not meaningful; h5py 

207 reads lazily regardless. 

208 **backend_kwargs 

209 Backend-specific options; none are currently used. 

210 """ 

211 with cls.open(path) as archive: 

212 if archive._get_main_json_path() is None: 212 ↛ 213line 212 didn't jump to line 213 because the condition on line 212 was never true

213 raise ArchiveReadError( 

214 f"{path!r} has no LSST JSON tree; only the symmetric read path is supported." 

215 ) 

216 info = archive.info 

217 tree_cls = tree_class_for_info(info, path) 

218 parameterized = parameterize_tree(tree_cls, NdfPointerModel) 

219 tree = archive.get_tree(parameterized) 

220 yield archive, tree, info 

221 

222 @classmethod 

223 @contextmanager 

224 def open(cls, path: ResourcePathExpression | IO[bytes]) -> Iterator[Self]: 

225 """Open an NDF file for reading and yield an `NdfInputArchive`. 

226 

227 Remote ResourcePaths are materialised locally first; fsspec-direct 

228 h5py reads are a deferred follow-up. 

229 

230 Parameters 

231 ---------- 

232 path 

233 Path to the NDF file to open, or a seekable binary stream 

234 containing the file's content. 

235 """ 

236 if _is_binary_stream(path): 

237 with h5py.File(path, "r") as f: 

238 yield cls(f) 

239 return 

240 rp = ResourcePath(path) 

241 with rp.as_local() as local: 

242 with h5py.File(local.ospath, "r") as f: 

243 yield cls(f) 

244 

245 def get_tree[T: ArchiveTree](self, model_type: type[T]) -> T: 

246 """Read and validate the main Pydantic tree at ``/MORE/LSST/JSON``. 

247 

248 Parameters 

249 ---------- 

250 model_type 

251 Archive tree model type to validate the JSON tree against. 

252 """ 

253 json_path = self._get_main_json_path() 

254 if json_path is None: 

255 raise ArchiveReadError( 

256 "File has no /MORE/LSST/JSON tree; this is either a " 

257 "Starlink-only NDF (use ndf.read_starlink() for auto-detect) or " 

258 "the file was written by an unrelated tool." 

259 ) 

260 json_text = _read_json_record(self._get_primitive(json_path), json_path) 

261 return model_type.model_validate_json(json_text, context=_ARCHIVE_READ_CONTEXT) 

262 

263 def deserialize_pointer[U: ArchiveTree, V]( 

264 self, 

265 pointer: NdfPointerModel, 

266 model_type: type[U], 

267 deserializer: Callable[[U, InputArchive[NdfPointerModel]], V], 

268 ) -> V: 

269 # Cache by pointer.path so repeated dereferences reuse the same 

270 # deserialised result and don't re-run the deserializer. 

271 if (cached := self._deserialized_pointer_cache.get(pointer.path)) is not None: 

272 return cached 

273 if not self._has_model_path(pointer.path): 273 ↛ 274line 273 didn't jump to line 274 because the condition on line 273 was never true

274 raise ArchiveReadError(f"Pointer reference {pointer.path!r} not found in NDF file.") 

275 primitive = self._get_primitive(pointer.path) 

276 json_text = _read_json_record(primitive, pointer.path) 

277 model = model_type.model_validate_json(json_text, context=_ARCHIVE_READ_CONTEXT) 

278 result = deserializer(model, self) 

279 self._deserialized_pointer_cache[pointer.path] = result 

280 if isinstance(result, FrameSet): 

281 self._frame_set_cache[pointer.path] = result 

282 return result 

283 

284 def get_frame_set(self, pointer: NdfPointerModel) -> FrameSet: 

285 try: 

286 return self._frame_set_cache[pointer.path] 

287 except KeyError: 

288 raise AssertionError( 

289 f"Frame set at {pointer.path!r} must be deserialised via " 

290 f"deserialize_pointer before any dependent transform can be." 

291 ) from None 

292 

293 def get_array( 

294 self, 

295 model: ArrayReferenceModel | InlineArrayModel, 

296 *, 

297 slices: tuple[slice, ...] | EllipsisType = ..., 

298 strip_header: Callable[[astropy.io.fits.Header], None] = no_header_updates, 

299 ) -> np.ndarray: 

300 if isinstance(model, InlineArrayModel): 

301 data: np.ndarray = np.array(model.data, dtype=model.datatype.to_numpy()) 

302 return data if slices is ... else data[slices] 

303 if not isinstance(model.source, str) or not model.source.startswith("ndf:"): 

304 raise ArchiveReadError( 

305 f"NdfInputArchive cannot resolve array source {model.source!r}; " 

306 f"expected an 'ndf:<HDF5-path>' reference." 

307 ) 

308 path = model.source[len("ndf:") :] 

309 if not self._has_model_path(path): 309 ↛ 310line 309 didn't jump to line 310 because the condition on line 309 was never true

310 raise ArchiveReadError(f"Array reference {path!r} not in file.") 

311 primitive = self._get_primitive(path) 

312 # h5py supports lazy slicing via dataset[slices]. 

313 if isinstance(primitive.data, h5py.Dataset): 313 ↛ 315line 313 didn't jump to line 315 because the condition on line 313 was always true

314 return primitive.data[()] if slices is ... else primitive.data[slices] 

315 data = primitive.read_array() 

316 return data if slices is ... else data[slices] 

317 

318 def get_table( 

319 self, 

320 model: TableModel, 

321 strip_header: Callable[[astropy.io.fits.Header], None] = no_header_updates, 

322 ) -> astropy.table.Table: 

323 result = astropy.table.Table(meta=model.meta) 

324 for column_model in model.columns: 

325 if isinstance(column_model.data, InlineArrayModel): 325 ↛ 326line 325 didn't jump to line 326 because the condition on line 325 was never true

326 data: Any = column_model.data.data 

327 else: 

328 data = self.get_array(column_model.data, strip_header=strip_header) 

329 result[column_model.name] = astropy.table.Column( 

330 data, 

331 name=column_model.name, 

332 dtype=column_model.data.datatype.to_numpy(), 

333 unit=column_model.unit, 

334 description=column_model.description, 

335 meta=column_model.meta, 

336 ) 

337 return result 

338 

339 def get_structured_array( 

340 self, 

341 model: TableModel, 

342 strip_header: Callable[[astropy.io.fits.Header], None] = no_header_updates, 

343 ) -> np.ndarray: 

344 return self.get_table(model, strip_header).as_array() 

345 

346 def _read_opaque_fits_metadata(self) -> None: 

347 if not self._has_model_path("/MORE/FITS"): 

348 return 

349 cards = self._get_primitive("/MORE/FITS").read_char_array() 

350 # FITS Header.fromstring expects fixed-width 80-char cards 

351 # concatenated; pad each card defensively so readers tolerate 

352 # files written with shorter widths. 

353 header = astropy.io.fits.Header.fromstring("".join(c.ljust(80) for c in cards)) 

354 self._opaque_metadata.add_header(header, name="", ver=1) 

355 

356 def get_opaque_metadata(self) -> FitsOpaqueMetadata: 

357 return self._opaque_metadata 

358 

359 @property 

360 def info(self) -> ArchiveInfo: 

361 """Schema/format info read from the open document's ``DATA_MODEL`` 

362 (`.serialization.ArchiveInfo`). 

363 """ 

364 return _read_archive_info(self._get_optional_primitive, repr(self._file.filename)) 

365 

366 def _get_main_json_path(self) -> str | None: 

367 """Return the path of the main LSST JSON tree, if present.""" 

368 for prefix in _LSST_EXTENSION_PREFIXES: 

369 path = f"{prefix}/JSON" 

370 if self._has_model_path(path): 

371 return path 

372 return None 

373 

374 def _check_format_version(self) -> None: 

375 """Read FORMAT_VERSION from the NDF top-level structure and check it. 

376 

377 A file carrying an LSST JSON tree was written by this package and so 

378 must carry the layout stamp too; a missing stamp there is a damaged 

379 file, not a legacy one. A Starlink NDF has no JSON tree and no layout 

380 of ours to version, so there is nothing to check -- those are read by 

381 `read_starlink`, which opens the archive through this same class. 

382 

383 DATA_MODEL is informational only on read; the JSON tree's 

384 ``schema_version`` / ``min_read_version`` drive data-model 

385 compatibility. 

386 """ 

387 on_disk = _read_format_version(self._get_optional_primitive) 

388 if on_disk is None: 

389 if self._get_main_json_path() is not None: 

390 raise ArchiveReadError( 

391 f"{self._file.filename!r} has an LSST JSON tree but no container format version " 

392 "at /MORE/LSST/FORMAT_VERSION or /LSST/FORMAT_VERSION." 

393 ) 

394 return 

395 _check_format_version("ndf", on_disk, _NDF_FORMAT_VERSION) 

396 

397 def _has_model_path(self, path: str) -> bool: 

398 """Return `True` if a path exists in the NDF document model.""" 

399 try: 

400 self._document.get(path) 

401 except KeyError: 

402 return False 

403 return True 

404 

405 def _get_primitive(self, path: str) -> HdsPrimitive: 

406 """Return a primitive component from the NDF document model.""" 

407 node = self._document.get(path) 

408 if not isinstance(node, HdsPrimitive): 408 ↛ 409line 408 didn't jump to line 409 because the condition on line 408 was never true

409 raise ArchiveReadError(f"NDF reference {path!r} is not a primitive dataset.") 

410 return node 

411 

412 def _get_optional_primitive(self, path: str) -> HdsPrimitive | None: 

413 """Return a primitive from the document model, or `None` if there is 

414 no primitive at ``path``. 

415 """ 

416 try: 

417 node = self._document.get(path) 

418 except KeyError: 

419 return None 

420 return node if isinstance(node, HdsPrimitive) else None 

421 

422 

423def read_starlink[T: Any](cls_: type[T], path: ResourcePathExpression) -> T: 

424 """Reconstruct an `~lsst.images.Image` or `~lsst.images.MaskedImage` 

425 from a schema-less Starlink NDF. 

426 

427 Files written by this package carry a ``/MORE/LSST/JSON`` tree and are 

428 read through the generic `lsst.images.serialization.read_archive` / 

429 `lsst.images.serialization.open_archive`. A Starlink-produced NDF has 

430 no such tree and therefore no schema, so it cannot go through that 

431 path; this function auto-detects a minimal recognised-component set 

432 (``DATA_ARRAY``, ``VARIANCE``, ``QUALITY``, ``MORE.FITS``) instead. 

433 ``WCS`` is reconstructed when possible; other components are 

434 logged-and-dropped. 

435 

436 Parameters 

437 ---------- 

438 cls_ 

439 Expected return type; `~lsst.images.Image` and 

440 `~lsst.images.MaskedImage` are the only types the auto-detect path 

441 can produce. 

442 path 

443 File path or `lsst.resources.ResourcePathExpression`. 

444 

445 Returns 

446 ------- 

447 object 

448 The deserialized ``cls`` instance. 

449 

450 Raises 

451 ------ 

452 ArchiveReadError 

453 If the file has an LSST JSON tree (use the generic ``read_archive`` 

454 instead) or no recognised ``DATA_ARRAY`` component. 

455 """ 

456 with NdfInputArchive.open(path) as archive: 

457 if archive._get_main_json_path() is not None: 457 ↛ 458line 457 didn't jump to line 458 because the condition on line 457 was never true

458 raise ArchiveReadError( 

459 f"{path!r} has an LSST JSON tree; read it with serialization.read_archive()/open_archive()." 

460 ) 

461 return _read_auto_detect(cls_, archive) 

462 

463 

464def _read_auto_detect[T: Any](cls: type[T], archive: NdfInputArchive) -> T: 

465 """Reconstruct an `Image` (or `MaskedImage`) from a Starlink NDF. 

466 

467 Recognised components: ``DATA_ARRAY`` (in either simple or complex 

468 form), ``VARIANCE``, ``QUALITY``, ``MORE.FITS``. Other components 

469 (``WCS``, ``HISTORY``, ``AXIS``, ``LABEL``, custom ``MORE.*``, 

470 ``_LOGICAL`` primitives) are warned-and-dropped. 

471 """ 

472 f = archive._file 

473 ndf_group = _locate_ndf_root(f) 

474 

475 # DATA_ARRAY is required. 

476 if "DATA_ARRAY" not in ndf_group: 

477 raise ArchiveReadError(f"Auto-detect read of {f.filename!r}: no DATA_ARRAY component.") 

478 data_arr, bbox = _read_data_array_with_bbox(ndf_group["DATA_ARRAY"]) 

479 

480 # VARIANCE / QUALITY are optional. 

481 variance_arr: np.ndarray | None = None 

482 variance_bbox: Any | None = None 

483 if "VARIANCE" in ndf_group: 

484 variance_arr, variance_bbox = _read_data_array_with_bbox(ndf_group["VARIANCE"]) 

485 quality_arr: np.ndarray | None = None 

486 quality_bbox: Any | None = None 

487 quality_badbits = 255 

488 if "QUALITY" in ndf_group and isinstance(ndf_group["QUALITY"], h5py.Group): 

489 q = ndf_group["QUALITY"] 

490 quality_badbits = _read_quality_badbits(q) 

491 if "QUALITY" in q and isinstance(q["QUALITY"], h5py.Dataset): 491 ↛ 492line 491 didn't jump to line 492 because the condition on line 491 was never true

492 quality_arr = _validate_quality_array(_hds.read_array(q["QUALITY"])) 

493 quality_bbox = _make_bbox(x_min=0, y_min=0, array=quality_arr) 

494 elif "QUALITY" in q and isinstance(q["QUALITY"], h5py.Group): 494 ↛ 498line 494 didn't jump to line 498 because the condition on line 494 was always true

495 quality_arr, quality_bbox = _read_data_array_with_bbox(q["QUALITY"]) 

496 quality_arr = _validate_quality_array(quality_arr) 

497 

498 sky_projection: SkyProjection | None = None 

499 if "WCS" in ndf_group: 

500 try: 

501 wcs_group = ndf_group["WCS"] 

502 if isinstance(wcs_group, h5py.Group) and "DATA" in wcs_group: 502 ↛ 520line 502 didn't jump to line 520 because the condition on line 502 was always true

503 wcs_lines = _hds.read_char_array(wcs_group["DATA"]) 

504 wcs_text = _hds.decode_ndf_ast_data(wcs_lines) 

505 ast_obj = astshim.Object.fromString(wcs_text) 

506 if isinstance(ast_obj, astshim.FrameSet): 506 ↛ 520line 506 didn't jump to line 520 because the condition on line 506 was always true

507 pixel_frame = GeneralFrame(unit=u.pix) 

508 sky_projection = SkyProjection.from_ast_frame_set( 

509 ast_obj, 

510 pixel_frame, 

511 pixel_bounds=bbox, 

512 ) 

513 except Exception: 

514 _LOG.warning( 

515 "Could not reconstruct Projection from WCS in %s; dropping.", 

516 f.filename, 

517 exc_info=True, 

518 ) 

519 

520 unit = _read_ndf_units(ndf_group) 

521 

522 # Anything unrecognised: warn-and-drop. 

523 recognised = { 

524 "DATA_ARRAY", 

525 "VARIANCE", 

526 "QUALITY", 

527 "WCS", 

528 "MORE", 

529 "TITLE", 

530 "LABEL", 

531 "UNITS", 

532 "HISTORY", 

533 "AXIS", 

534 } 

535 for name in ndf_group: 

536 if name not in recognised: 536 ↛ 537line 536 didn't jump to line 537 because the condition on line 536 was never true

537 _LOG.warning( 

538 "Ignoring unrecognised NDF component %s/%s during auto-detect read.", 

539 ndf_group.name, 

540 name, 

541 ) 

542 

543 # Build the requested in-memory object. Any NDF can be read as an Image; 

544 # MaskedImage construction uses whatever VARIANCE/QUALITY are present and 

545 # lets the MaskedImage constructor provide defaults for missing planes. 

546 image = Image(data_arr, bbox=bbox, unit=unit, sky_projection=sky_projection) 

547 obj: Any 

548 if cls is Image: 

549 obj = image 

550 elif issubclass(cls, MaskedImage): 

551 if quality_arr is not None: 

552 schema = _make_quality_mask_schema(quality_badbits) 

553 mask = Mask(quality_arr[:, :, np.newaxis], schema=schema, bbox=quality_bbox) 

554 else: 

555 schema = MaskSchema([MaskPlane(name="BAD", description="Bad pixel.")]) 

556 mask = None 

557 variance = Image(variance_arr, bbox=variance_bbox) if variance_arr is not None else None 

558 obj = cls( 

559 image=image, 

560 mask=mask, 

561 mask_schema=schema if mask is None else None, 

562 variance=variance, 

563 ) 

564 else: 

565 raise ArchiveReadError( 

566 f"Auto-detect can produce Image or MaskedImage, but caller asked for {cls.__name__}." 

567 ) 

568 obj._opaque_metadata = archive.get_opaque_metadata() 

569 return obj 

570 

571 

572def _read_ndf_units(ndf_group: h5py.Group) -> u.UnitBase | None: 

573 """Read the NDF UNITS component, if present.""" 

574 if "UNITS" not in ndf_group or not isinstance(ndf_group["UNITS"], h5py.Dataset): 

575 return None 

576 dataset = ndf_group["UNITS"] 

577 if dataset.dtype.kind != "S": 577 ↛ 578line 577 didn't jump to line 578 because the condition on line 577 was never true

578 _LOG.warning("Ignoring non-character NDF UNITS component in %s.", ndf_group.name) 

579 return None 

580 if dataset.ndim == 0: 580 ↛ 588line 580 didn't jump to line 588 because the condition on line 580 was always true

581 raw = dataset[()] 

582 if isinstance(raw, np.bytes_): 582 ↛ 584line 582 didn't jump to line 584 because the condition on line 582 was always true

583 raw = bytes(raw) 

584 if not isinstance(raw, bytes): 584 ↛ 585line 584 didn't jump to line 585 because the condition on line 584 was never true

585 return None 

586 units_text = raw.decode("ascii").rstrip(" ") 

587 else: 

588 records = _hds.read_char_array(dataset) 

589 units_text = records[0] if records else "" 

590 if not units_text: 590 ↛ 591line 590 didn't jump to line 591 because the condition on line 590 was never true

591 return None 

592 for kwargs in ({"format": "fits"}, {}): 592 ↛ 597line 592 didn't jump to line 597 because the loop on line 592 didn't complete

593 try: 

594 return u.Unit(units_text, **kwargs) 

595 except ValueError: 

596 continue 

597 _LOG.warning("Could not parse NDF UNITS value %r in %s.", units_text, ndf_group.name) 

598 return None 

599 

600 

601def _read_quality_badbits(quality_group: h5py.Group) -> int: 

602 """Read the scalar NDF QUALITY.BADBITS value.""" 

603 badbits = quality_group.get("BADBITS") 

604 if not isinstance(badbits, h5py.Dataset): 604 ↛ 605line 604 didn't jump to line 605 because the condition on line 604 was never true

605 return 255 

606 value = np.asarray(_hds.read_array(badbits)).reshape(-1) 

607 if value.size == 0: 607 ↛ 608line 607 didn't jump to line 608 because the condition on line 607 was never true

608 return 255 

609 return int(value[0]) 

610 

611 

612def _validate_quality_array(quality: np.ndarray) -> np.ndarray: 

613 """Return an NDF QUALITY array as a `numpy.uint8` mask plane.""" 

614 if quality.dtype != np.dtype(np.uint8): 614 ↛ 615line 614 didn't jump to line 615 because the condition on line 614 was never true

615 raise ArchiveReadError(f"NDF QUALITY array has dtype {quality.dtype}; expected uint8.") 

616 return quality 

617 

618 

619def _make_quality_mask_schema(badbits: int) -> MaskSchema: 

620 """Create a fallback `MaskSchema` for an unnamed 8-bit QUALITY array.""" 

621 planes = [] 

622 for bit in range(8): 

623 mask = 1 << bit 

624 description = f"NDF QUALITY bit {bit}." 

625 if badbits & mask: 

626 description += " Selected by BADBITS." 

627 planes.append(MaskPlane(name=f"MASK{bit}", description=description)) 

628 return MaskSchema(planes, dtype=np.uint8) 

629 

630 

631def _locate_ndf_root(f: h5py.File) -> h5py.Group: 

632 """Return the group representing the top-level NDF. 

633 

634 Most files have the NDF at the root group itself. A few wrap it 

635 in a single-child container at the root; we accept that shape 

636 too. Anything more elaborate raises. 

637 """ 

638 root_class = f["/"].attrs.get(_hds.ATTR_CLASS) 

639 if isinstance(root_class, bytes): 

640 root_class = root_class.decode("ascii") 

641 if root_class == "NDF": 641 ↛ 644line 641 didn't jump to line 644 because the condition on line 641 was always true

642 return f["/"] 

643 # Maybe a one-level container. 

644 candidates = [] 

645 for name, child in f["/"].items(): 

646 if isinstance(child, h5py.Group): 

647 cls_attr = child.attrs.get(_hds.ATTR_CLASS) 

648 if isinstance(cls_attr, bytes): 

649 cls_attr = cls_attr.decode("ascii") 

650 if cls_attr == "NDF": 

651 candidates.append(name) 

652 if len(candidates) == 1: 

653 return f[candidates[0]] 

654 raise ArchiveReadError( 

655 f"Could not locate top-level NDF in {f.filename!r}; " 

656 f"expected the root group or a single NDF-typed child." 

657 ) 

658 

659 

660def _read_data_array_with_bbox( 

661 obj: h5py.Group | h5py.Dataset, 

662) -> tuple[np.ndarray, Any]: 

663 """Read a DATA_ARRAY component in either simple or complex form. 

664 

665 The complex form (what our writer always produces) is an HDS 

666 ARRAY structure (h5py group with CLASS="ARRAY") containing 

667 ``DATA`` and ``ORIGIN`` primitives. The simple form is a bare 

668 primitive dataset. 

669 

670 Returns 

671 ------- 

672 array, bbox : tuple 

673 ``array`` is the C-order numpy data (shape ``(height, width)`` 

674 for 2D images). ``bbox`` is constructed from the ORIGIN if 

675 present, else from a default origin of (0, 0). 

676 """ 

677 if isinstance(obj, h5py.Dataset): 677 ↛ 679line 677 didn't jump to line 679 because the condition on line 677 was never true

678 # Simple form. 

679 array = _hds.read_array(obj) 

680 bbox = _make_bbox(x_min=0, y_min=0, array=array) 

681 return array, bbox 

682 # Complex form: an HDS structure with DATA + ORIGIN. 

683 data = _hds.read_array(obj["DATA"]) 

684 if "ORIGIN" in obj: 

685 origin = _hds.read_array(obj["ORIGIN"]) 

686 bbox = _make_bbox(x_min=int(origin[0]), y_min=int(origin[1]), array=data) 

687 else: 

688 bbox = _make_bbox(x_min=0, y_min=0, array=data) 

689 return data, bbox 

690 

691 

692def _read_json_record(primitive: HdsPrimitive, path: str) -> str: 

693 """Read a JSON document stored as a single _CHAR*N record. 

694 

695 Our writer always emits JSON trees as a single-element character 

696 array sized to the document. Joining multiple records would lose 

697 trailing whitespace inside JSON string values, since 

698 `read_char_array` strips trailing spaces per record. 

699 """ 

700 records = primitive.read_char_array() 

701 if len(records) != 1: 701 ↛ 702line 701 didn't jump to line 702 because the condition on line 701 was never true

702 raise ArchiveReadError(f"Expected a single _CHAR*N record at {path!r}, got {len(records)}.") 

703 return records[0] 

704 

705 

706def _make_bbox(*, x_min: int, y_min: int, array: np.ndarray) -> Any: 

707 """Build an lsst.images.Box for a 2D image array. 

708 

709 The array is C-order ``(height, width)``. NDF stores ``ORIGIN`` 

710 in Fortran axis order ``(x_min, y_min)``. 

711 """ 

712 if array.ndim != 2: 712 ↛ 713line 712 didn't jump to line 713 because the condition on line 712 was never true

713 raise ArchiveReadError(f"Auto-detect read only supports 2D arrays, got ndim={array.ndim}.") 

714 # Box.from_shape takes (height, width) and start=(y_start, x_start). 

715 return Box.from_shape(array.shape, start=(y_min, x_min))