Coverage for python/lsst/images/serialization/_common.py: 61%

170 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-09-30 11:45 +0000

1# This file is part of lsst-images. 

2# 

3# Developed for the LSST Data Management System. 

4# This product includes software developed by the LSST Project 

5# (https://www.lsst.org). 

6# See the COPYRIGHT file at the top-level directory of this distribution 

7# for details of code ownership. 

8# 

9# Use of this source code is governed by a 3-clause BSD-style 

10# license that can be found in the LICENSE file. 

11 

12from __future__ import annotations 

13 

14__all__ = ( 

15 "SCHEMA_URL_BASE", 

16 "ArchiveAccessRequiredError", 

17 "ArchiveReadError", 

18 "ArchiveTree", 

19 "ButlerInfo", 

20 "DevelopmentSchemaWarning", 

21 "InvalidComponentError", 

22 "InvalidParameterError", 

23 "JsonRef", 

24 "MetadataValue", 

25 "OpaqueArchiveMetadata", 

26 "is_development_version", 

27 "no_header_updates", 

28 "warn_for_development_schemas", 

29) 

30 

31import operator 

32import warnings 

33from abc import ABC, abstractmethod 

34from collections.abc import Iterator 

35from copy import deepcopy 

36from typing import TYPE_CHECKING, Any, ClassVar, Protocol, Self 

37 

38import astropy.table 

39import astropy.units 

40import pydantic 

41from packaging.version import Version 

42 

43from .._geom import Box 

44from ..utils import is_none 

45from ._external_metadata import ExternalMetadata 

46from ._migrations import _MIGRATABLE_NAMES, _MIGRATIONS 

47 

48try: 

49 from lsst.daf.butler import DatasetProvenance, SerializedDatasetRef 

50except ImportError: 

51 type DatasetProvenance = Any # type: ignore[no-redef] 

52 type SerializedDatasetRef = Any # type: ignore[no-redef] 

53 

54if TYPE_CHECKING: 

55 import astropy.io.fits 

56 

57 from ._input_archive import InputArchive 

58 

59 

60type MetadataValue = ( 

61 pydantic.StrictInt | pydantic.StrictFloat | pydantic.StrictStr | pydantic.StrictBool | None 

62) 

63 

64SCHEMA_URL_BASE = "https://images.lsst.io/schemas" 

65"""Base for the schema URLs of this package's own schemas, used as 

66``{SCHEMA_URL_BASE}/{name}-{version}``. 

67 

68External packages providing their own schemas override 

69`~lsst.images.serialization.ArchiveTree.SCHEMA_URL_BASE` instead, so their 

70schema URLs are minted under a site they control. 

71""" 

72 

73_ARCHIVE_READ_CONTEXT = object() 

74"""Pydantic context used when validating a tree read from an archive.""" 

75 

76 

77class ButlerInfo(pydantic.BaseModel): 

78 """Information about a butler dataset.""" 

79 

80 dataset: SerializedDatasetRef 

81 provenance: DatasetProvenance = pydantic.Field(default_factory=DatasetProvenance) 

82 

83 

84class JsonRef(pydantic.BaseModel, serialize_by_alias=True): 

85 """Pydantic model for JSON Reference / Pointer (IETF RFC 6901). 

86 

87 Notes 

88 ----- 

89 This model does not do any of the escaping or special-character 

90 interpretation required by the spec; it assumes that's already been done, 

91 so its job is *just* putting a ``$ref`` field inside another model. 

92 """ 

93 

94 ref: str = pydantic.Field(alias="$ref") 

95 

96 

97class ArchiveTree( 

98 pydantic.BaseModel, ABC, ser_json_inf_nan="constants", ser_json_bytes="base64", val_json_bytes="base64" 

99): 

100 """An intermediate base class of `pydantic.BaseModel` that should be used 

101 for all objects that may be used as the top-level tree models written to 

102 archives. 

103 

104 See :ref:`lsst.images-schema-versioning` for how the ``SCHEMA_NAME`` / 

105 ``SCHEMA_VERSION`` / ``MIN_READ_VERSION`` constants and the 

106 ``schema_version`` / ``min_read_version`` / ``schema_url`` fields are used. 

107 """ 

108 

109 SCHEMA_NAME: ClassVar[str] 

110 SCHEMA_VERSION: ClassVar[str] 

111 MIN_READ_VERSION: ClassVar[int] 

112 SCHEMA_URL_BASE: ClassVar[str] = SCHEMA_URL_BASE 

113 """Base for this schema's URL, as ``{SCHEMA_URL_BASE}/{name}-{version}``. 

114 

115 External packages providing their own schemas should override this (once, 

116 on a shared intermediate base class) so their schema URLs are minted 

117 under a documentation site they control rather than images.lsst.io.""" 

118 

119 PUBLIC_TYPE: ClassVar[type] 

120 """In-memory Python type produced by this tree's ``deserialize`` (e.g. 

121 `dict` for a mapping return). Declared explicitly by each concrete 

122 subclass and surfaced by 

123 `~lsst.images.serialization.public_type_for_schema`.""" 

124 

125 # These mirror the SCHEMA_VERSION / MIN_READ_VERSION class constants and 

126 # are never passed on construction, so repr omits them; they say nothing 

127 # about the object that its type does not already say. 

128 schema_version: str = pydantic.Field( 

129 default="1.0.0", 

130 description="Data-model schema version of this tree (major.minor.patch).", 

131 repr=False, 

132 ) 

133 min_read_version: int = pydantic.Field( 

134 default=1, 

135 description="Smallest reader major that can interpret this tree.", 

136 repr=False, 

137 ) 

138 metadata: dict[str, MetadataValue] = pydantic.Field( 

139 default_factory=dict, description="Additional unstructured metadata.", exclude_if=operator.not_ 

140 ) 

141 butler_info: ButlerInfo | None = pydantic.Field( 

142 default=None, 

143 description="Information about the butler dataset backed by this file.", 

144 exclude_if=is_none, 

145 ) 

146 # Populated by the archive machinery rather than by a caller, so repr 

147 # omits it as well. 

148 indirect: list[Any] = pydantic.Field( 

149 default_factory=list, 

150 description="Serialized nested objects that may be saved or read more than once.", 

151 exclude_if=operator.not_, 

152 repr=False, 

153 ) 

154 

155 @pydantic.computed_field(description="Canonical schema URL for this tree.") # type: ignore[prop-decorator] 

156 @property 

157 def schema_url(self) -> str: 

158 """Return the schema URL of this tree's class. 

159 

160 Computed from ``SCHEMA_NAME`` and ``SCHEMA_VERSION`` ClassVars. 

161 """ 

162 cls = type(self) 

163 return f"{cls.SCHEMA_URL_BASE}/{cls.SCHEMA_NAME}-{cls.SCHEMA_VERSION}" 

164 

165 @pydantic.model_validator(mode="before") 

166 @classmethod 

167 def _migrate_from_older_major(cls, data: Any, info: pydantic.ValidationInfo) -> Any: 

168 """Morph an older on-disk tree into the current shape. 

169 

170 Chains registered migrations until the tree reaches the in-code major, 

171 so the ``mode="after"`` validator below sees a current-shape tree. 

172 

173 A tree validated in the archive-read context, which the backends mark 

174 via ``info.context``, must carry a ``schema_version``; in-memory 

175 construction is unmarked and keeps the class-constant defaults. See 

176 :ref:`lsst.images-schema-versioning-migration`. 

177 """ 

178 if not isinstance(data, dict): 

179 return data 

180 if not hasattr(cls, "SCHEMA_NAME"): 180 ↛ 185line 180 didn't jump to line 185 because the condition on line 180 was never true

181 # ArchiveTree itself is abstract, and a subclass that has not yet 

182 # declared SCHEMA_NAME (during incremental rollout) has no 

183 # SCHEMA_VERSION to parse either; skip cleanly rather than 

184 # raising AttributeError, mirroring the after-validator below. 

185 return data 

186 name = cls.SCHEMA_NAME 

187 if "schema_version" not in data: 

188 if info.context is _ARCHIVE_READ_CONTEXT: 

189 raise _MissingSchemaVersionError( 

190 f"{name}: archive tree has no schema_version; unstamped archive data is not supported." 

191 ) 

192 return data 

193 if not _MIGRATABLE_NAMES: 193 ↛ 194line 193 didn't jump to line 194 because the condition on line 193 was never true

194 return data 

195 if name not in _MIGRATABLE_NAMES: 

196 # This schema has not opted into migrations at all, regardless of 

197 # its major: whether some *other*, unrelated schema has a 

198 # registered migration must not change what this one does with 

199 # its own tree. A schema in this state falls through to ordinary 

200 # validation -- in-model backfill for an additive change, or a 

201 # Pydantic validation error for a real incompatibility. 

202 return data 

203 on_disk_major = _parse_on_disk_major(data["schema_version"]) 

204 in_code_major = _parse_major(cls.SCHEMA_VERSION) 

205 if on_disk_major < in_code_major: 

206 # A failed union candidate must not mutate the input subsequently 

207 # tried by another candidate. Migrations may mutate freely, so 

208 # isolate the complete JSON-like tree, not just its top-level 

209 # dictionary. 

210 data = deepcopy(data) 

211 while on_disk_major < in_code_major: 

212 try: 

213 step = _MIGRATIONS[(name, on_disk_major)] 

214 except KeyError: 

215 raise _MigrationGapError( 

216 f"{name}: no migration from major {on_disk_major} to {on_disk_major + 1}." 

217 ) from None 

218 data = step(data) 

219 on_disk_major += 1 

220 data["schema_version"] = f"{on_disk_major}.0.0" 

221 return data 

222 

223 @pydantic.model_validator(mode="after") 

224 def _check_and_normalize_schema_version(self) -> Self: 

225 """Validate and normalise the schema version fields. 

226 

227 Compares the on-tree ``schema_version`` / ``min_read_version`` against 

228 the in-code values from the subclass's ClassVars; raises if 

229 incompatible, otherwise normalises the fields to the in-code values. 

230 """ 

231 cls = type(self) 

232 # ArchiveTree itself is abstract (deserialize is @abstractmethod). 

233 # Subclasses that haven't yet declared SCHEMA_NAME are skipped — this 

234 # matters during incremental rollout and remains a safe no-op 

235 # afterwards (a class-invariants test ensures every concrete subclass 

236 # has the constants). 

237 if not hasattr(cls, "SCHEMA_NAME"): 237 ↛ 238line 237 didn't jump to line 238 because the condition on line 237 was never true

238 return self 

239 _check_compat( 

240 cls.SCHEMA_NAME, 

241 self.schema_version, 

242 self.min_read_version, 

243 cls.SCHEMA_VERSION, 

244 ) 

245 if self.schema_version != cls.SCHEMA_VERSION: 

246 self.schema_version = cls.SCHEMA_VERSION 

247 if self.min_read_version != cls.MIN_READ_VERSION: 

248 self.min_read_version = cls.MIN_READ_VERSION 

249 return self 

250 

251 @classmethod 

252 def __pydantic_init_subclass__(cls, **kwargs: Any) -> None: 

253 """Inject ``$id`` and ``title`` into the subclass's JSON Schema, and 

254 register the subclass in the schema-name registry. 

255 

256 Populates ``model_config['json_schema_extra']`` with values derived 

257 from the subclass's ``SCHEMA_NAME`` / ``SCHEMA_VERSION`` ClassVars, 

258 then registers the subclass so it can be looked up by schema name. 

259 Subclasses that haven't declared the ClassVars are skipped. 

260 """ 

261 super().__pydantic_init_subclass__(**kwargs) 

262 name = cls.__dict__.get("SCHEMA_NAME") 

263 version = cls.__dict__.get("SCHEMA_VERSION") 

264 if name is None or version is None: 

265 return 

266 json_schema_extra = cls.model_config.get("json_schema_extra") or {} 

267 if isinstance(json_schema_extra, dict): 267 ↛ 279line 267 didn't jump to line 279 because the condition on line 267 was always true

268 existing = dict(json_schema_extra) 

269 # Always override: a subclass of a concrete schema (e.g. 

270 # visit_image subclassing masked_image) inherits its parent's 

271 # already-stamped values through the merged model_config, and 

272 # this hook only runs when the subclass declares its own 

273 # SCHEMA_NAME / SCHEMA_VERSION for these to be derived from. 

274 existing["$id"] = f"{cls.SCHEMA_URL_BASE}/{name}-{version}" 

275 existing["title"] = name 

276 cls.model_config = {**cls.model_config, "json_schema_extra": existing} 

277 # Local import to avoid the _io -> _common circular dependency at 

278 # module load time. 

279 from ._io import register_schema_class 

280 

281 register_schema_class(cls) 

282 

283 @abstractmethod 

284 def deserialize(self, archive: InputArchive[Any], **kwargs: Any) -> Any: 

285 """Return the in-memory object that was serialized to this tree. 

286 

287 Parameters 

288 ---------- 

289 archive 

290 The input archive to read from. 

291 **kwargs 

292 Additional keyword arguments specific to this type. 

293 

294 Raises 

295 ------ 

296 ~lsst.images.serialization.InvalidParameterError 

297 Raised for unsupported ``**kwargs``. 

298 

299 Notes 

300 ----- 

301 Subclass implementations may take additional keyword-only arguments. 

302 Callers that invoke this method without knowing what those might be 

303 should catch `TypeError` and re-raise as 

304 `~lsst.images.serialization.InvalidParameterError` if they pass 

305 additional keyword arguments. 

306 """ 

307 raise NotImplementedError() 

308 

309 def deserialize_component(self, component: str, archive: InputArchive[Any], **kwargs: Any) -> Any: 

310 """Return a component in-memory object that was serialized to this 

311 tree. 

312 

313 Parameters 

314 ---------- 

315 component 

316 Name of the component to read. 

317 archive 

318 The input archive to read from. 

319 **kwargs 

320 Additional keyword arguments specific to this type. 

321 

322 Raises 

323 ------ 

324 ~lsst.images.serialization.InvalidComponentError 

325 Raise if ``component`` is not recognized. 

326 ~lsst.images.serialization.InvalidParameterError 

327 Raised for unsupported ``**kwargs``. 

328 

329 Notes 

330 ----- 

331 The default implementation for this method tries to get an attribute 

332 with the component's name from ``self``, and then: 

333 

334 - returns `None` if it is `None`; 

335 - calls `deserialize` on that object if it is also an 

336 `~lsst.images.serialization.ArchiveTree`; 

337 - returns it directly otherwise. 

338 

339 If there is no such attribute, it raises 

340 `~lsst.images.serialization.InvalidComponentError`. 

341 

342 ``**kwargs`` are forwarded to component `deserialize` methods, but 

343 are otherwise not checked. Subclasses are generally expected to 

344 implement this method to do that checking and handle any components 

345 for which the other will not work, and then delegate to `super` at 

346 the end. 

347 """ 

348 try: 

349 component_model = getattr(self, component) 

350 except AttributeError: 

351 raise InvalidComponentError( 

352 f"Component {component!r} is not recognized by {type(self).__name__}." 

353 ) from None 

354 if component_model is None: 

355 return None 

356 if isinstance(component_model, ArchiveTree): 

357 return component_model.deserialize(archive, **kwargs) 

358 return component_model 

359 

360 

361class DevelopmentSchemaWarning(UserWarning): 

362 """Warning that a file is being written with a development schema.""" 

363 

364 

365def is_development_version(version: str) -> bool: 

366 """Return whether a schema version string is a PEP 440 development release. 

367 

368 Parameters 

369 ---------- 

370 version 

371 Schema version string, e.g. ``1.0.0`` or ``1.0.0.dev0``. 

372 """ 

373 return Version(version).is_devrelease 

374 

375 

376def _iter_archive_trees(obj: Any) -> Iterator[ArchiveTree]: 

377 """Yield every `ArchiveTree` embedded in a serialized tree.""" 

378 if isinstance(obj, ArchiveTree): 

379 yield obj 

380 if isinstance(obj, pydantic.BaseModel): 

381 for field_name in type(obj).model_fields: 

382 yield from _iter_archive_trees(getattr(obj, field_name)) 

383 elif isinstance(obj, list | tuple): 

384 for item in obj: 

385 yield from _iter_archive_trees(item) 

386 elif isinstance(obj, dict): 

387 for value in obj.values(): 

388 yield from _iter_archive_trees(value) 

389 

390 

391def warn_for_development_schemas(root: ArchiveTree) -> None: 

392 """Emit a `DevelopmentSchemaWarning` if a serialized tree contains any 

393 schema still in development. 

394 

395 Parameters 

396 ---------- 

397 root 

398 Top-level serialized tree about to be written. 

399 """ 

400 developing = sorted( 

401 { 

402 tree.schema_url 

403 for tree in _iter_archive_trees(root) 

404 if is_development_version(type(tree).SCHEMA_VERSION) 

405 } 

406 ) 

407 if developing: 

408 warnings.warn( 

409 "Writing a file with development schema(s) " 

410 f"{', '.join(developing)}; such files are not for production and " 

411 "may not remain readable.", 

412 DevelopmentSchemaWarning, 

413 stacklevel=3, 

414 ) 

415 

416 

417class ArchiveReadError(RuntimeError): 

418 """Exception raised when the contents of an archive cannot be read.""" 

419 

420 

421class InvalidParameterError(ArchiveReadError): 

422 """Exception raised by `ArchiveTree.deserialize` or 

423 `ArchiveTree.deserialize_component` when passed an invalid keyword 

424 argument. 

425 """ 

426 

427 

428class InvalidComponentError(ArchiveReadError): 

429 """Exception `ArchiveTree.deserialize_component` when passed an invalid 

430 component name. 

431 """ 

432 

433 

434class _MigrationGapError(ArchiveReadError, ValueError): 

435 """A migration chain has no registered step for the tree's on-disk major. 

436 

437 Raised from the ``mode="before"`` validator in 

438 `ArchiveTree._migrate_from_older_major`, which pydantic-core invokes as 

439 part of validating a field. Deriving only from `ArchiveReadError` (a 

440 `RuntimeError`) would let this exception escape a discriminated union 

441 entirely: pydantic only treats `ValueError`, `TypeError` and 

442 `AssertionError` raised by a validator as "this candidate failed, try the 

443 next one", so a `RuntimeError` would abort the whole union instead of 

444 falling through. Deriving from both means a ``left_to_right`` (or 

445 ``smart``) union tries the next variant on a migration gap exactly as it 

446 would on any other validation failure, while ``except ArchiveReadError`` 

447 callers outside a union still catch it unchanged. 

448 """ 

449 

450 

451class _MissingSchemaVersionError(ArchiveReadError, ValueError): 

452 """An archive tree omitted its required ``schema_version`` stamp.""" 

453 

454 

455class _MalformedSchemaVersionError(ArchiveReadError, ValueError): 

456 """An archive tree's ``schema_version`` stamp could not be parsed. 

457 

458 Also a `ValueError`, for the reason `_MigrationGapError` gives: the stamp 

459 is parsed in the ``mode="before"`` validator, which runs ahead of field 

460 validation for every migratable union candidate, including ones the 

461 payload does not belong to. A stamp the reader cannot parse therefore has 

462 to fail that one candidate rather than abort the whole union. 

463 

464 Only an *on-disk* stamp raises this. A malformed in-code 

465 ``SCHEMA_VERSION`` is a programming error and keeps raising the 

466 `RuntimeError`-only `ArchiveReadError`, so a union cannot swallow it. 

467 """ 

468 

469 

470class ArchiveAccessRequiredError(RuntimeError): 

471 """Exception raised when a deserialization needs data from the file. 

472 

473 Raised by all data-access methods of 

474 `~lsst.images.serialization.DetachedArchive`, signaling that the 

475 requested object cannot be deserialized from the JSON tree alone. 

476 

477 Notes 

478 ----- 

479 This deliberately does not inherit from `ArchiveReadError`: it signals 

480 that a live archive is required rather than that an archive is corrupt, 

481 and it must not be swallowed by ``except ArchiveReadError`` handlers. 

482 """ 

483 

484 

485class OpaqueArchiveMetadata(Protocol): 

486 """Interface for opaque archive metadata. 

487 

488 In addition to implementing the methods defined here, all implementations 

489 must be pickleable. 

490 """ 

491 

492 def copy(self) -> Self | None: 

493 """Copy, reference, or discard metadata when its holding object is 

494 copied. 

495 """ 

496 ... 

497 

498 def subset(self, bbox: Box) -> Self | None: 

499 """Copy, reference, or discard metadata when a subset of its its 

500 holding object is extracted. 

501 

502 Parameters 

503 ---------- 

504 bbox 

505 Bounding box of the subset being extracted. 

506 """ 

507 ... 

508 

509 def external_metadata(self) -> ExternalMetadata: 

510 """Return a read-only view of the metadata carried by this object 

511 that is not part of the data model. 

512 """ 

513 ... 

514 

515 

516def no_header_updates(header: astropy.io.fits.Header) -> None: 

517 """Do not make any modifications to the given FITS header. 

518 

519 Parameters 

520 ---------- 

521 header 

522 FITS header that is left unchanged. 

523 """ 

524 

525 

526def _parse_major(version: object) -> int: 

527 """Return the integer major component of a major.minor.patch string. 

528 

529 Accepts any object, because callers pass values that may have come 

530 straight from a file; use `_parse_on_disk_major` for those. 

531 

532 Raises 

533 ------ 

534 ArchiveReadError 

535 If ``version`` is not a non-empty string of the form 

536 ``major.minor.patch`` with integer components. 

537 """ 

538 if not isinstance(version, str) or not version: 

539 raise ArchiveReadError(f"Schema version {version!r} is not a non-empty string.") 

540 head = version.split(".", 1)[0] 

541 try: 

542 return int(head) 

543 except ValueError as exc: 

544 raise ArchiveReadError(f"Schema version {version!r} has non-integer major.") from exc 

545 

546 

547def _parse_on_disk_major(version: object) -> int: 

548 """Return the integer major of a version stamp that came from a file. 

549 

550 Raises 

551 ------ 

552 _MalformedSchemaVersionError 

553 If ``version`` is not a non-empty string of the form 

554 ``major.minor.patch`` with integer components. Unlike `_parse_major`, 

555 this is a `ValueError`, so a stamp the reader cannot parse fails one 

556 union candidate instead of aborting the union. 

557 """ 

558 try: 

559 return _parse_major(version) 

560 except ArchiveReadError as exc: 

561 raise _MalformedSchemaVersionError(str(exc)) from None 

562 

563 

564def _check_compat( 

565 name: str, 

566 on_disk_version: str, 

567 on_disk_min_read: int, 

568 in_code_version: str, 

569) -> None: 

570 """Raise `ArchiveReadError` if a tree written with the given 

571 schema_version/min_read_version cannot be read by the current code. 

572 

573 See :ref:`lsst.images-schema-versioning` for the compatibility rule. 

574 """ 

575 in_code_major = _parse_major(in_code_version) 

576 if on_disk_min_read > in_code_major: 

577 raise ArchiveReadError( 

578 f"{name}: tree requires reader major >= {on_disk_min_read}; this release is {in_code_version}." 

579 ) 

580 

581 

582def _check_format_version(name: str, on_disk: int, in_code: int) -> None: 

583 """Raise `ArchiveReadError` if a backend file's container layout 

584 version is newer than this release knows how to read. 

585 """ 

586 if on_disk > in_code: 

587 raise ArchiveReadError( 

588 f"{name}: on-disk container format version {on_disk} is " 

589 f"newer than this release ({in_code}); cannot read." 

590 )