Coverage for python/lsst/images/serialization/_common.py: 62%

167 statements  

« prev     ^ index     » next       coverage.py v7.15.4, created at 2026-08-29 02:31 -0700

1# This file is part of lsst-images. 

2# 

3# Developed for the LSST Data Management System. 

4# This product includes software developed by the LSST Project 

5# (https://www.lsst.org). 

6# See the COPYRIGHT file at the top-level directory of this distribution 

7# for details of code ownership. 

8# 

9# Use of this source code is governed by a 3-clause BSD-style 

10# license that can be found in the LICENSE file. 

11 

12from __future__ import annotations 

13 

14__all__ = ( 

15 "SCHEMA_URL_BASE", 

16 "ArchiveAccessRequiredError", 

17 "ArchiveReadError", 

18 "ArchiveTree", 

19 "ButlerInfo", 

20 "DevelopmentSchemaWarning", 

21 "InvalidComponentError", 

22 "InvalidParameterError", 

23 "JsonRef", 

24 "MetadataValue", 

25 "OpaqueArchiveMetadata", 

26 "is_development_version", 

27 "no_header_updates", 

28 "warn_for_development_schemas", 

29) 

30 

31import operator 

32import warnings 

33from abc import ABC, abstractmethod 

34from collections.abc import Iterator 

35from copy import deepcopy 

36from typing import TYPE_CHECKING, Any, ClassVar, Protocol, Self 

37 

38import astropy.table 

39import astropy.units 

40import pydantic 

41from packaging.version import Version 

42 

43from .._geom import Box 

44from ..utils import is_none 

45from ._migrations import _MIGRATABLE_NAMES, _MIGRATIONS 

46 

47try: 

48 from lsst.daf.butler import DatasetProvenance, SerializedDatasetRef 

49except ImportError: 

50 type DatasetProvenance = Any # type: ignore[no-redef] 

51 type SerializedDatasetRef = Any # type: ignore[no-redef] 

52 

53if TYPE_CHECKING: 

54 import astropy.io.fits 

55 

56 from ._input_archive import InputArchive 

57 

58 

59type MetadataValue = ( 

60 pydantic.StrictInt | pydantic.StrictFloat | pydantic.StrictStr | pydantic.StrictBool | None 

61) 

62 

63SCHEMA_URL_BASE = "https://images.lsst.io/schemas" 

64"""Base for the schema URLs of this package's own schemas, used as 

65``{SCHEMA_URL_BASE}/{name}-{version}``. 

66 

67External packages providing their own schemas override 

68`~lsst.images.serialization.ArchiveTree.SCHEMA_URL_BASE` instead, so their 

69schema URLs are minted under a site they control. 

70""" 

71 

72_ARCHIVE_READ_CONTEXT = object() 

73"""Pydantic context used when validating a tree read from an archive.""" 

74 

75 

76class ButlerInfo(pydantic.BaseModel): 

77 """Information about a butler dataset.""" 

78 

79 dataset: SerializedDatasetRef 

80 provenance: DatasetProvenance = pydantic.Field(default_factory=DatasetProvenance) 

81 

82 

83class JsonRef(pydantic.BaseModel, serialize_by_alias=True): 

84 """Pydantic model for JSON Reference / Pointer (IETF RFC 6901). 

85 

86 Notes 

87 ----- 

88 This model does not do any of the escaping or special-character 

89 interpretation required by the spec; it assumes that's already been done, 

90 so its job is *just* putting a ``$ref`` field inside another model. 

91 """ 

92 

93 ref: str = pydantic.Field(alias="$ref") 

94 

95 

96class ArchiveTree( 

97 pydantic.BaseModel, ABC, ser_json_inf_nan="constants", ser_json_bytes="base64", val_json_bytes="base64" 

98): 

99 """An intermediate base class of `pydantic.BaseModel` that should be used 

100 for all objects that may be used as the top-level tree models written to 

101 archives. 

102 

103 See :ref:`lsst.images-schema-versioning` for how the ``SCHEMA_NAME`` / 

104 ``SCHEMA_VERSION`` / ``MIN_READ_VERSION`` constants and the 

105 ``schema_version`` / ``min_read_version`` / ``schema_url`` fields are used. 

106 """ 

107 

108 SCHEMA_NAME: ClassVar[str] 

109 SCHEMA_VERSION: ClassVar[str] 

110 MIN_READ_VERSION: ClassVar[int] 

111 SCHEMA_URL_BASE: ClassVar[str] = SCHEMA_URL_BASE 

112 """Base for this schema's URL, as ``{SCHEMA_URL_BASE}/{name}-{version}``. 

113 

114 External packages providing their own schemas should override this (once, 

115 on a shared intermediate base class) so their schema URLs are minted 

116 under a documentation site they control rather than images.lsst.io.""" 

117 

118 PUBLIC_TYPE: ClassVar[type] 

119 """In-memory Python type produced by this tree's ``deserialize`` (e.g. 

120 `dict` for a mapping return). Declared explicitly by each concrete 

121 subclass and surfaced by 

122 `~lsst.images.serialization.public_type_for_schema`.""" 

123 

124 # These mirror the SCHEMA_VERSION / MIN_READ_VERSION class constants and 

125 # are never passed on construction, so repr omits them; they say nothing 

126 # about the object that its type does not already say. 

127 schema_version: str = pydantic.Field( 

128 default="1.0.0", 

129 description="Data-model schema version of this tree (major.minor.patch).", 

130 repr=False, 

131 ) 

132 min_read_version: int = pydantic.Field( 

133 default=1, 

134 description="Smallest reader major that can interpret this tree.", 

135 repr=False, 

136 ) 

137 metadata: dict[str, MetadataValue] = pydantic.Field( 

138 default_factory=dict, description="Additional unstructured metadata.", exclude_if=operator.not_ 

139 ) 

140 butler_info: ButlerInfo | None = pydantic.Field( 

141 default=None, 

142 description="Information about the butler dataset backed by this file.", 

143 exclude_if=is_none, 

144 ) 

145 # Populated by the archive machinery rather than by a caller, so repr 

146 # omits it as well. 

147 indirect: list[Any] = pydantic.Field( 

148 default_factory=list, 

149 description="Serialized nested objects that may be saved or read more than once.", 

150 exclude_if=operator.not_, 

151 repr=False, 

152 ) 

153 

154 @pydantic.computed_field(description="Canonical schema URL for this tree.") # type: ignore[prop-decorator] 

155 @property 

156 def schema_url(self) -> str: 

157 """Return the schema URL of this tree's class. 

158 

159 Computed from ``SCHEMA_NAME`` and ``SCHEMA_VERSION`` ClassVars. 

160 """ 

161 cls = type(self) 

162 return f"{cls.SCHEMA_URL_BASE}/{cls.SCHEMA_NAME}-{cls.SCHEMA_VERSION}" 

163 

164 @pydantic.model_validator(mode="before") 

165 @classmethod 

166 def _migrate_from_older_major(cls, data: Any, info: pydantic.ValidationInfo) -> Any: 

167 """Morph an older on-disk tree into the current shape. 

168 

169 Chains registered migrations until the tree reaches the in-code major, 

170 so the ``mode="after"`` validator below sees a current-shape tree. 

171 

172 A tree validated in the archive-read context, which the backends mark 

173 via ``info.context``, must carry a ``schema_version``; in-memory 

174 construction is unmarked and keeps the class-constant defaults. See 

175 :ref:`lsst.images-schema-versioning-migration`. 

176 """ 

177 if not isinstance(data, dict): 

178 return data 

179 if not hasattr(cls, "SCHEMA_NAME"): 179 ↛ 184line 179 didn't jump to line 184 because the condition on line 179 was never true

180 # ArchiveTree itself is abstract, and a subclass that has not yet 

181 # declared SCHEMA_NAME (during incremental rollout) has no 

182 # SCHEMA_VERSION to parse either; skip cleanly rather than 

183 # raising AttributeError, mirroring the after-validator below. 

184 return data 

185 name = cls.SCHEMA_NAME 

186 if "schema_version" not in data: 

187 if info.context is _ARCHIVE_READ_CONTEXT: 

188 raise _MissingSchemaVersionError( 

189 f"{name}: archive tree has no schema_version; unstamped archive data is not supported." 

190 ) 

191 return data 

192 if not _MIGRATABLE_NAMES: 192 ↛ 193line 192 didn't jump to line 193 because the condition on line 192 was never true

193 return data 

194 if name not in _MIGRATABLE_NAMES: 

195 # This schema has not opted into migrations at all, regardless of 

196 # its major: whether some *other*, unrelated schema has a 

197 # registered migration must not change what this one does with 

198 # its own tree. A schema in this state falls through to ordinary 

199 # validation -- in-model backfill for an additive change, or a 

200 # Pydantic validation error for a real incompatibility. 

201 return data 

202 on_disk_major = _parse_on_disk_major(data["schema_version"]) 

203 in_code_major = _parse_major(cls.SCHEMA_VERSION) 

204 if on_disk_major < in_code_major: 

205 # A failed union candidate must not mutate the input subsequently 

206 # tried by another candidate. Migrations may mutate freely, so 

207 # isolate the complete JSON-like tree, not just its top-level 

208 # dictionary. 

209 data = deepcopy(data) 

210 while on_disk_major < in_code_major: 

211 try: 

212 step = _MIGRATIONS[(name, on_disk_major)] 

213 except KeyError: 

214 raise _MigrationGapError( 

215 f"{name}: no migration from major {on_disk_major} to {on_disk_major + 1}." 

216 ) from None 

217 data = step(data) 

218 on_disk_major += 1 

219 data["schema_version"] = f"{on_disk_major}.0.0" 

220 return data 

221 

222 @pydantic.model_validator(mode="after") 

223 def _check_and_normalize_schema_version(self) -> Self: 

224 """Validate and normalise the schema version fields. 

225 

226 Compares the on-tree ``schema_version`` / ``min_read_version`` against 

227 the in-code values from the subclass's ClassVars; raises if 

228 incompatible, otherwise normalises the fields to the in-code values. 

229 """ 

230 cls = type(self) 

231 # ArchiveTree itself is abstract (deserialize is @abstractmethod). 

232 # Subclasses that haven't yet declared SCHEMA_NAME are skipped — this 

233 # matters during incremental rollout and remains a safe no-op 

234 # afterwards (a class-invariants test ensures every concrete subclass 

235 # has the constants). 

236 if not hasattr(cls, "SCHEMA_NAME"): 236 ↛ 237line 236 didn't jump to line 237 because the condition on line 236 was never true

237 return self 

238 _check_compat( 

239 cls.SCHEMA_NAME, 

240 self.schema_version, 

241 self.min_read_version, 

242 cls.SCHEMA_VERSION, 

243 ) 

244 if self.schema_version != cls.SCHEMA_VERSION: 

245 self.schema_version = cls.SCHEMA_VERSION 

246 if self.min_read_version != cls.MIN_READ_VERSION: 

247 self.min_read_version = cls.MIN_READ_VERSION 

248 return self 

249 

250 @classmethod 

251 def __pydantic_init_subclass__(cls, **kwargs: Any) -> None: 

252 """Inject ``$id`` and ``title`` into the subclass's JSON Schema, and 

253 register the subclass in the schema-name registry. 

254 

255 Populates ``model_config['json_schema_extra']`` with values derived 

256 from the subclass's ``SCHEMA_NAME`` / ``SCHEMA_VERSION`` ClassVars, 

257 then registers the subclass so it can be looked up by schema name. 

258 Subclasses that haven't declared the ClassVars are skipped. 

259 """ 

260 super().__pydantic_init_subclass__(**kwargs) 

261 name = cls.__dict__.get("SCHEMA_NAME") 

262 version = cls.__dict__.get("SCHEMA_VERSION") 

263 if name is None or version is None: 

264 return 

265 json_schema_extra = cls.model_config.get("json_schema_extra") or {} 

266 if isinstance(json_schema_extra, dict): 266 ↛ 278line 266 didn't jump to line 278 because the condition on line 266 was always true

267 existing = dict(json_schema_extra) 

268 # Always override: a subclass of a concrete schema (e.g. 

269 # visit_image subclassing masked_image) inherits its parent's 

270 # already-stamped values through the merged model_config, and 

271 # this hook only runs when the subclass declares its own 

272 # SCHEMA_NAME / SCHEMA_VERSION for these to be derived from. 

273 existing["$id"] = f"{cls.SCHEMA_URL_BASE}/{name}-{version}" 

274 existing["title"] = name 

275 cls.model_config = {**cls.model_config, "json_schema_extra": existing} 

276 # Local import to avoid the _io -> _common circular dependency at 

277 # module load time. 

278 from ._io import register_schema_class 

279 

280 register_schema_class(cls) 

281 

282 @abstractmethod 

283 def deserialize(self, archive: InputArchive[Any], **kwargs: Any) -> Any: 

284 """Return the in-memory object that was serialized to this tree. 

285 

286 Parameters 

287 ---------- 

288 archive 

289 The input archive to read from. 

290 **kwargs 

291 Additional keyword arguments specific to this type. 

292 

293 Raises 

294 ------ 

295 ~lsst.images.serialization.InvalidParameterError 

296 Raised for unsupported ``**kwargs``. 

297 

298 Notes 

299 ----- 

300 Subclass implementations may take additional keyword-only arguments. 

301 Callers that invoke this method without knowing what those might be 

302 should catch `TypeError` and re-raise as 

303 `~lsst.images.serialization.InvalidParameterError` if they pass 

304 additional keyword arguments. 

305 """ 

306 raise NotImplementedError() 

307 

308 def deserialize_component(self, component: str, archive: InputArchive[Any], **kwargs: Any) -> Any: 

309 """Return a component in-memory object that was serialized to this 

310 tree. 

311 

312 Parameters 

313 ---------- 

314 component 

315 Name of the component to read. 

316 archive 

317 The input archive to read from. 

318 **kwargs 

319 Additional keyword arguments specific to this type. 

320 

321 Raises 

322 ------ 

323 ~lsst.images.serialization.InvalidComponentError 

324 Raise if ``component`` is not recognized. 

325 ~lsst.images.serialization.InvalidParameterError 

326 Raised for unsupported ``**kwargs``. 

327 

328 Notes 

329 ----- 

330 The default implementation for this method tries to get an attribute 

331 with the component's name from ``self``, and then: 

332 

333 - returns `None` if it is `None`; 

334 - calls `deserialize` on that object if it is also an 

335 `~lsst.images.serialization.ArchiveTree`; 

336 - returns it directly otherwise. 

337 

338 If there is no such attribute, it raises 

339 `~lsst.images.serialization.InvalidComponentError`. 

340 

341 ``**kwargs`` are forwarded to component `deserialize` methods, but 

342 are otherwise not checked. Subclasses are generally expected to 

343 implement this method to do that checking and handle any components 

344 for which the other will not work, and then delegate to `super` at 

345 the end. 

346 """ 

347 try: 

348 component_model = getattr(self, component) 

349 except AttributeError: 

350 raise InvalidComponentError( 

351 f"Component {component!r} is not recognized by {type(self).__name__}." 

352 ) from None 

353 if component_model is None: 

354 return None 

355 if isinstance(component_model, ArchiveTree): 

356 return component_model.deserialize(archive, **kwargs) 

357 return component_model 

358 

359 

360class DevelopmentSchemaWarning(UserWarning): 

361 """Warning that a file is being written with a development schema.""" 

362 

363 

364def is_development_version(version: str) -> bool: 

365 """Return whether a schema version string is a PEP 440 development release. 

366 

367 Parameters 

368 ---------- 

369 version 

370 Schema version string, e.g. ``1.0.0`` or ``1.0.0.dev0``. 

371 """ 

372 return Version(version).is_devrelease 

373 

374 

375def _iter_archive_trees(obj: Any) -> Iterator[ArchiveTree]: 

376 """Yield every `ArchiveTree` embedded in a serialized tree.""" 

377 if isinstance(obj, ArchiveTree): 

378 yield obj 

379 if isinstance(obj, pydantic.BaseModel): 

380 for field_name in type(obj).model_fields: 

381 yield from _iter_archive_trees(getattr(obj, field_name)) 

382 elif isinstance(obj, list | tuple): 

383 for item in obj: 

384 yield from _iter_archive_trees(item) 

385 elif isinstance(obj, dict): 

386 for value in obj.values(): 

387 yield from _iter_archive_trees(value) 

388 

389 

390def warn_for_development_schemas(root: ArchiveTree) -> None: 

391 """Emit a `DevelopmentSchemaWarning` if a serialized tree contains any 

392 schema still in development. 

393 

394 Parameters 

395 ---------- 

396 root 

397 Top-level serialized tree about to be written. 

398 """ 

399 developing = sorted( 

400 { 

401 tree.schema_url 

402 for tree in _iter_archive_trees(root) 

403 if is_development_version(type(tree).SCHEMA_VERSION) 

404 } 

405 ) 

406 if developing: 

407 warnings.warn( 

408 "Writing a file with development schema(s) " 

409 f"{', '.join(developing)}; such files are not for production and " 

410 "may not remain readable.", 

411 DevelopmentSchemaWarning, 

412 stacklevel=3, 

413 ) 

414 

415 

416class ArchiveReadError(RuntimeError): 

417 """Exception raised when the contents of an archive cannot be read.""" 

418 

419 

420class InvalidParameterError(ArchiveReadError): 

421 """Exception raised by `ArchiveTree.deserialize` or 

422 `ArchiveTree.deserialize_component` when passed an invalid keyword 

423 argument. 

424 """ 

425 

426 

427class InvalidComponentError(ArchiveReadError): 

428 """Exception `ArchiveTree.deserialize_component` when passed an invalid 

429 component name. 

430 """ 

431 

432 

433class _MigrationGapError(ArchiveReadError, ValueError): 

434 """A migration chain has no registered step for the tree's on-disk major. 

435 

436 Raised from the ``mode="before"`` validator in 

437 `ArchiveTree._migrate_from_older_major`, which pydantic-core invokes as 

438 part of validating a field. Deriving only from `ArchiveReadError` (a 

439 `RuntimeError`) would let this exception escape a discriminated union 

440 entirely: pydantic only treats `ValueError`, `TypeError` and 

441 `AssertionError` raised by a validator as "this candidate failed, try the 

442 next one", so a `RuntimeError` would abort the whole union instead of 

443 falling through. Deriving from both means a ``left_to_right`` (or 

444 ``smart``) union tries the next variant on a migration gap exactly as it 

445 would on any other validation failure, while ``except ArchiveReadError`` 

446 callers outside a union still catch it unchanged. 

447 """ 

448 

449 

450class _MissingSchemaVersionError(ArchiveReadError, ValueError): 

451 """An archive tree omitted its required ``schema_version`` stamp.""" 

452 

453 

454class _MalformedSchemaVersionError(ArchiveReadError, ValueError): 

455 """An archive tree's ``schema_version`` stamp could not be parsed. 

456 

457 Also a `ValueError`, for the reason `_MigrationGapError` gives: the stamp 

458 is parsed in the ``mode="before"`` validator, which runs ahead of field 

459 validation for every migratable union candidate, including ones the 

460 payload does not belong to. A stamp the reader cannot parse therefore has 

461 to fail that one candidate rather than abort the whole union. 

462 

463 Only an *on-disk* stamp raises this. A malformed in-code 

464 ``SCHEMA_VERSION`` is a programming error and keeps raising the 

465 `RuntimeError`-only `ArchiveReadError`, so a union cannot swallow it. 

466 """ 

467 

468 

469class ArchiveAccessRequiredError(RuntimeError): 

470 """Exception raised when a deserialization needs data from the file. 

471 

472 Raised by all data-access methods of 

473 `~lsst.images.serialization.DetachedArchive`, signaling that the 

474 requested object cannot be deserialized from the JSON tree alone. 

475 

476 Notes 

477 ----- 

478 This deliberately does not inherit from `ArchiveReadError`: it signals 

479 that a live archive is required rather than that an archive is corrupt, 

480 and it must not be swallowed by ``except ArchiveReadError`` handlers. 

481 """ 

482 

483 

484class OpaqueArchiveMetadata(Protocol): 

485 """Interface for opaque archive metadata. 

486 

487 In addition to implementing the methods defined here, all implementations 

488 must be pickleable. 

489 """ 

490 

491 def copy(self) -> Self | None: 

492 """Copy, reference, or discard metadata when its holding object is 

493 copied. 

494 """ 

495 ... 

496 

497 def subset(self, bbox: Box) -> Self | None: 

498 """Copy, reference, or discard metadata when a subset of its its 

499 holding object is extracted. 

500 

501 Parameters 

502 ---------- 

503 bbox 

504 Bounding box of the subset being extracted. 

505 """ 

506 ... 

507 

508 

509def no_header_updates(header: astropy.io.fits.Header) -> None: 

510 """Do not make any modifications to the given FITS header. 

511 

512 Parameters 

513 ---------- 

514 header 

515 FITS header that is left unchanged. 

516 """ 

517 

518 

519def _parse_major(version: object) -> int: 

520 """Return the integer major component of a major.minor.patch string. 

521 

522 Accepts any object, because callers pass values that may have come 

523 straight from a file; use `_parse_on_disk_major` for those. 

524 

525 Raises 

526 ------ 

527 ArchiveReadError 

528 If ``version`` is not a non-empty string of the form 

529 ``major.minor.patch`` with integer components. 

530 """ 

531 if not isinstance(version, str) or not version: 

532 raise ArchiveReadError(f"Schema version {version!r} is not a non-empty string.") 

533 head = version.split(".", 1)[0] 

534 try: 

535 return int(head) 

536 except ValueError as exc: 

537 raise ArchiveReadError(f"Schema version {version!r} has non-integer major.") from exc 

538 

539 

540def _parse_on_disk_major(version: object) -> int: 

541 """Return the integer major of a version stamp that came from a file. 

542 

543 Raises 

544 ------ 

545 _MalformedSchemaVersionError 

546 If ``version`` is not a non-empty string of the form 

547 ``major.minor.patch`` with integer components. Unlike `_parse_major`, 

548 this is a `ValueError`, so a stamp the reader cannot parse fails one 

549 union candidate instead of aborting the union. 

550 """ 

551 try: 

552 return _parse_major(version) 

553 except ArchiveReadError as exc: 

554 raise _MalformedSchemaVersionError(str(exc)) from None 

555 

556 

557def _check_compat( 

558 name: str, 

559 on_disk_version: str, 

560 on_disk_min_read: int, 

561 in_code_version: str, 

562) -> None: 

563 """Raise `ArchiveReadError` if a tree written with the given 

564 schema_version/min_read_version cannot be read by the current code. 

565 

566 See :ref:`lsst.images-schema-versioning` for the compatibility rule. 

567 """ 

568 in_code_major = _parse_major(in_code_version) 

569 if on_disk_min_read > in_code_major: 

570 raise ArchiveReadError( 

571 f"{name}: tree requires reader major >= {on_disk_min_read}; this release is {in_code_version}." 

572 ) 

573 

574 

575def _check_format_version(name: str, on_disk: int, in_code: int) -> None: 

576 """Raise `ArchiveReadError` if a backend file's container layout 

577 version is newer than this release knows how to read. 

578 """ 

579 if on_disk > in_code: 

580 raise ArchiveReadError( 

581 f"{name}: on-disk container format version {on_disk} is " 

582 f"newer than this release ({in_code}); cannot read." 

583 )