Coverage for python/lsst/images/serialization/_common.py: 62%
167 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-08-17 21:43 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-08-17 21:43 +0000
1# This file is part of lsst-images.
2#
3# Developed for the LSST Data Management System.
4# This product includes software developed by the LSST Project
5# (https://www.lsst.org).
6# See the COPYRIGHT file at the top-level directory of this distribution
7# for details of code ownership.
8#
9# Use of this source code is governed by a 3-clause BSD-style
10# license that can be found in the LICENSE file.
12from __future__ import annotations
14__all__ = (
15 "SCHEMA_URL_BASE",
16 "ArchiveAccessRequiredError",
17 "ArchiveReadError",
18 "ArchiveTree",
19 "ButlerInfo",
20 "DevelopmentSchemaWarning",
21 "InvalidComponentError",
22 "InvalidParameterError",
23 "JsonRef",
24 "MetadataValue",
25 "OpaqueArchiveMetadata",
26 "is_development_version",
27 "no_header_updates",
28 "warn_for_development_schemas",
29)
31import operator
32import warnings
33from abc import ABC, abstractmethod
34from collections.abc import Iterator
35from copy import deepcopy
36from typing import TYPE_CHECKING, Any, ClassVar, Protocol, Self
38import astropy.table
39import astropy.units
40import pydantic
41from packaging.version import Version
43from .._geom import Box
44from ..utils import is_none
45from ._migrations import _MIGRATABLE_NAMES, _MIGRATIONS
47try:
48 from lsst.daf.butler import DatasetProvenance, SerializedDatasetRef
49except ImportError:
50 type DatasetProvenance = Any # type: ignore[no-redef]
51 type SerializedDatasetRef = Any # type: ignore[no-redef]
53if TYPE_CHECKING:
54 import astropy.io.fits
56 from ._input_archive import InputArchive
59type MetadataValue = (
60 pydantic.StrictInt | pydantic.StrictFloat | pydantic.StrictStr | pydantic.StrictBool | None
61)
63SCHEMA_URL_BASE = "https://images.lsst.io/schemas"
64"""Base for the schema URLs of this package's own schemas, used as
65``{SCHEMA_URL_BASE}/{name}-{version}``.
67External packages providing their own schemas override
68`~lsst.images.serialization.ArchiveTree.SCHEMA_URL_BASE` instead, so their
69schema URLs are minted under a site they control.
70"""
72_ARCHIVE_READ_CONTEXT = object()
73"""Pydantic context used when validating a tree read from an archive."""
76class ButlerInfo(pydantic.BaseModel):
77 """Information about a butler dataset."""
79 dataset: SerializedDatasetRef
80 provenance: DatasetProvenance = pydantic.Field(default_factory=DatasetProvenance)
83class JsonRef(pydantic.BaseModel, serialize_by_alias=True):
84 """Pydantic model for JSON Reference / Pointer (IETF RFC 6901).
86 Notes
87 -----
88 This model does not do any of the escaping or special-character
89 interpretation required by the spec; it assumes that's already been done,
90 so its job is *just* putting a ``$ref`` field inside another model.
91 """
93 ref: str = pydantic.Field(alias="$ref")
96class ArchiveTree(
97 pydantic.BaseModel, ABC, ser_json_inf_nan="constants", ser_json_bytes="base64", val_json_bytes="base64"
98):
99 """An intermediate base class of `pydantic.BaseModel` that should be used
100 for all objects that may be used as the top-level tree models written to
101 archives.
103 See :ref:`lsst.images-schema-versioning` for how the ``SCHEMA_NAME`` /
104 ``SCHEMA_VERSION`` / ``MIN_READ_VERSION`` constants and the
105 ``schema_version`` / ``min_read_version`` / ``schema_url`` fields are used.
106 """
108 SCHEMA_NAME: ClassVar[str]
109 SCHEMA_VERSION: ClassVar[str]
110 MIN_READ_VERSION: ClassVar[int]
111 SCHEMA_URL_BASE: ClassVar[str] = SCHEMA_URL_BASE
112 """Base for this schema's URL, as ``{SCHEMA_URL_BASE}/{name}-{version}``.
114 External packages providing their own schemas should override this (once,
115 on a shared intermediate base class) so their schema URLs are minted
116 under a documentation site they control rather than images.lsst.io."""
118 PUBLIC_TYPE: ClassVar[type]
119 """In-memory Python type produced by this tree's ``deserialize`` (e.g.
120 `dict` for a mapping return). Declared explicitly by each concrete
121 subclass and surfaced by
122 `~lsst.images.serialization.public_type_for_schema`."""
124 # These mirror the SCHEMA_VERSION / MIN_READ_VERSION class constants and
125 # are never passed on construction, so repr omits them; they say nothing
126 # about the object that its type does not already say.
127 schema_version: str = pydantic.Field(
128 default="1.0.0",
129 description="Data-model schema version of this tree (major.minor.patch).",
130 repr=False,
131 )
132 min_read_version: int = pydantic.Field(
133 default=1,
134 description="Smallest reader major that can interpret this tree.",
135 repr=False,
136 )
137 metadata: dict[str, MetadataValue] = pydantic.Field(
138 default_factory=dict, description="Additional unstructured metadata.", exclude_if=operator.not_
139 )
140 butler_info: ButlerInfo | None = pydantic.Field(
141 default=None,
142 description="Information about the butler dataset backed by this file.",
143 exclude_if=is_none,
144 )
145 # Populated by the archive machinery rather than by a caller, so repr
146 # omits it as well.
147 indirect: list[Any] = pydantic.Field(
148 default_factory=list,
149 description="Serialized nested objects that may be saved or read more than once.",
150 exclude_if=operator.not_,
151 repr=False,
152 )
154 @pydantic.computed_field(description="Canonical schema URL for this tree.") # type: ignore[prop-decorator]
155 @property
156 def schema_url(self) -> str:
157 """Return the schema URL of this tree's class.
159 Computed from ``SCHEMA_NAME`` and ``SCHEMA_VERSION`` ClassVars.
160 """
161 cls = type(self)
162 return f"{cls.SCHEMA_URL_BASE}/{cls.SCHEMA_NAME}-{cls.SCHEMA_VERSION}"
164 @pydantic.model_validator(mode="before")
165 @classmethod
166 def _migrate_from_older_major(cls, data: Any, info: pydantic.ValidationInfo) -> Any:
167 """Morph an older on-disk tree into the current shape.
169 Chains registered migrations until the tree reaches the in-code major,
170 so the ``mode="after"`` validator below sees a current-shape tree.
172 A tree validated in the archive-read context, which the backends mark
173 via ``info.context``, must carry a ``schema_version``; in-memory
174 construction is unmarked and keeps the class-constant defaults. See
175 :ref:`lsst.images-schema-versioning-migration`.
176 """
177 if not isinstance(data, dict):
178 return data
179 if not hasattr(cls, "SCHEMA_NAME"): 179 ↛ 184line 179 didn't jump to line 184 because the condition on line 179 was never true
180 # ArchiveTree itself is abstract, and a subclass that has not yet
181 # declared SCHEMA_NAME (during incremental rollout) has no
182 # SCHEMA_VERSION to parse either; skip cleanly rather than
183 # raising AttributeError, mirroring the after-validator below.
184 return data
185 name = cls.SCHEMA_NAME
186 if "schema_version" not in data:
187 if info.context is _ARCHIVE_READ_CONTEXT:
188 raise _MissingSchemaVersionError(
189 f"{name}: archive tree has no schema_version; unstamped archive data is not supported."
190 )
191 return data
192 if not _MIGRATABLE_NAMES: 192 ↛ 193line 192 didn't jump to line 193 because the condition on line 192 was never true
193 return data
194 if name not in _MIGRATABLE_NAMES:
195 # This schema has not opted into migrations at all, regardless of
196 # its major: whether some *other*, unrelated schema has a
197 # registered migration must not change what this one does with
198 # its own tree. A schema in this state falls through to ordinary
199 # validation -- in-model backfill for an additive change, or a
200 # Pydantic validation error for a real incompatibility.
201 return data
202 on_disk_major = _parse_on_disk_major(data["schema_version"])
203 in_code_major = _parse_major(cls.SCHEMA_VERSION)
204 if on_disk_major < in_code_major:
205 # A failed union candidate must not mutate the input subsequently
206 # tried by another candidate. Migrations may mutate freely, so
207 # isolate the complete JSON-like tree, not just its top-level
208 # dictionary.
209 data = deepcopy(data)
210 while on_disk_major < in_code_major:
211 try:
212 step = _MIGRATIONS[(name, on_disk_major)]
213 except KeyError:
214 raise _MigrationGapError(
215 f"{name}: no migration from major {on_disk_major} to {on_disk_major + 1}."
216 ) from None
217 data = step(data)
218 on_disk_major += 1
219 data["schema_version"] = f"{on_disk_major}.0.0"
220 return data
222 @pydantic.model_validator(mode="after")
223 def _check_and_normalize_schema_version(self) -> Self:
224 """Validate and normalise the schema version fields.
226 Compares the on-tree ``schema_version`` / ``min_read_version`` against
227 the in-code values from the subclass's ClassVars; raises if
228 incompatible, otherwise normalises the fields to the in-code values.
229 """
230 cls = type(self)
231 # ArchiveTree itself is abstract (deserialize is @abstractmethod).
232 # Subclasses that haven't yet declared SCHEMA_NAME are skipped — this
233 # matters during incremental rollout and remains a safe no-op
234 # afterwards (a class-invariants test ensures every concrete subclass
235 # has the constants).
236 if not hasattr(cls, "SCHEMA_NAME"): 236 ↛ 237line 236 didn't jump to line 237 because the condition on line 236 was never true
237 return self
238 _check_compat(
239 cls.SCHEMA_NAME,
240 self.schema_version,
241 self.min_read_version,
242 cls.SCHEMA_VERSION,
243 )
244 if self.schema_version != cls.SCHEMA_VERSION:
245 self.schema_version = cls.SCHEMA_VERSION
246 if self.min_read_version != cls.MIN_READ_VERSION:
247 self.min_read_version = cls.MIN_READ_VERSION
248 return self
250 @classmethod
251 def __pydantic_init_subclass__(cls, **kwargs: Any) -> None:
252 """Inject ``$id`` and ``title`` into the subclass's JSON Schema, and
253 register the subclass in the schema-name registry.
255 Populates ``model_config['json_schema_extra']`` with values derived
256 from the subclass's ``SCHEMA_NAME`` / ``SCHEMA_VERSION`` ClassVars,
257 then registers the subclass so it can be looked up by schema name.
258 Subclasses that haven't declared the ClassVars are skipped.
259 """
260 super().__pydantic_init_subclass__(**kwargs)
261 name = cls.__dict__.get("SCHEMA_NAME")
262 version = cls.__dict__.get("SCHEMA_VERSION")
263 if name is None or version is None:
264 return
265 json_schema_extra = cls.model_config.get("json_schema_extra") or {}
266 if isinstance(json_schema_extra, dict): 266 ↛ 278line 266 didn't jump to line 278 because the condition on line 266 was always true
267 existing = dict(json_schema_extra)
268 # Always override: a subclass of a concrete schema (e.g.
269 # visit_image subclassing masked_image) inherits its parent's
270 # already-stamped values through the merged model_config, and
271 # this hook only runs when the subclass declares its own
272 # SCHEMA_NAME / SCHEMA_VERSION for these to be derived from.
273 existing["$id"] = f"{cls.SCHEMA_URL_BASE}/{name}-{version}"
274 existing["title"] = name
275 cls.model_config = {**cls.model_config, "json_schema_extra": existing}
276 # Local import to avoid the _io -> _common circular dependency at
277 # module load time.
278 from ._io import register_schema_class
280 register_schema_class(cls)
282 @abstractmethod
283 def deserialize(self, archive: InputArchive[Any], **kwargs: Any) -> Any:
284 """Return the in-memory object that was serialized to this tree.
286 Parameters
287 ----------
288 archive
289 The input archive to read from.
290 **kwargs
291 Additional keyword arguments specific to this type.
293 Raises
294 ------
295 ~lsst.images.serialization.InvalidParameterError
296 Raised for unsupported ``**kwargs``.
298 Notes
299 -----
300 Subclass implementations may take additional keyword-only arguments.
301 Callers that invoke this method without knowing what those might be
302 should catch `TypeError` and re-raise as
303 `~lsst.images.serialization.InvalidParameterError` if they pass
304 additional keyword arguments.
305 """
306 raise NotImplementedError()
308 def deserialize_component(self, component: str, archive: InputArchive[Any], **kwargs: Any) -> Any:
309 """Return a component in-memory object that was serialized to this
310 tree.
312 Parameters
313 ----------
314 component
315 Name of the component to read.
316 archive
317 The input archive to read from.
318 **kwargs
319 Additional keyword arguments specific to this type.
321 Raises
322 ------
323 ~lsst.images.serialization.InvalidComponentError
324 Raise if ``component`` is not recognized.
325 ~lsst.images.serialization.InvalidParameterError
326 Raised for unsupported ``**kwargs``.
328 Notes
329 -----
330 The default implementation for this method tries to get an attribute
331 with the component's name from ``self``, and then:
333 - returns `None` if it is `None`;
334 - calls `deserialize` on that object if it is also an
335 `~lsst.images.serialization.ArchiveTree`;
336 - returns it directly otherwise.
338 If there is no such attribute, it raises
339 `~lsst.images.serialization.InvalidComponentError`.
341 ``**kwargs`` are forwarded to component `deserialize` methods, but
342 are otherwise not checked. Subclasses are generally expected to
343 implement this method to do that checking and handle any components
344 for which the other will not work, and then delegate to `super` at
345 the end.
346 """
347 try:
348 component_model = getattr(self, component)
349 except AttributeError:
350 raise InvalidComponentError(
351 f"Component {component!r} is not recognized by {type(self).__name__}."
352 ) from None
353 if component_model is None:
354 return None
355 if isinstance(component_model, ArchiveTree):
356 return component_model.deserialize(archive, **kwargs)
357 return component_model
360class DevelopmentSchemaWarning(UserWarning):
361 """Warning that a file is being written with a development schema."""
364def is_development_version(version: str) -> bool:
365 """Return whether a schema version string is a PEP 440 development release.
367 Parameters
368 ----------
369 version
370 Schema version string, e.g. ``1.0.0`` or ``1.0.0.dev0``.
371 """
372 return Version(version).is_devrelease
375def _iter_archive_trees(obj: Any) -> Iterator[ArchiveTree]:
376 """Yield every `ArchiveTree` embedded in a serialized tree."""
377 if isinstance(obj, ArchiveTree):
378 yield obj
379 if isinstance(obj, pydantic.BaseModel):
380 for field_name in type(obj).model_fields:
381 yield from _iter_archive_trees(getattr(obj, field_name))
382 elif isinstance(obj, list | tuple):
383 for item in obj:
384 yield from _iter_archive_trees(item)
385 elif isinstance(obj, dict):
386 for value in obj.values():
387 yield from _iter_archive_trees(value)
390def warn_for_development_schemas(root: ArchiveTree) -> None:
391 """Emit a `DevelopmentSchemaWarning` if a serialized tree contains any
392 schema still in development.
394 Parameters
395 ----------
396 root
397 Top-level serialized tree about to be written.
398 """
399 developing = sorted(
400 {
401 tree.schema_url
402 for tree in _iter_archive_trees(root)
403 if is_development_version(type(tree).SCHEMA_VERSION)
404 }
405 )
406 if developing:
407 warnings.warn(
408 "Writing a file with development schema(s) "
409 f"{', '.join(developing)}; such files are not for production and "
410 "may not remain readable.",
411 DevelopmentSchemaWarning,
412 stacklevel=3,
413 )
416class ArchiveReadError(RuntimeError):
417 """Exception raised when the contents of an archive cannot be read."""
420class InvalidParameterError(ArchiveReadError):
421 """Exception raised by `ArchiveTree.deserialize` or
422 `ArchiveTree.deserialize_component` when passed an invalid keyword
423 argument.
424 """
427class InvalidComponentError(ArchiveReadError):
428 """Exception `ArchiveTree.deserialize_component` when passed an invalid
429 component name.
430 """
433class _MigrationGapError(ArchiveReadError, ValueError):
434 """A migration chain has no registered step for the tree's on-disk major.
436 Raised from the ``mode="before"`` validator in
437 `ArchiveTree._migrate_from_older_major`, which pydantic-core invokes as
438 part of validating a field. Deriving only from `ArchiveReadError` (a
439 `RuntimeError`) would let this exception escape a discriminated union
440 entirely: pydantic only treats `ValueError`, `TypeError` and
441 `AssertionError` raised by a validator as "this candidate failed, try the
442 next one", so a `RuntimeError` would abort the whole union instead of
443 falling through. Deriving from both means a ``left_to_right`` (or
444 ``smart``) union tries the next variant on a migration gap exactly as it
445 would on any other validation failure, while ``except ArchiveReadError``
446 callers outside a union still catch it unchanged.
447 """
450class _MissingSchemaVersionError(ArchiveReadError, ValueError):
451 """An archive tree omitted its required ``schema_version`` stamp."""
454class _MalformedSchemaVersionError(ArchiveReadError, ValueError):
455 """An archive tree's ``schema_version`` stamp could not be parsed.
457 Also a `ValueError`, for the reason `_MigrationGapError` gives: the stamp
458 is parsed in the ``mode="before"`` validator, which runs ahead of field
459 validation for every migratable union candidate, including ones the
460 payload does not belong to. A stamp the reader cannot parse therefore has
461 to fail that one candidate rather than abort the whole union.
463 Only an *on-disk* stamp raises this. A malformed in-code
464 ``SCHEMA_VERSION`` is a programming error and keeps raising the
465 `RuntimeError`-only `ArchiveReadError`, so a union cannot swallow it.
466 """
469class ArchiveAccessRequiredError(RuntimeError):
470 """Exception raised when a deserialization needs data from the file.
472 Raised by all data-access methods of
473 `~lsst.images.serialization.DetachedArchive`, signaling that the
474 requested object cannot be deserialized from the JSON tree alone.
476 Notes
477 -----
478 This deliberately does not inherit from `ArchiveReadError`: it signals
479 that a live archive is required rather than that an archive is corrupt,
480 and it must not be swallowed by ``except ArchiveReadError`` handlers.
481 """
484class OpaqueArchiveMetadata(Protocol):
485 """Interface for opaque archive metadata.
487 In addition to implementing the methods defined here, all implementations
488 must be pickleable.
489 """
491 def copy(self) -> Self | None:
492 """Copy, reference, or discard metadata when its holding object is
493 copied.
494 """
495 ...
497 def subset(self, bbox: Box) -> Self | None:
498 """Copy, reference, or discard metadata when a subset of its its
499 holding object is extracted.
501 Parameters
502 ----------
503 bbox
504 Bounding box of the subset being extracted.
505 """
506 ...
509def no_header_updates(header: astropy.io.fits.Header) -> None:
510 """Do not make any modifications to the given FITS header.
512 Parameters
513 ----------
514 header
515 FITS header that is left unchanged.
516 """
519def _parse_major(version: object) -> int:
520 """Return the integer major component of a major.minor.patch string.
522 Accepts any object, because callers pass values that may have come
523 straight from a file; use `_parse_on_disk_major` for those.
525 Raises
526 ------
527 ArchiveReadError
528 If ``version`` is not a non-empty string of the form
529 ``major.minor.patch`` with integer components.
530 """
531 if not isinstance(version, str) or not version:
532 raise ArchiveReadError(f"Schema version {version!r} is not a non-empty string.")
533 head = version.split(".", 1)[0]
534 try:
535 return int(head)
536 except ValueError as exc:
537 raise ArchiveReadError(f"Schema version {version!r} has non-integer major.") from exc
540def _parse_on_disk_major(version: object) -> int:
541 """Return the integer major of a version stamp that came from a file.
543 Raises
544 ------
545 _MalformedSchemaVersionError
546 If ``version`` is not a non-empty string of the form
547 ``major.minor.patch`` with integer components. Unlike `_parse_major`,
548 this is a `ValueError`, so a stamp the reader cannot parse fails one
549 union candidate instead of aborting the union.
550 """
551 try:
552 return _parse_major(version)
553 except ArchiveReadError as exc:
554 raise _MalformedSchemaVersionError(str(exc)) from None
557def _check_compat(
558 name: str,
559 on_disk_version: str,
560 on_disk_min_read: int,
561 in_code_version: str,
562) -> None:
563 """Raise `ArchiveReadError` if a tree written with the given
564 schema_version/min_read_version cannot be read by the current code.
566 See :ref:`lsst.images-schema-versioning` for the compatibility rule.
567 """
568 in_code_major = _parse_major(in_code_version)
569 if on_disk_min_read > in_code_major:
570 raise ArchiveReadError(
571 f"{name}: tree requires reader major >= {on_disk_min_read}; this release is {in_code_version}."
572 )
575def _check_format_version(name: str, on_disk: int, in_code: int) -> None:
576 """Raise `ArchiveReadError` if a backend file's container layout
577 version is newer than this release knows how to read.
578 """
579 if on_disk > in_code:
580 raise ArchiveReadError(
581 f"{name}: on-disk container format version {on_disk} is "
582 f"newer than this release ({in_code}); cannot read."
583 )