Coverage for python/lsst/images/serialization/_common.py: 61%
170 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-09-30 11:30 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-09-30 11:30 +0000
1# This file is part of lsst-images.
2#
3# Developed for the LSST Data Management System.
4# This product includes software developed by the LSST Project
5# (https://www.lsst.org).
6# See the COPYRIGHT file at the top-level directory of this distribution
7# for details of code ownership.
8#
9# Use of this source code is governed by a 3-clause BSD-style
10# license that can be found in the LICENSE file.
12from __future__ import annotations
14__all__ = (
15 "SCHEMA_URL_BASE",
16 "ArchiveAccessRequiredError",
17 "ArchiveReadError",
18 "ArchiveTree",
19 "ButlerInfo",
20 "DevelopmentSchemaWarning",
21 "InvalidComponentError",
22 "InvalidParameterError",
23 "JsonRef",
24 "MetadataValue",
25 "OpaqueArchiveMetadata",
26 "is_development_version",
27 "no_header_updates",
28 "warn_for_development_schemas",
29)
31import operator
32import warnings
33from abc import ABC, abstractmethod
34from collections.abc import Iterator
35from copy import deepcopy
36from typing import TYPE_CHECKING, Any, ClassVar, Protocol, Self
38import astropy.table
39import astropy.units
40import pydantic
41from packaging.version import Version
43from .._geom import Box
44from ..utils import is_none
45from ._external_metadata import ExternalMetadata
46from ._migrations import _MIGRATABLE_NAMES, _MIGRATIONS
48try:
49 from lsst.daf.butler import DatasetProvenance, SerializedDatasetRef
50except ImportError:
51 type DatasetProvenance = Any # type: ignore[no-redef]
52 type SerializedDatasetRef = Any # type: ignore[no-redef]
54if TYPE_CHECKING:
55 import astropy.io.fits
57 from ._input_archive import InputArchive
60type MetadataValue = (
61 pydantic.StrictInt | pydantic.StrictFloat | pydantic.StrictStr | pydantic.StrictBool | None
62)
64SCHEMA_URL_BASE = "https://images.lsst.io/schemas"
65"""Base for the schema URLs of this package's own schemas, used as
66``{SCHEMA_URL_BASE}/{name}-{version}``.
68External packages providing their own schemas override
69`~lsst.images.serialization.ArchiveTree.SCHEMA_URL_BASE` instead, so their
70schema URLs are minted under a site they control.
71"""
73_ARCHIVE_READ_CONTEXT = object()
74"""Pydantic context used when validating a tree read from an archive."""
77class ButlerInfo(pydantic.BaseModel):
78 """Information about a butler dataset."""
80 dataset: SerializedDatasetRef
81 provenance: DatasetProvenance = pydantic.Field(default_factory=DatasetProvenance)
84class JsonRef(pydantic.BaseModel, serialize_by_alias=True):
85 """Pydantic model for JSON Reference / Pointer (IETF RFC 6901).
87 Notes
88 -----
89 This model does not do any of the escaping or special-character
90 interpretation required by the spec; it assumes that's already been done,
91 so its job is *just* putting a ``$ref`` field inside another model.
92 """
94 ref: str = pydantic.Field(alias="$ref")
97class ArchiveTree(
98 pydantic.BaseModel, ABC, ser_json_inf_nan="constants", ser_json_bytes="base64", val_json_bytes="base64"
99):
100 """An intermediate base class of `pydantic.BaseModel` that should be used
101 for all objects that may be used as the top-level tree models written to
102 archives.
104 See :ref:`lsst.images-schema-versioning` for how the ``SCHEMA_NAME`` /
105 ``SCHEMA_VERSION`` / ``MIN_READ_VERSION`` constants and the
106 ``schema_version`` / ``min_read_version`` / ``schema_url`` fields are used.
107 """
109 SCHEMA_NAME: ClassVar[str]
110 SCHEMA_VERSION: ClassVar[str]
111 MIN_READ_VERSION: ClassVar[int]
112 SCHEMA_URL_BASE: ClassVar[str] = SCHEMA_URL_BASE
113 """Base for this schema's URL, as ``{SCHEMA_URL_BASE}/{name}-{version}``.
115 External packages providing their own schemas should override this (once,
116 on a shared intermediate base class) so their schema URLs are minted
117 under a documentation site they control rather than images.lsst.io."""
119 PUBLIC_TYPE: ClassVar[type]
120 """In-memory Python type produced by this tree's ``deserialize`` (e.g.
121 `dict` for a mapping return). Declared explicitly by each concrete
122 subclass and surfaced by
123 `~lsst.images.serialization.public_type_for_schema`."""
125 # These mirror the SCHEMA_VERSION / MIN_READ_VERSION class constants and
126 # are never passed on construction, so repr omits them; they say nothing
127 # about the object that its type does not already say.
128 schema_version: str = pydantic.Field(
129 default="1.0.0",
130 description="Data-model schema version of this tree (major.minor.patch).",
131 repr=False,
132 )
133 min_read_version: int = pydantic.Field(
134 default=1,
135 description="Smallest reader major that can interpret this tree.",
136 repr=False,
137 )
138 metadata: dict[str, MetadataValue] = pydantic.Field(
139 default_factory=dict, description="Additional unstructured metadata.", exclude_if=operator.not_
140 )
141 butler_info: ButlerInfo | None = pydantic.Field(
142 default=None,
143 description="Information about the butler dataset backed by this file.",
144 exclude_if=is_none,
145 )
146 # Populated by the archive machinery rather than by a caller, so repr
147 # omits it as well.
148 indirect: list[Any] = pydantic.Field(
149 default_factory=list,
150 description="Serialized nested objects that may be saved or read more than once.",
151 exclude_if=operator.not_,
152 repr=False,
153 )
155 @pydantic.computed_field(description="Canonical schema URL for this tree.") # type: ignore[prop-decorator]
156 @property
157 def schema_url(self) -> str:
158 """Return the schema URL of this tree's class.
160 Computed from ``SCHEMA_NAME`` and ``SCHEMA_VERSION`` ClassVars.
161 """
162 cls = type(self)
163 return f"{cls.SCHEMA_URL_BASE}/{cls.SCHEMA_NAME}-{cls.SCHEMA_VERSION}"
165 @pydantic.model_validator(mode="before")
166 @classmethod
167 def _migrate_from_older_major(cls, data: Any, info: pydantic.ValidationInfo) -> Any:
168 """Morph an older on-disk tree into the current shape.
170 Chains registered migrations until the tree reaches the in-code major,
171 so the ``mode="after"`` validator below sees a current-shape tree.
173 A tree validated in the archive-read context, which the backends mark
174 via ``info.context``, must carry a ``schema_version``; in-memory
175 construction is unmarked and keeps the class-constant defaults. See
176 :ref:`lsst.images-schema-versioning-migration`.
177 """
178 if not isinstance(data, dict):
179 return data
180 if not hasattr(cls, "SCHEMA_NAME"): 180 ↛ 185line 180 didn't jump to line 185 because the condition on line 180 was never true
181 # ArchiveTree itself is abstract, and a subclass that has not yet
182 # declared SCHEMA_NAME (during incremental rollout) has no
183 # SCHEMA_VERSION to parse either; skip cleanly rather than
184 # raising AttributeError, mirroring the after-validator below.
185 return data
186 name = cls.SCHEMA_NAME
187 if "schema_version" not in data:
188 if info.context is _ARCHIVE_READ_CONTEXT:
189 raise _MissingSchemaVersionError(
190 f"{name}: archive tree has no schema_version; unstamped archive data is not supported."
191 )
192 return data
193 if not _MIGRATABLE_NAMES: 193 ↛ 194line 193 didn't jump to line 194 because the condition on line 193 was never true
194 return data
195 if name not in _MIGRATABLE_NAMES:
196 # This schema has not opted into migrations at all, regardless of
197 # its major: whether some *other*, unrelated schema has a
198 # registered migration must not change what this one does with
199 # its own tree. A schema in this state falls through to ordinary
200 # validation -- in-model backfill for an additive change, or a
201 # Pydantic validation error for a real incompatibility.
202 return data
203 on_disk_major = _parse_on_disk_major(data["schema_version"])
204 in_code_major = _parse_major(cls.SCHEMA_VERSION)
205 if on_disk_major < in_code_major:
206 # A failed union candidate must not mutate the input subsequently
207 # tried by another candidate. Migrations may mutate freely, so
208 # isolate the complete JSON-like tree, not just its top-level
209 # dictionary.
210 data = deepcopy(data)
211 while on_disk_major < in_code_major:
212 try:
213 step = _MIGRATIONS[(name, on_disk_major)]
214 except KeyError:
215 raise _MigrationGapError(
216 f"{name}: no migration from major {on_disk_major} to {on_disk_major + 1}."
217 ) from None
218 data = step(data)
219 on_disk_major += 1
220 data["schema_version"] = f"{on_disk_major}.0.0"
221 return data
223 @pydantic.model_validator(mode="after")
224 def _check_and_normalize_schema_version(self) -> Self:
225 """Validate and normalise the schema version fields.
227 Compares the on-tree ``schema_version`` / ``min_read_version`` against
228 the in-code values from the subclass's ClassVars; raises if
229 incompatible, otherwise normalises the fields to the in-code values.
230 """
231 cls = type(self)
232 # ArchiveTree itself is abstract (deserialize is @abstractmethod).
233 # Subclasses that haven't yet declared SCHEMA_NAME are skipped — this
234 # matters during incremental rollout and remains a safe no-op
235 # afterwards (a class-invariants test ensures every concrete subclass
236 # has the constants).
237 if not hasattr(cls, "SCHEMA_NAME"): 237 ↛ 238line 237 didn't jump to line 238 because the condition on line 237 was never true
238 return self
239 _check_compat(
240 cls.SCHEMA_NAME,
241 self.schema_version,
242 self.min_read_version,
243 cls.SCHEMA_VERSION,
244 )
245 if self.schema_version != cls.SCHEMA_VERSION:
246 self.schema_version = cls.SCHEMA_VERSION
247 if self.min_read_version != cls.MIN_READ_VERSION:
248 self.min_read_version = cls.MIN_READ_VERSION
249 return self
251 @classmethod
252 def __pydantic_init_subclass__(cls, **kwargs: Any) -> None:
253 """Inject ``$id`` and ``title`` into the subclass's JSON Schema, and
254 register the subclass in the schema-name registry.
256 Populates ``model_config['json_schema_extra']`` with values derived
257 from the subclass's ``SCHEMA_NAME`` / ``SCHEMA_VERSION`` ClassVars,
258 then registers the subclass so it can be looked up by schema name.
259 Subclasses that haven't declared the ClassVars are skipped.
260 """
261 super().__pydantic_init_subclass__(**kwargs)
262 name = cls.__dict__.get("SCHEMA_NAME")
263 version = cls.__dict__.get("SCHEMA_VERSION")
264 if name is None or version is None:
265 return
266 json_schema_extra = cls.model_config.get("json_schema_extra") or {}
267 if isinstance(json_schema_extra, dict): 267 ↛ 279line 267 didn't jump to line 279 because the condition on line 267 was always true
268 existing = dict(json_schema_extra)
269 # Always override: a subclass of a concrete schema (e.g.
270 # visit_image subclassing masked_image) inherits its parent's
271 # already-stamped values through the merged model_config, and
272 # this hook only runs when the subclass declares its own
273 # SCHEMA_NAME / SCHEMA_VERSION for these to be derived from.
274 existing["$id"] = f"{cls.SCHEMA_URL_BASE}/{name}-{version}"
275 existing["title"] = name
276 cls.model_config = {**cls.model_config, "json_schema_extra": existing}
277 # Local import to avoid the _io -> _common circular dependency at
278 # module load time.
279 from ._io import register_schema_class
281 register_schema_class(cls)
283 @abstractmethod
284 def deserialize(self, archive: InputArchive[Any], **kwargs: Any) -> Any:
285 """Return the in-memory object that was serialized to this tree.
287 Parameters
288 ----------
289 archive
290 The input archive to read from.
291 **kwargs
292 Additional keyword arguments specific to this type.
294 Raises
295 ------
296 ~lsst.images.serialization.InvalidParameterError
297 Raised for unsupported ``**kwargs``.
299 Notes
300 -----
301 Subclass implementations may take additional keyword-only arguments.
302 Callers that invoke this method without knowing what those might be
303 should catch `TypeError` and re-raise as
304 `~lsst.images.serialization.InvalidParameterError` if they pass
305 additional keyword arguments.
306 """
307 raise NotImplementedError()
309 def deserialize_component(self, component: str, archive: InputArchive[Any], **kwargs: Any) -> Any:
310 """Return a component in-memory object that was serialized to this
311 tree.
313 Parameters
314 ----------
315 component
316 Name of the component to read.
317 archive
318 The input archive to read from.
319 **kwargs
320 Additional keyword arguments specific to this type.
322 Raises
323 ------
324 ~lsst.images.serialization.InvalidComponentError
325 Raise if ``component`` is not recognized.
326 ~lsst.images.serialization.InvalidParameterError
327 Raised for unsupported ``**kwargs``.
329 Notes
330 -----
331 The default implementation for this method tries to get an attribute
332 with the component's name from ``self``, and then:
334 - returns `None` if it is `None`;
335 - calls `deserialize` on that object if it is also an
336 `~lsst.images.serialization.ArchiveTree`;
337 - returns it directly otherwise.
339 If there is no such attribute, it raises
340 `~lsst.images.serialization.InvalidComponentError`.
342 ``**kwargs`` are forwarded to component `deserialize` methods, but
343 are otherwise not checked. Subclasses are generally expected to
344 implement this method to do that checking and handle any components
345 for which the other will not work, and then delegate to `super` at
346 the end.
347 """
348 try:
349 component_model = getattr(self, component)
350 except AttributeError:
351 raise InvalidComponentError(
352 f"Component {component!r} is not recognized by {type(self).__name__}."
353 ) from None
354 if component_model is None:
355 return None
356 if isinstance(component_model, ArchiveTree):
357 return component_model.deserialize(archive, **kwargs)
358 return component_model
361class DevelopmentSchemaWarning(UserWarning):
362 """Warning that a file is being written with a development schema."""
365def is_development_version(version: str) -> bool:
366 """Return whether a schema version string is a PEP 440 development release.
368 Parameters
369 ----------
370 version
371 Schema version string, e.g. ``1.0.0`` or ``1.0.0.dev0``.
372 """
373 return Version(version).is_devrelease
376def _iter_archive_trees(obj: Any) -> Iterator[ArchiveTree]:
377 """Yield every `ArchiveTree` embedded in a serialized tree."""
378 if isinstance(obj, ArchiveTree):
379 yield obj
380 if isinstance(obj, pydantic.BaseModel):
381 for field_name in type(obj).model_fields:
382 yield from _iter_archive_trees(getattr(obj, field_name))
383 elif isinstance(obj, list | tuple):
384 for item in obj:
385 yield from _iter_archive_trees(item)
386 elif isinstance(obj, dict):
387 for value in obj.values():
388 yield from _iter_archive_trees(value)
391def warn_for_development_schemas(root: ArchiveTree) -> None:
392 """Emit a `DevelopmentSchemaWarning` if a serialized tree contains any
393 schema still in development.
395 Parameters
396 ----------
397 root
398 Top-level serialized tree about to be written.
399 """
400 developing = sorted(
401 {
402 tree.schema_url
403 for tree in _iter_archive_trees(root)
404 if is_development_version(type(tree).SCHEMA_VERSION)
405 }
406 )
407 if developing:
408 warnings.warn(
409 "Writing a file with development schema(s) "
410 f"{', '.join(developing)}; such files are not for production and "
411 "may not remain readable.",
412 DevelopmentSchemaWarning,
413 stacklevel=3,
414 )
417class ArchiveReadError(RuntimeError):
418 """Exception raised when the contents of an archive cannot be read."""
421class InvalidParameterError(ArchiveReadError):
422 """Exception raised by `ArchiveTree.deserialize` or
423 `ArchiveTree.deserialize_component` when passed an invalid keyword
424 argument.
425 """
428class InvalidComponentError(ArchiveReadError):
429 """Exception `ArchiveTree.deserialize_component` when passed an invalid
430 component name.
431 """
434class _MigrationGapError(ArchiveReadError, ValueError):
435 """A migration chain has no registered step for the tree's on-disk major.
437 Raised from the ``mode="before"`` validator in
438 `ArchiveTree._migrate_from_older_major`, which pydantic-core invokes as
439 part of validating a field. Deriving only from `ArchiveReadError` (a
440 `RuntimeError`) would let this exception escape a discriminated union
441 entirely: pydantic only treats `ValueError`, `TypeError` and
442 `AssertionError` raised by a validator as "this candidate failed, try the
443 next one", so a `RuntimeError` would abort the whole union instead of
444 falling through. Deriving from both means a ``left_to_right`` (or
445 ``smart``) union tries the next variant on a migration gap exactly as it
446 would on any other validation failure, while ``except ArchiveReadError``
447 callers outside a union still catch it unchanged.
448 """
451class _MissingSchemaVersionError(ArchiveReadError, ValueError):
452 """An archive tree omitted its required ``schema_version`` stamp."""
455class _MalformedSchemaVersionError(ArchiveReadError, ValueError):
456 """An archive tree's ``schema_version`` stamp could not be parsed.
458 Also a `ValueError`, for the reason `_MigrationGapError` gives: the stamp
459 is parsed in the ``mode="before"`` validator, which runs ahead of field
460 validation for every migratable union candidate, including ones the
461 payload does not belong to. A stamp the reader cannot parse therefore has
462 to fail that one candidate rather than abort the whole union.
464 Only an *on-disk* stamp raises this. A malformed in-code
465 ``SCHEMA_VERSION`` is a programming error and keeps raising the
466 `RuntimeError`-only `ArchiveReadError`, so a union cannot swallow it.
467 """
470class ArchiveAccessRequiredError(RuntimeError):
471 """Exception raised when a deserialization needs data from the file.
473 Raised by all data-access methods of
474 `~lsst.images.serialization.DetachedArchive`, signaling that the
475 requested object cannot be deserialized from the JSON tree alone.
477 Notes
478 -----
479 This deliberately does not inherit from `ArchiveReadError`: it signals
480 that a live archive is required rather than that an archive is corrupt,
481 and it must not be swallowed by ``except ArchiveReadError`` handlers.
482 """
485class OpaqueArchiveMetadata(Protocol):
486 """Interface for opaque archive metadata.
488 In addition to implementing the methods defined here, all implementations
489 must be pickleable.
490 """
492 def copy(self) -> Self | None:
493 """Copy, reference, or discard metadata when its holding object is
494 copied.
495 """
496 ...
498 def subset(self, bbox: Box) -> Self | None:
499 """Copy, reference, or discard metadata when a subset of its its
500 holding object is extracted.
502 Parameters
503 ----------
504 bbox
505 Bounding box of the subset being extracted.
506 """
507 ...
509 def external_metadata(self) -> ExternalMetadata:
510 """Return a read-only view of the metadata carried by this object
511 that is not part of the data model.
512 """
513 ...
516def no_header_updates(header: astropy.io.fits.Header) -> None:
517 """Do not make any modifications to the given FITS header.
519 Parameters
520 ----------
521 header
522 FITS header that is left unchanged.
523 """
526def _parse_major(version: object) -> int:
527 """Return the integer major component of a major.minor.patch string.
529 Accepts any object, because callers pass values that may have come
530 straight from a file; use `_parse_on_disk_major` for those.
532 Raises
533 ------
534 ArchiveReadError
535 If ``version`` is not a non-empty string of the form
536 ``major.minor.patch`` with integer components.
537 """
538 if not isinstance(version, str) or not version:
539 raise ArchiveReadError(f"Schema version {version!r} is not a non-empty string.")
540 head = version.split(".", 1)[0]
541 try:
542 return int(head)
543 except ValueError as exc:
544 raise ArchiveReadError(f"Schema version {version!r} has non-integer major.") from exc
547def _parse_on_disk_major(version: object) -> int:
548 """Return the integer major of a version stamp that came from a file.
550 Raises
551 ------
552 _MalformedSchemaVersionError
553 If ``version`` is not a non-empty string of the form
554 ``major.minor.patch`` with integer components. Unlike `_parse_major`,
555 this is a `ValueError`, so a stamp the reader cannot parse fails one
556 union candidate instead of aborting the union.
557 """
558 try:
559 return _parse_major(version)
560 except ArchiveReadError as exc:
561 raise _MalformedSchemaVersionError(str(exc)) from None
564def _check_compat(
565 name: str,
566 on_disk_version: str,
567 on_disk_min_read: int,
568 in_code_version: str,
569) -> None:
570 """Raise `ArchiveReadError` if a tree written with the given
571 schema_version/min_read_version cannot be read by the current code.
573 See :ref:`lsst.images-schema-versioning` for the compatibility rule.
574 """
575 in_code_major = _parse_major(in_code_version)
576 if on_disk_min_read > in_code_major:
577 raise ArchiveReadError(
578 f"{name}: tree requires reader major >= {on_disk_min_read}; this release is {in_code_version}."
579 )
582def _check_format_version(name: str, on_disk: int, in_code: int) -> None:
583 """Raise `ArchiveReadError` if a backend file's container layout
584 version is newer than this release knows how to read.
585 """
586 if on_disk > in_code:
587 raise ArchiveReadError(
588 f"{name}: on-disk container format version {on_disk} is "
589 f"newer than this release ({in_code}); cannot read."
590 )