Coverage for python/lsst/images/json/_input_archive.py: 70%
71 statements
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-19 10:11 +0000
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-19 10:11 +0000
1# This file is part of lsst-images.
2#
3# Developed for the LSST Data Management System.
4# This product includes software developed by the LSST Project
5# (https://www.lsst.org).
6# See the COPYRIGHT file at the top-level directory of this distribution
7# for details of code ownership.
8#
9# Use of this source code is governed by a 3-clause BSD-style
10# license that can be found in the LICENSE file.
12from __future__ import annotations
14__all__ = ("JsonInputArchive",)
16from collections.abc import Callable, Iterator
17from contextlib import contextmanager
18from types import EllipsisType
19from typing import IO, TYPE_CHECKING, Any, Self
21import astropy.table
22import numpy as np
23from pydantic_core import from_json
25from lsst.resources import ResourcePath, ResourcePathExpression
27from .._transforms import FrameSet
28from ..serialization import (
29 ArchiveInfo,
30 ArchiveReadError,
31 ArchiveTree,
32 ArrayReferenceModel,
33 InlineArrayModel,
34 InputArchive,
35 JsonRef,
36 TableModel,
37 no_header_updates,
38 parameterize_tree,
39 tree_class_for_info,
40)
41from ..serialization._backends import _is_binary_stream
42from ..serialization._common import _ARCHIVE_READ_CONTEXT
44if TYPE_CHECKING:
45 import astropy.io.fits
48class JsonInputArchive(InputArchive[JsonRef]):
49 """An implementation of the `.serialization.InputArchive` interface that
50 reads from JSON files.
52 Parameters
53 ----------
54 indirect
55 The `.serialization.ArchiveTree.indirect` attribute of the root
56 serialization model.
57 """
59 @classmethod
60 def get_basic_info(cls, path: ResourcePathExpression) -> ArchiveInfo:
61 """Read the top-level tree's ``schema_url``; JSON has no container
62 format version.
64 This parses the whole document. Unlike the FITS and NDF backends
65 there is no cheap header to read: ``schema_url`` is a computed field
66 serialized after the (potentially large) ``indirect`` payload, and
67 nested trees carry their own ``schema_url``, so a bounded prefix
68 cannot identify the top-level tree reliably. JSON is not intended
69 for large pixel archives, where FITS or NDF should be used instead.
71 Parameters
72 ----------
73 path
74 Path to the archive to read.
75 """
76 raw = from_json(ResourcePath(path).read())
77 if not isinstance(raw, dict) or not raw.get("schema_url"): 77 ↛ 78line 77 didn't jump to line 78 because the condition on line 77 was never true
78 raise ArchiveReadError(f"{path!r} has no schema_url in its top-level JSON tree.")
79 return ArchiveInfo.from_schema_url(raw["schema_url"], format_version=None)
81 @classmethod
82 @contextmanager
83 def open_tree(
84 cls,
85 path: ResourcePathExpression | IO[bytes],
86 *,
87 partial: bool = True,
88 **backend_kwargs: Any,
89 ) -> Iterator[tuple[Self, ArchiveTree, ArchiveInfo]]:
90 """Parse the JSON tree and yield ``(archive, tree, info)``.
92 Parameters
93 ----------
94 path
95 File resource to open, or a seekable binary stream containing
96 the file's content.
97 partial
98 Ignored. The entire JSON file is always read into memory.
99 **backend_kwargs
100 No keyword parameters are supported by this backend.
101 """
102 if _is_binary_stream(path):
103 raw = path.read()
104 else:
105 raw = ResourcePath(path).read()
106 parsed = from_json(raw)
107 if not isinstance(parsed, dict) or not parsed.get("schema_url"): 107 ↛ 108line 107 didn't jump to line 108 because the condition on line 107 was never true
108 raise ArchiveReadError(f"{path!r} has no schema_url in its top-level JSON tree.")
109 info = ArchiveInfo.from_schema_url(parsed["schema_url"], format_version=None)
110 tree_cls = tree_class_for_info(info, path)
111 parameterized = parameterize_tree(tree_cls, JsonRef)
112 tree = parameterized.model_validate_json(raw, context=_ARCHIVE_READ_CONTEXT)
113 archive = cls(tree.indirect)
114 try:
115 yield archive, tree, info
116 finally:
117 tree.indirect = []
119 def __init__(self, indirect: list[Any] | None = None) -> None:
120 self._indirect = indirect if indirect is not None else []
121 self._deserialized_pointer_cache: dict[int, Any] = {}
123 def deserialize_pointer[U: ArchiveTree, V](
124 self,
125 pointer: JsonRef,
126 model_type: type[U],
127 deserializer: Callable[[U, InputArchive[JsonRef]], V],
128 ) -> V:
129 index = int(pointer.ref.removeprefix("#/indirect/"))
130 if (existing := self._deserialized_pointer_cache.get(index)) is not None:
131 return existing
132 model = model_type.model_validate(self._indirect[index], context=_ARCHIVE_READ_CONTEXT)
133 result = deserializer(model, self)
134 self._deserialized_pointer_cache[index] = result
135 return result
137 def get_frame_set(self, ref: JsonRef) -> FrameSet:
138 index = int(ref.ref.removeprefix("#/indirect/"))
139 try:
140 result = self._deserialized_pointer_cache[index]
141 except KeyError:
142 raise AssertionError(
143 f"Frame set at {ref.model_dump_json(indent=2)} must be deserialized "
144 "before any dependent transform can be."
145 ) from None
146 if not isinstance(result, FrameSet):
147 raise ArchiveReadError(f"Expected a FrameSet instance at {ref.model_dump_json(indent=2)}.")
148 return result
150 def get_array(
151 self,
152 model: ArrayReferenceModel | InlineArrayModel,
153 *,
154 slices: tuple[slice, ...] | EllipsisType = ...,
155 strip_header: Callable[[astropy.io.fits.Header], None] = no_header_updates,
156 ) -> np.ndarray:
157 if not isinstance(model, InlineArrayModel): 157 ↛ 158line 157 didn't jump to line 158 because the condition on line 157 was never true
158 raise ArchiveReadError("Only inline arrays are supported in JSON archives.")
159 return np.array(model.data, dtype=model.datatype.to_numpy())[slices]
161 def get_table(
162 self,
163 model: TableModel,
164 strip_header: Callable[[astropy.io.fits.Header], None] = no_header_updates,
165 ) -> astropy.table.Table:
166 result = astropy.table.Table(meta=model.meta)
167 for column_model in model.columns:
168 if not isinstance(column_model.data, InlineArrayModel): 168 ↛ 169line 168 didn't jump to line 169 because the condition on line 168 was never true
169 raise ArchiveReadError("Only inline arrays are supported in JSON archives.")
170 result[column_model.name] = astropy.table.Column(
171 column_model.data.data,
172 name=column_model.name,
173 dtype=column_model.data.datatype.to_numpy(),
174 unit=column_model.unit,
175 description=column_model.description,
176 meta=column_model.meta,
177 )
178 return result
180 def get_structured_array(
181 self,
182 model: TableModel,
183 strip_header: Callable[[astropy.io.fits.Header], None] = no_header_updates,
184 ) -> np.ndarray:
185 table = self.get_table(model)
186 return table.as_array()