Coverage for python/lsst/images/json/_input_archive.py: 70%

71 statements  

« prev     ^ index     » next       coverage.py v7.16.0, created at 2026-09-06 09:22 +0000

1# This file is part of lsst-images. 

2# 

3# Developed for the LSST Data Management System. 

4# This product includes software developed by the LSST Project 

5# (https://www.lsst.org). 

6# See the COPYRIGHT file at the top-level directory of this distribution 

7# for details of code ownership. 

8# 

9# Use of this source code is governed by a 3-clause BSD-style 

10# license that can be found in the LICENSE file. 

11 

12from __future__ import annotations 

13 

14__all__ = ("JsonInputArchive",) 

15 

16from collections.abc import Callable, Iterator 

17from contextlib import contextmanager 

18from types import EllipsisType 

19from typing import IO, TYPE_CHECKING, Any, Self 

20 

21import astropy.table 

22import numpy as np 

23from pydantic_core import from_json 

24 

25from lsst.resources import ResourcePath, ResourcePathExpression 

26 

27from .._transforms import FrameSet 

28from ..serialization import ( 

29 ArchiveInfo, 

30 ArchiveReadError, 

31 ArchiveTree, 

32 ArrayReferenceModel, 

33 InlineArrayModel, 

34 InputArchive, 

35 JsonRef, 

36 TableModel, 

37 no_header_updates, 

38 parameterize_tree, 

39 tree_class_for_info, 

40) 

41from ..serialization._backends import _is_binary_stream 

42from ..serialization._common import _ARCHIVE_READ_CONTEXT 

43 

44if TYPE_CHECKING: 

45 import astropy.io.fits 

46 

47 

48class JsonInputArchive(InputArchive[JsonRef]): 

49 """An implementation of the `.serialization.InputArchive` interface that 

50 reads from JSON files. 

51 

52 Parameters 

53 ---------- 

54 indirect 

55 The `.serialization.ArchiveTree.indirect` attribute of the root 

56 serialization model. 

57 """ 

58 

59 @classmethod 

60 def get_basic_info(cls, path: ResourcePathExpression) -> ArchiveInfo: 

61 """Read the top-level tree's ``schema_url``; JSON has no container 

62 format version. 

63 

64 This parses the whole document. Unlike the FITS and NDF backends 

65 there is no cheap header to read: ``schema_url`` is a computed field 

66 serialized after the (potentially large) ``indirect`` payload, and 

67 nested trees carry their own ``schema_url``, so a bounded prefix 

68 cannot identify the top-level tree reliably. JSON is not intended 

69 for large pixel archives, where FITS or NDF should be used instead. 

70 

71 Parameters 

72 ---------- 

73 path 

74 Path to the archive to read. 

75 """ 

76 raw = from_json(ResourcePath(path).read()) 

77 if not isinstance(raw, dict) or not raw.get("schema_url"): 77 ↛ 78line 77 didn't jump to line 78 because the condition on line 77 was never true

78 raise ArchiveReadError(f"{path!r} has no schema_url in its top-level JSON tree.") 

79 return ArchiveInfo.from_schema_url(raw["schema_url"], format_version=None) 

80 

81 @classmethod 

82 @contextmanager 

83 def open_tree( 

84 cls, 

85 path: ResourcePathExpression | IO[bytes], 

86 *, 

87 partial: bool = True, 

88 **backend_kwargs: Any, 

89 ) -> Iterator[tuple[Self, ArchiveTree, ArchiveInfo]]: 

90 """Parse the JSON tree and yield ``(archive, tree, info)``. 

91 

92 Parameters 

93 ---------- 

94 path 

95 File resource to open, or a seekable binary stream containing 

96 the file's content. 

97 partial 

98 Ignored. The entire JSON file is always read into memory. 

99 **backend_kwargs 

100 No keyword parameters are supported by this backend. 

101 """ 

102 if _is_binary_stream(path): 

103 raw = path.read() 

104 else: 

105 raw = ResourcePath(path).read() 

106 parsed = from_json(raw) 

107 if not isinstance(parsed, dict) or not parsed.get("schema_url"): 107 ↛ 108line 107 didn't jump to line 108 because the condition on line 107 was never true

108 raise ArchiveReadError(f"{path!r} has no schema_url in its top-level JSON tree.") 

109 info = ArchiveInfo.from_schema_url(parsed["schema_url"], format_version=None) 

110 tree_cls = tree_class_for_info(info, path) 

111 parameterized = parameterize_tree(tree_cls, JsonRef) 

112 tree = parameterized.model_validate_json(raw, context=_ARCHIVE_READ_CONTEXT) 

113 archive = cls(tree.indirect) 

114 try: 

115 yield archive, tree, info 

116 finally: 

117 tree.indirect = [] 

118 

119 def __init__(self, indirect: list[Any] | None = None) -> None: 

120 self._indirect = indirect if indirect is not None else [] 

121 self._deserialized_pointer_cache: dict[int, Any] = {} 

122 

123 def deserialize_pointer[U: ArchiveTree, V]( 

124 self, 

125 pointer: JsonRef, 

126 model_type: type[U], 

127 deserializer: Callable[[U, InputArchive[JsonRef]], V], 

128 ) -> V: 

129 index = int(pointer.ref.removeprefix("#/indirect/")) 

130 if (existing := self._deserialized_pointer_cache.get(index)) is not None: 

131 return existing 

132 model = model_type.model_validate(self._indirect[index], context=_ARCHIVE_READ_CONTEXT) 

133 result = deserializer(model, self) 

134 self._deserialized_pointer_cache[index] = result 

135 return result 

136 

137 def get_frame_set(self, ref: JsonRef) -> FrameSet: 

138 index = int(ref.ref.removeprefix("#/indirect/")) 

139 try: 

140 result = self._deserialized_pointer_cache[index] 

141 except KeyError: 

142 raise AssertionError( 

143 f"Frame set at {ref.model_dump_json(indent=2)} must be deserialized " 

144 "before any dependent transform can be." 

145 ) from None 

146 if not isinstance(result, FrameSet): 

147 raise ArchiveReadError(f"Expected a FrameSet instance at {ref.model_dump_json(indent=2)}.") 

148 return result 

149 

150 def get_array( 

151 self, 

152 model: ArrayReferenceModel | InlineArrayModel, 

153 *, 

154 slices: tuple[slice, ...] | EllipsisType = ..., 

155 strip_header: Callable[[astropy.io.fits.Header], None] = no_header_updates, 

156 ) -> np.ndarray: 

157 if not isinstance(model, InlineArrayModel): 157 ↛ 158line 157 didn't jump to line 158 because the condition on line 157 was never true

158 raise ArchiveReadError("Only inline arrays are supported in JSON archives.") 

159 return np.array(model.data, dtype=model.datatype.to_numpy())[slices] 

160 

161 def get_table( 

162 self, 

163 model: TableModel, 

164 strip_header: Callable[[astropy.io.fits.Header], None] = no_header_updates, 

165 ) -> astropy.table.Table: 

166 result = astropy.table.Table(meta=model.meta) 

167 for column_model in model.columns: 

168 if not isinstance(column_model.data, InlineArrayModel): 168 ↛ 169line 168 didn't jump to line 169 because the condition on line 168 was never true

169 raise ArchiveReadError("Only inline arrays are supported in JSON archives.") 

170 result[column_model.name] = astropy.table.Column( 

171 column_model.data.data, 

172 name=column_model.name, 

173 dtype=column_model.data.datatype.to_numpy(), 

174 unit=column_model.unit, 

175 description=column_model.description, 

176 meta=column_model.meta, 

177 ) 

178 return result 

179 

180 def get_structured_array( 

181 self, 

182 model: TableModel, 

183 strip_header: Callable[[astropy.io.fits.Header], None] = no_header_updates, 

184 ) -> np.ndarray: 

185 table = self.get_table(model) 

186 return table.as_array()