Coverage for python/lsst/images/serialization/_output_archive.py: 35%

67 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-08-20 02:04 -0700

1# This file is part of lsst-images. 

2# 

3# Developed for the LSST Data Management System. 

4# This product includes software developed by the LSST Project 

5# (https://www.lsst.org). 

6# See the COPYRIGHT file at the top-level directory of this distribution 

7# for details of code ownership. 

8# 

9# Use of this source code is governed by a 3-clause BSD-style 

10# license that can be found in the LICENSE file. 

11 

12from __future__ import annotations 

13 

14__all__ = ( 

15 "NestedOutputArchive", 

16 "OutputArchive", 

17) 

18 

19from abc import ABC, abstractmethod 

20from collections.abc import Callable, Hashable, Iterator, Mapping 

21from typing import TYPE_CHECKING, Any, TypeVar 

22 

23import astropy.io.fits 

24import astropy.table 

25import astropy.units 

26import numpy as np 

27import pydantic 

28 

29from ._asdf_utils import ArrayReferenceModel, InlineArrayModel 

30from ._common import ( 

31 ArchiveTree, 

32 ButlerInfo, 

33 MetadataValue, 

34 no_header_updates, 

35 warn_for_development_schemas, 

36) 

37from ._tables import TableModel 

38 

39if TYPE_CHECKING: 

40 from .._transforms import FrameSet 

41 

42# This pre-python-3.12 declaration is needed by Sphinx (probably the 

43# autodoc-typehints plugin. 

44P = TypeVar("P", bound=pydantic.BaseModel) 

45 

46 

47class OutputArchive[P](ABC): 

48 """Abstract interface for writing to a file format. 

49 

50 Notes 

51 ----- 

52 An output archive instance is assumed to be paired with a Pydantic model 

53 that represents a JSON tree, with the archive used to serialize data that 

54 is not natively JSON into data that is (which may just be a reference to 

55 binary data stored elsewhere in the file). The archive doesn't actually 

56 hold that model instance because we don't want to assume it can be built 

57 via default-initialization and assignment, and because we'd prefer to avoid 

58 making the output archive generic over the model type. It is expected that 

59 most concrete archive implementations will accept the paired model in some 

60 sort of finalization method in order to write it into the file, but this is 

61 not part of the base class interface. 

62 """ 

63 

64 def __init__(self) -> None: 

65 self._name_versions: dict[str, int] = {} 

66 """Per-name occurrence count, used by `_register_name` to disambiguate 

67 repeated logical names within a single write (e.g. each operand of a 

68 `SumField` calling ``add_array(name="data")`` from the same nested 

69 archive). 

70 """ 

71 

72 def serialize_root( 

73 self, 

74 obj: Any, 

75 metadata: dict[str, MetadataValue] | None = None, 

76 butler_info: ButlerInfo | None = None, 

77 ) -> ArchiveTree: 

78 """Serialize ``obj`` to a root tree, apply write-time overrides, and 

79 warn if the tree uses a development schema. 

80 

81 Every backend's top-level write funnel calls this so the metadata and 

82 butler overrides and the development-schema warning are applied 

83 uniformly, regardless of file format. 

84 

85 Parameters 

86 ---------- 

87 obj 

88 Object with a ``serialize`` method (and an optional 

89 ``_archive_default_name``) to serialize. 

90 metadata 

91 Extra metadata to merge into the tree, or `None`. 

92 butler_info 

93 Butler information to set on the tree, or `None`. 

94 

95 Returns 

96 ------- 

97 `~lsst.images.serialization.ArchiveTree` 

98 The serialized root tree. 

99 """ 

100 name = getattr(obj, "_archive_default_name", None) 

101 tree = self.serialize_direct(name, obj.serialize) if name is not None else obj.serialize(self) 

102 # Serializers generally transfer an object's metadata mapping by 

103 # reference. Detach it before applying write-only overrides so a 

104 # write never mutates the in-memory object. 

105 tree.metadata = tree.metadata.copy() 

106 if metadata is not None: 

107 tree.metadata.update(metadata) 

108 if butler_info is not None: 

109 tree.butler_info = butler_info 

110 warn_for_development_schemas(tree) 

111 return tree 

112 

113 def _register_name(self, name: str) -> tuple[str, int]: 

114 """Return the input name and its 1-based occurrence count. 

115 

116 Parameters 

117 ---------- 

118 name 

119 The logical archive name being saved (typically the absolute 

120 archive path of an array, table, or pointer target). 

121 

122 Returns 

123 ------- 

124 name : `str` 

125 The input name, returned unchanged so that the caller controls 

126 how the version is rendered into the on-disk identifier. 

127 version : `int` 

128 ``1`` the first time a given name is registered, then ``2``, 

129 ``3`` and so on for subsequent calls with the same name. 

130 

131 Notes 

132 ----- 

133 Backends should call this from `add_array`, `add_table`, 

134 `add_structured_array`, and `serialize_pointer` to detect repeated 

135 names; the registry lives on the root archive so that nested 

136 archives share a single namespace. Each backend chooses how to 

137 encode ``version > 1`` on disk: FITS uses the FITS ``EXTVER`` 

138 keyword without modifying the extension name, while hierarchical 

139 backends can append ``_{version}`` to the leaf component of the path. 

140 """ 

141 version = self._name_versions.get(name, 0) + 1 

142 self._name_versions[name] = version 

143 return name, version 

144 

145 @abstractmethod 

146 def serialize_direct[T: pydantic.BaseModel | None]( 

147 self, name: str, serializer: Callable[[OutputArchive], T] 

148 ) -> T: 

149 """Use a serializer function to save a nested object. 

150 

151 Parameters 

152 ---------- 

153 name 

154 Attribute of the paired Pydantic model that will be assigned the 

155 result of this call. If it will not be assigned to a direct 

156 attribute, it may be a JSON Pointer path (relative to the paired 

157 Pydantic model) to the location where it will be added. 

158 serializer 

159 Callable that takes an `~lsst.serialization.OutputArchive` and 

160 returns a Pydantic model. This will be passed a new 

161 `~lsst.serialization.OutputArchive` that automatically prepends 

162 ``{name}/`` (and any root path added by this archive) to names 

163 passed to it, so the ``serializer`` does not need to know where it 

164 appears in the overall tree. 

165 

166 Returns 

167 ------- 

168 T 

169 Result of the call to the serializer. 

170 """ 

171 raise NotImplementedError() 

172 

173 @abstractmethod 

174 def serialize_pointer[T: ArchiveTree]( 

175 self, name: str, serializer: Callable[[OutputArchive], T], key: Hashable 

176 ) -> T | P: 

177 """Use a serializer function to save a nested object that may be 

178 referenced in multiple locations in the same archive. 

179 

180 Parameters 

181 ---------- 

182 name 

183 Attribute of the paired Pydantic model that will be assigned the 

184 result of this call. If it will not be assigned to a direct 

185 attribute, it may be a JSON Pointer path (relative to the paired 

186 Pydantic model) to the location where it will be added. 

187 serializer 

188 Callable that takes an `~lsst.serialization.OutputArchive` and 

189 returns a Pydantic model. This will be passed a new 

190 `~lsst.serialization.OutputArchive` that automatically prepends 

191 ``{name}/`` (and any root path added by this archive) to names 

192 passed to it, so the ``serializer`` does not need to know where it 

193 appears in the overall tree. 

194 key 

195 A unique identifier for the in-memory object the serializer saves, 

196 e.g. a call to the built-in `id` function. 

197 

198 Returns 

199 ------- 

200 T | P 

201 Either the result of the call to the serializer, or a Pydantic 

202 model that can be considered a reference to it and added to a 

203 larger model in its place. 

204 """ 

205 # Since Pydantic doesn't provide us a good way to "dereference" a JSON 

206 # Pointer (i.e. traversing the tree to extract the original model), it 

207 # is probably easier to implement an `InputArchive` for the case where 

208 # the `~lsst.serialization.OutputArchive` opts to stuff all pointer 

209 # serializations into a standard location outside the user-controlled 

210 # Pydantic model tree, and always returned a JSON pointer to that 

211 # standard location from this function. 

212 raise NotImplementedError() 

213 

214 @abstractmethod 

215 def serialize_frame_set[T: ArchiveTree]( 

216 self, name: str, frame_set: FrameSet, serializer: Callable[[OutputArchive], T], key: Hashable 

217 ) -> T | P: 

218 """Serialize a frame set and make it available to objects saved later. 

219 

220 Parameters 

221 ---------- 

222 name 

223 Attribute of the paired Pydantic model that will be assigned the 

224 result of this call. If it will not be assigned to a direct 

225 attribute, it may be a JSON Pointer path (relative to the paired 

226 Pydantic model) to the location where it will be added. 

227 frame_set 

228 The frame set being saved. This will be returned in later calls 

229 to `iter_frame_sets`, along with the returned reference object. 

230 serializer 

231 Callable that takes an `~lsst.serialization.OutputArchive` and 

232 returns a Pydantic model. This will be passed a new 

233 `~lsst.serialization.OutputArchive` that automatically prepends 

234 ``{name}/`` (and any root path added by this archive) to names 

235 passed to it, so the ``serializer`` does not need to know where it 

236 appears in the overall tree. 

237 key 

238 A unique identifier for the in-memory object the serializer saves, 

239 e.g. a call to the built-in `id` function. 

240 

241 Returns 

242 ------- 

243 T | P 

244 Either the result of the call to the serializer, or a Pydantic 

245 model that can be considered a reference to it and added to a 

246 larger model in its place. 

247 """ 

248 raise NotImplementedError() 

249 

250 @abstractmethod 

251 def iter_frame_sets(self) -> Iterator[tuple[FrameSet, P]]: 

252 """Iterate over the frame sets already serialized to this archive. 

253 

254 Yields 

255 ------ 

256 frame_set 

257 A frame set that has already been written to this archive. 

258 reference 

259 An implementation-specific reference model that points to the 

260 frame set. 

261 """ 

262 raise NotImplementedError() 

263 

264 @abstractmethod 

265 def add_array( 

266 self, 

267 array: np.ndarray, 

268 *, 

269 name: str | None = None, 

270 update_header: Callable[[astropy.io.fits.Header], None] = no_header_updates, 

271 tile_shape: tuple[int, ...] | None = None, 

272 options_name: str | None = None, 

273 ) -> ArrayReferenceModel | InlineArrayModel: 

274 """Add an array to the archive. 

275 

276 Parameters 

277 ---------- 

278 array 

279 Array to save. 

280 name 

281 Name of the array. This should generally be the name of the 

282 Pydantic model attribute to which the result will be assigned. It 

283 may be left `None` if there is only one [structured] array or 

284 table in a nested object that is being saved. 

285 update_header 

286 A callback that will be given the FITS header for the HDU 

287 containing this array in order to add keys to it. This callback 

288 may be provided but will not be called if the output format is not 

289 FITS. 

290 tile_shape 

291 The recommended shape of each tile if the implementation will save 

292 the array in distinct tiles for faster subarray retrieval. 

293 This is a hint; implementations are not required to use this value. 

294 options_name 

295 Use the options (e.g. for compression) associated with this name 

296 when saving this array. 

297 

298 Returns 

299 ------- 

300 `~lsst.images.serialization.ArrayReferenceModel` |\ 

301 `~lsst.images.serialization.InlineArrayModel` 

302 A Pydantic model that references or holds the stored array. 

303 """ 

304 raise NotImplementedError() 

305 

306 @abstractmethod 

307 def add_table( 

308 self, 

309 table: astropy.table.Table, 

310 *, 

311 name: str | None = None, 

312 update_header: Callable[[astropy.io.fits.Header], None] = no_header_updates, 

313 ) -> TableModel: 

314 """Add a table to the archive. 

315 

316 Parameters 

317 ---------- 

318 table 

319 Table to save. 

320 name 

321 Name of the table. This should generally be the name of the 

322 Pydantic model attribute to which the result will be assigned. It 

323 may be left `None` if there is only one [structured] array or 

324 table in a nested object that is being saved. 

325 update_header 

326 A callback that will be given the FITS header for the HDU 

327 containing this table in order to add keys to it. This callback 

328 may be provided but will not be called if the output format is not 

329 FITS. 

330 

331 Returns 

332 ------- 

333 TableModel 

334 A Pydantic model that represents the table. 

335 """ 

336 raise NotImplementedError() 

337 

338 @abstractmethod 

339 def add_structured_array( 

340 self, 

341 array: np.ndarray, 

342 *, 

343 name: str | None = None, 

344 units: Mapping[str, astropy.units.Unit] | None = None, 

345 descriptions: Mapping[str, str] | None = None, 

346 update_header: Callable[[astropy.io.fits.Header], None] = no_header_updates, 

347 ) -> TableModel: 

348 """Add a table to the archive. 

349 

350 Parameters 

351 ---------- 

352 array 

353 A structured numpy array. 

354 name 

355 Name of the array. This should generally be the name of the 

356 Pydantic model attribute to which the result will be assigned. It 

357 may be left `None` if there is only one [structured] array or 

358 table in a nested object that is being saved. 

359 units 

360 A mapping of units for columns. Need not be complete. 

361 descriptions 

362 A mapping of descriptions for columns. Need not be complete. 

363 update_header 

364 A callback that will be given the FITS header for the HDU 

365 containing this table in order to add keys to it. This callback 

366 may be provided but will not be called if the output format is not 

367 FITS. 

368 

369 Returns 

370 ------- 

371 TableModel 

372 A Pydantic model that represents the table. 

373 """ 

374 raise NotImplementedError() 

375 

376 

377class NestedOutputArchive[P: pydantic.BaseModel](OutputArchive[P]): 

378 """A proxy output archive that joins a root path into all names before 

379 delegating back to its parent archive. 

380 

381 This is intended to be used in the implementation of most 

382 `~lsst.serialization.OutputArchive.serialize_direct` and 

383 `~lsst.serialization.OutputArchive.serialize_pointer` implementations. 

384 

385 Parameters 

386 ---------- 

387 root 

388 Root of all JSON Pointer paths. Should include a leading slash (as we 

389 always use absolute JSON Pointers) but no trailing slash. 

390 parent 

391 Parent output archive to delegate to. 

392 """ 

393 

394 def __init__(self, root: str, parent: OutputArchive) -> None: 

395 super().__init__() 

396 self._root = root 

397 self._parent = parent 

398 

399 def serialize_direct[T: pydantic.BaseModel | None]( 

400 self, name: str, serializer: Callable[[OutputArchive[P]], T] 

401 ) -> T: 

402 return self._parent.serialize_direct(self._join_path(name), serializer) 

403 

404 def serialize_pointer[T: ArchiveTree]( 

405 self, name: str, serializer: Callable[[OutputArchive[P]], T], key: Hashable 

406 ) -> T | P: 

407 return self._parent.serialize_pointer(self._join_path(name), serializer, key) 

408 

409 def serialize_frame_set[T: ArchiveTree]( 

410 self, name: str, frame_set: FrameSet, serializer: Callable[[OutputArchive], T], key: Hashable 

411 ) -> T | P: 

412 return self._parent.serialize_frame_set(self._join_path(name), frame_set, serializer, key) 

413 

414 def iter_frame_sets(self) -> Iterator[tuple[FrameSet, P]]: 

415 return self._parent.iter_frame_sets() 

416 

417 def add_array( 

418 self, 

419 array: np.ndarray, 

420 *, 

421 name: str | None = None, 

422 update_header: Callable[[astropy.io.fits.Header], None] = no_header_updates, 

423 tile_shape: tuple[int, ...] | None = None, 

424 options_name: str | None = None, 

425 ) -> ArrayReferenceModel | InlineArrayModel: 

426 return self._parent.add_array( 

427 array, 

428 name=self._join_path(name), 

429 update_header=update_header, 

430 tile_shape=tile_shape, 

431 options_name=options_name, 

432 ) 

433 

434 def add_table( 

435 self, 

436 table: astropy.table.Table, 

437 *, 

438 name: str | None = None, 

439 update_header: Callable[[astropy.io.fits.Header], None] = no_header_updates, 

440 ) -> TableModel: 

441 return self._parent.add_table(table, name=self._join_path(name), update_header=update_header) 

442 

443 def add_structured_array( 

444 self, 

445 array: np.ndarray, 

446 *, 

447 name: str | None = None, 

448 units: Mapping[str, astropy.units.Unit] | None = None, 

449 descriptions: Mapping[str, str] | None = None, 

450 update_header: Callable[[astropy.io.fits.Header], None] = no_header_updates, 

451 ) -> TableModel: 

452 return self._parent.add_structured_array( 

453 array, 

454 name=self._join_path(name), 

455 units=units, 

456 descriptions=descriptions, 

457 update_header=update_header, 

458 ) 

459 

460 def _join_path(self, name: str | None) -> str: 

461 return f"{self._root}/{name}" if name is not None else self._root