Coverage for python/lsst/daf/butler/tests/_testRepo.py: 97%

146 statements  

« prev     ^ index     » next       coverage.py v7.16.0, created at 2026-09-19 09:05 +0000

1# This file is part of daf_butler. 

2# 

3# Developed for the LSST Data Management System. 

4# This product includes software developed by the LSST Project 

5# (http://www.lsst.org). 

6# See the COPYRIGHT file at the top-level directory of this distribution 

7# for details of code ownership. 

8# 

9# This software is dual licensed under the GNU General Public License and also 

10# under a 3-clause BSD license. Recipients may choose which of these licenses 

11# to use; please see the files gpl-3.0.txt and/or bsd_license.txt, 

12# respectively. If you choose the GPL option then the following text applies 

13# (but note that there is still no warranty even if you opt for BSD instead): 

14# 

15# This program is free software: you can redistribute it and/or modify 

16# it under the terms of the GNU General Public License as published by 

17# the Free Software Foundation, either version 3 of the License, or 

18# (at your option) any later version. 

19# 

20# This program is distributed in the hope that it will be useful, 

21# but WITHOUT ANY WARRANTY; without even the implied warranty of 

22# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the 

23# GNU General Public License for more details. 

24# 

25# You should have received a copy of the GNU General Public License 

26# along with this program. If not, see <http://www.gnu.org/licenses/>. 

27 

28from __future__ import annotations 

29 

30__all__ = [ 

31 "DatastoreMock", 

32 "addDataIdValue", 

33 "addDatasetType", 

34 "expandUniqueId", 

35 "makeTestCollection", 

36 "makeTestRepo", 

37] 

38 

39import random 

40from collections.abc import Iterable, Mapping 

41from typing import TYPE_CHECKING, Any 

42from unittest.mock import MagicMock 

43 

44import sqlalchemy 

45 

46from lsst.daf.butler import ( 

47 Butler, 

48 Config, 

49 DataCoordinate, 

50 DatasetRef, 

51 DatasetType, 

52 Dimension, 

53 DimensionUniverse, 

54 FileDataset, 

55 StorageClass, 

56) 

57 

58from ._repo_template_cache import make_repo_for_test 

59 

60if TYPE_CHECKING: 

61 from lsst.daf.butler import DatasetId 

62 

63 

64def makeTestRepo( 

65 root: str, dataIds: Mapping[str, Iterable] | None = None, *, config: Config | None = None, **kwargs: Any 

66) -> Butler: 

67 """Create an empty test repository. 

68 

69 Parameters 

70 ---------- 

71 root : `str` 

72 The location of the root directory for the repository. 

73 dataIds : `~collections.abc.Mapping` \ 

74 [`str`, `~collections.abc.Iterable`], optional 

75 A mapping keyed by the dimensions used in the test. Each value is an 

76 iterable of names for that dimension (e.g., detector IDs for 

77 ``"detector"``). Related dimensions (e.g., instruments and detectors) 

78 are linked arbitrarily, with values created for implied dimensions only 

79 when needed. This parameter is provided for compatibility with old 

80 code; newer code should make the repository, then call 

81 `~lsst.daf.butler.tests.addDataIdValue`. 

82 config : `lsst.daf.butler.Config`, optional 

83 A configuration for the repository (for details, see 

84 `lsst.daf.butler.Butler.makeRepo`). If omitted, creates a repository 

85 with default dataset and storage types, but optimized for speed. The 

86 defaults set ``.datastore.cls``, ``.datastore.checksum`` and 

87 ``.registry.db``. If a supplied config does not specify these values 

88 the internal defaults will be used to ensure that we have a usable 

89 configuration. 

90 **kwargs 

91 Extra arguments to `lsst.daf.butler.Butler.makeRepo`. 

92 

93 Returns 

94 ------- 

95 butler : `lsst.daf.butler.Butler` 

96 A Butler referring to the new repository. This Butler is provided only 

97 for additional setup; to keep test cases isolated, it is highly 

98 recommended that each test create its own Butler with a unique 

99 run/collection. See `makeTestCollection`. 

100 

101 Notes 

102 ----- 

103 This function provides a "quick and dirty" repository for simple unit tests 

104 that don't depend on complex data relationships. It is ill-suited for tests 

105 where the structure of the data matters. If you need such a dataset, create 

106 it directly or use a saved test dataset. 

107 """ 

108 defaults = Config() 

109 defaults["datastore", "cls"] = "lsst.daf.butler.datastores.inMemoryDatastore.InMemoryDatastore" 

110 defaults["datastore", "checksum"] = False # In case of future changes 

111 defaults["registry", "db"] = "sqlite:///<butlerRoot>/gen3.sqlite3" 

112 

113 if config: 

114 defaults.update(config) 

115 

116 if not dataIds: 

117 dataIds = {} 

118 

119 # Disable config root by default so that our registry override will 

120 # not be ignored. 

121 # newConfig guards against location-related keywords like outfile 

122 newConfig = make_repo_for_test(root, config=defaults, forceConfigRoot=False, **kwargs) 

123 butler = Butler.from_config(newConfig, writeable=True) 

124 dimensionRecords = _makeRecords(dataIds, butler.dimensions) 

125 for dimension, records in dimensionRecords.items(): 

126 if butler.dimensions[dimension].has_own_table: 

127 butler.registry.insertDimensionData(dimension, *records) 

128 return butler 

129 

130 

131def makeTestCollection(repo: Butler, uniqueId: str | None = None) -> Butler: 

132 """Create a read/write Butler to a fresh collection. 

133 

134 Parameters 

135 ---------- 

136 repo : `lsst.daf.butler.Butler` 

137 A previously existing Butler to a repository, such as that returned by 

138 `~lsst.daf.butler.Butler.makeRepo` or `makeTestRepo`. 

139 uniqueId : `str`, optional 

140 A collection ID guaranteed by external code to be unique across all 

141 calls to ``makeTestCollection`` for the same repository. 

142 

143 Returns 

144 ------- 

145 butler : `lsst.daf.butler.Butler` 

146 A Butler referring to a new collection in the repository at ``root``. 

147 The collection is (almost) guaranteed to be new. 

148 

149 Notes 

150 ----- 

151 This function creates a single run collection that does not necessarily 

152 conform to any repository conventions. It is only suitable for creating an 

153 isolated test area, and not for repositories intended for real data 

154 processing or analysis. 

155 """ 

156 if not uniqueId: 

157 # Create a "random" collection name 

158 # Speed matters more than cryptographic guarantees 

159 uniqueId = str(random.randrange(1_000_000_000)) 

160 collection = "test_" + uniqueId 

161 return Butler.from_config(butler=repo, run=collection) 

162 

163 

164def _makeRecords(dataIds: Mapping[str, Iterable], universe: DimensionUniverse) -> Mapping[str, Iterable]: 

165 """Create cross-linked dimension records from a collection of 

166 data ID values. 

167 

168 Parameters 

169 ---------- 

170 dataIds : `~collections.abc.Mapping` [`str`, `~collections.abc.Iterable`] 

171 A mapping keyed by the dimensions of interest. Each value is an 

172 iterable of names for that dimension (e.g., detector IDs for 

173 ``"detector"``). 

174 universe : lsst.daf.butler.DimensionUniverse 

175 Set of all known dimensions and their relationships. 

176 

177 Returns 

178 ------- 

179 dataIds : `~collections.abc.Mapping` [`str`, `~collections.abc.Iterable`] 

180 A mapping keyed by the dimensions of interest, giving one 

181 `~lsst.daf.butler.DimensionRecord` for each input name. Related 

182 dimensions (e.g., instruments and detectors) are linked arbitrarily. 

183 """ 

184 # Create values for all dimensions that are (recursive) required or implied 

185 # dependencies of the given ones. 

186 complete_data_id_values = {} 

187 for dimension_name in universe.conform(dataIds.keys()).names: 

188 if dimension_name in dataIds: 

189 complete_data_id_values[dimension_name] = list(dataIds[dimension_name]) 

190 if dimension_name not in complete_data_id_values: 

191 complete_data_id_values[dimension_name] = [ 

192 _makeRandomDataIdValue(universe.dimensions[dimension_name]) 

193 ] 

194 

195 # Start populating dicts that will become DimensionRecords by providing 

196 # alternate keys like detector names 

197 record_dicts_by_dimension_name: dict[str, list[dict[str, str | int | bytes]]] = {} 

198 for name, values in complete_data_id_values.items(): 

199 record_dicts_by_dimension_name[name] = [] 

200 dimension_el = universe[name] 

201 for value in values: 

202 # _fillAllKeys wants Dimension and not DimensionElement. 

203 # universe.__getitem__ says it returns DimensionElement but this 

204 # really does also seem to be a Dimension here. 

205 record_dicts_by_dimension_name[name].append( 

206 _fillAllKeys(dimension_el, value) # type: ignore[arg-type] 

207 ) 

208 

209 # Pick cross-relationships arbitrarily 

210 for name, record_dicts in record_dicts_by_dimension_name.items(): 

211 dimension_el = universe[name] 

212 for record_dict in record_dicts: 

213 for other in dimension_el.dimensions: 

214 if other != dimension_el: 

215 relation = record_dicts_by_dimension_name[other.name][0] 

216 record_dict[other.name] = relation[other.primaryKey.name] 

217 

218 return { 

219 dimension: [universe[dimension].RecordClass(**record_dict) for record_dict in record_dicts] 

220 for dimension, record_dicts in record_dicts_by_dimension_name.items() 

221 } 

222 

223 

224def _fillAllKeys(dimension: Dimension, value: str | int) -> dict[str, str | int | bytes]: 

225 """Create an arbitrary mapping of all required keys for a given dimension 

226 that do not refer to other dimensions. 

227 

228 Parameters 

229 ---------- 

230 dimension : `lsst.daf.butler.Dimension` 

231 The dimension for which to generate a set of keys (e.g., detector). 

232 value 

233 The value assigned to ``dimension`` (e.g., detector ID). 

234 

235 Returns 

236 ------- 

237 expandedValue : `dict` [`str`] 

238 A mapping of dimension keys to values. ``dimension's`` primary key 

239 maps to ``value``, but all other mappings (e.g., detector name) 

240 are arbitrary. 

241 """ 

242 expandedValue: dict[str, str | int | bytes] = {} 

243 for key in dimension.uniqueKeys: 

244 if key.nbytes: 244 ↛ 253line 244 didn't jump to line 253 because the condition on line 244 was never true

245 # For `bytes` fields, we want something that casts at least `str` 

246 # and `int` values to bytes and yields b'' when called with no 

247 # arguments (as in the except block below). Unfortunately, the 

248 # `bytes` type itself fails for both `str` and `int`, but this 

249 # lambda does what we need. This particularly important for the 

250 # skymap dimensions' bytes 'hash' field, which has a unique 

251 # constraint; without this, all skymaps would get a hash of b'' 

252 # and end up conflicting. 

253 castType = lambda *args: str(*args).encode() # noqa: E731 

254 else: 

255 castType = key.dtype().python_type 

256 try: 

257 castValue = castType(value) 

258 except TypeError: 

259 castValue = castType() 

260 expandedValue[key.name] = castValue 

261 for key in dimension.metadata: 

262 if not key.nullable: 262 ↛ 263line 262 didn't jump to line 263 because the condition on line 262 was never true

263 expandedValue[key.name] = key.dtype().python_type(value) 

264 return expandedValue 

265 

266 

267def _makeRandomDataIdValue(dimension: Dimension) -> int | str: 

268 """Generate a random value of the appropriate type for a data ID key. 

269 

270 Parameters 

271 ---------- 

272 dimension : `Dimension` 

273 Dimension the value corresponds to. 

274 

275 Returns 

276 ------- 

277 value : `int` or `str` 

278 Random value. 

279 """ 

280 if dimension.primaryKey.getPythonType() is str: 

281 return str(random.randrange(1000)) 

282 else: 

283 return random.randrange(1000) 

284 

285 

286def expandUniqueId(butler: Butler, partialId: Mapping[str, Any]) -> DataCoordinate: 

287 """Return a complete data ID matching some criterion. 

288 

289 Parameters 

290 ---------- 

291 butler : `lsst.daf.butler.Butler` 

292 The repository to query. 

293 partialId : `~collections.abc.Mapping` [`str`] 

294 A mapping of known dimensions and values. 

295 

296 Returns 

297 ------- 

298 dataId : `lsst.daf.butler.DataCoordinate` 

299 The unique data ID that matches ``partialId``. 

300 

301 Raises 

302 ------ 

303 ValueError 

304 Raised if ``partialId`` does not uniquely identify a data ID. 

305 

306 Notes 

307 ----- 

308 This method will only work correctly if all dimensions attached to the 

309 target dimension (eg., "physical_filter" for "visit") are known to the 

310 repository, even if they're not needed to identify a dataset. This function 

311 is only suitable for certain kinds of test repositories, and not for 

312 repositories intended for real data processing or analysis. 

313 

314 Examples 

315 -------- 

316 .. code-block:: py 

317 

318 >>> butler = makeTestRepo( 

319 "testdir", {"instrument": ["notACam"], "detector": [1]}) 

320 >>> expandUniqueId(butler, {"detector": 1}) 

321 DataCoordinate({instrument, detector}, ('notACam', 1)) 

322 """ 

323 # The example is *not* a doctest because it requires dangerous I/O 

324 registry = butler.registry 

325 dimensions = registry.dimensions.conform(partialId.keys()).required 

326 

327 query = " AND ".join(f"{dimension} = {value!r}" for dimension, value in partialId.items()) 

328 

329 # Much of the purpose of this function is to do something we explicitly 

330 # reject most of the time: query for a governor dimension (e.g. instrument) 

331 # given something that depends on it (e.g. visit), hence check=False. 

332 dataId = list(registry.queryDataIds(dimensions, where=query, check=False)) 

333 if len(dataId) == 1: 

334 return dataId[0] 

335 else: 

336 raise ValueError(f"Found {len(dataId)} matches for {partialId}, expected 1.") 

337 

338 

339def _findOrInventDataIdValue( 

340 butler: Butler, data_id: dict[str, str | int], dimension: Dimension 

341) -> tuple[str | int, bool]: 

342 """Look up an arbitrary value for a dimension that is consistent with a 

343 partial data ID that does not specify that dimension, or invent one if no 

344 such value exists. 

345 

346 Parameters 

347 ---------- 

348 butler : `Butler` 

349 Butler to use to look up data ID values. 

350 data_id : `dict` [ `str`, `str` or `int` ] 

351 Dictionary of possibly-related data ID values. 

352 dimension : `Dimension` 

353 Dimension to obtain a value for. 

354 

355 Returns 

356 ------- 

357 value : `int` or `str` 

358 Value for this dimension. 

359 invented : `bool` 

360 `True` if the value had to be invented, `False` if a compatible value 

361 already existed. 

362 """ 

363 # No values given by caller for this dimension. See if any exist 

364 # in the registry that are consistent with the values of dimensions 

365 # we do have: 

366 match_data_id = {key: data_id[key] for key in data_id.keys() & dimension.dimensions.names} 

367 matches = list(butler.registry.queryDimensionRecords(dimension, dataId=match_data_id).limit(1)) 

368 if not matches: 

369 # Nothing in the registry matches: invent a data ID value 

370 # with the right type (actual value does not matter). 

371 # We may or may not actually make a record with this; that's 

372 # easier to check later. 

373 dimension_value = _makeRandomDataIdValue(dimension) 

374 return dimension_value, True 

375 else: 

376 # A record does exist in the registry. Use its data ID value. 

377 dim_value = matches[0].dataId[dimension.name] 

378 assert dim_value is not None 

379 return dim_value, False 

380 

381 

382def _makeDimensionRecordDict(data_id: dict[str, str | int], dimension: Dimension) -> dict[str, Any]: 

383 """Create a dictionary that can be used to build a `DimensionRecord` that 

384 is consistent with the given data ID. 

385 

386 Parameters 

387 ---------- 

388 data_id : `dict` [ `str`, `str` or `int` ] 

389 Dictionary that contains values for at least all of 

390 ``dimension.dimensions.names`` (the main dimension, its recursive 

391 required dependencies, and its non-recursive implied dependencies). 

392 dimension : `Dimension` 

393 Dimension to build a record dictionary for. 

394 

395 Returns 

396 ------- 

397 record_dict : `dict` [ `str`, `object` ] 

398 Dictionary that can be passed as ``**kwargs`` to this dimensions 

399 record class constructor. 

400 """ 

401 # Add the primary key field for this dimension. 

402 record_dict: dict[str, Any] = {dimension.primaryKey.name: data_id[dimension.name]} 

403 # Define secondary keys (e.g., detector name given detector id) 

404 record_dict.update(_fillAllKeys(dimension, data_id[dimension.name])) 

405 # Set the foreign key values for any related dimensions that should 

406 # appear in the record. 

407 for related_dimension in dimension.dimensions: 

408 if related_dimension.name != dimension.name: 

409 record_dict[related_dimension.name] = data_id[related_dimension.name] 

410 return record_dict 

411 

412 

413def addDataIdValue(butler: Butler, dimension: str, value: str | int, **related: str | int) -> None: 

414 """Add the records that back a new data ID to a repository. 

415 

416 Parameters 

417 ---------- 

418 butler : `lsst.daf.butler.Butler` 

419 The repository to update. 

420 dimension : `str` 

421 The name of the dimension to gain a new value. 

422 value : `str` or `int` 

423 The value to register for the dimension. 

424 **related : `typing.Any` 

425 Any existing dimensions to be linked to ``value``. 

426 

427 Notes 

428 ----- 

429 Related dimensions (e.g., the instrument associated with a detector) may be 

430 specified using ``related``, which requires a value for those dimensions to 

431 have been added to the repository already (generally with a previous call 

432 to `addDataIdValue`. Any dependencies of the given dimension that are not 

433 included in ``related`` will be linked to existing values arbitrarily, and 

434 (for implied dependencies only) created and also inserted into the registry 

435 if they do not exist. Values for required dimensions and those given in 

436 ``related`` are never created. 

437 

438 Because this function creates filler data, it is only suitable for test 

439 repositories. It should not be used for repositories intended for real data 

440 processing or analysis, which have known dimension values. 

441 

442 Examples 

443 -------- 

444 See the guide on :ref:`using-butler-in-tests-make-repo` for usage examples. 

445 """ 

446 # Example is not doctest, because it's probably unsafe to create even an 

447 # in-memory butler in that environment. 

448 try: 

449 full_dimension = butler.dimensions[dimension] 

450 except KeyError as e: 

451 raise ValueError from e 

452 # Bad keys ignored by registry code 

453 extra_keys = related.keys() - full_dimension.minimal_group.names 

454 if extra_keys: 

455 raise ValueError( 

456 f"Unexpected keywords {extra_keys} not found in {full_dimension.minimal_group.names}" 

457 ) 

458 

459 # Assemble a dictionary data ID holding the given primary dimension value 

460 # and all of the related ones. 

461 data_id: dict[str, int | str] = {dimension: value} 

462 data_id.update(related) 

463 

464 # Compute the set of all dimensions that these recursively depend on. 

465 all_dimensions = butler.dimensions.conform(data_id.keys()) 

466 

467 # Create dicts that will become DimensionRecords for all of these data IDs. 

468 # This iteration is guaranteed to be in topological order, so we can count 

469 # on new data ID values being invented before they are needed. 

470 record_dicts_by_dimension: dict[Dimension, dict[str, Any]] = {} 

471 for dimension_name in all_dimensions.names: 

472 dimension_obj = butler.dimensions.dimensions[dimension_name] 

473 dimension_value = data_id.get(dimension_name) 

474 if dimension_value is None: 

475 data_id[dimension_name], invented = _findOrInventDataIdValue(butler, data_id, dimension_obj) 

476 if not invented: 

477 # No need to make a new record; one already exists. 

478 continue 

479 if dimension_name in related: 

480 # Caller passed in a value of this dimension explicitly, but it 

481 # isn't the primary dimension they asked to have a record created 

482 # for. That means they expect this record to already exist. 

483 continue 

484 if dimension_name != dimension and dimension_name in all_dimensions.required: 

485 # We also don't want to automatically create new dimension records 

486 # for required dimensions (except for the main dimension the caller 

487 # asked for); those are also asserted by the caller to already 

488 # exist. 

489 continue 

490 if not dimension_obj.has_own_table: 490 ↛ 493line 490 didn't jump to line 493 because the condition on line 490 was never true

491 # Don't need to bother generating full records for dimensions whose 

492 # records are not actually stored. 

493 continue 

494 record_dicts_by_dimension[dimension_obj] = _makeDimensionRecordDict(data_id, dimension_obj) 

495 

496 # Sync those dimension record dictionaries with the database. 

497 for dimension_obj, record_dict in record_dicts_by_dimension.items(): 

498 record = dimension_obj.RecordClass(**record_dict) 

499 try: 

500 butler.registry.syncDimensionData(dimension_obj, record) 

501 except sqlalchemy.exc.IntegrityError as e: 

502 raise RuntimeError( 

503 "Could not create data ID value. Automatic relationship generation " 

504 "may have failed; try adding keywords to assign a specific instrument, " 

505 "physical_filter, etc. based on the nested exception message." 

506 ) from e 

507 

508 

509def addDatasetType(butler: Butler, name: str, dimensions: set[str], storageClass: str) -> DatasetType: 

510 """Add a new dataset type to a repository. 

511 

512 Parameters 

513 ---------- 

514 butler : `lsst.daf.butler.Butler` 

515 The repository to update. 

516 name : `str` 

517 The name of the dataset type. 

518 dimensions : `set` [`str`] 

519 The dimensions of the new dataset type. 

520 storageClass : `str` 

521 The storage class the dataset will use. 

522 

523 Returns 

524 ------- 

525 datasetType : `lsst.daf.butler.DatasetType` 

526 The new type. 

527 

528 Raises 

529 ------ 

530 ValueError 

531 Raised if the dimensions or storage class is invalid. 

532 

533 Notes 

534 ----- 

535 Dataset types are shared across all collections in a repository, so this 

536 function does not need to be run for each collection. 

537 """ 

538 try: 

539 datasetType = DatasetType(name, dimensions, storageClass, universe=butler.dimensions) 

540 butler.registry.registerDatasetType(datasetType) 

541 return datasetType 

542 except KeyError as e: 

543 raise ValueError from e 

544 

545 

546class DatastoreMock: 

547 """Mocks a butler datastore. 

548 

549 Has functions that mock the datastore in a butler. Provides an `apply` 

550 function to replace the relevent butler datastore functions with the mock 

551 functions. 

552 """ 

553 

554 @staticmethod 

555 def apply(butler: Butler) -> None: 

556 """Apply datastore mocks to a butler. 

557 

558 Parameters 

559 ---------- 

560 butler : `~lsst.daf.butler.Butler` 

561 Butler to be modified. 

562 """ 

563 butler._datastore.export = DatastoreMock._mock_export # type: ignore 

564 butler._datastore.get = DatastoreMock._mock_get # type: ignore 

565 butler._datastore.ingest = MagicMock() # type: ignore 

566 

567 @staticmethod 

568 def _mock_export( 

569 refs: Iterable[DatasetRef], *, directory: str | None = None, transfer: str | None = None 

570 ) -> Iterable[FileDataset]: 

571 """Mock of `Datastore.export` that satisfies the requirement that 

572 the refs passed in are included in the `FileDataset` objects 

573 returned. 

574 

575 This can be used to construct a `Datastore` mock that can be used 

576 in repository export via:: 

577 

578 datastore = unittest.mock.Mock(spec=Datastore) 

579 datastore.export = DatastoreMock._mock_export 

580 

581 """ 

582 for ref in refs: 

583 yield FileDataset( 

584 refs=[ref], path="mock/path", formatter="lsst.daf.butler.formatters.json.JsonFormatter" 

585 ) 

586 

587 @staticmethod 

588 def _mock_get( 

589 ref: DatasetRef, 

590 parameters: Mapping[str, Any] | None = None, 

591 storageClass: StorageClass | str | None = None, 

592 ) -> tuple[DatasetId, Mapping[str, Any] | None]: 

593 """Mock of `Datastore.get` that just returns the integer dataset ID 

594 value and parameters it was given. 

595 """ 

596 return (ref.id, parameters)