Coverage for python/lsst/daf/butler/tests/_testRepo.py: 97%
146 statements
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-02 09:06 +0000
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-02 09:06 +0000
1# This file is part of daf_butler.
2#
3# Developed for the LSST Data Management System.
4# This product includes software developed by the LSST Project
5# (http://www.lsst.org).
6# See the COPYRIGHT file at the top-level directory of this distribution
7# for details of code ownership.
8#
9# This software is dual licensed under the GNU General Public License and also
10# under a 3-clause BSD license. Recipients may choose which of these licenses
11# to use; please see the files gpl-3.0.txt and/or bsd_license.txt,
12# respectively. If you choose the GPL option then the following text applies
13# (but note that there is still no warranty even if you opt for BSD instead):
14#
15# This program is free software: you can redistribute it and/or modify
16# it under the terms of the GNU General Public License as published by
17# the Free Software Foundation, either version 3 of the License, or
18# (at your option) any later version.
19#
20# This program is distributed in the hope that it will be useful,
21# but WITHOUT ANY WARRANTY; without even the implied warranty of
22# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
23# GNU General Public License for more details.
24#
25# You should have received a copy of the GNU General Public License
26# along with this program. If not, see <http://www.gnu.org/licenses/>.
28from __future__ import annotations
30__all__ = [
31 "DatastoreMock",
32 "addDataIdValue",
33 "addDatasetType",
34 "expandUniqueId",
35 "makeTestCollection",
36 "makeTestRepo",
37]
39import random
40from collections.abc import Iterable, Mapping
41from typing import TYPE_CHECKING, Any
42from unittest.mock import MagicMock
44import sqlalchemy
46from lsst.daf.butler import (
47 Butler,
48 Config,
49 DataCoordinate,
50 DatasetRef,
51 DatasetType,
52 Dimension,
53 DimensionUniverse,
54 FileDataset,
55 StorageClass,
56)
58from ._repo_template_cache import make_repo_for_test
60if TYPE_CHECKING:
61 from lsst.daf.butler import DatasetId
64def makeTestRepo(
65 root: str, dataIds: Mapping[str, Iterable] | None = None, *, config: Config | None = None, **kwargs: Any
66) -> Butler:
67 """Create an empty test repository.
69 Parameters
70 ----------
71 root : `str`
72 The location of the root directory for the repository.
73 dataIds : `~collections.abc.Mapping` \
74 [`str`, `~collections.abc.Iterable`], optional
75 A mapping keyed by the dimensions used in the test. Each value is an
76 iterable of names for that dimension (e.g., detector IDs for
77 ``"detector"``). Related dimensions (e.g., instruments and detectors)
78 are linked arbitrarily, with values created for implied dimensions only
79 when needed. This parameter is provided for compatibility with old
80 code; newer code should make the repository, then call
81 `~lsst.daf.butler.tests.addDataIdValue`.
82 config : `lsst.daf.butler.Config`, optional
83 A configuration for the repository (for details, see
84 `lsst.daf.butler.Butler.makeRepo`). If omitted, creates a repository
85 with default dataset and storage types, but optimized for speed. The
86 defaults set ``.datastore.cls``, ``.datastore.checksum`` and
87 ``.registry.db``. If a supplied config does not specify these values
88 the internal defaults will be used to ensure that we have a usable
89 configuration.
90 **kwargs
91 Extra arguments to `lsst.daf.butler.Butler.makeRepo`.
93 Returns
94 -------
95 butler : `lsst.daf.butler.Butler`
96 A Butler referring to the new repository. This Butler is provided only
97 for additional setup; to keep test cases isolated, it is highly
98 recommended that each test create its own Butler with a unique
99 run/collection. See `makeTestCollection`.
101 Notes
102 -----
103 This function provides a "quick and dirty" repository for simple unit tests
104 that don't depend on complex data relationships. It is ill-suited for tests
105 where the structure of the data matters. If you need such a dataset, create
106 it directly or use a saved test dataset.
107 """
108 defaults = Config()
109 defaults["datastore", "cls"] = "lsst.daf.butler.datastores.inMemoryDatastore.InMemoryDatastore"
110 defaults["datastore", "checksum"] = False # In case of future changes
111 defaults["registry", "db"] = "sqlite:///<butlerRoot>/gen3.sqlite3"
113 if config:
114 defaults.update(config)
116 if not dataIds:
117 dataIds = {}
119 # Disable config root by default so that our registry override will
120 # not be ignored.
121 # newConfig guards against location-related keywords like outfile
122 newConfig = make_repo_for_test(root, config=defaults, forceConfigRoot=False, **kwargs)
123 butler = Butler.from_config(newConfig, writeable=True)
124 dimensionRecords = _makeRecords(dataIds, butler.dimensions)
125 for dimension, records in dimensionRecords.items():
126 if butler.dimensions[dimension].has_own_table:
127 butler.registry.insertDimensionData(dimension, *records)
128 return butler
131def makeTestCollection(repo: Butler, uniqueId: str | None = None) -> Butler:
132 """Create a read/write Butler to a fresh collection.
134 Parameters
135 ----------
136 repo : `lsst.daf.butler.Butler`
137 A previously existing Butler to a repository, such as that returned by
138 `~lsst.daf.butler.Butler.makeRepo` or `makeTestRepo`.
139 uniqueId : `str`, optional
140 A collection ID guaranteed by external code to be unique across all
141 calls to ``makeTestCollection`` for the same repository.
143 Returns
144 -------
145 butler : `lsst.daf.butler.Butler`
146 A Butler referring to a new collection in the repository at ``root``.
147 The collection is (almost) guaranteed to be new.
149 Notes
150 -----
151 This function creates a single run collection that does not necessarily
152 conform to any repository conventions. It is only suitable for creating an
153 isolated test area, and not for repositories intended for real data
154 processing or analysis.
155 """
156 if not uniqueId:
157 # Create a "random" collection name
158 # Speed matters more than cryptographic guarantees
159 uniqueId = str(random.randrange(1_000_000_000))
160 collection = "test_" + uniqueId
161 return Butler.from_config(butler=repo, run=collection)
164def _makeRecords(dataIds: Mapping[str, Iterable], universe: DimensionUniverse) -> Mapping[str, Iterable]:
165 """Create cross-linked dimension records from a collection of
166 data ID values.
168 Parameters
169 ----------
170 dataIds : `~collections.abc.Mapping` [`str`, `~collections.abc.Iterable`]
171 A mapping keyed by the dimensions of interest. Each value is an
172 iterable of names for that dimension (e.g., detector IDs for
173 ``"detector"``).
174 universe : lsst.daf.butler.DimensionUniverse
175 Set of all known dimensions and their relationships.
177 Returns
178 -------
179 dataIds : `~collections.abc.Mapping` [`str`, `~collections.abc.Iterable`]
180 A mapping keyed by the dimensions of interest, giving one
181 `~lsst.daf.butler.DimensionRecord` for each input name. Related
182 dimensions (e.g., instruments and detectors) are linked arbitrarily.
183 """
184 # Create values for all dimensions that are (recursive) required or implied
185 # dependencies of the given ones.
186 complete_data_id_values = {}
187 for dimension_name in universe.conform(dataIds.keys()).names:
188 if dimension_name in dataIds:
189 complete_data_id_values[dimension_name] = list(dataIds[dimension_name])
190 if dimension_name not in complete_data_id_values:
191 complete_data_id_values[dimension_name] = [
192 _makeRandomDataIdValue(universe.dimensions[dimension_name])
193 ]
195 # Start populating dicts that will become DimensionRecords by providing
196 # alternate keys like detector names
197 record_dicts_by_dimension_name: dict[str, list[dict[str, str | int | bytes]]] = {}
198 for name, values in complete_data_id_values.items():
199 record_dicts_by_dimension_name[name] = []
200 dimension_el = universe[name]
201 for value in values:
202 # _fillAllKeys wants Dimension and not DimensionElement.
203 # universe.__getitem__ says it returns DimensionElement but this
204 # really does also seem to be a Dimension here.
205 record_dicts_by_dimension_name[name].append(
206 _fillAllKeys(dimension_el, value) # type: ignore[arg-type]
207 )
209 # Pick cross-relationships arbitrarily
210 for name, record_dicts in record_dicts_by_dimension_name.items():
211 dimension_el = universe[name]
212 for record_dict in record_dicts:
213 for other in dimension_el.dimensions:
214 if other != dimension_el:
215 relation = record_dicts_by_dimension_name[other.name][0]
216 record_dict[other.name] = relation[other.primaryKey.name]
218 return {
219 dimension: [universe[dimension].RecordClass(**record_dict) for record_dict in record_dicts]
220 for dimension, record_dicts in record_dicts_by_dimension_name.items()
221 }
224def _fillAllKeys(dimension: Dimension, value: str | int) -> dict[str, str | int | bytes]:
225 """Create an arbitrary mapping of all required keys for a given dimension
226 that do not refer to other dimensions.
228 Parameters
229 ----------
230 dimension : `lsst.daf.butler.Dimension`
231 The dimension for which to generate a set of keys (e.g., detector).
232 value
233 The value assigned to ``dimension`` (e.g., detector ID).
235 Returns
236 -------
237 expandedValue : `dict` [`str`]
238 A mapping of dimension keys to values. ``dimension's`` primary key
239 maps to ``value``, but all other mappings (e.g., detector name)
240 are arbitrary.
241 """
242 expandedValue: dict[str, str | int | bytes] = {}
243 for key in dimension.uniqueKeys:
244 if key.nbytes: 244 ↛ 253line 244 didn't jump to line 253 because the condition on line 244 was never true
245 # For `bytes` fields, we want something that casts at least `str`
246 # and `int` values to bytes and yields b'' when called with no
247 # arguments (as in the except block below). Unfortunately, the
248 # `bytes` type itself fails for both `str` and `int`, but this
249 # lambda does what we need. This particularly important for the
250 # skymap dimensions' bytes 'hash' field, which has a unique
251 # constraint; without this, all skymaps would get a hash of b''
252 # and end up conflicting.
253 castType = lambda *args: str(*args).encode() # noqa: E731
254 else:
255 castType = key.dtype().python_type
256 try:
257 castValue = castType(value)
258 except TypeError:
259 castValue = castType()
260 expandedValue[key.name] = castValue
261 for key in dimension.metadata:
262 if not key.nullable: 262 ↛ 263line 262 didn't jump to line 263 because the condition on line 262 was never true
263 expandedValue[key.name] = key.dtype().python_type(value)
264 return expandedValue
267def _makeRandomDataIdValue(dimension: Dimension) -> int | str:
268 """Generate a random value of the appropriate type for a data ID key.
270 Parameters
271 ----------
272 dimension : `Dimension`
273 Dimension the value corresponds to.
275 Returns
276 -------
277 value : `int` or `str`
278 Random value.
279 """
280 if dimension.primaryKey.getPythonType() is str:
281 return str(random.randrange(1000))
282 else:
283 return random.randrange(1000)
286def expandUniqueId(butler: Butler, partialId: Mapping[str, Any]) -> DataCoordinate:
287 """Return a complete data ID matching some criterion.
289 Parameters
290 ----------
291 butler : `lsst.daf.butler.Butler`
292 The repository to query.
293 partialId : `~collections.abc.Mapping` [`str`]
294 A mapping of known dimensions and values.
296 Returns
297 -------
298 dataId : `lsst.daf.butler.DataCoordinate`
299 The unique data ID that matches ``partialId``.
301 Raises
302 ------
303 ValueError
304 Raised if ``partialId`` does not uniquely identify a data ID.
306 Notes
307 -----
308 This method will only work correctly if all dimensions attached to the
309 target dimension (eg., "physical_filter" for "visit") are known to the
310 repository, even if they're not needed to identify a dataset. This function
311 is only suitable for certain kinds of test repositories, and not for
312 repositories intended for real data processing or analysis.
314 Examples
315 --------
316 .. code-block:: py
318 >>> butler = makeTestRepo(
319 "testdir", {"instrument": ["notACam"], "detector": [1]})
320 >>> expandUniqueId(butler, {"detector": 1})
321 DataCoordinate({instrument, detector}, ('notACam', 1))
322 """
323 # The example is *not* a doctest because it requires dangerous I/O
324 registry = butler.registry
325 dimensions = registry.dimensions.conform(partialId.keys()).required
327 query = " AND ".join(f"{dimension} = {value!r}" for dimension, value in partialId.items())
329 # Much of the purpose of this function is to do something we explicitly
330 # reject most of the time: query for a governor dimension (e.g. instrument)
331 # given something that depends on it (e.g. visit), hence check=False.
332 dataId = list(registry.queryDataIds(dimensions, where=query, check=False))
333 if len(dataId) == 1:
334 return dataId[0]
335 else:
336 raise ValueError(f"Found {len(dataId)} matches for {partialId}, expected 1.")
339def _findOrInventDataIdValue(
340 butler: Butler, data_id: dict[str, str | int], dimension: Dimension
341) -> tuple[str | int, bool]:
342 """Look up an arbitrary value for a dimension that is consistent with a
343 partial data ID that does not specify that dimension, or invent one if no
344 such value exists.
346 Parameters
347 ----------
348 butler : `Butler`
349 Butler to use to look up data ID values.
350 data_id : `dict` [ `str`, `str` or `int` ]
351 Dictionary of possibly-related data ID values.
352 dimension : `Dimension`
353 Dimension to obtain a value for.
355 Returns
356 -------
357 value : `int` or `str`
358 Value for this dimension.
359 invented : `bool`
360 `True` if the value had to be invented, `False` if a compatible value
361 already existed.
362 """
363 # No values given by caller for this dimension. See if any exist
364 # in the registry that are consistent with the values of dimensions
365 # we do have:
366 match_data_id = {key: data_id[key] for key in data_id.keys() & dimension.dimensions.names}
367 matches = list(butler.registry.queryDimensionRecords(dimension, dataId=match_data_id).limit(1))
368 if not matches:
369 # Nothing in the registry matches: invent a data ID value
370 # with the right type (actual value does not matter).
371 # We may or may not actually make a record with this; that's
372 # easier to check later.
373 dimension_value = _makeRandomDataIdValue(dimension)
374 return dimension_value, True
375 else:
376 # A record does exist in the registry. Use its data ID value.
377 dim_value = matches[0].dataId[dimension.name]
378 assert dim_value is not None
379 return dim_value, False
382def _makeDimensionRecordDict(data_id: dict[str, str | int], dimension: Dimension) -> dict[str, Any]:
383 """Create a dictionary that can be used to build a `DimensionRecord` that
384 is consistent with the given data ID.
386 Parameters
387 ----------
388 data_id : `dict` [ `str`, `str` or `int` ]
389 Dictionary that contains values for at least all of
390 ``dimension.dimensions.names`` (the main dimension, its recursive
391 required dependencies, and its non-recursive implied dependencies).
392 dimension : `Dimension`
393 Dimension to build a record dictionary for.
395 Returns
396 -------
397 record_dict : `dict` [ `str`, `object` ]
398 Dictionary that can be passed as ``**kwargs`` to this dimensions
399 record class constructor.
400 """
401 # Add the primary key field for this dimension.
402 record_dict: dict[str, Any] = {dimension.primaryKey.name: data_id[dimension.name]}
403 # Define secondary keys (e.g., detector name given detector id)
404 record_dict.update(_fillAllKeys(dimension, data_id[dimension.name]))
405 # Set the foreign key values for any related dimensions that should
406 # appear in the record.
407 for related_dimension in dimension.dimensions:
408 if related_dimension.name != dimension.name:
409 record_dict[related_dimension.name] = data_id[related_dimension.name]
410 return record_dict
413def addDataIdValue(butler: Butler, dimension: str, value: str | int, **related: str | int) -> None:
414 """Add the records that back a new data ID to a repository.
416 Parameters
417 ----------
418 butler : `lsst.daf.butler.Butler`
419 The repository to update.
420 dimension : `str`
421 The name of the dimension to gain a new value.
422 value : `str` or `int`
423 The value to register for the dimension.
424 **related : `typing.Any`
425 Any existing dimensions to be linked to ``value``.
427 Notes
428 -----
429 Related dimensions (e.g., the instrument associated with a detector) may be
430 specified using ``related``, which requires a value for those dimensions to
431 have been added to the repository already (generally with a previous call
432 to `addDataIdValue`. Any dependencies of the given dimension that are not
433 included in ``related`` will be linked to existing values arbitrarily, and
434 (for implied dependencies only) created and also inserted into the registry
435 if they do not exist. Values for required dimensions and those given in
436 ``related`` are never created.
438 Because this function creates filler data, it is only suitable for test
439 repositories. It should not be used for repositories intended for real data
440 processing or analysis, which have known dimension values.
442 Examples
443 --------
444 See the guide on :ref:`using-butler-in-tests-make-repo` for usage examples.
445 """
446 # Example is not doctest, because it's probably unsafe to create even an
447 # in-memory butler in that environment.
448 try:
449 full_dimension = butler.dimensions[dimension]
450 except KeyError as e:
451 raise ValueError from e
452 # Bad keys ignored by registry code
453 extra_keys = related.keys() - full_dimension.minimal_group.names
454 if extra_keys:
455 raise ValueError(
456 f"Unexpected keywords {extra_keys} not found in {full_dimension.minimal_group.names}"
457 )
459 # Assemble a dictionary data ID holding the given primary dimension value
460 # and all of the related ones.
461 data_id: dict[str, int | str] = {dimension: value}
462 data_id.update(related)
464 # Compute the set of all dimensions that these recursively depend on.
465 all_dimensions = butler.dimensions.conform(data_id.keys())
467 # Create dicts that will become DimensionRecords for all of these data IDs.
468 # This iteration is guaranteed to be in topological order, so we can count
469 # on new data ID values being invented before they are needed.
470 record_dicts_by_dimension: dict[Dimension, dict[str, Any]] = {}
471 for dimension_name in all_dimensions.names:
472 dimension_obj = butler.dimensions.dimensions[dimension_name]
473 dimension_value = data_id.get(dimension_name)
474 if dimension_value is None:
475 data_id[dimension_name], invented = _findOrInventDataIdValue(butler, data_id, dimension_obj)
476 if not invented:
477 # No need to make a new record; one already exists.
478 continue
479 if dimension_name in related:
480 # Caller passed in a value of this dimension explicitly, but it
481 # isn't the primary dimension they asked to have a record created
482 # for. That means they expect this record to already exist.
483 continue
484 if dimension_name != dimension and dimension_name in all_dimensions.required:
485 # We also don't want to automatically create new dimension records
486 # for required dimensions (except for the main dimension the caller
487 # asked for); those are also asserted by the caller to already
488 # exist.
489 continue
490 if not dimension_obj.has_own_table: 490 ↛ 493line 490 didn't jump to line 493 because the condition on line 490 was never true
491 # Don't need to bother generating full records for dimensions whose
492 # records are not actually stored.
493 continue
494 record_dicts_by_dimension[dimension_obj] = _makeDimensionRecordDict(data_id, dimension_obj)
496 # Sync those dimension record dictionaries with the database.
497 for dimension_obj, record_dict in record_dicts_by_dimension.items():
498 record = dimension_obj.RecordClass(**record_dict)
499 try:
500 butler.registry.syncDimensionData(dimension_obj, record)
501 except sqlalchemy.exc.IntegrityError as e:
502 raise RuntimeError(
503 "Could not create data ID value. Automatic relationship generation "
504 "may have failed; try adding keywords to assign a specific instrument, "
505 "physical_filter, etc. based on the nested exception message."
506 ) from e
509def addDatasetType(butler: Butler, name: str, dimensions: set[str], storageClass: str) -> DatasetType:
510 """Add a new dataset type to a repository.
512 Parameters
513 ----------
514 butler : `lsst.daf.butler.Butler`
515 The repository to update.
516 name : `str`
517 The name of the dataset type.
518 dimensions : `set` [`str`]
519 The dimensions of the new dataset type.
520 storageClass : `str`
521 The storage class the dataset will use.
523 Returns
524 -------
525 datasetType : `lsst.daf.butler.DatasetType`
526 The new type.
528 Raises
529 ------
530 ValueError
531 Raised if the dimensions or storage class is invalid.
533 Notes
534 -----
535 Dataset types are shared across all collections in a repository, so this
536 function does not need to be run for each collection.
537 """
538 try:
539 datasetType = DatasetType(name, dimensions, storageClass, universe=butler.dimensions)
540 butler.registry.registerDatasetType(datasetType)
541 return datasetType
542 except KeyError as e:
543 raise ValueError from e
546class DatastoreMock:
547 """Mocks a butler datastore.
549 Has functions that mock the datastore in a butler. Provides an `apply`
550 function to replace the relevent butler datastore functions with the mock
551 functions.
552 """
554 @staticmethod
555 def apply(butler: Butler) -> None:
556 """Apply datastore mocks to a butler.
558 Parameters
559 ----------
560 butler : `~lsst.daf.butler.Butler`
561 Butler to be modified.
562 """
563 butler._datastore.export = DatastoreMock._mock_export # type: ignore
564 butler._datastore.get = DatastoreMock._mock_get # type: ignore
565 butler._datastore.ingest = MagicMock() # type: ignore
567 @staticmethod
568 def _mock_export(
569 refs: Iterable[DatasetRef], *, directory: str | None = None, transfer: str | None = None
570 ) -> Iterable[FileDataset]:
571 """Mock of `Datastore.export` that satisfies the requirement that
572 the refs passed in are included in the `FileDataset` objects
573 returned.
575 This can be used to construct a `Datastore` mock that can be used
576 in repository export via::
578 datastore = unittest.mock.Mock(spec=Datastore)
579 datastore.export = DatastoreMock._mock_export
581 """
582 for ref in refs:
583 yield FileDataset(
584 refs=[ref], path="mock/path", formatter="lsst.daf.butler.formatters.json.JsonFormatter"
585 )
587 @staticmethod
588 def _mock_get(
589 ref: DatasetRef,
590 parameters: Mapping[str, Any] | None = None,
591 storageClass: StorageClass | str | None = None,
592 ) -> tuple[DatasetId, Mapping[str, Any] | None]:
593 """Mock of `Datastore.get` that just returns the integer dataset ID
594 value and parameters it was given.
595 """
596 return (ref.id, parameters)