Coverage for python/lsst/daf/butler/_butler.py: 94%
331 statements
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-23 02:16 -0700
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-23 02:16 -0700
1# This file is part of daf_butler.
2#
3# Developed for the LSST Data Management System.
4# This product includes software developed by the LSST Project
5# (http://www.lsst.org).
6# See the COPYRIGHT file at the top-level directory of this distribution
7# for details of code ownership.
8#
9# This software is dual licensed under the GNU General Public License and also
10# under a 3-clause BSD license. Recipients may choose which of these licenses
11# to use; please see the files gpl-3.0.txt and/or bsd_license.txt,
12# respectively. If you choose the GPL option then the following text applies
13# (but note that there is still no warranty even if you opt for BSD instead):
14#
15# This program is free software: you can redistribute it and/or modify
16# it under the terms of the GNU General Public License as published by
17# the Free Software Foundation, either version 3 of the License, or
18# (at your option) any later version.
19#
20# This program is distributed in the hope that it will be useful,
21# but WITHOUT ANY WARRANTY; without even the implied warranty of
22# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
23# GNU General Public License for more details.
24#
25# You should have received a copy of the GNU General Public License
26# along with this program. If not, see <http://www.gnu.org/licenses/>.
28from __future__ import annotations
30__all__ = ["Butler", "ParsedButlerDatasetURI", "SpecificButlerDataset"]
32import dataclasses
33import urllib.parse
34import uuid
35import warnings
36from abc import abstractmethod
37from collections.abc import Collection, Iterable, Iterator, Mapping, Sequence
38from contextlib import AbstractContextManager
39from types import EllipsisType
40from typing import TYPE_CHECKING, Any, TextIO
42from lsst.resources import ResourcePath, ResourcePathExpression
43from lsst.utils import doImportType
44from lsst.utils.iteration import ensure_iterable
45from lsst.utils.logging import getLogger
47from ._butler_collections import ButlerCollections
48from ._butler_config import ButlerConfig, ButlerType
49from ._butler_instance_options import ButlerInstanceOptions
50from ._butler_metrics import ButlerMetrics
51from ._butler_repo_index import ButlerRepoIndex
52from ._config import Config, ConfigSubset
53from ._exceptions import EmptyQueryResultError, InvalidQueryError
54from ._limited_butler import LimitedButler
55from ._query_all_datasets import QueryAllDatasetsParameters
56from .datastore import Datastore
57from .dimensions import DataCoordinate, DimensionConfig
58from .registry import RegistryConfig, _RegistryFactory
59from .repo_relocation import BUTLER_ROOT_TAG
60from .utils import has_globs
62if TYPE_CHECKING:
63 from ._dataset_existence import DatasetExistence
64 from ._dataset_provenance import DatasetProvenance
65 from ._dataset_ref import DatasetId, DatasetRef
66 from ._dataset_type import DatasetType
67 from ._deferredDatasetHandle import DeferredDatasetHandle
68 from ._file_dataset import FileDataset
69 from ._labeled_butler_factory import LabeledButlerFactoryProtocol
70 from ._storage_class import StorageClass
71 from ._timespan import Timespan
72 from .datastore import DatasetRefURIs
73 from .dimensions import DataId, DimensionGroup, DimensionRecord
74 from .queries import Query
75 from .registry import CollectionArgType, Registry
76 from .transfers import RepoExportContext
78_LOG = getLogger(__name__)
81@dataclasses.dataclass
82class ParsedButlerDatasetURI:
83 """Representation of the contents of an IVOA IVOID or dataset URI."""
85 label: str
86 """Label of the associated butler repository. (`str`)"""
87 dataset_id: uuid.UUID
88 """Dataset ID of the referenced dataset within the labeled repository.
89 (`uuid.UUID`)"""
90 uri: str
91 """The original URI that was parsed (`str`)."""
94@dataclasses.dataclass
95class SpecificButlerDataset:
96 """A dataset ref associated with a specific butler."""
98 butler: Butler
99 """A specific butler repository (`Butler`)."""
100 dataset: DatasetRef | None
101 """The reference of a specific dataset in that butler (`DatasetRef`)."""
104class _DeprecatedDefault:
105 """Default value for a deprecated parameter."""
108class Butler(LimitedButler): # numpydoc ignore=PR02
109 """Interface for data butler and factory for Butler instances.
111 Parameters
112 ----------
113 config : `ButlerConfig`, `Config` or `str`, optional
114 Configuration. Anything acceptable to the `ButlerConfig` constructor.
115 If a directory path is given the configuration will be read from a
116 ``butler.yaml`` file in that location. If `None` is given default
117 values will be used. If ``config`` contains "cls" key then its value is
118 used as a name of butler class and it must be a sub-class of this
119 class, otherwise `DirectButler` is instantiated.
120 collections : `str` or `~collections.abc.Iterable` [ `str` ], optional
121 An expression specifying the collections to be searched (in order) when
122 reading datasets.
123 This may be a `str` collection name or an iterable thereof.
124 See :ref:`daf_butler_collection_expressions` for more information.
125 These collections are not registered automatically and must be
126 manually registered before they are used by any method, but they may be
127 manually registered after the `Butler` is initialized.
128 run : `str`, optional
129 Name of the `~CollectionType.RUN` collection new datasets should be
130 inserted into. If ``collections`` is `None` and ``run`` is not `None`,
131 ``collections`` will be set to ``[run]``. If not `None`, this
132 collection will automatically be registered. If this is not set (and
133 ``writeable`` is not set either), a read-only butler will be created.
134 searchPaths : `list` of `str`, optional
135 Directory paths to search when calculating the full Butler
136 configuration. Not used if the supplied config is already a
137 `ButlerConfig`.
138 writeable : `bool`, optional
139 Explicitly sets whether the butler supports write operations. If not
140 provided, a read-write butler is created if any of ``run``, ``tags``,
141 or ``chains`` is non-empty.
142 inferDefaults : `bool`, optional
143 If `True` (default) infer default data ID values from the values
144 present in the datasets in ``collections``: if all collections have the
145 same value (or no value) for a governor dimension, that value will be
146 the default for that dimension. Nonexistent collections are ignored.
147 If a default value is provided explicitly for a governor dimension via
148 ``**kwargs``, no default will be inferred for that dimension.
149 without_datastore : `bool`, optional
150 If `True` do not attach a datastore to this butler. Any attempts
151 to use a datastore will fail.
152 metrics : `ButlerMetrics` or `None`
153 External metrics object to be used for tracking butler usage. If `None`
154 a new metrics object is created.
155 **kwargs : `typing.Any`
156 Additional keyword arguments passed to a constructor of actual butler
157 class.
159 Notes
160 -----
161 The preferred way to instantiate Butler is via the `from_config` method.
162 The call to ``Butler(...)`` is equivalent to ``Butler.from_config(...)``,
163 but ``mypy`` will complain about the former.
164 """
166 def __new__(
167 cls,
168 config: Config | ResourcePathExpression | None = None,
169 *,
170 collections: Any = None,
171 run: str | None = None,
172 searchPaths: Sequence[ResourcePathExpression] | None = None,
173 writeable: bool | None = None,
174 inferDefaults: bool = True,
175 without_datastore: bool = False,
176 metrics: ButlerMetrics | None = None,
177 **kwargs: Any,
178 ) -> Butler:
179 if cls is Butler:
180 return Butler.from_config(
181 config=config,
182 collections=collections,
183 run=run,
184 searchPaths=searchPaths,
185 writeable=writeable,
186 inferDefaults=inferDefaults,
187 without_datastore=without_datastore,
188 metrics=metrics,
189 **kwargs,
190 )
192 # Note: we do not pass any parameters to __new__, Python will pass them
193 # to __init__ after __new__ returns sub-class instance.
194 return super().__new__(cls)
196 @classmethod
197 def from_config(
198 cls,
199 config: Config | ResourcePathExpression | None = None,
200 *,
201 collections: Any = None,
202 run: str | None = None,
203 searchPaths: Sequence[ResourcePathExpression] | None = None,
204 writeable: bool | None = None,
205 inferDefaults: bool = True,
206 without_datastore: bool = False,
207 metrics: ButlerMetrics | None = None,
208 **kwargs: Any,
209 ) -> Butler:
210 """Create butler instance from configuration.
212 Parameters
213 ----------
214 config : `ButlerConfig`, `Config` or `str`, optional
215 Configuration. Anything acceptable to the `ButlerConfig`
216 constructor. If a directory path is given the configuration will be
217 read from a ``butler.yaml`` file in that location. If `None` is
218 given default values will be used. If ``config`` contains "cls" key
219 then its value is used as a name of butler class and it must be a
220 sub-class of this class, otherwise `DirectButler` is instantiated.
221 collections : `str` or `~collections.abc.Iterable` [ `str` ], optional
222 An expression specifying the collections to be searched (in order)
223 when reading datasets.
224 This may be a `str` collection name or an iterable thereof.
225 See :ref:`daf_butler_collection_expressions` for more information.
226 These collections are not registered automatically and must be
227 manually registered before they are used by any method, but they
228 may be manually registered after the `Butler` is initialized.
229 run : `str`, optional
230 Name of the `~CollectionType.RUN` collection new datasets should be
231 inserted into. If ``collections`` is `None` and ``run`` is not
232 `None`, ``collections`` will be set to ``[run]``. If not `None`,
233 this collection will automatically be registered. If this is not
234 set (and ``writeable`` is not set either), a read-only butler will
235 be created.
236 searchPaths : `list` of `str`, optional
237 Directory paths to search when calculating the full Butler
238 configuration. Not used if the supplied config is already a
239 `ButlerConfig`.
240 writeable : `bool`, optional
241 Explicitly sets whether the butler supports write operations. If
242 not provided, a read-write butler is created if any of ``run``,
243 ``tags``, or ``chains`` is non-empty.
244 inferDefaults : `bool`, optional
245 If `True` (default) infer default data ID values from the values
246 present in the datasets in ``collections``: if all collections have
247 the same value (or no value) for a governor dimension, that value
248 will be the default for that dimension. Nonexistent collections
249 are ignored. If a default value is provided explicitly for a
250 governor dimension via ``**kwargs``, no default will be inferred
251 for that dimension.
252 without_datastore : `bool`, optional
253 If `True` do not attach a datastore to this butler. Any attempts
254 to use a datastore will fail.
255 metrics : `ButlerMetrics` or `None`, optional
256 Metrics object to record butler usage statistics.
257 **kwargs : `typing.Any`
258 Default data ID key-value pairs. These may only identify
259 "governor" dimensions like ``instrument`` and ``skymap``.
261 Returns
262 -------
263 butler : `Butler`
264 A `Butler` constructed from the given configuration.
266 Notes
267 -----
268 Calling this factory method is identical to calling
269 ``Butler(config, ...)``. Its only raison d'être is that ``mypy``
270 complains about ``Butler()`` call.
272 Examples
273 --------
274 While there are many ways to control exactly how a `Butler` interacts
275 with the collections in its `Registry`, the most common cases are still
276 simple.
278 For a read-only `Butler` that searches one collection, do::
280 butler = Butler.from_config(
281 "/path/to/repo", collections=["u/alice/DM-50000"]
282 )
284 For a read-write `Butler` that writes to and reads from a
285 `~CollectionType.RUN` collection::
287 butler = Butler.from_config(
288 "/path/to/repo", run="u/alice/DM-50000/a"
289 )
291 The `Butler` passed to a ``PipelineTask`` is often much more complex,
292 because we want to write to one `~CollectionType.RUN` collection but
293 read from several others (as well)::
295 butler = Butler.from_config(
296 "/path/to/repo",
297 run="u/alice/DM-50000/a",
298 collections=[
299 "u/alice/DM-50000/a",
300 "u/bob/DM-49998",
301 "HSC/defaults",
302 ],
303 )
305 This butler will `put` new datasets to the run ``u/alice/DM-50000/a``.
306 Datasets will be read first from that run (since it appears first in
307 the chain), and then from ``u/bob/DM-49998`` and finally
308 ``HSC/defaults``.
310 Finally, one can always create a `Butler` with no collections::
312 butler = Butler.from_config("/path/to/repo", writeable=True)
314 This can be extremely useful when you just want to use
315 ``butler.registry``, e.g. for inserting dimension data or managing
316 collections, or when the collections you want to use with the butler
317 are not consistent. Passing ``writeable`` explicitly here is only
318 necessary if you want to be able to make changes to the repo - usually
319 the value for ``writeable`` can be guessed from the collection
320 arguments provided, but it defaults to `False` when there are not
321 collection arguments.
322 """
323 # DirectButler used to have a way to specify a "copy constructor" by
324 # passing the "butler" parameter to its constructor. This has
325 # been moved out of the constructor into Butler.clone().
326 butler = kwargs.pop("butler", None)
327 metrics = metrics if metrics is not None else ButlerMetrics()
328 if butler is not None:
329 if not isinstance(butler, Butler): 329 ↛ 330line 329 didn't jump to line 330 because the condition on line 329 was never true
330 raise TypeError("'butler' parameter must be a Butler instance")
331 if config is not None or searchPaths is not None or writeable is not None: 331 ↛ 332line 331 didn't jump to line 332 because the condition on line 331 was never true
332 raise TypeError(
333 "Cannot pass 'config', 'searchPaths', or 'writeable' arguments with 'butler' argument."
334 )
335 return butler.clone(
336 collections=collections, run=run, inferDefaults=inferDefaults, metrics=metrics, dataId=kwargs
337 )
339 options = ButlerInstanceOptions(
340 collections=collections,
341 run=run,
342 writeable=writeable,
343 inferDefaults=inferDefaults,
344 metrics=metrics,
345 kwargs=kwargs,
346 )
348 # Load the Butler configuration. This may involve searching the
349 # environment to locate a configuration file.
350 butler_config = ButlerConfig(config, searchPaths=searchPaths, without_datastore=without_datastore)
351 butler_type = butler_config.get_butler_type()
353 # Make DirectButler if class is not specified.
354 match butler_type:
355 case ButlerType.DIRECT: 355 ↛ 363line 355 didn't jump to line 363 because the pattern on line 355 always matched
356 from .direct_butler import DirectButler
358 return DirectButler.create_from_config(
359 butler_config,
360 options=options,
361 without_datastore=without_datastore,
362 )
363 case ButlerType.REMOTE:
364 from .remote_butler._factory import RemoteButlerFactory
366 # Assume this is being created by a client who would like
367 # default caching of remote datasets.
368 factory = RemoteButlerFactory.create_factory_from_config(butler_config)
369 return factory.create_butler_with_credentials_from_environment(
370 butler_options=options, enable_datastore_cache=True
371 )
372 case _:
373 raise TypeError(f"Unknown Butler type '{butler_type}'")
375 @staticmethod
376 def has_repo_config(root: ResourcePathExpression) -> bool:
377 """Check whether the given directory path contains a Butler
378 configuration or not.
380 Parameters
381 ----------
382 root : `lsst.resources.ResourcePathExpression`
383 The directory URI to check.
385 Returns
386 -------
387 is_root : `bool`
388 `True` if this is a directory containing a butler configuration.
389 """
390 root_uri = ResourcePath(root, forceDirectory=True)
391 return root_uri.join("butler.yaml").exists()
393 @staticmethod
394 def makeRepo(
395 root: ResourcePathExpression,
396 config: Config | str | None = None,
397 dimensionConfig: Config | str | None = None,
398 standalone: bool = False,
399 searchPaths: list[str] | None = None,
400 forceConfigRoot: bool = True,
401 outfile: ResourcePathExpression | None = None,
402 overwrite: bool = False,
403 ) -> Config:
404 """Create an empty data repository by adding a butler.yaml config
405 to a repository root directory.
407 Parameters
408 ----------
409 root : `lsst.resources.ResourcePathExpression`
410 Path or URI to the root location of the new repository. Will be
411 created if it does not exist.
412 config : `Config` or `str`, optional
413 Configuration to write to the repository, after setting any
414 root-dependent Registry or Datastore config options. Can not
415 be a `ButlerConfig` or a `ConfigSubset`. If `None`, default
416 configuration will be used. Root-dependent config options
417 specified in this config are overwritten if ``forceConfigRoot``
418 is `True`.
419 dimensionConfig : `Config` or `str`, optional
420 Configuration for dimensions, will be used to initialize registry
421 database.
422 standalone : `bool`
423 If True, write all expanded defaults, not just customized or
424 repository-specific settings.
425 This (mostly) decouples the repository from the default
426 configuration, insulating it from changes to the defaults (which
427 may be good or bad, depending on the nature of the changes).
428 Future *additions* to the defaults will still be picked up when
429 initializing a `Butler` for repos created with ``standalone=True``.
430 searchPaths : `list` of `str`, optional
431 Directory paths to search when calculating the full butler
432 configuration.
433 forceConfigRoot : `bool`, optional
434 If `False`, any values present in the supplied ``config`` that
435 would normally be reset are not overridden and will appear
436 directly in the output config. This allows non-standard overrides
437 of the root directory for a datastore or registry to be given.
438 If this parameter is `True` the values for ``root`` will be
439 forced into the resulting config if appropriate.
440 outfile : `lsst.resources.ResourcePathExpression`, optional
441 If not-`None`, the output configuration will be written to this
442 location rather than into the repository itself. Can be a URI
443 string. Can refer to a directory that will be used to write
444 ``butler.yaml``.
445 overwrite : `bool`, optional
446 Create a new configuration file even if one already exists
447 in the specified output location. Default is to raise
448 an exception.
450 Returns
451 -------
452 config : `Config`
453 The updated `Config` instance written to the repo.
455 Raises
456 ------
457 ValueError
458 Raised if a ButlerConfig or ConfigSubset is passed instead of a
459 regular Config (as these subclasses would make it impossible to
460 support ``standalone=False``).
461 FileExistsError
462 Raised if the output config file already exists.
463 os.error
464 Raised if the directory does not exist, exists but is not a
465 directory, or cannot be created.
467 Notes
468 -----
469 Note that when ``standalone=False`` (the default), the configuration
470 search path (see `ConfigSubset.defaultSearchPaths`) that was used to
471 construct the repository should also be used to construct any Butlers
472 to avoid configuration inconsistencies.
473 """
474 config, root_uri = Butler._make_repo_butler_config(
475 root,
476 config=config,
477 standalone=standalone,
478 searchPaths=searchPaths,
479 forceConfigRoot=forceConfigRoot,
480 outfile=outfile,
481 overwrite=overwrite,
482 )
483 Butler._make_repo_registry(config, dimensionConfig=dimensionConfig, root_uri=root_uri)
484 return config
486 @staticmethod
487 def _make_repo_butler_config(
488 root: ResourcePathExpression,
489 config: Config | str | None = None,
490 standalone: bool = False,
491 searchPaths: list[str] | None = None,
492 forceConfigRoot: bool = True,
493 outfile: ResourcePathExpression | None = None,
494 overwrite: bool = False,
495 ) -> tuple[Config, ResourcePath]:
496 """Create the repository directory and write its configuration.
498 This is the first half of `makeRepo`; it does everything except
499 creating the registry database.
501 Parameters
502 ----------
503 root : `lsst.resources.ResourcePathExpression`
504 Path or URI to the root location of the new repository.
505 config : `Config` or `str`, optional
506 Configuration to write to the repository.
507 standalone : `bool`, optional
508 If `True`, write all expanded defaults.
509 searchPaths : `list` of `str`, optional
510 Directory paths to search when calculating the full configuration.
511 forceConfigRoot : `bool`, optional
512 If `False`, any values present in ``config`` that would normally be
513 reset are not overridden.
514 outfile : `lsst.resources.ResourcePathExpression`, optional
515 If not-`None`, write the configuration here instead of into the
516 repository.
517 overwrite : `bool`, optional
518 If `False` an existing config file will cause an exception.
520 Returns
521 -------
522 config : `Config`
523 The configuration that was written.
524 root_uri : `lsst.resources.ResourcePath`
525 The root of the new repository.
527 Raises
528 ------
529 ValueError
530 Raised if a `ButlerConfig` or `ConfigSubset` is passed instead of a
531 regular `Config`.
532 """
533 if isinstance(config, ButlerConfig | ConfigSubset): 533 ↛ 534line 533 didn't jump to line 534 because the condition on line 533 was never true
534 raise ValueError("makeRepo must be passed a regular Config without defaults applied.")
536 # Ensure that the root of the repository exists or can be made
537 root_uri = ResourcePath(root, forceDirectory=True)
538 root_uri.mkdir()
540 config = Config(config)
542 # If we are creating a new repo from scratch with relative roots,
543 # do not propagate an explicit root from the config file
544 if "root" in config: 544 ↛ 545line 544 didn't jump to line 545 because the condition on line 544 was never true
545 del config["root"]
547 full = ButlerConfig(config, searchPaths=searchPaths) # this applies defaults
548 imported_class = doImportType(full["datastore", "cls"])
549 if not issubclass(imported_class, Datastore): 549 ↛ 550line 549 didn't jump to line 550 because the condition on line 549 was never true
550 raise TypeError(f"Imported datastore class {full['datastore', 'cls']} is not a Datastore")
551 datastoreClass: type[Datastore] = imported_class
552 datastoreClass.setConfigRoot(BUTLER_ROOT_TAG, config, full, overwrite=forceConfigRoot)
554 # if key exists in given config, parse it, otherwise parse the defaults
555 # in the expanded config
556 if config.get(("registry", "db")):
557 registryConfig = RegistryConfig(config)
558 else:
559 registryConfig = RegistryConfig(full)
560 defaultDatabaseUri = registryConfig.makeDefaultDatabaseUri(BUTLER_ROOT_TAG)
561 if defaultDatabaseUri is not None:
562 Config.updateParameters(
563 RegistryConfig, config, full, toUpdate={"db": defaultDatabaseUri}, overwrite=forceConfigRoot
564 )
565 else:
566 Config.updateParameters(RegistryConfig, config, full, toCopy=("db",), overwrite=forceConfigRoot)
568 if standalone:
569 config.merge(full)
570 else:
571 # Always expand the registry.managers section into the per-repo
572 # config, because after the database schema is created, it's not
573 # allowed to change anymore. Note that in the standalone=True
574 # branch, _everything_ in the config is expanded, so there's no
575 # need to special case this.
576 Config.updateParameters(RegistryConfig, config, full, toMerge=("managers",), overwrite=False)
577 configURI: ResourcePathExpression
578 if outfile is not None:
579 # When writing to a separate location we must include
580 # the root of the butler repo in the config else it won't know
581 # where to look.
582 config["root"] = root_uri.geturl()
583 configURI = outfile
584 else:
585 configURI = root_uri
586 # Check that if obscore key is present then its config must be there
587 # too, this is to avoid common mistake when people copy butler.yaml
588 # from existing repo with obscore but do not fill its config.
589 if (obscore_key := ("registry", "managers", "obscore")) in config:
590 obscore_config_key = ("registry", "managers", "obscore", "config")
591 if obscore_config_key not in config or not config[obscore_config_key]:
592 warnings.warn(
593 "Obscore manager is declared in registry configuration, "
594 "but obscore configuration is missing, obscore manager will be removed.",
595 stacklevel=2,
596 )
597 del config[obscore_key]
598 # Strip obscore configuration, if it is present, before writing config
599 # to a file, obscore config will be stored in registry.
600 if (obscore_config_key := ("registry", "managers", "obscore", "config")) in config:
601 config_to_write = config.copy()
602 del config_to_write[obscore_config_key]
603 config_to_write.dumpToUri(configURI, overwrite=overwrite)
604 # configFile attribute is updated, need to copy it to original.
605 config.configFile = config_to_write.configFile
606 else:
607 config.dumpToUri(configURI, overwrite=overwrite)
609 _LOG.verbose("Wrote new Butler configuration file to %s", configURI)
611 return config, root_uri
613 @staticmethod
614 def _make_repo_registry(
615 config: Config,
616 dimensionConfig: Config | str | None = None,
617 root_uri: ResourcePath | None = None,
618 ) -> None:
619 """Create the registry database for a new repository.
621 This is the second half of `makeRepo`; it assumes the repository
622 directory and its configuration already exist.
624 Parameters
625 ----------
626 config : `Config`
627 The repository configuration, as returned by
628 `_make_repo_butler_config`.
629 dimensionConfig : `Config` or `str`, optional
630 Configuration for dimensions, used to initialize the database.
631 root_uri : `lsst.resources.ResourcePath`, optional
632 Root of the repository, used to resolve a relative database
633 location.
634 """
635 registryConfig = RegistryConfig(config.get("registry"))
636 registry = _RegistryFactory(registryConfig).create_from_config(
637 dimensionConfig=DimensionConfig(dimensionConfig), butlerRoot=root_uri
638 )
639 registry.close()
641 @classmethod
642 def get_repo_uri(cls, label: str, return_label: bool = False) -> ResourcePath:
643 """Look up the label in a butler repository index.
645 Parameters
646 ----------
647 label : `str`
648 Label of the Butler repository to look up.
649 return_label : `bool`, optional
650 If ``label`` cannot be found in the repository index (either
651 because index is not defined or ``label`` is not in the index) and
652 ``return_label`` is `True` then return ``ResourcePath(label)``.
653 If ``return_label`` is `False` (default) then an exception will be
654 raised instead.
656 Returns
657 -------
658 uri : `lsst.resources.ResourcePath`
659 URI to the Butler repository associated with the given label or
660 default value if it is provided.
662 Raises
663 ------
664 KeyError
665 Raised if the label is not found in the index, or if an index
666 is not defined, and ``return_label`` is `False`.
668 Notes
669 -----
670 See `~lsst.daf.butler.ButlerRepoIndex` for details on how the
671 information is discovered.
672 """
673 return ButlerRepoIndex.get_repo_uri(label, return_label)
675 @classmethod
676 def get_known_repos(cls) -> set[str]:
677 """Retrieve the list of known repository labels.
679 Returns
680 -------
681 repos : `set` of `str`
682 All the known labels. Can be empty if no index can be found.
684 Notes
685 -----
686 See `~lsst.daf.butler.ButlerRepoIndex` for details on how the
687 information is discovered.
688 """
689 return ButlerRepoIndex.get_known_repos()
691 @classmethod
692 def parse_dataset_uri(cls, uri: str) -> ParsedButlerDatasetURI:
693 """Extract the butler label and dataset ID from a dataset URI.
695 Parameters
696 ----------
697 uri : `str`
698 The dataset URI to parse.
700 Returns
701 -------
702 parsed : `ParsedButlerDatasetURI`
703 The label associated with the butler repository from which this
704 dataset originates and the ID of the dataset.
706 Notes
707 -----
708 Supports dataset URIs of the forms
709 ``ivo://org.rubinobs/usdac/dr1?repo=butler_label&id=UUID`` (see
710 DMTN-302) and ``butler://butler_label/UUID``. The ``butler`` URI is
711 deprecated and can not include ``/`` in the label string. ``ivo`` URIs
712 can include anything supported by the `Butler` constructor, including
713 paths to repositories and alias labels.
715 ivo://org.rubinobs/dr1?repo=/repo/main&id=UUID
717 will return a label of ``/repo/main``.
719 This method does not attempt to check that the dataset exists in the
720 labeled butler.
722 Since the IVOID can be issued by any publisher to represent a Butler
723 dataset there is no validation of the path or netloc component of the
724 URI. The only requirement is that there are ``id`` and ``repo`` keys
725 in the ``ivo`` URI query component.
726 """
727 parsed = urllib.parse.urlparse(uri)
728 parsed_scheme = parsed.scheme.lower()
729 if parsed_scheme == "ivo":
730 # Do not validate the netloc or the path values.
731 qs = urllib.parse.parse_qs(parsed.query)
732 if "repo" not in qs or "id" not in qs:
733 raise ValueError(f"Missing 'repo' and/or 'id' query parameters in IVOID {uri}.")
734 if len(qs["repo"]) != 1 or len(qs["id"]) != 1:
735 raise ValueError(f"Butler IVOID only supports a single value of repo and id, got {uri}")
736 label = qs["repo"][0]
737 id_ = qs["id"][0]
738 elif parsed_scheme == "butler":
739 label = parsed.netloc # Butler label is case sensitive.
740 # Need to strip the leading /.
741 id_ = parsed.path[1:]
742 else:
743 raise ValueError(f"Unrecognized URI scheme: {uri!r}")
744 # Strip trailing/leading whitespace from label.
745 label = label.strip()
746 if not label:
747 raise ValueError(f"No butler repository label found in uri {uri!r}")
748 try:
749 dataset_id = uuid.UUID(hex=id_)
750 except Exception as e:
751 e.add_note(f"Error extracting dataset ID from uri {uri!r} with dataset ID string {id_!r}")
752 raise
754 return ParsedButlerDatasetURI(label=label, dataset_id=dataset_id, uri=uri)
756 @classmethod
757 def get_dataset_from_uri(
758 cls, uri: str, factory: LabeledButlerFactoryProtocol | None = None
759 ) -> SpecificButlerDataset:
760 """Get the dataset associated with the given dataset URI.
762 Parameters
763 ----------
764 uri : `str`
765 The URI associated with a dataset.
766 factory : `LabeledButlerFactoryProtocol` or `None`, optional
767 Bound factory function that will be given the butler label
768 and receive a `Butler`. If this is not provided the label
769 will be tried directly.
771 Returns
772 -------
773 result : `SpecificButlerDataset`
774 The butler associated with this URI and the dataset itself.
775 The dataset can be `None` if the UUID is valid but the dataset
776 is not known to this butler.
777 """
778 parsed = cls.parse_dataset_uri(uri)
779 butler: Butler | None = None
780 if factory is not None:
781 # If the label is not recognized, it might be a path.
782 try:
783 butler = factory(parsed.label)
784 except KeyError:
785 pass
786 if butler is None:
787 butler = cls.from_config(parsed.label)
788 return SpecificButlerDataset(butler=butler, dataset=butler.get_dataset(parsed.dataset_id))
790 @abstractmethod
791 def _caching_context(self) -> AbstractContextManager[None]:
792 """Context manager that enables caching."""
793 raise NotImplementedError()
795 @abstractmethod
796 def transaction(self) -> AbstractContextManager[None]:
797 """Context manager supporting `Butler` transactions.
799 Transactions can be nested.
800 """
801 raise NotImplementedError()
803 @abstractmethod
804 def put(
805 self,
806 obj: Any,
807 datasetRefOrType: DatasetRef | DatasetType | str,
808 /,
809 dataId: DataId | None = None,
810 *,
811 run: str | None = None,
812 provenance: DatasetProvenance | None = None,
813 **kwargs: Any,
814 ) -> DatasetRef:
815 """Store and register a dataset.
817 Parameters
818 ----------
819 obj : `object`
820 The dataset.
821 datasetRefOrType : `DatasetRef`, `DatasetType`, or `str`
822 When `DatasetRef` is provided, ``dataId`` should be `None`.
823 Otherwise the `DatasetType` or name thereof. If a fully resolved
824 `DatasetRef` is given the run and ID are used directly.
825 dataId : `dict` or `DataCoordinate`
826 A `dict` of `Dimension` link name, value pairs that label the
827 `DatasetRef` within a Collection. When `None`, a `DatasetRef`
828 should be provided as the second argument.
829 run : `str`, optional
830 The name of the run the dataset should be added to, overriding
831 ``self.run``. Not used if a resolved `DatasetRef` is provided.
832 provenance : `DatasetProvenance` or `None`, optional
833 Any provenance that should be attached to the serialized dataset.
834 Not supported by all serialization mechanisms.
835 **kwargs
836 Additional keyword arguments used to augment or construct a
837 `DataCoordinate`. See `DataCoordinate.standardize`
838 parameters. Not used if a resolve `DatasetRef` is provided.
840 Returns
841 -------
842 ref : `DatasetRef`
843 A reference to the stored dataset, updated with the correct id if
844 given.
846 Raises
847 ------
848 TypeError
849 Raised if the butler is read-only or if no run has been provided.
850 """
851 raise NotImplementedError()
853 @abstractmethod
854 def getDeferred(
855 self,
856 datasetRefOrType: DatasetRef | DatasetType | str,
857 /,
858 dataId: DataId | None = None,
859 *,
860 parameters: dict | None = None,
861 collections: Any = None,
862 storageClass: str | StorageClass | None = None,
863 timespan: Timespan | None = None,
864 **kwargs: Any,
865 ) -> DeferredDatasetHandle:
866 """Create a `DeferredDatasetHandle` which can later retrieve a dataset,
867 after an immediate registry lookup.
869 Parameters
870 ----------
871 datasetRefOrType : `DatasetRef`, `DatasetType`, or `str`
872 When `DatasetRef` the `dataId` should be `None`.
873 Otherwise the `DatasetType` or name thereof.
874 dataId : `dict` or `DataCoordinate`, optional
875 A `dict` of `Dimension` link name, value pairs that label the
876 `DatasetRef` within a Collection. When `None`, a `DatasetRef`
877 should be provided as the first argument.
878 parameters : `dict`
879 Additional StorageClass-defined options to control reading,
880 typically used to efficiently read only a subset of the dataset.
881 collections : Any, optional
882 Collections to be searched, overriding ``self.collections``.
883 Can be any of the types supported by the ``collections`` argument
884 to butler construction.
885 storageClass : `StorageClass` or `str`, optional
886 The storage class to be used to override the Python type
887 returned by this method. By default the returned type matches
888 the dataset type definition for this dataset. Specifying a
889 read `StorageClass` can force a different type to be returned.
890 This type must be compatible with the original type.
891 timespan : `Timespan` or `None`, optional
892 A timespan that the validity range of the dataset must overlap.
893 If not provided and this is a calibration dataset type, an attempt
894 will be made to find the timespan from any temporal coordinate
895 in the data ID.
896 **kwargs
897 Additional keyword arguments used to augment or construct a
898 `DataId`. See `DataId` parameters.
900 Returns
901 -------
902 obj : `DeferredDatasetHandle`
903 A handle which can be used to retrieve a dataset at a later time.
905 Raises
906 ------
907 LookupError
908 Raised if no matching dataset exists in the `Registry` or
909 datastore.
910 ValueError
911 Raised if a resolved `DatasetRef` was passed as an input, but it
912 differs from the one found in the registry.
913 TypeError
914 Raised if no collections were provided.
915 """
916 raise NotImplementedError()
918 @abstractmethod
919 def get(
920 self,
921 datasetRefOrType: DatasetRef | DatasetType | str,
922 /,
923 dataId: DataId | None = None,
924 *,
925 parameters: dict[str, Any] | None = None,
926 collections: Any = None,
927 storageClass: StorageClass | str | None = None,
928 timespan: Timespan | None = None,
929 **kwargs: Any,
930 ) -> Any:
931 """Retrieve a stored dataset.
933 Parameters
934 ----------
935 datasetRefOrType : `DatasetRef`, `DatasetType`, or `str`
936 When `DatasetRef` the `dataId` should be `None`.
937 Otherwise the `DatasetType` or name thereof.
938 If a resolved `DatasetRef`, the associated dataset
939 is returned directly without additional querying.
940 dataId : `dict` or `DataCoordinate`
941 A `dict` of `Dimension` link name, value pairs that label the
942 `DatasetRef` within a Collection. When `None`, a `DatasetRef`
943 should be provided as the first argument.
944 parameters : `dict`
945 Additional StorageClass-defined options to control reading,
946 typically used to efficiently read only a subset of the dataset.
947 collections : Any, optional
948 Collections to be searched, overriding ``self.collections``.
949 Can be any of the types supported by the ``collections`` argument
950 to butler construction.
951 storageClass : `StorageClass` or `str`, optional
952 The storage class to be used to override the Python type
953 returned by this method. By default the returned type matches
954 the dataset type definition for this dataset. Specifying a
955 read `StorageClass` can force a different type to be returned.
956 This type must be compatible with the original type.
957 timespan : `Timespan` or `None`, optional
958 A timespan that the validity range of the dataset must overlap.
959 If not provided and this is a calibration dataset type, an attempt
960 will be made to find the timespan from any temporal coordinate
961 in the data ID.
962 **kwargs
963 Additional keyword arguments used to augment or construct a
964 `DataCoordinate`. See `DataCoordinate.standardize`
965 parameters.
967 Returns
968 -------
969 obj : `object`
970 The dataset.
972 Raises
973 ------
974 LookupError
975 Raised if no matching dataset exists in the `Registry`.
976 TypeError
977 Raised if no collections were provided.
979 Notes
980 -----
981 When looking up datasets in a `~CollectionType.CALIBRATION` collection,
982 this method requires that the given data ID include temporal dimensions
983 beyond the dimensions of the dataset type itself, in order to find the
984 dataset with the appropriate validity range. For example, a "bias"
985 dataset with native dimensions ``{instrument, detector}`` could be
986 fetched with a ``{instrument, detector, exposure}`` data ID, because
987 ``exposure`` is a temporal dimension.
988 """
989 raise NotImplementedError()
991 @abstractmethod
992 def getURIs(
993 self,
994 datasetRefOrType: DatasetRef | DatasetType | str,
995 /,
996 dataId: DataId | None = None,
997 *,
998 predict: bool = False,
999 collections: Any = None,
1000 run: str | None = None,
1001 **kwargs: Any,
1002 ) -> DatasetRefURIs:
1003 """Return the URIs associated with the dataset.
1005 Parameters
1006 ----------
1007 datasetRefOrType : `DatasetRef`, `DatasetType`, or `str`
1008 When `DatasetRef` the `dataId` should be `None`.
1009 Otherwise the `DatasetType` or name thereof.
1010 dataId : `dict` or `DataCoordinate`
1011 A `dict` of `Dimension` link name, value pairs that label the
1012 `DatasetRef` within a Collection. When `None`, a `DatasetRef`
1013 should be provided as the first argument.
1014 predict : `bool`
1015 If `True`, allow URIs to be returned of datasets that have not
1016 been written.
1017 collections : Any, optional
1018 Collections to be searched, overriding ``self.collections``.
1019 Can be any of the types supported by the ``collections`` argument
1020 to butler construction.
1021 run : `str`, optional
1022 Run to use for predictions, overriding ``self.run``.
1023 **kwargs
1024 Additional keyword arguments used to augment or construct a
1025 `DataCoordinate`. See `DataCoordinate.standardize`
1026 parameters.
1028 Returns
1029 -------
1030 uris : `DatasetRefURIs`
1031 The URI to the primary artifact associated with this dataset (if
1032 the dataset was disassembled within the datastore this may be
1033 `None`), and the URIs to any components associated with the dataset
1034 artifact. (can be empty if there are no components).
1035 """
1036 raise NotImplementedError()
1038 def getURI(
1039 self,
1040 datasetRefOrType: DatasetRef | DatasetType | str,
1041 /,
1042 dataId: DataId | None = None,
1043 *,
1044 predict: bool = False,
1045 collections: Any = None,
1046 run: str | None = None,
1047 **kwargs: Any,
1048 ) -> ResourcePath:
1049 """Return the URI to the Dataset.
1051 Parameters
1052 ----------
1053 datasetRefOrType : `DatasetRef`, `DatasetType`, or `str`
1054 When `DatasetRef` the `dataId` should be `None`.
1055 Otherwise the `DatasetType` or name thereof.
1056 dataId : `dict` or `DataCoordinate`
1057 A `dict` of `Dimension` link name, value pairs that label the
1058 `DatasetRef` within a Collection. When `None`, a `DatasetRef`
1059 should be provided as the first argument.
1060 predict : `bool`
1061 If `True`, allow URIs to be returned of datasets that have not
1062 been written.
1063 collections : Any, optional
1064 Collections to be searched, overriding ``self.collections``.
1065 Can be any of the types supported by the ``collections`` argument
1066 to butler construction.
1067 run : `str`, optional
1068 Run to use for predictions, overriding ``self.run``.
1069 **kwargs
1070 Additional keyword arguments used to augment or construct a
1071 `DataCoordinate`. See `DataCoordinate.standardize`
1072 parameters.
1074 Returns
1075 -------
1076 uri : `lsst.resources.ResourcePath`
1077 URI pointing to the Dataset within the datastore. If the
1078 Dataset does not exist in the datastore, and if ``predict`` is
1079 `True`, the URI will be a prediction and will include a URI
1080 fragment "#predicted".
1081 If the datastore does not have entities that relate well
1082 to the concept of a URI the returned URI string will be
1083 descriptive. The returned URI is not guaranteed to be obtainable.
1085 Raises
1086 ------
1087 LookupError
1088 A URI has been requested for a dataset that does not exist and
1089 guessing is not allowed.
1090 ValueError
1091 Raised if a resolved `DatasetRef` was passed as an input, but it
1092 differs from the one found in the registry.
1093 TypeError
1094 Raised if no collections were provided.
1095 RuntimeError
1096 Raised if a URI is requested for a dataset that consists of
1097 multiple artifacts.
1098 """
1099 primary, components = self.getURIs(
1100 datasetRefOrType, dataId=dataId, predict=predict, collections=collections, run=run, **kwargs
1101 )
1103 if primary is None or components: 1103 ↛ 1104line 1103 didn't jump to line 1104 because the condition on line 1103 was never true
1104 raise RuntimeError(
1105 f"Dataset ({datasetRefOrType}) includes distinct URIs for components. "
1106 "Use Butler.getURIs() instead."
1107 )
1108 return primary
1110 @abstractmethod
1111 def get_dataset_type(self, name: str) -> DatasetType:
1112 """Get the `DatasetType`.
1114 Parameters
1115 ----------
1116 name : `str`
1117 Name of the type.
1119 Returns
1120 -------
1121 type : `DatasetType`
1122 The `DatasetType` associated with the given name.
1124 Raises
1125 ------
1126 lsst.daf.butler.MissingDatasetTypeError
1127 Raised if the requested dataset type has not been registered.
1129 Notes
1130 -----
1131 This method handles component dataset types automatically, though most
1132 other operations do not.
1133 """
1134 raise NotImplementedError()
1136 @abstractmethod
1137 def get_dataset(
1138 self,
1139 id: DatasetId | str,
1140 *,
1141 storage_class: str | StorageClass | None = None,
1142 dimension_records: bool = False,
1143 datastore_records: bool = False,
1144 ) -> DatasetRef | None:
1145 """Retrieve a Dataset entry.
1147 Parameters
1148 ----------
1149 id : `DatasetId`
1150 The unique identifier for the dataset, as an instance of
1151 `uuid.UUID` or a string containing a hexadecimal number.
1152 storage_class : `str` or `StorageClass` or `None`
1153 A storage class to use when creating the returned entry. If given
1154 it must be compatible with the default storage class.
1155 dimension_records : `bool`, optional
1156 If `True` the ref will be expanded and contain dimension records.
1157 datastore_records : `bool`, optional
1158 If `True` the ref will contain associated datastore records.
1160 Returns
1161 -------
1162 ref : `DatasetRef` or `None`
1163 A ref to the Dataset, or `None` if no matching Dataset
1164 was found.
1165 """
1166 raise NotImplementedError()
1168 @abstractmethod
1169 def get_many_datasets(self, ids: Iterable[DatasetId | str]) -> list[DatasetRef]:
1170 """Retrieve a list of dataset entries.
1172 Parameters
1173 ----------
1174 ids : `~collections.abc.Iterable` [ `DatasetId` or `str` ]
1175 The unique identifiers for the datasets, as instances of
1176 `uuid.UUID` or strings containing a hexadecimal number.
1178 Returns
1179 -------
1180 refs : `list` [ `DatasetRef` ]
1181 A list containing a `DatasetRef` for each of the given dataset IDs.
1182 If a dataset was not found, no error is thrown -- it is just not
1183 included in the list. The returned datasets are in no particular
1184 order.
1185 """
1186 raise NotImplementedError()
1188 @abstractmethod
1189 def find_dataset(
1190 self,
1191 dataset_type: DatasetType | str,
1192 data_id: DataId | None = None,
1193 *,
1194 collections: str | Sequence[str] | None = None,
1195 timespan: Timespan | None = None,
1196 storage_class: str | StorageClass | None = None,
1197 dimension_records: bool = False,
1198 datastore_records: bool = False,
1199 **kwargs: Any,
1200 ) -> DatasetRef | None:
1201 """Find a dataset given its `DatasetType` and data ID.
1203 This can be used to obtain a `DatasetRef` that permits the dataset to
1204 be read from a `Datastore`. If the dataset is a component and can not
1205 be found using the provided dataset type, a dataset ref for the parent
1206 will be returned instead but with the correct dataset type.
1208 Parameters
1209 ----------
1210 dataset_type : `DatasetType` or `str`
1211 A `DatasetType` or the name of one. If this is a `DatasetType`
1212 instance, its storage class will be respected and propagated to
1213 the output, even if it differs from the dataset type definition
1214 in the registry, as long as the storage classes are convertible.
1215 data_id : `dict` or `DataCoordinate`, optional
1216 A `dict`-like object containing the `Dimension` links that identify
1217 the dataset within a collection. If it is a `dict` the dataId
1218 can include dimension record values such as ``day_obs`` and
1219 ``seq_num`` or ``full_name`` that can be used to derive the
1220 primary dimension.
1221 collections : `str` or `list` [`str`], optional
1222 A an ordered list of collections to search for the dataset.
1223 Defaults to ``self.defaults.collections``.
1224 timespan : `Timespan`, optional
1225 A timespan that the validity range of the dataset must overlap.
1226 If not provided, any `~CollectionType.CALIBRATION` collections
1227 matched by the ``collections`` argument will not be searched.
1228 storage_class : `str` or `StorageClass` or `None`
1229 A storage class to use when creating the returned entry. If given
1230 it must be compatible with the default storage class.
1231 dimension_records : `bool`, optional
1232 If `True` the ref will be expanded and contain dimension records.
1233 datastore_records : `bool`, optional
1234 If `True` the ref will contain associated datastore records.
1235 **kwargs
1236 Additional keyword arguments passed to
1237 `DataCoordinate.standardize` to convert ``dataId`` to a true
1238 `DataCoordinate` or augment an existing one. This can also include
1239 dimension record metadata that can be used to derive a primary
1240 dimension value.
1242 Returns
1243 -------
1244 ref : `DatasetRef`
1245 A reference to the dataset, or `None` if no matching Dataset
1246 was found.
1248 Raises
1249 ------
1250 lsst.daf.butler.NoDefaultCollectionError
1251 Raised if ``collections`` is `None` and
1252 ``self.collections`` is `None`.
1253 LookupError
1254 Raised if one or more data ID keys are missing.
1255 lsst.daf.butler.MissingDatasetTypeError
1256 Raised if the dataset type does not exist.
1257 lsst.daf.butler.MissingCollectionError
1258 Raised if any of ``collections`` does not exist in the registry.
1260 Notes
1261 -----
1262 This method simply returns `None` and does not raise an exception even
1263 when the set of collections searched is intrinsically incompatible with
1264 the dataset type, e.g. if ``datasetType.isCalibration() is False``, but
1265 only `~CollectionType.CALIBRATION` collections are being searched.
1266 This may make it harder to debug some lookup failures, but the behavior
1267 is intentional; we consider it more important that failed searches are
1268 reported consistently, regardless of the reason, and that adding
1269 additional collections that do not contain a match to the search path
1270 never changes the behavior.
1272 This method handles component dataset types automatically, though most
1273 other query operations do not.
1274 """
1275 raise NotImplementedError()
1277 @abstractmethod
1278 def retrieve_artifacts_zip(
1279 self,
1280 refs: Iterable[DatasetRef],
1281 destination: ResourcePathExpression,
1282 overwrite: bool = True,
1283 ) -> ResourcePath:
1284 """Retrieve artifacts from a Butler and place in ZIP file.
1286 Parameters
1287 ----------
1288 refs : `~collections.abc.Iterable` [ `DatasetRef` ]
1289 The datasets to be included in the zip file.
1290 destination : `lsst.resources.ResourcePathExpression`
1291 Directory to write the new ZIP file. This directory will
1292 also be used as a staging area for the datasets being downloaded
1293 from the datastore.
1294 overwrite : `bool`, optional
1295 If `False` the output Zip will not be written if a file of the
1296 same name is already present in ``destination``.
1298 Returns
1299 -------
1300 zip_file : `lsst.resources.ResourcePath`
1301 The path to the new ZIP file.
1303 Raises
1304 ------
1305 ValueError
1306 Raised if there are no refs to retrieve.
1307 """
1308 raise NotImplementedError()
1310 @abstractmethod
1311 def retrieveArtifacts(
1312 self,
1313 refs: Iterable[DatasetRef],
1314 destination: ResourcePathExpression,
1315 transfer: str = "auto",
1316 preserve_path: bool = True,
1317 overwrite: bool = False,
1318 ) -> list[ResourcePath]:
1319 """Retrieve the artifacts associated with the supplied refs.
1321 Parameters
1322 ----------
1323 refs : `~collections.abc.Iterable` of `DatasetRef`
1324 The datasets for which artifacts are to be retrieved.
1325 A single ref can result in multiple artifacts. The refs must
1326 be resolved.
1327 destination : `lsst.resources.ResourcePath` or `str`
1328 Location to write the artifacts.
1329 transfer : `str`, optional
1330 Method to use to transfer the artifacts. Must be one of the options
1331 supported by `~lsst.resources.ResourcePath.transfer_from`.
1332 "move" is not allowed.
1333 preserve_path : `bool`, optional
1334 If `True` the full path of the artifact within the datastore
1335 is preserved. If `False` the final file component of the path
1336 is used.
1337 overwrite : `bool`, optional
1338 If `True` allow transfers to overwrite existing files at the
1339 destination.
1341 Returns
1342 -------
1343 targets : `list` of `lsst.resources.ResourcePath`
1344 URIs of file artifacts in destination location. Order is not
1345 preserved.
1347 Notes
1348 -----
1349 For non-file datastores the artifacts written to the destination
1350 may not match the representation inside the datastore. For example
1351 a hierarchical data structure in a NoSQL database may well be stored
1352 as a JSON file.
1353 """
1354 raise NotImplementedError()
1356 @abstractmethod
1357 def exists(
1358 self,
1359 dataset_ref_or_type: DatasetRef | DatasetType | str,
1360 /,
1361 data_id: DataId | None = None,
1362 *,
1363 full_check: bool = True,
1364 collections: Any = None,
1365 **kwargs: Any,
1366 ) -> DatasetExistence:
1367 """Indicate whether a dataset is known to Butler registry and
1368 datastore.
1370 Parameters
1371 ----------
1372 dataset_ref_or_type : `DatasetRef`, `DatasetType`, or `str`
1373 When `DatasetRef` the `dataId` should be `None`.
1374 Otherwise the `DatasetType` or name thereof.
1375 data_id : `dict` or `DataCoordinate`
1376 A `dict` of `Dimension` link name, value pairs that label the
1377 `DatasetRef` within a Collection. When `None`, a `DatasetRef`
1378 should be provided as the first argument.
1379 full_check : `bool`, optional
1380 If `True`, a check will be made for the actual existence of a
1381 dataset artifact. This will involve additional overhead due to
1382 the need to query an external system. If `False`, this check will
1383 be omitted, and the registry and datastore will solely be asked
1384 if they know about the dataset but no direct check for the
1385 artifact will be performed.
1386 collections : Any, optional
1387 Collections to be searched, overriding ``self.collections``.
1388 Can be any of the types supported by the ``collections`` argument
1389 to butler construction.
1390 **kwargs
1391 Additional keyword arguments used to augment or construct a
1392 `DataCoordinate`. See `DataCoordinate.standardize`
1393 parameters.
1395 Returns
1396 -------
1397 existence : `DatasetExistence`
1398 Object indicating whether the dataset is known to registry and
1399 datastore. Evaluates to `True` if the dataset is present and known
1400 to both.
1401 """
1402 raise NotImplementedError()
1404 @abstractmethod
1405 def _exists_many(
1406 self,
1407 refs: Iterable[DatasetRef],
1408 /,
1409 *,
1410 full_check: bool = True,
1411 ) -> dict[DatasetRef, DatasetExistence]:
1412 """Indicate whether multiple datasets are known to Butler registry and
1413 datastore.
1415 This is an experimental API that may change at any moment.
1417 Parameters
1418 ----------
1419 refs : `~collections.abc.Iterable` of `DatasetRef`
1420 The datasets to be checked.
1421 full_check : `bool`, optional
1422 If `True`, a check will be made for the actual existence of each
1423 dataset artifact. This will involve additional overhead due to
1424 the need to query an external system. If `False`, this check will
1425 be omitted, and the registry and datastore will solely be asked
1426 if they know about the dataset(s) but no direct check for the
1427 artifact(s) will be performed.
1429 Returns
1430 -------
1431 existence : `dict` [`DatasetRef`, `DatasetExistence`]
1432 Mapping from the given dataset refs to an enum indicating the
1433 status of the dataset in registry and datastore.
1434 Each value evaluates to `True` if the dataset is present and known
1435 to both.
1436 """
1437 raise NotImplementedError()
1439 @abstractmethod
1440 def removeRuns(
1441 self,
1442 names: Iterable[str],
1443 unstore: bool | type[_DeprecatedDefault] = _DeprecatedDefault,
1444 *,
1445 unlink_from_chains: bool = False,
1446 ) -> None:
1447 """Remove one or more `~CollectionType.RUN` collections and the
1448 datasets within them.
1450 Parameters
1451 ----------
1452 names : `~collections.abc.Iterable` [ `str` ]
1453 The names of the collections to remove.
1454 unstore : `bool`, optional
1455 If `True` (default), delete datasets from all datastores in which
1456 they are present, and attempt to rollback the registry deletions if
1457 datastore deletions fail (which may not always be possible). If
1458 `False`, datastore records for these datasets are still removed,
1459 but any artifacts (e.g. files) will not be. This parameter is now
1460 deprecated and no longer has any effect. Files are always deleted
1461 from datastores unless they were ingested using full URIs.
1462 unlink_from_chains : `bool`, optional
1463 If `True` remove the RUN collection from any chains prior to
1464 removing the RUN. If `False` the removal will fail if any chains
1465 still refer to the RUN.
1467 Raises
1468 ------
1469 TypeError
1470 Raised if one or more collections are not of type
1471 `~CollectionType.RUN`.
1472 """
1473 raise NotImplementedError()
1475 @abstractmethod
1476 def ingest(
1477 self,
1478 *datasets: FileDataset,
1479 transfer: str | None = "auto",
1480 record_validation_info: bool = True,
1481 skip_existing: bool = False,
1482 ) -> None:
1483 """Store and register one or more datasets that already exist on disk.
1485 Parameters
1486 ----------
1487 *datasets : `FileDataset`
1488 Each positional argument is a struct containing information about
1489 a file to be ingested, including its URI (either absolute or
1490 relative to the datastore root, if applicable), a resolved
1491 `DatasetRef`, and optionally a formatter class or its
1492 fully-qualified string name. If a formatter is not provided, the
1493 formatter that would be used for `put` is assumed. On successful
1494 ingest all `FileDataset.formatter` attributes will be set to the
1495 formatter class used. `FileDataset.path` attributes may be modified
1496 to put paths in whatever the datastore considers a standardized
1497 form.
1498 transfer : `str`, optional
1499 If not `None`, must be one of 'auto', 'move', 'copy', 'direct',
1500 'split', 'hardlink', 'relsymlink' or 'symlink', indicating how to
1501 transfer the file.
1502 record_validation_info : `bool`, optional
1503 If `True`, the default, the datastore can record validation
1504 information associated with the file. If `False` the datastore
1505 will not attempt to track any information such as checksums
1506 or file sizes. This can be useful if such information is tracked
1507 in an external system or if the file is to be compressed in place.
1508 It is up to the datastore whether this parameter is relevant.
1509 skip_existing : `bool`, optional
1510 If `True`, a dataset will not be ingested if a dataset with the
1511 same dataset ID already exists in the datastore.
1512 If `False` (the default), a `ConflictingDefinitionError` will be
1513 raised if any datasets with the same dataset ID already exist
1514 in the datastore.
1516 Returns
1517 -------
1518 None
1520 Raises
1521 ------
1522 TypeError
1523 Raised if the butler is read-only or if no run was provided.
1524 NotImplementedError
1525 Raised if the `Datastore` does not support the given transfer mode.
1526 DatasetTypeNotSupportedError
1527 Raised if one or more files to be ingested have a dataset type that
1528 is not supported by the `Datastore`..
1529 FileNotFoundError
1530 Raised if one of the given files does not exist.
1531 FileExistsError
1532 Raised if transfer is not `None` but the (internal) location the
1533 file would be moved to is already occupied.
1534 ConflictingDefinitionError
1535 Raised if a dataset already exists in the repository and
1536 ``skip_existing`` is `False`.
1538 Notes
1539 -----
1540 This operation is not fully exception safe: if a database operation
1541 fails, the given `FileDataset` instances may be only partially updated.
1543 It is atomic in terms of database operations (they will either all
1544 succeed or all fail) providing the database engine implements
1545 transactions correctly. It will attempt to be atomic in terms of
1546 filesystem operations as well, but this cannot be implemented
1547 rigorously for most datastores.
1548 """
1549 raise NotImplementedError()
1551 @abstractmethod
1552 def ingest_zip(
1553 self,
1554 zip_file: ResourcePathExpression,
1555 transfer: str = "auto",
1556 *,
1557 transfer_dimensions: bool = False,
1558 dry_run: bool = False,
1559 skip_existing: bool = False,
1560 ) -> None:
1561 """Ingest a Zip file into this butler.
1563 The Zip file must have been created by `retrieve_artifacts_zip`.
1565 Parameters
1566 ----------
1567 zip_file : `lsst.resources.ResourcePathExpression`
1568 Path to the Zip file.
1569 transfer : `str`, optional
1570 Method to use to transfer the Zip into the datastore.
1571 transfer_dimensions : `bool`, optional
1572 If `True`, dimension record data associated with the new datasets
1573 will be transferred from the Zip file, if present.
1574 dry_run : `bool`, optional
1575 If `True` the ingest will be processed without any modifications
1576 made to the target butler and as if the target butler did not
1577 have any of the datasets.
1578 skip_existing : `bool`, optional
1579 If `True`, a zip will not be ingested if the dataset entries listed
1580 in the index with the same dataset ID already exists in the butler.
1581 If `False` (the default), a `ConflictingDefinitionError` will be
1582 raised if any datasets with the same dataset ID already exist
1583 in the repository. If, somehow, some datasets are known to the
1584 butler and some are not, this is currently treated as an error
1585 rather than attempting to do a partial ingest.
1587 Notes
1588 -----
1589 Run collections and dataset types are created as needed.
1590 """
1591 raise NotImplementedError()
1593 @abstractmethod
1594 def export(
1595 self,
1596 *,
1597 directory: str | None = None,
1598 filename: str | None = None,
1599 format: str | None = None,
1600 transfer: str | None = None,
1601 ) -> AbstractContextManager[RepoExportContext]:
1602 """Export datasets from the repository represented by this `Butler`.
1604 This method is a context manager that returns a helper object
1605 (`RepoExportContext`) that is used to indicate what information from
1606 the repository should be exported.
1608 Parameters
1609 ----------
1610 directory : `str`, optional
1611 Directory dataset files should be written to if ``transfer`` is not
1612 `None`.
1613 filename : `str`, optional
1614 Name for the file that will include database information associated
1615 with the exported datasets. If this is not an absolute path and
1616 ``directory`` is not `None`, it will be written to ``directory``
1617 instead of the current working directory. Defaults to
1618 "export.{format}".
1619 format : `str`, optional
1620 File format for the database information file. If `None`, the
1621 extension of ``filename`` will be used.
1622 transfer : `str`, optional
1623 Transfer mode passed to `Datastore.export`.
1625 Raises
1626 ------
1627 TypeError
1628 Raised if the set of arguments passed is inconsistent.
1630 Examples
1631 --------
1632 Typically the `Registry.queryDataIds` and `Registry.queryDatasets`
1633 methods are used to provide the iterables over data IDs and/or datasets
1634 to be exported::
1636 with butler.export("exports.yaml") as export:
1637 # Export all flats, but none of the dimension element rows
1638 # (i.e. data ID information) associated with them.
1639 export.saveDatasets(
1640 butler.registry.queryDatasets("flat"), elements=()
1641 )
1642 # Export all datasets that start with "deepCoadd_" and all of
1643 # their associated data ID information.
1644 export.saveDatasets(butler.registry.queryDatasets("deepCoadd_*"))
1645 """
1646 raise NotImplementedError()
1648 @abstractmethod
1649 def import_(
1650 self,
1651 *,
1652 directory: ResourcePathExpression | None = None,
1653 filename: ResourcePathExpression | TextIO | None = None,
1654 format: str | None = None,
1655 transfer: str | None = None,
1656 skip_dimensions: set | None = None,
1657 record_validation_info: bool = True,
1658 without_datastore: bool = False,
1659 ) -> None:
1660 """Import datasets into this repository that were exported from a
1661 different butler repository via `~lsst.daf.butler.Butler.export`.
1663 Parameters
1664 ----------
1665 directory : `~lsst.resources.ResourcePathExpression`, optional
1666 Directory containing dataset files to import from. If `None`,
1667 ``filename`` and all dataset file paths specified therein must
1668 be absolute.
1669 filename : `~lsst.resources.ResourcePathExpression` or `typing.TextIO`
1670 A stream or name of file that contains database information
1671 associated with the exported datasets, typically generated by
1672 `~lsst.daf.butler.Butler.export`. If this a string (name) or
1673 `~lsst.resources.ResourcePath` and is not an absolute path,
1674 it will first be looked for relative to ``directory`` and if not
1675 found there it will be looked for in the current working
1676 directory. Defaults to "export.{format}".
1677 format : `str`, optional
1678 File format for ``filename``. If `None`, the extension of
1679 ``filename`` will be used.
1680 transfer : `str`, optional
1681 Transfer mode passed to `~lsst.daf.butler.Datastore.ingest`.
1682 skip_dimensions : `set`, optional
1683 Names of dimensions that should be skipped and not imported.
1684 record_validation_info : `bool`, optional
1685 If `True`, the default, the datastore can record validation
1686 information associated with the file. If `False` the datastore
1687 will not attempt to track any information such as checksums
1688 or file sizes. This can be useful if such information is tracked
1689 in an external system or if the file is to be compressed in place.
1690 It is up to the datastore whether this parameter is relevant.
1691 without_datastore : `bool`, optional
1692 If `True` only registry records will be imported and the datastore
1693 will be ignored.
1695 Raises
1696 ------
1697 TypeError
1698 Raised if the set of arguments passed is inconsistent, or if the
1699 butler is read-only.
1700 """
1701 raise NotImplementedError()
1703 @abstractmethod
1704 def transfer_dimension_records_from(
1705 self, source_butler: LimitedButler | Butler, source_refs: Iterable[DatasetRef | DataCoordinate]
1706 ) -> None:
1707 """Transfer dimension records to this Butler from another Butler.
1709 Parameters
1710 ----------
1711 source_butler : `LimitedButler` or `Butler`
1712 Butler from which the records are to be transferred. If data IDs
1713 in ``source_refs`` are not expanded then this has to be a full
1714 `Butler` whose registry will be used to expand data IDs. If the
1715 source refs contain coordinates that are used to populate other
1716 records then this will also need to be a full `Butler`.
1717 source_refs : `~collections.abc.Iterable` [`DatasetRef` |\
1718 `DataCoordinate`]
1719 Datasets or data IDs defined in the source butler whose dimension
1720 records should be transferred to this butler.
1721 """
1722 raise NotImplementedError()
1724 @abstractmethod
1725 def transfer_from(
1726 self,
1727 source_butler: LimitedButler,
1728 source_refs: Iterable[DatasetRef],
1729 transfer: str = "auto",
1730 skip_missing: bool = True,
1731 register_dataset_types: bool = False,
1732 transfer_dimensions: bool = False,
1733 dry_run: bool = False,
1734 ) -> Collection[DatasetRef]:
1735 """Transfer datasets to this Butler from a run in another Butler.
1737 Parameters
1738 ----------
1739 source_butler : `LimitedButler`
1740 Butler from which the datasets are to be transferred. If data IDs
1741 in ``source_refs`` are not expanded then this has to be a full
1742 `Butler` whose registry will be used to expand data IDs.
1743 source_refs : `~collections.abc.Iterable` of `DatasetRef`
1744 Datasets defined in the source butler that should be transferred to
1745 this butler. In most circumstances, ``transfer_from`` is faster if
1746 the dataset refs are expanded.
1747 transfer : `str`, optional
1748 Transfer mode passed to `~lsst.daf.butler.Datastore.transfer_from`.
1749 skip_missing : `bool`
1750 If `True`, datasets with no datastore artifact associated with
1751 them are not transferred. If `False` a registry entry will be
1752 created even if no datastore record is created (and so will
1753 look equivalent to the dataset being unstored).
1754 register_dataset_types : `bool`
1755 If `True` any missing dataset types are registered. Otherwise
1756 an exception is raised.
1757 transfer_dimensions : `bool`, optional
1758 If `True`, dimension record data associated with the new datasets
1759 will be transferred.
1760 dry_run : `bool`, optional
1761 If `True` the transfer will be processed without any modifications
1762 made to the target butler and as if the target butler did not
1763 have any of the datasets.
1765 Returns
1766 -------
1767 refs : `list` of `DatasetRef`
1768 The refs added to this Butler.
1770 Notes
1771 -----
1772 The datastore artifact has to exist for a transfer
1773 to be made but non-existence is not an error.
1775 Datasets that already exist in this run will be skipped.
1777 The datasets are imported as part of a transaction, although
1778 dataset types are registered before the transaction is started.
1779 This means that it is possible for a dataset type to be registered
1780 even though transfer has failed.
1781 """
1782 raise NotImplementedError()
1784 @abstractmethod
1785 def validateConfiguration(
1786 self,
1787 logFailures: bool = False,
1788 datasetTypeNames: Iterable[str] | None = None,
1789 ignore: Iterable[str] | None = None,
1790 ) -> None:
1791 """Validate butler configuration.
1793 Checks that each `DatasetType` can be stored in the `Datastore`.
1795 Parameters
1796 ----------
1797 logFailures : `bool`, optional
1798 If `True`, output a log message for every validation error
1799 detected.
1800 datasetTypeNames : `~collections.abc.Iterable` of `str`, optional
1801 The `DatasetType` names that should be checked. This allows
1802 only a subset to be selected.
1803 ignore : `~collections.abc.Iterable` of `str`, optional
1804 Names of DatasetTypes to skip over. This can be used to skip
1805 known problems. If a named `DatasetType` corresponds to a
1806 composite, all components of that `DatasetType` will also be
1807 ignored.
1809 Raises
1810 ------
1811 ButlerValidationError
1812 Raised if there is some inconsistency with how this Butler
1813 is configured.
1814 """
1815 raise NotImplementedError()
1817 @property
1818 @abstractmethod
1819 def collection_chains(self) -> ButlerCollections:
1820 """Object with methods for modifying collection chains
1821 (`~lsst.daf.butler.ButlerCollections`).
1823 Deprecated. Replaced with ``collections`` property.
1824 """
1825 raise NotImplementedError()
1827 @property
1828 @abstractmethod
1829 def collections(self) -> ButlerCollections:
1830 """Object with methods for modifying and querying collections
1831 (`~lsst.daf.butler.ButlerCollections`).
1833 Use of this object is preferred over `registry` wherever possible.
1834 """
1835 raise NotImplementedError()
1837 @property
1838 @abstractmethod
1839 def run(self) -> str | None:
1840 """Name of the run this butler writes outputs to by default (`str` or
1841 `None`).
1842 """
1843 raise NotImplementedError()
1845 @property
1846 @abstractmethod
1847 def registry(self) -> Registry:
1848 """The object that manages dataset metadata and relationships
1849 (`Registry`).
1851 Many operations that don't involve reading or writing butler datasets
1852 are accessible only via `Registry` methods. Eventually these methods
1853 will be replaced by equivalent `Butler` methods.
1854 """
1855 raise NotImplementedError()
1857 @abstractmethod
1858 def query(self) -> AbstractContextManager[Query]:
1859 """Context manager returning a `.queries.Query` object used for
1860 construction and execution of complex queries.
1861 """
1862 raise NotImplementedError()
1864 def query_data_ids(
1865 self,
1866 dimensions: DimensionGroup | Iterable[str] | str,
1867 *,
1868 data_id: DataId | None = None,
1869 where: str = "",
1870 bind: Mapping[str, Any] | None = None,
1871 with_dimension_records: bool = False,
1872 order_by: Iterable[str] | str | None = None,
1873 limit: int | None = -20_000,
1874 explain: bool = True,
1875 **kwargs: Any,
1876 ) -> list[DataCoordinate]:
1877 """Query for data IDs matching user-provided criteria.
1879 Parameters
1880 ----------
1881 dimensions : `DimensionGroup`, `str`, or \
1882 `~collections.abc.Iterable` [`str`]
1883 The dimensions of the data IDs to yield, as either `DimensionGroup`
1884 instances or `str`. Will be automatically expanded to a complete
1885 `DimensionGroup`.
1886 data_id : `dict` or `DataCoordinate`, optional
1887 A data ID whose key-value pairs are used as equality constraints
1888 in the query.
1889 where : `str`, optional
1890 A string expression similar to a SQL WHERE clause. May involve
1891 any column of a dimension table or (as a shortcut for the primary
1892 key column of a dimension table) dimension name. See
1893 :ref:`daf_butler_dimension_expressions` for more information.
1894 bind : `~collections.abc.Mapping`, optional
1895 Mapping containing literal values that should be injected into the
1896 ``where`` expression, keyed by the identifiers they replace.
1897 Values of collection type can be expanded in some cases; see
1898 :ref:`daf_butler_dimension_expressions_identifiers` for more
1899 information.
1900 with_dimension_records : `bool`, optional
1901 If `True` (default is `False`) then returned data IDs will have
1902 dimension records.
1903 order_by : `~collections.abc.Iterable` [`str`] or `str`, optional
1904 Names of the columns/dimensions to use for ordering returned data
1905 IDs. Column name can be prefixed with minus (``-``) to use
1906 descending ordering.
1907 limit : `int` or `None`, optional
1908 Upper limit on the number of returned records. `None` can be used
1909 if no limit is wanted. A limit of ``0`` means that the query will
1910 be executed and validated but no results will be returned. In this
1911 case there will be no exception even if ``explain`` is `True`.
1912 If a negative value is given a warning will be issued if the number
1913 of results is capped by that limit.
1914 explain : `bool`, optional
1915 If `True` (default) then `EmptyQueryResultError` exception is
1916 raised when resulting list is empty. The exception contains
1917 non-empty list of strings explaining possible causes for empty
1918 result.
1919 **kwargs
1920 Additional keyword arguments are forwarded to
1921 `DataCoordinate.standardize` when processing the ``data_id``
1922 argument (and may be used to provide a constraining data ID even
1923 when the ``data_id`` argument is `None`).
1925 Returns
1926 -------
1927 dataIds : `list` [`DataCoordinate`]
1928 Data IDs matching the given query parameters. These are always
1929 guaranteed to identify all dimensions (`DataCoordinate.hasFull`
1930 returns `True`).
1932 Raises
1933 ------
1934 lsst.daf.butler.registry.DataIdError
1935 Raised when ``data_id`` or keyword arguments specify unknown
1936 dimensions or values, or when they contain inconsistent values.
1937 lsst.daf.butler.registry.UserExpressionError
1938 Raised when ``where`` expression is invalid.
1939 lsst.daf.butler.EmptyQueryResultError
1940 Raised when query generates empty result and ``explain`` is set to
1941 `True`.
1942 TypeError
1943 Raised when the arguments are incompatible.
1944 """
1945 if data_id is None: 1945 ↛ 1947line 1945 didn't jump to line 1947 because the condition on line 1945 was always true
1946 data_id = DataCoordinate.make_empty(self.dimensions)
1947 if order_by is None:
1948 order_by = []
1949 query_limit = limit
1950 warn_limit = False
1951 if limit is not None and limit < 0:
1952 query_limit = abs(limit) + 1
1953 warn_limit = True
1954 with self.query() as query:
1955 result = (
1956 query.data_ids(dimensions)
1957 .where(data_id, where, bind=bind, **kwargs)
1958 .order_by(*ensure_iterable(order_by))
1959 .limit(query_limit)
1960 )
1961 if with_dimension_records: 1961 ↛ 1962line 1961 didn't jump to line 1962 because the condition on line 1961 was never true
1962 result = result.with_dimension_records()
1963 data_ids = list(result)
1964 if warn_limit and len(data_ids) == query_limit:
1965 # We asked for one too many so must remove that from the list.
1966 data_ids.pop(-1)
1967 assert limit is not None # For mypy.
1968 _LOG.warning("More data IDs are available than the requested limit of %d.", abs(limit))
1969 if explain and (limit is None or limit != 0) and not data_ids: 1969 ↛ 1970line 1969 didn't jump to line 1970 because the condition on line 1969 was never true
1970 raise EmptyQueryResultError(list(result.explain_no_results()))
1971 return data_ids
1973 def query_datasets(
1974 self,
1975 dataset_type: str | DatasetType,
1976 collections: str | Iterable[str] | None = None,
1977 *,
1978 find_first: bool = True,
1979 data_id: DataId | None = None,
1980 where: str = "",
1981 bind: Mapping[str, Any] | None = None,
1982 with_dimension_records: bool = False,
1983 order_by: Iterable[str] | str | None = None,
1984 limit: int | None = -20_000,
1985 explain: bool = True,
1986 **kwargs: Any,
1987 ) -> list[DatasetRef]:
1988 """Query for dataset references matching user-provided criteria.
1990 Parameters
1991 ----------
1992 dataset_type : `str` or `DatasetType`
1993 Dataset type object or name to search for.
1994 collections : collection expression, optional
1995 A collection name or iterable of collection names to search. If not
1996 provided, the default collections are used. Can be a wildcard if
1997 ``find_first`` is `False` (if find first is requested the order
1998 of collections matters and wildcards make the order indeterminate).
1999 See :ref:`daf_butler_collection_expressions` for more information.
2000 find_first : `bool`, optional
2001 If `True` (default), for each result data ID, only yield one
2002 `DatasetRef` of each `DatasetType`, from the first collection in
2003 which a dataset of that dataset type appears (according to the
2004 order of ``collections`` passed in). If `True`, ``collections``
2005 must not contain wildcards.
2006 data_id : `dict` or `DataCoordinate`, optional
2007 A data ID whose key-value pairs are used as equality constraints in
2008 the query.
2009 where : `str`, optional
2010 A string expression similar to a SQL WHERE clause. May involve any
2011 column of a dimension table or (as a shortcut for the primary key
2012 column of a dimension table) dimension name. See
2013 :ref:`daf_butler_dimension_expressions` for more information.
2014 bind : `~collections.abc.Mapping`, optional
2015 Mapping containing literal values that should be injected into the
2016 ``where`` expression, keyed by the identifiers they replace. Values
2017 of collection type can be expanded in some cases; see
2018 :ref:`daf_butler_dimension_expressions_identifiers` for more
2019 information.
2020 with_dimension_records : `bool`, optional
2021 If `True` (default is `False`) then returned data IDs will have
2022 dimension records.
2023 order_by : `~collections.abc.Iterable` [`str`] or `str`, optional
2024 Names of the columns/dimensions to use for ordering returned data
2025 IDs. Column name can be prefixed with minus (``-``) to use
2026 descending ordering.
2027 limit : `int` or `None`, optional
2028 Upper limit on the number of returned records. `None` can be used
2029 if no limit is wanted. A limit of ``0`` means that the query will
2030 be executed and validated but no results will be returned. In this
2031 case there will be no exception even if ``explain`` is `True`.
2032 If a negative value is given a warning will be issued if the number
2033 of results is capped by that limit.
2034 explain : `bool`, optional
2035 If `True` (default) then `EmptyQueryResultError` exception is
2036 raised when resulting list is empty. The exception contains
2037 non-empty list of strings explaining possible causes for empty
2038 result.
2039 **kwargs
2040 Additional keyword arguments are forwarded to
2041 `DataCoordinate.standardize` when processing the ``data_id``
2042 argument (and may be used to provide a constraining data ID even
2043 when the ``data_id`` argument is `None`).
2045 Returns
2046 -------
2047 refs : `.queries.DatasetRefQueryResults`
2048 Dataset references matching the given query criteria. Nested data
2049 IDs are guaranteed to include values for all implied dimensions
2050 (i.e. `DataCoordinate.hasFull` will return `True`).
2052 Raises
2053 ------
2054 lsst.daf.butler.DatasetTypeExpressionError
2055 Raised when ``dataset_type`` expression is invalid.
2056 lsst.daf.butler.registry.DataIdError
2057 Raised when ``data_id`` or keyword arguments specify unknown
2058 dimensions or values, or when they contain inconsistent values.
2059 lsst.daf.butler.registry.UserExpressionError
2060 Raised when ``where`` expression is invalid.
2061 lsst.daf.butler.EmptyQueryResultError
2062 Raised when query generates empty result and ``explain`` is set to
2063 `True`.
2064 TypeError
2065 Raised when the arguments are incompatible, such as when a
2066 collection wildcard is passed when ``find_first`` is `True`, or
2067 when ``collections`` is `None` and default butler collections are
2068 not defined.
2069 """
2070 if data_id is None: 2070 ↛ 2072line 2070 didn't jump to line 2072 because the condition on line 2070 was always true
2071 data_id = DataCoordinate.make_empty(self.dimensions)
2072 if order_by is None:
2073 order_by = []
2074 if collections and has_globs(collections):
2075 # Wild cards need to be expanded but can only be allowed if
2076 # find_first=False because expanding wildcards does not return
2077 # a guaranteed ordering. Querying collection registry to expand
2078 # collections when we do not have wildcards is expensive so only
2079 # do it if we need it.
2080 if find_first:
2081 raise InvalidQueryError(
2082 f"Can not use wildcards in collections when find_first=True (given {collections})"
2083 )
2084 collections = self.collections.query(collections)
2085 query_limit = limit
2086 warn_limit = False
2087 if limit is not None and limit < 0:
2088 query_limit = abs(limit) + 1
2089 warn_limit = True
2090 with self.query() as query:
2091 result = (
2092 query.datasets(dataset_type, collections=collections, find_first=find_first)
2093 .where(data_id, where, bind=bind, **kwargs)
2094 .order_by(*ensure_iterable(order_by))
2095 .limit(query_limit)
2096 )
2097 if with_dimension_records:
2098 result = result.with_dimension_records()
2099 refs = list(result)
2100 if warn_limit and len(refs) == query_limit:
2101 # We asked for one too many so must remove that from the list.
2102 refs.pop(-1)
2103 assert limit is not None # For mypy.
2104 _LOG.warning("More datasets are available than the requested limit of %d.", abs(limit))
2105 if explain and (limit is None or limit != 0) and not refs:
2106 raise EmptyQueryResultError(list(result.explain_no_results()))
2107 return refs
2109 def query_dimension_records(
2110 self,
2111 element: str,
2112 *,
2113 data_id: DataId | None = None,
2114 where: str = "",
2115 bind: Mapping[str, Any] | None = None,
2116 order_by: Iterable[str] | str | None = None,
2117 limit: int | None = -20_000,
2118 explain: bool = True,
2119 **kwargs: Any,
2120 ) -> list[DimensionRecord]:
2121 """Query for dimension information matching user-provided criteria.
2123 Parameters
2124 ----------
2125 element : `str`
2126 The name of a dimension element to obtain records for.
2127 data_id : `dict` or `DataCoordinate`, optional
2128 A data ID whose key-value pairs are used as equality constraints
2129 in the query.
2130 where : `str`, optional
2131 A string expression similar to a SQL WHERE clause. See
2132 `Registry.queryDataIds` and :ref:`daf_butler_dimension_expressions`
2133 for more information.
2134 bind : `~collections.abc.Mapping`, optional
2135 Mapping containing literal values that should be injected into the
2136 ``where`` expression, keyed by the identifiers they replace.
2137 Values of collection type can be expanded in some cases; see
2138 :ref:`daf_butler_dimension_expressions_identifiers` for more
2139 information.
2140 order_by : `~collections.abc.Iterable` [`str`] or `str`, optional
2141 Names of the columns/dimensions to use for ordering returned data
2142 IDs. Column name can be prefixed with minus (``-``) to use
2143 descending ordering.
2144 limit : `int` or `None`, optional
2145 Upper limit on the number of returned records. `None` can be used
2146 if no limit is wanted. A limit of ``0`` means that the query will
2147 be executed and validated but no results will be returned. In this
2148 case there will be no exception even if ``explain`` is `True`.
2149 If a negative value is given a warning will be issued if the number
2150 of results is capped by that limit.
2151 explain : `bool`, optional
2152 If `True` (default) then `EmptyQueryResultError` exception is
2153 raised when resulting list is empty. The exception contains
2154 non-empty list of strings explaining possible causes for empty
2155 result.
2156 **kwargs
2157 Additional keyword arguments are forwarded to
2158 `DataCoordinate.standardize` when processing the ``data_id``
2159 argument (and may be used to provide a constraining data ID even
2160 when the ``data_id`` argument is `None`).
2162 Returns
2163 -------
2164 records : `list` [`DimensionRecord`]
2165 Dimension records matching the given query parameters.
2167 Raises
2168 ------
2169 lsst.daf.butler.registry.DataIdError
2170 Raised when ``data_id`` or keyword arguments specify unknown
2171 dimensions or values, or when they contain inconsistent values.
2172 lsst.daf.butler.registry.UserExpressionError
2173 Raised when ``where`` expression is invalid.
2174 lsst.daf.butler.EmptyQueryResultError
2175 Raised when query generates empty result and ``explain`` is set to
2176 `True`.
2177 TypeError
2178 Raised when the arguments are incompatible, such as when a
2179 collection wildcard is passed when ``find_first`` is `True`, or
2180 when ``collections`` is `None` and default butler collections are
2181 not defined.
2182 """
2183 if data_id is None:
2184 data_id = DataCoordinate.make_empty(self.dimensions)
2185 if order_by is None:
2186 order_by = []
2187 query_limit = limit
2188 warn_limit = False
2189 if limit is not None and limit < 0:
2190 query_limit = abs(limit) + 1
2191 warn_limit = True
2192 with self.query() as query:
2193 result = (
2194 query.dimension_records(element)
2195 .where(data_id, where, bind=bind, **kwargs)
2196 .order_by(*ensure_iterable(order_by))
2197 .limit(query_limit)
2198 )
2199 dimension_records = list(result)
2200 if warn_limit and len(dimension_records) == query_limit:
2201 # We asked for one too many so must remove that from the list.
2202 dimension_records.pop(-1)
2203 assert limit is not None # For mypy.
2204 _LOG.warning(
2205 "More dimension records are available than the requested limit of %d.", abs(limit)
2206 )
2207 if explain and (limit is None or limit != 0) and not dimension_records:
2208 raise EmptyQueryResultError(list(result.explain_no_results()))
2209 return dimension_records
2211 def query_all_datasets(
2212 self,
2213 collections: str | Iterable[str] | None = None,
2214 *,
2215 name: str | Iterable[str] = "*",
2216 find_first: bool = True,
2217 data_id: DataId | None = None,
2218 where: str = "",
2219 bind: Mapping[str, Any] | None = None,
2220 limit: int | None = -20_000,
2221 **kwargs: Any,
2222 ) -> list[DatasetRef]:
2223 """Query for datasets of potentially multiple types.
2225 Parameters
2226 ----------
2227 collections : `str` or `~collections.abc.Iterable` [ `str` ], optional
2228 The collection or collections to search, in order. If not provided
2229 or `None`, the default collection search path for this butler is
2230 used.
2231 name : `str` or `~collections.abc.Iterable` [ `str` ], optional
2232 Names or name patterns (glob-style) that returned dataset type
2233 names must match. If an iterable, items are OR'd together. The
2234 default is to include all dataset types in the given collections.
2235 find_first : `bool`, optional
2236 If `True` (default), for each result data ID, only yield one
2237 `DatasetRef` of each `DatasetType`, from the first collection in
2238 which a dataset of that dataset type appears (according to the
2239 order of ``collections`` passed in).
2240 data_id : `dict` or `DataCoordinate`, optional
2241 A data ID whose key-value pairs are used as equality constraints in
2242 the query.
2243 where : `str`, optional
2244 A string expression similar to a SQL WHERE clause. May involve any
2245 column of a dimension table or (as a shortcut for the primary key
2246 column of a dimension table) dimension name. See
2247 :ref:`daf_butler_dimension_expressions` for more information.
2248 bind : `~collections.abc.Mapping`, optional
2249 Mapping containing literal values that should be injected into the
2250 ``where`` expression, keyed by the identifiers they replace. Values
2251 of collection type can be expanded in some cases; see
2252 :ref:`daf_butler_dimension_expressions_identifiers` for more
2253 information.
2254 limit : `int` or `None`, optional
2255 Upper limit on the number of returned records. `None` can be used
2256 if no limit is wanted. A limit of ``0`` means that the query will
2257 be executed and validated but no results will be returned.
2258 If a negative value is given a warning will be issued if the number
2259 of results is capped by that limit. If no limit is provided, by
2260 default a maximum of 20,000 records will be returned.
2261 **kwargs
2262 Additional keyword arguments are forwarded to
2263 `DataCoordinate.standardize` when processing the ``data_id``
2264 argument (and may be used to provide a constraining data ID even
2265 when the ``data_id`` argument is `None`).
2267 Raises
2268 ------
2269 MissingDatasetTypeError
2270 When no dataset types match ``name``, or an explicit (non-glob)
2271 dataset type in ``name`` does not exist.
2272 InvalidQueryError
2273 If the parameters to the query are inconsistent or malformed.
2274 MissingCollectionError
2275 If a given collection is not found.
2277 Returns
2278 -------
2279 refs : `list` [ `DatasetRef` ]
2280 Dataset references matching the given query criteria. Nested data
2281 IDs are guaranteed to include values for all implied dimensions
2282 (i.e. `DataCoordinate.hasFull` will return `True`), but will not
2283 include dimension records (`DataCoordinate.hasRecords` will be
2284 `False`).
2285 """
2286 if collections is None:
2287 collections = list(self.collections.defaults)
2288 else:
2289 collections = list(ensure_iterable(collections))
2291 if bind is None:
2292 bind = {}
2293 if data_id is None:
2294 data_id = {}
2296 warn_limit = False
2297 if limit is not None and limit < 0:
2298 # Add one to the limit so we can detect if we have exceeded it.
2299 limit = abs(limit) + 1
2300 warn_limit = True
2302 args = QueryAllDatasetsParameters(
2303 collections=collections,
2304 name=list(ensure_iterable(name)),
2305 find_first=find_first,
2306 data_id=data_id,
2307 where=where,
2308 limit=limit,
2309 bind=bind,
2310 kwargs=kwargs,
2311 with_dimension_records=False,
2312 )
2313 with self._query_all_datasets_by_page(args) as pages:
2314 result = []
2315 for page in pages:
2316 result.extend(page)
2318 if warn_limit and limit is not None and len(result) >= limit:
2319 # Remove the extra dataset we added for the limit check.
2320 result.pop()
2321 _LOG.warning("More datasets are available than the requested limit of %d.", limit - 1)
2323 return result
2325 @abstractmethod
2326 def _query_all_datasets_by_page(
2327 self, args: QueryAllDatasetsParameters
2328 ) -> AbstractContextManager[Iterator[list[DatasetRef]]]:
2329 raise NotImplementedError()
2331 def clone(
2332 self,
2333 *,
2334 collections: CollectionArgType | None | EllipsisType = ...,
2335 run: str | None | EllipsisType = ...,
2336 inferDefaults: bool | EllipsisType = ...,
2337 dataId: dict[str, str] | EllipsisType = ...,
2338 metrics: ButlerMetrics | None = None,
2339 ) -> Butler:
2340 """Return a new Butler instance connected to the same repository
2341 as this one, optionally overriding ``collections``, ``run``,
2342 ``inferDefaults``, and default data ID.
2344 Parameters
2345 ----------
2346 collections : `~lsst.daf.butler.registry.CollectionArgType` or `None`,\
2347 optional
2348 Same as constructor. If omitted, uses value from original object.
2349 run : `str` or `None`, optional
2350 Same as constructor. If `None`, no default run is used. If
2351 omitted, copies value from original object.
2352 inferDefaults : `bool`, optional
2353 Same as constructor. If omitted, copies value from original
2354 object.
2355 dataId : `str`
2356 Same as ``kwargs`` passed to the constructor. If omitted, copies
2357 values from original object.
2358 metrics : `ButlerMetrics` or `None`, optional
2359 Metrics object to record butler statistics.
2360 """
2361 raise NotImplementedError()
2363 @abstractmethod
2364 def close(self) -> None:
2365 raise NotImplementedError()
2367 @abstractmethod
2368 def _expand_data_ids(self, data_ids: Iterable[DataCoordinate]) -> list[DataCoordinate]:
2369 raise NotImplementedError()