Coverage for python/lsst/daf/butler/registry/_registry_base.py: 100%
79 statements
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-16 09:11 +0000
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-16 09:11 +0000
1# This file is part of daf_butler.
2#
3# Developed for the LSST Data Management System.
4# This product includes software developed by the LSST Project
5# (http://www.lsst.org).
6# See the COPYRIGHT file at the top-level directory of this distribution
7# for details of code ownership.
8#
9# This software is dual licensed under the GNU General Public License and also
10# under a 3-clause BSD license. Recipients may choose which of these licenses
11# to use; please see the files gpl-3.0.txt and/or bsd_license.txt,
12# respectively. If you choose the GPL option then the following text applies
13# (but note that there is still no warranty even if you opt for BSD instead):
14#
15# This program is free software: you can redistribute it and/or modify
16# it under the terms of the GNU General Public License as published by
17# the Free Software Foundation, either version 3 of the License, or
18# (at your option) any later version.
19#
20# This program is distributed in the hope that it will be useful,
21# but WITHOUT ANY WARRANTY; without even the implied warranty of
22# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
23# GNU General Public License for more details.
24#
25# You should have received a copy of the GNU General Public License
26# along with this program. If not, see <http://www.gnu.org/licenses/>.
28from __future__ import annotations
30__all__ = ("RegistryBase",)
32from collections.abc import Iterable, Iterator, Mapping
33from typing import Any
35from lsst.utils.iteration import ensure_iterable
37from .._butler import Butler
38from .._collection_type import CollectionType
39from .._dataset_association import DatasetAssociation
40from .._dataset_type import DatasetType
41from .._exceptions import DatasetTypeExpressionError
42from ..dimensions import DataId, DimensionElement, DimensionGroup
43from ..registry.wildcards import CollectionWildcard, DatasetTypeWildcard
44from ._exceptions import ArgumentError, NoDefaultCollectionError
45from ._registry import CollectionArgType, Registry
46from .queries import (
47 ChainedDatasetQueryResults,
48 DataCoordinateQueryResults,
49 DatasetQueryResults,
50 DimensionRecordQueryResults,
51)
52from .queries._query_common import CommonQueryArguments, resolve_collections
53from .queries._query_data_coordinates import QueryDriverDataCoordinateQueryResults
54from .queries._query_datasets import QueryDriverDatasetRefQueryResults
55from .queries._query_dimension_records import QueryDriverDimensionRecordQueryResults
58class RegistryBase(Registry):
59 """Common implementation for `Registry` methods shared between
60 DirectButler's RegistryShim and RemoteButlerRegistry.
62 Parameters
63 ----------
64 butler : `Butler`
65 Butler instance to which this registry delegates operations.
66 """
68 def __init__(self, butler: Butler) -> None:
69 self._butler = butler
71 def queryDatasets(
72 self,
73 datasetType: Any,
74 *,
75 collections: CollectionArgType | None = None,
76 dimensions: Iterable[str] | None = None,
77 dataId: DataId | None = None,
78 where: str = "",
79 findFirst: bool = False,
80 components: bool = False,
81 bind: Mapping[str, Any] | None = None,
82 check: bool = True,
83 **kwargs: Any,
84 ) -> DatasetQueryResults:
85 doomed_by: list[str] = []
86 dimension_group = self.dimensions.conform(dimensions) if dimensions is not None else None
88 if collections is None and not self.defaults.collections:
89 raise NoDefaultCollectionError("No collections provided, and no default collections set")
90 if findFirst and collections is not None:
91 wildcard = CollectionWildcard.from_expression(collections)
92 if wildcard.patterns:
93 raise TypeError(
94 "Collection search patterns not allowed in findFirst search, "
95 "because collections must be in a specific order."
96 )
98 args = self._convert_common_query_arguments(
99 dataId=dataId,
100 where=where,
101 bind=bind,
102 kwargs=kwargs,
103 datasets=None,
104 collections=collections,
105 doomed_by=doomed_by,
106 check=check,
107 )
109 if not args.collections:
110 doomed_by.append("No datasets can be found because collection list is empty.")
112 missing_dataset_types: list[str] = []
113 dataset_types = list(self.queryDatasetTypes(datasetType, missing=missing_dataset_types))
114 if missing_dataset_types:
115 doomed_by.extend(f"Dataset type {name} is not registered." for name in missing_dataset_types)
117 if len(dataset_types) == 0:
118 doomed_by.extend(
119 [
120 f"No registered dataset type matching {t!r} found, so no matching datasets can "
121 "exist in any collection."
122 for t in ensure_iterable(datasetType)
123 ]
124 )
125 return ChainedDatasetQueryResults([], doomed_by=doomed_by)
127 query_results = [
128 QueryDriverDatasetRefQueryResults(
129 self._butler,
130 args,
131 dataset_type=dt,
132 find_first=findFirst,
133 extra_dimensions=dimension_group,
134 doomed_by=doomed_by,
135 expanded=False,
136 )
137 for dt in dataset_types
138 ]
139 if len(query_results) == 1:
140 return query_results[0]
141 else:
142 return ChainedDatasetQueryResults(query_results)
144 def queryDataIds(
145 self,
146 dimensions: DimensionGroup | Iterable[str] | str,
147 *,
148 dataId: DataId | None = None,
149 datasets: Any = None,
150 collections: CollectionArgType | None = None,
151 where: str = "",
152 components: bool = False,
153 bind: Mapping[str, Any] | None = None,
154 check: bool = True,
155 **kwargs: Any,
156 ) -> DataCoordinateQueryResults:
157 if collections is not None and datasets is None:
158 raise ArgumentError(f"Cannot pass 'collections' (='{collections}') without 'datasets'.")
160 dimensions = self.dimensions.conform(dimensions)
161 args = self._convert_common_query_arguments(
162 dataId=dataId,
163 where=where,
164 bind=bind,
165 kwargs=kwargs,
166 datasets=datasets,
167 collections=collections,
168 check=check,
169 )
170 return QueryDriverDataCoordinateQueryResults(
171 self._butler, dimensions=dimensions, expanded=False, args=args
172 )
174 def queryDimensionRecords(
175 self,
176 element: DimensionElement | str,
177 *,
178 dataId: DataId | None = None,
179 datasets: Any = None,
180 collections: CollectionArgType | None = None,
181 where: str = "",
182 components: bool = False,
183 bind: Mapping[str, Any] | None = None,
184 check: bool = True,
185 **kwargs: Any,
186 ) -> DimensionRecordQueryResults:
187 if not isinstance(element, DimensionElement):
188 element = self.dimensions.elements[element]
190 args = self._convert_common_query_arguments(
191 dataId=dataId,
192 where=where,
193 bind=bind,
194 kwargs=kwargs,
195 datasets=datasets,
196 collections=collections,
197 check=check,
198 )
200 return QueryDriverDimensionRecordQueryResults(self._butler, element, args)
202 def _convert_common_query_arguments(
203 self,
204 *,
205 dataId: DataId | None = None,
206 datasets: object | None = None,
207 collections: CollectionArgType | None = None,
208 where: str = "",
209 bind: Mapping[str, Any] | None = None,
210 kwargs: dict[str, int | str],
211 doomed_by: list[str] | None = None,
212 check: bool = True,
213 ) -> CommonQueryArguments:
214 dataset_types = self._resolve_dataset_types(datasets)
215 if dataset_types and collections is None and not self.defaults.collections:
216 raise NoDefaultCollectionError("'collections' must be provided if 'datasets' is provided")
217 return CommonQueryArguments(
218 dataId=dataId,
219 where=where,
220 bind=dict(bind) if bind else None,
221 kwargs=dict(kwargs),
222 dataset_types=dataset_types,
223 collections=resolve_collections(self._butler, collections, doomed_by),
224 check=check,
225 )
227 def queryDatasetAssociations(
228 self,
229 datasetType: str | DatasetType,
230 collections: CollectionArgType | None = ...,
231 *,
232 collectionTypes: Iterable[CollectionType] = CollectionType.all(),
233 flattenChains: bool = False,
234 ) -> Iterator[DatasetAssociation]:
235 if isinstance(datasetType, str):
236 datasetType = self.getDatasetType(datasetType)
237 with self._butler.query() as query:
238 resolved_collections = self.queryCollections(
239 collections,
240 collectionTypes=collectionTypes,
241 flattenChains=True,
242 )
243 # It's annoyingly difficult to just do the collection query once,
244 # since query_info doesn't accept all the expression types that
245 # queryCollections does. But it's all cached anyway.
246 collection_info = {
247 info.name: info for info in self._butler.collections.query_info(resolved_collections)
248 }
249 query = query.join_dataset_search(datasetType, resolved_collections)
250 result = query.general(
251 datasetType.dimensions,
252 dataset_fields={datasetType.name: {"dataset_id", "run", "collection", "timespan"}},
253 find_first=False,
254 )
255 yield from DatasetAssociation.from_query_result(result, datasetType, collection_info)
257 def _resolve_dataset_types(self, dataset_types: object | None) -> list[str]:
258 if dataset_types is None:
259 return []
261 if dataset_types is ...:
262 raise TypeError(
263 "'...' not permitted for 'datasets'"
264 " -- searching for all dataset types does not constrain the search."
265 )
267 wildcard = DatasetTypeWildcard.from_expression(dataset_types)
268 if wildcard.patterns:
269 raise DatasetTypeExpressionError(
270 "Dataset type wildcard expressions are not supported in this context."
271 )
272 else:
273 return list(wildcard.values.keys())