Coverage for python/lsst/daf/butler/registry/_registry_base.py: 100%

79 statements  

« prev     ^ index     » next       coverage.py v7.16.0, created at 2026-09-19 09:05 +0000

1# This file is part of daf_butler. 

2# 

3# Developed for the LSST Data Management System. 

4# This product includes software developed by the LSST Project 

5# (http://www.lsst.org). 

6# See the COPYRIGHT file at the top-level directory of this distribution 

7# for details of code ownership. 

8# 

9# This software is dual licensed under the GNU General Public License and also 

10# under a 3-clause BSD license. Recipients may choose which of these licenses 

11# to use; please see the files gpl-3.0.txt and/or bsd_license.txt, 

12# respectively. If you choose the GPL option then the following text applies 

13# (but note that there is still no warranty even if you opt for BSD instead): 

14# 

15# This program is free software: you can redistribute it and/or modify 

16# it under the terms of the GNU General Public License as published by 

17# the Free Software Foundation, either version 3 of the License, or 

18# (at your option) any later version. 

19# 

20# This program is distributed in the hope that it will be useful, 

21# but WITHOUT ANY WARRANTY; without even the implied warranty of 

22# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the 

23# GNU General Public License for more details. 

24# 

25# You should have received a copy of the GNU General Public License 

26# along with this program. If not, see <http://www.gnu.org/licenses/>. 

27 

28from __future__ import annotations 

29 

30__all__ = ("RegistryBase",) 

31 

32from collections.abc import Iterable, Iterator, Mapping 

33from typing import Any 

34 

35from lsst.utils.iteration import ensure_iterable 

36 

37from .._butler import Butler 

38from .._collection_type import CollectionType 

39from .._dataset_association import DatasetAssociation 

40from .._dataset_type import DatasetType 

41from .._exceptions import DatasetTypeExpressionError 

42from ..dimensions import DataId, DimensionElement, DimensionGroup 

43from ..registry.wildcards import CollectionWildcard, DatasetTypeWildcard 

44from ._exceptions import ArgumentError, NoDefaultCollectionError 

45from ._registry import CollectionArgType, Registry 

46from .queries import ( 

47 ChainedDatasetQueryResults, 

48 DataCoordinateQueryResults, 

49 DatasetQueryResults, 

50 DimensionRecordQueryResults, 

51) 

52from .queries._query_common import CommonQueryArguments, resolve_collections 

53from .queries._query_data_coordinates import QueryDriverDataCoordinateQueryResults 

54from .queries._query_datasets import QueryDriverDatasetRefQueryResults 

55from .queries._query_dimension_records import QueryDriverDimensionRecordQueryResults 

56 

57 

58class RegistryBase(Registry): 

59 """Common implementation for `Registry` methods shared between 

60 DirectButler's RegistryShim and RemoteButlerRegistry. 

61 

62 Parameters 

63 ---------- 

64 butler : `Butler` 

65 Butler instance to which this registry delegates operations. 

66 """ 

67 

68 def __init__(self, butler: Butler) -> None: 

69 self._butler = butler 

70 

71 def queryDatasets( 

72 self, 

73 datasetType: Any, 

74 *, 

75 collections: CollectionArgType | None = None, 

76 dimensions: Iterable[str] | None = None, 

77 dataId: DataId | None = None, 

78 where: str = "", 

79 findFirst: bool = False, 

80 components: bool = False, 

81 bind: Mapping[str, Any] | None = None, 

82 check: bool = True, 

83 **kwargs: Any, 

84 ) -> DatasetQueryResults: 

85 doomed_by: list[str] = [] 

86 dimension_group = self.dimensions.conform(dimensions) if dimensions is not None else None 

87 

88 if collections is None and not self.defaults.collections: 

89 raise NoDefaultCollectionError("No collections provided, and no default collections set") 

90 if findFirst and collections is not None: 

91 wildcard = CollectionWildcard.from_expression(collections) 

92 if wildcard.patterns: 

93 raise TypeError( 

94 "Collection search patterns not allowed in findFirst search, " 

95 "because collections must be in a specific order." 

96 ) 

97 

98 args = self._convert_common_query_arguments( 

99 dataId=dataId, 

100 where=where, 

101 bind=bind, 

102 kwargs=kwargs, 

103 datasets=None, 

104 collections=collections, 

105 doomed_by=doomed_by, 

106 check=check, 

107 ) 

108 

109 if not args.collections: 

110 doomed_by.append("No datasets can be found because collection list is empty.") 

111 

112 missing_dataset_types: list[str] = [] 

113 dataset_types = list(self.queryDatasetTypes(datasetType, missing=missing_dataset_types)) 

114 if missing_dataset_types: 

115 doomed_by.extend(f"Dataset type {name} is not registered." for name in missing_dataset_types) 

116 

117 if len(dataset_types) == 0: 

118 doomed_by.extend( 

119 [ 

120 f"No registered dataset type matching {t!r} found, so no matching datasets can " 

121 "exist in any collection." 

122 for t in ensure_iterable(datasetType) 

123 ] 

124 ) 

125 return ChainedDatasetQueryResults([], doomed_by=doomed_by) 

126 

127 query_results = [ 

128 QueryDriverDatasetRefQueryResults( 

129 self._butler, 

130 args, 

131 dataset_type=dt, 

132 find_first=findFirst, 

133 extra_dimensions=dimension_group, 

134 doomed_by=doomed_by, 

135 expanded=False, 

136 ) 

137 for dt in dataset_types 

138 ] 

139 if len(query_results) == 1: 

140 return query_results[0] 

141 else: 

142 return ChainedDatasetQueryResults(query_results) 

143 

144 def queryDataIds( 

145 self, 

146 dimensions: DimensionGroup | Iterable[str] | str, 

147 *, 

148 dataId: DataId | None = None, 

149 datasets: Any = None, 

150 collections: CollectionArgType | None = None, 

151 where: str = "", 

152 components: bool = False, 

153 bind: Mapping[str, Any] | None = None, 

154 check: bool = True, 

155 **kwargs: Any, 

156 ) -> DataCoordinateQueryResults: 

157 if collections is not None and datasets is None: 

158 raise ArgumentError(f"Cannot pass 'collections' (='{collections}') without 'datasets'.") 

159 

160 dimensions = self.dimensions.conform(dimensions) 

161 args = self._convert_common_query_arguments( 

162 dataId=dataId, 

163 where=where, 

164 bind=bind, 

165 kwargs=kwargs, 

166 datasets=datasets, 

167 collections=collections, 

168 check=check, 

169 ) 

170 return QueryDriverDataCoordinateQueryResults( 

171 self._butler, dimensions=dimensions, expanded=False, args=args 

172 ) 

173 

174 def queryDimensionRecords( 

175 self, 

176 element: DimensionElement | str, 

177 *, 

178 dataId: DataId | None = None, 

179 datasets: Any = None, 

180 collections: CollectionArgType | None = None, 

181 where: str = "", 

182 components: bool = False, 

183 bind: Mapping[str, Any] | None = None, 

184 check: bool = True, 

185 **kwargs: Any, 

186 ) -> DimensionRecordQueryResults: 

187 if not isinstance(element, DimensionElement): 

188 element = self.dimensions.elements[element] 

189 

190 args = self._convert_common_query_arguments( 

191 dataId=dataId, 

192 where=where, 

193 bind=bind, 

194 kwargs=kwargs, 

195 datasets=datasets, 

196 collections=collections, 

197 check=check, 

198 ) 

199 

200 return QueryDriverDimensionRecordQueryResults(self._butler, element, args) 

201 

202 def _convert_common_query_arguments( 

203 self, 

204 *, 

205 dataId: DataId | None = None, 

206 datasets: object | None = None, 

207 collections: CollectionArgType | None = None, 

208 where: str = "", 

209 bind: Mapping[str, Any] | None = None, 

210 kwargs: dict[str, int | str], 

211 doomed_by: list[str] | None = None, 

212 check: bool = True, 

213 ) -> CommonQueryArguments: 

214 dataset_types = self._resolve_dataset_types(datasets) 

215 if dataset_types and collections is None and not self.defaults.collections: 

216 raise NoDefaultCollectionError("'collections' must be provided if 'datasets' is provided") 

217 return CommonQueryArguments( 

218 dataId=dataId, 

219 where=where, 

220 bind=dict(bind) if bind else None, 

221 kwargs=dict(kwargs), 

222 dataset_types=dataset_types, 

223 collections=resolve_collections(self._butler, collections, doomed_by), 

224 check=check, 

225 ) 

226 

227 def queryDatasetAssociations( 

228 self, 

229 datasetType: str | DatasetType, 

230 collections: CollectionArgType | None = ..., 

231 *, 

232 collectionTypes: Iterable[CollectionType] = CollectionType.all(), 

233 flattenChains: bool = False, 

234 ) -> Iterator[DatasetAssociation]: 

235 if isinstance(datasetType, str): 

236 datasetType = self.getDatasetType(datasetType) 

237 with self._butler.query() as query: 

238 resolved_collections = self.queryCollections( 

239 collections, 

240 collectionTypes=collectionTypes, 

241 flattenChains=True, 

242 ) 

243 # It's annoyingly difficult to just do the collection query once, 

244 # since query_info doesn't accept all the expression types that 

245 # queryCollections does. But it's all cached anyway. 

246 collection_info = { 

247 info.name: info for info in self._butler.collections.query_info(resolved_collections) 

248 } 

249 query = query.join_dataset_search(datasetType, resolved_collections) 

250 result = query.general( 

251 datasetType.dimensions, 

252 dataset_fields={datasetType.name: {"dataset_id", "run", "collection", "timespan"}}, 

253 find_first=False, 

254 ) 

255 yield from DatasetAssociation.from_query_result(result, datasetType, collection_info) 

256 

257 def _resolve_dataset_types(self, dataset_types: object | None) -> list[str]: 

258 if dataset_types is None: 

259 return [] 

260 

261 if dataset_types is ...: 

262 raise TypeError( 

263 "'...' not permitted for 'datasets'" 

264 " -- searching for all dataset types does not constrain the search." 

265 ) 

266 

267 wildcard = DatasetTypeWildcard.from_expression(dataset_types) 

268 if wildcard.patterns: 

269 raise DatasetTypeExpressionError( 

270 "Dataset type wildcard expressions are not supported in this context." 

271 ) 

272 else: 

273 return list(wildcard.values.keys())