Coverage for python/lsst/summit/utils/butlerUtils.py: 16%

282 statements  

« prev     ^ index     » next       coverage.py v7.16.0, created at 2026-09-16 03:29 -0700

1# This file is part of summit_utils. 

2# 

3# Developed for the LSST Data Management System. 

4# This product includes software developed by the LSST Project 

5# (https://www.lsst.org). 

6# See the COPYRIGHT file at the top-level directory of this distribution 

7# for details of code ownership. 

8# 

9# This program is free software: you can redistribute it and/or modify 

10# it under the terms of the GNU General Public License as published by 

11# the Free Software Foundation, either version 3 of the License, or 

12# (at your option) any later version. 

13# 

14# This program is distributed in the hope that it will be useful, 

15# but WITHOUT ANY WARRANTY; without even the implied warranty of 

16# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the 

17# GNU General Public License for more details. 

18# 

19# You should have received a copy of the GNU General Public License 

20# along with this program. If not, see <https://www.gnu.org/licenses/>. 

21 

22import copy 

23import itertools 

24import logging 

25from collections.abc import Iterable, Mapping 

26from typing import Any 

27 

28from deprecated.sphinx import deprecated 

29 

30import lsst.daf.butler as dafButler 

31from lsst.daf.butler.direct_butler import DirectButler 

32from lsst.summit.utils.utils import getSite 

33from lsst.utils.iteration import ensure_iterable 

34 

35__all__ = [ 

36 "makeDefaultLatissButler", # deprecated 

37 "makeDefaultButler", 

38 "updateDataId", 

39 "sanitizeDayObs", 

40 "getMostRecentDayObs", 

41 "getSeqNumsForDayObs", 

42 "getMostRecentDataId", 

43 "getDatasetRefForDataId", 

44 "getDayObs", 

45 "getSeqNum", 

46 "getExpId", 

47 "datasetExists", # deprecated 

48 "sortRecordsByDayObsThenSeqNum", 

49 "getDaysWithData", 

50 "getExpIdFromDayObsSeqNum", 

51 "updateDataIdOrDataCord", 

52 "fillDataId", 

53 "getExpRecordFromDataId", 

54 "getDayObsSeqNumFromExposureId", 

55 "removeDataProduct", 

56 "getLatissOnSkyDataIds", 

57 "getExpRecord", 

58] 

59 

60_LATISS_DEFAULT_COLLECTIONS = ["LATISS/raw/all", "LATISS/calib", "LATISS/runs/quickLook"] 

61 

62# RECENT_DAY must be in the past *and have data* (otherwise some tests are 

63# no-ops), to speed up queries by restricting them significantly, 

64# but data must definitely been taken since. Should 

65# also not be more than 2 months in the past due to 60 day lookback time on the 

66# summit. All this means it should be updated by an informed human. 

67RECENT_DAY = 20220503 

68 

69 

70def _configureForSite() -> None: 

71 try: 

72 site = getSite() 

73 except ValueError: 

74 # this method is run automatically on module import, so 

75 # don't fail for k8s where this cannot yet be determined 

76 print("WARNING: failed to automatically determine site") 

77 site = None 

78 if site == "tucson": 78 ↛ 80line 78 didn't jump to line 80 because the condition on line 78 was never true

79 global RECENT_DAY 

80 RECENT_DAY = 20211104 # TTS has limited data, so use this day 

81 

82 

83_configureForSite() 

84 

85 

86def getLatissDefaultCollections() -> list[str]: 

87 """Get the default set of LATISS collections, updated for the site at 

88 which the code is being run. 

89 

90 Returns 

91 ------- 

92 collections : `list` of `str` 

93 The default collections for the site. 

94 """ 

95 collections = _LATISS_DEFAULT_COLLECTIONS 

96 try: 

97 site = getSite() 

98 except ValueError: 

99 site = "" 

100 

101 if site == "tucson": 101 ↛ 102line 101 didn't jump to line 102 because the condition on line 101 was never true

102 collections.append("LATISS-test-data") 

103 return collections 

104 if site == "summit": 104 ↛ 105line 104 didn't jump to line 105 because the condition on line 104 was never true

105 collections.append("LATISS-test-data") 

106 return collections 

107 return collections 

108 

109 

110def _update_RECENT_DAY(day: int) -> None: 

111 """Update the value for RECENT_DAY once we have a value for free.""" 

112 global RECENT_DAY 

113 RECENT_DAY = max(day - 1, RECENT_DAY) 

114 

115 

116@deprecated( 

117 reason="Use the more generic makeDefaultButler('LATISS'). Will be removed after v28.0.", 

118 version="v27.0", 

119 category=FutureWarning, 

120) 

121def makeDefaultLatissButler( 

122 *, 

123 extraCollections: list[str] | None = None, 

124 writeable: bool = False, 

125 embargo: bool = False, 

126) -> dafButler.Butler: 

127 """Create a butler for LATISS using the default collections. 

128 

129 Parameters 

130 ---------- 

131 extraCollections : `list` of `str` 

132 Extra input collections to supply to the butler init. 

133 writable : `bool`, optional 

134 Whether to make a writable butler. 

135 embargo : `bool`, optional 

136 Use the embargo repo instead of the main one. Needed to access 

137 embargoed data. 

138 

139 Returns 

140 ------- 

141 butler : `lsst.daf.butler.Butler` 

142 The butler. 

143 """ 

144 return makeDefaultButler( 

145 "LATISS", extraCollections=extraCollections, writeable=writeable, embargo=embargo 

146 ) 

147 

148 

149def makeDefaultButler( 

150 instrument: str, 

151 *, 

152 extraCollections: list[str] | None = None, 

153 writeable: bool = False, 

154 embargo: bool = True, 

155) -> dafButler.Butler: 

156 """Create a butler for the instrument using default collections, regardless 

157 of the location. 

158 

159 Parameters 

160 ---------- 

161 extraCollections : `list` of `str` 

162 Extra input collections to supply to the butler init. 

163 writable : `bool`, optional 

164 Whether to make a writable butler. 

165 embargo : `bool`, optional 

166 Use the embargo repo instead of the main one. Needed to access 

167 embargoed data if not at a summit-like location. 

168 

169 Returns 

170 ------- 

171 butler : `lsst.daf.butler.Butler` 

172 The butler. 

173 

174 Raises 

175 ------ 

176 FileNotFoundError 

177 Raised if the butler cannot be created, because this is the error 

178 when the DAF_BUTLER_REPOSITORY_INDEX is not set correctly. 

179 """ 

180 SUPPORTED_SITES = [ 

181 "summit", 

182 "tucson", 

183 "base", 

184 "staff-rsp", 

185 "rubin-devl", 

186 "usdf-k8s", 

187 ] 

188 

189 site = getSite() 

190 if site not in SUPPORTED_SITES: 

191 # This might look like a slightly weird error to raise, but this is 

192 # the same error that's raised when the DAF_BUTLER_REPOSITORY_INDEX 

193 # isn't set, so it's the most appropriate error to raise here, i.e. 

194 # this is what would be raised if this function was allowed to continue 

195 # only this is a less confusing version of it. 

196 raise FileNotFoundError(f"Default butler creation only available at: {SUPPORTED_SITES}, got {site=}") 

197 

198 summitLike = site in ["summit", "tucson", "base"] 

199 if summitLike: 

200 if embargo is True: 

201 logger = logging.getLogger(__name__) 

202 logger.debug("embargo option is irrelevant on the summit, ignoring") 

203 embargo = False # there's only one repo too, so this makes the code more simple too 

204 

205 collections: list[str] = [f"{instrument}/defaults"] 

206 raCollection = ( 

207 [f"{instrument}/runs/quickLook"] if summitLike else [f"{instrument}/runs/nightlyValidation"] 

208 ) 

209 collections.extend(raCollection) 

210 if instrument == "LSSTCam": 

211 collections.append("LSSTCam/raw/guider") 

212 

213 repo = instrument if embargo is False else "embargo" 

214 

215 if extraCollections is not None: 

216 assert extraCollections is not None # just for mypy 

217 extraCollectionsList = ensure_iterable(extraCollections) 

218 collections.extend(extraCollectionsList) 

219 

220 try: 

221 butler = dafButler.Butler.from_config( 

222 repo, collections=collections, writeable=writeable, instrument=instrument 

223 ) 

224 except (FileNotFoundError, RuntimeError): 

225 # Depending on the value of DAF_BUTLER_REPOSITORY_INDEX and whether 

226 # it is present and blank, or just not set, both these exception 

227 # types can be raised, see tests/test_butlerUtils.py:ButlerInitTestCase 

228 # for details and tests which confirm these have not changed 

229 raise FileNotFoundError # unify exception type 

230 return butler 

231 

232 

233@deprecated( 

234 reason="datasExists has been replaced by Butler.exists(). Will be removed after v26.0.", 

235 version="v26.0", 

236 category=FutureWarning, 

237) 

238def datasetExists( 

239 butler: dafButler.Butler, dataProduct: str, dataId: dafButler.DataId, **kwargs: Any 

240) -> bool: 

241 """Collapse the tri-state behaviour of butler.datasetExists to a boolean. 

242 

243 Parameters 

244 ---------- 

245 butler : `lsst.daf.butler.Butler` 

246 The butler 

247 dataProduct : `str` 

248 The type of data product to check for 

249 dataId : `dafButler.DataId` 

250 The dataId of the dataProduct to check for 

251 

252 Returns 

253 ------- 

254 exists : `bool` 

255 True if the dataProduct exists for the dataProduct and can be retreived 

256 else False. 

257 """ 

258 return bool(butler.exists(dataProduct, dataId, **kwargs)) 

259 

260 

261def updateDataId(dataId: dafButler.DataId, **kwargs: Any) -> dafButler.DataId: 

262 """Update a DataCoordinate or dataId dict with kwargs. 

263 

264 Provides a single interface for adding the detector key (or others) to a 

265 dataId whether it's a DataCoordinate or a dict 

266 

267 Parameters 

268 ---------- 

269 dataId : `dafButler.DataId` 

270 The dataId to update. 

271 kwargs : `dict` 

272 The keys and values to add to the dataId. 

273 

274 Returns 

275 ------- 

276 dataId : `dict` or `lsst.daf.butler.DataCoordinate` 

277 The updated dataId, with the same type as the input. 

278 """ 

279 

280 match dataId: 

281 case dafButler.DataCoordinate(): 

282 return dafButler.DataCoordinate.standardize(dataId, **kwargs) 

283 case dict() as dataId: 

284 return dict(dataId, **kwargs) 

285 raise ValueError(f"Unknown dataId type {type(dataId)}") 

286 

287 

288def sanitizeDayObs(day_obs: int | str) -> int: 

289 """Take string or int day_obs and turn it into the int version. 

290 

291 Parameters 

292 ---------- 

293 day_obs : `str` or `int` 

294 The day_obs to sanitize. 

295 

296 Returns 

297 ------- 

298 day_obs : `int` 

299 The sanitized day_obs. 

300 

301 Raises 

302 ------ 

303 ValueError 

304 Raised if the day_obs fails to translate for any reason. 

305 """ 

306 if isinstance(day_obs, int): 

307 return day_obs 

308 elif isinstance(day_obs, str): 

309 try: 

310 return int(day_obs.replace("-", "")) 

311 except Exception: 

312 ValueError(f"Failed to sanitize {day_obs!r} to a day_obs") 

313 raise ValueError(f"Cannot sanitize {day_obs!r} to a day_obs") 

314 

315 

316def getMostRecentDayObs(butler: dafButler.Butler) -> int: 

317 """Get the most recent day_obs for which there is data. 

318 

319 Parameters 

320 ---------- 

321 butler : `lsst.daf.butler.Butler 

322 The butler to query. 

323 

324 Returns 

325 ------- 

326 day_obs : `int` 

327 The day_obs. 

328 """ 

329 where = "exposure.day_obs>=RECENT_DAY" 

330 records = butler.registry.queryDimensionRecords( 

331 "exposure", where=where, datasets="raw", bind={"RECENT_DAY": RECENT_DAY} 

332 ) 

333 recentDay = max(r.day_obs for r in records) 

334 _update_RECENT_DAY(recentDay) 

335 return recentDay 

336 

337 

338def getSeqNumsForDayObs(butler: dafButler.Butler, day_obs: int, extraWhere: str = "") -> list[int]: 

339 """Get a list of all seq_nums taken on a given day_obs. 

340 

341 Parameters 

342 ---------- 

343 butler : `lsst.daf.butler.Butler 

344 The butler to query. 

345 day_obs : `int` or `str` 

346 The day_obs for which the seq_nums are desired. 

347 extraWhere : `str` 

348 Any extra where conditions to add to the queryDimensionRecords call. 

349 

350 Returns 

351 ------- 

352 seq_nums : `list` of `int` 

353 The seq_nums taken on the corresponding day_obs in ascending numerical 

354 order. 

355 """ 

356 day_obs = sanitizeDayObs(day_obs) 

357 where = "exposure.day_obs=dayObs" 

358 if extraWhere: 

359 extraWhere = extraWhere.replace('"', "'") 

360 where += f" and {extraWhere}" 

361 records = butler.registry.queryDimensionRecords( 

362 "exposure", where=where, bind={"dayObs": day_obs}, datasets="raw" 

363 ) 

364 return sorted(set([r.seq_num for r in records])) 

365 

366 

367def sortRecordsByDayObsThenSeqNum( 

368 records: dafButler.registry.queries.DimensionRecordQueryResults | list[dafButler.DimensionRecord], 

369) -> list[dafButler.DimensionRecord]: 

370 """Sort a set of records by dayObs, then seqNum to get the order in which 

371 they were taken. 

372 

373 Parameters 

374 ---------- 

375 records : `list` of `dafButler.DimensionRecord` 

376 The records to be sorted. 

377 

378 Returns 

379 ------- 

380 sortedRecords : `list` of `dafButler.DimensionRecord` 

381 The sorted records 

382 

383 Raises 

384 ------ 

385 ValueError 

386 Raised if the recordSet contains duplicate records, or if it contains 

387 (dayObs, seqNum) collisions. 

388 """ 

389 records = list(records) # must call list in case we have a generator 

390 recordSet = set(records) 

391 if len(records) != len(recordSet): 

392 raise ValueError("Record set contains duplicate records and therefore cannot be sorted unambiguously") 

393 

394 daySeqTuples = [(r.day_obs, r.seq_num) for r in records] 

395 if len(daySeqTuples) != len(set(daySeqTuples)): 

396 raise ValueError( 

397 "Record set contains dayObs/seqNum collisions, and therefore cannot be sorted " "unambiguously" 

398 ) 

399 

400 records.sort(key=lambda r: (r.day_obs, r.seq_num)) 

401 return records 

402 

403 

404def getDaysWithData(butler: dafButler.Butler, datasetType: str = "raw") -> list[int]: 

405 """Get all the days for which LATISS has taken data on the mountain. 

406 

407 Parameters 

408 ---------- 

409 butler : `lsst.daf.butler.Butler 

410 The butler to query. 

411 datasetType : `str` 

412 The datasetType to query. 

413 

414 Returns 

415 ------- 

416 days : `list` of `int` 

417 A sorted list of the day_obs values for which mountain-top data exists. 

418 """ 

419 # 20200101 is a day between shipping LATISS and going on sky 

420 # We used to constrain on exposure.seq_num<50 to massively reduce the 

421 # number of returned records whilst being large enough to ensure that no 

422 # days are missed because early seq_nums were skipped. However, because 

423 # we have test datasets like LATISS-test-data-tts where we only kept 

424 # seqNums from 950 on one day, we can no longer assume this so don't be 

425 # tempted to add such a constraint back in here for speed. 

426 where = "exposure.day_obs>20200101" 

427 records = butler.registry.queryDimensionRecords("exposure", where=where, datasets=datasetType) 

428 return sorted(set([r.day_obs for r in records])) 

429 

430 

431def getMostRecentDataId(butler: dafButler.Butler) -> dict[str, Any]: 

432 """Get the dataId for the most recent observation. 

433 

434 Parameters 

435 ---------- 

436 butler : `lsst.daf.butler.Butler 

437 The butler to query. 

438 

439 Returns 

440 ------- 

441 dataId : `dict[str, Any]` 

442 The dataId of the most recent exposure. 

443 """ 

444 lastDay = getMostRecentDayObs(butler) 

445 seqNum = getSeqNumsForDayObs(butler, lastDay)[-1] 

446 dataId = {"day_obs": lastDay, "seq_num": seqNum, "detector": 0} 

447 dataId.update(getExpIdFromDayObsSeqNum(butler, dataId)) 

448 return dataId 

449 

450 

451def getExpIdFromDayObsSeqNum(butler: dafButler.Butler, dataId: dafButler.DataId) -> dict[str, int]: 

452 """Get the exposure id for the dataId. 

453 

454 Parameters 

455 ---------- 

456 butler : `lsst.daf.butler.Butler 

457 The butler to query. 

458 dataId : `dafButler.DataId` 

459 The dataId for which to return the exposure id. 

460 

461 Returns 

462 ------- 

463 dataId : `dict[str, int]` 

464 The dataId of the most recent exposure. 

465 """ 

466 expRecord = getExpRecordFromDataId(butler, dataId) 

467 return {"exposure": expRecord.id} 

468 

469 

470def updateDataIdOrDataCord( 

471 dataId: dafButler.DataId | dafButler.DimensionRecord, **updateKwargs: Any 

472) -> Mapping[str, Any]: 

473 """Add key, value pairs to a dataId or data coordinate. 

474 

475 Parameters 

476 ---------- 

477 dataId : `dafButler.DataId` 

478 The dataId for which to return the exposure id. 

479 updateKwargs : `dict[str, Any]` 

480 The key value pairs add to the dataId or dataCoord. 

481 

482 Returns 

483 ------- 

484 dataId : `Mapping[str, Any]` 

485 The updated dataId. 

486 

487 Notes 

488 ----- 

489 Always returns a dict, so note that if a data coordinate is supplied, a 

490 dict is returned, changing the type. 

491 """ 

492 newId = copy.copy(dataId) 

493 newId = _assureDict(newId) 

494 newId.update(updateKwargs) 

495 return newId 

496 

497 

498def fillDataId( 

499 butler: DirectButler, dataId: dafButler.DataId | dafButler.DimensionRecord 

500) -> Mapping[str, Any]: 

501 """Given a dataId, fill it with values for all available dimensions. 

502 

503 Parameters 

504 ---------- 

505 butler : `lsst.daf.butler.Butler` 

506 The butler. 

507 dataId : `dafButler.DataId` 

508 The dataId to fill. 

509 

510 Returns 

511 ------- 

512 dataId : `Mapping[str, Any]` 

513 The filled dataId. 

514 

515 Notes 

516 ----- 

517 This function is *slow*! Running this on 20,000 dataIds takes approximately 

518 7 minutes. Virtually all the slowdown is in the 

519 butler.registry.expandDataId() call though, so this wrapper is not to blame 

520 here, and might speed up in future with butler improvements. 

521 """ 

522 # ensure it's a dict to deal with records etc 

523 dictId = _assureDict(dataId) 

524 

525 # this removes extraneous keys that would trip up the registry call 

526 # using _rewrite_data_id is perhaps ever so slightly slower than popping 

527 # the bad keys, or making a minimal dataId by hand, but is more 

528 # reliable/general, so we choose that over the other approach here 

529 realDataId, _ = butler._rewrite_data_id(dictId, butler.get_dataset_type("raw")) 

530 

531 # now expand and turn back to a dict - this call is VERY slow 

532 expandedDictDataId = dict(butler.registry.expandDataId(realDataId, detector=0).mapping) 

533 

534 missingExpId = getExpId(expandedDictDataId) is None 

535 missingDayObs = getDayObs(expandedDictDataId) is None 

536 missingSeqNum = getSeqNum(expandedDictDataId) is None 

537 

538 if missingDayObs or missingSeqNum: 

539 dayObsSeqNum = getDayObsSeqNumFromExposureId(butler, expandedDictDataId) 

540 expandedDictDataId.update(dayObsSeqNum) 

541 

542 if missingExpId: 

543 expId = getExpIdFromDayObsSeqNum(butler, expandedDictDataId) 

544 expandedDictDataId.update(expId) 

545 

546 return expandedDictDataId 

547 

548 

549def _assureDict( 

550 dataId: dafButler.DataId | dafButler.DimensionRecord, 

551) -> dict[str, Any]: 

552 """Turn any data-identifier-like object into a dict. 

553 

554 Parameters 

555 ---------- 

556 dataId : `dafButler.DataId` or 

557 `lsst.daf.butler.dimensions.DimensionRecord` 

558 The data identifier. 

559 

560 Returns 

561 ------- 

562 dataId : `dict[str, Any]` 

563 The data identifier as a dict. 

564 """ 

565 if isinstance(dataId, dict): 

566 return dataId 

567 elif hasattr(dataId, "mapping"): # dafButler.DataCoordinate 

568 return {str(k): v for k, v in dataId.mapping.items()} 

569 elif hasattr(dataId, "dataId"): # dafButler.DimensionRecord 

570 return {str(k): v for k, v in dataId.dataId.mapping.items()} 

571 elif hasattr(dataId, "keys"): # some other mapping, assume str keys 

572 return dict(dataId) 

573 else: 

574 raise RuntimeError(f"Failed to coerce {type(dataId)} to dict") 

575 

576 

577def getExpRecordFromDataId(butler: dafButler.Butler, dataId: dafButler.DataId) -> dafButler.DimensionRecord: 

578 """Get the exposure record for a given dataId. 

579 

580 Parameters 

581 ---------- 

582 butler : `lsst.daf.butler.Butler` 

583 The butler. 

584 dataId : `dafButler.DataId` 

585 The dataId. 

586 

587 Returns 

588 ------- 

589 expRecord : `lsst.daf.butler.dimensions.DimensionRecord` 

590 The exposure record. 

591 """ 

592 dataId = _assureDict(dataId) 

593 assert isinstance(dataId, dict), f"dataId must be a dict or DimensionRecord, got {type(dataId)}" 

594 

595 if expId := getExpId(dataId): 

596 where = "exposure.id=expId" 

597 expRecords = butler.registry.queryDimensionRecords( 

598 "exposure", where=where, bind={"expId": expId}, datasets="raw" 

599 ) 

600 

601 else: 

602 dayObs = getDayObs(dataId) 

603 seqNum = getSeqNum(dataId) 

604 if not (dayObs and seqNum): 

605 raise RuntimeError(f"Failed to find either expId or day_obs and seq_num in dataId {dataId}") 

606 where = "exposure.day_obs=dayObs AND exposure.seq_num=seq_num" 

607 expRecords = butler.registry.queryDimensionRecords( 

608 "exposure", where=where, bind={"dayObs": dayObs, "seq_num": seqNum}, datasets="raw" 

609 ) 

610 

611 filteredExpRecords = set(expRecords) 

612 if not filteredExpRecords: 

613 raise LookupError(f"No exposure records found for {dataId}") 

614 assert len(filteredExpRecords) == 1, f"Found {len(filteredExpRecords)} exposure records for {dataId}" 

615 return filteredExpRecords.pop() 

616 

617 

618def getDayObsSeqNumFromExposureId(butler: dafButler.Butler, dataId: Mapping[str, Any]) -> dict[str, int]: 

619 """Get the day_obs and seq_num for an exposure id. 

620 

621 Parameters 

622 ---------- 

623 butler : `lsst.daf.butler.Butler` 

624 The butler. 

625 dataId : ` Mapping[str, Any]` 

626 The dataId containing the exposure id. 

627 

628 Returns 

629 ------- 

630 dataId : ` dict[str, int]` 

631 A dict containing only the day_obs and seq_num. 

632 """ 

633 if (dayObs := getDayObs(dataId)) and (seqNum := getSeqNum(dataId)): 

634 return {"day_obs": dayObs, "seq_num": seqNum} 

635 

636 if isinstance(dataId, int): 

637 dataId = {"exposure": dataId} 

638 else: 

639 dataId = _assureDict(dataId) 

640 assert isinstance(dataId, dict) 

641 

642 if not (expId := getExpId(dataId)): 

643 raise RuntimeError(f"Failed to find exposure id in {dataId}") 

644 

645 where = "exposure.id=expId" 

646 expRecords = list( 

647 butler.registry.queryDimensionRecords("exposure", where=where, bind={"expId": expId}, datasets="raw") 

648 ) 

649 uniqueExpRecords = set(expRecords) 

650 if not uniqueExpRecords: 

651 raise LookupError(f"No exposure records found for {dataId}") 

652 assert len(uniqueExpRecords) == 1, f"Found {len(uniqueExpRecords)} exposure records for {dataId}" 

653 record = uniqueExpRecords.pop() 

654 return {"day_obs": record.day_obs, "seq_num": record.seq_num} 

655 

656 

657def getDatasetRefForDataId( 

658 butler: dafButler.Butler, datasetType: str | dafButler.DatasetType, dataId: dafButler.DataId 

659) -> dafButler.DatasetRef | None: 

660 """Get the datasetReference for a dataId. 

661 

662 Parameters 

663 ---------- 

664 butler : `lsst.daf.butler.Butler` 

665 The butler. 

666 datasetType : `str` or `datasetType` 

667 The dataset type. 

668 dataId : `dafButler.DataId` 

669 The dataId. 

670 

671 Returns 

672 ------- 

673 datasetRef : `lsst.daf.butler.dimensions.DatasetReference` 

674 The dataset reference. 

675 """ 

676 dictId = _assureDict(dataId) 

677 if not _expid_present(dictId): 

678 assert _dayobs_present(dictId) and _seqnum_present(dictId) 

679 dictId.update(getExpIdFromDayObsSeqNum(butler, dictId)) 

680 

681 dRef = butler.find_dataset(datasetType, dictId) 

682 return dRef 

683 

684 

685def removeDataProduct( 

686 butler: dafButler.Butler, datasetType: str | dafButler.DatasetType, dataId: dict[str, Any] 

687) -> None: 

688 """Remove a data prodcut from the registry. Use with caution. 

689 

690 Parameters 

691 ---------- 

692 butler : `lsst.daf.butler.Butler` 

693 The butler. 

694 datasetType : `str` or `datasetType` 

695 The dataset type. 

696 dataId : `dict[str, Any]` 

697 The dataId. 

698 

699 """ 

700 if datasetType == "raw": 

701 raise RuntimeError("I'm sorry, Dave, I'm afraid I can't do that.") 

702 dRef = getDatasetRefForDataId(butler, datasetType, dataId) 

703 if dRef is None: 

704 return 

705 butler.pruneDatasets([dRef], disassociate=True, unstore=True, purge=True) 

706 return 

707 

708 

709def _dayobs_present(dataId: Mapping[str, Any]) -> bool: 

710 return _get_dayobs_key(dataId) is not None 

711 

712 

713def _seqnum_present(dataId: Mapping[str, Any]) -> bool: 

714 return _get_seqnum_key(dataId) is not None 

715 

716 

717def _expid_present(dataId: Mapping[str, Any]) -> bool: 

718 return _get_expid_key(dataId) is not None 

719 

720 

721def _get_dayobs_key(dataId: Mapping[str, Any]) -> str | None: 

722 """Return the key for day_obs if present, else None""" 

723 keys = [k for k in dataId.keys() if k.find("day_obs") != -1] 

724 if not keys: 

725 return None 

726 return keys[0] 

727 

728 

729def _get_seqnum_key(dataId: Mapping[str, Any]) -> str | None: 

730 """Return the key for seq_num if present, else None""" 

731 keys = [k for k in dataId.keys() if k.find("seq_num") != -1] 

732 if not keys: 

733 return None 

734 return keys[0] 

735 

736 

737def _get_expid_key(dataId: Mapping[str, Any]) -> str | None: 

738 """Return the key for expId if present, else None""" 

739 if "exposure.id" in dataId: 

740 return "exposure.id" 

741 elif "exposure" in dataId: 

742 return "exposure" 

743 return None 

744 

745 

746def getDayObs(dataId: Mapping[str, Any] | dafButler.DimensionRecord) -> int | None: 

747 """Get the day_obs from a dataId. 

748 

749 Parameters 

750 ---------- 

751 dataId : `Mapping[str, Any]` or `lsst.daf.butler.DimensionRecord` 

752 The dataId. 

753 

754 Returns 

755 ------- 

756 day_obs : `int` or `None` 

757 The day_obs value if present, else None. 

758 """ 

759 if hasattr(dataId, "day_obs"): 

760 return getattr(dataId, "day_obs") 

761 if not _dayobs_present(dataId): 

762 return None 

763 return dataId["day_obs"] if "day_obs" in dataId else dataId["exposure.day_obs"] 

764 

765 

766def getSeqNum(dataId: Mapping[str, Any] | dafButler.DimensionRecord) -> int | None: 

767 """Get the seq_num from a dataId. 

768 

769 Parameters 

770 ---------- 

771 dataId : `Mapping[str, Any]` or `lsst.daf.butler.DimensionRecord` 

772 The dataId. 

773 

774 Returns 

775 ------- 

776 seq_num : `int` or `None` 

777 The seq_num value if present, else None. 

778 """ 

779 if hasattr(dataId, "seq_num"): 

780 return getattr(dataId, "seq_num") 

781 if not _seqnum_present(dataId): 

782 return None 

783 return dataId["seq_num"] if "seq_num" in dataId else dataId["exposure.seq_num"] 

784 

785 

786def getExpId(dataId: Mapping[str, Any] | dafButler.DimensionRecord) -> int | None: 

787 """Get the expId from a dataId. 

788 

789 Parameters 

790 ---------- 

791 dataId : `Mapping[str, Any]` or `lsst.daf.butler.DimensionRecord` 

792 The dataId. 

793 

794 Returns 

795 ------- 

796 expId : `int` or `None` 

797 The expId value if present, else None. 

798 """ 

799 if hasattr(dataId, "id"): 

800 return getattr(dataId, "id") 

801 if not _expid_present(dataId): 

802 return None 

803 return dataId["exposure"] if "exposure" in dataId else dataId["exposure.id"] 

804 

805 

806def getLatissOnSkyDataIds( 

807 butler: DirectButler, 

808 skipTypes: Iterable[str] = ("bias", "dark", "flat"), 

809 checkObject: bool = True, 

810 full: bool = True, 

811 startDate: int | None = None, 

812 endDate: int | None = None, 

813) -> list[Mapping[str, int | str | None]]: 

814 """Get a list of all on-sky dataIds taken. 

815 

816 Parameters 

817 ---------- 

818 butler : `lsst.daf.butler.Butler` 

819 The butler. 

820 skipTypes : `list` of `str` 

821 Image types to exclude. 

822 checkObject : `bool` 

823 Check if the value of target_name (formerly OBJECT) is set and exlude 

824 if it is not. 

825 full : `bool` 

826 Return filled dataIds. Required for some analyses, but runs much 

827 (~30x) slower. 

828 startDate : `int` 

829 The day_obs to start at, inclusive. 

830 endDate : `int` 

831 The day_obs to end at, inclusive. 

832 

833 Returns 

834 ------- 

835 dataIds : `list` or `dataIds` 

836 The dataIds. 

837 """ 

838 

839 def isOnSky(expRecord: dafButler.DimensionRecord) -> bool: 

840 imageType = expRecord.observation_type 

841 obj = expRecord.target_name 

842 if checkObject and obj == "NOTSET": 

843 return False 

844 if imageType not in skipTypes: 

845 return True 

846 return False 

847 

848 recordSets = [] 

849 days = getDaysWithData(butler) 

850 if startDate: 

851 days = [d for d in days if d >= startDate] 

852 if endDate: 

853 days = [d for d in days if d <= endDate] 

854 days = sorted(set(days)) 

855 

856 where = "exposure.day_obs=dayObs" 

857 for day in days: 

858 # queryDataIds would be better here, but it's then hard/impossible 

859 # to do the filtering for which is on sky, so just take the dataIds 

860 records = butler.registry.queryDimensionRecords( 

861 "exposure", where=where, bind={"dayObs": day}, datasets="raw" 

862 ) 

863 recordSets.append(sortRecordsByDayObsThenSeqNum(records)) 

864 

865 dataIds = [r.dataId for r in filter(isOnSky, itertools.chain(*recordSets))] 

866 if full: 

867 expandedIds = [ 

868 updateDataIdOrDataCord(butler.registry.expandDataId(dataId, detector=0).mapping) 

869 for dataId in dataIds 

870 ] 

871 filledIds = [fillDataId(butler, dataId) for dataId in expandedIds] 

872 return filledIds 

873 else: 

874 return [updateDataIdOrDataCord(dataId, detector=0) for dataId in dataIds] 

875 

876 

877def getExpRecord( 

878 butler: dafButler.Butler, 

879 instrument: str, 

880 expId: int | None = None, 

881 dayObs: int | None = None, 

882 seqNum: int | None = None, 

883) -> dafButler.DimensionRecord: 

884 """Get the exposure record for a given exposure ID or dayObs+seqNum. 

885 

886 Parameters 

887 ---------- 

888 butler : `lsst.daf.butler.Butler` 

889 The butler. 

890 instrument : `str` 

891 The instrument name, e.g. 'LSSTCam'. 

892 expId : `int` 

893 The exposure ID. 

894 

895 Returns 

896 ------- 

897 expRecord : `lsst.daf.butler.DimensionRecord` 

898 The exposure record. 

899 """ 

900 if expId is None and (dayObs is None or seqNum is None): 

901 raise ValueError("Must supply either expId or (dayObs AND seqNum)") 

902 

903 where = "instrument=inst" # Note you can't use =instrument as bind-strings can't clash with dimensions 

904 bind: "dict[str, str | int]" = {"inst": instrument} 

905 if expId: 

906 where += " AND exposure.id=expId" 

907 bind.update({"expId": expId}) 

908 if dayObs and seqNum: 

909 where += " AND exposure.day_obs=dayObs AND exposure.seq_num=seqNum" 

910 bind.update({"dayObs": dayObs, "seqNum": seqNum}) 

911 

912 expRecords = butler.registry.queryDimensionRecords("exposure", where=where, bind=bind, datasets="raw") 

913 filteredExpRecords = list(set(expRecords)) # must call set as this may contain many duplicates 

914 if len(filteredExpRecords) != 1: 

915 raise RuntimeError( 

916 f"Failed to find unique exposure record for {instrument=} with" 

917 f" {expId=}, {dayObs=}, {seqNum=}, got {len(filteredExpRecords)} records" 

918 ) 

919 return filteredExpRecords[0]