Coverage for python/lsst/summit/utils/butlerUtils.py: 16%
282 statements
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-16 03:29 -0700
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-16 03:29 -0700
1# This file is part of summit_utils.
2#
3# Developed for the LSST Data Management System.
4# This product includes software developed by the LSST Project
5# (https://www.lsst.org).
6# See the COPYRIGHT file at the top-level directory of this distribution
7# for details of code ownership.
8#
9# This program is free software: you can redistribute it and/or modify
10# it under the terms of the GNU General Public License as published by
11# the Free Software Foundation, either version 3 of the License, or
12# (at your option) any later version.
13#
14# This program is distributed in the hope that it will be useful,
15# but WITHOUT ANY WARRANTY; without even the implied warranty of
16# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
17# GNU General Public License for more details.
18#
19# You should have received a copy of the GNU General Public License
20# along with this program. If not, see <https://www.gnu.org/licenses/>.
22import copy
23import itertools
24import logging
25from collections.abc import Iterable, Mapping
26from typing import Any
28from deprecated.sphinx import deprecated
30import lsst.daf.butler as dafButler
31from lsst.daf.butler.direct_butler import DirectButler
32from lsst.summit.utils.utils import getSite
33from lsst.utils.iteration import ensure_iterable
35__all__ = [
36 "makeDefaultLatissButler", # deprecated
37 "makeDefaultButler",
38 "updateDataId",
39 "sanitizeDayObs",
40 "getMostRecentDayObs",
41 "getSeqNumsForDayObs",
42 "getMostRecentDataId",
43 "getDatasetRefForDataId",
44 "getDayObs",
45 "getSeqNum",
46 "getExpId",
47 "datasetExists", # deprecated
48 "sortRecordsByDayObsThenSeqNum",
49 "getDaysWithData",
50 "getExpIdFromDayObsSeqNum",
51 "updateDataIdOrDataCord",
52 "fillDataId",
53 "getExpRecordFromDataId",
54 "getDayObsSeqNumFromExposureId",
55 "removeDataProduct",
56 "getLatissOnSkyDataIds",
57 "getExpRecord",
58]
60_LATISS_DEFAULT_COLLECTIONS = ["LATISS/raw/all", "LATISS/calib", "LATISS/runs/quickLook"]
62# RECENT_DAY must be in the past *and have data* (otherwise some tests are
63# no-ops), to speed up queries by restricting them significantly,
64# but data must definitely been taken since. Should
65# also not be more than 2 months in the past due to 60 day lookback time on the
66# summit. All this means it should be updated by an informed human.
67RECENT_DAY = 20220503
70def _configureForSite() -> None:
71 try:
72 site = getSite()
73 except ValueError:
74 # this method is run automatically on module import, so
75 # don't fail for k8s where this cannot yet be determined
76 print("WARNING: failed to automatically determine site")
77 site = None
78 if site == "tucson": 78 ↛ 80line 78 didn't jump to line 80 because the condition on line 78 was never true
79 global RECENT_DAY
80 RECENT_DAY = 20211104 # TTS has limited data, so use this day
83_configureForSite()
86def getLatissDefaultCollections() -> list[str]:
87 """Get the default set of LATISS collections, updated for the site at
88 which the code is being run.
90 Returns
91 -------
92 collections : `list` of `str`
93 The default collections for the site.
94 """
95 collections = _LATISS_DEFAULT_COLLECTIONS
96 try:
97 site = getSite()
98 except ValueError:
99 site = ""
101 if site == "tucson": 101 ↛ 102line 101 didn't jump to line 102 because the condition on line 101 was never true
102 collections.append("LATISS-test-data")
103 return collections
104 if site == "summit": 104 ↛ 105line 104 didn't jump to line 105 because the condition on line 104 was never true
105 collections.append("LATISS-test-data")
106 return collections
107 return collections
110def _update_RECENT_DAY(day: int) -> None:
111 """Update the value for RECENT_DAY once we have a value for free."""
112 global RECENT_DAY
113 RECENT_DAY = max(day - 1, RECENT_DAY)
116@deprecated(
117 reason="Use the more generic makeDefaultButler('LATISS'). Will be removed after v28.0.",
118 version="v27.0",
119 category=FutureWarning,
120)
121def makeDefaultLatissButler(
122 *,
123 extraCollections: list[str] | None = None,
124 writeable: bool = False,
125 embargo: bool = False,
126) -> dafButler.Butler:
127 """Create a butler for LATISS using the default collections.
129 Parameters
130 ----------
131 extraCollections : `list` of `str`
132 Extra input collections to supply to the butler init.
133 writable : `bool`, optional
134 Whether to make a writable butler.
135 embargo : `bool`, optional
136 Use the embargo repo instead of the main one. Needed to access
137 embargoed data.
139 Returns
140 -------
141 butler : `lsst.daf.butler.Butler`
142 The butler.
143 """
144 return makeDefaultButler(
145 "LATISS", extraCollections=extraCollections, writeable=writeable, embargo=embargo
146 )
149def makeDefaultButler(
150 instrument: str,
151 *,
152 extraCollections: list[str] | None = None,
153 writeable: bool = False,
154 embargo: bool = True,
155) -> dafButler.Butler:
156 """Create a butler for the instrument using default collections, regardless
157 of the location.
159 Parameters
160 ----------
161 extraCollections : `list` of `str`
162 Extra input collections to supply to the butler init.
163 writable : `bool`, optional
164 Whether to make a writable butler.
165 embargo : `bool`, optional
166 Use the embargo repo instead of the main one. Needed to access
167 embargoed data if not at a summit-like location.
169 Returns
170 -------
171 butler : `lsst.daf.butler.Butler`
172 The butler.
174 Raises
175 ------
176 FileNotFoundError
177 Raised if the butler cannot be created, because this is the error
178 when the DAF_BUTLER_REPOSITORY_INDEX is not set correctly.
179 """
180 SUPPORTED_SITES = [
181 "summit",
182 "tucson",
183 "base",
184 "staff-rsp",
185 "rubin-devl",
186 "usdf-k8s",
187 ]
189 site = getSite()
190 if site not in SUPPORTED_SITES:
191 # This might look like a slightly weird error to raise, but this is
192 # the same error that's raised when the DAF_BUTLER_REPOSITORY_INDEX
193 # isn't set, so it's the most appropriate error to raise here, i.e.
194 # this is what would be raised if this function was allowed to continue
195 # only this is a less confusing version of it.
196 raise FileNotFoundError(f"Default butler creation only available at: {SUPPORTED_SITES}, got {site=}")
198 summitLike = site in ["summit", "tucson", "base"]
199 if summitLike:
200 if embargo is True:
201 logger = logging.getLogger(__name__)
202 logger.debug("embargo option is irrelevant on the summit, ignoring")
203 embargo = False # there's only one repo too, so this makes the code more simple too
205 collections: list[str] = [f"{instrument}/defaults"]
206 raCollection = (
207 [f"{instrument}/runs/quickLook"] if summitLike else [f"{instrument}/runs/nightlyValidation"]
208 )
209 collections.extend(raCollection)
210 if instrument == "LSSTCam":
211 collections.append("LSSTCam/raw/guider")
213 repo = instrument if embargo is False else "embargo"
215 if extraCollections is not None:
216 assert extraCollections is not None # just for mypy
217 extraCollectionsList = ensure_iterable(extraCollections)
218 collections.extend(extraCollectionsList)
220 try:
221 butler = dafButler.Butler.from_config(
222 repo, collections=collections, writeable=writeable, instrument=instrument
223 )
224 except (FileNotFoundError, RuntimeError):
225 # Depending on the value of DAF_BUTLER_REPOSITORY_INDEX and whether
226 # it is present and blank, or just not set, both these exception
227 # types can be raised, see tests/test_butlerUtils.py:ButlerInitTestCase
228 # for details and tests which confirm these have not changed
229 raise FileNotFoundError # unify exception type
230 return butler
233@deprecated(
234 reason="datasExists has been replaced by Butler.exists(). Will be removed after v26.0.",
235 version="v26.0",
236 category=FutureWarning,
237)
238def datasetExists(
239 butler: dafButler.Butler, dataProduct: str, dataId: dafButler.DataId, **kwargs: Any
240) -> bool:
241 """Collapse the tri-state behaviour of butler.datasetExists to a boolean.
243 Parameters
244 ----------
245 butler : `lsst.daf.butler.Butler`
246 The butler
247 dataProduct : `str`
248 The type of data product to check for
249 dataId : `dafButler.DataId`
250 The dataId of the dataProduct to check for
252 Returns
253 -------
254 exists : `bool`
255 True if the dataProduct exists for the dataProduct and can be retreived
256 else False.
257 """
258 return bool(butler.exists(dataProduct, dataId, **kwargs))
261def updateDataId(dataId: dafButler.DataId, **kwargs: Any) -> dafButler.DataId:
262 """Update a DataCoordinate or dataId dict with kwargs.
264 Provides a single interface for adding the detector key (or others) to a
265 dataId whether it's a DataCoordinate or a dict
267 Parameters
268 ----------
269 dataId : `dafButler.DataId`
270 The dataId to update.
271 kwargs : `dict`
272 The keys and values to add to the dataId.
274 Returns
275 -------
276 dataId : `dict` or `lsst.daf.butler.DataCoordinate`
277 The updated dataId, with the same type as the input.
278 """
280 match dataId:
281 case dafButler.DataCoordinate():
282 return dafButler.DataCoordinate.standardize(dataId, **kwargs)
283 case dict() as dataId:
284 return dict(dataId, **kwargs)
285 raise ValueError(f"Unknown dataId type {type(dataId)}")
288def sanitizeDayObs(day_obs: int | str) -> int:
289 """Take string or int day_obs and turn it into the int version.
291 Parameters
292 ----------
293 day_obs : `str` or `int`
294 The day_obs to sanitize.
296 Returns
297 -------
298 day_obs : `int`
299 The sanitized day_obs.
301 Raises
302 ------
303 ValueError
304 Raised if the day_obs fails to translate for any reason.
305 """
306 if isinstance(day_obs, int):
307 return day_obs
308 elif isinstance(day_obs, str):
309 try:
310 return int(day_obs.replace("-", ""))
311 except Exception:
312 ValueError(f"Failed to sanitize {day_obs!r} to a day_obs")
313 raise ValueError(f"Cannot sanitize {day_obs!r} to a day_obs")
316def getMostRecentDayObs(butler: dafButler.Butler) -> int:
317 """Get the most recent day_obs for which there is data.
319 Parameters
320 ----------
321 butler : `lsst.daf.butler.Butler
322 The butler to query.
324 Returns
325 -------
326 day_obs : `int`
327 The day_obs.
328 """
329 where = "exposure.day_obs>=RECENT_DAY"
330 records = butler.registry.queryDimensionRecords(
331 "exposure", where=where, datasets="raw", bind={"RECENT_DAY": RECENT_DAY}
332 )
333 recentDay = max(r.day_obs for r in records)
334 _update_RECENT_DAY(recentDay)
335 return recentDay
338def getSeqNumsForDayObs(butler: dafButler.Butler, day_obs: int, extraWhere: str = "") -> list[int]:
339 """Get a list of all seq_nums taken on a given day_obs.
341 Parameters
342 ----------
343 butler : `lsst.daf.butler.Butler
344 The butler to query.
345 day_obs : `int` or `str`
346 The day_obs for which the seq_nums are desired.
347 extraWhere : `str`
348 Any extra where conditions to add to the queryDimensionRecords call.
350 Returns
351 -------
352 seq_nums : `list` of `int`
353 The seq_nums taken on the corresponding day_obs in ascending numerical
354 order.
355 """
356 day_obs = sanitizeDayObs(day_obs)
357 where = "exposure.day_obs=dayObs"
358 if extraWhere:
359 extraWhere = extraWhere.replace('"', "'")
360 where += f" and {extraWhere}"
361 records = butler.registry.queryDimensionRecords(
362 "exposure", where=where, bind={"dayObs": day_obs}, datasets="raw"
363 )
364 return sorted(set([r.seq_num for r in records]))
367def sortRecordsByDayObsThenSeqNum(
368 records: dafButler.registry.queries.DimensionRecordQueryResults | list[dafButler.DimensionRecord],
369) -> list[dafButler.DimensionRecord]:
370 """Sort a set of records by dayObs, then seqNum to get the order in which
371 they were taken.
373 Parameters
374 ----------
375 records : `list` of `dafButler.DimensionRecord`
376 The records to be sorted.
378 Returns
379 -------
380 sortedRecords : `list` of `dafButler.DimensionRecord`
381 The sorted records
383 Raises
384 ------
385 ValueError
386 Raised if the recordSet contains duplicate records, or if it contains
387 (dayObs, seqNum) collisions.
388 """
389 records = list(records) # must call list in case we have a generator
390 recordSet = set(records)
391 if len(records) != len(recordSet):
392 raise ValueError("Record set contains duplicate records and therefore cannot be sorted unambiguously")
394 daySeqTuples = [(r.day_obs, r.seq_num) for r in records]
395 if len(daySeqTuples) != len(set(daySeqTuples)):
396 raise ValueError(
397 "Record set contains dayObs/seqNum collisions, and therefore cannot be sorted " "unambiguously"
398 )
400 records.sort(key=lambda r: (r.day_obs, r.seq_num))
401 return records
404def getDaysWithData(butler: dafButler.Butler, datasetType: str = "raw") -> list[int]:
405 """Get all the days for which LATISS has taken data on the mountain.
407 Parameters
408 ----------
409 butler : `lsst.daf.butler.Butler
410 The butler to query.
411 datasetType : `str`
412 The datasetType to query.
414 Returns
415 -------
416 days : `list` of `int`
417 A sorted list of the day_obs values for which mountain-top data exists.
418 """
419 # 20200101 is a day between shipping LATISS and going on sky
420 # We used to constrain on exposure.seq_num<50 to massively reduce the
421 # number of returned records whilst being large enough to ensure that no
422 # days are missed because early seq_nums were skipped. However, because
423 # we have test datasets like LATISS-test-data-tts where we only kept
424 # seqNums from 950 on one day, we can no longer assume this so don't be
425 # tempted to add such a constraint back in here for speed.
426 where = "exposure.day_obs>20200101"
427 records = butler.registry.queryDimensionRecords("exposure", where=where, datasets=datasetType)
428 return sorted(set([r.day_obs for r in records]))
431def getMostRecentDataId(butler: dafButler.Butler) -> dict[str, Any]:
432 """Get the dataId for the most recent observation.
434 Parameters
435 ----------
436 butler : `lsst.daf.butler.Butler
437 The butler to query.
439 Returns
440 -------
441 dataId : `dict[str, Any]`
442 The dataId of the most recent exposure.
443 """
444 lastDay = getMostRecentDayObs(butler)
445 seqNum = getSeqNumsForDayObs(butler, lastDay)[-1]
446 dataId = {"day_obs": lastDay, "seq_num": seqNum, "detector": 0}
447 dataId.update(getExpIdFromDayObsSeqNum(butler, dataId))
448 return dataId
451def getExpIdFromDayObsSeqNum(butler: dafButler.Butler, dataId: dafButler.DataId) -> dict[str, int]:
452 """Get the exposure id for the dataId.
454 Parameters
455 ----------
456 butler : `lsst.daf.butler.Butler
457 The butler to query.
458 dataId : `dafButler.DataId`
459 The dataId for which to return the exposure id.
461 Returns
462 -------
463 dataId : `dict[str, int]`
464 The dataId of the most recent exposure.
465 """
466 expRecord = getExpRecordFromDataId(butler, dataId)
467 return {"exposure": expRecord.id}
470def updateDataIdOrDataCord(
471 dataId: dafButler.DataId | dafButler.DimensionRecord, **updateKwargs: Any
472) -> Mapping[str, Any]:
473 """Add key, value pairs to a dataId or data coordinate.
475 Parameters
476 ----------
477 dataId : `dafButler.DataId`
478 The dataId for which to return the exposure id.
479 updateKwargs : `dict[str, Any]`
480 The key value pairs add to the dataId or dataCoord.
482 Returns
483 -------
484 dataId : `Mapping[str, Any]`
485 The updated dataId.
487 Notes
488 -----
489 Always returns a dict, so note that if a data coordinate is supplied, a
490 dict is returned, changing the type.
491 """
492 newId = copy.copy(dataId)
493 newId = _assureDict(newId)
494 newId.update(updateKwargs)
495 return newId
498def fillDataId(
499 butler: DirectButler, dataId: dafButler.DataId | dafButler.DimensionRecord
500) -> Mapping[str, Any]:
501 """Given a dataId, fill it with values for all available dimensions.
503 Parameters
504 ----------
505 butler : `lsst.daf.butler.Butler`
506 The butler.
507 dataId : `dafButler.DataId`
508 The dataId to fill.
510 Returns
511 -------
512 dataId : `Mapping[str, Any]`
513 The filled dataId.
515 Notes
516 -----
517 This function is *slow*! Running this on 20,000 dataIds takes approximately
518 7 minutes. Virtually all the slowdown is in the
519 butler.registry.expandDataId() call though, so this wrapper is not to blame
520 here, and might speed up in future with butler improvements.
521 """
522 # ensure it's a dict to deal with records etc
523 dictId = _assureDict(dataId)
525 # this removes extraneous keys that would trip up the registry call
526 # using _rewrite_data_id is perhaps ever so slightly slower than popping
527 # the bad keys, or making a minimal dataId by hand, but is more
528 # reliable/general, so we choose that over the other approach here
529 realDataId, _ = butler._rewrite_data_id(dictId, butler.get_dataset_type("raw"))
531 # now expand and turn back to a dict - this call is VERY slow
532 expandedDictDataId = dict(butler.registry.expandDataId(realDataId, detector=0).mapping)
534 missingExpId = getExpId(expandedDictDataId) is None
535 missingDayObs = getDayObs(expandedDictDataId) is None
536 missingSeqNum = getSeqNum(expandedDictDataId) is None
538 if missingDayObs or missingSeqNum:
539 dayObsSeqNum = getDayObsSeqNumFromExposureId(butler, expandedDictDataId)
540 expandedDictDataId.update(dayObsSeqNum)
542 if missingExpId:
543 expId = getExpIdFromDayObsSeqNum(butler, expandedDictDataId)
544 expandedDictDataId.update(expId)
546 return expandedDictDataId
549def _assureDict(
550 dataId: dafButler.DataId | dafButler.DimensionRecord,
551) -> dict[str, Any]:
552 """Turn any data-identifier-like object into a dict.
554 Parameters
555 ----------
556 dataId : `dafButler.DataId` or
557 `lsst.daf.butler.dimensions.DimensionRecord`
558 The data identifier.
560 Returns
561 -------
562 dataId : `dict[str, Any]`
563 The data identifier as a dict.
564 """
565 if isinstance(dataId, dict):
566 return dataId
567 elif hasattr(dataId, "mapping"): # dafButler.DataCoordinate
568 return {str(k): v for k, v in dataId.mapping.items()}
569 elif hasattr(dataId, "dataId"): # dafButler.DimensionRecord
570 return {str(k): v for k, v in dataId.dataId.mapping.items()}
571 elif hasattr(dataId, "keys"): # some other mapping, assume str keys
572 return dict(dataId)
573 else:
574 raise RuntimeError(f"Failed to coerce {type(dataId)} to dict")
577def getExpRecordFromDataId(butler: dafButler.Butler, dataId: dafButler.DataId) -> dafButler.DimensionRecord:
578 """Get the exposure record for a given dataId.
580 Parameters
581 ----------
582 butler : `lsst.daf.butler.Butler`
583 The butler.
584 dataId : `dafButler.DataId`
585 The dataId.
587 Returns
588 -------
589 expRecord : `lsst.daf.butler.dimensions.DimensionRecord`
590 The exposure record.
591 """
592 dataId = _assureDict(dataId)
593 assert isinstance(dataId, dict), f"dataId must be a dict or DimensionRecord, got {type(dataId)}"
595 if expId := getExpId(dataId):
596 where = "exposure.id=expId"
597 expRecords = butler.registry.queryDimensionRecords(
598 "exposure", where=where, bind={"expId": expId}, datasets="raw"
599 )
601 else:
602 dayObs = getDayObs(dataId)
603 seqNum = getSeqNum(dataId)
604 if not (dayObs and seqNum):
605 raise RuntimeError(f"Failed to find either expId or day_obs and seq_num in dataId {dataId}")
606 where = "exposure.day_obs=dayObs AND exposure.seq_num=seq_num"
607 expRecords = butler.registry.queryDimensionRecords(
608 "exposure", where=where, bind={"dayObs": dayObs, "seq_num": seqNum}, datasets="raw"
609 )
611 filteredExpRecords = set(expRecords)
612 if not filteredExpRecords:
613 raise LookupError(f"No exposure records found for {dataId}")
614 assert len(filteredExpRecords) == 1, f"Found {len(filteredExpRecords)} exposure records for {dataId}"
615 return filteredExpRecords.pop()
618def getDayObsSeqNumFromExposureId(butler: dafButler.Butler, dataId: Mapping[str, Any]) -> dict[str, int]:
619 """Get the day_obs and seq_num for an exposure id.
621 Parameters
622 ----------
623 butler : `lsst.daf.butler.Butler`
624 The butler.
625 dataId : ` Mapping[str, Any]`
626 The dataId containing the exposure id.
628 Returns
629 -------
630 dataId : ` dict[str, int]`
631 A dict containing only the day_obs and seq_num.
632 """
633 if (dayObs := getDayObs(dataId)) and (seqNum := getSeqNum(dataId)):
634 return {"day_obs": dayObs, "seq_num": seqNum}
636 if isinstance(dataId, int):
637 dataId = {"exposure": dataId}
638 else:
639 dataId = _assureDict(dataId)
640 assert isinstance(dataId, dict)
642 if not (expId := getExpId(dataId)):
643 raise RuntimeError(f"Failed to find exposure id in {dataId}")
645 where = "exposure.id=expId"
646 expRecords = list(
647 butler.registry.queryDimensionRecords("exposure", where=where, bind={"expId": expId}, datasets="raw")
648 )
649 uniqueExpRecords = set(expRecords)
650 if not uniqueExpRecords:
651 raise LookupError(f"No exposure records found for {dataId}")
652 assert len(uniqueExpRecords) == 1, f"Found {len(uniqueExpRecords)} exposure records for {dataId}"
653 record = uniqueExpRecords.pop()
654 return {"day_obs": record.day_obs, "seq_num": record.seq_num}
657def getDatasetRefForDataId(
658 butler: dafButler.Butler, datasetType: str | dafButler.DatasetType, dataId: dafButler.DataId
659) -> dafButler.DatasetRef | None:
660 """Get the datasetReference for a dataId.
662 Parameters
663 ----------
664 butler : `lsst.daf.butler.Butler`
665 The butler.
666 datasetType : `str` or `datasetType`
667 The dataset type.
668 dataId : `dafButler.DataId`
669 The dataId.
671 Returns
672 -------
673 datasetRef : `lsst.daf.butler.dimensions.DatasetReference`
674 The dataset reference.
675 """
676 dictId = _assureDict(dataId)
677 if not _expid_present(dictId):
678 assert _dayobs_present(dictId) and _seqnum_present(dictId)
679 dictId.update(getExpIdFromDayObsSeqNum(butler, dictId))
681 dRef = butler.find_dataset(datasetType, dictId)
682 return dRef
685def removeDataProduct(
686 butler: dafButler.Butler, datasetType: str | dafButler.DatasetType, dataId: dict[str, Any]
687) -> None:
688 """Remove a data prodcut from the registry. Use with caution.
690 Parameters
691 ----------
692 butler : `lsst.daf.butler.Butler`
693 The butler.
694 datasetType : `str` or `datasetType`
695 The dataset type.
696 dataId : `dict[str, Any]`
697 The dataId.
699 """
700 if datasetType == "raw":
701 raise RuntimeError("I'm sorry, Dave, I'm afraid I can't do that.")
702 dRef = getDatasetRefForDataId(butler, datasetType, dataId)
703 if dRef is None:
704 return
705 butler.pruneDatasets([dRef], disassociate=True, unstore=True, purge=True)
706 return
709def _dayobs_present(dataId: Mapping[str, Any]) -> bool:
710 return _get_dayobs_key(dataId) is not None
713def _seqnum_present(dataId: Mapping[str, Any]) -> bool:
714 return _get_seqnum_key(dataId) is not None
717def _expid_present(dataId: Mapping[str, Any]) -> bool:
718 return _get_expid_key(dataId) is not None
721def _get_dayobs_key(dataId: Mapping[str, Any]) -> str | None:
722 """Return the key for day_obs if present, else None"""
723 keys = [k for k in dataId.keys() if k.find("day_obs") != -1]
724 if not keys:
725 return None
726 return keys[0]
729def _get_seqnum_key(dataId: Mapping[str, Any]) -> str | None:
730 """Return the key for seq_num if present, else None"""
731 keys = [k for k in dataId.keys() if k.find("seq_num") != -1]
732 if not keys:
733 return None
734 return keys[0]
737def _get_expid_key(dataId: Mapping[str, Any]) -> str | None:
738 """Return the key for expId if present, else None"""
739 if "exposure.id" in dataId:
740 return "exposure.id"
741 elif "exposure" in dataId:
742 return "exposure"
743 return None
746def getDayObs(dataId: Mapping[str, Any] | dafButler.DimensionRecord) -> int | None:
747 """Get the day_obs from a dataId.
749 Parameters
750 ----------
751 dataId : `Mapping[str, Any]` or `lsst.daf.butler.DimensionRecord`
752 The dataId.
754 Returns
755 -------
756 day_obs : `int` or `None`
757 The day_obs value if present, else None.
758 """
759 if hasattr(dataId, "day_obs"):
760 return getattr(dataId, "day_obs")
761 if not _dayobs_present(dataId):
762 return None
763 return dataId["day_obs"] if "day_obs" in dataId else dataId["exposure.day_obs"]
766def getSeqNum(dataId: Mapping[str, Any] | dafButler.DimensionRecord) -> int | None:
767 """Get the seq_num from a dataId.
769 Parameters
770 ----------
771 dataId : `Mapping[str, Any]` or `lsst.daf.butler.DimensionRecord`
772 The dataId.
774 Returns
775 -------
776 seq_num : `int` or `None`
777 The seq_num value if present, else None.
778 """
779 if hasattr(dataId, "seq_num"):
780 return getattr(dataId, "seq_num")
781 if not _seqnum_present(dataId):
782 return None
783 return dataId["seq_num"] if "seq_num" in dataId else dataId["exposure.seq_num"]
786def getExpId(dataId: Mapping[str, Any] | dafButler.DimensionRecord) -> int | None:
787 """Get the expId from a dataId.
789 Parameters
790 ----------
791 dataId : `Mapping[str, Any]` or `lsst.daf.butler.DimensionRecord`
792 The dataId.
794 Returns
795 -------
796 expId : `int` or `None`
797 The expId value if present, else None.
798 """
799 if hasattr(dataId, "id"):
800 return getattr(dataId, "id")
801 if not _expid_present(dataId):
802 return None
803 return dataId["exposure"] if "exposure" in dataId else dataId["exposure.id"]
806def getLatissOnSkyDataIds(
807 butler: DirectButler,
808 skipTypes: Iterable[str] = ("bias", "dark", "flat"),
809 checkObject: bool = True,
810 full: bool = True,
811 startDate: int | None = None,
812 endDate: int | None = None,
813) -> list[Mapping[str, int | str | None]]:
814 """Get a list of all on-sky dataIds taken.
816 Parameters
817 ----------
818 butler : `lsst.daf.butler.Butler`
819 The butler.
820 skipTypes : `list` of `str`
821 Image types to exclude.
822 checkObject : `bool`
823 Check if the value of target_name (formerly OBJECT) is set and exlude
824 if it is not.
825 full : `bool`
826 Return filled dataIds. Required for some analyses, but runs much
827 (~30x) slower.
828 startDate : `int`
829 The day_obs to start at, inclusive.
830 endDate : `int`
831 The day_obs to end at, inclusive.
833 Returns
834 -------
835 dataIds : `list` or `dataIds`
836 The dataIds.
837 """
839 def isOnSky(expRecord: dafButler.DimensionRecord) -> bool:
840 imageType = expRecord.observation_type
841 obj = expRecord.target_name
842 if checkObject and obj == "NOTSET":
843 return False
844 if imageType not in skipTypes:
845 return True
846 return False
848 recordSets = []
849 days = getDaysWithData(butler)
850 if startDate:
851 days = [d for d in days if d >= startDate]
852 if endDate:
853 days = [d for d in days if d <= endDate]
854 days = sorted(set(days))
856 where = "exposure.day_obs=dayObs"
857 for day in days:
858 # queryDataIds would be better here, but it's then hard/impossible
859 # to do the filtering for which is on sky, so just take the dataIds
860 records = butler.registry.queryDimensionRecords(
861 "exposure", where=where, bind={"dayObs": day}, datasets="raw"
862 )
863 recordSets.append(sortRecordsByDayObsThenSeqNum(records))
865 dataIds = [r.dataId for r in filter(isOnSky, itertools.chain(*recordSets))]
866 if full:
867 expandedIds = [
868 updateDataIdOrDataCord(butler.registry.expandDataId(dataId, detector=0).mapping)
869 for dataId in dataIds
870 ]
871 filledIds = [fillDataId(butler, dataId) for dataId in expandedIds]
872 return filledIds
873 else:
874 return [updateDataIdOrDataCord(dataId, detector=0) for dataId in dataIds]
877def getExpRecord(
878 butler: dafButler.Butler,
879 instrument: str,
880 expId: int | None = None,
881 dayObs: int | None = None,
882 seqNum: int | None = None,
883) -> dafButler.DimensionRecord:
884 """Get the exposure record for a given exposure ID or dayObs+seqNum.
886 Parameters
887 ----------
888 butler : `lsst.daf.butler.Butler`
889 The butler.
890 instrument : `str`
891 The instrument name, e.g. 'LSSTCam'.
892 expId : `int`
893 The exposure ID.
895 Returns
896 -------
897 expRecord : `lsst.daf.butler.DimensionRecord`
898 The exposure record.
899 """
900 if expId is None and (dayObs is None or seqNum is None):
901 raise ValueError("Must supply either expId or (dayObs AND seqNum)")
903 where = "instrument=inst" # Note you can't use =instrument as bind-strings can't clash with dimensions
904 bind: "dict[str, str | int]" = {"inst": instrument}
905 if expId:
906 where += " AND exposure.id=expId"
907 bind.update({"expId": expId})
908 if dayObs and seqNum:
909 where += " AND exposure.day_obs=dayObs AND exposure.seq_num=seqNum"
910 bind.update({"dayObs": dayObs, "seqNum": seqNum})
912 expRecords = butler.registry.queryDimensionRecords("exposure", where=where, bind=bind, datasets="raw")
913 filteredExpRecords = list(set(expRecords)) # must call set as this may contain many duplicates
914 if len(filteredExpRecords) != 1:
915 raise RuntimeError(
916 f"Failed to find unique exposure record for {instrument=} with"
917 f" {expId=}, {dayObs=}, {seqNum=}, got {len(filteredExpRecords)} records"
918 )
919 return filteredExpRecords[0]