Coverage for python/lsst/daf/butler/registry/tests/_registry.py: 98%
1714 statements
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-22 02:19 -0700
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-22 02:19 -0700
1# This file is part of daf_butler.
2#
3# Developed for the LSST Data Management System.
4# This product includes software developed by the LSST Project
5# (http://www.lsst.org).
6# See the COPYRIGHT file at the top-level directory of this distribution
7# for details of code ownership.
8#
9# This software is dual licensed under the GNU General Public License and also
10# under a 3-clause BSD license. Recipients may choose which of these licenses
11# to use; please see the files gpl-3.0.txt and/or bsd_license.txt,
12# respectively. If you choose the GPL option then the following text applies
13# (but note that there is still no warranty even if you opt for BSD instead):
14#
15# This program is free software: you can redistribute it and/or modify
16# it under the terms of the GNU General Public License as published by
17# the Free Software Foundation, either version 3 of the License, or
18# (at your option) any later version.
19#
20# This program is distributed in the hope that it will be useful,
21# but WITHOUT ANY WARRANTY; without even the implied warranty of
22# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
23# GNU General Public License for more details.
24#
25# You should have received a copy of the GNU General Public License
26# along with this program. If not, see <http://www.gnu.org/licenses/>.
27from __future__ import annotations
29from ... import ddl
31__all__ = ["RegistryTests"]
33import contextlib
34import datetime
35import itertools
36import re
37import time
38import unittest
39import uuid
40from abc import ABC, abstractmethod
41from collections import defaultdict, namedtuple
42from collections.abc import Callable, Iterator
43from concurrent.futures import ThreadPoolExecutor
44from contextlib import ExitStack
45from datetime import timedelta
46from threading import Barrier
47from typing import TypeVar
49import astropy.time
50import sqlalchemy
52try:
53 import numpy as np
54except ImportError:
55 np = None
57import lsst.sphgeom
59from ... import Butler
60from ..._collection_type import CollectionType
61from ..._dataset_association import DatasetAssociation
62from ..._dataset_ref import DatasetIdFactory, DatasetIdGenEnum, DatasetRef
63from ..._dataset_type import DatasetType
64from ..._exceptions import (
65 CalibrationLookupError,
66 CollectionTypeError,
67 DataIdValueError,
68 DatasetTypeExpressionError,
69 InconsistentDataIdError,
70 InvalidQueryError,
71 MissingCollectionError,
72 MissingDatasetTypeError,
73)
74from ..._exceptions_legacy import DatasetTypeError
75from ..._storage_class import StorageClass
76from ..._timespan import Timespan
77from ...dimensions import DataCoordinate, DataCoordinateSet, DimensionUniverse, SkyPixDimension
78from ...direct_butler import DirectButler
79from .._collection_summary import CollectionSummary
80from .._config import RegistryConfig
81from .._exceptions import (
82 ArgumentError,
83 CollectionError,
84 ConflictingDefinitionError,
85 NoDefaultCollectionError,
86 OrphanedRecordError,
87)
88from ..interfaces import ButlerAttributeExistsError, ReadOnlyDatabaseError
89from ..queries import ParentDatasetQueryResults
90from ..sql_registry import SqlRegistry
92_T = TypeVar("_T")
95class RegistryTests(ABC):
96 """Generic tests for the `SqlRegistry` class that can be subclassed to
97 generate tests for different configurations.
98 """
100 collectionsManager: str | None = None
101 """Name of the collections manager class, if subclass provides value for
102 this member then it overrides name specified in default configuration
103 (`str`).
104 """
106 datasetsManager: str | dict[str, str] | None = None
107 """Name or configuration dictionary of the datasets manager class, if
108 subclass provides value for this member then it overrides name specified
109 in default configuration (`str` or `dict`).
110 """
112 supportsCollectionRegex: bool = False
113 """True if the registry class being tested supports regex searches for
114 collections."""
116 def makeRegistryConfig(self) -> RegistryConfig:
117 """Create RegistryConfig used to create a registry.
119 This method should be called by a subclass from `makeRegistry`.
120 Returned instance will be pre-configured based on the values of class
121 members, and default-configured for all other parameters. Subclasses
122 that need default configuration should just instantiate
123 `RegistryConfig` directly.
124 """
125 config = RegistryConfig()
126 if self.collectionsManager: 126 ↛ 128line 126 didn't jump to line 128 because the condition on line 126 was always true
127 config["managers", "collections"] = self.collectionsManager
128 if self.datasetsManager: 128 ↛ 130line 128 didn't jump to line 130 because the condition on line 128 was always true
129 config["managers", "datasets"] = self.datasetsManager
130 return config
132 @abstractmethod
133 def make_butler(self, registry_config: RegistryConfig | None = None) -> Butler:
134 """Return the butler to be tested.
136 Parameters
137 ----------
138 registry_config : `RegistryConfig`, optional
139 Registry configuration used when instantiating the Butler.
141 Returns
142 -------
143 butler : `~lsst.daf.butler.Butler`
144 The butler with a registry to be tested.
145 """
146 raise NotImplementedError()
148 def load_data(self, butler: Butler, *filenames: str) -> None:
149 """Load registry test data from
150 ``resource://lsst.daf.butler/tests/registry_data/<filename>``,
151 which should be a YAML import/export file.
153 Parameters
154 ----------
155 butler : `Butler`
156 The butler to load into.
157 *filenames : `str`
158 The names of the files to load.
159 """
160 for filename in filenames:
161 butler.import_(
162 filename=f"resource://lsst.daf.butler/tests/registry_data/{filename}", without_datastore=True
163 )
165 def checkQueryResults(self, results, expected):
166 """Check that a query results object contains expected values.
168 Parameters
169 ----------
170 results : `DataCoordinateQueryResults` or `DatasetQueryResults`
171 A lazy-evaluation query results object.
172 expected : `list`
173 A list of `DataCoordinate` o `DatasetRef` objects that should be
174 equal to results of the query, aside from ordering.
175 """
176 self.assertCountEqual(list(results), expected)
177 self.assertEqual(results.count(), len(expected))
178 if expected:
179 self.assertTrue(results.any())
180 else:
181 self.assertFalse(results.any())
183 def testOpaque(self):
184 """Tests for `SqlRegistry.registerOpaqueTable`,
185 `SqlRegistry.insertOpaqueData`, `SqlRegistry.fetchOpaqueData`, and
186 `SqlRegistry.deleteOpaqueData`.
187 """
188 butler = self.make_butler()
189 registry = butler._registry
190 table = "opaque_table_for_testing"
191 registry.registerOpaqueTable(
192 table,
193 spec=ddl.TableSpec(
194 fields=[
195 ddl.FieldSpec("id", dtype=sqlalchemy.BigInteger, primaryKey=True),
196 ddl.FieldSpec("name", dtype=sqlalchemy.String, length=16, nullable=False),
197 ddl.FieldSpec("count", dtype=sqlalchemy.SmallInteger, nullable=True),
198 ],
199 ),
200 )
201 rows = [
202 {"id": 1, "name": "one", "count": None},
203 {"id": 2, "name": "two", "count": 5},
204 {"id": 3, "name": "three", "count": 6},
205 ]
206 registry.insertOpaqueData(table, *rows)
207 self.assertCountEqual(rows, list(registry.fetchOpaqueData(table)))
208 self.assertEqual(rows[0:1], list(registry.fetchOpaqueData(table, id=1)))
209 self.assertEqual(rows[1:2], list(registry.fetchOpaqueData(table, name="two")))
210 self.assertEqual(rows[0:1], list(registry.fetchOpaqueData(table, id=(1, 3), name=("one", "two"))))
211 self.assertEqual(rows, list(registry.fetchOpaqueData(table, id=(1, 2, 3))))
212 # Test very long IN clause which exceeds sqlite limit on number of
213 # parameters. SQLite says the limit is 32k but it looks like it is
214 # much higher.
215 self.assertEqual(rows, list(registry.fetchOpaqueData(table, id=list(range(300_000)))))
216 # Two IN clauses, each longer than 1k batch size, first with
217 # duplicates, second has matching elements in different batches (after
218 # sorting).
219 self.assertEqual(
220 rows[0:2],
221 list(
222 registry.fetchOpaqueData(
223 table,
224 id=list(range(1000)) + list(range(100, 0, -1)),
225 name=["one"] + [f"q{i}" for i in range(2200)] + ["two"],
226 )
227 ),
228 )
229 self.assertEqual([], list(registry.fetchOpaqueData(table, id=1, name="two")))
230 registry.deleteOpaqueData(table, id=3)
231 self.assertCountEqual(rows[:2], list(registry.fetchOpaqueData(table)))
232 registry.deleteOpaqueData(table)
233 self.assertEqual([], list(registry.fetchOpaqueData(table)))
235 def testDatasetType(self):
236 """Tests for `SqlRegistry.registerDatasetType` and
237 `SqlRegistry.getDatasetType`.
238 """
239 butler = self.make_butler()
240 registry = butler.registry
241 # Check valid insert
242 datasetTypeName = "test"
243 storageClass = StorageClass("testDatasetType")
244 registry.storageClasses.registerStorageClass(storageClass)
245 dimensions = registry.dimensions.conform(("instrument", "visit"))
246 differentDimensions = registry.dimensions.conform(("instrument", "patch"))
247 inDatasetType = DatasetType(datasetTypeName, dimensions, storageClass)
248 # Inserting for the first time should return True
249 self.assertTrue(registry.registerDatasetType(inDatasetType))
250 outDatasetType1 = registry.getDatasetType(datasetTypeName)
251 self.assertEqual(outDatasetType1, inDatasetType)
253 # Re-inserting should work
254 self.assertFalse(registry.registerDatasetType(inDatasetType))
255 # Except when they are not identical
256 with self.assertRaises(ConflictingDefinitionError):
257 nonIdenticalDatasetType = DatasetType(datasetTypeName, differentDimensions, storageClass)
258 registry.registerDatasetType(nonIdenticalDatasetType)
260 # Template can be None
261 datasetTypeName = "testNoneTemplate"
262 storageClass = StorageClass("testDatasetType2")
263 registry.storageClasses.registerStorageClass(storageClass)
264 dimensions = registry.dimensions.conform(("instrument", "visit"))
265 inDatasetType = DatasetType(datasetTypeName, dimensions, storageClass)
266 registry.registerDatasetType(inDatasetType)
267 outDatasetType2 = registry.getDatasetType(datasetTypeName)
268 self.assertEqual(outDatasetType2, inDatasetType)
270 allTypes = set(registry.queryDatasetTypes())
271 self.assertEqual(allTypes, {outDatasetType1, outDatasetType2})
273 # Test some basic queryDatasetTypes functionality
274 missing: list[str] = []
275 types = registry.queryDatasetTypes(["te*", "notarealdatasettype"], missing=missing)
276 self.assertCountEqual([dt.name for dt in types], ["test", "testNoneTemplate"])
277 self.assertEqual(missing, ["notarealdatasettype"])
279 # Trying to register a dataset type with different universe version or
280 # namespace will raise.
281 wrong_universes = (DimensionUniverse(version=-1), DimensionUniverse(namespace="🔭"))
282 for universe in wrong_universes:
283 storageClass = StorageClass("testDatasetType")
284 dataset_type = DatasetType(
285 "wrong_universe", ("instrument", "visit"), storageClass, universe=universe
286 )
287 with self.assertRaisesRegex(ValueError, "Incompatible dimension universe versions"):
288 registry.registerDatasetType(dataset_type)
290 def testDatasetTypeCache(self):
291 """Test for dataset type cache update logic after a cache miss."""
292 butler1 = self.make_butler()
293 butler2 = butler1.clone()
294 self.load_data(butler1, "base.yaml")
296 # Trigger full cache load.
297 butler2.get_dataset_type("flat")
298 # Have an external process register a dataset type.
299 butler1.registry.registerDatasetType(
300 DatasetType("test_type", ["instrument"], "int", universe=butler1.dimensions)
301 )
302 # Try to read the new dataset type -- this is a cache miss that
303 # triggers fetch of a single dataset type.
304 dt = butler2.get_dataset_type("test_type")
305 self.assertEqual(dt.name, "test_type")
306 self.assertEqual(list(dt.dimensions.names), ["instrument"])
307 # Read it again -- this time it should pull from the cache.
308 dt = butler2.get_dataset_type("test_type")
309 self.assertEqual(dt.name, "test_type")
310 self.assertEqual(list(dt.dimensions.names), ["instrument"])
311 # Do a query that uses the dataset type's tags table.
312 self.assertEqual(
313 butler2.query_datasets("test_type", collections="*", find_first=False, explain=False), []
314 )
316 def testDimensions(self):
317 """Tests for `SqlRegistry.insertDimensionData`,
318 `SqlRegistry.syncDimensionData`, and `SqlRegistry.expandDataId`.
319 """
320 butler = self.make_butler()
321 registry = butler.registry
322 dimensionName = "instrument"
323 dimension = registry.dimensions[dimensionName]
324 dimensionValue = {
325 "name": "DummyCam",
326 "visit_max": 10,
327 "visit_system": 0,
328 "exposure_max": 10,
329 "detector_max": 2,
330 "class_name": "lsst.pipe.base.Instrument",
331 }
332 registry.insertDimensionData(dimensionName, dimensionValue)
333 # Inserting the same value twice should fail
334 with self.assertRaises(sqlalchemy.exc.IntegrityError):
335 registry.insertDimensionData(dimensionName, dimensionValue)
336 # expandDataId should retrieve the record we just inserted
337 self.assertEqual(
338 registry.expandDataId(instrument="DummyCam", dimensions=dimension.minimal_group)
339 .records[dimensionName]
340 .toDict(),
341 dimensionValue,
342 )
343 # expandDataId should raise if there is no record with the given ID.
344 with self.assertRaises(DataIdValueError):
345 registry.expandDataId({"instrument": "Unknown"}, dimensions=dimension.minimal_group)
346 # band doesn't have a table; insert should fail.
347 with self.assertRaises(TypeError):
348 registry.insertDimensionData("band", {"band": "i"})
349 dimensionName2 = "physical_filter"
350 dimension2 = registry.dimensions[dimensionName2]
351 dimensionValue2 = {"name": "DummyCam_i", "band": "i"}
352 # Missing required dependency ("instrument") should fail
353 with self.assertRaises(KeyError):
354 registry.insertDimensionData(dimensionName2, dimensionValue2)
355 # Adding required dependency should fix the failure
356 dimensionValue2["instrument"] = "DummyCam"
357 registry.insertDimensionData(dimensionName2, dimensionValue2)
358 # expandDataId should retrieve the record we just inserted.
359 self.assertEqual(
360 registry.expandDataId(
361 instrument="DummyCam", physical_filter="DummyCam_i", dimensions=dimension2.minimal_group
362 )
363 .records[dimensionName2]
364 .toDict(),
365 dimensionValue2,
366 )
367 # Use syncDimensionData to insert a new record successfully.
368 dimensionName3 = "detector"
369 dimensionValue3 = {
370 "instrument": "DummyCam",
371 "id": 1,
372 "full_name": "one",
373 "name_in_raft": "zero",
374 "purpose": "SCIENCE",
375 }
376 self.assertTrue(registry.syncDimensionData(dimensionName3, dimensionValue3))
377 # Sync that again. Note that one field ("raft") is NULL, and that
378 # should be okay.
379 self.assertFalse(registry.syncDimensionData(dimensionName3, dimensionValue3))
380 # Now try that sync with the same primary key but a different value.
381 # This should fail.
382 with self.assertRaises(ConflictingDefinitionError):
383 registry.syncDimensionData(
384 dimensionName3,
385 {
386 "instrument": "DummyCam",
387 "id": 1,
388 "full_name": "one",
389 "name_in_raft": "four",
390 "purpose": "SCIENCE",
391 },
392 )
394 @unittest.skipIf(np is None, "numpy not available.")
395 def testNumpyDataId(self):
396 """Test that we can use a numpy int in a dataId."""
397 butler = self.make_butler()
398 registry = butler.registry
399 dimensionEntries = [
400 ("instrument", {"instrument": "DummyCam"}),
401 ("physical_filter", {"instrument": "DummyCam", "name": "d-r", "band": "R"}),
402 ("day_obs", {"instrument": "DummyCam", "id": 20250101}),
403 # Using an np.int64 here fails unless Records.fromDict is also
404 # patched to look for numbers.Integral
405 (
406 "visit",
407 {
408 "instrument": "DummyCam",
409 "id": 42,
410 "name": "fortytwo",
411 "physical_filter": "d-r",
412 "day_obs": 20250101,
413 },
414 ),
415 ]
416 for args in dimensionEntries:
417 registry.insertDimensionData(*args)
419 # Try a normal integer and something that looks like an int but
420 # is not.
421 for visit_id in (42, np.int64(42)):
422 with self.subTest(visit_id=repr(visit_id), id_type=type(visit_id).__name__):
423 expanded = registry.expandDataId({"instrument": "DummyCam", "visit": visit_id})
424 self.assertEqual(expanded["visit"], int(visit_id))
425 self.assertIsInstance(expanded["visit"], int)
427 def testDataIdRelationships(self):
428 """Test that `SqlRegistry.expandDataId` raises an exception when the
429 given keys are inconsistent.
430 """
431 butler = self.make_butler()
432 self.load_data(butler, "base.yaml")
433 registry = butler.registry
434 # Insert a few more dimension records for the next test.
435 registry.insertDimensionData(
436 "day_obs",
437 {"instrument": "Cam1", "id": 20250101},
438 )
439 registry.insertDimensionData(
440 "group",
441 {"instrument": "Cam1", "name": "group1"},
442 )
443 registry.insertDimensionData(
444 "exposure",
445 {
446 "instrument": "Cam1",
447 "id": 1,
448 "obs_id": "one",
449 "physical_filter": "Cam1-G",
450 "group": "group1",
451 "day_obs": 20250101,
452 },
453 )
454 registry.insertDimensionData(
455 "group",
456 {"instrument": "Cam1", "name": "group2"},
457 )
458 registry.insertDimensionData(
459 "exposure",
460 {
461 "instrument": "Cam1",
462 "id": 2,
463 "obs_id": "two",
464 "physical_filter": "Cam1-G",
465 "group": "group2",
466 "day_obs": 20250101,
467 },
468 )
469 registry.insertDimensionData(
470 "visit_system",
471 {"instrument": "Cam1", "id": 0, "name": "one-to-one"},
472 )
473 registry.insertDimensionData(
474 "visit",
475 {"instrument": "Cam1", "id": 1, "name": "one", "physical_filter": "Cam1-G", "day_obs": 20250101},
476 )
477 registry.insertDimensionData(
478 "visit_definition",
479 {"instrument": "Cam1", "visit": 1, "exposure": 1},
480 )
481 with self.assertRaises(InconsistentDataIdError):
482 registry.expandDataId(
483 {"instrument": "Cam1", "visit": 1, "exposure": 2},
484 )
486 def testDataset(self):
487 """Basic tests for `SqlRegistry.insertDatasets`,
488 `SqlRegistry.getDataset`, and `SqlRegistry.removeDatasets`.
489 """
490 butler = self.make_butler()
491 registry = butler.registry
492 self.load_data(butler, "base.yaml")
493 run = "tésτ"
494 registry.registerRun(run)
495 datasetType = registry.getDatasetType("bias")
496 dataId = {"instrument": "Cam1", "detector": 2}
497 (ref,) = registry.insertDatasets(datasetType, dataIds=[dataId], run=run)
498 outRef = registry.getDataset(ref.id)
499 self.assertIsNotNone(ref.id)
500 self.assertEqual(ref, outRef)
501 with self.assertRaises(ConflictingDefinitionError):
502 registry.insertDatasets(datasetType, dataIds=[dataId], run=run)
503 registry.removeDatasets([ref])
504 self.assertIsNone(registry.findDataset(datasetType, dataId, collections=[run]))
506 def test_get_many_datasets(self):
507 butler = self.make_butler()
508 self.load_data(butler, "base.yaml", "datasets.yaml")
509 expected_refs = {
510 str(ref.id): ref
511 for ref in butler.query_all_datasets(["imported_g", "imported_r"], find_first=False)
512 }
514 # Set up a tagged collection containing a dataset used by the tests
515 # below. get_many_datasets() queries on tables shared between run
516 # collections and tagged collections, so this makes sure the tags don't
517 # interfere.
518 butler.collections.register("tagged", CollectionType.TAGGED)
519 butler.registry.associate("tagged", [expected_refs["60c8a65c-7290-4c38-b1de-e3b1cdcf872d"]])
521 # Empty input returns empty output.
522 self.assertEqual(butler.get_many_datasets([]), [])
523 # Datasets all of one type, but in different collections.
524 self.assertCountEqual(
525 butler.get_many_datasets(
526 ["60c8a65c-7290-4c38-b1de-e3b1cdcf872d", "d0bb04cd-d697-4a83-ba53-cdfcd58e3a0c"]
527 ),
528 [
529 expected_refs["60c8a65c-7290-4c38-b1de-e3b1cdcf872d"],
530 expected_refs["d0bb04cd-d697-4a83-ba53-cdfcd58e3a0c"],
531 ],
532 )
533 # Datasets of multiple types with different dimension groups.
534 self.assertCountEqual(
535 butler.get_many_datasets(
536 [
537 "60c8a65c-7290-4c38-b1de-e3b1cdcf872d",
538 "d0bb04cd-d697-4a83-ba53-cdfcd58e3a0c",
539 "87f3e68d-258d-41b7-8ea5-edf3557ccb30",
540 ]
541 ),
542 [
543 expected_refs["60c8a65c-7290-4c38-b1de-e3b1cdcf872d"],
544 expected_refs["d0bb04cd-d697-4a83-ba53-cdfcd58e3a0c"],
545 expected_refs["87f3e68d-258d-41b7-8ea5-edf3557ccb30"],
546 ],
547 )
548 # Missing datasets are omitted from the result.
549 self.assertCountEqual(
550 butler.get_many_datasets(
551 [
552 "238c3b83-f6e5-4ccb-a7b0-5028dec1dcbb",
553 "60c8a65c-7290-4c38-b1de-e3b1cdcf872d",
554 ]
555 ),
556 [expected_refs["60c8a65c-7290-4c38-b1de-e3b1cdcf872d"]],
557 )
558 # Duplicates are squashed in the result.
559 self.assertCountEqual(
560 butler.get_many_datasets(
561 [
562 "60c8a65c-7290-4c38-b1de-e3b1cdcf872d",
563 "60c8a65c-7290-4c38-b1de-e3b1cdcf872d",
564 ]
565 ),
566 [expected_refs["60c8a65c-7290-4c38-b1de-e3b1cdcf872d"]],
567 )
568 # Can use UUID instances as inputs instead of strings.
569 self.assertCountEqual(
570 butler.get_many_datasets(
571 [
572 uuid.UUID("60c8a65c-7290-4c38-b1de-e3b1cdcf872d"),
573 ]
574 ),
575 [expected_refs["60c8a65c-7290-4c38-b1de-e3b1cdcf872d"]],
576 )
577 # Bad ID format raises ValueError.
578 with self.assertRaises(ValueError):
579 butler.get_many_datasets(["not-a-valid-uuid"])
580 # Works with arbitrary iterables as input.
581 self.assertCountEqual(
582 butler.get_many_datasets(
583 itertools.chain(
584 ["60c8a65c-7290-4c38-b1de-e3b1cdcf872d", "d0bb04cd-d697-4a83-ba53-cdfcd58e3a0c"]
585 )
586 ),
587 [
588 expected_refs["60c8a65c-7290-4c38-b1de-e3b1cdcf872d"],
589 expected_refs["d0bb04cd-d697-4a83-ba53-cdfcd58e3a0c"],
590 ],
591 )
593 def test_fetch_run_dataset_ids(self):
594 butler = self.make_butler()
595 registry = butler._registry
596 self.load_data(butler, "base.yaml", "datasets.yaml")
597 dataset_ids = registry._fetch_run_dataset_ids("imported_r")
598 self.assertEqual(len(dataset_ids), 7)
599 refs = butler.query_all_datasets("imported_r")
600 self.assertCountEqual(dataset_ids, [ref.id for ref in refs])
602 def testFindDataset(self):
603 """Tests for `SqlRegistry.findDataset`."""
604 butler = self.make_butler()
605 registry = butler.registry
606 self.load_data(butler, "base.yaml")
607 run = "tésτ"
608 datasetType = registry.getDatasetType("bias")
609 dataId = {"instrument": "Cam1", "detector": 4}
610 registry.registerRun(run)
611 (inputRef,) = registry.insertDatasets(datasetType, dataIds=[dataId], run=run)
612 outputRef = registry.findDataset(datasetType, dataId, collections=[run])
613 self.assertEqual(outputRef, inputRef)
614 # Check that retrieval with invalid dataId raises
615 with self.assertRaises(LookupError):
616 dataId = {"instrument": "Cam1"} # no detector
617 registry.findDataset(datasetType, dataId, collections=run)
618 # Check that different dataIds match to different datasets
619 dataId1 = {"instrument": "Cam1", "detector": 1}
620 (inputRef1,) = registry.insertDatasets(datasetType, dataIds=[dataId1], run=run)
621 dataId2 = {"instrument": "Cam1", "detector": 2}
622 (inputRef2,) = registry.insertDatasets(datasetType, dataIds=[dataId2], run=run)
623 self.assertEqual(registry.findDataset(datasetType, dataId1, collections=run), inputRef1)
624 self.assertEqual(registry.findDataset(datasetType, dataId2, collections=run), inputRef2)
625 self.assertNotEqual(registry.findDataset(datasetType, dataId1, collections=run), inputRef2)
626 self.assertNotEqual(registry.findDataset(datasetType, dataId2, collections=run), inputRef1)
627 # Check that requesting a non-existing dataId returns None
628 nonExistingDataId = {"instrument": "Cam1", "detector": 3}
629 self.assertIsNone(registry.findDataset(datasetType, nonExistingDataId, collections=run))
630 # Search more than one collection, in which two have the right
631 # dataset type and another does not.
632 registry.registerRun("empty")
633 self.load_data(butler, "datasets.yaml")
634 bias1 = registry.findDataset("bias", instrument="Cam1", detector=2, collections=["imported_g"])
635 self.assertIsNotNone(bias1)
636 bias2 = registry.findDataset("bias", instrument="Cam1", detector=2, collections=["imported_r"])
637 self.assertIsNotNone(bias2)
638 self.assertEqual(
639 bias1,
640 registry.findDataset(
641 "bias", instrument="Cam1", detector=2, collections=["empty", "imported_g", "imported_r"]
642 ),
643 )
644 self.assertEqual(
645 bias2,
646 registry.findDataset(
647 "bias", instrument="Cam1", detector=2, collections=["empty", "imported_r", "imported_g"]
648 ),
649 )
650 # If the input data ID was an expanded DataCoordinate with records,
651 # then the output ref has records, too.
652 expanded_id = registry.expandDataId({"instrument": "Cam1", "detector": 2})
653 expanded_ref = registry.findDataset("bias", expanded_id, collections=["imported_r"])
654 self.assertTrue(expanded_ref.dataId.hasRecords())
655 # Search more than one collection, with one of them a CALIBRATION
656 # collection.
657 registry.registerCollection("Cam1/calib", CollectionType.CALIBRATION)
658 timespan = Timespan(
659 begin=astropy.time.Time("2020-01-01T01:00:00", format="isot", scale="tai"),
660 end=astropy.time.Time("2020-01-01T02:00:00", format="isot", scale="tai"),
661 )
662 registry.certify("Cam1/calib", [bias2], timespan=timespan)
663 self.assertEqual(
664 bias1,
665 registry.findDataset(
666 "bias",
667 instrument="Cam1",
668 detector=2,
669 collections=["empty", "imported_g", "Cam1/calib"],
670 timespan=timespan,
671 ),
672 )
673 self.assertEqual(
674 bias1,
675 registry.findDataset(
676 "bias",
677 instrument="Cam1",
678 detector=2,
679 # Calibration dataset type, with no calibration collection, but
680 # a timespan was provided.
681 collections=["imported_g"],
682 timespan=timespan,
683 ),
684 )
685 self.assertEqual(
686 bias2,
687 registry.findDataset(
688 "bias",
689 instrument="Cam1",
690 detector=2,
691 collections=["empty", "Cam1/calib", "imported_g"],
692 timespan=timespan,
693 ),
694 )
695 # If we try to search those same collections without a timespan, it
696 # should still work, since the CALIBRATION collection is ignored.
697 self.assertEqual(
698 bias1,
699 registry.findDataset(
700 "bias", instrument="Cam1", detector=2, collections=["empty", "imported_g", "Cam1/calib"]
701 ),
702 )
703 self.assertEqual(
704 bias1,
705 registry.findDataset(
706 "bias", instrument="Cam1", detector=2, collections=["empty", "Cam1/calib", "imported_g"]
707 ),
708 )
709 self.assertIsNone(
710 registry.findDataset("bias", instrument="Cam1", detector=2, collections=["Cam1/calib"])
711 )
712 # Test non-calibration dataset type.
713 registry.registerDatasetType(
714 DatasetType("noncalibration", ["instrument", "detector"], "int", universe=butler.dimensions)
715 )
716 (non_calibration_ref,) = registry.insertDatasets("noncalibration", dataIds=[dataId2], run=run)
717 self.assertIsNone(
718 registry.findDataset("noncalibration", instrument="Cam1", detector=2, collections=["imported_g"])
719 )
720 self.assertEqual(
721 non_calibration_ref,
722 registry.findDataset("noncalibration", instrument="Cam1", detector=2, collections=[run]),
723 )
724 # Timespan parameter is ignored for non-calibration dataset types.
725 self.assertIsNone(
726 registry.findDataset(
727 "noncalibration", instrument="Cam1", detector=2, collections=["imported_g"], timespan=timespan
728 )
729 )
730 self.assertEqual(
731 non_calibration_ref,
732 registry.findDataset(
733 "noncalibration", instrument="Cam1", detector=2, collections=[run], timespan=timespan
734 ),
735 )
736 self.assertEqual(
737 non_calibration_ref,
738 registry.findDataset(
739 "noncalibration",
740 instrument="Cam1",
741 detector=2,
742 collections=["Cam1/calib", run],
743 timespan=timespan,
744 ),
745 )
746 # Add a dataset type whose dimension group involves an "implied"
747 # dimension. ("physical_filter" implies "band".)
748 registry.registerDatasetType(
749 DatasetType(
750 "dt_with_implied",
751 [
752 "instrument",
753 "physical_filter",
754 ],
755 "int",
756 universe=butler.dimensions,
757 )
758 )
759 data_id = {"instrument": "Cam1", "physical_filter": "Cam1-G"}
760 (implied_ref,) = registry.insertDatasets("dt_with_implied", dataIds=[data_id], run=run)
761 found_ref = registry.findDataset("dt_with_implied", data_id, collections=[run])
762 self.assertEqual(implied_ref, found_ref)
763 # The "full" data ID with implied values is looked up, even though we
764 # provided only the "required" values.
765 self.assertTrue(found_ref.dataId.hasFull())
766 # The search ignores excess data ID values beyond the 'required' set.
767 # This is not the correct band value for this physical_filter, but
768 # the mismatch is ignored.
769 self.assertEqual(
770 implied_ref,
771 registry.findDataset(
772 "dt_with_implied",
773 {"instrument": "Cam1", "physical_filter": "Cam1-G", "band": "r"},
774 collections=[run],
775 ),
776 )
777 # Correct band value, wrong physical_filter.
778 self.assertIsNone(
779 registry.findDataset(
780 "dt_with_implied",
781 {"instrument": "Cam1", "physical_filter": "Cam1-R1", "band": "g"},
782 collections=[run],
783 ),
784 )
786 def testRemoveDatasetTypeSuccess(self):
787 """Test that SqlRegistry.removeDatasetType works when there are no
788 datasets of that type present.
789 """
790 butler = self.make_butler()
791 registry = butler.registry
792 self.load_data(butler, "base.yaml")
793 registry.removeDatasetType("flat")
794 with self.assertRaises(MissingDatasetTypeError):
795 registry.getDatasetType("flat")
797 def testRemoveDatasetTypeFailure(self):
798 """Test that SqlRegistry.removeDatasetType raises when there are
799 datasets of that type present or if the dataset type is for a
800 component.
801 """
802 butler = self.make_butler()
803 registry = butler.registry
804 self.load_data(butler, "base.yaml", "datasets.yaml")
805 with self.assertRaises(OrphanedRecordError):
806 registry.removeDatasetType("flat")
807 with self.assertRaises(DatasetTypeError):
808 registry.removeDatasetType(DatasetType.nameWithComponent("flat", "image"))
810 def testImportDatasetsUUID(self):
811 """Test for `SqlRegistry._importDatasets` with UUID dataset ID."""
812 if isinstance(self.datasetsManager, str):
813 if not self.datasetsManager.endswith(".ByDimensionsDatasetRecordStorageManagerUUID"): 813 ↛ 814line 813 didn't jump to line 814 because the condition on line 813 was never true
814 self.skipTest(f"Unexpected dataset manager {self.datasetsManager}")
815 elif isinstance(self.datasetsManager, dict) and not self.datasetsManager["cls"].endswith( 815 ↛ 818line 815 didn't jump to line 818 because the condition on line 815 was never true
816 ".ByDimensionsDatasetRecordStorageManagerUUID"
817 ):
818 self.skipTest(f"Unexpected dataset manager {self.datasetsManager['cls']}")
820 butler = self.make_butler()
821 registry = butler.registry
822 self.load_data(butler, "base.yaml")
823 for run in range(6):
824 registry.registerRun(f"run{run}")
825 datasetTypeBias = registry.getDatasetType("bias")
826 datasetTypeFlat = registry.getDatasetType("flat")
827 dataIdBias1 = {"instrument": "Cam1", "detector": 1}
828 dataIdBias2 = {"instrument": "Cam1", "detector": 2}
829 dataIdFlat1 = {"instrument": "Cam1", "detector": 1, "physical_filter": "Cam1-G", "band": "g"}
831 ref = DatasetRef(datasetTypeBias, dataIdBias1, run="run0")
832 (ref1,) = registry._importDatasets([ref], assume_new=True)
833 # UUID is used without change
834 self.assertEqual(ref.id, ref1.id)
836 # Inserting this ref with assume_new=True should fail, since this
837 # dataset exists.
838 with self.assertRaises(ConflictingDefinitionError):
839 registry._importDatasets([ref], assume_new=True)
841 # All different failure modes
842 refs = (
843 # Importing same DatasetRef with different dataset ID is an error
844 DatasetRef(datasetTypeBias, dataIdBias1, run="run0"),
845 # Same DatasetId but different DataId
846 DatasetRef(datasetTypeBias, dataIdBias2, id=ref1.id, run="run0"),
847 DatasetRef(datasetTypeFlat, dataIdFlat1, id=ref1.id, run="run0"),
848 # Same DatasetRef and DatasetId but different run
849 DatasetRef(datasetTypeBias, dataIdBias1, id=ref1.id, run="run1"),
850 )
851 for ref in refs:
852 with self.assertRaises(ConflictingDefinitionError):
853 registry._importDatasets([ref])
855 # Test for non-unique IDs, they can be re-imported multiple times.
856 for run, idGenMode in ((2, DatasetIdGenEnum.DATAID_TYPE), (4, DatasetIdGenEnum.DATAID_TYPE_RUN)):
857 with self.subTest(idGenMode=repr(idGenMode)):
858 # Make dataset ref with reproducible dataset ID.
859 ref = DatasetRef(datasetTypeBias, dataIdBias1, run=f"run{run}", id_generation_mode=idGenMode)
860 (ref1,) = registry._importDatasets([ref])
861 self.assertIsInstance(ref1.id, uuid.UUID)
862 self.assertEqual(ref1.id.version, 5)
863 self.assertEqual(ref1.id, ref.id)
865 # Importing it again is OK
866 (ref2,) = registry._importDatasets([ref1])
867 self.assertEqual(ref2.id, ref1.id)
869 # Cannot import to different run with the same ID
870 ref = DatasetRef(datasetTypeBias, dataIdBias1, id=ref1.id, run=f"run{run + 1}")
871 with self.assertRaises(ConflictingDefinitionError):
872 registry._importDatasets([ref])
874 ref = DatasetRef(
875 datasetTypeBias, dataIdBias1, run=f"run{run + 1}", id_generation_mode=idGenMode
876 )
877 if idGenMode is DatasetIdGenEnum.DATAID_TYPE:
878 # Cannot import same DATAID_TYPE ref into a new run
879 with self.assertRaises(ConflictingDefinitionError):
880 (ref2,) = registry._importDatasets([ref])
881 else:
882 # DATAID_TYPE_RUN ref can be imported into a new run
883 (ref2,) = registry._importDatasets([ref])
885 def testComponentLookups(self):
886 """Test searching for component datasets via their parents.
888 Components can no longer be found by registry. This test checks
889 that this now fails.
890 """
891 butler = self.make_butler()
892 registry = butler.registry
893 self.load_data(butler, "base.yaml", "datasets.yaml")
894 # Test getting the child dataset type (which does still exist in the
895 # Registry), and check for consistency with
896 # DatasetRef.makeComponentRef.
897 collection = "imported_g"
898 parentType = registry.getDatasetType("bias")
899 childType = registry.getDatasetType("bias.wcs")
900 parentRefResolved = registry.findDataset(
901 parentType, collections=collection, instrument="Cam1", detector=1
902 )
903 self.assertIsInstance(parentRefResolved, DatasetRef)
904 self.assertEqual(childType, parentRefResolved.makeComponentRef("wcs").datasetType)
905 # Search for a single dataset with findDataset.
906 with self.assertRaises(DatasetTypeError):
907 registry.findDataset("bias.wcs", collections=collection, dataId=parentRefResolved.dataId)
909 def testInvalidDatasetTypeName(self):
910 """Test that a syntactically invalid dataset type name is reported as
911 a bad expression rather than as a missing dataset type.
912 """
913 butler = self.make_butler()
914 registry = butler.registry
915 self.load_data(butler, "base.yaml", "datasets.yaml")
916 with self.assertRaisesRegex(DatasetTypeExpressionError, "does not look like a valid"):
917 registry.getDatasetType("...")
918 # A search expression reports the name as invalid rather than
919 # complaining that "." introduces an unsupported component.
920 with self.assertRaisesRegex(DatasetTypeExpressionError, "does not look like a valid"):
921 registry.queryDatasetTypes("...")
922 with self.assertRaisesRegex(DatasetTypeExpressionError, "Component dataset types"):
923 registry.queryDatasetTypes("bias.image")
924 # A well-formed name that is simply not registered is still missing
925 # rather than invalid.
926 with self.assertRaises(MissingDatasetTypeError):
927 registry.getDatasetType("not_bias")
928 # Names that cannot be registered do not prevent a search from
929 # reporting the other names it could not find.
930 missing: list[str] = []
931 registry.queryDatasetTypes(["bias", "not-a-legal-name"], missing=missing)
932 self.assertEqual(missing, ["not-a-legal-name"])
934 def testCollections(self):
935 """Tests for registry methods that manage collections."""
936 butler = self.make_butler()
937 registry = butler.registry
938 other_registry = butler.clone().registry
939 self.load_data(butler, "base.yaml", "datasets.yaml")
940 run1 = "imported_g"
941 run2 = "imported_r"
942 # Test setting a collection docstring after it has been created.
943 registry.setCollectionDocumentation(run1, "doc for run1")
944 self.assertEqual(registry.getCollectionDocumentation(run1), "doc for run1")
945 registry.setCollectionDocumentation(run1, None)
946 self.assertIsNone(registry.getCollectionDocumentation(run1))
947 datasetType = "bias"
948 # Find some datasets via their run's collection.
949 dataId1 = {"instrument": "Cam1", "detector": 1}
950 ref1 = registry.findDataset(datasetType, dataId1, collections=run1)
951 self.assertIsNotNone(ref1)
952 dataId2 = {"instrument": "Cam1", "detector": 2}
953 ref2 = registry.findDataset(datasetType, dataId2, collections=run1)
954 self.assertIsNotNone(ref2)
955 # Associate those into a new collection, then look for them there.
956 tag1 = "tag1"
957 registry.registerCollection(tag1, type=CollectionType.TAGGED, doc="doc for tag1")
958 # Check that we can query for old and new collections by type.
959 self.assertEqual(set(registry.queryCollections(collectionTypes=CollectionType.RUN)), {run1, run2})
960 self.assertEqual(
961 set(registry.queryCollections(collectionTypes={CollectionType.TAGGED, CollectionType.RUN})),
962 {tag1, run1, run2},
963 )
964 self.assertEqual(registry.getCollectionDocumentation(tag1), "doc for tag1")
965 registry.associate(tag1, [ref1, ref2])
966 self.assertEqual(registry.findDataset(datasetType, dataId1, collections=tag1), ref1)
967 self.assertEqual(registry.findDataset(datasetType, dataId2, collections=tag1), ref2)
968 # Disassociate one and verify that we can't it there anymore...
969 registry.disassociate(tag1, [ref1])
970 self.assertIsNone(registry.findDataset(datasetType, dataId1, collections=tag1))
971 # ...but we can still find ref2 in tag1, and ref1 in the run.
972 self.assertEqual(registry.findDataset(datasetType, dataId1, collections=run1), ref1)
973 self.assertEqual(registry.findDataset(datasetType, dataId2, collections=tag1), ref2)
974 collections = set(registry.queryCollections())
975 self.assertEqual(collections, {run1, run2, tag1})
976 # Associate both refs into tag1 again; ref2 is already there, but that
977 # should be a harmless no-op.
978 registry.associate(tag1, [ref1, ref2])
979 self.assertEqual(registry.findDataset(datasetType, dataId1, collections=tag1), ref1)
980 self.assertEqual(registry.findDataset(datasetType, dataId2, collections=tag1), ref2)
981 # Get a different dataset (from a different run) that has the same
982 # dataset type and data ID as ref2.
983 ref2b = registry.findDataset(datasetType, dataId2, collections=run2)
984 self.assertNotEqual(ref2, ref2b)
985 # Attempting to associate that into tag1 should be an error.
986 with self.assertRaises(ConflictingDefinitionError):
987 registry.associate(tag1, [ref2b])
988 # That error shouldn't have messed up what we had before.
989 self.assertEqual(registry.findDataset(datasetType, dataId1, collections=tag1), ref1)
990 self.assertEqual(registry.findDataset(datasetType, dataId2, collections=tag1), ref2)
991 # Attempt to associate the conflicting dataset again, this time with
992 # a dataset that isn't in the collection and won't cause a conflict.
993 # Should also fail without modifying anything.
994 dataId3 = {"instrument": "Cam1", "detector": 3}
995 ref3 = registry.findDataset(datasetType, dataId3, collections=run1)
996 with self.assertRaises(ConflictingDefinitionError):
997 registry.associate(tag1, [ref3, ref2b])
998 self.assertEqual(registry.findDataset(datasetType, dataId1, collections=tag1), ref1)
999 self.assertEqual(registry.findDataset(datasetType, dataId2, collections=tag1), ref2)
1000 self.assertIsNone(registry.findDataset(datasetType, dataId3, collections=tag1))
1001 # Register a chained collection that searches [tag1, run2]
1002 chain1 = "chain1"
1003 registry.registerCollection(chain1, type=CollectionType.CHAINED)
1004 self.assertIs(registry.getCollectionType(chain1), CollectionType.CHAINED)
1005 # Chained collection exists, but has no collections in it.
1006 self.assertFalse(registry.getCollectionChain(chain1))
1007 # If we query for all collections, we should get the chained collection
1008 # if we don't ask to flatten it (i.e. yield only its children) or if we
1009 # explicitly ask to include it too.
1010 self.assertEqual(set(registry.queryCollections(flattenChains=False)), {tag1, run1, run2, chain1})
1011 self.assertEqual(set(registry.queryCollections(flattenChains=True)), {tag1, run1, run2})
1012 self.assertEqual(
1013 set(registry.queryCollections(flattenChains=True, includeChains=True)), {tag1, run1, run2, chain1}
1014 )
1015 # Attempt to set its child collections to something circular; that
1016 # should fail.
1017 with self.assertRaises(ValueError):
1018 registry.setCollectionChain(chain1, [tag1, chain1])
1019 # Add the child collections.
1020 registry.setCollectionChain(chain1, [tag1, run2])
1021 self.assertEqual(list(registry.getCollectionChain(chain1)), [tag1, run2])
1022 self.assertEqual(registry.getCollectionParentChains(tag1), {chain1})
1023 self.assertEqual(registry.getCollectionParentChains(run2), {chain1})
1024 # Refresh the other registry that points to the same repo, and make
1025 # sure it can see the things we've done (note that this does require
1026 # an explicit refresh(); that's the documented behavior, because
1027 # caching is ~impossible otherwise).
1028 if other_registry is not None: 1028 ↛ 1035line 1028 didn't jump to line 1035 because the condition on line 1028 was always true
1029 other_registry.refresh()
1030 self.assertEqual(list(other_registry.getCollectionChain(chain1)), [tag1, run2])
1031 self.assertEqual(other_registry.getCollectionParentChains(tag1), {chain1})
1032 self.assertEqual(other_registry.getCollectionParentChains(run2), {chain1})
1033 # Searching for dataId1 or dataId2 in the chain should return ref1 and
1034 # ref2, because both are in tag1.
1035 self.assertEqual(registry.findDataset(datasetType, dataId1, collections=chain1), ref1)
1036 self.assertEqual(registry.findDataset(datasetType, dataId2, collections=chain1), ref2)
1037 # Now disassociate ref2 from tag1. The search (for bias) with
1038 # dataId2 in chain1 should then:
1039 # 1. not find it in tag1
1040 # 2. find a different dataset in run2
1041 registry.disassociate(tag1, [ref2])
1042 ref2b = registry.findDataset(datasetType, dataId2, collections=chain1)
1043 self.assertNotEqual(ref2b, ref2)
1044 self.assertEqual(ref2b, registry.findDataset(datasetType, dataId2, collections=run2))
1045 # Define a new chain so we can test recursive chains.
1046 chain2 = "chain2"
1047 registry.registerCollection(chain2, type=CollectionType.CHAINED)
1048 registry.setCollectionChain(chain2, [run2, chain1])
1049 self.assertEqual(registry.getCollectionParentChains(chain1), {chain2})
1050 self.assertEqual(registry.getCollectionParentChains(run2), {chain1, chain2})
1052 if self.supportsCollectionRegex: 1052 ↛ 1054line 1052 didn't jump to line 1054 because the condition on line 1052 was never true
1053 # Query for collections matching a regex.
1054 with self.assertWarns(FutureWarning):
1055 self.assertCountEqual(
1056 list(registry.queryCollections(re.compile("imported_."), flattenChains=False)),
1057 ["imported_r", "imported_g"],
1058 )
1059 # Query for collections matching a regex or an explicit str.
1060 with self.assertWarns(FutureWarning):
1061 self.assertCountEqual(
1062 list(
1063 registry.queryCollections([re.compile("imported_."), "chain1"], flattenChains=False)
1064 ),
1065 ["imported_r", "imported_g", "chain1"],
1066 )
1067 # Same queries as the regex ones above, but using globs instead of
1068 # regex.
1069 self.assertCountEqual(
1070 list(registry.queryCollections("imported_*", flattenChains=False)),
1071 ["imported_r", "imported_g"],
1072 )
1073 # Query for collections matching a regex or an explicit str.
1074 self.assertCountEqual(
1075 list(registry.queryCollections(["imported_*", "chain1"], flattenChains=False)),
1076 ["imported_r", "imported_g", "chain1"],
1077 )
1078 # Query for collection matching chain names, by flattening it should
1079 # only return non-chain names.
1080 self.assertCountEqual(list(registry.queryCollections("chain?", flattenChains=True)), [tag1, run2])
1081 # Query for collection matching chain name, by flattening and
1082 # asking to include chains it should return everything.
1083 self.assertCountEqual(
1084 list(registry.queryCollections("chain*", flattenChains=True, includeChains=True)),
1085 [tag1, run2, chain1, chain2],
1086 )
1087 # Order of children in chained collections is preserved.
1088 self.assertEqual(list(registry.queryCollections("chain1", flattenChains=True)), [tag1, run2])
1089 self.assertEqual(list(registry.queryCollections("cha*2", flattenChains=True)), [run2, tag1])
1090 self.assertEqual(
1091 list(registry.queryCollections("chain1", flattenChains=True, includeChains=True)),
1092 [chain1, tag1, run2],
1093 )
1094 self.assertEqual(
1095 list(registry.queryCollections("chain2", flattenChains=True, includeChains=True)),
1096 [chain2, run2, chain1, tag1],
1097 )
1099 # Search for bias with dataId1 should find it via tag1 in chain2,
1100 # recursing, because is not in run1.
1101 self.assertIsNone(registry.findDataset(datasetType, dataId1, collections=run2))
1102 self.assertEqual(registry.findDataset(datasetType, dataId1, collections=chain2), ref1)
1103 # Search for bias with dataId2 should find it in run2 (ref2b).
1104 self.assertEqual(registry.findDataset(datasetType, dataId2, collections=chain2), ref2b)
1105 # Search for a flat that is in run2. That should not be found
1106 # at the front of chain2, because of the restriction to bias
1107 # on run2 there, but it should be found in at the end of chain1.
1108 dataId4 = {"instrument": "Cam1", "detector": 3, "physical_filter": "Cam1-R2"}
1109 ref4 = registry.findDataset("flat", dataId4, collections=run2)
1110 self.assertIsNotNone(ref4)
1111 self.assertEqual(ref4, registry.findDataset("flat", dataId4, collections=chain2))
1112 # Deleting a collection that's part of a CHAINED collection is not
1113 # allowed, and is exception-safe.
1114 with self.assertRaises(sqlalchemy.exc.IntegrityError):
1115 registry.removeCollection(run2)
1116 self.assertEqual(registry.getCollectionType(run2), CollectionType.RUN)
1117 with self.assertRaises(sqlalchemy.exc.IntegrityError):
1118 registry.removeCollection(chain1)
1119 self.assertEqual(registry.getCollectionType(chain1), CollectionType.CHAINED)
1120 # Actually remove chain2, test that it's gone by asking for its type.
1121 registry.removeCollection(chain2)
1122 with self.assertRaises(MissingCollectionError):
1123 registry.getCollectionType(chain2)
1124 # Actually remove run2 and chain1, which should work now.
1125 registry.removeCollection(chain1)
1126 registry.removeCollection(run2)
1127 with self.assertRaises(MissingCollectionError):
1128 registry.getCollectionType(run2)
1129 with self.assertRaises(MissingCollectionError):
1130 registry.getCollectionType(chain1)
1131 # Remove tag1 as well, just to test that we can remove TAGGED
1132 # collections.
1133 registry.removeCollection(tag1)
1134 with self.assertRaises(MissingCollectionError):
1135 registry.getCollectionType(tag1)
1137 def test_collection_clearing(self) -> None:
1138 """Test that we can delete TAGGED and CALIBRATION collections without
1139 manually removing all associated datasets first.
1140 """
1141 butler = self.make_butler()
1142 self.load_data(butler, "base.yaml", "datasets.yaml")
1144 # This brings in datasets of two different types, with the same
1145 # dimension group.
1146 original_datasets = tuple(butler.query_all_datasets("imported_r", instrument="Cam1", detector=2))
1147 self.assertEqual(len(original_datasets), 2)
1149 # Test tagged collections.
1150 butler.collections.register("tag1", CollectionType.TAGGED)
1151 butler.collections.register("tag2", CollectionType.TAGGED)
1152 butler.registry.associate("tag1", original_datasets)
1153 butler.registry.associate("tag2", original_datasets)
1154 butler.collections.x_remove("tag1")
1155 with self.assertRaises(MissingCollectionError):
1156 butler.collections.get_info("tag1")
1157 # Make sure there was no collateral damage -- tag2 should still be
1158 # intact.
1159 self.assertEqual(set(butler.query_all_datasets("tag2")), set(original_datasets))
1161 # Test calibration collections.
1162 butler.collections.register("calib1", CollectionType.CALIBRATION)
1163 butler.collections.register("calib2", CollectionType.CALIBRATION)
1164 butler.registry.certify("calib1", original_datasets, Timespan(None, None))
1165 butler.registry.certify("calib2", original_datasets, Timespan(None, None))
1166 butler.collections.x_remove("calib1")
1167 with self.assertRaises(MissingCollectionError):
1168 butler.collections.get_info("calib1")
1169 # Make sure there was no collateral damage -- calib2 should still be
1170 # intact.
1171 self.assertEqual(set(butler.query_all_datasets("calib2")), set(original_datasets))
1173 def testCollectionChainCaching(self):
1174 butler = self.make_butler()
1175 registry = butler.registry
1176 with registry.caching_context():
1177 registry.registerCollection("a")
1178 registry.registerCollection("chain", CollectionType.CHAINED)
1179 # There used to be a caching bug (DM-43750) that would throw an
1180 # exception if you modified a collection chain for a collection
1181 # that was already in the cache.
1182 registry.setCollectionChain("chain", ["a"])
1183 self.assertEqual(list(registry.getCollectionChain("chain")), ["a"])
1185 def testCollectionChainFlatten(self):
1186 """Test that `SqlRegistry.setCollectionChain` obeys its 'flatten'
1187 option.
1188 """
1189 butler = self.make_butler()
1190 registry = butler.registry
1191 registry.registerCollection("inner", CollectionType.CHAINED)
1192 registry.registerCollection("innermost", CollectionType.RUN)
1193 registry.setCollectionChain("inner", ["innermost"])
1194 registry.registerCollection("outer", CollectionType.CHAINED)
1195 registry.setCollectionChain("outer", ["inner"], flatten=False)
1196 self.assertEqual(list(registry.getCollectionChain("outer")), ["inner"])
1197 registry.setCollectionChain("outer", ["inner"], flatten=True)
1198 self.assertEqual(list(registry.getCollectionChain("outer")), ["innermost"])
1200 def testCollectionChainPrependConcurrency(self):
1201 """Verify that locking via database row locks is working as
1202 expected.
1203 """
1205 def blocked_thread_func(butler: Butler):
1206 # This call will become blocked after it has decided on positions
1207 # for the new children in the collection chain, but before
1208 # inserting them.
1209 butler.collections.prepend_chain("chain", ["a"])
1211 def unblocked_thread_func(butler: Butler):
1212 butler.collections.prepend_chain("chain", ["b"])
1214 registry = self._do_collection_concurrency_test(blocked_thread_func, unblocked_thread_func)
1216 # blocked_thread_func should have finished first, inserting "a".
1217 # unblocked_thread_func should have finished second, prepending "b".
1218 self.assertEqual(("b", "a"), registry.getCollectionChain("chain"))
1220 def testCollectionChainReplaceConcurrency(self):
1221 """Verify that locking via database row locks is working as
1222 expected.
1223 """
1225 def blocked_thread_func(butler: Butler):
1226 # This call will become blocked after deleting children, but before
1227 # inserting new ones.
1228 butler.collections.redefine_chain("chain", ["a"])
1230 def unblocked_thread_func(butler: Butler):
1231 butler.collections.redefine_chain("chain", ["b"])
1233 registry = self._do_collection_concurrency_test(blocked_thread_func, unblocked_thread_func)
1235 # blocked_thread_func should have finished first.
1236 # unblocked_thread_func should have finished second, overwriting the
1237 # chain with "b".
1238 self.assertEqual(("b",), registry.getCollectionChain("chain"))
1240 def testCollectionChainRemoveConcurrency(self):
1241 def blocked_thread_func(butler: Butler):
1242 # This call will become blocked after taking the lock, but before
1243 # deleting the children.
1244 butler.collections.remove_from_chain("chain", ["b"])
1246 def unblocked_thread_func(butler: Butler):
1247 butler.collections.redefine_chain("chain", ["b", "a"])
1249 registry = self._do_collection_concurrency_test(blocked_thread_func, unblocked_thread_func)
1251 # blocked_thread_func should have finished first, removing "b".
1252 # unblocked_thread_func should have finished second, putting "b" back.
1253 self.assertEqual(("b", "a"), registry.getCollectionChain("chain"))
1255 def _do_collection_concurrency_test(
1256 self, blocked_thread_func: Callable[[Butler]], unblocked_thread_func: Callable[[Butler]]
1257 ) -> SqlRegistry:
1258 # This function:
1259 # 1. Sets up two registries pointing at the same database.
1260 # 2. Start running 'blocked_thread_func' in a background thread,
1261 # arranging for it to become blocked during a critical section in
1262 # the collections manager.
1263 # 3. Wait for 'blocked_thread_func' to reach the critical section
1264 # 4. Start running 'unblocked_thread_func'.
1265 # 5. Allow both functions to run to completion.
1267 # Set up two registries pointing to the same DB
1268 butler1 = self.make_butler()
1269 butler2 = butler1.clone()
1270 registry1 = butler1._registry
1271 assert isinstance(registry1, SqlRegistry)
1272 registry2 = butler2._registry
1274 with contextlib.suppress(AttributeError):
1275 if ":memory:" in str(registry2._db):
1276 raise unittest.SkipTest("Testing concurrency requires two connections to the same DB.")
1278 registry1.registerCollection("chain", CollectionType.CHAINED)
1279 for collection in ["a", "b"]:
1280 registry1.registerCollection(collection)
1282 # Arrange for registry1 to block during its critical section, allowing
1283 # us to detect this and control when it becomes unblocked.
1284 enter_barrier = Barrier(2, timeout=60)
1285 exit_barrier = Barrier(2, timeout=60)
1287 def wait_for_barrier():
1288 enter_barrier.wait()
1289 exit_barrier.wait()
1291 registry1._managers.collections._block_for_concurrency_test = wait_for_barrier
1293 with ThreadPoolExecutor(max_workers=1) as exec1:
1294 with ThreadPoolExecutor(max_workers=1) as exec2:
1295 future1 = exec1.submit(blocked_thread_func, butler1)
1296 enter_barrier.wait()
1298 # At this point registry 1 has entered the critical section and
1299 # is waiting for us to release it. Start the other thread.
1300 future2 = exec2.submit(unblocked_thread_func, butler2)
1301 # thread2 should block inside a database call, but we have no
1302 # way to detect when it is in this state.
1303 time.sleep(0.200)
1305 # Let the threads run to completion.
1306 exit_barrier.wait()
1307 future1.result()
1308 future2.result()
1310 return registry1
1312 def testBasicTransaction(self):
1313 """Test that all operations within a single transaction block are
1314 rolled back if an exception propagates out of the block.
1315 """
1316 butler = self.make_butler()
1317 registry = butler.registry
1318 storageClass = StorageClass("testDatasetType")
1319 registry.storageClasses.registerStorageClass(storageClass)
1320 with registry.transaction():
1321 registry.insertDimensionData("instrument", {"name": "Cam1", "class_name": "A"})
1322 with self.assertRaises(ValueError):
1323 with registry.transaction():
1324 registry.insertDimensionData("instrument", {"name": "Cam2"})
1325 raise ValueError("Oops, something went wrong")
1326 # Cam1 should exist
1327 self.assertEqual(registry.expandDataId(instrument="Cam1").records["instrument"].class_name, "A")
1328 # But Cam2 and Cam3 should both not exist
1329 with self.assertRaises(DataIdValueError):
1330 registry.expandDataId(instrument="Cam2")
1331 with self.assertRaises(DataIdValueError):
1332 registry.expandDataId(instrument="Cam3")
1334 def testNestedTransaction(self):
1335 """Test that operations within a transaction block are not rolled back
1336 if an exception propagates out of an inner transaction block and is
1337 then caught.
1338 """
1339 butler = self.make_butler()
1340 registry = butler.registry
1341 dimension = registry.dimensions["instrument"]
1342 dataId1 = {"instrument": "DummyCam"}
1343 dataId2 = {"instrument": "DummyCam2"}
1344 checkpointReached = False
1345 with registry.transaction():
1346 # This should be added and (ultimately) committed.
1347 registry.insertDimensionData(dimension, dataId1)
1348 with self.assertRaises(sqlalchemy.exc.IntegrityError):
1349 with registry.transaction(savepoint=True):
1350 # This does not conflict, and should succeed (but not
1351 # be committed).
1352 registry.insertDimensionData(dimension, dataId2)
1353 checkpointReached = True
1354 # This should conflict and raise, triggering a rollback
1355 # of the previous insertion within the same transaction
1356 # context, but not the original insertion in the outer
1357 # block.
1358 registry.insertDimensionData(dimension, dataId1)
1359 self.assertTrue(checkpointReached)
1360 self.assertIsNotNone(registry.expandDataId(dataId1, dimensions=dimension.minimal_group))
1361 with self.assertRaises(DataIdValueError):
1362 registry.expandDataId(dataId2, dimensions=dimension.minimal_group)
1364 def testInstrumentDimensions(self):
1365 """Test queries involving only instrument dimensions, with no joins to
1366 skymap.
1367 """
1368 butler = self.make_butler()
1369 registry = butler.registry
1371 # need a bunch of dimensions and datasets for test
1372 registry.insertDimensionData(
1373 "instrument", dict(name="DummyCam", visit_max=25, exposure_max=300, detector_max=6)
1374 )
1375 registry.insertDimensionData("day_obs", dict(instrument="DummyCam", id=20250101))
1376 registry.insertDimensionData(
1377 "physical_filter",
1378 dict(instrument="DummyCam", name="dummy_r", band="r"),
1379 dict(instrument="DummyCam", name="dummy_i", band="i"),
1380 )
1381 registry.insertDimensionData(
1382 "detector", *[dict(instrument="DummyCam", id=i, full_name=str(i)) for i in range(1, 6)]
1383 )
1384 registry.insertDimensionData(
1385 "visit",
1386 dict(instrument="DummyCam", id=10, name="ten", physical_filter="dummy_i", day_obs=20250101),
1387 dict(instrument="DummyCam", id=11, name="eleven", physical_filter="dummy_r", day_obs=20250101),
1388 dict(instrument="DummyCam", id=20, name="twelve", physical_filter="dummy_r", day_obs=20250101),
1389 )
1390 registry.insertDimensionData(
1391 "group",
1392 dict(instrument="DummyCam", name="ten"),
1393 dict(instrument="DummyCam", name="eleven"),
1394 dict(instrument="DummyCam", name="twelve"),
1395 )
1396 for i in range(1, 6):
1397 registry.insertDimensionData(
1398 "visit_detector_region",
1399 dict(instrument="DummyCam", visit=10, detector=i),
1400 dict(instrument="DummyCam", visit=11, detector=i),
1401 dict(instrument="DummyCam", visit=20, detector=i),
1402 )
1403 registry.insertDimensionData(
1404 "exposure",
1405 dict(
1406 instrument="DummyCam",
1407 id=100,
1408 obs_id="100",
1409 physical_filter="dummy_i",
1410 group="ten",
1411 day_obs=20250101,
1412 ),
1413 dict(
1414 instrument="DummyCam",
1415 id=101,
1416 obs_id="101",
1417 physical_filter="dummy_i",
1418 group="ten",
1419 day_obs=20250101,
1420 ),
1421 dict(
1422 instrument="DummyCam",
1423 id=110,
1424 obs_id="110",
1425 physical_filter="dummy_r",
1426 group="eleven",
1427 day_obs=20250101,
1428 ),
1429 dict(
1430 instrument="DummyCam",
1431 id=111,
1432 obs_id="111",
1433 physical_filter="dummy_r",
1434 group="eleven",
1435 day_obs=20250101,
1436 ),
1437 dict(
1438 instrument="DummyCam",
1439 id=200,
1440 obs_id="200",
1441 physical_filter="dummy_r",
1442 group="twelve",
1443 day_obs=20250101,
1444 ),
1445 dict(
1446 instrument="DummyCam",
1447 id=201,
1448 obs_id="201",
1449 physical_filter="dummy_r",
1450 group="twelve",
1451 day_obs=20250101,
1452 ),
1453 )
1454 registry.insertDimensionData(
1455 "visit_definition",
1456 dict(instrument="DummyCam", exposure=100, visit=10),
1457 dict(instrument="DummyCam", exposure=101, visit=10),
1458 dict(instrument="DummyCam", exposure=110, visit=11),
1459 dict(instrument="DummyCam", exposure=111, visit=11),
1460 dict(instrument="DummyCam", exposure=200, visit=20),
1461 dict(instrument="DummyCam", exposure=201, visit=20),
1462 )
1463 # dataset types
1464 run1 = "test1_r"
1465 run2 = "test2_r"
1466 tagged2 = "test2_t"
1467 registry.registerRun(run1)
1468 registry.registerRun(run2)
1469 registry.registerCollection(tagged2)
1470 storageClass = StorageClass("testDataset")
1471 registry.storageClasses.registerStorageClass(storageClass)
1472 rawType = DatasetType(
1473 name="RAW",
1474 dimensions=registry.dimensions.conform(("instrument", "exposure", "detector")),
1475 storageClass=storageClass,
1476 )
1477 registry.registerDatasetType(rawType)
1478 calexpType = DatasetType(
1479 name="CALEXP",
1480 dimensions=registry.dimensions.conform(("instrument", "visit", "detector")),
1481 storageClass=storageClass,
1482 )
1483 registry.registerDatasetType(calexpType)
1485 # add pre-existing datasets
1486 for exposure in (100, 101, 110, 111):
1487 for detector in (1, 2, 3):
1488 # note that only 3 of 5 detectors have datasets
1489 dataId = dict(instrument="DummyCam", exposure=exposure, detector=detector)
1490 (ref,) = registry.insertDatasets(rawType, dataIds=[dataId], run=run1)
1491 # exposures 100 and 101 appear in both run1 and tagged2.
1492 # 100 has different datasets in the different collections
1493 # 101 has the same dataset in both collections.
1494 if exposure == 100:
1495 (ref,) = registry.insertDatasets(rawType, dataIds=[dataId], run=run2)
1496 if exposure in (100, 101):
1497 registry.associate(tagged2, [ref])
1498 # Add pre-existing datasets to tagged2.
1499 for exposure in (200, 201):
1500 for detector in (3, 4, 5):
1501 # note that only 3 of 5 detectors have datasets
1502 dataId = dict(instrument="DummyCam", exposure=exposure, detector=detector)
1503 (ref,) = registry.insertDatasets(rawType, dataIds=[dataId], run=run2)
1504 registry.associate(tagged2, [ref])
1506 dimensions = registry.dimensions.conform(rawType.dimensions.required | calexpType.dimensions.required)
1507 # Test that single dim string works as well as list of str
1508 rows = registry.queryDataIds("visit", datasets=rawType, collections=run1).expanded().toSet()
1509 rowsI = registry.queryDataIds(["visit"], datasets=rawType, collections=run1).expanded().toSet()
1510 self.assertEqual(rows, rowsI)
1511 # with empty expression
1512 rows = registry.queryDataIds(dimensions, datasets=rawType, collections=run1).expanded().toSet()
1513 self.assertEqual(len(rows), 4 * 3) # 4 exposures times 3 detectors
1514 self.assertCountEqual({dataId["exposure"] for dataId in rows}, (100, 101, 110, 111))
1515 self.assertCountEqual({dataId["visit"] for dataId in rows}, (10, 11))
1516 self.assertCountEqual({dataId["detector"] for dataId in rows}, (1, 2, 3))
1518 # second collection
1519 rows = registry.queryDataIds(dimensions, datasets=rawType, collections=tagged2).toSet()
1520 self.assertEqual(len(rows), 4 * 3) # 4 exposures times 3 detectors
1521 for dataId in rows:
1522 self.assertCountEqual(dataId.dimensions.required, ("instrument", "detector", "exposure", "visit"))
1523 self.assertCountEqual({dataId["exposure"] for dataId in rows}, (100, 101, 200, 201))
1524 self.assertCountEqual({dataId["visit"] for dataId in rows}, (10, 20))
1525 self.assertCountEqual({dataId["detector"] for dataId in rows}, (1, 2, 3, 4, 5))
1527 # with two input datasets
1528 rows = registry.queryDataIds(dimensions, datasets=rawType, collections=[run1, tagged2]).toSet()
1529 self.assertEqual(len(set(rows)), 6 * 3) # 6 exposures times 3 detectors; set needed to de-dupe
1530 for dataId in rows:
1531 self.assertCountEqual(dataId.dimensions.required, ("instrument", "detector", "exposure", "visit"))
1532 self.assertCountEqual({dataId["exposure"] for dataId in rows}, (100, 101, 110, 111, 200, 201))
1533 self.assertCountEqual({dataId["visit"] for dataId in rows}, (10, 11, 20))
1534 self.assertCountEqual({dataId["detector"] for dataId in rows}, (1, 2, 3, 4, 5))
1536 # limit to single visit
1537 rows = registry.queryDataIds(
1538 dimensions, datasets=rawType, collections=run1, where="visit = 10", instrument="DummyCam"
1539 ).toSet()
1540 self.assertEqual(len(rows), 2 * 3) # 2 exposures times 3 detectors
1541 self.assertCountEqual({dataId["exposure"] for dataId in rows}, (100, 101))
1542 self.assertCountEqual({dataId["visit"] for dataId in rows}, (10,))
1543 self.assertCountEqual({dataId["detector"] for dataId in rows}, (1, 2, 3))
1545 # more limiting expression, using link names instead of Table.column
1546 rows = registry.queryDataIds(
1547 dimensions,
1548 datasets=rawType,
1549 collections=run1,
1550 where="visit = 10 and detector > 1 and 'DummyCam'=instrument",
1551 ).toSet()
1552 self.assertEqual(len(rows), 2 * 2) # 2 exposures times 2 detectors
1553 self.assertCountEqual({dataId["exposure"] for dataId in rows}, (100, 101))
1554 self.assertCountEqual({dataId["visit"] for dataId in rows}, (10,))
1555 self.assertCountEqual({dataId["detector"] for dataId in rows}, (2, 3))
1557 # queryDataIds with only one of `datasets` and `collections` is an
1558 # error.
1559 with self.assertRaises(CollectionError):
1560 registry.queryDataIds(dimensions, datasets=rawType)
1561 with self.assertRaises(ArgumentError):
1562 registry.queryDataIds(dimensions, collections=run1)
1564 # expression excludes everything
1565 rows = registry.queryDataIds(
1566 dimensions, datasets=rawType, collections=run1, where="visit > 1000", instrument="DummyCam"
1567 ).toSet()
1568 self.assertEqual(len(rows), 0)
1570 # Selecting by physical_filter, this is not in the dimensions, but it
1571 # is a part of the full expression so it should work too.
1572 rows = registry.queryDataIds(
1573 dimensions,
1574 datasets=rawType,
1575 collections=run1,
1576 where="physical_filter = 'dummy_r'",
1577 instrument="DummyCam",
1578 ).toSet()
1579 self.assertEqual(len(rows), 2 * 3) # 2 exposures times 3 detectors
1580 self.assertCountEqual({dataId["exposure"] for dataId in rows}, (110, 111))
1581 self.assertCountEqual({dataId["visit"] for dataId in rows}, (11,))
1582 self.assertCountEqual({dataId["detector"] for dataId in rows}, (1, 2, 3))
1584 def testSkyMapDimensions(self):
1585 """Tests involving only skymap dimensions, no joins to instrument."""
1586 butler = self.make_butler()
1587 registry = butler.registry
1589 # need a bunch of dimensions and datasets for test, we want
1590 # "band" in the test so also have to add physical_filter
1591 # dimensions
1592 registry.insertDimensionData("instrument", dict(instrument="DummyCam"))
1593 registry.insertDimensionData(
1594 "physical_filter",
1595 dict(instrument="DummyCam", name="dummy_r", band="r"),
1596 dict(instrument="DummyCam", name="dummy_i", band="i"),
1597 )
1598 registry.insertDimensionData("skymap", dict(name="DummyMap", hash=b"sha!"))
1599 for tract in range(10):
1600 registry.insertDimensionData("tract", dict(skymap="DummyMap", id=tract))
1601 registry.insertDimensionData(
1602 "patch",
1603 *[dict(skymap="DummyMap", tract=tract, id=patch, cell_x=0, cell_y=0) for patch in range(10)],
1604 )
1606 # dataset types
1607 run = "tésτ"
1608 registry.registerRun(run)
1609 storageClass = StorageClass("testDataset")
1610 registry.storageClasses.registerStorageClass(storageClass)
1611 calexpType = DatasetType(
1612 name="deepCoadd_calexp",
1613 dimensions=registry.dimensions.conform(("skymap", "tract", "patch", "band")),
1614 storageClass=storageClass,
1615 )
1616 registry.registerDatasetType(calexpType)
1617 mergeType = DatasetType(
1618 name="deepCoadd_mergeDet",
1619 dimensions=registry.dimensions.conform(("skymap", "tract", "patch")),
1620 storageClass=storageClass,
1621 )
1622 registry.registerDatasetType(mergeType)
1623 measType = DatasetType(
1624 name="deepCoadd_meas",
1625 dimensions=registry.dimensions.conform(("skymap", "tract", "patch", "band")),
1626 storageClass=storageClass,
1627 )
1628 registry.registerDatasetType(measType)
1630 dimensions = registry.dimensions.conform(
1631 calexpType.dimensions.required | mergeType.dimensions.required | measType.dimensions.required
1632 )
1634 # add pre-existing datasets
1635 for tract in (1, 3, 5):
1636 for patch in (2, 4, 6, 7):
1637 dataId = dict(skymap="DummyMap", tract=tract, patch=patch)
1638 registry.insertDatasets(mergeType, dataIds=[dataId], run=run)
1639 for aFilter in ("i", "r"):
1640 dataId = dict(skymap="DummyMap", tract=tract, patch=patch, band=aFilter)
1641 registry.insertDatasets(calexpType, dataIds=[dataId], run=run)
1643 # with empty expression
1644 rows = registry.queryDataIds(dimensions, datasets=[calexpType, mergeType], collections=run).toSet()
1645 self.assertEqual(len(rows), 3 * 4 * 2) # 4 tracts x 4 patches x 2 filters
1646 for dataId in rows:
1647 self.assertCountEqual(dataId.dimensions.required, ("skymap", "tract", "patch", "band"))
1648 self.assertCountEqual({dataId["tract"] for dataId in rows}, (1, 3, 5))
1649 self.assertCountEqual({dataId["patch"] for dataId in rows}, (2, 4, 6, 7))
1650 self.assertCountEqual({dataId["band"] for dataId in rows}, ("i", "r"))
1652 # limit to 2 tracts and 2 patches
1653 rows = registry.queryDataIds(
1654 dimensions,
1655 datasets=[calexpType, mergeType],
1656 collections=run,
1657 where="tract IN (1, 5) AND patch IN (2, 7)",
1658 skymap="DummyMap",
1659 ).toSet()
1660 self.assertEqual(len(rows), 2 * 2 * 2) # 2 tracts x 2 patches x 2 filters
1661 self.assertCountEqual({dataId["tract"] for dataId in rows}, (1, 5))
1662 self.assertCountEqual({dataId["patch"] for dataId in rows}, (2, 7))
1663 self.assertCountEqual({dataId["band"] for dataId in rows}, ("i", "r"))
1665 # limit to single filter
1666 rows = registry.queryDataIds(
1667 dimensions, datasets=[calexpType, mergeType], collections=run, where="band = 'i'"
1668 ).toSet()
1669 self.assertEqual(len(rows), 3 * 4 * 1) # 4 tracts x 4 patches x 2 filters
1670 self.assertCountEqual({dataId["tract"] for dataId in rows}, (1, 3, 5))
1671 self.assertCountEqual({dataId["patch"] for dataId in rows}, (2, 4, 6, 7))
1672 self.assertCountEqual({dataId["band"] for dataId in rows}, ("i",))
1674 def do_query():
1675 return registry.queryDataIds(
1676 dimensions, datasets=[calexpType, mergeType], collections=run, where="skymap = 'Mars'"
1677 ).toSet()
1679 self.assertEqual(len(do_query()), 0)
1681 def testSpatialJoin(self):
1682 """Test queries that involve spatial overlap joins."""
1683 butler = self.make_butler()
1684 registry = butler.registry
1685 self.load_data(butler, "base.yaml", "spatial.yaml")
1687 # Dictionary of spatial DatabaseDimensionElements, keyed by the name of
1688 # the TopologicalFamily they belong to. We'll relate all elements in
1689 # each family to all of the elements in each other family.
1690 families = defaultdict(set)
1691 # Dictionary of {element.name: {dataId: region}}.
1692 regions = {}
1693 for element in registry.dimensions.database_elements:
1694 if element.spatial is not None:
1695 families[element.spatial.name].add(element)
1696 regions[element.name] = {
1697 record.dataId: record.region for record in registry.queryDimensionRecords(element)
1698 }
1700 # If this check fails, it's not necessarily a problem - it may just be
1701 # a reasonable change to the default dimension definitions - but the
1702 # test below depends on there being more than one family to do anything
1703 # useful.
1704 self.assertEqual(len(families), 2)
1706 # Overlap DatabaseDimensionElements with each other.
1707 for family1, family2 in itertools.combinations(families, 2):
1708 for element1, element2 in itertools.product(families[family1], families[family2]):
1709 dimensions = element1.minimal_group | element2.minimal_group
1710 # Construct expected set of overlapping data IDs via a
1711 # brute-force comparison of the regions we've already fetched.
1712 expected = {
1713 DataCoordinate.standardize(
1714 {**dataId1.required, **dataId2.required}, dimensions=dimensions
1715 )
1716 for (dataId1, region1), (dataId2, region2) in itertools.product(
1717 regions[element1.name].items(), regions[element2.name].items()
1718 )
1719 if not region1.isDisjointFrom(region2)
1720 }
1721 self.assertGreater(len(expected), 2, msg="Test that we aren't just comparing empty sets.")
1722 queried = set(registry.queryDataIds(dimensions))
1723 self.assertEqual(expected, queried)
1725 # Overlap each DatabaseDimensionElement with the commonSkyPix system.
1726 commonSkyPix = registry.dimensions.commonSkyPix
1727 for elementName, these_regions in regions.items():
1728 dimensions = registry.dimensions[elementName].minimal_group | commonSkyPix.minimal_group
1729 expected = set()
1730 for dataId, region in these_regions.items():
1731 for begin, end in commonSkyPix.pixelization.envelope(region):
1732 expected.update(
1733 DataCoordinate.standardize(
1734 {commonSkyPix.name: index, **dataId.required}, dimensions=dimensions
1735 )
1736 for index in range(begin, end)
1737 )
1738 self.assertGreater(len(expected), 2, msg="Test that we aren't just comparing empty sets.")
1739 queried = set(registry.queryDataIds(dimensions))
1740 self.assertEqual(expected, queried)
1742 def testAbstractQuery(self):
1743 """Test that we can run a query that just lists the known
1744 bands. This is tricky because band is
1745 backed by a query against physical_filter.
1746 """
1747 butler = self.make_butler()
1748 registry = butler.registry
1749 registry.insertDimensionData("instrument", dict(name="DummyCam"))
1750 registry.insertDimensionData(
1751 "physical_filter",
1752 dict(instrument="DummyCam", name="dummy_i", band="i"),
1753 dict(instrument="DummyCam", name="dummy_i2", band="i"),
1754 dict(instrument="DummyCam", name="dummy_r", band="r"),
1755 )
1756 rows = registry.queryDataIds(["band"]).toSet()
1757 self.assertCountEqual(
1758 rows,
1759 [
1760 DataCoordinate.standardize(band="i", universe=registry.dimensions),
1761 DataCoordinate.standardize(band="r", universe=registry.dimensions),
1762 ],
1763 )
1765 def testAttributeManager(self):
1766 """Test basic functionality of attribute manager."""
1767 # number of attributes with schema versions in a fresh database,
1768 # 6 managers with 2 records per manager, plus config for dimensions
1769 VERSION_COUNT = 6 * 2 + 1
1771 butler = self.make_butler()
1772 registry = butler._registry
1773 attributes = registry._managers.attributes
1775 # check what get() returns for non-existing key
1776 self.assertIsNone(attributes.get("attr"))
1777 self.assertEqual(attributes.get("attr", ""), "")
1778 self.assertEqual(attributes.get("attr", "Value"), "Value")
1779 self.assertEqual(len(list(attributes.items())), VERSION_COUNT)
1781 # cannot store empty key or value
1782 with self.assertRaises(ValueError):
1783 attributes.set("", "value")
1784 with self.assertRaises(ValueError):
1785 attributes.set("attr", "")
1787 # set value of non-existing key
1788 attributes.set("attr", "value")
1789 self.assertEqual(len(list(attributes.items())), VERSION_COUNT + 1)
1790 self.assertEqual(attributes.get("attr"), "value")
1792 # update value of existing key
1793 with self.assertRaises(ButlerAttributeExistsError):
1794 attributes.set("attr", "value2")
1796 attributes.set("attr", "value2", force=True)
1797 self.assertEqual(len(list(attributes.items())), VERSION_COUNT + 1)
1798 self.assertEqual(attributes.get("attr"), "value2")
1800 # delete existing key
1801 self.assertTrue(attributes.delete("attr"))
1802 self.assertEqual(len(list(attributes.items())), VERSION_COUNT)
1804 # delete non-existing key
1805 self.assertFalse(attributes.delete("non-attr"))
1807 # store bunch of keys and get the list back
1808 data = [
1809 ("version.core", "1.2.3"),
1810 ("version.dimensions", "3.2.1"),
1811 ("config.managers.opaque", "ByNameOpaqueTableStorageManager"),
1812 ]
1813 for key, value in data:
1814 attributes.set(key, value)
1815 items = dict(attributes.items())
1816 for key, value in data:
1817 self.assertEqual(items[key], value)
1819 def testQueryDatasetsDeduplication(self):
1820 """Test that the findFirst option to queryDatasets selects datasets
1821 from collections in the order given".
1822 """
1823 butler = self.make_butler()
1824 registry = butler.registry
1825 self.load_data(butler, "base.yaml", "datasets.yaml")
1826 self.assertCountEqual(
1827 list(registry.queryDatasets("bias", collections=["imported_g", "imported_r"])),
1828 [
1829 registry.findDataset("bias", instrument="Cam1", detector=1, collections="imported_g"),
1830 registry.findDataset("bias", instrument="Cam1", detector=2, collections="imported_g"),
1831 registry.findDataset("bias", instrument="Cam1", detector=3, collections="imported_g"),
1832 registry.findDataset("bias", instrument="Cam1", detector=2, collections="imported_r"),
1833 registry.findDataset("bias", instrument="Cam1", detector=3, collections="imported_r"),
1834 registry.findDataset("bias", instrument="Cam1", detector=4, collections="imported_r"),
1835 ],
1836 )
1837 self.assertCountEqual(
1838 list(registry.queryDatasets("bias", collections=["imported_g", "imported_r"], findFirst=True)),
1839 [
1840 registry.findDataset("bias", instrument="Cam1", detector=1, collections="imported_g"),
1841 registry.findDataset("bias", instrument="Cam1", detector=2, collections="imported_g"),
1842 registry.findDataset("bias", instrument="Cam1", detector=3, collections="imported_g"),
1843 registry.findDataset("bias", instrument="Cam1", detector=4, collections="imported_r"),
1844 ],
1845 )
1846 self.assertCountEqual(
1847 list(registry.queryDatasets("bias", collections=["imported_r", "imported_g"], findFirst=True)),
1848 [
1849 registry.findDataset("bias", instrument="Cam1", detector=1, collections="imported_g"),
1850 registry.findDataset("bias", instrument="Cam1", detector=2, collections="imported_r"),
1851 registry.findDataset("bias", instrument="Cam1", detector=3, collections="imported_r"),
1852 registry.findDataset("bias", instrument="Cam1", detector=4, collections="imported_r"),
1853 ],
1854 )
1856 with self.assertRaises(TypeError):
1857 # Collection wildcards not allowed in find-first searches because
1858 # they do not guarantee the ordering of collections.
1859 registry.queryDatasets("bias", collections="imported_*", findFirst=True)
1861 def testQueryDatasetsExtraDimensions(self):
1862 butler = self.make_butler()
1863 registry = butler.registry
1864 self.load_data(butler, "base.yaml", "datasets.yaml")
1865 # Bias dataset type does not include physical filter. By adding
1866 # "physical_filter" to dimensions, we are effectively searching here
1867 # for bias datasets with an instrument that has a specific filter
1868 # available, even though that filter has nothing to do with the bias
1869 # datasets we are finding.
1870 self.assertEqual(
1871 0,
1872 registry.queryDatasets(
1873 "bias",
1874 collections=...,
1875 dimensions=["physical_filter"],
1876 dataId={
1877 "instrument": "Cam1",
1878 "band": "not_a_real_band",
1879 },
1880 ).count(),
1881 )
1882 self.assertEqual(
1883 6,
1884 len(
1885 set(
1886 registry.queryDatasets(
1887 "bias",
1888 collections=...,
1889 dimensions=["physical_filter"],
1890 dataId={
1891 "instrument": "Cam1",
1892 "band": "r",
1893 },
1894 )
1895 )
1896 ),
1897 )
1899 def testQueryResults(self):
1900 """Test querying for data IDs and then manipulating the QueryResults
1901 object returned to perform other queries.
1902 """
1903 butler = self.make_butler()
1904 registry = butler.registry
1905 self.load_data(butler, "base.yaml", "datasets.yaml")
1906 bias = registry.getDatasetType("bias")
1907 flat = registry.getDatasetType("flat")
1908 # Obtain expected results from methods other than those we're testing
1909 # here. That includes:
1910 # - the dimensions of the data IDs we want to query:
1911 expected_dimensions = registry.dimensions.conform(["detector", "physical_filter"])
1912 # - the dimensions of some other data IDs we'll extract from that:
1913 expected_subset_dimensions = registry.dimensions.conform(["detector"])
1914 # - the data IDs we expect to obtain from the first queries:
1915 expectedDataIds = DataCoordinateSet(
1916 {
1917 DataCoordinate.standardize(
1918 instrument="Cam1", detector=d, physical_filter=p, universe=registry.dimensions
1919 )
1920 for d, p in itertools.product({1, 2, 3}, {"Cam1-G", "Cam1-R1", "Cam1-R2"})
1921 },
1922 dimensions=expected_dimensions,
1923 hasFull=False,
1924 hasRecords=False,
1925 )
1926 # - the flat datasets we expect to find from those data IDs, in just
1927 # one collection (so deduplication is irrelevant):
1928 expectedFlats = [
1929 registry.findDataset(
1930 flat, instrument="Cam1", detector=1, physical_filter="Cam1-R1", collections="imported_r"
1931 ),
1932 registry.findDataset(
1933 flat, instrument="Cam1", detector=2, physical_filter="Cam1-R1", collections="imported_r"
1934 ),
1935 registry.findDataset(
1936 flat, instrument="Cam1", detector=3, physical_filter="Cam1-R2", collections="imported_r"
1937 ),
1938 ]
1939 # - the data IDs we expect to extract from that:
1940 expectedSubsetDataIds = expectedDataIds.subset(expected_subset_dimensions)
1941 # - the bias datasets we expect to find from those data IDs, after we
1942 # subset-out the physical_filter dimension, both with duplicates:
1943 expectedAllBiases = [
1944 registry.findDataset(bias, instrument="Cam1", detector=1, collections="imported_g"),
1945 registry.findDataset(bias, instrument="Cam1", detector=2, collections="imported_g"),
1946 registry.findDataset(bias, instrument="Cam1", detector=3, collections="imported_g"),
1947 registry.findDataset(bias, instrument="Cam1", detector=2, collections="imported_r"),
1948 registry.findDataset(bias, instrument="Cam1", detector=3, collections="imported_r"),
1949 ]
1950 # - ...and without duplicates:
1951 expectedDeduplicatedBiases = [
1952 registry.findDataset(bias, instrument="Cam1", detector=1, collections="imported_g"),
1953 registry.findDataset(bias, instrument="Cam1", detector=2, collections="imported_r"),
1954 registry.findDataset(bias, instrument="Cam1", detector=3, collections="imported_r"),
1955 ]
1956 # Test against those expected results, using a "lazy" query for the
1957 # data IDs (which re-executes that query each time we use it to do
1958 # something new).
1959 dataIds = registry.queryDataIds(
1960 ["detector", "physical_filter"],
1961 where="detector.purpose = 'SCIENCE'", # this rejects detector=4
1962 instrument="Cam1",
1963 )
1964 self.assertEqual(dataIds.dimensions, expected_dimensions)
1965 self.assertEqual(dataIds.toSet(), expectedDataIds)
1966 self.assertCountEqual(
1967 list(
1968 dataIds.findDatasets(
1969 flat,
1970 collections=["imported_r"],
1971 )
1972 ),
1973 expectedFlats,
1974 )
1975 subsetDataIds = dataIds.subset(expected_subset_dimensions, unique=True)
1976 self.assertEqual(subsetDataIds.dimensions, expected_subset_dimensions)
1977 self.assertEqual(subsetDataIds.toSet(), expectedSubsetDataIds)
1978 self.assertCountEqual(
1979 list(subsetDataIds.findDatasets(bias, collections=["imported_r", "imported_g"], findFirst=False)),
1980 expectedAllBiases,
1981 )
1982 self.assertCountEqual(
1983 list(subsetDataIds.findDatasets(bias, collections=["imported_r", "imported_g"], findFirst=True)),
1984 expectedDeduplicatedBiases,
1985 )
1987 # Searching for a dataset with dimensions we had projected away
1988 # restores those dimensions.
1989 self.assertCountEqual(
1990 list(subsetDataIds.findDatasets("flat", collections=["imported_r"], findFirst=True)),
1991 expectedFlats,
1992 )
1994 # Use a named dataset type that does not exist and a dataset type
1995 # object that does not exist.
1996 unknown_type = DatasetType("not_known", dimensions=bias.dimensions, storageClass="Exposure")
1998 # Test both string name and dataset type object.
1999 test_type: str | DatasetType
2000 for test_type, test_type_name in (
2001 (unknown_type, unknown_type.name),
2002 (unknown_type.name, unknown_type.name),
2003 ):
2004 with self.assertRaisesRegex(DatasetTypeError, expected_regex=test_type_name):
2005 list(
2006 subsetDataIds.findDatasets(
2007 test_type, collections=["imported_r", "imported_g"], findFirst=True
2008 )
2009 )
2011 # Materialize the data ID subset query, but not the dataset queries.
2012 with subsetDataIds.materialize() as subsetDataIds:
2013 self.assertEqual(subsetDataIds.dimensions, expected_subset_dimensions)
2014 self.assertEqual(subsetDataIds.toSet(), expectedSubsetDataIds)
2015 self.assertCountEqual(
2016 list(
2017 subsetDataIds.findDatasets(
2018 bias, collections=["imported_r", "imported_g"], findFirst=False
2019 )
2020 ),
2021 expectedAllBiases,
2022 )
2023 self.assertCountEqual(
2024 list(
2025 subsetDataIds.findDatasets(bias, collections=["imported_r", "imported_g"], findFirst=True)
2026 ),
2027 expectedDeduplicatedBiases,
2028 )
2029 # Materialize the original query, but none of the follow-up queries.
2030 with dataIds.materialize() as dataIds:
2031 self.assertEqual(dataIds.dimensions, expected_dimensions)
2032 self.assertEqual(dataIds.toSet(), expectedDataIds)
2033 self.assertCountEqual(
2034 list(
2035 dataIds.findDatasets(
2036 flat,
2037 collections=["imported_r"],
2038 )
2039 ),
2040 expectedFlats,
2041 )
2042 subsetDataIds = dataIds.subset(expected_subset_dimensions, unique=True)
2043 self.assertEqual(subsetDataIds.dimensions, expected_subset_dimensions)
2044 self.assertEqual(subsetDataIds.toSet(), expectedSubsetDataIds)
2045 self.assertCountEqual(
2046 list(
2047 subsetDataIds.findDatasets(
2048 bias, collections=["imported_r", "imported_g"], findFirst=False
2049 )
2050 ),
2051 expectedAllBiases,
2052 )
2053 self.assertCountEqual(
2054 list(
2055 subsetDataIds.findDatasets(bias, collections=["imported_r", "imported_g"], findFirst=True)
2056 ),
2057 expectedDeduplicatedBiases,
2058 )
2059 # Materialize the subset data ID query, but not the dataset
2060 # queries.
2061 with subsetDataIds.materialize() as subsetDataIds:
2062 self.assertEqual(subsetDataIds.dimensions, expected_subset_dimensions)
2063 self.assertEqual(subsetDataIds.toSet(), expectedSubsetDataIds)
2064 self.assertCountEqual(
2065 list(
2066 subsetDataIds.findDatasets(
2067 bias, collections=["imported_r", "imported_g"], findFirst=False
2068 )
2069 ),
2070 expectedAllBiases,
2071 )
2072 self.assertCountEqual(
2073 list(
2074 subsetDataIds.findDatasets(
2075 bias, collections=["imported_r", "imported_g"], findFirst=True
2076 )
2077 ),
2078 expectedDeduplicatedBiases,
2079 )
2081 def testStorageClassPropagation(self):
2082 """Test that queries for datasets respect the storage class passed in
2083 as part of a full dataset type.
2084 """
2085 butler = self.make_butler()
2086 registry = butler.registry
2087 self.load_data(butler, "base.yaml")
2088 dataset_type_in_registry = DatasetType(
2089 "tbl", dimensions=["instrument"], storageClass="Packages", universe=registry.dimensions
2090 )
2091 registry.registerDatasetType(dataset_type_in_registry)
2092 run = "run1"
2093 registry.registerRun(run)
2094 (inserted_ref,) = registry.insertDatasets(
2095 dataset_type_in_registry, [registry.expandDataId(instrument="Cam1")], run=run
2096 )
2097 self.assertEqual(inserted_ref.datasetType, dataset_type_in_registry)
2098 query_dataset_type = DatasetType(
2099 "tbl", dimensions=["instrument"], storageClass="StructuredDataDict", universe=registry.dimensions
2100 )
2101 self.assertNotEqual(dataset_type_in_registry, query_dataset_type)
2102 query_datasets_result = registry.queryDatasets(query_dataset_type, collections=[run])
2103 self.assertEqual(query_datasets_result.parentDatasetType, query_dataset_type) # type: ignore
2104 (query_datasets_ref,) = query_datasets_result
2105 self.assertEqual(query_datasets_ref.datasetType, query_dataset_type)
2106 query_data_ids_find_datasets_result = registry.queryDataIds(["instrument"]).findDatasets(
2107 query_dataset_type, collections=[run]
2108 )
2109 self.assertEqual(query_data_ids_find_datasets_result.parentDatasetType, query_dataset_type)
2110 (query_data_ids_find_datasets_ref,) = query_data_ids_find_datasets_result
2111 self.assertEqual(query_data_ids_find_datasets_ref.datasetType, query_dataset_type)
2112 query_dataset_types_result = registry.queryDatasetTypes(query_dataset_type)
2113 self.assertEqual(list(query_dataset_types_result), [query_dataset_type])
2114 find_dataset_ref = registry.findDataset(query_dataset_type, instrument="Cam1", collections=[run])
2115 self.assertEqual(find_dataset_ref.datasetType, query_dataset_type)
2117 def testEmptyDimensionsQueries(self):
2118 """Test Query and QueryResults objects in the case where there are no
2119 dimensions.
2120 """
2121 # Set up test data: one dataset type, two runs, one dataset in each.
2122 butler = self.make_butler()
2123 registry = butler.registry
2124 self.load_data(butler, "base.yaml")
2125 schema = DatasetType("schema", dimensions=registry.dimensions.empty, storageClass="Catalog")
2126 registry.registerDatasetType(schema)
2127 dataId = DataCoordinate.make_empty(registry.dimensions)
2128 run1 = "run1"
2129 run2 = "run2"
2130 registry.registerRun(run1)
2131 registry.registerRun(run2)
2132 (dataset1,) = registry.insertDatasets(schema, dataIds=[dataId], run=run1)
2133 (dataset2,) = registry.insertDatasets(schema, dataIds=[dataId], run=run2)
2134 # Query directly for both of the datasets, and each one, one at a time.
2135 self.checkQueryResults(
2136 registry.queryDatasets(schema, collections=[run1, run2], findFirst=False), [dataset1, dataset2]
2137 )
2138 self.checkQueryResults(
2139 registry.queryDatasets(schema, collections=[run1, run2], findFirst=True),
2140 [dataset1],
2141 )
2142 self.checkQueryResults(
2143 registry.queryDatasets(schema, collections=[run2, run1], findFirst=True),
2144 [dataset2],
2145 )
2146 # Query for data IDs with no dimensions.
2147 dataIds = registry.queryDataIds([])
2148 self.checkQueryResults(dataIds, [dataId])
2149 # Use queried data IDs to find the datasets.
2150 self.checkQueryResults(
2151 dataIds.findDatasets(schema, collections=[run1, run2], findFirst=False),
2152 [dataset1, dataset2],
2153 )
2154 self.checkQueryResults(
2155 dataIds.findDatasets(schema, collections=[run1, run2], findFirst=True),
2156 [dataset1],
2157 )
2158 self.checkQueryResults(
2159 dataIds.findDatasets(schema, collections=[run2, run1], findFirst=True),
2160 [dataset2],
2161 )
2162 # Now materialize the data ID query results and repeat those tests.
2163 with dataIds.materialize() as dataIds:
2164 self.checkQueryResults(dataIds, [dataId])
2165 self.checkQueryResults(
2166 dataIds.findDatasets(schema, collections=[run1, run2], findFirst=True),
2167 [dataset1],
2168 )
2169 self.checkQueryResults(
2170 dataIds.findDatasets(schema, collections=[run2, run1], findFirst=True),
2171 [dataset2],
2172 )
2173 # Query for non-empty data IDs, then subset that to get the empty one.
2174 # Repeat the above tests starting from that.
2175 dataIds = registry.queryDataIds(["instrument"]).subset(registry.dimensions.empty, unique=True)
2176 self.checkQueryResults(dataIds, [dataId])
2177 self.checkQueryResults(
2178 dataIds.findDatasets(schema, collections=[run1, run2], findFirst=False),
2179 [dataset1, dataset2],
2180 )
2181 self.checkQueryResults(
2182 dataIds.findDatasets(schema, collections=[run1, run2], findFirst=True),
2183 [dataset1],
2184 )
2185 self.checkQueryResults(
2186 dataIds.findDatasets(schema, collections=[run2, run1], findFirst=True),
2187 [dataset2],
2188 )
2189 with dataIds.materialize() as dataIds:
2190 self.checkQueryResults(dataIds, [dataId])
2191 self.checkQueryResults(
2192 dataIds.findDatasets(schema, collections=[run1, run2], findFirst=False),
2193 [dataset1, dataset2],
2194 )
2195 self.checkQueryResults(
2196 dataIds.findDatasets(schema, collections=[run1, run2], findFirst=True),
2197 [dataset1],
2198 )
2199 self.checkQueryResults(
2200 dataIds.findDatasets(schema, collections=[run2, run1], findFirst=True),
2201 [dataset2],
2202 )
2203 # Query for non-empty data IDs, then materialize, then subset to get
2204 # the empty one. Repeat again.
2205 with registry.queryDataIds(["instrument"]).materialize() as nonEmptyDataIds:
2206 dataIds = nonEmptyDataIds.subset(registry.dimensions.empty, unique=True)
2207 self.checkQueryResults(dataIds, [dataId])
2208 self.checkQueryResults(
2209 dataIds.findDatasets(schema, collections=[run1, run2], findFirst=False),
2210 [dataset1, dataset2],
2211 )
2212 self.checkQueryResults(
2213 dataIds.findDatasets(schema, collections=[run1, run2], findFirst=True),
2214 [dataset1],
2215 )
2216 self.checkQueryResults(
2217 dataIds.findDatasets(schema, collections=[run2, run1], findFirst=True),
2218 [dataset2],
2219 )
2220 with dataIds.materialize() as dataIds:
2221 self.checkQueryResults(dataIds, [dataId])
2222 self.checkQueryResults(
2223 dataIds.findDatasets(schema, collections=[run1, run2], findFirst=False),
2224 [dataset1, dataset2],
2225 )
2226 self.checkQueryResults(
2227 dataIds.findDatasets(schema, collections=[run1, run2], findFirst=True),
2228 [dataset1],
2229 )
2230 self.checkQueryResults(
2231 dataIds.findDatasets(schema, collections=[run2, run1], findFirst=True),
2232 [dataset2],
2233 )
2234 # Query for non-empty data IDs with a constraint on an empty-data-ID
2235 # dataset that exists.
2236 dataIds = registry.queryDataIds(["instrument"], datasets="schema", collections=...)
2237 self.checkQueryResults(
2238 dataIds.subset(unique=True),
2239 [DataCoordinate.standardize(instrument="Cam1", universe=registry.dimensions)],
2240 )
2241 # Again query for non-empty data IDs with a constraint on empty-data-ID
2242 # datasets, but when the datasets don't exist. We delete the existing
2243 # dataset and query just that collection rather than creating a new
2244 # empty collection because this is a bit less likely for our build-time
2245 # logic to shortcut-out (via the collection summaries), and such a
2246 # shortcut would make this test a bit more trivial than we'd like.
2247 registry.removeDatasets([dataset2])
2248 dataIds = registry.queryDataIds(["instrument"], datasets="schema", collections=run2)
2249 self.checkQueryResults(dataIds, [])
2251 def testDimensionDataModifications(self):
2252 """Test that modifying dimension records via:
2253 syncDimensionData(..., update=True) and
2254 insertDimensionData(..., replace=True) works as expected, even in the
2255 presence of datasets using those dimensions and spatial overlap
2256 relationships.
2257 """
2259 def _unpack_range_set(ranges: lsst.sphgeom.RangeSet) -> Iterator[int]:
2260 """Unpack a sphgeom.RangeSet into the integers it contains."""
2261 for begin, end in ranges:
2262 yield from range(begin, end)
2264 def _range_set_hull(
2265 ranges: lsst.sphgeom.RangeSet,
2266 pixelization: lsst.sphgeom.HtmPixelization,
2267 ) -> lsst.sphgeom.ConvexPolygon:
2268 """Create a ConvexPolygon hull of the region defined by a set of
2269 HTM pixelization index ranges.
2270 """
2271 points = []
2272 for index in _unpack_range_set(ranges):
2273 points.extend(pixelization.triangle(index).getVertices())
2274 return lsst.sphgeom.ConvexPolygon(points)
2276 # Use HTM to set up an initial parent region (one arbitrary trixel)
2277 # and four child regions (the trixels within the parent at the next
2278 # level. We'll use the parent as a tract/visit region and the children
2279 # as its patch/visit_detector regions.
2280 butler = self.make_butler()
2281 registry = butler.registry
2282 htm6 = registry.dimensions.skypix["htm"][6].pixelization
2283 commonSkyPix = registry.dimensions.commonSkyPix.pixelization
2284 index = 12288
2285 child_ranges_small = lsst.sphgeom.RangeSet(index).scaled(4)
2286 assert htm6.universe().contains(child_ranges_small)
2287 child_regions_small = [htm6.triangle(i) for i in _unpack_range_set(child_ranges_small)]
2288 parent_region_small = lsst.sphgeom.ConvexPolygon(
2289 list(itertools.chain.from_iterable(c.getVertices() for c in child_regions_small))
2290 )
2291 assert all(parent_region_small.contains(c) for c in child_regions_small)
2292 # Make a larger version of each child region, defined to be the set of
2293 # htm6 trixels that overlap the original's bounding circle. Make a new
2294 # parent that's the convex hull of the new children.
2295 child_regions_large = [
2296 _range_set_hull(htm6.envelope(c.getBoundingCircle()), htm6) for c in child_regions_small
2297 ]
2298 assert all(
2299 large.contains(small)
2300 for large, small in zip(child_regions_large, child_regions_small, strict=True)
2301 )
2302 parent_region_large = lsst.sphgeom.ConvexPolygon(
2303 list(itertools.chain.from_iterable(c.getVertices() for c in child_regions_large))
2304 )
2305 assert all(parent_region_large.contains(c) for c in child_regions_large)
2306 assert parent_region_large.contains(parent_region_small)
2307 assert not parent_region_small.contains(parent_region_large)
2308 assert not all(parent_region_small.contains(c) for c in child_regions_large)
2309 # Find some commonSkyPix indices that overlap the large regions but not
2310 # overlap the small regions. We use commonSkyPix here to make sure the
2311 # real tests later involve what's in the database, not just post-query
2312 # filtering of regions.
2313 child_difference_indices = []
2314 for large, small in zip(child_regions_large, child_regions_small, strict=True):
2315 difference = list(_unpack_range_set(commonSkyPix.envelope(large) - commonSkyPix.envelope(small)))
2316 assert difference, "if this is empty, we can't test anything useful with these regions"
2317 assert all(
2318 not commonSkyPix.triangle(d).isDisjointFrom(large)
2319 and commonSkyPix.triangle(d).isDisjointFrom(small)
2320 for d in difference
2321 )
2322 child_difference_indices.append(difference)
2323 parent_difference_indices = list(
2324 _unpack_range_set(
2325 commonSkyPix.envelope(parent_region_large) - commonSkyPix.envelope(parent_region_small)
2326 )
2327 )
2328 assert parent_difference_indices, "if this is empty, we can't test anything useful with these regions"
2329 assert all(
2330 (
2331 not commonSkyPix.triangle(d).isDisjointFrom(parent_region_large)
2332 and commonSkyPix.triangle(d).isDisjointFrom(parent_region_small)
2333 )
2334 for d in parent_difference_indices
2335 )
2336 # Now that we've finally got those regions, we'll insert the large ones
2337 # as tract/patch dimension records.
2338 skymap_name = "testing_v1"
2339 registry.insertDimensionData(
2340 "skymap",
2341 {
2342 "name": skymap_name,
2343 "hash": bytes([42]),
2344 "tract_max": 1,
2345 "patch_nx_max": 2,
2346 "patch_ny_max": 2,
2347 },
2348 )
2349 registry.insertDimensionData("tract", {"skymap": skymap_name, "id": 0, "region": parent_region_large})
2350 registry.insertDimensionData(
2351 "patch",
2352 *[
2353 {"skymap": skymap_name, "tract": 0, "id": n, "cell_x": n % 2, "cell_y": n // 2, "region": c}
2354 for n, c in enumerate(child_regions_large)
2355 ],
2356 )
2357 # Add at dataset that uses these dimensions to make sure that modifying
2358 # them doesn't disrupt foreign keys (need to make sure DB doesn't
2359 # implement insert with replace=True as delete-then-insert).
2360 dataset_type = DatasetType(
2361 "coadd",
2362 dimensions=["tract", "patch"],
2363 universe=registry.dimensions,
2364 storageClass="Exposure",
2365 )
2366 registry.registerDatasetType(dataset_type)
2367 registry.registerCollection("the_run", CollectionType.RUN)
2368 registry.insertDatasets(
2369 dataset_type,
2370 [{"skymap": skymap_name, "tract": 0, "patch": 2}],
2371 run="the_run",
2372 )
2373 # Query for tracts and patches that overlap some "difference" htm9
2374 # pixels; there should be overlaps, because the database has
2375 # the "large" suite of regions.
2376 self.assertEqual(
2377 {0},
2378 {
2379 data_id["tract"]
2380 for data_id in registry.queryDataIds(
2381 ["tract"],
2382 skymap=skymap_name,
2383 dataId={registry.dimensions.commonSkyPix.name: parent_difference_indices[0]},
2384 )
2385 },
2386 )
2387 for patch_id, patch_difference_indices in enumerate(child_difference_indices):
2388 self.assertIn(
2389 patch_id,
2390 {
2391 data_id["patch"]
2392 for data_id in registry.queryDataIds(
2393 ["patch"],
2394 skymap=skymap_name,
2395 dataId={registry.dimensions.commonSkyPix.name: patch_difference_indices[0]},
2396 )
2397 },
2398 )
2399 # Use sync to update the tract region and insert to update the regions
2400 # of the patches, to the "small" suite.
2401 updated = registry.syncDimensionData(
2402 "tract",
2403 {"skymap": skymap_name, "id": 0, "region": parent_region_small},
2404 update=True,
2405 )
2406 self.assertEqual(updated, {"region": parent_region_large})
2407 registry.insertDimensionData(
2408 "patch",
2409 *[
2410 {"skymap": skymap_name, "tract": 0, "id": n, "cell_x": n % 2, "cell_y": n // 2, "region": c}
2411 for n, c in enumerate(child_regions_small)
2412 ],
2413 replace=True,
2414 )
2415 # Query again; there now should be no such overlaps, because the
2416 # database has the "small" suite of regions.
2417 self.assertFalse(
2418 set(
2419 registry.queryDataIds(
2420 ["tract"],
2421 skymap=skymap_name,
2422 dataId={registry.dimensions.commonSkyPix.name: parent_difference_indices[0]},
2423 )
2424 )
2425 )
2426 for patch_id, patch_difference_indices in enumerate(child_difference_indices):
2427 self.assertNotIn(
2428 patch_id,
2429 {
2430 data_id["patch"]
2431 for data_id in registry.queryDataIds(
2432 ["patch"],
2433 skymap=skymap_name,
2434 dataId={registry.dimensions.commonSkyPix.name: patch_difference_indices[0]},
2435 )
2436 },
2437 )
2438 # Update back to the large regions and query one more time.
2439 updated = registry.syncDimensionData(
2440 "tract",
2441 {"skymap": skymap_name, "id": 0, "region": parent_region_large},
2442 update=True,
2443 )
2444 self.assertEqual(updated, {"region": parent_region_small})
2445 registry.insertDimensionData(
2446 "patch",
2447 *[
2448 {"skymap": skymap_name, "tract": 0, "id": n, "cell_x": n % 2, "cell_y": n // 2, "region": c}
2449 for n, c in enumerate(child_regions_large)
2450 ],
2451 replace=True,
2452 )
2453 self.assertEqual(
2454 {0},
2455 {
2456 data_id["tract"]
2457 for data_id in registry.queryDataIds(
2458 ["tract"],
2459 skymap=skymap_name,
2460 dataId={registry.dimensions.commonSkyPix.name: parent_difference_indices[0]},
2461 )
2462 },
2463 )
2464 for patch_id, patch_difference_indices in enumerate(child_difference_indices):
2465 self.assertIn(
2466 patch_id,
2467 {
2468 data_id["patch"]
2469 for data_id in registry.queryDataIds(
2470 ["patch"],
2471 skymap=skymap_name,
2472 dataId={registry.dimensions.commonSkyPix.name: patch_difference_indices[0]},
2473 )
2474 },
2475 )
2477 def testCalibrationCollections(self):
2478 """Test operations on `~CollectionType.CALIBRATION` collections,
2479 including `SqlRegistry.certify`, `SqlRegistry.decertify`,
2480 `SqlRegistry.findDataset`, and
2481 `DataCoordinateQueryResults.findRelatedDatasets`.
2482 """
2483 # Setup - make a Registry, fill it with some datasets in
2484 # non-calibration collections.
2485 butler = self.make_butler()
2486 registry = butler.registry
2487 self.load_data(butler, "base.yaml", "datasets.yaml")
2488 # Set up some timestamps.
2489 t1 = astropy.time.Time("2020-01-01T01:00:00", format="isot", scale="tai")
2490 t2 = astropy.time.Time("2020-01-01T02:00:00", format="isot", scale="tai")
2491 t3 = astropy.time.Time("2020-01-01T03:00:00", format="isot", scale="tai")
2492 t4 = astropy.time.Time("2020-01-01T04:00:00", format="isot", scale="tai")
2493 t5 = astropy.time.Time("2020-01-01T05:00:00", format="isot", scale="tai")
2494 allTimespans = [
2495 Timespan(a, b) for a, b in itertools.combinations([None, t1, t2, t3, t4, t5, None], r=2)
2496 ]
2497 # Insert some exposure records with timespans between each sequential
2498 # pair of those.
2499 registry.insertDimensionData(
2500 "day_obs", {"instrument": "Cam1", "id": 20200101, "timespan": Timespan(t1, t5)}
2501 )
2502 registry.insertDimensionData(
2503 "group",
2504 {"instrument": "Cam1", "name": "group0"},
2505 {"instrument": "Cam1", "name": "group1"},
2506 {"instrument": "Cam1", "name": "group2"},
2507 {"instrument": "Cam1", "name": "group3"},
2508 )
2509 registry.insertDimensionData(
2510 "exposure",
2511 {
2512 "instrument": "Cam1",
2513 "id": 0,
2514 "group": "group0",
2515 "obs_id": "zero",
2516 "physical_filter": "Cam1-G",
2517 "day_obs": 20200101,
2518 "timespan": Timespan(t1, t2),
2519 },
2520 {
2521 "instrument": "Cam1",
2522 "id": 1,
2523 "group": "group1",
2524 "obs_id": "one",
2525 "physical_filter": "Cam1-G",
2526 "day_obs": 20200101,
2527 "timespan": Timespan(t2, t3),
2528 },
2529 {
2530 "instrument": "Cam1",
2531 "id": 2,
2532 "group": "group2",
2533 "obs_id": "two",
2534 "physical_filter": "Cam1-G",
2535 "day_obs": 20200101,
2536 "timespan": Timespan(t3, t4),
2537 },
2538 {
2539 "instrument": "Cam1",
2540 "id": 3,
2541 "group": "group3",
2542 "obs_id": "three",
2543 "physical_filter": "Cam1-G",
2544 "day_obs": 20200101,
2545 "timespan": Timespan(t4, t5),
2546 },
2547 )
2548 # Get references to some datasets.
2549 bias2a = registry.findDataset("bias", instrument="Cam1", detector=2, collections="imported_g")
2550 bias3a = registry.findDataset("bias", instrument="Cam1", detector=3, collections="imported_g")
2551 bias2b = registry.findDataset("bias", instrument="Cam1", detector=2, collections="imported_r")
2552 bias3b = registry.findDataset("bias", instrument="Cam1", detector=3, collections="imported_r")
2553 # Register the main calibration collection we'll be working with.
2554 collection = "Cam1/calibs/default"
2555 registry.registerCollection(collection, type=CollectionType.CALIBRATION)
2556 # Cannot associate into a calibration collection (no timespan).
2557 with self.assertRaises(CollectionTypeError):
2558 registry.associate(collection, [bias2a])
2559 # Certify 2a dataset with [t2, t4) validity.
2560 registry.certify(collection, [bias2a], Timespan(begin=t2, end=t4))
2561 # Test that we can query for this dataset via the new collection, both
2562 # on its own and with a RUN collection.
2563 self.assertEqual(
2564 set(registry.queryDatasets("bias", findFirst=False, collections=collection)),
2565 {bias2a},
2566 )
2567 self.assertEqual(
2568 set(registry.queryDatasets("bias", findFirst=False, collections=[collection, "imported_r"])),
2569 {
2570 bias2a,
2571 bias2b,
2572 bias3b,
2573 registry.findDataset("bias", instrument="Cam1", detector=4, collections="imported_r"),
2574 },
2575 )
2576 self.assertEqual(
2577 set(registry.queryDataIds("detector", datasets="bias", collections=collection)),
2578 {registry.expandDataId(instrument="Cam1", detector=2)},
2579 )
2580 self.assertEqual(
2581 set(registry.queryDataIds("detector", datasets="bias", collections=[collection, "imported_r"])),
2582 {
2583 registry.expandDataId(instrument="Cam1", detector=2),
2584 registry.expandDataId(instrument="Cam1", detector=3),
2585 registry.expandDataId(instrument="Cam1", detector=4),
2586 },
2587 )
2588 self.assertEqual(
2589 set(
2590 registry.queryDataIds(["exposure", "detector"]).findRelatedDatasets(
2591 "bias", findFirst=True, collections=[collection]
2592 )
2593 ),
2594 {
2595 (registry.expandDataId(instrument="Cam1", detector=2, exposure=1), bias2a),
2596 (registry.expandDataId(instrument="Cam1", detector=2, exposure=2), bias2a),
2597 },
2598 )
2600 # We should not be able to certify 2b with anything overlapping that
2601 # window.
2602 with self.assertRaises(ConflictingDefinitionError):
2603 registry.certify(collection, [bias2b], Timespan(begin=None, end=t3))
2604 with self.assertRaises(ConflictingDefinitionError):
2605 registry.certify(collection, [bias2b], Timespan(begin=None, end=t5))
2606 with self.assertRaises(ConflictingDefinitionError):
2607 registry.certify(collection, [bias2b], Timespan(begin=t1, end=t3))
2608 with self.assertRaises(ConflictingDefinitionError):
2609 registry.certify(collection, [bias2b], Timespan(begin=t1, end=t5))
2610 with self.assertRaises(ConflictingDefinitionError):
2611 registry.certify(collection, [bias2b], Timespan(begin=t1, end=None))
2612 with self.assertRaises(ConflictingDefinitionError):
2613 registry.certify(collection, [bias2b], Timespan(begin=t2, end=t3))
2614 with self.assertRaises(ConflictingDefinitionError):
2615 registry.certify(collection, [bias2b], Timespan(begin=t2, end=t5))
2616 with self.assertRaises(ConflictingDefinitionError):
2617 registry.certify(collection, [bias2b], Timespan(begin=t2, end=None))
2618 # We should be able to certify 3a with a range overlapping that window,
2619 # because it's for a different detector.
2620 # We'll certify 3a over [t1, t3).
2621 registry.certify(collection, [bias3a], Timespan(begin=t1, end=t3))
2622 # Now we'll certify 2b and 3b together over [t4, ∞).
2623 registry.certify(collection, [bias2b, bias3b], Timespan(begin=t4, end=None))
2625 # Fetch all associations and check that they are what we expect.
2626 self.assertCountEqual(
2627 list(
2628 registry.queryDatasetAssociations(
2629 "bias",
2630 collections=[collection, "imported_g", "imported_r"],
2631 )
2632 ),
2633 [
2634 DatasetAssociation(
2635 ref=registry.findDataset("bias", instrument="Cam1", detector=1, collections="imported_g"),
2636 collection="imported_g",
2637 timespan=None,
2638 ),
2639 DatasetAssociation(
2640 ref=registry.findDataset("bias", instrument="Cam1", detector=4, collections="imported_r"),
2641 collection="imported_r",
2642 timespan=None,
2643 ),
2644 DatasetAssociation(ref=bias2a, collection="imported_g", timespan=None),
2645 DatasetAssociation(ref=bias3a, collection="imported_g", timespan=None),
2646 DatasetAssociation(ref=bias2b, collection="imported_r", timespan=None),
2647 DatasetAssociation(ref=bias3b, collection="imported_r", timespan=None),
2648 DatasetAssociation(ref=bias2a, collection=collection, timespan=Timespan(begin=t2, end=t4)),
2649 DatasetAssociation(ref=bias3a, collection=collection, timespan=Timespan(begin=t1, end=t3)),
2650 DatasetAssociation(ref=bias2b, collection=collection, timespan=Timespan(begin=t4, end=None)),
2651 DatasetAssociation(ref=bias3b, collection=collection, timespan=Timespan(begin=t4, end=None)),
2652 ],
2653 )
2655 # Test dataset association query against a chained collection.
2656 # This is a regression test for DM-53179, as well as verification
2657 # that the flattenChains parameter has never had any effect.
2658 butler.collections.register("chain", CollectionType.CHAINED)
2659 butler.collections.redefine_chain("chain", [collection])
2660 expected_datasets = (
2661 DatasetAssociation(ref=bias2a, collection=collection, timespan=Timespan(begin=t2, end=t4)),
2662 DatasetAssociation(ref=bias3a, collection=collection, timespan=Timespan(begin=t1, end=t3)),
2663 DatasetAssociation(ref=bias2b, collection=collection, timespan=Timespan(begin=t4, end=None)),
2664 DatasetAssociation(ref=bias3b, collection=collection, timespan=Timespan(begin=t4, end=None)),
2665 )
2666 self.assertCountEqual(
2667 list(registry.queryDatasetAssociations("bias", collections=["chain"], flattenChains=False)),
2668 expected_datasets,
2669 )
2670 self.assertCountEqual(
2671 list(registry.queryDatasetAssociations("bias", collections=["chain"], flattenChains=True)),
2672 expected_datasets,
2673 )
2675 class Ambiguous:
2676 """Tag class to denote lookups that should be ambiguous."""
2678 pass
2680 def _assertLookup(
2681 detector: int, timespan: Timespan, expected: DatasetRef | type[Ambiguous] | None
2682 ) -> None:
2683 """Local function that asserts that a bias lookup returns the given
2684 expected result.
2685 """
2686 if expected is Ambiguous:
2687 with self.assertRaises((DatasetTypeError, LookupError)):
2688 registry.findDataset(
2689 "bias",
2690 collections=collection,
2691 instrument="Cam1",
2692 detector=detector,
2693 timespan=timespan,
2694 )
2695 else:
2696 self.assertEqual(
2697 expected,
2698 registry.findDataset(
2699 "bias",
2700 collections=collection,
2701 instrument="Cam1",
2702 detector=detector,
2703 timespan=timespan,
2704 ),
2705 )
2707 # Systematically test lookups against expected results.
2708 _assertLookup(detector=2, timespan=Timespan(None, t1), expected=None)
2709 _assertLookup(detector=2, timespan=Timespan(None, t2), expected=None)
2710 _assertLookup(detector=2, timespan=Timespan(None, t3), expected=bias2a)
2711 _assertLookup(detector=2, timespan=Timespan(None, t4), expected=bias2a)
2712 _assertLookup(detector=2, timespan=Timespan(None, t5), expected=Ambiguous)
2713 _assertLookup(detector=2, timespan=Timespan(None, None), expected=Ambiguous)
2714 _assertLookup(detector=2, timespan=Timespan(t1, t2), expected=None)
2715 _assertLookup(detector=2, timespan=Timespan(t1, t3), expected=bias2a)
2716 _assertLookup(detector=2, timespan=Timespan(t1, t4), expected=bias2a)
2717 _assertLookup(detector=2, timespan=Timespan(t1, t5), expected=Ambiguous)
2718 _assertLookup(detector=2, timespan=Timespan(t1, None), expected=Ambiguous)
2719 _assertLookup(detector=2, timespan=Timespan(t2, t3), expected=bias2a)
2720 _assertLookup(detector=2, timespan=Timespan(t2, t4), expected=bias2a)
2721 _assertLookup(detector=2, timespan=Timespan(t2, t5), expected=Ambiguous)
2722 _assertLookup(detector=2, timespan=Timespan(t2, None), expected=Ambiguous)
2723 _assertLookup(detector=2, timespan=Timespan(t3, t4), expected=bias2a)
2724 _assertLookup(detector=2, timespan=Timespan(t3, t5), expected=Ambiguous)
2725 _assertLookup(detector=2, timespan=Timespan(t3, None), expected=Ambiguous)
2726 _assertLookup(detector=2, timespan=Timespan(t4, t5), expected=bias2b)
2727 _assertLookup(detector=2, timespan=Timespan(t4, None), expected=bias2b)
2728 _assertLookup(detector=2, timespan=Timespan(t5, None), expected=bias2b)
2729 _assertLookup(detector=3, timespan=Timespan(None, t1), expected=None)
2730 _assertLookup(detector=3, timespan=Timespan(None, t2), expected=bias3a)
2731 _assertLookup(detector=3, timespan=Timespan(None, t3), expected=bias3a)
2732 _assertLookup(detector=3, timespan=Timespan(None, t4), expected=bias3a)
2733 _assertLookup(detector=3, timespan=Timespan(None, t5), expected=Ambiguous)
2734 _assertLookup(detector=3, timespan=Timespan(None, None), expected=Ambiguous)
2735 _assertLookup(detector=3, timespan=Timespan(t1, t2), expected=bias3a)
2736 _assertLookup(detector=3, timespan=Timespan(t1, t3), expected=bias3a)
2737 _assertLookup(detector=3, timespan=Timespan(t1, t4), expected=bias3a)
2738 _assertLookup(detector=3, timespan=Timespan(t1, t5), expected=Ambiguous)
2739 _assertLookup(detector=3, timespan=Timespan(t1, None), expected=Ambiguous)
2740 _assertLookup(detector=3, timespan=Timespan(t2, t3), expected=bias3a)
2741 _assertLookup(detector=3, timespan=Timespan(t2, t4), expected=bias3a)
2742 _assertLookup(detector=3, timespan=Timespan(t2, t5), expected=Ambiguous)
2743 _assertLookup(detector=3, timespan=Timespan(t2, None), expected=Ambiguous)
2744 _assertLookup(detector=3, timespan=Timespan(t3, t4), expected=None)
2745 _assertLookup(detector=3, timespan=Timespan(t3, t5), expected=bias3b)
2746 _assertLookup(detector=3, timespan=Timespan(t3, None), expected=bias3b)
2747 _assertLookup(detector=3, timespan=Timespan(t4, t5), expected=bias3b)
2748 _assertLookup(detector=3, timespan=Timespan(t4, None), expected=bias3b)
2749 _assertLookup(detector=3, timespan=Timespan(t5, None), expected=bias3b)
2751 # Test lookups via temporal joins to exposures.
2752 self.assertEqual(
2753 set(
2754 registry.queryDataIds(
2755 ["exposure", "detector"], instrument="Cam1", detector=2
2756 ).findRelatedDatasets("bias", collections=[collection])
2757 ),
2758 {
2759 (registry.expandDataId(instrument="Cam1", exposure=1, detector=2), bias2a),
2760 (registry.expandDataId(instrument="Cam1", exposure=2, detector=2), bias2a),
2761 (registry.expandDataId(instrument="Cam1", exposure=3, detector=2), bias2b),
2762 },
2763 )
2764 self.assertEqual(
2765 set(
2766 registry.queryDataIds(
2767 ["exposure", "detector"], instrument="Cam1", detector=3
2768 ).findRelatedDatasets("bias", collections=[collection])
2769 ),
2770 {
2771 (registry.expandDataId(instrument="Cam1", exposure=0, detector=3), bias3a),
2772 (registry.expandDataId(instrument="Cam1", exposure=1, detector=3), bias3a),
2773 (registry.expandDataId(instrument="Cam1", exposure=3, detector=3), bias3b),
2774 },
2775 )
2777 # Decertify [t3, t5) for all data IDs, and do test lookups again.
2778 # This should truncate bias2a to [t2, t3), leave bias3a unchanged at
2779 # [t1, t3), and truncate bias2b and bias3b to [t5, ∞).
2780 registry.decertify(collection=collection, datasetType="bias", timespan=Timespan(t3, t5))
2781 _assertLookup(detector=2, timespan=Timespan(None, t1), expected=None)
2782 _assertLookup(detector=2, timespan=Timespan(None, t2), expected=None)
2783 _assertLookup(detector=2, timespan=Timespan(None, t3), expected=bias2a)
2784 _assertLookup(detector=2, timespan=Timespan(None, t4), expected=bias2a)
2785 _assertLookup(detector=2, timespan=Timespan(None, t5), expected=bias2a)
2786 _assertLookup(detector=2, timespan=Timespan(None, None), expected=Ambiguous)
2787 _assertLookup(detector=2, timespan=Timespan(t1, t2), expected=None)
2788 _assertLookup(detector=2, timespan=Timespan(t1, t3), expected=bias2a)
2789 _assertLookup(detector=2, timespan=Timespan(t1, t4), expected=bias2a)
2790 _assertLookup(detector=2, timespan=Timespan(t1, t5), expected=bias2a)
2791 _assertLookup(detector=2, timespan=Timespan(t1, None), expected=Ambiguous)
2792 _assertLookup(detector=2, timespan=Timespan(t2, t3), expected=bias2a)
2793 _assertLookup(detector=2, timespan=Timespan(t2, t4), expected=bias2a)
2794 _assertLookup(detector=2, timespan=Timespan(t2, t5), expected=bias2a)
2795 _assertLookup(detector=2, timespan=Timespan(t2, None), expected=Ambiguous)
2796 _assertLookup(detector=2, timespan=Timespan(t3, t4), expected=None)
2797 _assertLookup(detector=2, timespan=Timespan(t3, t5), expected=None)
2798 _assertLookup(detector=2, timespan=Timespan(t3, None), expected=bias2b)
2799 _assertLookup(detector=2, timespan=Timespan(t4, t5), expected=None)
2800 _assertLookup(detector=2, timespan=Timespan(t4, None), expected=bias2b)
2801 _assertLookup(detector=2, timespan=Timespan(t5, None), expected=bias2b)
2802 _assertLookup(detector=3, timespan=Timespan(None, t1), expected=None)
2803 _assertLookup(detector=3, timespan=Timespan(None, t2), expected=bias3a)
2804 _assertLookup(detector=3, timespan=Timespan(None, t3), expected=bias3a)
2805 _assertLookup(detector=3, timespan=Timespan(None, t4), expected=bias3a)
2806 _assertLookup(detector=3, timespan=Timespan(None, t5), expected=bias3a)
2807 _assertLookup(detector=3, timespan=Timespan(None, None), expected=Ambiguous)
2808 _assertLookup(detector=3, timespan=Timespan(t1, t2), expected=bias3a)
2809 _assertLookup(detector=3, timespan=Timespan(t1, t3), expected=bias3a)
2810 _assertLookup(detector=3, timespan=Timespan(t1, t4), expected=bias3a)
2811 _assertLookup(detector=3, timespan=Timespan(t1, t5), expected=bias3a)
2812 _assertLookup(detector=3, timespan=Timespan(t1, None), expected=Ambiguous)
2813 _assertLookup(detector=3, timespan=Timespan(t2, t3), expected=bias3a)
2814 _assertLookup(detector=3, timespan=Timespan(t2, t4), expected=bias3a)
2815 _assertLookup(detector=3, timespan=Timespan(t2, t5), expected=bias3a)
2816 _assertLookup(detector=3, timespan=Timespan(t2, None), expected=Ambiguous)
2817 _assertLookup(detector=3, timespan=Timespan(t3, t4), expected=None)
2818 _assertLookup(detector=3, timespan=Timespan(t3, t5), expected=None)
2819 _assertLookup(detector=3, timespan=Timespan(t3, None), expected=bias3b)
2820 _assertLookup(detector=3, timespan=Timespan(t4, t5), expected=None)
2821 _assertLookup(detector=3, timespan=Timespan(t4, None), expected=bias3b)
2822 _assertLookup(detector=3, timespan=Timespan(t5, None), expected=bias3b)
2824 # Decertify everything, this time with explicit data IDs, then check
2825 # that no lookups succeed.
2826 registry.decertify(
2827 collection,
2828 "bias",
2829 Timespan(None, None),
2830 dataIds=[
2831 dict(instrument="Cam1", detector=2),
2832 dict(instrument="Cam1", detector=3),
2833 ],
2834 )
2835 for detector in (2, 3):
2836 for timespan in allTimespans:
2837 _assertLookup(detector=detector, timespan=timespan, expected=None)
2838 # Certify bias2a and bias3a over (-∞, ∞), check that all lookups return
2839 # those.
2840 registry.certify(
2841 collection,
2842 [bias2a, bias3a],
2843 Timespan(None, None),
2844 )
2845 for timespan in allTimespans:
2846 _assertLookup(detector=2, timespan=timespan, expected=bias2a)
2847 _assertLookup(detector=3, timespan=timespan, expected=bias3a)
2848 # Decertify just bias2 over [t2, t4).
2849 # This should split a single certification row into two (and leave the
2850 # other existing row, for bias3a, alone).
2851 registry.decertify(
2852 collection, "bias", Timespan(t2, t4), dataIds=[dict(instrument="Cam1", detector=2)]
2853 )
2854 for timespan in allTimespans:
2855 _assertLookup(detector=3, timespan=timespan, expected=bias3a)
2856 overlapsBefore = timespan.overlaps(Timespan(None, t2))
2857 overlapsAfter = timespan.overlaps(Timespan(t4, None))
2858 if overlapsBefore and overlapsAfter:
2859 expected = Ambiguous
2860 elif overlapsBefore or overlapsAfter:
2861 expected = bias2a
2862 else:
2863 expected = None
2864 _assertLookup(detector=2, timespan=timespan, expected=expected)
2866 def testSkipCalibs(self):
2867 """Test how queries handle skipping of calibration collections."""
2868 butler = self.make_butler()
2869 registry = butler.registry
2870 self.load_data(butler, "base.yaml", "datasets.yaml")
2872 coll_calib = "Cam1/calibs/default"
2873 registry.registerCollection(coll_calib, type=CollectionType.CALIBRATION)
2875 # Add all biases to the calibration collection.
2876 # Without this, the logic that prunes dataset subqueries based on
2877 # datasetType-collection summary information will fire before the logic
2878 # we want to test below. This is a good thing (it avoids the dreaded
2879 # NotImplementedError a bit more often) everywhere but here.
2880 registry.certify(coll_calib, registry.queryDatasets("bias", collections=...), Timespan(None, None))
2882 coll_list = [coll_calib, "imported_g", "imported_r"]
2883 chain = "Cam1/chain"
2884 registry.registerCollection(chain, type=CollectionType.CHAINED)
2885 registry.setCollectionChain(chain, coll_list)
2887 # Lookup is ambiguous due to multiple datasets with the same data ID
2888 # in the calibration collection.
2889 with self.assertRaises(CalibrationLookupError):
2890 list(registry.queryDatasets("bias", collections=coll_list, findFirst=True))
2892 # chain will skip
2893 datasets = list(registry.queryDatasets("bias", collections=chain))
2894 self.assertGreater(len(datasets), 0)
2896 dataIds = list(registry.queryDataIds(["instrument", "detector"], datasets="bias", collections=chain))
2897 self.assertGreater(len(dataIds), 0)
2899 # glob will skip too
2900 datasets = list(registry.queryDatasets("bias", collections="*d*"))
2901 self.assertGreater(len(datasets), 0)
2903 # regular expression will skip too
2904 if self.supportsCollectionRegex: 2904 ↛ 2905line 2904 didn't jump to line 2905 because the condition on line 2904 was never true
2905 pattern = re.compile(".*")
2906 with self.assertWarns(FutureWarning):
2907 datasets = list(registry.queryDatasets("bias", collections=pattern))
2908 self.assertGreater(len(datasets), 0)
2910 # ellipsis should work as usual
2911 datasets = list(registry.queryDatasets("bias", collections=...))
2912 self.assertGreater(len(datasets), 0)
2914 # New query system correctly determines that this search is
2915 # ambiguous, because there are multiple datasets with the same
2916 # {instrument=Cam1, detector=2} data ID in the calibration
2917 # collection at the beginning of the chain.
2918 with self.assertRaises(CalibrationLookupError):
2919 datasets = list(registry.queryDatasets("bias", collections=chain, findFirst=True))
2921 def testIngestTimeQuery(self):
2922 butler = self.make_butler()
2923 registry = butler.registry
2924 dt0 = datetime.datetime.now(datetime.UTC)
2925 self.load_data(butler, "base.yaml", "datasets.yaml")
2926 dt1 = datetime.datetime.now(datetime.UTC)
2928 datasets = list(registry.queryDatasets(..., collections=...))
2929 len0 = len(datasets)
2930 self.assertGreater(len0, 0)
2932 for where in ("ingest_date > T'2000-01-01'", "T'2000-01-01' < ingest_date"):
2933 datasets = list(registry.queryDatasets(..., collections=..., where=where))
2934 len1 = len(datasets)
2935 self.assertEqual(len0, len1)
2937 # no one will ever use this piece of software in 30 years
2938 for where in ("ingest_date > T'2050-01-01'", "T'2050-01-01' < ingest_date"):
2939 datasets = list(registry.queryDatasets(..., collections=..., where=where))
2940 len2 = len(datasets)
2941 self.assertEqual(len2, 0)
2943 # Check more exact timing to make sure there is no 37 seconds offset
2944 # (after fixing DM-30124). SQLite time precision is 1 second, make
2945 # sure that we don't test with higher precision.
2946 tests = [
2947 # format: (timestamp, operator, expected_len)
2948 (dt0 - timedelta(seconds=1), ">", len0),
2949 (dt0 - timedelta(seconds=1), "<", 0),
2950 (dt1 + timedelta(seconds=1), "<", len0),
2951 (dt1 + timedelta(seconds=1), ">", 0),
2952 ]
2953 for dt, op, expect_len in tests:
2954 dt_str = dt.isoformat(sep=" ")
2956 where = f"ingest_date {op} T'{dt_str}'"
2957 datasets = list(registry.queryDatasets(..., collections=..., where=where))
2958 self.assertEqual(len(datasets), expect_len)
2960 # same with bind using datetime or astropy Time
2961 where = f"ingest_date {op} :ingest_time"
2962 datasets = list(
2963 registry.queryDatasets(..., collections=..., where=where, bind={"ingest_time": dt})
2964 )
2965 self.assertEqual(len(datasets), expect_len)
2967 dt_astropy = astropy.time.Time(dt, format="datetime")
2968 datasets = list(
2969 registry.queryDatasets(..., collections=..., where=where, bind={"ingest_time": dt_astropy})
2970 )
2971 self.assertEqual(len(datasets), expect_len)
2973 def testTimespanQueries(self):
2974 """Test query expressions involving timespans."""
2975 butler = self.make_butler()
2976 registry = butler.registry
2977 self.load_data(butler, "ci_hsc-subset.yaml")
2978 # All exposures in the database; mapping from ID to timespan.
2979 visits = {record.id: record.timespan for record in registry.queryDimensionRecords("visit")}
2980 # Just those IDs, sorted (which is also temporal sorting, because HSC
2981 # exposure IDs are monotonically increasing).
2982 ids = sorted(visits.keys())
2983 self.assertEqual(len(ids), 11)
2984 # Pick some quasi-random indexes into `ids` to play with.
2985 i1 = int(len(ids) * 0.1)
2986 i2 = int(len(ids) * 0.3)
2987 i3 = int(len(ids) * 0.6)
2988 i4 = int(len(ids) * 0.8)
2989 # Extract some times from those: just before the beginning of i1 (which
2990 # should be after the end of the exposure before), exactly the
2991 # beginning of i2, just after the beginning of i3 (and before its end),
2992 # and the exact end of i4.
2993 t1 = visits[ids[i1]].begin - astropy.time.TimeDelta(1.0, format="sec")
2994 self.assertGreater(t1, visits[ids[i1 - 1]].end)
2995 t2 = visits[ids[i2]].begin
2996 t3 = visits[ids[i3]].begin + astropy.time.TimeDelta(1.0, format="sec")
2997 self.assertLess(t3, visits[ids[i3]].end)
2998 t4 = visits[ids[i4]].end
2999 # Make sure those are actually in order.
3000 self.assertEqual([t1, t2, t3, t4], sorted([t4, t3, t2, t1]))
3002 bind = {
3003 "t1": t1,
3004 "t2": t2,
3005 "t3": t3,
3006 "t4": t4,
3007 "ts23": Timespan(t2, t3),
3008 }
3010 def query(where):
3011 """Return results as a sorted, deduplicated list of visit IDs.
3013 Parameters
3014 ----------
3015 where : `str`
3016 The WHERE clause for the query.
3017 """
3018 return sorted(
3019 {
3020 dataId["visit"]
3021 for dataId in registry.queryDataIds("visit", instrument="HSC", bind=bind, where=where)
3022 }
3023 )
3025 # Try a bunch of timespan queries, mixing up the bounds themselves,
3026 # where they appear in the expression, and how we get the timespan into
3027 # the expression.
3029 # t1 is before the start of i1, so this should not include i1.
3030 self.assertEqual(ids[:i1], query("visit.timespan OVERLAPS (null, :t1)"))
3031 # t2 is exactly at the start of i2, but ends are exclusive, so these
3032 # should not include i2.
3033 self.assertEqual(ids[i1:i2], query("(:t1, :t2) OVERLAPS visit.timespan"))
3034 # t3 is in the middle of i3, so this should include i3.
3035 self.assertEqual(ids[i2 : i3 + 1], query("visit.timespan OVERLAPS :ts23"))
3036 # This one should not include t3 by the same reasoning.
3037 # t4 is exactly at the end of i4, so this should include i4.
3038 self.assertEqual(ids[i3 : i4 + 1], query(f"visit.timespan OVERLAPS (T'{t3.tai.isot}/tai', :t4)"))
3039 # i4's upper bound of t4 is exclusive so this should not include t4.
3040 self.assertEqual(ids[i4 + 1 :], query("visit.timespan OVERLAPS (:t4, NULL)"))
3042 # Now some timespan vs. time scalar queries.
3043 self.assertEqual(ids[i3 : i3 + 1], query("visit.timespan OVERLAPS :t3"))
3044 self.assertEqual(ids[i3 : i3 + 1], query(f"T'{t3.tai.isot}/tai' OVERLAPS visit.timespan"))
3046 # Empty timespans should not overlap anything.
3047 self.assertEqual([], query("visit.timespan OVERLAPS (:t3, :t2)"))
3049 # Make sure that expanded data IDs include the timespans.
3050 results = list(
3051 registry.queryDataIds(["visit"], dataId={"instrument": "HSC", "visit": ids[1]}).expanded()
3052 )
3053 self.assertEqual(len(results), 1)
3054 visit_timespan = visits[ids[1]]
3055 self.assertEqual(results[0].timespan, visit_timespan)
3056 visit_record = results[0].records["visit"]
3057 assert visit_record is not None
3058 self.assertEqual(visit_record.timespan, visit_timespan)
3059 day_obs_record = results[0].records["day_obs"]
3060 assert day_obs_record is not None
3061 self.assertEqual(day_obs_record.id, 20130617)
3062 self.assertEqual(
3063 day_obs_record.timespan,
3064 Timespan(
3065 astropy.time.Time("2013-06-17T00:00:00", scale="tai"),
3066 astropy.time.Time("2013-06-18T00:00:00", scale="tai"),
3067 ),
3068 )
3070 def testCollectionSummaries(self):
3071 """Test recording and retrieval of collection summaries."""
3072 self.maxDiff = None
3073 butler = self.make_butler()
3074 registry = butler.registry
3075 # Importing datasets from yaml should go through the code path where
3076 # we update collection summaries as we insert datasets.
3077 self.load_data(butler, "base.yaml", "datasets.yaml")
3078 flat = registry.getDatasetType("flat")
3079 expected1 = CollectionSummary()
3080 expected1.dataset_types.add(registry.getDatasetType("bias"))
3081 expected1.add_data_ids(
3082 flat, [DataCoordinate.standardize(instrument="Cam1", universe=registry.dimensions)]
3083 )
3084 self.assertEqual(registry.getCollectionSummary("imported_g"), expected1)
3085 self.assertEqual(registry.getCollectionSummary("imported_r"), expected1)
3086 # Create a chained collection with both of the imported runs; the
3087 # summary should be the same, because it's a union with itself.
3088 chain = "chain"
3089 registry.registerCollection(chain, CollectionType.CHAINED)
3090 registry.setCollectionChain(chain, ["imported_r", "imported_g"])
3091 self.assertEqual(registry.getCollectionSummary(chain), expected1)
3092 # Associate flats only into a tagged collection and a calibration
3093 # collection to check summaries of those.
3094 tag = "tag"
3095 registry.registerCollection(tag, CollectionType.TAGGED)
3096 registry.associate(tag, registry.queryDatasets(flat, collections="imported_g"))
3097 calibs = "calibs"
3098 registry.registerCollection(calibs, CollectionType.CALIBRATION)
3099 registry.certify(
3100 calibs, registry.queryDatasets(flat, collections="imported_g"), timespan=Timespan(None, None)
3101 )
3102 expected2 = expected1.copy()
3103 expected2.dataset_types.discard("bias")
3104 self.assertEqual(registry.getCollectionSummary(tag), expected2)
3105 self.assertEqual(registry.getCollectionSummary(calibs), expected2)
3106 # Explicitly calling SqlRegistry.refresh() should load those same
3107 # summaries, via a totally different code path.
3108 registry.refresh()
3109 self.assertEqual(registry.getCollectionSummary("imported_g"), expected1)
3110 self.assertEqual(registry.getCollectionSummary("imported_r"), expected1)
3111 self.assertEqual(registry.getCollectionSummary(tag), expected2)
3112 self.assertEqual(registry.getCollectionSummary(calibs), expected2)
3114 def testBindInQueryDatasets(self):
3115 """Test that the bind parameter is correctly forwarded in
3116 queryDatasets recursion.
3117 """
3118 butler = self.make_butler()
3119 registry = butler.registry
3120 # Importing datasets from yaml should go through the code path where
3121 # we update collection summaries as we insert datasets.
3122 self.load_data(butler, "base.yaml", "datasets.yaml")
3123 self.assertEqual(
3124 set(registry.queryDatasets("flat", band="r", collections=...)),
3125 set(
3126 registry.queryDatasets("flat", where="band=:my_band", bind={"my_band": "r"}, collections=...)
3127 ),
3128 )
3130 def testQueryIntRangeExpressions(self):
3131 """Test integer range expressions in ``where`` arguments.
3133 Note that our expressions use inclusive stop values, unlike Python's.
3134 """
3135 butler = self.make_butler()
3136 registry = butler.registry
3137 self.load_data(butler, "base.yaml")
3138 self.assertEqual(
3139 set(registry.queryDataIds(["detector"], instrument="Cam1", where="detector IN (1..2)")),
3140 {registry.expandDataId(instrument="Cam1", detector=n) for n in [1, 2]},
3141 )
3142 self.assertEqual(
3143 set(registry.queryDataIds(["detector"], instrument="Cam1", where="detector IN (1..4:2)")),
3144 {registry.expandDataId(instrument="Cam1", detector=n) for n in [1, 3]},
3145 )
3146 self.assertEqual(
3147 set(registry.queryDataIds(["detector"], instrument="Cam1", where="detector IN (2..4:2)")),
3148 {registry.expandDataId(instrument="Cam1", detector=n) for n in [2, 4]},
3149 )
3151 def testQueryResultSummaries(self):
3152 """Test summary methods like `count`, `any`, and `explain_no_results`
3153 on `DataCoordinateQueryResults` and `DatasetQueryResults`.
3154 """
3155 butler = self.make_butler()
3156 registry = butler.registry
3157 self.load_data(butler, "base.yaml", "datasets.yaml", "spatial.yaml")
3158 # Default test dataset has two collections, each with both flats and
3159 # biases. Add a new collection with only biases.
3160 registry.registerCollection("biases", CollectionType.TAGGED)
3161 registry.associate("biases", registry.queryDatasets("bias", collections=["imported_g"]))
3162 # First query yields two results, and involves no postprocessing.
3163 query1 = registry.queryDataIds(["physical_filter"], band="r")
3164 self.assertTrue(query1.any(execute=False, exact=False))
3165 self.assertTrue(query1.any(execute=True, exact=False))
3166 self.assertTrue(query1.any(execute=True, exact=True))
3167 self.assertEqual(query1.count(exact=False), 2)
3168 self.assertEqual(query1.count(exact=True), 2)
3169 self.assertFalse(list(query1.explain_no_results()))
3170 # Second query should yield no results, which we should see when
3171 # we attempt to expand the data ID.
3172 query2 = registry.queryDataIds(["physical_filter"], band="h")
3173 # There's no execute=False, exact=False test here because the behavior
3174 # not something we want to guarantee in this case (and exact=False
3175 # says either answer is legal).
3176 self.assertFalse(query2.any(execute=True, exact=False))
3177 self.assertFalse(query2.any(execute=True, exact=True))
3178 self.assertEqual(query2.count(exact=False), 0)
3179 self.assertEqual(query2.count(exact=True), 0)
3180 # These queries yield no results due to various problems that can be
3181 # spotted prior to execution, yielding helpful diagnostics.
3182 base_query = registry.queryDataIds(["detector", "physical_filter"])
3183 queries_and_snippets = [
3184 (
3185 # Dataset type name doesn't match any existing dataset types.
3186 registry.queryDatasets("nonexistent", collections=...),
3187 ["nonexistent"],
3188 ),
3189 (
3190 # Dataset type object isn't registered.
3191 registry.queryDatasets(
3192 DatasetType(
3193 "nonexistent",
3194 dimensions=["instrument"],
3195 universe=registry.dimensions,
3196 storageClass="Image",
3197 ),
3198 collections=...,
3199 ),
3200 ["nonexistent"],
3201 ),
3202 (
3203 # No datasets of this type in this collection.
3204 registry.queryDatasets("flat", collections=["biases"]),
3205 ["flat", "biases"],
3206 ),
3207 (
3208 # No datasets of this type in this collection.
3209 base_query.findDatasets("flat", collections=["biases"]),
3210 ["flat", "biases"],
3211 ),
3212 (
3213 # No collections matching at all.
3214 registry.queryDatasets("flat", collections="potato*"),
3215 ["potato"],
3216 ),
3217 ]
3218 with self.assertRaises(MissingDatasetTypeError):
3219 # Dataset type name doesn't match any existing dataset types.
3220 list(registry.queryDataIds(["detector"], datasets=["nonexistent"], collections=...))
3221 with self.assertRaises(MissingDatasetTypeError):
3222 # Dataset type name doesn't match any existing dataset types.
3223 registry.queryDimensionRecords("detector", datasets=["nonexistent"], collections=...).any()
3224 with self.assertRaises(DatasetTypeExpressionError):
3225 # queryDimensionRecords does not allow dataset type wildcards.
3226 registry.queryDimensionRecords("detector", datasets=["f*"], collections=...).any()
3227 for query, snippets in queries_and_snippets:
3228 self.assertFalse(query.any(execute=False, exact=False))
3229 self.assertFalse(query.any(execute=True, exact=False))
3230 self.assertFalse(query.any(execute=True, exact=True))
3231 self.assertEqual(query.count(exact=False), 0)
3232 self.assertEqual(query.count(exact=True), 0)
3233 messages = list(query.explain_no_results())
3234 self.assertTrue(messages)
3235 # Want all expected snippets to appear in at least one message.
3236 self.assertTrue(
3237 any(
3238 all(snippet in message for snippet in snippets) for message in query.explain_no_results()
3239 ),
3240 messages,
3241 )
3243 # Wildcards on dataset types are not permitted in queryDataIds.
3244 with self.assertRaises(DatasetTypeExpressionError):
3245 registry.queryDataIds(["detector"], datasets=re.compile("^nonexistent$"), collections=...)
3247 # This query yields four overlaps in the database, but one is filtered
3248 # out in postprocessing. The count queries aren't accurate because
3249 # they don't account for duplication that happens due to an internal
3250 # join against commonSkyPix.
3251 query3 = registry.queryDataIds(["visit", "tract"], instrument="Cam1", skymap="SkyMap1")
3252 self.assertEqual(
3253 {
3254 DataCoordinate.standardize(
3255 instrument="Cam1",
3256 skymap="SkyMap1",
3257 visit=v,
3258 tract=t,
3259 universe=registry.dimensions,
3260 )
3261 for v, t in [(1, 0), (2, 0), (2, 1)]
3262 },
3263 set(query3),
3264 )
3265 self.assertTrue(query3.any(execute=False, exact=False))
3266 self.assertTrue(query3.any(execute=True, exact=False))
3267 self.assertTrue(query3.any(execute=True, exact=True))
3268 self.assertGreaterEqual(query3.count(exact=False), 3)
3269 self.assertGreaterEqual(query3.count(exact=True, discard=True), 3)
3270 self.assertFalse(list(query3.explain_no_results()))
3271 # This query yields overlaps in the database, but all are filtered
3272 # out in postprocessing. The count queries again aren't very useful.
3273 # We have to use `where=` here to avoid an optimization that
3274 # (currently) skips the spatial postprocess-filtering because it
3275 # recognizes that no spatial join is necessary. That's not ideal, but
3276 # fixing it is out of scope for this ticket.
3277 query4 = registry.queryDataIds(
3278 ["visit", "tract"],
3279 instrument="Cam1",
3280 skymap="SkyMap1",
3281 where="visit=1 AND detector=1 AND tract=0 AND patch=4",
3282 )
3283 self.assertFalse(set(query4))
3284 self.assertTrue(query4.any(execute=False, exact=False))
3285 self.assertTrue(query4.any(execute=True, exact=False))
3286 self.assertFalse(query4.any(execute=True, exact=True))
3287 self.assertGreaterEqual(query4.count(exact=False), 1)
3288 self.assertEqual(query4.count(exact=True, discard=True), 0)
3289 # This query should yield results from one dataset type but not the
3290 # other, which is not registered.
3291 query5 = registry.queryDatasets(["bias", "nonexistent"], collections=["biases"])
3292 self.assertTrue(set(query5))
3293 self.assertTrue(query5.any(execute=False, exact=False))
3294 self.assertTrue(query5.any(execute=True, exact=False))
3295 self.assertTrue(query5.any(execute=True, exact=True))
3296 self.assertGreaterEqual(query5.count(exact=False), 1)
3297 self.assertGreaterEqual(query5.count(exact=True), 1)
3298 # This query applies a selection that yields no results, fully in the
3299 # database. Explaining why it fails involves traversing the relation
3300 # tree and running a LIMIT 1 query at each level that has the potential
3301 # to remove rows.
3302 query6 = registry.queryDimensionRecords(
3303 "detector", where="detector.purpose = 'no-purpose'", instrument="Cam1"
3304 )
3305 self.assertEqual(query6.count(exact=True), 0)
3306 self.assertFalse(query6.any())
3308 def testQueryDataIdsExpressionError(self):
3309 """Test error checking of 'where' expressions in queryDataIds."""
3310 butler = self.make_butler()
3311 registry = butler.registry
3312 self.load_data(butler, "base.yaml")
3313 bind = {"time": astropy.time.Time("2020-01-01T01:00:00", format="isot", scale="tai")}
3314 # The diagnostics raised are slightly different between the old query
3315 # system (ValueError, first error string) and the new query system
3316 # (InvalidQueryError, second error string).
3317 with self.assertRaisesRegex(
3318 (LookupError, InvalidQueryError),
3319 r"(No dimension element with name 'foo' in 'foo\.bar'\.)|(Unrecognized identifier 'foo.bar')",
3320 ):
3321 list(registry.queryDataIds(["detector"], where="foo.bar = 12"))
3322 with self.assertRaisesRegex(
3323 (LookupError, InvalidQueryError),
3324 "(Dimension element name cannot be inferred in this context.)"
3325 "|(Unrecognized identifier 'timespan')",
3326 ):
3327 list(registry.queryDataIds(["detector"], where="timespan.end < :time", bind=bind))
3329 def testQueryDataIdsOrderBy(self):
3330 """Test order_by and limit on result returned by queryDataIds()."""
3331 butler = self.make_butler()
3332 registry = butler.registry
3333 self.load_data(butler, "base.yaml", "datasets.yaml", "spatial.yaml")
3335 def do_query(dimensions=("visit", "tract"), datasets=None, collections=None):
3336 return registry.queryDataIds(
3337 dimensions, datasets=datasets, collections=collections, instrument="Cam1", skymap="SkyMap1"
3338 )
3340 Test = namedtuple(
3341 "testQueryDataIdsOrderByTest",
3342 ("order_by", "keys", "result", "limit", "datasets", "collections"),
3343 defaults=(None, None, None),
3344 )
3346 test_data = [
3347 Test("tract,visit", "tract,visit", ((0, 1), (0, 2), (1, 2))),
3348 Test("-tract,visit", "tract,visit", ((1, 2), (0, 1), (0, 2))),
3349 Test("tract,-visit", "tract,visit", ((0, 2), (0, 1), (1, 2))),
3350 Test("-tract,-visit", "tract,visit", ((1, 2), (0, 2), (0, 1))),
3351 Test("tract.id,visit.id", "tract,visit", ((0, 1),), limit=(1,)),
3352 Test("-tract,-visit", "tract,visit", ((1, 2),), limit=(1,)),
3353 Test("tract,visit.exposure_time", "tract,visit", ((0, 2), (0, 1), (1, 2))),
3354 Test("-tract,-visit.exposure_time", "tract,visit", ((1, 2), (0, 1), (0, 2))),
3355 Test("tract,-exposure_time", "tract,visit", ((0, 1), (0, 2), (1, 2))),
3356 Test("tract,visit.name", "tract,visit", ((0, 1), (0, 2), (1, 2))),
3357 Test(
3358 "tract,-visit.timespan.begin,visit.timespan.end",
3359 "tract,visit",
3360 ((0, 2), (0, 1), (1, 2)),
3361 ),
3362 Test("visit.day_obs,exposure.day_obs", "visit,exposure", ()),
3363 Test("visit.timespan.begin,-exposure.timespan.begin", "visit,exposure", ()),
3364 Test(
3365 "tract,detector",
3366 "tract,detector",
3367 ((0, 1), (0, 2), (0, 3), (0, 4), (1, 1), (1, 2), (1, 3), (1, 4)),
3368 datasets="flat",
3369 collections="imported_r",
3370 ),
3371 Test(
3372 "tract,detector.full_name",
3373 "tract,detector",
3374 ((0, 1), (0, 2), (0, 3), (0, 4), (1, 1), (1, 2), (1, 3), (1, 4)),
3375 datasets="flat",
3376 collections="imported_r",
3377 ),
3378 Test(
3379 "tract,detector.raft,detector.name_in_raft",
3380 "tract,detector",
3381 ((0, 1), (0, 2), (0, 3), (0, 4), (1, 1), (1, 2), (1, 3), (1, 4)),
3382 datasets="flat",
3383 collections="imported_r",
3384 ),
3385 ]
3387 for test in test_data:
3388 with self.subTest(test=repr(test)):
3389 order_by = test.order_by.split(",")
3390 keys = test.keys.split(",")
3391 query = do_query(keys, test.datasets, test.collections).order_by(*order_by)
3392 if test.limit is not None:
3393 query = query.limit(*test.limit)
3394 dataIds = tuple(tuple(dataId[k] for k in keys) for dataId in query)
3395 self.assertEqual(dataIds, test.result)
3397 # and materialize
3398 query = do_query(keys).order_by(*order_by)
3399 if test.limit is not None:
3400 query = query.limit(*test.limit)
3402 # Test exceptions for errors in a name.
3403 # Many of these raise slightly different diagnostics in the old query
3404 # system (ValueError, first error string) than the new query system
3405 # (InvalidQueryError, second error string).
3406 for order_by in ("", "-"):
3407 with self.assertRaisesRegex((ValueError, InvalidQueryError), "Empty dimension name in ORDER BY"):
3408 list(do_query().order_by(order_by))
3410 for order_by in ("undimension.name", "-undimension.name"):
3411 with self.assertRaisesRegex(
3412 (ValueError, InvalidQueryError),
3413 "(Unknown dimension element 'undimension')|(Unrecognized identifier 'undimension.name')",
3414 ):
3415 list(do_query().order_by(order_by))
3417 for order_by in ("attract", "-attract"):
3418 with self.assertRaisesRegex(
3419 (ValueError, InvalidQueryError),
3420 "(Metadata 'attract' cannot be found in any dimension)|(Unrecognized identifier 'attract')",
3421 ):
3422 list(do_query().order_by(order_by))
3424 with self.assertRaisesRegex(
3425 (ValueError, InvalidQueryError),
3426 "(Metadata 'exposure_time' exists in more than one dimension)"
3427 "|(Ambiguous identifier 'exposure_time' matches multiple fields)",
3428 ):
3429 list(do_query(("exposure", "visit")).order_by("exposure_time"))
3431 with self.assertRaisesRegex(
3432 (ValueError, InvalidQueryError),
3433 r"(Timespan exists in more than one dimension element \(day_obs, exposure, visit\); "
3434 r"qualify timespan with specific dimension name\.)|"
3435 r"(Ambiguous identifier 'timespan' matches multiple fields)",
3436 ):
3437 list(do_query(("exposure", "visit")).order_by("timespan.begin"))
3439 with self.assertRaisesRegex(
3440 (ValueError, InvalidQueryError),
3441 "(Cannot find any temporal dimension element for 'timespan.begin')"
3442 "|(Unrecognized identifier 'timespan')",
3443 ):
3444 list(do_query("tract").order_by("timespan.begin"))
3446 with self.assertRaisesRegex(
3447 (ValueError, InvalidQueryError),
3448 "(Cannot use 'timespan.begin' with non-temporal element)"
3449 "|(Unrecognized field 'timespan' for tract)",
3450 ):
3451 list(do_query("tract").order_by("tract.timespan.begin"))
3453 with self.assertRaisesRegex(
3454 (ValueError, InvalidQueryError),
3455 "(Field 'name' does not exist in 'tract')|(Unrecognized field 'name' for tract.)",
3456 ):
3457 list(do_query("tract").order_by("tract.name"))
3459 with self.assertRaisesRegex(
3460 (ValueError, InvalidQueryError),
3461 r"(Unknown dimension element 'timestamp'; perhaps you meant 'timespan.begin'\?)"
3462 r"|(Unrecognized identifier 'timestamp.begin')",
3463 ):
3464 list(do_query("visit").order_by("timestamp.begin"))
3466 def testQueryDataIdsGovernorExceptions(self):
3467 """Test exceptions raised by queryDataIds() for incorrect governors."""
3468 butler = self.make_butler()
3469 registry = butler.registry
3470 self.load_data(butler, "base.yaml", "datasets.yaml", "spatial.yaml")
3472 def do_query(dimensions, dataId=None, where="", bind=None, **kwargs):
3473 return registry.queryDataIds(dimensions, dataId=dataId, where=where, bind=bind, **kwargs)
3475 Test = namedtuple(
3476 "testQueryDataIdExceptionsTest",
3477 ("dimensions", "dataId", "where", "bind", "kwargs", "exception", "count"),
3478 defaults=(None, None, None, {}, None, 0),
3479 )
3481 test_data = (
3482 Test("tract,visit", count=3),
3483 Test("tract,visit", kwargs={"instrument": "Cam1", "skymap": "SkyMap1"}, count=3),
3484 Test("tract,visit", kwargs={"instrument": "Cam2", "skymap": "SkyMap1"}, count=0),
3485 Test("tract,visit", dataId={"instrument": "Cam1", "skymap": "SkyMap1"}, count=3),
3486 Test("tract,visit", dataId={"instrument": "Cam1", "skymap": "SkyMap2"}, count=0),
3487 Test("tract,visit", where="instrument='Cam1' AND skymap='SkyMap1'", count=3),
3488 Test("tract,visit", where="instrument='Cam1' AND skymap='SkyMap5'", count=0),
3489 Test(
3490 "tract,visit",
3491 where="instrument=:cam AND skymap=:map",
3492 bind={"cam": "Cam1", "map": "SkyMap1"},
3493 count=3,
3494 ),
3495 Test(
3496 "tract,visit",
3497 where="instrument=:cam AND skymap=:map",
3498 bind={"cam": "Cam", "map": "SkyMap"},
3499 count=0,
3500 ),
3501 )
3503 for test in test_data:
3504 print(test)
3505 dimensions = test.dimensions.split(",")
3506 if test.exception: 3506 ↛ 3507line 3506 didn't jump to line 3507 because the condition on line 3506 was never true
3507 with self.assertRaises(test.exception):
3508 with ExitStack() as stack:
3509 if test.exception == DataIdValueError:
3510 stack.enter_context(self.assertWarns(FutureWarning))
3511 do_query(dimensions, test.dataId, test.where, bind=test.bind, **test.kwargs).count()
3512 else:
3513 query = do_query(dimensions, test.dataId, test.where, bind=test.bind, **test.kwargs)
3514 print(list(query))
3515 self.assertEqual(query.count(discard=True), test.count)
3517 # and materialize
3518 if test.exception: 3518 ↛ 3519line 3518 didn't jump to line 3519 because the condition on line 3518 was never true
3519 with self.assertRaises(test.exception):
3520 with ExitStack() as stack:
3521 if test.exception == DataIdValueError:
3522 stack.enter_context(self.assertWarns(FutureWarning))
3523 query = do_query(dimensions, test.dataId, test.where, bind=test.bind, **test.kwargs)
3524 with query.materialize() as materialized:
3525 materialized.count(discard=True)
3526 else:
3527 query = do_query(dimensions, test.dataId, test.where, bind=test.bind, **test.kwargs)
3528 with query.materialize() as materialized:
3529 self.assertEqual(materialized.count(discard=True), test.count)
3531 def testQueryDimensionRecordsOrderBy(self):
3532 """Test order_by and limit on result returned by
3533 queryDimensionRecords().
3534 """
3535 butler = self.make_butler()
3536 registry = butler.registry
3537 self.load_data(butler, "base.yaml", "datasets.yaml", "spatial.yaml")
3539 def do_query(element, datasets=None, collections=None):
3540 return registry.queryDimensionRecords(
3541 element, instrument="Cam1", datasets=datasets, collections=collections
3542 )
3544 query = do_query("detector")
3545 self.assertEqual(len(list(query)), 4)
3547 Test = namedtuple(
3548 "testQueryDataIdsOrderByTest",
3549 ("element", "order_by", "result", "limit", "datasets", "collections"),
3550 defaults=(None, None, None),
3551 )
3553 test_data = [
3554 Test("detector", "detector", (1, 2, 3, 4)),
3555 Test("detector", "-detector", (4, 3, 2, 1)),
3556 Test("detector", "raft,-name_in_raft", (2, 1, 4, 3)),
3557 Test("detector", "-detector.purpose", (4,), limit=(1,)),
3558 Test("visit", "visit", (1, 2)),
3559 Test("visit", "-visit.id", (2, 1)),
3560 Test("visit", "zenith_angle", (1, 2)),
3561 Test("visit", "-visit.name", (2, 1)),
3562 Test("visit", "day_obs,-visit.timespan.begin", (2, 1)),
3563 ]
3565 def do_test(test: Test):
3566 order_by = test.order_by.split(",")
3567 query = do_query(test.element).order_by(*order_by)
3568 if test.limit is not None:
3569 query = query.limit(*test.limit)
3570 dataIds = tuple(rec.id for rec in query)
3571 self.assertEqual(dataIds, test.result)
3573 for test in test_data:
3574 do_test(test)
3576 # errors in a name
3577 for order_by in ("", "-"):
3578 with self.assertRaisesRegex(
3579 (ValueError, InvalidQueryError),
3580 "(Empty dimension name in ORDER BY)|(Unrecognized identifier)",
3581 ):
3582 list(do_query("detector").order_by(order_by))
3584 for order_by in ("undimension.name", "-undimension.name"):
3585 with self.assertRaisesRegex(
3586 (ValueError, InvalidQueryError),
3587 "(Element name mismatch: 'undimension')|(Unrecognized identifier)",
3588 ):
3589 list(do_query("detector").order_by(order_by))
3591 for order_by in ("attract", "-attract"):
3592 with self.assertRaisesRegex(
3593 (ValueError, InvalidQueryError),
3594 "(Field 'attract' does not exist in 'detector'.)|(Unrecognized identifier)",
3595 ):
3596 list(do_query("detector").order_by(order_by))
3598 for order_by in ("timestamp.begin", "-timestamp.begin"):
3599 with self.assertRaisesRegex(
3600 (ValueError, InvalidQueryError),
3601 r"(Element name mismatch: 'timestamp' instead of 'visit'; "
3602 r"perhaps you meant 'timespan.begin'\?)"
3603 r"|(Unrecognized identifier)",
3604 ):
3605 list(do_query("visit").order_by(order_by))
3607 def testQueryDimensionRecordsExceptions(self):
3608 """Test exceptions raised by queryDimensionRecords()."""
3609 butler = self.make_butler()
3610 registry = butler.registry
3611 self.load_data(butler, "base.yaml", "datasets.yaml", "spatial.yaml")
3613 result = registry.queryDimensionRecords("detector")
3614 self.assertEqual(result.count(), 4)
3615 result = registry.queryDimensionRecords("detector", instrument="Cam1")
3616 self.assertEqual(result.count(), 4)
3617 result = registry.queryDimensionRecords("detector", dataId={"instrument": "Cam1"})
3618 self.assertEqual(result.count(), 4)
3620 # Test that values specified in kwargs override those specified in
3621 # dataId.
3622 result = registry.queryDimensionRecords(
3623 "detector", dataId={"instrument": "NotCam1"}, instrument="Cam1"
3624 )
3625 self.assertEqual(result.count(), 4)
3627 result = registry.queryDimensionRecords("detector", where="instrument='Cam1'")
3628 self.assertEqual(result.count(), 4)
3629 result = registry.queryDimensionRecords("detector", where="instrument=:instr", bind={"instr": "Cam1"})
3630 self.assertTrue(result.any())
3631 self.assertEqual(result.count(), 4)
3633 def testDatasetConstrainedDimensionRecordQueries(self):
3634 """Test that queryDimensionRecords works even when given a dataset
3635 constraint whose dimensions extend beyond the requested dimension
3636 element's.
3637 """
3638 butler = self.make_butler()
3639 registry = butler.registry
3640 self.load_data(butler, "base.yaml", "datasets.yaml")
3641 # Query for physical_filter dimension records, using a dataset that
3642 # has both physical_filter and dataset dimensions.
3643 records = registry.queryDimensionRecords(
3644 "physical_filter",
3645 datasets=["flat"],
3646 collections="imported_r",
3647 )
3648 self.assertEqual({record.name for record in records}, {"Cam1-R1", "Cam1-R2"})
3649 # Trying to constrain by all dataset types is an error.
3650 with self.assertRaises(TypeError):
3651 list(registry.queryDimensionRecords("physical_filter", datasets=..., collections="imported_r"))
3653 def testSkyPixDatasetQueries(self):
3654 """Test that we can build queries involving skypix dimensions as long
3655 as a dataset type that uses those dimensions is included.
3656 """
3657 butler = self.make_butler()
3658 registry = butler.registry
3659 self.load_data(butler, "base.yaml")
3660 dataset_type = DatasetType(
3661 "a", dimensions=["htm7", "instrument"], universe=registry.dimensions, storageClass="int"
3662 )
3663 registry.registerDatasetType(dataset_type)
3664 run = "r"
3665 registry.registerRun(run)
3666 # First try queries where there are no datasets; the concern is whether
3667 # we can even build and execute these queries without raising, even
3668 # when "doomed" query shortcuts are in play.
3669 self.assertFalse(
3670 list(registry.queryDataIds(["htm7", "instrument"], datasets=dataset_type, collections=run))
3671 )
3672 self.assertFalse(list(registry.queryDatasets(dataset_type, collections=run)))
3673 # Now add a dataset and see that we can get it back.
3674 htm7 = registry.dimensions.skypix["htm"][7].pixelization
3675 data_id = registry.expandDataId(instrument="Cam1", htm7=htm7.universe()[0][0])
3676 (ref,) = registry.insertDatasets(dataset_type, [data_id], run=run)
3677 self.assertEqual(
3678 set(registry.queryDataIds(["htm7", "instrument"], datasets=dataset_type, collections=run)),
3679 {data_id},
3680 )
3681 self.assertEqual(set(registry.queryDatasets(dataset_type, collections=run)), {ref})
3683 def testDatasetIdFactory(self):
3684 """Simple test for DatasetIdFactory, mostly to catch potential changes
3685 in its API.
3686 """
3687 butler = self.make_butler()
3688 registry = butler.registry
3689 factory = DatasetIdFactory()
3690 dataset_type = DatasetType(
3691 "datasetType",
3692 dimensions=["detector", "instrument"],
3693 universe=registry.dimensions,
3694 storageClass="int",
3695 )
3696 run = "run"
3697 data_id = DataCoordinate.standardize(
3698 instrument="Cam1", detector=1, dimensions=dataset_type.dimensions
3699 )
3701 datasetId = factory.makeDatasetId(run, dataset_type, data_id, DatasetIdGenEnum.UNIQUE)
3702 self.assertIsInstance(datasetId, uuid.UUID)
3703 self.assertEqual(datasetId.version, 7)
3705 datasetId = factory.makeDatasetId(run, dataset_type, data_id, DatasetIdGenEnum.DATAID_TYPE)
3706 self.assertIsInstance(datasetId, uuid.UUID)
3707 self.assertEqual(datasetId.version, 5)
3709 datasetId = factory.makeDatasetId(run, dataset_type, data_id, DatasetIdGenEnum.DATAID_TYPE_RUN)
3710 self.assertIsInstance(datasetId, uuid.UUID)
3711 self.assertEqual(datasetId.version, 5)
3713 def testExposureQueries(self):
3714 """Test query methods using arguments sourced from the exposure log
3715 service.
3717 The most complete test dataset currently available to daf_butler tests
3718 is ci_hsc-subset.yaml export , but that does not have 'exposure'
3719 dimension records. So in this test we need to translate queries that
3720 originally used the exposure dimension to use the (very similar) visit
3721 dimension instead.
3722 """
3723 butler = self.make_butler()
3724 registry = butler.registry
3725 self.load_data(butler, "ci_hsc-subset.yaml")
3726 self.assertEqual(
3727 [
3728 record.id
3729 for record in registry.queryDimensionRecords("visit", instrument="HSC")
3730 .order_by("visit")
3731 .limit(5)
3732 ],
3733 [903334, 903336, 903338, 903342, 903344],
3734 )
3735 self.assertEqual(
3736 [
3737 data_id["visit"]
3738 for data_id in registry.queryDataIds(["visit"], instrument="HSC").order_by("visit").limit(5)
3739 ],
3740 [903334, 903336, 903338, 903342, 903344],
3741 )
3742 self.assertEqual(
3743 [
3744 record.id
3745 for record in registry.queryDimensionRecords("detector", instrument="HSC")
3746 .order_by("full_name")
3747 .limit(5)
3748 ],
3749 [25, 24, 23, 22, 18],
3750 )
3751 self.assertEqual(
3752 [
3753 data_id["detector"]
3754 for data_id in registry.queryDataIds(["detector"], instrument="HSC")
3755 .order_by("full_name")
3756 .limit(5)
3757 ],
3758 [25, 24, 23, 22, 18],
3759 )
3761 def test_long_query_names(self) -> None:
3762 """Test that queries involving very long names are handled correctly.
3764 This is especially important for PostgreSQL, which truncates symbols
3765 longer than 64 chars, but it's worth testing for all DBs.
3766 """
3767 butler = self.make_butler()
3768 registry = butler.registry
3769 name = "abcd" * 17
3770 registry.registerDatasetType(
3771 DatasetType(
3772 name,
3773 dimensions=(),
3774 storageClass="Exposure",
3775 universe=registry.dimensions,
3776 )
3777 )
3778 # Need to search more than one collection actually containing a
3779 # matching dataset to avoid optimizations that sidestep bugs due to
3780 # truncation by making findFirst=True a no-op.
3781 run1 = "run1"
3782 registry.registerRun(run1)
3783 run2 = "run2"
3784 registry.registerRun(run2)
3785 (ref1,) = registry.insertDatasets(name, [DataCoordinate.make_empty(registry.dimensions)], run1)
3786 registry.insertDatasets(name, [DataCoordinate.make_empty(registry.dimensions)], run2)
3787 self.assertEqual(
3788 set(registry.queryDatasets(name, collections=[run1, run2], findFirst=True)),
3789 {ref1},
3790 )
3792 def test_skypix_constraint_queries(self) -> None:
3793 """Test queries spatially constrained by a skypix data ID."""
3794 butler = self.make_butler()
3795 registry = butler.registry
3796 self.load_data(butler, "base.yaml", "spatial.yaml")
3797 patch_regions = {
3798 (data_id["tract"], data_id["patch"]): data_id.region
3799 for data_id in registry.queryDataIds(["patch"]).expanded()
3800 }
3801 skypix_dimension: SkyPixDimension = registry.dimensions["htm11"]
3802 # This check ensures the test doesn't become trivial due to a config
3803 # change; if it does, just pick a different HTML level.
3804 self.assertNotEqual(skypix_dimension, registry.dimensions.commonSkyPix)
3805 # Gather all skypix IDs that definitely overlap at least one of these
3806 # patches.
3807 relevant_skypix_ids = lsst.sphgeom.RangeSet()
3808 for patch_region in patch_regions.values():
3809 relevant_skypix_ids |= skypix_dimension.pixelization.interior(patch_region)
3810 # Look for a "nontrivial" skypix_id that overlaps at least one patch
3811 # and does not overlap at least one other patch.
3812 for skypix_id in itertools.chain.from_iterable( 3812 ↛ 3824line 3812 didn't jump to line 3824 because the loop on line 3812 didn't complete
3813 range(begin, end) for begin, end in relevant_skypix_ids
3814 ):
3815 skypix_region = skypix_dimension.pixelization.pixel(skypix_id)
3816 overlapping_patches = {
3817 patch_key
3818 for patch_key, patch_region in patch_regions.items()
3819 if not patch_region.isDisjointFrom(skypix_region)
3820 }
3821 if overlapping_patches and overlapping_patches != patch_regions.keys(): 3821 ↛ 3812line 3821 didn't jump to line 3812 because the condition on line 3821 was always true
3822 break
3823 else:
3824 raise RuntimeError("Could not find usable skypix ID for this dimension configuration.")
3825 # Test that a three-way join that includes the common skypix system in
3826 # the dimensions doesn't generate redundant join terms in the query.
3827 with self.assertRaises(InvalidQueryError):
3828 set(
3829 registry.queryDataIds(
3830 ["tract", "visit", "htm7"], skymap="SkyMap1", instrument="Cam1"
3831 ).expanded()
3832 )
3834 def test_spatial_constraint_queries(self) -> None:
3835 """Test queries in which one spatial dimension in the constraint (data
3836 ID or ``where`` string) constrains a different spatial dimension in the
3837 query result columns.
3838 """
3839 butler = self.make_butler()
3840 registry = butler.registry
3841 self.load_data(butler, "base.yaml", "spatial.yaml")
3842 patch_regions = {
3843 (data_id["tract"], data_id["patch"]): data_id.region
3844 for data_id in registry.queryDataIds(["patch"]).expanded()
3845 }
3846 observation_regions = {
3847 (data_id["visit"], data_id["detector"]): data_id.region
3848 for data_id in registry.queryDataIds(["visit", "detector"]).expanded()
3849 }
3850 all_combos = {
3851 (patch_key, observation_key)
3852 for patch_key, observation_key in itertools.product(patch_regions, observation_regions)
3853 }
3854 overlapping_combos = {
3855 (patch_key, observation_key)
3856 for patch_key, observation_key in all_combos
3857 if not patch_regions[patch_key].isDisjointFrom(observation_regions[observation_key])
3858 }
3859 # Check a direct spatial join with no constraint first.
3860 self.assertEqual(
3861 {
3862 ((data_id["tract"], data_id["patch"]), (data_id["visit"], data_id["detector"]))
3863 for data_id in registry.queryDataIds(["patch", "visit", "detector"])
3864 },
3865 overlapping_combos,
3866 )
3867 overlaps_by_patch: defaultdict[tuple[int, int], set[tuple[str, str]]] = defaultdict(set)
3868 overlaps_by_observation: defaultdict[tuple[int, int], set[tuple[str, str]]] = defaultdict(set)
3869 for patch_key, observation_key in overlapping_combos:
3870 overlaps_by_patch[patch_key].add(observation_key)
3871 overlaps_by_observation[observation_key].add(patch_key)
3872 # Find patches and observations that overlap at least one of the other
3873 # but not all of the other.
3874 nontrivial_patch = next(
3875 iter(
3876 patch_key
3877 for patch_key, observation_keys in overlaps_by_patch.items()
3878 if observation_keys and observation_keys != observation_regions.keys()
3879 )
3880 )
3881 nontrivial_observation = next(
3882 iter(
3883 observation_key
3884 for observation_key, patch_keys in overlaps_by_observation.items()
3885 if patch_keys and patch_keys != patch_regions.keys()
3886 )
3887 )
3888 # Use the nontrivial patches and observations as constraints on the
3889 # other dimensions in various ways, first via a 'where' expression.
3890 # It's better in general to us 'bind' instead of f-strings, but these
3891 # all integers so there are no quoting concerns.
3892 self.assertEqual(
3893 {
3894 (data_id["visit"], data_id["detector"])
3895 for data_id in registry.queryDataIds(
3896 ["visit", "detector"],
3897 where=f"tract={nontrivial_patch[0]} AND patch={nontrivial_patch[1]}",
3898 skymap="SkyMap1",
3899 )
3900 },
3901 overlaps_by_patch[nontrivial_patch],
3902 )
3903 self.assertEqual(
3904 {
3905 (data_id["tract"], data_id["patch"])
3906 for data_id in registry.queryDataIds(
3907 ["patch"],
3908 where=f"visit={nontrivial_observation[0]} AND detector={nontrivial_observation[1]}",
3909 instrument="Cam1",
3910 )
3911 },
3912 overlaps_by_observation[nontrivial_observation],
3913 )
3914 # and then via the dataId argument.
3915 self.assertEqual(
3916 {
3917 (data_id["visit"], data_id["detector"])
3918 for data_id in registry.queryDataIds(
3919 ["visit", "detector"],
3920 dataId={
3921 "tract": nontrivial_patch[0],
3922 "patch": nontrivial_patch[1],
3923 },
3924 skymap="SkyMap1",
3925 )
3926 },
3927 overlaps_by_patch[nontrivial_patch],
3928 )
3929 self.assertEqual(
3930 {
3931 (data_id["tract"], data_id["patch"])
3932 for data_id in registry.queryDataIds(
3933 ["patch"],
3934 dataId={
3935 "visit": nontrivial_observation[0],
3936 "detector": nontrivial_observation[1],
3937 },
3938 instrument="Cam1",
3939 )
3940 },
3941 overlaps_by_observation[nontrivial_observation],
3942 )
3944 def test_query_empty_collections(self) -> None:
3945 """Test for registry query methods with empty collections. The methods
3946 should return empty result set (or None when applicable) and provide
3947 "doomed" diagnostics.
3948 """
3949 butler = self.make_butler()
3950 registry = butler.registry
3951 self.load_data(butler, "base.yaml", "datasets.yaml")
3953 # Tests for registry.findDataset()
3954 with self.assertRaises(NoDefaultCollectionError):
3955 registry.findDataset("bias", instrument="Cam1", detector=1)
3956 self.assertIsNotNone(registry.findDataset("bias", instrument="Cam1", detector=1, collections=...))
3957 self.assertIsNone(registry.findDataset("bias", instrument="Cam1", detector=1, collections=[]))
3959 # Tests for registry.queryDatasets()
3960 with self.assertRaises(NoDefaultCollectionError):
3961 registry.queryDatasets("bias")
3962 self.assertTrue(list(registry.queryDatasets("bias", collections=...)))
3964 result = registry.queryDatasets("bias", collections=[])
3965 self.assertEqual(len(list(result)), 0)
3966 messages = list(result.explain_no_results())
3967 self.assertTrue(messages)
3968 self.assertTrue(any("because collection list is empty" in message for message in messages))
3970 # Tests for registry.queryDataIds()
3971 with self.assertRaises(NoDefaultCollectionError):
3972 registry.queryDataIds("detector", datasets="bias")
3973 self.assertTrue(list(registry.queryDataIds("detector", datasets="bias", collections=...)))
3975 result = registry.queryDataIds("detector", datasets="bias", collections=[])
3976 self.assertEqual(len(list(result)), 0)
3977 messages = list(result.explain_no_results())
3978 self.assertTrue(messages)
3979 self.assertTrue(any("because collection list is empty" in message for message in messages))
3981 # Tests for registry.queryDimensionRecords()
3982 with self.assertRaises(NoDefaultCollectionError):
3983 registry.queryDimensionRecords("detector", datasets="bias")
3984 self.assertTrue(list(registry.queryDimensionRecords("detector", datasets="bias", collections=...)))
3986 result = registry.queryDimensionRecords("detector", datasets="bias", collections=[])
3987 self.assertEqual(len(list(result)), 0)
3988 messages = list(result.explain_no_results())
3989 self.assertTrue(messages)
3990 self.assertTrue(any("because collection list is empty" in message for message in messages))
3992 def test_dataset_followup_spatial_joins(self) -> None:
3993 """Test queryDataIds(...).findRelatedDatasets(...) where a spatial join
3994 is involved.
3995 """
3996 butler = self.make_butler()
3997 registry = butler.registry
3998 self.load_data(butler, "base.yaml", "spatial.yaml")
3999 pvi_dataset_type = DatasetType(
4000 "pvi", {"visit", "detector"}, storageClass="StructuredDataDict", universe=registry.dimensions
4001 )
4002 registry.registerDatasetType(pvi_dataset_type)
4003 collection = "datasets"
4004 registry.registerRun(collection)
4005 (pvi1,) = registry.insertDatasets(
4006 pvi_dataset_type, [{"instrument": "Cam1", "visit": 1, "detector": 1}], run=collection
4007 )
4008 (pvi2,) = registry.insertDatasets(
4009 pvi_dataset_type, [{"instrument": "Cam1", "visit": 1, "detector": 2}], run=collection
4010 )
4011 (pvi3,) = registry.insertDatasets(
4012 pvi_dataset_type, [{"instrument": "Cam1", "visit": 1, "detector": 3}], run=collection
4013 )
4014 self.assertEqual(
4015 set(
4016 registry.queryDataIds(["patch"], skymap="SkyMap1", tract=0)
4017 .expanded()
4018 .findRelatedDatasets("pvi", [collection])
4019 ),
4020 {
4021 (registry.expandDataId(skymap="SkyMap1", tract=0, patch=0), pvi1),
4022 (registry.expandDataId(skymap="SkyMap1", tract=0, patch=0), pvi2),
4023 (registry.expandDataId(skymap="SkyMap1", tract=0, patch=1), pvi2),
4024 (registry.expandDataId(skymap="SkyMap1", tract=0, patch=2), pvi1),
4025 (registry.expandDataId(skymap="SkyMap1", tract=0, patch=2), pvi2),
4026 (registry.expandDataId(skymap="SkyMap1", tract=0, patch=2), pvi3),
4027 (registry.expandDataId(skymap="SkyMap1", tract=0, patch=3), pvi2),
4028 (registry.expandDataId(skymap="SkyMap1", tract=0, patch=4), pvi3),
4029 },
4030 )
4032 def test_expanded_data_id_queries(self) -> None:
4033 """Tests for basic functionality of expanded() on queryDataIds and
4034 queryDatasets.
4035 """
4036 butler = self.make_butler()
4037 registry = butler.registry
4038 self.load_data(butler, "base.yaml", "spatial.yaml", "datasets.yaml")
4040 result_obj = (
4041 registry.queryDataIds(["visit"], where="instrument = 'Cam1' and (visit.id = 1 or visit.id = 2)")
4042 .expanded()
4043 .order_by("visit.id")
4044 )
4045 self.assertTrue(result_obj.hasRecords())
4046 visits = list(result_obj)
4047 self.assertEqual(len(visits), 2)
4049 self.assertEqual(visits[0]["visit"], 1)
4050 self.assertEqual(visits[1]["visit"], 2)
4051 self.assertEqual(visits[0].records["visit"].exposure_time, 60.0)
4052 self.assertEqual(visits[1].records["visit"].exposure_time, 45.0)
4053 # physical_filter is a "cacheable" dimension, so its records are loaded
4054 # from local cache rather than being part of the DB rows.
4055 self.assertEqual(visits[0].records["physical_filter"].band, "g")
4056 self.assertEqual(visits[1].records["physical_filter"].band, "r")
4058 # Make sure that we can fetch nulls in dimension records
4059 registry.insertDimensionData(
4060 "detector",
4061 {
4062 "instrument": "Cam1",
4063 "id": 5,
4064 "raft": "Z",
4065 "name_in_raft": "z",
4066 "full_name": "Zz",
4067 "purpose": None,
4068 },
4069 )
4070 detectors = list(
4071 registry.queryDataIds("detector", dataId={"instrument": "Cam1", "detector": 5}).expanded()
4072 )
4073 self.assertIsNone(detectors[0].records["detector"].purpose)
4075 datasets_query = registry.queryDatasets(
4076 "flat", collections="imported_g", where="instrument = 'Cam1' and detector <= 3"
4077 ).expanded()
4078 datasets = list(datasets_query)
4079 datasets.sort(key=lambda ref: ref.dataId["detector"])
4080 self.assertEqual(len(datasets), 2)
4081 self.assertEqual(datasets[0].id, uuid.UUID("60c8a65c-7290-4c38-b1de-e3b1cdcf872d"))
4082 self.assertEqual(datasets[1].id, uuid.UUID("84239e7f-c41f-46d5-97b9-a27976b98ceb"))
4083 # All of the dimensions for flat are "cached" dimensions.
4084 self.assertEqual(datasets[0].dataId.records["detector"].full_name, "Ab")
4085 self.assertEqual(datasets[1].dataId.records["detector"].full_name, "Ba")
4086 self.assertEqual(datasets[0].dataId.records["instrument"].visit_system, 1)
4087 assert isinstance(datasets_query, ParentDatasetQueryResults)
4088 data_ids = list(datasets_query.dataIds)
4089 data_ids.sort(key=lambda data_id: data_id["detector"])
4090 self.assertEqual(len(data_ids), 2)
4091 self.assertEqual(data_ids[0].records["detector"].full_name, "Ab")
4092 self.assertEqual(data_ids[1].records["detector"].full_name, "Ba")
4093 self.assertEqual(data_ids[0].records["instrument"].visit_system, 1)
4095 # None of the datasets in the test data include any uncached
4096 # dimensions, so we have to set one up.
4097 registry.registerDatasetType(DatasetType("test", ["visit"], "int", universe=registry.dimensions))
4098 registry.insertDatasets("test", [{"instrument": "Cam1", "visit": 1}], run="imported_g")
4099 ref = list(registry.queryDatasets("test", collections="imported_g").expanded())[0]
4100 self.assertEqual(ref.dataId.records["visit"].zenith_angle, 5.0)
4101 self.assertEqual(ref.dataId.records["physical_filter"].band, "g")
4102 self.assertEqual(
4103 ref.dataId.timespan,
4104 Timespan(
4105 begin=astropy.time.Time("2021-09-09 03:00:00.000000000", scale="tai"),
4106 end=astropy.time.Time("2021-09-09 03:01:00.000000000", scale="tai"),
4107 ),
4108 )
4110 def test_collection_summary(self) -> None:
4111 """Test for collection summary methods."""
4112 butler = self.make_butler()
4113 registry = butler.registry
4114 self.load_data(butler, "base.yaml", "datasets.yaml", "spatial.yaml")
4116 # Add one more dataset type, just for its existence to trigger a bug
4117 # in `associate` (DM-44311).
4118 test_dataset_type = DatasetType("test", ["tract", "patch"], "int", universe=registry.dimensions)
4119 registry.registerDatasetType(test_dataset_type)
4121 # Check for what has been imported.
4122 summary = registry.getCollectionSummary("imported_g")
4123 self.assertEqual(summary.dataset_types.names, {"bias", "flat"})
4124 self.assertEqual(summary.governors, {"instrument": {"Cam1"}})
4126 # Make a tagged collection and associate some datasets.
4127 tagged_coll = "tagged"
4128 registry.registerCollection(tagged_coll, CollectionType.TAGGED)
4129 refsets = registry.queryDatasets(..., collections=["imported_g"]).byParentDatasetType()
4130 for refs in refsets:
4131 registry.associate(tagged_coll, refs)
4133 # Summary has to have the same dataset types.
4134 summary = registry.getCollectionSummary(tagged_coll)
4135 self.assertEqual(summary.dataset_types.names, {"bias", "flat"})
4136 self.assertEqual(summary.governors, {"instrument": {"Cam1"}})
4138 # Remove all datasets from the tagged collection.
4139 refs = list(registry.queryDatasets(..., collections=[tagged_coll]))
4140 registry.disassociate(tagged_coll, refs)
4142 # Summaries should not have changed.
4143 summary = registry.getCollectionSummary(tagged_coll)
4144 self.assertEqual(summary.dataset_types.names, {"bias", "flat"})
4145 self.assertEqual(summary.governors, {"instrument": {"Cam1"}})
4147 # Cleanup summaries.
4148 registry.refresh_collection_summaries()
4149 summary = registry.getCollectionSummary(tagged_coll)
4150 self.assertFalse(summary.dataset_types.names)
4151 # We do not clean governor summaries yet, but because how the query is
4152 # run, it returns empty governors when collection is missing from
4153 # summaries.
4154 self.assertFalse(summary.governors)
4156 # Add dataset with different governor, this is to test that governors
4157 # are not actually cleaned.
4158 refs = registry.insertDatasets("test", [{"skymap": "SkyMap1", "tract": 0, "patch": 0}], "imported_g")
4159 registry.associate(tagged_coll, refs)
4160 summary = registry.getCollectionSummary(tagged_coll)
4161 self.assertEqual(summary.dataset_types.names, {"test"})
4162 # Note that instrument governor resurrects here, even though there are
4163 # no datasets left with that governor.
4164 self.assertEqual(summary.governors, {"instrument": {"Cam1"}, "skymap": {"SkyMap1"}})
4166 def test_temp_table_config(self) -> None:
4167 config = self.makeRegistryConfig()
4168 config["temporary_tables"] = False
4169 self.assertEqual(config.areTemporaryTablesAllowed, False)
4170 butler = self.make_butler(config)
4171 if not isinstance(butler, DirectButler): 4171 ↛ 4172line 4171 didn't jump to line 4172 because the condition on line 4171 was never true
4172 raise unittest.SkipTest("Test only makes sense for registry with direct database connection.")
4173 self.assertEqual(butler._registry._db.supports_temporary_tables, False)
4174 with self.assertRaisesRegex(ReadOnlyDatabaseError, "temporary tables"):
4175 with butler._registry._db.temporary_table(...):
4176 pass