Coverage for python/lsst/resources/_resourcePath.py: 92%
583 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-08-17 20:47 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-08-17 20:47 +0000
1# This file is part of lsst-resources.
2#
3# Developed for the LSST Data Management System.
4# This product includes software developed by the LSST Project
5# (https://www.lsst.org).
6# See the COPYRIGHT file at the top-level directory of this distribution
7# for details of code ownership.
8#
9# Use of this source code is governed by a 3-clause BSD-style
10# license that can be found in the LICENSE file.
12from __future__ import annotations
14__all__ = ("ResourceInfo", "ResourcePath", "ResourcePathExpression")
16import concurrent.futures
17import contextlib
18import copy
19import dataclasses
20import datetime
21import io
22import locale
23import logging
24import os
25import posixpath
26import re
27import sys
28import urllib.parse
29from collections import defaultdict
30from pathlib import Path, PurePath, PurePosixPath
31from random import Random
32from typing import TYPE_CHECKING, TypeAlias
34try:
35 import fsspec
36 from fsspec.spec import AbstractFileSystem
37except ImportError:
38 # Hidden from type checkers so that the names above keep the types they
39 # have when fsspec is installed.
40 if not TYPE_CHECKING:
41 fsspec = None
42 AbstractFileSystem = type
44from collections.abc import Generator, Iterable, Iterator
45from typing import Any, Literal, NamedTuple, overload
47from ._resourceHandles._baseResourceHandle import ResourceHandleProtocol
48from .utils import _get_num_workers, get_tempdir
50if TYPE_CHECKING:
51 from .utils import TransactionProtocol
54log = logging.getLogger(__name__)
56# Regex for looking for URI escapes
57ESCAPES_RE = re.compile(r"%[A-F0-9]{2}")
59# Precomputed escaped hash
60ESCAPED_HASH = urllib.parse.quote("#")
63class MBulkResult(NamedTuple):
64 """Report on a bulk operation."""
66 success: bool
67 exception: Exception | None
70_EXECUTOR_TYPE: TypeAlias = type[
71 concurrent.futures.ThreadPoolExecutor | concurrent.futures.ProcessPoolExecutor
72]
74# Cache value for executor class so as not to issue warning multiple
75# times but still allow tests to override the value.
76_POOL_EXECUTOR_CLASS: _EXECUTOR_TYPE | None = None
79def _get_executor_class() -> _EXECUTOR_TYPE:
80 """Return the executor class used for parallelized execution.
82 Returns
83 -------
84 cls : `concurrent.futures.Executor`
85 The ``Executor`` class. Default is
86 `concurrent.futures.ThreadPoolExecutor`. Can be set explicitly by
87 setting the ``$LSST_RESOURCES_EXECUTOR`` environment variable to
88 "thread" or "process". Returns "thread" pool if the value of the
89 variable is not recognized.
90 """
91 global _POOL_EXECUTOR_CLASS
93 if _POOL_EXECUTOR_CLASS is not None:
94 return _POOL_EXECUTOR_CLASS
96 pool_executor_classes = {
97 "threads": concurrent.futures.ThreadPoolExecutor,
98 "process": concurrent.futures.ProcessPoolExecutor,
99 }
100 default_executor = "threads"
101 external = os.getenv("LSST_RESOURCES_EXECUTOR", default_executor)
102 if not external: 102 ↛ 103line 102 didn't jump to line 103 because the condition on line 102 was never true
103 external = default_executor
104 if external not in pool_executor_classes: 104 ↛ 105line 104 didn't jump to line 105 because the condition on line 104 was never true
105 log.warning(
106 "Unrecognized value of '%s' for LSST_RESOURCES_EXECUTOR env var. Using '%s'",
107 external,
108 default_executor,
109 )
110 external = default_executor
111 _POOL_EXECUTOR_CLASS = pool_executor_classes[external]
112 return _POOL_EXECUTOR_CLASS
115@contextlib.contextmanager
116def _patch_environ(new_values: dict[str, str]) -> Generator[None]:
117 """Patch os.environ temporarily using the supplied values.
119 Parameters
120 ----------
121 new_values : `dict` [ `str`, `str` ]
122 New values to be stored in the environment.
123 """
124 old_values: dict[str, str] = {}
125 for k, v in new_values.items():
126 if k in os.environ: 126 ↛ 127line 126 didn't jump to line 127 because the condition on line 126 was never true
127 old_values[k] = os.environ[k]
128 os.environ[k] = v
130 try:
131 yield
132 finally:
133 for k in new_values:
134 del os.environ[k]
135 if k in old_values: 135 ↛ 136line 135 didn't jump to line 136 because the condition on line 135 was never true
136 os.environ[k] = old_values[k]
139@dataclasses.dataclass(frozen=True)
140class ResourceInfo:
141 """Information about this resource."""
143 uri: str
144 """URI in string form of the resource from which this information is
145 derived.
146 """
147 is_file: bool
148 """Indicate whether the resource is a file or a directory."""
149 size: int
150 """Size of the file in bytes. A directory or a URI that has no concept
151 of size returns 0."""
152 last_modified: datetime.datetime | None
153 """Modification date of the resource, if known."""
154 checksums: dict[str, Any]
155 """Checksums for this file. Supported checksum implementations are
156 backend dependent.
157 """
160class ResourcePath: # numpydoc ignore=PR02
161 """Convenience wrapper around URI parsers.
163 Provides access to URI components and can convert file
164 paths into absolute path URIs. Scheme-less URIs are treated as if
165 they are local file system paths and are converted to absolute URIs.
167 A specialist subclass is created for each supported URI scheme.
169 Parameters
170 ----------
171 uri : `str`, `pathlib.Path`, `urllib.parse.ParseResult`, or `ResourcePath`
172 URI in string form. Can be scheme-less if referring to a relative
173 path or an absolute path on the local file system.
174 root : `str` or `ResourcePath`, optional
175 When fixing up a relative path in a ``file`` scheme or if scheme-less,
176 use this as the root. Must be absolute. If `None` the current
177 working directory will be used. Can be any supported URI scheme.
178 Not used if ``forceAbsolute`` is `False`.
179 forceAbsolute : `bool`, optional
180 If `True`, scheme-less relative URI will be converted to an absolute
181 path using a ``file`` scheme. If `False` scheme-less URI will remain
182 scheme-less and will not be updated to ``file`` or absolute path unless
183 it is already an absolute path, in which case it will be updated to
184 a ``file`` scheme.
185 forceDirectory : `bool` or `None`, optional
186 If `True` forces the URI to end with a separator. If `False` the URI
187 is interpreted as a file-like entity. Default, `None`, is that the
188 given URI is interpreted as a directory if there is a trailing ``/`` or
189 for some schemes the system will check to see if it is a file or a
190 directory.
191 isTemporary : `bool`, optional
192 If `True` indicates that this URI points to a temporary resource.
193 The default is `False`, unless ``uri`` is already a `ResourcePath`
194 instance and ``uri.isTemporary is True``.
196 Notes
197 -----
198 A non-standard URI of the form ``file:dir/file.txt`` is always converted
199 to an absolute ``file`` URI.
200 """
202 _pathLib: type[PurePath] = PurePosixPath
203 """Path library to use for this scheme."""
205 _pathModule = posixpath
206 """Path module to use for this scheme."""
208 transferModes: tuple[str, ...] = ("copy", "auto", "move")
209 """Transfer modes supported by this implementation.
211 Move is special in that it is generally a copy followed by an unlink.
212 Whether that unlink works depends critically on whether the source URI
213 implements unlink. If it does not the move will be reported as a failure.
214 """
216 transferDefault: str = "copy"
217 """Default mode to use for transferring if ``auto`` is specified."""
219 quotePaths = True
220 """True if path-like elements modifying a URI should be quoted.
222 All non-schemeless URIs have to internally use quoted paths. Therefore
223 if a new file name is given (e.g. to updatedFile or join) a decision must
224 be made whether to quote it to be consistent.
225 """
227 isLocal = False
228 """If `True` this URI refers to a local file."""
230 # This is not an ABC with abstract methods because the __new__ being
231 # a factory confuses mypy such that it assumes that every constructor
232 # returns a ResourcePath and then determines that all the abstract methods
233 # are still abstract. If they are not marked abstract but just raise
234 # mypy is fine with it.
236 # mypy is confused without these
237 _uri: urllib.parse.ParseResult
238 isTemporary: bool
239 dirLike: bool | None
240 """Whether the resource looks like a directory resource. `None` means that
241 the status is uncertain."""
243 def __new__(
244 cls,
245 uri: ResourcePathExpression,
246 root: str | ResourcePath | None = None,
247 forceAbsolute: bool = True,
248 forceDirectory: bool | None = None,
249 isTemporary: bool | None = None,
250 ) -> ResourcePath:
251 """Create and return new specialist ResourcePath subclass."""
252 parsed: urllib.parse.ParseResult
253 dirLike: bool | None = forceDirectory
254 subclass: type[ResourcePath] | None = None
256 # Force root to be a ResourcePath -- this simplifies downstream
257 # code.
258 if root is None:
259 root_uri = None
260 elif isinstance(root, str):
261 root_uri = ResourcePath(root, forceDirectory=True, forceAbsolute=True)
262 else:
263 root_uri = root
265 if isinstance(uri, os.PathLike):
266 uri = str(uri)
268 # Record if we need to post process the URI components
269 # or if the instance is already fully configured
270 if isinstance(uri, str):
271 # Since local file names can have special characters in them
272 # we need to quote them for the parser but we can unquote
273 # later. Assume that all other URI schemes are quoted.
274 # Since sometimes people write file:/a/b and not file:///a/b
275 # we should not quote in the explicit case of file:
276 if "://" not in uri and not uri.startswith("file:"):
277 if ESCAPES_RE.search(uri):
278 log.warning("Possible double encoding of %s", uri)
279 else:
280 # Fragments are generally not encoded so we must search
281 # for the fragment boundary ourselves. This is making
282 # an assumption that the filename does not include a "#"
283 # and also that there is no "/" in the fragment itself.
284 to_encode = uri
285 fragment = ""
286 if "#" in uri:
287 dirpos = uri.rfind("/")
288 trailing = uri[dirpos + 1 :]
289 hashpos = trailing.rfind("#")
290 if hashpos != -1:
291 fragment = trailing[hashpos:]
292 to_encode = uri[: dirpos + hashpos + 1]
294 uri = urllib.parse.quote(to_encode) + fragment
296 parsed = urllib.parse.urlparse(uri)
297 elif isinstance(uri, urllib.parse.ParseResult):
298 parsed = copy.copy(uri)
299 # If we are being instantiated with a subclass, rather than
300 # ResourcePath, ensure that that subclass is used directly.
301 # This could lead to inconsistencies if this constructor
302 # is used externally outside of the ResourcePath.replace() method.
303 # S3ResourcePath(urllib.parse.urlparse("file://a/b.txt"))
304 # will be a problem.
305 # This is needed to prevent a schemeless absolute URI become
306 # a file URI unexpectedly when calling updatedFile or
307 # updatedExtension
308 if cls is not ResourcePath:
309 parsed, dirLike = cls._fixDirectorySep(parsed, forceDirectory)
310 subclass = cls
312 elif isinstance(uri, ResourcePath):
313 # Since ResourcePath is immutable we can return the argument
314 # unchanged if it already agrees with forceDirectory, isTemporary,
315 # and forceAbsolute.
316 # We invoke __new__ again with str(self) to add a scheme for
317 # forceAbsolute, but for the others that seems more likely to paper
318 # over logic errors than do something useful, so we just raise.
319 if forceDirectory is not None and uri.dirLike is not None and forceDirectory is not uri.dirLike:
320 # Can not force a file-like URI to become a dir-like one or
321 # vice versa.
322 raise RuntimeError(
323 f"{uri} can not be forced to change directory vs file state when previously declared."
324 )
325 if isTemporary is not None and isTemporary is not uri.isTemporary:
326 raise RuntimeError(
327 f"{uri} is already a {'temporary' if uri.isTemporary else 'permanent'} "
328 f"ResourcePath; cannot make it {'temporary' if isTemporary else 'permanent'}."
329 )
331 if forceAbsolute and not uri.scheme:
332 # Create new absolute from relative.
333 return ResourcePath(
334 str(uri),
335 root=root,
336 forceAbsolute=forceAbsolute,
337 forceDirectory=forceDirectory or uri.dirLike,
338 isTemporary=uri.isTemporary,
339 )
340 elif forceDirectory is not None and uri.dirLike is None:
341 # Clone but with a new dirLike status.
342 return uri.replace(forceDirectory=forceDirectory)
343 return uri
344 else:
345 raise ValueError(
346 f"Supplied URI must be string, Path, ResourcePath, or ParseResult but got '{uri!r}'"
347 )
349 if subclass is None:
350 # Work out the subclass from the URI scheme
351 if not parsed.scheme:
352 # Root may be specified as a ResourcePath that overrides
353 # the schemeless determination.
354 if (
355 root_uri is not None
356 and root_uri.scheme != "file" # file scheme has different code path
357 and not parsed.path.startswith("/") # Not already absolute path
358 ):
359 if root_uri.dirLike is False:
360 raise ValueError(
361 f"Root URI ({root}) was not a directory so can not be joined with"
362 f" path {parsed.path!r}"
363 )
364 # If root is temporary or this schemeless is temporary we
365 # assume this URI is temporary.
366 isTemporary = isTemporary or root_uri.isTemporary
367 joined = root_uri.join(
368 parsed.path, forceDirectory=forceDirectory, isTemporary=isTemporary
369 )
371 # Rather than returning this new ResourcePath directly we
372 # instead extract the path and the scheme and adjust the
373 # URI we were given -- we need to do this to preserve
374 # fragments since join() will drop them.
375 parsed = parsed._replace(scheme=joined.scheme, path=joined.path, netloc=joined.netloc)
376 subclass = type(joined)
378 # Clear the root parameter to indicate that it has
379 # been applied already.
380 root_uri = None
381 else:
382 from .schemeless import SchemelessResourcePath
384 subclass = SchemelessResourcePath
385 elif parsed.scheme == "file":
386 from .file import FileResourcePath
388 subclass = FileResourcePath
389 elif parsed.scheme == "s3":
390 from .s3 import S3ResourcePath
392 subclass = S3ResourcePath
393 elif parsed.scheme.startswith("http"):
394 from .http import HttpResourcePath
396 subclass = HttpResourcePath
397 elif parsed.scheme in {"dav", "davs"}:
398 from .dav import DavResourcePath
400 subclass = DavResourcePath
401 elif parsed.scheme == "gs":
402 from .gs import GSResourcePath
404 subclass = GSResourcePath
405 elif parsed.scheme == "resource":
406 # Rules for scheme names disallow pkg_resource
407 from .packageresource import PackageResourcePath
409 subclass = PackageResourcePath
410 elif parsed.scheme == "mem":
411 # in-memory datastore object
412 from .mem import InMemoryResourcePath
414 subclass = InMemoryResourcePath
415 elif parsed.scheme == "eups":
416 # EUPS package root.
417 from .eups import EupsResourcePath
419 subclass = EupsResourcePath
420 elif parsed.scheme == "remote-test":
421 # EUPS package root.
422 from .remote_test import RemoteTestResourcePath
424 subclass = RemoteTestResourcePath
425 else:
426 raise NotImplementedError(
427 f"No URI support for scheme: '{parsed.scheme}' in {parsed.geturl()}"
428 )
430 parsed, dirLike = subclass._fixupPathUri(
431 parsed, root=root_uri, forceAbsolute=forceAbsolute, forceDirectory=forceDirectory
432 )
434 # It is possible for the class to change from schemeless
435 # to file or eups so handle that
436 if parsed.scheme == "file":
437 from .file import FileResourcePath
439 subclass = FileResourcePath
440 elif parsed.scheme == "eups":
441 from .eups import EupsResourcePath
443 subclass = EupsResourcePath
445 # Now create an instance of the correct subclass and set the
446 # attributes directly
447 self = object.__new__(subclass)
448 self._uri = parsed
449 self.dirLike = dirLike
450 if isTemporary is None:
451 isTemporary = False
452 self.isTemporary = isTemporary
453 self._set_proxy()
454 return self
456 def _set_proxy(self) -> None:
457 """Calculate internal proxy for externally visible resource path."""
458 pass
460 @property
461 def scheme(self) -> str:
462 """Return the URI scheme.
464 Notes
465 -----
466 (``://`` is not part of the scheme).
467 """
468 return self._uri.scheme
470 @property
471 def netloc(self) -> str:
472 """Return the URI network location."""
473 return self._uri.netloc
475 @property
476 def path(self) -> str:
477 """Return the path component of the URI."""
478 return self._uri.path
480 @property
481 def unquoted_path(self) -> str:
482 """Return path component of the URI with any URI quoting reversed."""
483 return urllib.parse.unquote(self._uri.path)
485 @property
486 def ospath(self) -> str:
487 """Return the path component of the URI localized to current OS."""
488 raise AttributeError(f"Non-file URI ({self}) has no local OS path.")
490 @property
491 def relativeToPathRoot(self) -> str:
492 """Return path relative to network location.
494 This is the path property with posix separator stripped
495 from the left hand side of the path.
497 Always unquotes.
498 """
499 relToRoot = self.path.lstrip("/")
500 if relToRoot == "":
501 return "./"
502 return urllib.parse.unquote(relToRoot)
504 @property
505 def is_root(self) -> bool:
506 """Return whether this URI points to the root of the network location.
508 This means that the path components refers to the top level.
509 """
510 relpath = self.relativeToPathRoot
511 if relpath == "./":
512 return True
513 return False
515 @property
516 def fragment(self) -> str:
517 """Return the fragment component of the URI. May be quoted."""
518 return self._uri.fragment
520 @property
521 def unquoted_fragment(self) -> str:
522 """Return unquoted fragment."""
523 return urllib.parse.unquote(self.fragment)
525 @property
526 def params(self) -> str:
527 """Return any parameters included in the URI."""
528 return self._uri.params
530 @property
531 def query(self) -> str:
532 """Return any query strings included in the URI."""
533 return self._uri.query
535 def geturl(self) -> str:
536 """Return the URI in string form.
538 Returns
539 -------
540 url : `str`
541 String form of URI.
542 """
543 return self._uri.geturl()
545 def to_fsspec(self) -> tuple[AbstractFileSystem, str]:
546 """Return an abstract file system and path that can be used by fsspec.
548 Returns
549 -------
550 fs : `fsspec.spec.AbstractFileSystem`
551 A file system object suitable for use with the returned path.
552 path : `str`
553 A path that can be opened by the file system object.
554 """
555 if fsspec is None:
556 raise ImportError("fsspec is not available")
557 # By default give the URL to fsspec and hope.
558 return fsspec.url_to_fs(self.geturl())
560 def root_uri(self) -> ResourcePath:
561 """Return the base root URI.
563 Returns
564 -------
565 uri : `ResourcePath`
566 Root URI.
567 """
568 return self.replace(path="", query="", fragment="", params="", forceDirectory=True)
570 def split(self) -> tuple[ResourcePath, str]:
571 """Split URI into head and tail.
573 Returns
574 -------
575 head: `ResourcePath`
576 Everything leading up to tail, expanded and normalized as per
577 ResourcePath rules.
578 tail : `str`
579 Last path component. Tail will be empty if path ends on a
580 separator or if the URI is known to be associated with a directory.
581 Tail will never contain separators. It will be unquoted.
583 Notes
584 -----
585 Equivalent to `os.path.split` where head preserves the URI
586 components. In some cases this method can result in a file system
587 check to verify whether the URI is a directory or not (only if
588 ``forceDirectory`` was `None` during construction). For a scheme-less
589 URI this can mean that the result might change depending on current
590 working directory.
591 """
592 if self.isdir():
593 # This is known to be a directory so must return itself and
594 # the empty string.
595 return self, ""
597 head, tail = self._pathModule.split(self.path)
598 headuri = self._uri._replace(path=head, fragment="", query="", params="")
600 # The file part should never include quoted metacharacters
601 tail = urllib.parse.unquote(tail)
603 # Schemeless is special in that it can be a relative path.
604 # We need to ensure that it stays that way. All other URIs will
605 # be absolute already.
606 forceAbsolute = self.isabs()
607 return ResourcePath(headuri, forceDirectory=True, forceAbsolute=forceAbsolute), tail
609 def basename(self) -> str:
610 """Return the base name, last element of path, of the URI.
612 Returns
613 -------
614 tail : `str`
615 Last part of the path attribute. Trail will be empty if path ends
616 on a separator.
618 Notes
619 -----
620 If URI ends on a slash returns an empty string. This is the second
621 element returned by `split()`.
623 Equivalent of `os.path.basename`.
624 """
625 return self.split()[1]
627 def dirname(self) -> ResourcePath:
628 """Return the directory component of the path as a new `ResourcePath`.
630 Returns
631 -------
632 head : `ResourcePath`
633 Everything except the tail of path attribute, expanded and
634 normalized as per ResourcePath rules.
636 Notes
637 -----
638 Equivalent of `os.path.dirname`. If this is a directory URI it will
639 be returned unchanged. If the parent directory is always required
640 use `parent`.
641 """
642 return self.split()[0]
644 def parent(self) -> ResourcePath:
645 """Return a `ResourcePath` of the parent directory.
647 Returns
648 -------
649 head : `ResourcePath`
650 Everything except the tail of path attribute, expanded and
651 normalized as per `ResourcePath` rules.
653 Notes
654 -----
655 For a file-like URI this will be the same as calling `dirname`.
656 For a directory-like URI this will always return the parent directory
657 whereas `dirname()` will return the original URI. This is consistent
658 with `os.path.dirname` compared to the `pathlib.Path` property
659 ``parent``.
660 """
661 if self.dirLike is False:
662 # os.path.split() is slightly faster than calling Path().parent.
663 return self.dirname()
664 # When self is dir-like, returns its parent directory,
665 # regardless of the presence of a trailing separator
666 originalPath = self._pathLib(self.path)
667 parentPath = originalPath.parent
668 return self.replace(path=str(parentPath), forceDirectory=True, fragment="", query="", params="")
670 def replace(
671 self, forceDirectory: bool | None = None, isTemporary: bool = False, **kwargs: Any
672 ) -> ResourcePath:
673 """Return new `ResourcePath` with specified components replaced.
675 Parameters
676 ----------
677 forceDirectory : `bool` or `None`, optional
678 Parameter passed to ResourcePath constructor to force this
679 new URI to be dir-like or file-like.
680 isTemporary : `bool`, optional
681 Indicate that the resulting URI is temporary resource.
682 **kwargs
683 Components of a `urllib.parse.ParseResult` that should be
684 modified for the newly-created `ResourcePath`.
686 Returns
687 -------
688 new : `ResourcePath`
689 New `ResourcePath` object with updated values.
691 Notes
692 -----
693 Does not, for now, allow a change in URI scheme.
694 """
695 # Disallow a change in scheme
696 if "scheme" in kwargs:
697 raise ValueError(f"Can not use replace() method to change URI scheme for {self}")
698 result = self.__class__(
699 self._uri._replace(**kwargs), forceDirectory=forceDirectory, isTemporary=isTemporary
700 )
701 result._copy_extra_attributes(self)
702 return result
704 def updatedFile(self, newfile: str) -> ResourcePath:
705 """Return new URI with an updated final component of the path.
707 Parameters
708 ----------
709 newfile : `str`
710 File name with no path component.
712 Returns
713 -------
714 updated : `ResourcePath`
715 Updated `ResourcePath` with new updated final component.
717 Notes
718 -----
719 Forces the ``ResourcePath.dirLike`` attribute to be false. The new file
720 path will be quoted if necessary. If the current URI is known to
721 refer to a directory, the new file will be joined to the current file.
722 It is recommended that this behavior no longer be used and a call
723 to `isdir` by the caller should be used to decide whether to join or
724 replace. In the future this method may be modified to always replace
725 the final element of the path.
726 """
727 if self.dirLike:
728 return self.join(newfile, forceDirectory=False)
729 return self.parent().join(newfile, forceDirectory=False)
731 def updatedExtension(self, ext: str | None) -> ResourcePath:
732 """Return a new `ResourcePath` with updated file extension.
734 All file extensions are replaced.
736 Parameters
737 ----------
738 ext : `str` or `None`
739 New extension. If an empty string is given any extension will
740 be removed. If `None` is given there will be no change.
742 Returns
743 -------
744 updated : `ResourcePath`
745 URI with the specified extension. Can return itself if
746 no extension was specified.
747 """
748 if ext is None:
749 return self
751 # Get the extension
752 current = self.getExtension()
754 # Nothing to do if the extension already matches
755 if current == ext:
756 return self
758 # Remove the current extension from the path
759 # .fits.gz counts as one extension do not use os.path.splitext
760 path = self.path
761 if current:
762 path = path.removesuffix(current)
764 # Ensure that we have a leading "." on file extension (and we do not
765 # try to modify the empty string)
766 if ext and not ext.startswith("."):
767 ext = "." + ext
769 return self.replace(path=path + ext, forceDirectory=False)
771 def getExtension(self) -> str:
772 """Return the extension(s) associated with this URI path.
774 Returns
775 -------
776 ext : `str`
777 The file extension (including the ``.``). Can be empty string
778 if there is no file extension. Usually returns only the last
779 file extension unless there is a special extension modifier
780 indicating file compression, in which case the combined
781 extension (e.g. ``.fits.gz``) will be returned.
783 Notes
784 -----
785 Does not distinguish between file and directory URIs when determining
786 a suffix. An extension is only determined from the final component
787 of the path.
788 """
789 special = {".gz", ".bz2", ".xz", ".fz"}
791 # path lib will ignore any "." in directories.
792 # path lib works well:
793 # extensions = self._pathLib(self.path).suffixes
794 # But the constructor is slow. Therefore write our own implementation.
795 # Strip trailing separator if present, do not care if this is a
796 # directory or not.
797 parts = self.path.rstrip("/").rsplit(self._pathModule.sep, 1)
798 _, *extensions = parts[-1].split(".")
800 if not extensions:
801 return ""
802 extensions = ["." + x for x in extensions]
804 ext = extensions.pop()
806 # Multiple extensions, decide whether to include the final two
807 if extensions and ext in special:
808 ext = f"{extensions[-1]}{ext}"
810 return ext
812 def join(
813 self, path: str | ResourcePath, isTemporary: bool | None = None, forceDirectory: bool | None = None
814 ) -> ResourcePath:
815 """Return new `ResourcePath` with additional path components.
817 Parameters
818 ----------
819 path : `str`, `ResourcePath`
820 Additional file components to append to the current URI. Will be
821 quoted depending on the associated URI scheme. If the path looks
822 like a URI referring to an absolute location, it will be returned
823 directly (matching the behavior of `os.path.join`). It can
824 also be a `ResourcePath`. Fragments are propagated.
825 isTemporary : `bool`, optional
826 Indicate that the resulting URI represents a temporary resource.
827 Default is ``self.isTemporary``.
828 forceDirectory : `bool` or `None`, optional
829 If `True` forces the URI to end with a separator. If `False` the
830 resultant URI is declared to refer to a file. `None` indicates
831 that the file directory status is unknown.
833 Returns
834 -------
835 new : `ResourcePath`
836 New URI with the path appended.
838 Notes
839 -----
840 Schemeless URIs assume local path separator but all other URIs assume
841 POSIX separator if the supplied path has directory structure. It
842 may be this never becomes a problem but datastore templates assume
843 POSIX separator is being used.
845 If an absolute `ResourcePath` is given for ``path`` is is assumed that
846 this should be returned directly. Giving a ``path`` of an absolute
847 scheme-less URI is not allowed for safety reasons as it may indicate
848 a mistake in the calling code.
850 It is an error to attempt to join to something that is known to
851 refer to a file. Use `updatedFile` if the file is to be
852 replaced.
854 If an unquoted ``#`` is included in the path it is assumed to be
855 referring to a fragment and not part of the file name.
857 Raises
858 ------
859 ValueError
860 Raised if the given path object refers to a directory but the
861 ``forceDirectory`` parameter insists the outcome should be a file,
862 and vice versa. Also raised if the URI being joined with is known
863 to refer to a file.
864 RuntimeError
865 Raised if this attempts to join a temporary URI to a non-temporary
866 URI.
867 """
868 if self.dirLike is False:
869 raise ValueError("Can not join a new path component to a file.")
870 if isTemporary is None:
871 isTemporary = self.isTemporary
872 elif not isTemporary and self.isTemporary:
873 raise RuntimeError("Cannot join temporary URI to non-temporary URI.")
874 # If we have a full URI in path we will use it directly
875 # but without forcing to absolute so that we can trap the
876 # expected option of relative path.
877 path_uri = ResourcePath(
878 path, forceAbsolute=False, forceDirectory=forceDirectory, isTemporary=isTemporary
879 )
880 if forceDirectory is not None and path_uri.dirLike is not forceDirectory: 880 ↛ 881line 880 didn't jump to line 881 because the condition on line 880 was never true
881 raise ValueError(
882 "The supplied path URI to join has inconsistent directory state "
883 f"with forceDirectory parameter: {path_uri.dirLike} vs {forceDirectory}"
884 )
885 forceDirectory = path_uri.dirLike
887 if path_uri.isabs():
888 # Absolute URI so return it directly.
889 return path_uri
891 # We want to propagate fragments to the joined path and we rely on
892 # the ResourcePath parser to find these fragments for us even in plain
893 # strings. Must assume there are no `#` characters in filenames.
894 if not isinstance(path, str) or path_uri.fragment:
895 path = path_uri.unquoted_path
897 # Might need to quote the path.
898 if self.quotePaths:
899 path = urllib.parse.quote(path)
901 newpath = self._pathModule.normpath(self._pathModule.join(self.path, path))
903 # normpath can strip trailing / so we force directory if the supplied
904 # path ended with a /
905 has_dir_sep = path.endswith(self._pathModule.sep)
906 if forceDirectory is None and has_dir_sep: 906 ↛ 907line 906 didn't jump to line 907 because the condition on line 906 was never true
907 forceDirectory = True
908 elif forceDirectory is False and has_dir_sep: 908 ↛ 909line 908 didn't jump to line 909 because the condition on line 908 was never true
909 raise ValueError("Path to join has trailing / but is being forced to be a file.")
910 return self.replace(
911 path=newpath,
912 forceDirectory=forceDirectory,
913 isTemporary=isTemporary,
914 fragment=path_uri.fragment,
915 query=path_uri.query,
916 params=path_uri.params,
917 )
919 def relative_to(self, other: ResourcePath, walk_up: bool = False) -> str | None:
920 """Return the relative path from this URI to the other URI.
922 Parameters
923 ----------
924 other : `ResourcePath`
925 URI to use to calculate the relative path. Must be a parent
926 of this URI.
927 walk_up : `bool`, optional
928 Control whether "``..``" can be used to resolve a relative path.
929 Default is `False`. Can not be `True` on Python version 3.11.
931 Returns
932 -------
933 subpath : `str`
934 The sub path of this URI relative to the supplied other URI.
935 Returns `None` if there is no parent child relationship.
936 Scheme and netloc must match.
937 """
938 # Scheme-less self is handled elsewhere.
939 if self.scheme != other.scheme:
940 return None
941 if self.netloc != other.netloc:
942 # Special case for localhost vs empty string.
943 # There can be many variants of localhost.
944 local_netlocs = {"", "localhost", "localhost.localdomain", "127.0.0.1"}
945 if not {self.netloc, other.netloc}.issubset(local_netlocs):
946 return None
948 # Rather than trying to guess a failure reason from the TypeError
949 # explicitly check for python 3.11. Doing this will simplify the
950 # rediscovery of a useless python version check when we set a new
951 # minimum version.
952 kwargs = {}
953 if walk_up:
954 if sys.version_info < (3, 12, 0): 954 ↛ 955line 954 didn't jump to line 955 because the condition on line 954 was never true
955 raise TypeError("walk_up parameter can not be true in python 3.11 and older")
957 kwargs["walk_up"] = True
959 enclosed_path = self._pathLib(self.relativeToPathRoot)
960 parent_path = other.relativeToPathRoot
961 subpath: str | None
962 try:
963 subpath = str(enclosed_path.relative_to(parent_path, **kwargs))
964 except ValueError:
965 subpath = None
966 else:
967 subpath = urllib.parse.unquote(subpath)
968 return subpath
970 def exists(self) -> bool:
971 """Indicate that the resource is available.
973 Returns
974 -------
975 exists : `bool`
976 `True` if the resource exists.
977 """
978 raise NotImplementedError()
980 @classmethod
981 def _group_uris(cls, uris: Iterable[ResourcePath]) -> dict[type[ResourcePath], list[ResourcePath]]:
982 """Group URIs by class/scheme."""
983 grouped: dict[type[ResourcePath], list[ResourcePath]] = defaultdict(list)
984 for uri in uris:
985 grouped[uri.__class__].append(uri)
986 return grouped
988 @classmethod
989 def mexists(
990 cls, uris: Iterable[ResourcePath], *, num_workers: int | None = None
991 ) -> dict[ResourcePath, bool]:
992 """Check for existence of multiple URIs at once.
994 Parameters
995 ----------
996 uris : iterable of `ResourcePath`
997 The URIs to test.
998 num_workers : `int` or `None`, optional
999 The number of parallel workers to use when checking for existence
1000 If `None`, the default value will be taken from the environment.
1001 If this number is higher than the default and a thread pool is
1002 used, there may not be enough cached connections available.
1004 Returns
1005 -------
1006 existence : `dict` of [`ResourcePath`, `bool`]
1007 Mapping of original URI to boolean indicating existence.
1008 """
1009 existence: dict[ResourcePath, bool] = {}
1010 for uri_class, group in cls._group_uris(uris).items():
1011 existence.update(uri_class._mexists(group, num_workers=num_workers))
1013 return existence
1015 @classmethod
1016 def _mexists(
1017 cls, uris: Iterable[ResourcePath], *, num_workers: int | None = None
1018 ) -> dict[ResourcePath, bool]:
1019 """Check for existence of multiple URIs at once.
1021 Implementation helper method for `mexists`.
1024 Parameters
1025 ----------
1026 uris : iterable of `ResourcePath`
1027 The URIs to test.
1028 num_workers : `int` or `None`, optional
1029 The number of parallel workers to use when checking for existence
1030 If `None`, the default value will be taken from the environment.
1032 Returns
1033 -------
1034 existence : `dict` of [`ResourcePath`, `bool`]
1035 Mapping of original URI to boolean indicating existence.
1036 """
1037 pool_executor_class = _get_executor_class()
1038 if issubclass(pool_executor_class, concurrent.futures.ProcessPoolExecutor):
1039 # Patch the environment to make it think there is only one worker
1040 # for each subprocess.
1041 with _patch_environ({"LSST_RESOURCES_NUM_WORKERS": "1"}):
1042 return cls._mexists_pool(pool_executor_class, uris)
1043 else:
1044 return cls._mexists_pool(pool_executor_class, uris, num_workers=num_workers)
1046 @classmethod
1047 def _mexists_pool(
1048 cls,
1049 pool_executor_class: _EXECUTOR_TYPE,
1050 uris: Iterable[ResourcePath],
1051 *,
1052 num_workers: int | None = None,
1053 ) -> dict[ResourcePath, bool]:
1054 """Check for existence of multiple URIs at once using specified pool
1055 executor.
1057 Implementation helper method for `_mexists`.
1059 Parameters
1060 ----------
1061 pool_executor_class : `type` [ `concurrent.futures.Executor` ]
1062 Type of executor pool to use.
1063 uris : iterable of `ResourcePath`
1064 The URIs to test.
1065 num_workers : `int` or `None`, optional
1066 The number of parallel workers to use when checking for existence
1067 If `None`, the default value will be taken from the environment.
1069 Returns
1070 -------
1071 existence : `dict` of [`ResourcePath`, `bool`]
1072 Mapping of original URI to boolean indicating existence.
1073 """
1074 max_workers = num_workers if num_workers is not None else _get_num_workers()
1075 with pool_executor_class(max_workers=max_workers) as exists_executor:
1076 future_exists = {exists_executor.submit(uri.exists): uri for uri in uris}
1078 results: dict[ResourcePath, bool] = {}
1079 for future in concurrent.futures.as_completed(future_exists):
1080 uri = future_exists[future]
1081 try:
1082 exists = future.result()
1083 except Exception:
1084 exists = False
1085 results[uri] = exists
1086 return results
1088 @classmethod
1089 def mtransfer(
1090 cls,
1091 transfer: str,
1092 from_to: Iterable[tuple[ResourcePath, ResourcePath]],
1093 overwrite: bool = False,
1094 transaction: TransactionProtocol | None = None,
1095 do_raise: bool = True,
1096 ) -> dict[ResourcePath, MBulkResult]:
1097 """Transfer many files in bulk.
1099 Parameters
1100 ----------
1101 transfer : `str`
1102 Mode to use for transferring the resource. Generically there are
1103 many standard options: copy, link, symlink, hardlink, relsymlink.
1104 Not all URIs support all modes.
1105 from_to : `list` [ `tuple` [ `ResourcePath`, `ResourcePath` ] ]
1106 A sequence of the source URIs and the target URIs.
1107 overwrite : `bool`, optional
1108 Allow an existing file to be overwritten. Defaults to `False`.
1109 transaction : `~lsst.resources.utils.TransactionProtocol`, optional
1110 A transaction object that can (depending on implementation)
1111 rollback transfers on error. Not guaranteed to be implemented.
1112 The transaction object must be thread safe.
1113 do_raise : `bool`, optional
1114 If `True` an `ExceptionGroup` will be raised containing any
1115 exceptions raised by the individual transfers. If `False`, or if
1116 there were no exceptions, a dict reporting the status of each
1117 `ResourcePath` will be returned.
1119 Returns
1120 -------
1121 copy_status : `dict` [ `ResourcePath`, `MBulkResult` ]
1122 A dict of all the transfer attempts with a value indicating
1123 whether the transfer succeeded for the target URI. If ``do_raise``
1124 is `True`, this will only be returned if there are no errors.
1125 """
1126 pool_executor_class = _get_executor_class()
1127 if issubclass(pool_executor_class, concurrent.futures.ProcessPoolExecutor):
1128 # Patch the environment to make it think there is only one worker
1129 # for each subprocess.
1130 with _patch_environ({"LSST_RESOURCES_NUM_WORKERS": "1"}):
1131 return cls._mtransfer(
1132 pool_executor_class,
1133 transfer,
1134 from_to,
1135 overwrite=overwrite,
1136 transaction=transaction,
1137 do_raise=do_raise,
1138 )
1139 return cls._mtransfer(
1140 pool_executor_class,
1141 transfer,
1142 from_to,
1143 overwrite=overwrite,
1144 transaction=transaction,
1145 do_raise=do_raise,
1146 )
1148 @classmethod
1149 def _mtransfer(
1150 cls,
1151 pool_executor_class: _EXECUTOR_TYPE,
1152 transfer: str,
1153 from_to: Iterable[tuple[ResourcePath, ResourcePath]],
1154 overwrite: bool = False,
1155 transaction: TransactionProtocol | None = None,
1156 do_raise: bool = True,
1157 ) -> dict[ResourcePath, MBulkResult]:
1158 """Transfer many files in bulk.
1160 Parameters
1161 ----------
1162 transfer : `str`
1163 Mode to use for transferring the resource. Generically there are
1164 many standard options: copy, link, symlink, hardlink, relsymlink.
1165 Not all URIs support all modes.
1166 from_to : `list` [ `tuple` [ `ResourcePath`, `ResourcePath` ] ]
1167 A sequence of the source URIs and the target URIs.
1168 overwrite : `bool`, optional
1169 Allow an existing file to be overwritten. Defaults to `False`.
1170 transaction : `~lsst.resources.utils.TransactionProtocol`, optional
1171 A transaction object that can (depending on implementation)
1172 rollback transfers on error. Not guaranteed to be implemented.
1173 The transaction object must be thread safe.
1174 do_raise : `bool`, optional
1175 If `True` an `ExceptionGroup` will be raised containing any
1176 exceptions raised by the individual transfers. Else a dict
1177 reporting the status of each `ResourcePath` will be returned.
1179 Returns
1180 -------
1181 copy_status : `dict` [ `ResourcePath`, `MBulkResult` ]
1182 A dict of all the transfer attempts with a value indicating
1183 whether the transfer succeeded for the target URI.
1184 """
1185 with pool_executor_class(max_workers=_get_num_workers()) as transfer_executor:
1186 future_transfers = {
1187 transfer_executor.submit(
1188 to_uri.transfer_from,
1189 from_uri,
1190 transfer=transfer,
1191 overwrite=overwrite,
1192 transaction=transaction,
1193 multithreaded=False,
1194 ): to_uri
1195 for from_uri, to_uri in from_to
1196 }
1197 results: dict[ResourcePath, MBulkResult] = {}
1198 failed = False
1199 for future in concurrent.futures.as_completed(future_transfers):
1200 to_uri = future_transfers[future]
1201 try:
1202 future.result()
1203 except Exception as e:
1204 transferred = MBulkResult(False, e)
1205 failed = True
1206 else:
1207 transferred = MBulkResult(True, None)
1208 results[to_uri] = transferred
1210 if do_raise and failed:
1211 raise ExceptionGroup(
1212 f"Errors transferring {len(results)} artifacts",
1213 tuple(res.exception for res in results.values() if res.exception is not None),
1214 )
1216 return results
1218 def remove(self) -> None:
1219 """Remove the resource."""
1220 raise NotImplementedError()
1222 @classmethod
1223 def mremove(
1224 cls, uris: Iterable[ResourcePath], *, do_raise: bool = True
1225 ) -> dict[ResourcePath, MBulkResult]:
1226 """Remove multiple URIs at once.
1228 Parameters
1229 ----------
1230 uris : iterable of `ResourcePath`
1231 URIs to remove.
1232 do_raise : `bool`, optional
1233 If `True` an `ExceptionGroup` will be raised containing any
1234 exceptions raised by the individual transfers. If `False`, or if
1235 there were no exceptions, a dict reporting the status of each
1236 `ResourcePath` will be returned.
1238 Returns
1239 -------
1240 results : `dict` [ `ResourcePath`, `MBulkResult` ]
1241 Dictionary mapping each URI to a result object indicating whether
1242 the removal succeeded or resulted in an exception. If ``do_raise``
1243 is `True` this will only be returned if everything succeeded.
1244 """
1245 # Group URIs by scheme since some URI schemes support native bulk
1246 # APIs.
1247 results: dict[ResourcePath, MBulkResult] = {}
1248 for uri_class, group in cls._group_uris(uris).items():
1249 results.update(uri_class._mremove(group))
1250 if do_raise: 1250 ↛ 1251line 1250 didn't jump to line 1251 because the condition on line 1250 was never true
1251 failed = any(not r.success for r in results.values())
1252 if failed:
1253 s = "s" if len(results) != 1 else ""
1254 raise ExceptionGroup(
1255 f"Error{s} removing {len(results)} artifact{s}",
1256 tuple(res.exception for res in results.values() if res.exception is not None),
1257 )
1259 return results
1261 @classmethod
1262 def _mremove(cls, uris: Iterable[ResourcePath]) -> dict[ResourcePath, MBulkResult]:
1263 """Remove multiple URIs using futures."""
1264 pool_executor_class = _get_executor_class()
1265 if issubclass(pool_executor_class, concurrent.futures.ProcessPoolExecutor):
1266 # Patch the environment to make it think there is only one worker
1267 # for each subprocess.
1268 with _patch_environ({"LSST_RESOURCES_NUM_WORKERS": "1"}):
1269 return cls._mremove_pool(pool_executor_class, uris)
1270 else:
1271 return cls._mremove_pool(pool_executor_class, uris)
1273 @classmethod
1274 def _mremove_pool(
1275 cls,
1276 pool_executor_class: _EXECUTOR_TYPE,
1277 uris: Iterable[ResourcePath],
1278 *,
1279 num_workers: int | None = None,
1280 ) -> dict[ResourcePath, MBulkResult]:
1281 """Remove URIs using a futures pool."""
1282 max_workers = num_workers if num_workers is not None else _get_num_workers()
1283 results: dict[ResourcePath, MBulkResult] = {}
1284 with pool_executor_class(max_workers=max_workers) as remove_executor:
1285 future_remove = {remove_executor.submit(uri.remove): uri for uri in uris}
1286 for future in concurrent.futures.as_completed(future_remove):
1287 try:
1288 future.result()
1289 except Exception as e:
1290 removed = MBulkResult(False, e)
1291 else:
1292 removed = MBulkResult(True, None)
1293 uri = future_remove[future]
1294 results[uri] = removed
1295 return results
1297 def isabs(self) -> bool:
1298 """Indicate that the resource is fully specified.
1300 For non-schemeless URIs this is always true.
1302 Returns
1303 -------
1304 isabs : `bool`
1305 `True` in all cases except schemeless URI.
1306 """
1307 return True
1309 def abspath(self) -> ResourcePath:
1310 """Return URI using an absolute path.
1312 Returns
1313 -------
1314 abs : `ResourcePath`
1315 Absolute URI. For non-schemeless URIs this always returns itself.
1316 Schemeless URIs are upgraded to file URIs.
1317 """
1318 return self
1320 @contextlib.contextmanager
1321 def _as_local(
1322 self, multithreaded: bool = True, tmpdir: ResourcePath | None = None
1323 ) -> Generator[ResourcePath]:
1324 """Return the location of the (possibly remote) resource as local file.
1326 This is a helper function for `as_local` context manager.
1328 Parameters
1329 ----------
1330 multithreaded : `bool`, optional
1331 If `True` the transfer will be allowed to attempt to improve
1332 throughput by using parallel download streams. This may of no
1333 effect if the URI scheme does not support parallel streams or
1334 if a global override has been applied. If `False` parallel
1335 streams will be disabled.
1336 tmpdir : `ResourcePath` or `None`, optional
1337 Explicit override of the temporary directory to use for remote
1338 downloads.
1340 Returns
1341 -------
1342 local_uri : `ResourcePath`
1343 A URI to a local POSIX file. This can either be the same resource
1344 or a local downloaded copy of the resource.
1345 """
1346 raise NotImplementedError()
1348 @contextlib.contextmanager
1349 def as_local(
1350 self, multithreaded: bool = True, tmpdir: ResourcePathExpression | None = None
1351 ) -> Generator[ResourcePath]:
1352 """Return the location of the (possibly remote) resource as local file.
1354 Parameters
1355 ----------
1356 multithreaded : `bool`, optional
1357 If `True` the transfer will be allowed to attempt to improve
1358 throughput by using parallel download streams. This may of no
1359 effect if the URI scheme does not support parallel streams or
1360 if a global override has been applied. If `False` parallel
1361 streams will be disabled.
1362 tmpdir : `lsst.resources.ResourcePathExpression` or `None`, optional
1363 Explicit override of the temporary directory to use for remote
1364 downloads. This directory must be a local POSIX directory and
1365 must exist.
1367 Yields
1368 ------
1369 local : `ResourcePath`
1370 If this is a remote resource, it will be a copy of the resource
1371 on the local file system, probably in a temporary directory.
1372 For a local resource this should be the actual path to the
1373 resource.
1375 Notes
1376 -----
1377 The context manager will automatically delete any local temporary
1378 file.
1380 Examples
1381 --------
1382 Should be used as a context manager:
1384 .. code-block:: py
1386 with uri.as_local() as local:
1387 ospath = local.ospath
1388 """
1389 if self.isdir():
1390 raise IsADirectoryError(f"Directory-like URI {self} cannot be fetched as local.")
1391 temp_dir = ResourcePath(tmpdir, forceDirectory=True) if tmpdir is not None else None
1392 if temp_dir is not None and not temp_dir.isLocal:
1393 raise ValueError(f"Temporary directory for as_local must be local resource not {temp_dir}")
1394 with self._as_local(multithreaded=multithreaded, tmpdir=temp_dir) as local_uri:
1395 yield local_uri
1397 @classmethod
1398 @contextlib.contextmanager
1399 def temporary_uri(
1400 cls,
1401 prefix: ResourcePath | None = None,
1402 suffix: str | None = None,
1403 delete: bool = True,
1404 ) -> Generator[ResourcePath]:
1405 """Create a temporary file-like URI.
1407 Parameters
1408 ----------
1409 prefix : `ResourcePath`, optional
1410 Temporary directory to use (can be any scheme). Without this the
1411 path will be formed as a local file URI in a temporary directory
1412 obtained from `lsst.resources.utils.get_tempdir`. Ensuring that the
1413 prefix location exists is the responsibility of the caller.
1414 suffix : `str`, optional
1415 A file suffix to be used. The ``.`` should be included in this
1416 suffix.
1417 delete : `bool`, optional
1418 By default the resource will be deleted when the context manager
1419 is exited. Setting this flag to `False` will leave the resource
1420 alone.
1422 Yields
1423 ------
1424 uri : `ResourcePath`
1425 The temporary URI. Will be removed when the context is completed.
1426 """
1427 if prefix is None:
1428 prefix = ResourcePath(get_tempdir(), forceDirectory=True)
1430 # Need to create a randomized file name. For consistency do not
1431 # use mkstemp for local and something else for remote. Additionally
1432 # this method does not create the file to prevent name clashes.
1433 characters = "abcdefghijklmnopqrstuvwxyz0123456789_"
1434 rng = Random()
1435 tempname = "".join(rng.choice(characters) for _ in range(16))
1436 if suffix:
1437 tempname += suffix
1438 temporary_uri = prefix.join(tempname, isTemporary=True)
1439 if temporary_uri.isdir():
1440 # If we had a safe way to clean up a remote temporary directory, we
1441 # could support this.
1442 raise NotImplementedError("temporary_uri cannot be used to create a temporary directory.")
1443 try:
1444 yield temporary_uri
1445 finally:
1446 if delete:
1447 with contextlib.suppress(FileNotFoundError):
1448 # It's okay if this does not work because the user
1449 # removed the file.
1450 temporary_uri.remove()
1452 def read(self, size: int = -1) -> bytes:
1453 """Open the resource and return the contents in bytes.
1455 Parameters
1456 ----------
1457 size : `int`, optional
1458 The number of bytes to read. Negative or omitted indicates
1459 that all data should be read.
1460 """
1461 raise NotImplementedError()
1463 def write(self, data: bytes, overwrite: bool = True) -> None:
1464 """Write the supplied bytes to the new resource.
1466 Parameters
1467 ----------
1468 data : `bytes`
1469 The bytes to write to the resource. The entire contents of the
1470 resource will be replaced.
1471 overwrite : `bool`, optional
1472 If `True` the resource will be overwritten if it exists. Otherwise
1473 the write will fail.
1474 """
1475 raise NotImplementedError()
1477 def mkdir(self) -> None:
1478 """For a dir-like URI, create the directory resource if needed."""
1479 raise NotImplementedError()
1481 def isdir(self) -> bool:
1482 """Return True if this URI looks like a directory, else False."""
1483 return bool(self.dirLike)
1485 def size(self) -> int:
1486 """For non-dir-like URI, return the size of the resource.
1488 Returns
1489 -------
1490 sz : `int`
1491 The size in bytes of the resource associated with this URI.
1492 Returns 0 if dir-like.
1493 """
1494 raise NotImplementedError()
1496 def __str__(self) -> str:
1497 """Convert the URI to its native string form."""
1498 return self.geturl()
1500 def __repr__(self) -> str:
1501 """Return string representation suitable for evaluation."""
1502 return f'ResourcePath("{self.geturl()}")'
1504 def __eq__(self, other: Any) -> bool:
1505 """Compare supplied object with this `ResourcePath`."""
1506 if not isinstance(other, ResourcePath):
1507 return NotImplemented
1508 return self.geturl() == other.geturl()
1510 def __hash__(self) -> int:
1511 """Return hash of this object."""
1512 return hash(str(self))
1514 def __lt__(self, other: ResourcePath) -> bool:
1515 return self.geturl() < other.geturl()
1517 def __le__(self, other: ResourcePath) -> bool:
1518 return self.geturl() <= other.geturl()
1520 def __gt__(self, other: ResourcePath) -> bool:
1521 return self.geturl() > other.geturl()
1523 def __ge__(self, other: ResourcePath) -> bool:
1524 return self.geturl() >= other.geturl()
1526 def __copy__(self) -> ResourcePath:
1527 """Copy constructor.
1529 Object is immutable so copy can return itself.
1530 """
1531 # Implement here because the __new__ method confuses things
1532 return self
1534 def __deepcopy__(self, memo: Any) -> ResourcePath:
1535 """Deepcopy the object.
1537 Object is immutable so copy can return itself.
1538 """
1539 # Implement here because the __new__ method confuses things
1540 return self
1542 def __getnewargs__(self) -> tuple:
1543 """Support pickling."""
1544 return (str(self),)
1546 @classmethod
1547 def _fixDirectorySep(
1548 cls, parsed: urllib.parse.ParseResult, forceDirectory: bool | None = None
1549 ) -> tuple[urllib.parse.ParseResult, bool | None]:
1550 """Ensure that a path separator is present on directory paths.
1552 Parameters
1553 ----------
1554 parsed : `~urllib.parse.ParseResult`
1555 The result from parsing a URI using `urllib.parse`.
1556 forceDirectory : `bool` or `None`, optional
1557 If `True` forces the URI to end with a separator, otherwise given
1558 URI is interpreted as is. Specifying that the URI is conceptually
1559 equivalent to a directory can break some ambiguities when
1560 interpreting the last element of a path.
1562 Returns
1563 -------
1564 modified : `~urllib.parse.ParseResult`
1565 Update result if a URI is being handled.
1566 dirLike : `bool` or `None`
1567 `True` if given parsed URI has a trailing separator or
1568 ``forceDirectory`` is `True`. Otherwise returns the given value of
1569 ``forceDirectory``.
1570 """
1571 # Assume the forceDirectory flag can give us a clue.
1572 dirLike = forceDirectory
1574 # Directory separator
1575 sep = cls._pathModule.sep
1577 # URI is dir-like if explicitly stated or if it ends on a separator
1578 endsOnSep = parsed.path.endswith(sep)
1580 if forceDirectory is False and endsOnSep:
1581 raise ValueError(
1582 f"URI {parsed.geturl()} ends with {sep} but "
1583 "forceDirectory parameter declares it to be a file."
1584 )
1586 if forceDirectory or endsOnSep:
1587 dirLike = True
1588 # only add the separator if it's not already there
1589 if not endsOnSep:
1590 parsed = parsed._replace(path=parsed.path + sep)
1592 return parsed, dirLike
1594 @classmethod
1595 def _fixupPathUri(
1596 cls,
1597 parsed: urllib.parse.ParseResult,
1598 root: ResourcePath | None = None,
1599 forceAbsolute: bool = False,
1600 forceDirectory: bool | None = None,
1601 ) -> tuple[urllib.parse.ParseResult, bool | None]:
1602 """Correct any issues with the supplied URI.
1604 Parameters
1605 ----------
1606 parsed : `~urllib.parse.ParseResult`
1607 The result from parsing a URI using `urllib.parse`.
1608 root : `ResourcePath`, ignored
1609 Not used by the this implementation since all URIs are
1610 absolute except for those representing the local file system.
1611 forceAbsolute : `bool`, ignored.
1612 Not used by this implementation. URIs are generally always
1613 absolute.
1614 forceDirectory : `bool` or `None`, optional
1615 If `True` forces the URI to end with a separator, otherwise given
1616 URI is interpreted as is. Specifying that the URI is conceptually
1617 equivalent to a directory can break some ambiguities when
1618 interpreting the last element of a path.
1620 Returns
1621 -------
1622 modified : `~urllib.parse.ParseResult`
1623 Update result if a URI is being handled.
1624 dirLike : `bool`
1625 `True` if given parsed URI has a trailing separator or
1626 ``forceDirectory`` is `True`. Otherwise returns the given value
1627 of ``forceDirectory``.
1629 Notes
1630 -----
1631 Relative paths are explicitly not supported by RFC8089 but `urllib`
1632 does accept URIs of the form ``file:relative/path.ext``. They need
1633 to be turned into absolute paths before they can be used. This is
1634 always done regardless of the ``forceAbsolute`` parameter.
1636 AWS S3 differentiates between keys with trailing POSIX separators (i.e
1637 ``/dir`` and ``/dir/``) whereas POSIX does not necessarily.
1639 Scheme-less paths are normalized.
1640 """
1641 return cls._fixDirectorySep(parsed, forceDirectory)
1643 def transfer_from(
1644 self,
1645 src: ResourcePath,
1646 transfer: str,
1647 overwrite: bool = False,
1648 transaction: TransactionProtocol | None = None,
1649 multithreaded: bool = True,
1650 ) -> None:
1651 """Transfer to this URI from another.
1653 Parameters
1654 ----------
1655 src : `ResourcePath`
1656 Source URI.
1657 transfer : `str`
1658 Mode to use for transferring the resource. Generically there are
1659 many standard options: copy, link, symlink, hardlink, relsymlink.
1660 Not all URIs support all modes.
1661 overwrite : `bool`, optional
1662 Allow an existing file to be overwritten. Defaults to `False`.
1663 transaction : `~lsst.resources.utils.TransactionProtocol`, optional
1664 A transaction object that can (depending on implementation)
1665 rollback transfers on error. Not guaranteed to be implemented.
1666 multithreaded : `bool`, optional
1667 If `True` the transfer will be allowed to attempt to improve
1668 throughput by using parallel download streams. This may of no
1669 effect if the URI scheme does not support parallel streams or
1670 if a global override has been applied. If `False` parallel
1671 streams will be disabled.
1673 Notes
1674 -----
1675 Conceptually this is hard to scale as the number of URI schemes
1676 grow. The destination URI is more important than the source URI
1677 since that is where all the transfer modes are relevant (with the
1678 complication that "move" deletes the source).
1680 Local file to local file is the fundamental use case but every
1681 other scheme has to support "copy" to local file (with implicit
1682 support for "move") and copy from local file.
1683 All the "link" options tend to be specific to local file systems.
1685 "move" is a "copy" where the remote resource is deleted at the end.
1686 Whether this works depends on the source URI rather than the
1687 destination URI. Reverting a move on transaction rollback is
1688 expected to be problematic if a remote resource was involved.
1689 """
1690 raise NotImplementedError(f"No transfer modes supported by URI scheme {self.scheme}")
1692 def walk(
1693 self, file_filter: str | re.Pattern | None = None
1694 ) -> Iterator[list | tuple[ResourcePath, list[str], list[str]]]:
1695 """Walk the directory tree returning matching files and directories.
1697 Parameters
1698 ----------
1699 file_filter : `str` or `re.Pattern`, optional
1700 Regex to filter out files from the list before it is returned.
1702 Yields
1703 ------
1704 dirpath : `ResourcePath`
1705 Current directory being examined.
1706 dirnames : `list` of `str`
1707 Names of subdirectories within dirpath.
1708 filenames : `list` of `str`
1709 Names of all the files within dirpath.
1710 """
1711 raise NotImplementedError()
1713 @overload
1714 @classmethod
1715 def findFileResources( 1715 ↛ exitline 1715 didn't return from function 'findFileResources' because
1716 cls,
1717 candidates: Iterable[ResourcePathExpression],
1718 file_filter: str | re.Pattern | None,
1719 grouped: Literal[True],
1720 ) -> Iterator[Iterator[ResourcePath]]: ...
1722 @overload
1723 @classmethod
1724 def findFileResources( 1724 ↛ exitline 1724 didn't return from function 'findFileResources' because
1725 cls,
1726 candidates: Iterable[ResourcePathExpression],
1727 *,
1728 grouped: Literal[True],
1729 ) -> Iterator[Iterator[ResourcePath]]: ...
1731 @overload
1732 @classmethod
1733 def findFileResources( 1733 ↛ exitline 1733 didn't return from function 'findFileResources' because
1734 cls,
1735 candidates: Iterable[ResourcePathExpression],
1736 file_filter: str | re.Pattern | None = None,
1737 grouped: Literal[False] = False,
1738 ) -> Iterator[ResourcePath]: ...
1740 @classmethod
1741 def findFileResources(
1742 cls,
1743 candidates: Iterable[ResourcePathExpression],
1744 file_filter: str | re.Pattern | None = None,
1745 grouped: bool = False,
1746 ) -> Iterator[ResourcePath | Iterator[ResourcePath]]:
1747 """Get all the files from a list of values.
1749 Parameters
1750 ----------
1751 candidates : iterable [`str` or `ResourcePath`]
1752 The files to return and directories in which to look for files to
1753 return.
1754 file_filter : `str` or `re.Pattern`, optional
1755 The regex to use when searching for files within directories.
1756 By default returns all the found files.
1757 grouped : `bool`, optional
1758 If `True` the results will be grouped by directory and each
1759 yielded value will be an iterator over URIs. If `False` each
1760 URI will be returned separately.
1762 Yields
1763 ------
1764 found_file: `ResourcePath`
1765 The passed-in URIs and URIs found in passed-in directories.
1766 If grouping is enabled, each of the yielded values will be an
1767 iterator yielding members of the group. Files given explicitly
1768 will be returned as a single group at the end.
1770 Notes
1771 -----
1772 If a value is a file it is yielded immediately without checking that it
1773 exists. If a value is a directory, all the files in the directory
1774 (recursively) that match the regex will be yielded in turn.
1775 """
1776 fileRegex = None if file_filter is None else re.compile(file_filter)
1778 singles = []
1780 # Find all the files of interest
1781 for location in candidates:
1782 uri = ResourcePath(location)
1783 if uri.isdir():
1784 for found in uri.walk(fileRegex):
1785 if not found: 1785 ↛ 1788line 1785 didn't jump to line 1788 because the condition on line 1785 was never true
1786 # This means the uri does not exist and by
1787 # convention we ignore it
1788 continue
1789 root, dirs, files = found
1790 if not files:
1791 continue
1792 if grouped:
1793 yield (root.join(name) for name in files)
1794 else:
1795 for name in files:
1796 yield root.join(name)
1797 else:
1798 if grouped:
1799 singles.append(uri)
1800 else:
1801 yield uri
1803 # Finally, return any explicitly given files in one group
1804 if grouped and singles:
1805 yield iter(singles)
1807 @contextlib.contextmanager
1808 def open(
1809 self,
1810 mode: str = "r",
1811 *,
1812 encoding: str | None = None,
1813 prefer_file_temporary: bool = False,
1814 ) -> Generator[ResourceHandleProtocol]:
1815 """Return a context manager that wraps an object that behaves like an
1816 open file at the location of the URI.
1818 Parameters
1819 ----------
1820 mode : `str`
1821 String indicating the mode in which to open the file. Values are
1822 the same as those accepted by `open`, though intrinsically
1823 read-only URI types may only support read modes, and
1824 `io.IOBase.seekable` is not guaranteed to be `True` on the returned
1825 object.
1826 encoding : `str`, optional
1827 Unicode encoding for text IO; ignored for binary IO. Defaults to
1828 ``locale.getpreferredencoding(False)``, just as `open`
1829 does.
1830 prefer_file_temporary : `bool`, optional
1831 If `True`, for implementations that require transfers from a remote
1832 system to temporary local storage and/or back, use a temporary file
1833 instead of an in-memory buffer; this is generally slower, but it
1834 may be necessary to avoid excessive memory usage by large files.
1835 Ignored by implementations that do not require a temporary.
1837 Yields
1838 ------
1839 cm : `~contextlib.AbstractContextManager`
1840 A context manager that wraps a `ResourceHandleProtocol` file-like
1841 object.
1843 Notes
1844 -----
1845 The default implementation of this method uses a local temporary buffer
1846 (in-memory or file, depending on ``prefer_file_temporary``) with calls
1847 to `read`, `write`, `as_local`, and `transfer_from` as necessary to
1848 read and write from/to remote systems. Remote writes thus occur only
1849 when the context manager is exited. `ResourcePath` implementations
1850 that can return a more efficient native buffer should do so whenever
1851 possible (as is guaranteed for local files). `ResourcePath`
1852 implementations for which `as_local` does not return a temporary are
1853 required to reimplement `open`, though they may delegate to `super`
1854 when ``prefer_file_temporary`` is `False`.
1855 """
1856 if self.isdir():
1857 raise IsADirectoryError(f"Directory-like URI {self} cannot be opened.")
1858 if "x" in mode and self.exists():
1859 raise FileExistsError(f"File at {self} already exists.")
1860 if prefer_file_temporary:
1861 if "r" in mode or "a" in mode:
1862 local_cm = self.as_local()
1863 else:
1864 local_cm = self.temporary_uri(suffix=self.getExtension())
1865 with local_cm as local_uri:
1866 assert local_uri.isTemporary, (
1867 "ResourcePath implementations for which as_local is not "
1868 "a temporary must reimplement `open`."
1869 )
1870 # An encoding is ignored for binary IO, so do not let the
1871 # builtin open() reject it.
1872 encoding_arg = None if "b" in mode else encoding
1873 with open(local_uri.ospath, mode=mode, encoding=encoding_arg) as file_buffer:
1874 if "a" in mode:
1875 file_buffer.seek(0, io.SEEK_END)
1876 yield file_buffer
1877 if "r" not in mode or "+" in mode:
1878 self.transfer_from(local_uri, transfer="copy", overwrite=("x" not in mode))
1879 else:
1880 with self._openImpl(mode, encoding=encoding) as handle:
1881 yield handle
1883 @contextlib.contextmanager
1884 def _openImpl(self, mode: str = "r", *, encoding: str | None = None) -> Generator[ResourceHandleProtocol]:
1885 """Implement opening of a resource handle.
1887 This private method may be overridden by specific `ResourcePath`
1888 implementations to provide a customized handle like interface.
1890 Parameters
1891 ----------
1892 mode : `str`
1893 The mode the handle should be opened with
1894 encoding : `str`, optional
1895 The byte encoding of any binary text
1897 Yields
1898 ------
1899 handle : `~._resourceHandles.BaseResourceHandle`
1900 A handle that conforms to the
1901 `~._resourceHandles.BaseResourceHandle` interface
1903 Notes
1904 -----
1905 The base implementation of a file handle reads in a files entire
1906 contents into a buffer for manipulation, and then writes it back out
1907 upon close. Subclasses of this class may offer more fine grained
1908 control.
1909 """
1910 in_bytes = self.read() if "r" in mode or "a" in mode else b""
1911 if "b" in mode: 1911 ↛ 1919line 1911 didn't jump to line 1919 because the condition on line 1911 was always true
1912 bytes_buffer = io.BytesIO(in_bytes)
1913 bytes_buffer.name = str(self)
1914 if "a" in mode: 1914 ↛ 1915line 1914 didn't jump to line 1915 because the condition on line 1914 was never true
1915 bytes_buffer.seek(0, io.SEEK_END)
1916 yield bytes_buffer
1917 out_bytes = bytes_buffer.getvalue()
1918 else:
1919 if encoding is None:
1920 encoding = locale.getpreferredencoding(False)
1921 str_buffer = io.StringIO(in_bytes.decode(encoding))
1922 str_buffer.name = str(self)
1923 if "a" in mode:
1924 str_buffer.seek(0, io.SEEK_END)
1925 yield str_buffer
1926 out_bytes = str_buffer.getvalue().encode(encoding)
1927 if "r" not in mode or "+" in mode: 1927 ↛ 1928line 1927 didn't jump to line 1928 because the condition on line 1927 was never true
1928 self.write(out_bytes, overwrite=("x" not in mode))
1930 def generate_presigned_get_url(self, *, expiration_time_seconds: int) -> str:
1931 """Return a pre-signed URL that can be used to retrieve this resource
1932 using an HTTP GET without supplying any access credentials.
1934 Parameters
1935 ----------
1936 expiration_time_seconds : `int`
1937 Number of seconds until the generated URL is no longer valid.
1939 Returns
1940 -------
1941 url : `str`
1942 HTTP URL signed for GET.
1943 """
1944 raise NotImplementedError(f"URL signing is not supported for '{self.scheme}'")
1946 def generate_presigned_put_url(self, *, expiration_time_seconds: int) -> str:
1947 """Return a pre-signed URL that can be used to upload a file to this
1948 path using an HTTP PUT without supplying any access credentials.
1950 Parameters
1951 ----------
1952 expiration_time_seconds : `int`
1953 Number of seconds until the generated URL is no longer valid.
1955 Returns
1956 -------
1957 url : `str`
1958 HTTP URL signed for PUT.
1959 """
1960 raise NotImplementedError(f"URL signing is not supported for '{self.scheme}'")
1962 def _copy_extra_attributes(self, original_uri: ResourcePath) -> None:
1963 # May be overridden by subclasses to transfer attributes when a
1964 # ResourcePath is constructed using the "clone" version of the
1965 # ResourcePath constructor by passing in a ResourcePath object.
1966 pass
1968 def get_info(self) -> ResourceInfo:
1969 """Return lightweight metadata about this resource.
1971 Returns
1972 -------
1973 info : `ResourceInfo`
1974 The information about this resource that can be obtained from
1975 the backend. Will not read the file contents.
1976 """
1977 raise NotImplementedError("")
1980ResourcePathExpression = str | urllib.parse.ParseResult | ResourcePath | Path
1981"""Type-annotation alias for objects that can be coerced to ResourcePath.
1982"""