Coverage for python/lsst/resources/_resourcePath.py: 92%
597 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-23 09:30 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-23 09:30 +0000
1# This file is part of lsst-resources.
2#
3# Developed for the LSST Data Management System.
4# This product includes software developed by the LSST Project
5# (https://www.lsst.org).
6# See the COPYRIGHT file at the top-level directory of this distribution
7# for details of code ownership.
8#
9# Use of this source code is governed by a 3-clause BSD-style
10# license that can be found in the LICENSE file.
12from __future__ import annotations
14__all__ = ("ResourceInfo", "ResourcePath", "ResourcePathExpression")
16import concurrent.futures
17import contextlib
18import copy
19import dataclasses
20import datetime
21import io
22import locale
23import logging
24import os
25import posixpath
26import re
27import sys
28import urllib.parse
29from collections import defaultdict
30from pathlib import Path, PurePath, PurePosixPath
31from random import Random
32from typing import TYPE_CHECKING
34try:
35 import fsspec
36 from fsspec.spec import AbstractFileSystem
37except ImportError:
38 # Hidden from type checkers so that the names above keep the types they
39 # have when fsspec is installed.
40 if not TYPE_CHECKING:
41 fsspec = None
42 AbstractFileSystem = type
44from collections.abc import Generator, Iterable, Iterator
45from typing import Any, Literal, NamedTuple, TypeVar, overload
47from lsst.utils.iteration import chunk_iterable
49from ._resourceHandles._baseResourceHandle import ResourceHandleProtocol
50from .utils import MAX_WORKERS, _get_num_workers, get_tempdir
52if TYPE_CHECKING:
53 from .utils import TransactionProtocol
56log = logging.getLogger(__name__)
58# Regex for looking for URI escapes
59ESCAPES_RE = re.compile(r"%[A-F0-9]{2}")
61# Precomputed escaped hash
62ESCAPED_HASH = urllib.parse.quote("#")
64_T = TypeVar("_T")
67class MBulkResult(NamedTuple):
68 """Report on a bulk operation."""
70 success: bool
71 exception: Exception | None
74@dataclasses.dataclass(frozen=True)
75class ResourceInfo:
76 """Information about this resource."""
78 uri: str
79 """URI in string form of the resource from which this information is
80 derived.
81 """
82 is_file: bool
83 """Indicate whether the resource is a file or a directory."""
84 size: int
85 """Size of the file in bytes. A directory or a URI that has no concept
86 of size returns 0."""
87 last_modified: datetime.datetime | None
88 """Modification date of the resource, if known."""
89 checksums: dict[str, Any]
90 """Checksums for this file. Supported checksum implementations are
91 backend dependent.
92 """
95class ResourcePath: # numpydoc ignore=PR02
96 """Convenience wrapper around URI parsers.
98 Provides access to URI components and can convert file
99 paths into absolute path URIs. Scheme-less URIs are treated as if
100 they are local file system paths and are converted to absolute URIs.
102 A specialist subclass is created for each supported URI scheme.
104 Parameters
105 ----------
106 uri : `str`, `pathlib.Path`, `urllib.parse.ParseResult`, or `ResourcePath`
107 URI in string form. Can be scheme-less if referring to a relative
108 path or an absolute path on the local file system.
109 root : `str` or `ResourcePath`, optional
110 When fixing up a relative path in a ``file`` scheme or if scheme-less,
111 use this as the root. Must be absolute. If `None` the current
112 working directory will be used. Can be any supported URI scheme.
113 Not used if ``forceAbsolute`` is `False`.
114 forceAbsolute : `bool`, optional
115 If `True`, scheme-less relative URI will be converted to an absolute
116 path using a ``file`` scheme. If `False` scheme-less URI will remain
117 scheme-less and will not be updated to ``file`` or absolute path unless
118 it is already an absolute path, in which case it will be updated to
119 a ``file`` scheme.
120 forceDirectory : `bool` or `None`, optional
121 If `True` forces the URI to end with a separator. If `False` the URI
122 is interpreted as a file-like entity. Default, `None`, is that the
123 given URI is interpreted as a directory if there is a trailing ``/`` or
124 for some schemes the system will check to see if it is a file or a
125 directory.
126 isTemporary : `bool`, optional
127 If `True` indicates that this URI points to a temporary resource.
128 The default is `False`, unless ``uri`` is already a `ResourcePath`
129 instance and ``uri.isTemporary is True``.
131 Notes
132 -----
133 A non-standard URI of the form ``file:dir/file.txt`` is always converted
134 to an absolute ``file`` URI.
135 """
137 _pathLib: type[PurePath] = PurePosixPath
138 """Path library to use for this scheme."""
140 _pathModule = posixpath
141 """Path module to use for this scheme."""
143 transferModes: tuple[str, ...] = ("copy", "auto", "move")
144 """Transfer modes supported by this implementation.
146 Move is special in that it is generally a copy followed by an unlink.
147 Whether that unlink works depends critically on whether the source URI
148 implements unlink. If it does not the move will be reported as a failure.
149 """
151 transferDefault: str = "copy"
152 """Default mode to use for transferring if ``auto`` is specified."""
154 quotePaths = True
155 """True if path-like elements modifying a URI should be quoted.
157 All non-schemeless URIs have to internally use quoted paths. Therefore
158 if a new file name is given (e.g. to updatedFile or join) a decision must
159 be made whether to quote it to be consistent.
160 """
162 isLocal = False
163 """If `True` this URI refers to a local file."""
165 _max_workers: int = MAX_WORKERS
166 """Upper bound on workers for parallel operations on this scheme.
168 Schemes backed by a connection pool keep this modest because the pool is
169 sized to match it; schemes with no pool can raise it.
170 """
172 _chunk_size: int = 1
173 """Number of URIs given to each worker by `mexists` and `mremove`.
175 A batch that fits in a single chunk is handled in the calling thread
176 rather than being sent to a pool. The default suits a scheme where every
177 operation is a network round trip, which is expensive enough that handing
178 over one URI at a time costs nothing and gives the pool the freedom to
179 balance itself. A scheme whose operations are cheap should raise it until
180 a chunk is worth handing over.
181 """
183 _transfer_chunk_size: int = 1
184 """Number of files given to each worker by `mtransfer`.
186 Kept separate from ``_chunk_size`` because a transfer costs orders of
187 magnitude more than an existence check on the same scheme, and its cost
188 scales with a file size the caller does not know in advance, so chunks
189 have to stay small enough for the pool queue to balance them.
190 """
192 # This is not an ABC with abstract methods because the __new__ being
193 # a factory confuses mypy such that it assumes that every constructor
194 # returns a ResourcePath and then determines that all the abstract methods
195 # are still abstract. If they are not marked abstract but just raise
196 # mypy is fine with it.
198 # mypy is confused without these
199 _uri: urllib.parse.ParseResult
200 isTemporary: bool
201 dirLike: bool | None
202 """Whether the resource looks like a directory resource. `None` means that
203 the status is uncertain."""
205 def __new__(
206 cls,
207 uri: ResourcePathExpression,
208 root: str | ResourcePath | None = None,
209 forceAbsolute: bool = True,
210 forceDirectory: bool | None = None,
211 isTemporary: bool | None = None,
212 ) -> ResourcePath:
213 """Create and return new specialist ResourcePath subclass."""
214 parsed: urllib.parse.ParseResult
215 dirLike: bool | None = forceDirectory
216 subclass: type[ResourcePath] | None = None
218 # Force root to be a ResourcePath -- this simplifies downstream
219 # code.
220 if root is None:
221 root_uri = None
222 elif isinstance(root, str):
223 root_uri = ResourcePath(root, forceDirectory=True, forceAbsolute=True)
224 else:
225 root_uri = root
227 if isinstance(uri, os.PathLike):
228 uri = str(uri)
230 # Record if we need to post process the URI components
231 # or if the instance is already fully configured
232 if isinstance(uri, str):
233 # Since local file names can have special characters in them
234 # we need to quote them for the parser but we can unquote
235 # later. Assume that all other URI schemes are quoted.
236 # Since sometimes people write file:/a/b and not file:///a/b
237 # we should not quote in the explicit case of file:
238 if "://" not in uri and not uri.startswith("file:"):
239 if ESCAPES_RE.search(uri):
240 log.warning("Possible double encoding of %s", uri)
241 else:
242 # Fragments are generally not encoded so we must search
243 # for the fragment boundary ourselves. This is making
244 # an assumption that the filename does not include a "#"
245 # and also that there is no "/" in the fragment itself.
246 to_encode = uri
247 fragment = ""
248 if "#" in uri:
249 dirpos = uri.rfind("/")
250 trailing = uri[dirpos + 1 :]
251 hashpos = trailing.rfind("#")
252 if hashpos != -1:
253 fragment = trailing[hashpos:]
254 to_encode = uri[: dirpos + hashpos + 1]
256 uri = urllib.parse.quote(to_encode) + fragment
258 parsed = urllib.parse.urlparse(uri)
259 elif isinstance(uri, urllib.parse.ParseResult):
260 parsed = copy.copy(uri)
261 # If we are being instantiated with a subclass, rather than
262 # ResourcePath, ensure that that subclass is used directly.
263 # This could lead to inconsistencies if this constructor
264 # is used externally outside of the ResourcePath.replace() method.
265 # S3ResourcePath(urllib.parse.urlparse("file://a/b.txt"))
266 # will be a problem.
267 # This is needed to prevent a schemeless absolute URI become
268 # a file URI unexpectedly when calling updatedFile or
269 # updatedExtension
270 if cls is not ResourcePath:
271 parsed, dirLike = cls._fixDirectorySep(parsed, forceDirectory)
272 subclass = cls
274 elif isinstance(uri, ResourcePath):
275 # Since ResourcePath is immutable we can return the argument
276 # unchanged if it already agrees with forceDirectory, isTemporary,
277 # and forceAbsolute.
278 # We invoke __new__ again with str(self) to add a scheme for
279 # forceAbsolute, but for the others that seems more likely to paper
280 # over logic errors than do something useful, so we just raise.
281 if forceDirectory is not None and uri.dirLike is not None and forceDirectory is not uri.dirLike:
282 # Can not force a file-like URI to become a dir-like one or
283 # vice versa.
284 raise RuntimeError(
285 f"{uri} can not be forced to change directory vs file state when previously declared."
286 )
287 if isTemporary is not None and isTemporary is not uri.isTemporary:
288 raise RuntimeError(
289 f"{uri} is already a {'temporary' if uri.isTemporary else 'permanent'} "
290 f"ResourcePath; cannot make it {'temporary' if isTemporary else 'permanent'}."
291 )
293 if forceAbsolute and not uri.scheme:
294 # Create new absolute from relative.
295 return ResourcePath(
296 str(uri),
297 root=root,
298 forceAbsolute=forceAbsolute,
299 forceDirectory=forceDirectory or uri.dirLike,
300 isTemporary=uri.isTemporary,
301 )
302 elif forceDirectory is not None and uri.dirLike is None:
303 # Clone but with a new dirLike status.
304 return uri.replace(forceDirectory=forceDirectory)
305 return uri
306 else:
307 raise ValueError(
308 f"Supplied URI must be string, Path, ResourcePath, or ParseResult but got '{uri!r}'"
309 )
311 if subclass is None:
312 # Work out the subclass from the URI scheme
313 if not parsed.scheme:
314 # Root may be specified as a ResourcePath that overrides
315 # the schemeless determination.
316 if (
317 root_uri is not None
318 and root_uri.scheme != "file" # file scheme has different code path
319 and not parsed.path.startswith("/") # Not already absolute path
320 ):
321 if root_uri.dirLike is False:
322 raise ValueError(
323 f"Root URI ({root}) was not a directory so can not be joined with"
324 f" path {parsed.path!r}"
325 )
326 # If root is temporary or this schemeless is temporary we
327 # assume this URI is temporary.
328 isTemporary = isTemporary or root_uri.isTemporary
329 joined = root_uri.join(
330 parsed.path, forceDirectory=forceDirectory, isTemporary=isTemporary
331 )
333 # Rather than returning this new ResourcePath directly we
334 # instead extract the path and the scheme and adjust the
335 # URI we were given -- we need to do this to preserve
336 # fragments since join() will drop them.
337 parsed = parsed._replace(scheme=joined.scheme, path=joined.path, netloc=joined.netloc)
338 subclass = type(joined)
340 # Clear the root parameter to indicate that it has
341 # been applied already.
342 root_uri = None
343 else:
344 from .schemeless import SchemelessResourcePath
346 subclass = SchemelessResourcePath
347 elif parsed.scheme == "file":
348 from .file import FileResourcePath
350 subclass = FileResourcePath
351 elif parsed.scheme == "s3":
352 from .s3 import S3ResourcePath
354 subclass = S3ResourcePath
355 elif parsed.scheme.startswith("http"):
356 from .http import HttpResourcePath
358 subclass = HttpResourcePath
359 elif parsed.scheme in {"dav", "davs"}:
360 from .dav import DavResourcePath
362 subclass = DavResourcePath
363 elif parsed.scheme == "gs":
364 from .gs import GSResourcePath
366 subclass = GSResourcePath
367 elif parsed.scheme == "resource":
368 # Rules for scheme names disallow pkg_resource
369 from .packageresource import PackageResourcePath
371 subclass = PackageResourcePath
372 elif parsed.scheme == "mem":
373 # in-memory datastore object
374 from .mem import InMemoryResourcePath
376 subclass = InMemoryResourcePath
377 elif parsed.scheme == "eups":
378 # EUPS package root.
379 from .eups import EupsResourcePath
381 subclass = EupsResourcePath
382 elif parsed.scheme == "remote-test":
383 # EUPS package root.
384 from .remote_test import RemoteTestResourcePath
386 subclass = RemoteTestResourcePath
387 else:
388 raise NotImplementedError(
389 f"No URI support for scheme: '{parsed.scheme}' in {parsed.geturl()}"
390 )
392 parsed, dirLike = subclass._fixupPathUri(
393 parsed, root=root_uri, forceAbsolute=forceAbsolute, forceDirectory=forceDirectory
394 )
396 # It is possible for the class to change from schemeless
397 # to file or eups so handle that
398 if parsed.scheme == "file":
399 from .file import FileResourcePath
401 subclass = FileResourcePath
402 elif parsed.scheme == "eups":
403 from .eups import EupsResourcePath
405 subclass = EupsResourcePath
407 # Now create an instance of the correct subclass and set the
408 # attributes directly
409 self = object.__new__(subclass)
410 self._uri = parsed
411 self.dirLike = dirLike
412 if isTemporary is None:
413 isTemporary = False
414 self.isTemporary = isTemporary
415 self._set_proxy()
416 return self
418 def _set_proxy(self) -> None:
419 """Calculate internal proxy for externally visible resource path."""
420 pass
422 @property
423 def scheme(self) -> str:
424 """Return the URI scheme.
426 Notes
427 -----
428 (``://`` is not part of the scheme).
429 """
430 return self._uri.scheme
432 @property
433 def netloc(self) -> str:
434 """Return the URI network location."""
435 return self._uri.netloc
437 @property
438 def path(self) -> str:
439 """Return the path component of the URI."""
440 return self._uri.path
442 @property
443 def unquoted_path(self) -> str:
444 """Return path component of the URI with any URI quoting reversed."""
445 return urllib.parse.unquote(self._uri.path)
447 @property
448 def ospath(self) -> str:
449 """Return the path component of the URI localized to current OS."""
450 raise AttributeError(f"Non-file URI ({self}) has no local OS path.")
452 @property
453 def relativeToPathRoot(self) -> str:
454 """Return path relative to network location.
456 This is the path property with posix separator stripped
457 from the left hand side of the path.
459 Always unquotes.
460 """
461 relToRoot = self.path.lstrip("/")
462 if relToRoot == "":
463 return "./"
464 return urllib.parse.unquote(relToRoot)
466 @property
467 def is_root(self) -> bool:
468 """Return whether this URI points to the root of the network location.
470 This means that the path components refers to the top level.
471 """
472 relpath = self.relativeToPathRoot
473 if relpath == "./":
474 return True
475 return False
477 @property
478 def fragment(self) -> str:
479 """Return the fragment component of the URI. May be quoted."""
480 return self._uri.fragment
482 @property
483 def unquoted_fragment(self) -> str:
484 """Return unquoted fragment."""
485 return urllib.parse.unquote(self.fragment)
487 @property
488 def params(self) -> str:
489 """Return any parameters included in the URI."""
490 return self._uri.params
492 @property
493 def query(self) -> str:
494 """Return any query strings included in the URI."""
495 return self._uri.query
497 def geturl(self) -> str:
498 """Return the URI in string form.
500 Returns
501 -------
502 url : `str`
503 String form of URI.
504 """
505 return self._uri.geturl()
507 def to_fsspec(self) -> tuple[AbstractFileSystem, str]:
508 """Return an abstract file system and path that can be used by fsspec.
510 Returns
511 -------
512 fs : `fsspec.spec.AbstractFileSystem`
513 A file system object suitable for use with the returned path.
514 path : `str`
515 A path that can be opened by the file system object.
516 """
517 if fsspec is None:
518 raise ImportError("fsspec is not available")
519 # By default give the URL to fsspec and hope.
520 return fsspec.url_to_fs(self.geturl())
522 def root_uri(self) -> ResourcePath:
523 """Return the base root URI.
525 Returns
526 -------
527 uri : `ResourcePath`
528 Root URI.
529 """
530 return self.replace(path="", query="", fragment="", params="", forceDirectory=True)
532 def split(self) -> tuple[ResourcePath, str]:
533 """Split URI into head and tail.
535 Returns
536 -------
537 head: `ResourcePath`
538 Everything leading up to tail, expanded and normalized as per
539 ResourcePath rules.
540 tail : `str`
541 Last path component. Tail will be empty if path ends on a
542 separator or if the URI is known to be associated with a directory.
543 Tail will never contain separators. It will be unquoted.
545 Notes
546 -----
547 Equivalent to `os.path.split` where head preserves the URI
548 components. In some cases this method can result in a file system
549 check to verify whether the URI is a directory or not (only if
550 ``forceDirectory`` was `None` during construction). For a scheme-less
551 URI this can mean that the result might change depending on current
552 working directory.
553 """
554 if self.isdir():
555 # This is known to be a directory so must return itself and
556 # the empty string.
557 return self, ""
559 head, tail = self._pathModule.split(self.path)
560 headuri = self._uri._replace(path=head, fragment="", query="", params="")
562 # The file part should never include quoted metacharacters
563 tail = urllib.parse.unquote(tail)
565 # Schemeless is special in that it can be a relative path.
566 # We need to ensure that it stays that way. All other URIs will
567 # be absolute already.
568 forceAbsolute = self.isabs()
569 return ResourcePath(headuri, forceDirectory=True, forceAbsolute=forceAbsolute), tail
571 def basename(self) -> str:
572 """Return the base name, last element of path, of the URI.
574 Returns
575 -------
576 tail : `str`
577 Last part of the path attribute. Trail will be empty if path ends
578 on a separator.
580 Notes
581 -----
582 If URI ends on a slash returns an empty string. This is the second
583 element returned by `split()`.
585 Equivalent of `os.path.basename`.
586 """
587 return self.split()[1]
589 def dirname(self) -> ResourcePath:
590 """Return the directory component of the path as a new `ResourcePath`.
592 Returns
593 -------
594 head : `ResourcePath`
595 Everything except the tail of path attribute, expanded and
596 normalized as per ResourcePath rules.
598 Notes
599 -----
600 Equivalent of `os.path.dirname`. If this is a directory URI it will
601 be returned unchanged. If the parent directory is always required
602 use `parent`.
603 """
604 return self.split()[0]
606 def parent(self) -> ResourcePath:
607 """Return a `ResourcePath` of the parent directory.
609 Returns
610 -------
611 head : `ResourcePath`
612 Everything except the tail of path attribute, expanded and
613 normalized as per `ResourcePath` rules.
615 Notes
616 -----
617 For a file-like URI this will be the same as calling `dirname`.
618 For a directory-like URI this will always return the parent directory
619 whereas `dirname()` will return the original URI. This is consistent
620 with `os.path.dirname` compared to the `pathlib.Path` property
621 ``parent``.
622 """
623 if self.dirLike is False:
624 # os.path.split() is slightly faster than calling Path().parent.
625 return self.dirname()
626 # When self is dir-like, returns its parent directory,
627 # regardless of the presence of a trailing separator
628 originalPath = self._pathLib(self.path)
629 parentPath = originalPath.parent
630 return self.replace(path=str(parentPath), forceDirectory=True, fragment="", query="", params="")
632 def replace(
633 self, forceDirectory: bool | None = None, isTemporary: bool = False, **kwargs: Any
634 ) -> ResourcePath:
635 """Return new `ResourcePath` with specified components replaced.
637 Parameters
638 ----------
639 forceDirectory : `bool` or `None`, optional
640 Parameter passed to ResourcePath constructor to force this
641 new URI to be dir-like or file-like.
642 isTemporary : `bool`, optional
643 Indicate that the resulting URI is temporary resource.
644 **kwargs
645 Components of a `urllib.parse.ParseResult` that should be
646 modified for the newly-created `ResourcePath`.
648 Returns
649 -------
650 new : `ResourcePath`
651 New `ResourcePath` object with updated values.
653 Notes
654 -----
655 Does not, for now, allow a change in URI scheme.
656 """
657 # Disallow a change in scheme
658 if "scheme" in kwargs:
659 raise ValueError(f"Can not use replace() method to change URI scheme for {self}")
660 result = self.__class__(
661 self._uri._replace(**kwargs), forceDirectory=forceDirectory, isTemporary=isTemporary
662 )
663 result._copy_extra_attributes(self)
664 return result
666 def updatedFile(self, newfile: str) -> ResourcePath:
667 """Return new URI with an updated final component of the path.
669 Parameters
670 ----------
671 newfile : `str`
672 File name with no path component.
674 Returns
675 -------
676 updated : `ResourcePath`
677 Updated `ResourcePath` with new updated final component.
679 Notes
680 -----
681 Forces the ``ResourcePath.dirLike`` attribute to be false. The new file
682 path will be quoted if necessary. If the current URI is known to
683 refer to a directory, the new file will be joined to the current file.
684 It is recommended that this behavior no longer be used and a call
685 to `isdir` by the caller should be used to decide whether to join or
686 replace. In the future this method may be modified to always replace
687 the final element of the path.
688 """
689 if self.dirLike:
690 return self.join(newfile, forceDirectory=False)
691 return self.parent().join(newfile, forceDirectory=False)
693 def updatedExtension(self, ext: str | None) -> ResourcePath:
694 """Return a new `ResourcePath` with updated file extension.
696 All file extensions are replaced.
698 Parameters
699 ----------
700 ext : `str` or `None`
701 New extension. If an empty string is given any extension will
702 be removed. If `None` is given there will be no change.
704 Returns
705 -------
706 updated : `ResourcePath`
707 URI with the specified extension. Can return itself if
708 no extension was specified.
709 """
710 if ext is None:
711 return self
713 # Get the extension
714 current = self.getExtension()
716 # Nothing to do if the extension already matches
717 if current == ext:
718 return self
720 # Remove the current extension from the path
721 # .fits.gz counts as one extension do not use os.path.splitext
722 path = self.path
723 if current:
724 path = path.removesuffix(current)
726 # Ensure that we have a leading "." on file extension (and we do not
727 # try to modify the empty string)
728 if ext and not ext.startswith("."):
729 ext = "." + ext
731 return self.replace(path=path + ext, forceDirectory=False)
733 def getExtension(self) -> str:
734 """Return the extension(s) associated with this URI path.
736 Returns
737 -------
738 ext : `str`
739 The file extension (including the ``.``). Can be empty string
740 if there is no file extension. Usually returns only the last
741 file extension unless there is a special extension modifier
742 indicating file compression, in which case the combined
743 extension (e.g. ``.fits.gz``) will be returned.
745 Notes
746 -----
747 Does not distinguish between file and directory URIs when determining
748 a suffix. An extension is only determined from the final component
749 of the path.
750 """
751 special = {".gz", ".bz2", ".xz", ".fz"}
753 # path lib will ignore any "." in directories.
754 # path lib works well:
755 # extensions = self._pathLib(self.path).suffixes
756 # But the constructor is slow. Therefore write our own implementation.
757 # Strip trailing separator if present, do not care if this is a
758 # directory or not.
759 parts = self.path.rstrip("/").rsplit(self._pathModule.sep, 1)
760 _, *extensions = parts[-1].split(".")
762 if not extensions:
763 return ""
764 extensions = ["." + x for x in extensions]
766 ext = extensions.pop()
768 # Multiple extensions, decide whether to include the final two
769 if extensions and ext in special:
770 ext = f"{extensions[-1]}{ext}"
772 return ext
774 def join(
775 self, path: str | ResourcePath, isTemporary: bool | None = None, forceDirectory: bool | None = None
776 ) -> ResourcePath:
777 """Return new `ResourcePath` with additional path components.
779 Parameters
780 ----------
781 path : `str`, `ResourcePath`
782 Additional file components to append to the current URI. Will be
783 quoted depending on the associated URI scheme. If the path looks
784 like a URI referring to an absolute location, it will be returned
785 directly (matching the behavior of `os.path.join`). It can
786 also be a `ResourcePath`. Fragments are propagated.
787 isTemporary : `bool`, optional
788 Indicate that the resulting URI represents a temporary resource.
789 Default is ``self.isTemporary``.
790 forceDirectory : `bool` or `None`, optional
791 If `True` forces the URI to end with a separator. If `False` the
792 resultant URI is declared to refer to a file. `None` indicates
793 that the file directory status is unknown.
795 Returns
796 -------
797 new : `ResourcePath`
798 New URI with the path appended.
800 Notes
801 -----
802 Schemeless URIs assume local path separator but all other URIs assume
803 POSIX separator if the supplied path has directory structure. It
804 may be this never becomes a problem but datastore templates assume
805 POSIX separator is being used.
807 If an absolute `ResourcePath` is given for ``path`` is is assumed that
808 this should be returned directly. Giving a ``path`` of an absolute
809 scheme-less URI is not allowed for safety reasons as it may indicate
810 a mistake in the calling code.
812 It is an error to attempt to join to something that is known to
813 refer to a file. Use `updatedFile` if the file is to be
814 replaced.
816 If an unquoted ``#`` is included in the path it is assumed to be
817 referring to a fragment and not part of the file name.
819 Raises
820 ------
821 ValueError
822 Raised if the given path object refers to a directory but the
823 ``forceDirectory`` parameter insists the outcome should be a file,
824 and vice versa. Also raised if the URI being joined with is known
825 to refer to a file.
826 RuntimeError
827 Raised if this attempts to join a temporary URI to a non-temporary
828 URI.
829 """
830 if self.dirLike is False:
831 raise ValueError("Can not join a new path component to a file.")
832 if isTemporary is None:
833 isTemporary = self.isTemporary
834 elif not isTemporary and self.isTemporary:
835 raise RuntimeError("Cannot join temporary URI to non-temporary URI.")
836 # If we have a full URI in path we will use it directly
837 # but without forcing to absolute so that we can trap the
838 # expected option of relative path.
839 path_uri = ResourcePath(
840 path, forceAbsolute=False, forceDirectory=forceDirectory, isTemporary=isTemporary
841 )
842 if forceDirectory is not None and path_uri.dirLike is not forceDirectory: 842 ↛ 843line 842 didn't jump to line 843 because the condition on line 842 was never true
843 raise ValueError(
844 "The supplied path URI to join has inconsistent directory state "
845 f"with forceDirectory parameter: {path_uri.dirLike} vs {forceDirectory}"
846 )
847 forceDirectory = path_uri.dirLike
849 if path_uri.isabs():
850 # Absolute URI so return it directly.
851 return path_uri
853 # We want to propagate fragments to the joined path and we rely on
854 # the ResourcePath parser to find these fragments for us even in plain
855 # strings. Must assume there are no `#` characters in filenames.
856 if not isinstance(path, str) or path_uri.fragment:
857 path = path_uri.unquoted_path
859 # Might need to quote the path.
860 if self.quotePaths:
861 path = urllib.parse.quote(path)
863 newpath = self._pathModule.normpath(self._pathModule.join(self.path, path))
865 # normpath can strip trailing / so we force directory if the supplied
866 # path ended with a /
867 has_dir_sep = path.endswith(self._pathModule.sep)
868 if forceDirectory is None and has_dir_sep: 868 ↛ 869line 868 didn't jump to line 869 because the condition on line 868 was never true
869 forceDirectory = True
870 elif forceDirectory is False and has_dir_sep: 870 ↛ 871line 870 didn't jump to line 871 because the condition on line 870 was never true
871 raise ValueError("Path to join has trailing / but is being forced to be a file.")
872 return self.replace(
873 path=newpath,
874 forceDirectory=forceDirectory,
875 isTemporary=isTemporary,
876 fragment=path_uri.fragment,
877 query=path_uri.query,
878 params=path_uri.params,
879 )
881 def relative_to(self, other: ResourcePath, walk_up: bool = False) -> str | None:
882 """Return the relative path from this URI to the other URI.
884 Parameters
885 ----------
886 other : `ResourcePath`
887 URI to use to calculate the relative path. Must be a parent
888 of this URI.
889 walk_up : `bool`, optional
890 Control whether "``..``" can be used to resolve a relative path.
891 Default is `False`. Can not be `True` on Python version 3.11.
893 Returns
894 -------
895 subpath : `str`
896 The sub path of this URI relative to the supplied other URI.
897 Returns `None` if there is no parent child relationship.
898 Scheme and netloc must match.
899 """
900 # Scheme-less self is handled elsewhere.
901 if self.scheme != other.scheme:
902 return None
903 if self.netloc != other.netloc:
904 # Special case for localhost vs empty string.
905 # There can be many variants of localhost.
906 local_netlocs = {"", "localhost", "localhost.localdomain", "127.0.0.1"}
907 if not {self.netloc, other.netloc}.issubset(local_netlocs):
908 return None
910 # Rather than trying to guess a failure reason from the TypeError
911 # explicitly check for python 3.11. Doing this will simplify the
912 # rediscovery of a useless python version check when we set a new
913 # minimum version.
914 kwargs = {}
915 if walk_up:
916 if sys.version_info < (3, 12, 0): 916 ↛ 917line 916 didn't jump to line 917 because the condition on line 916 was never true
917 raise TypeError("walk_up parameter can not be true in python 3.11 and older")
919 kwargs["walk_up"] = True
921 enclosed_path = self._pathLib(self.relativeToPathRoot)
922 parent_path = other.relativeToPathRoot
923 subpath: str | None
924 try:
925 subpath = str(enclosed_path.relative_to(parent_path, **kwargs))
926 except ValueError:
927 subpath = None
928 else:
929 subpath = urllib.parse.unquote(subpath)
930 return subpath
932 def exists(self) -> bool:
933 """Indicate that the resource is available.
935 Returns
936 -------
937 exists : `bool`
938 `True` if the resource exists.
939 """
940 raise NotImplementedError()
942 @classmethod
943 def _group_uris(cls, uris: Iterable[ResourcePath]) -> dict[type[ResourcePath], list[ResourcePath]]:
944 """Group URIs by class/scheme."""
945 grouped: dict[type[ResourcePath], list[ResourcePath]] = defaultdict(list)
946 for uri in uris:
947 grouped[uri.__class__].append(uri)
948 return grouped
950 @classmethod
951 def mexists(
952 cls, uris: Iterable[ResourcePath], *, num_workers: int | None = None
953 ) -> dict[ResourcePath, bool]:
954 """Check for existence of multiple URIs at once.
956 Parameters
957 ----------
958 uris : iterable of `ResourcePath`
959 The URIs to test.
960 num_workers : `int` or `None`, optional
961 The number of parallel workers to use when checking for existence.
962 If `None`, the default value will be taken from the environment
963 and bounded by the limit for this scheme.
965 Returns
966 -------
967 existence : `dict` of [`ResourcePath`, `bool`]
968 Mapping of original URI to boolean indicating existence.
969 """
970 existence: dict[ResourcePath, bool] = {}
971 for uri_class, group in cls._group_uris(uris).items():
972 existence.update(uri_class._mexists(group, num_workers=num_workers))
974 return existence
976 @classmethod
977 def _mexists(
978 cls, uris: Iterable[ResourcePath], *, num_workers: int | None = None
979 ) -> dict[ResourcePath, bool]:
980 """Check for existence of multiple URIs at once.
982 Implementation helper method for `mexists`.
984 Parameters
985 ----------
986 uris : iterable of `ResourcePath`
987 The URIs to test.
988 num_workers : `int` or `None`, optional
989 The number of parallel workers to use when checking for existence
990 If `None`, the default value will be taken from the environment.
992 Returns
993 -------
994 existence : `dict` of [`ResourcePath`, `bool`]
995 Mapping of original URI to boolean indicating existence.
996 """
997 uri_list = list(uris)
998 max_workers = num_workers if num_workers is not None else _get_num_workers(cls._max_workers)
999 chunks = cls._chunk_work(uri_list, cls._chunk_size)
1000 if not chunks: 1000 ↛ 1001line 1000 didn't jump to line 1001 because the condition on line 1000 was never true
1001 return {}
1002 if len(chunks) == 1:
1003 # Not enough work to be worth handing to another thread.
1004 return cls._exists_chunk(chunks[0])
1006 results: dict[ResourcePath, bool] = {}
1007 with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as exists_executor:
1008 future_exists = {exists_executor.submit(cls._exists_chunk, chunk): chunk for chunk in chunks}
1009 for future in concurrent.futures.as_completed(future_exists):
1010 try:
1011 results.update(future.result())
1012 except Exception:
1013 # The chunk failed as a whole, for example because the
1014 # pool could not start a thread.
1015 for uri in future_exists[future]:
1016 results[uri] = False
1017 return results
1019 @classmethod
1020 def _exists_chunk(cls, uris: tuple[ResourcePath, ...]) -> dict[ResourcePath, bool]:
1021 """Check a batch of URIs for existence.
1023 Parameters
1024 ----------
1025 uris : `tuple` [ `ResourcePath`, ... ]
1026 The URIs to check.
1028 Returns
1029 -------
1030 results : `dict` [ `ResourcePath`, `bool` ]
1031 An entry for every URI in ``uris``. A URI that cannot be checked
1032 is reported as absent, and does not prevent the URIs after it in
1033 the batch from being checked.
1034 """
1035 results: dict[ResourcePath, bool] = {}
1036 for uri in uris:
1037 try:
1038 results[uri] = uri.exists()
1039 except Exception:
1040 results[uri] = False
1041 return results
1043 @classmethod
1044 def mtransfer(
1045 cls,
1046 transfer: str,
1047 from_to: Iterable[tuple[ResourcePath, ResourcePath]],
1048 overwrite: bool = False,
1049 transaction: TransactionProtocol | None = None,
1050 do_raise: bool = True,
1051 ) -> dict[ResourcePath, MBulkResult]:
1052 """Transfer many files in bulk.
1054 Parameters
1055 ----------
1056 transfer : `str`
1057 Mode to use for transferring the resource. Generically there are
1058 many standard options: copy, link, symlink, hardlink, relsymlink.
1059 Not all URIs support all modes.
1060 from_to : `list` [ `tuple` [ `ResourcePath`, `ResourcePath` ] ]
1061 A sequence of the source URIs and the target URIs.
1062 overwrite : `bool`, optional
1063 Allow an existing file to be overwritten. Defaults to `False`.
1064 transaction : `~lsst.resources.utils.TransactionProtocol`, optional
1065 A transaction object that can (depending on implementation)
1066 rollback transfers on error. Not guaranteed to be implemented.
1067 The transaction object must be thread safe.
1068 do_raise : `bool`, optional
1069 If `True` an `ExceptionGroup` will be raised containing any
1070 exceptions raised by the individual transfers. If `False`, or if
1071 there were no exceptions, a dict reporting the status of each
1072 `ResourcePath` will be returned.
1074 Returns
1075 -------
1076 copy_status : `dict` [ `ResourcePath`, `MBulkResult` ]
1077 A dict of all the transfer attempts with a value indicating
1078 whether the transfer succeeded for the target URI. If ``do_raise``
1079 is `True`, this will only be returned if there are no errors.
1080 """
1081 # A transfer is driven by the target, so group by the target scheme
1082 # and let each scheme decide how many workers to use.
1083 grouped: dict[type[ResourcePath], list[tuple[ResourcePath, ResourcePath]]] = defaultdict(list)
1084 for from_uri, to_uri in from_to:
1085 grouped[to_uri.__class__].append((from_uri, to_uri))
1087 results: dict[ResourcePath, MBulkResult] = {}
1088 for uri_class, group in grouped.items():
1089 results.update(
1090 uri_class._mtransfer(transfer, group, overwrite=overwrite, transaction=transaction)
1091 )
1093 if do_raise and any(not res.success for res in results.values()):
1094 raise ExceptionGroup(
1095 f"Errors transferring {len(results)} artifacts",
1096 tuple(res.exception for res in results.values() if res.exception is not None),
1097 )
1099 return results
1101 @classmethod
1102 def _mtransfer(
1103 cls,
1104 transfer: str,
1105 from_to: Iterable[tuple[ResourcePath, ResourcePath]],
1106 *,
1107 overwrite: bool = False,
1108 transaction: TransactionProtocol | None = None,
1109 ) -> dict[ResourcePath, MBulkResult]:
1110 """Transfer many files in bulk to targets of this scheme.
1112 Implementation helper method for `mtransfer`.
1114 Parameters
1115 ----------
1116 transfer : `str`
1117 Mode to use for transferring the resource.
1118 from_to : iterable [ `tuple` [ `ResourcePath`, `ResourcePath` ] ]
1119 A sequence of the source URIs and the target URIs.
1120 overwrite : `bool`, optional
1121 Allow an existing file to be overwritten. Defaults to `False`.
1122 transaction : `~lsst.resources.utils.TransactionProtocol`, optional
1123 A transaction object that can (depending on implementation)
1124 rollback transfers on error. Not guaranteed to be implemented.
1125 The transaction object must be thread safe.
1127 Returns
1128 -------
1129 copy_status : `dict` [ `ResourcePath`, `MBulkResult` ]
1130 A dict of all the transfer attempts with a value indicating
1131 whether the transfer succeeded for the target URI.
1132 """
1133 pairs = list(from_to)
1134 max_workers = _get_num_workers(cls._max_workers)
1135 chunks = cls._chunk_work(pairs, cls._transfer_chunk_size)
1136 if not chunks: 1136 ↛ 1137line 1136 didn't jump to line 1137 because the condition on line 1136 was never true
1137 return {}
1138 if len(chunks) == 1:
1139 # Not enough work to be worth handing to another thread.
1140 return cls._transfer_chunk(chunks[0], transfer, overwrite, transaction)
1142 results: dict[ResourcePath, MBulkResult] = {}
1143 with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as transfer_executor:
1144 future_transfers = {
1145 transfer_executor.submit(cls._transfer_chunk, chunk, transfer, overwrite, transaction): chunk
1146 for chunk in chunks
1147 }
1148 for future in concurrent.futures.as_completed(future_transfers):
1149 try:
1150 results.update(future.result())
1151 except Exception as e:
1152 # The chunk failed as a whole, for example because the
1153 # pool could not start a thread.
1154 for _, to_uri in future_transfers[future]:
1155 results[to_uri] = MBulkResult(False, e)
1156 return results
1158 @classmethod
1159 def _transfer_chunk(
1160 cls,
1161 from_to: tuple[tuple[ResourcePath, ResourcePath], ...],
1162 transfer: str,
1163 overwrite: bool,
1164 transaction: TransactionProtocol | None,
1165 ) -> dict[ResourcePath, MBulkResult]:
1166 """Transfer a batch of files, reporting each result independently.
1168 Parameters
1169 ----------
1170 from_to : `tuple` [ `tuple` [ `ResourcePath`, `ResourcePath` ], ... ]
1171 The source and target URIs to transfer.
1172 transfer : `str`
1173 Mode to use for transferring the resource.
1174 overwrite : `bool`
1175 Allow an existing file to be overwritten.
1176 transaction : `~lsst.resources.utils.TransactionProtocol` or `None`
1177 A transaction object that can (depending on implementation)
1178 rollback transfers on error.
1180 Returns
1181 -------
1182 results : `dict` [ `ResourcePath`, `MBulkResult` ]
1183 An entry for every target URI in ``from_to``. A transfer that
1184 fails does not prevent the transfers after it in the batch.
1185 """
1186 results: dict[ResourcePath, MBulkResult] = {}
1187 for from_uri, to_uri in from_to:
1188 try:
1189 to_uri.transfer_from(
1190 from_uri,
1191 transfer=transfer,
1192 overwrite=overwrite,
1193 transaction=transaction,
1194 multithreaded=False,
1195 )
1196 except Exception as e:
1197 results[to_uri] = MBulkResult(False, e)
1198 else:
1199 results[to_uri] = MBulkResult(True, None)
1200 return results
1202 def remove(self) -> None:
1203 """Remove the resource."""
1204 raise NotImplementedError()
1206 @classmethod
1207 def mremove(
1208 cls, uris: Iterable[ResourcePath], *, do_raise: bool = True
1209 ) -> dict[ResourcePath, MBulkResult]:
1210 """Remove multiple URIs at once.
1212 Parameters
1213 ----------
1214 uris : iterable of `ResourcePath`
1215 URIs to remove.
1216 do_raise : `bool`, optional
1217 If `True` an `ExceptionGroup` will be raised containing any
1218 exceptions raised by the individual transfers. If `False`, or if
1219 there were no exceptions, a dict reporting the status of each
1220 `ResourcePath` will be returned.
1222 Returns
1223 -------
1224 results : `dict` [ `ResourcePath`, `MBulkResult` ]
1225 Dictionary mapping each URI to a result object indicating whether
1226 the removal succeeded or resulted in an exception. If ``do_raise``
1227 is `True` this will only be returned if everything succeeded.
1228 """
1229 # Group URIs by scheme since some URI schemes support native bulk
1230 # APIs.
1231 results: dict[ResourcePath, MBulkResult] = {}
1232 for uri_class, group in cls._group_uris(uris).items():
1233 results.update(uri_class._mremove(group))
1234 if do_raise:
1235 failed = any(not r.success for r in results.values())
1236 if failed: 1236 ↛ 1237line 1236 didn't jump to line 1237 because the condition on line 1236 was never true
1237 s = "s" if len(results) != 1 else ""
1238 raise ExceptionGroup(
1239 f"Error{s} removing {len(results)} artifact{s}",
1240 tuple(res.exception for res in results.values() if res.exception is not None),
1241 )
1243 return results
1245 @staticmethod
1246 def _chunk_work(items: list[_T], chunk_size: int) -> list[tuple[_T, ...]]:
1247 """Split work items into batches of a fixed size.
1249 Parameters
1250 ----------
1251 items : `list`
1252 The work items to split.
1253 chunk_size : `int`
1254 Number of items to put in each batch.
1256 Returns
1257 -------
1258 chunks : `list` [ `tuple` ]
1259 The batches. Empty if ``items`` is empty. A single batch means the
1260 work is not worth spreading, and callers run it directly.
1262 Notes
1263 -----
1264 The batch size does not depend on how many items there are. Sizing it
1265 from the total would make a batch grow without bound as the total
1266 grows, and a batch that draws a run of slow URIs then stalls a worker
1267 for the rest of the operation with no way to rebalance. Asking for
1268 more batches than there are workers is harmless, since a pool only
1269 starts a thread when there is a batch waiting for it.
1270 """
1271 if not items:
1272 return []
1273 return list(chunk_iterable(items, chunk_size=chunk_size))
1275 @classmethod
1276 def _remove_chunk(cls, uris: tuple[ResourcePath, ...]) -> dict[ResourcePath, MBulkResult]:
1277 """Remove a batch of URIs, reporting each result independently.
1279 Parameters
1280 ----------
1281 uris : `tuple` [ `ResourcePath`, ... ]
1282 The URIs to remove.
1284 Returns
1285 -------
1286 results : `dict` [ `ResourcePath`, `MBulkResult` ]
1287 An entry for every URI in ``uris``. A URI that cannot be removed
1288 does not prevent the removal of the URIs after it.
1289 """
1290 results: dict[ResourcePath, MBulkResult] = {}
1291 for uri in uris:
1292 try:
1293 uri.remove()
1294 except Exception as e:
1295 results[uri] = MBulkResult(False, e)
1296 else:
1297 results[uri] = MBulkResult(True, None)
1298 return results
1300 @classmethod
1301 def _mremove(cls, uris: Iterable[ResourcePath]) -> dict[ResourcePath, MBulkResult]:
1302 """Remove multiple URIs using threads.
1304 Implementation helper method for `mremove`.
1306 Parameters
1307 ----------
1308 uris : iterable of `ResourcePath`
1309 The URIs to remove.
1311 Returns
1312 -------
1313 removal : `dict` of [`ResourcePath`, `MBulkResult`]
1314 Mapping of original URI to the result of removing it.
1315 """
1316 uri_list = list(uris)
1317 max_workers = _get_num_workers(cls._max_workers)
1318 chunks = cls._chunk_work(uri_list, cls._chunk_size)
1319 if not chunks: 1319 ↛ 1320line 1319 didn't jump to line 1320 because the condition on line 1319 was never true
1320 return {}
1321 if len(chunks) == 1:
1322 # Not enough work to be worth handing to another thread.
1323 return cls._remove_chunk(chunks[0])
1325 results: dict[ResourcePath, MBulkResult] = {}
1326 with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as remove_executor:
1327 future_remove = {remove_executor.submit(cls._remove_chunk, chunk): chunk for chunk in chunks}
1328 for future in concurrent.futures.as_completed(future_remove):
1329 try:
1330 results.update(future.result())
1331 except Exception as e:
1332 # The chunk failed as a whole, for example because the
1333 # pool could not start a thread.
1334 for uri in future_remove[future]:
1335 results[uri] = MBulkResult(False, e)
1336 return results
1338 def isabs(self) -> bool:
1339 """Indicate that the resource is fully specified.
1341 For non-schemeless URIs this is always true.
1343 Returns
1344 -------
1345 isabs : `bool`
1346 `True` in all cases except schemeless URI.
1347 """
1348 return True
1350 def abspath(self) -> ResourcePath:
1351 """Return URI using an absolute path.
1353 Returns
1354 -------
1355 abs : `ResourcePath`
1356 Absolute URI. For non-schemeless URIs this always returns itself.
1357 Schemeless URIs are upgraded to file URIs.
1358 """
1359 return self
1361 @contextlib.contextmanager
1362 def _as_local(
1363 self, multithreaded: bool = True, tmpdir: ResourcePath | None = None
1364 ) -> Generator[ResourcePath]:
1365 """Return the location of the (possibly remote) resource as local file.
1367 This is a helper function for `as_local` context manager.
1369 Parameters
1370 ----------
1371 multithreaded : `bool`, optional
1372 If `True` the transfer will be allowed to attempt to improve
1373 throughput by using parallel download streams. This may of no
1374 effect if the URI scheme does not support parallel streams or
1375 if a global override has been applied. If `False` parallel
1376 streams will be disabled.
1377 tmpdir : `ResourcePath` or `None`, optional
1378 Explicit override of the temporary directory to use for remote
1379 downloads.
1381 Returns
1382 -------
1383 local_uri : `ResourcePath`
1384 A URI to a local POSIX file. This can either be the same resource
1385 or a local downloaded copy of the resource.
1386 """
1387 raise NotImplementedError()
1389 @contextlib.contextmanager
1390 def as_local(
1391 self, multithreaded: bool = True, tmpdir: ResourcePathExpression | None = None
1392 ) -> Generator[ResourcePath]:
1393 """Return the location of the (possibly remote) resource as local file.
1395 Parameters
1396 ----------
1397 multithreaded : `bool`, optional
1398 If `True` the transfer will be allowed to attempt to improve
1399 throughput by using parallel download streams. This may of no
1400 effect if the URI scheme does not support parallel streams or
1401 if a global override has been applied. If `False` parallel
1402 streams will be disabled.
1403 tmpdir : `lsst.resources.ResourcePathExpression` or `None`, optional
1404 Explicit override of the temporary directory to use for remote
1405 downloads. This directory must be a local POSIX directory and
1406 must exist.
1408 Yields
1409 ------
1410 local : `ResourcePath`
1411 If this is a remote resource, it will be a copy of the resource
1412 on the local file system, probably in a temporary directory.
1413 For a local resource this should be the actual path to the
1414 resource.
1416 Notes
1417 -----
1418 The context manager will automatically delete any local temporary
1419 file.
1421 Examples
1422 --------
1423 Should be used as a context manager:
1425 .. code-block:: py
1427 with uri.as_local() as local:
1428 ospath = local.ospath
1429 """
1430 if self.isdir():
1431 raise IsADirectoryError(f"Directory-like URI {self} cannot be fetched as local.")
1432 temp_dir = ResourcePath(tmpdir, forceDirectory=True) if tmpdir is not None else None
1433 if temp_dir is not None and not temp_dir.isLocal:
1434 raise ValueError(f"Temporary directory for as_local must be local resource not {temp_dir}")
1435 with self._as_local(multithreaded=multithreaded, tmpdir=temp_dir) as local_uri:
1436 yield local_uri
1438 @classmethod
1439 @contextlib.contextmanager
1440 def temporary_uri(
1441 cls,
1442 prefix: ResourcePath | None = None,
1443 suffix: str | None = None,
1444 delete: bool = True,
1445 ) -> Generator[ResourcePath]:
1446 """Create a temporary file-like URI.
1448 Parameters
1449 ----------
1450 prefix : `ResourcePath`, optional
1451 Temporary directory to use (can be any scheme). Without this the
1452 path will be formed as a local file URI in a temporary directory
1453 obtained from `lsst.resources.utils.get_tempdir`. Ensuring that the
1454 prefix location exists is the responsibility of the caller.
1455 suffix : `str`, optional
1456 A file suffix to be used. The ``.`` should be included in this
1457 suffix.
1458 delete : `bool`, optional
1459 By default the resource will be deleted when the context manager
1460 is exited. Setting this flag to `False` will leave the resource
1461 alone.
1463 Yields
1464 ------
1465 uri : `ResourcePath`
1466 The temporary URI. Will be removed when the context is completed.
1467 """
1468 if prefix is None:
1469 prefix = ResourcePath(get_tempdir(), forceDirectory=True)
1471 # Need to create a randomized file name. For consistency do not
1472 # use mkstemp for local and something else for remote. Additionally
1473 # this method does not create the file to prevent name clashes.
1474 characters = "abcdefghijklmnopqrstuvwxyz0123456789_"
1475 rng = Random()
1476 tempname = "".join(rng.choice(characters) for _ in range(16))
1477 if suffix:
1478 tempname += suffix
1479 temporary_uri = prefix.join(tempname, isTemporary=True)
1480 if temporary_uri.isdir():
1481 # If we had a safe way to clean up a remote temporary directory, we
1482 # could support this.
1483 raise NotImplementedError("temporary_uri cannot be used to create a temporary directory.")
1484 try:
1485 yield temporary_uri
1486 finally:
1487 if delete:
1488 with contextlib.suppress(FileNotFoundError):
1489 # It's okay if this does not work because the user
1490 # removed the file.
1491 temporary_uri.remove()
1493 def read(self, size: int = -1) -> bytes:
1494 """Open the resource and return the contents in bytes.
1496 Parameters
1497 ----------
1498 size : `int`, optional
1499 The number of bytes to read. Negative or omitted indicates
1500 that all data should be read.
1501 """
1502 raise NotImplementedError()
1504 def write(self, data: bytes, overwrite: bool = True) -> None:
1505 """Write the supplied bytes to the new resource.
1507 Parameters
1508 ----------
1509 data : `bytes`
1510 The bytes to write to the resource. The entire contents of the
1511 resource will be replaced.
1512 overwrite : `bool`, optional
1513 If `True` the resource will be overwritten if it exists. Otherwise
1514 the write will fail.
1515 """
1516 raise NotImplementedError()
1518 def mkdir(self) -> None:
1519 """For a dir-like URI, create the directory resource if needed."""
1520 raise NotImplementedError()
1522 def isdir(self) -> bool:
1523 """Return True if this URI looks like a directory, else False."""
1524 return bool(self.dirLike)
1526 def size(self) -> int:
1527 """For non-dir-like URI, return the size of the resource.
1529 Returns
1530 -------
1531 sz : `int`
1532 The size in bytes of the resource associated with this URI.
1533 Returns 0 if dir-like.
1534 """
1535 raise NotImplementedError()
1537 def __str__(self) -> str:
1538 """Convert the URI to its native string form."""
1539 return self.geturl()
1541 def __repr__(self) -> str:
1542 """Return string representation suitable for evaluation."""
1543 return f'ResourcePath("{self.geturl()}")'
1545 def __eq__(self, other: Any) -> bool:
1546 """Compare supplied object with this `ResourcePath`."""
1547 if not isinstance(other, ResourcePath):
1548 return NotImplemented
1549 return self.geturl() == other.geturl()
1551 def __hash__(self) -> int:
1552 """Return hash of this object."""
1553 return hash(str(self))
1555 def __lt__(self, other: ResourcePath) -> bool:
1556 return self.geturl() < other.geturl()
1558 def __le__(self, other: ResourcePath) -> bool:
1559 return self.geturl() <= other.geturl()
1561 def __gt__(self, other: ResourcePath) -> bool:
1562 return self.geturl() > other.geturl()
1564 def __ge__(self, other: ResourcePath) -> bool:
1565 return self.geturl() >= other.geturl()
1567 def __copy__(self) -> ResourcePath:
1568 """Copy constructor.
1570 Object is immutable so copy can return itself.
1571 """
1572 # Implement here because the __new__ method confuses things
1573 return self
1575 def __deepcopy__(self, memo: Any) -> ResourcePath:
1576 """Deepcopy the object.
1578 Object is immutable so copy can return itself.
1579 """
1580 # Implement here because the __new__ method confuses things
1581 return self
1583 def __getnewargs__(self) -> tuple:
1584 """Support pickling."""
1585 return (str(self),)
1587 @classmethod
1588 def _fixDirectorySep(
1589 cls, parsed: urllib.parse.ParseResult, forceDirectory: bool | None = None
1590 ) -> tuple[urllib.parse.ParseResult, bool | None]:
1591 """Ensure that a path separator is present on directory paths.
1593 Parameters
1594 ----------
1595 parsed : `~urllib.parse.ParseResult`
1596 The result from parsing a URI using `urllib.parse`.
1597 forceDirectory : `bool` or `None`, optional
1598 If `True` forces the URI to end with a separator, otherwise given
1599 URI is interpreted as is. Specifying that the URI is conceptually
1600 equivalent to a directory can break some ambiguities when
1601 interpreting the last element of a path.
1603 Returns
1604 -------
1605 modified : `~urllib.parse.ParseResult`
1606 Update result if a URI is being handled.
1607 dirLike : `bool` or `None`
1608 `True` if given parsed URI has a trailing separator or
1609 ``forceDirectory`` is `True`. Otherwise returns the given value of
1610 ``forceDirectory``.
1611 """
1612 # Assume the forceDirectory flag can give us a clue.
1613 dirLike = forceDirectory
1615 # Directory separator
1616 sep = cls._pathModule.sep
1618 # URI is dir-like if explicitly stated or if it ends on a separator
1619 endsOnSep = parsed.path.endswith(sep)
1621 if forceDirectory is False and endsOnSep:
1622 raise ValueError(
1623 f"URI {parsed.geturl()} ends with {sep} but "
1624 "forceDirectory parameter declares it to be a file."
1625 )
1627 if forceDirectory or endsOnSep:
1628 dirLike = True
1629 # only add the separator if it's not already there
1630 if not endsOnSep:
1631 parsed = parsed._replace(path=parsed.path + sep)
1633 return parsed, dirLike
1635 @classmethod
1636 def _fixupPathUri(
1637 cls,
1638 parsed: urllib.parse.ParseResult,
1639 root: ResourcePath | None = None,
1640 forceAbsolute: bool = False,
1641 forceDirectory: bool | None = None,
1642 ) -> tuple[urllib.parse.ParseResult, bool | None]:
1643 """Correct any issues with the supplied URI.
1645 Parameters
1646 ----------
1647 parsed : `~urllib.parse.ParseResult`
1648 The result from parsing a URI using `urllib.parse`.
1649 root : `ResourcePath`, ignored
1650 Not used by the this implementation since all URIs are
1651 absolute except for those representing the local file system.
1652 forceAbsolute : `bool`, ignored.
1653 Not used by this implementation. URIs are generally always
1654 absolute.
1655 forceDirectory : `bool` or `None`, optional
1656 If `True` forces the URI to end with a separator, otherwise given
1657 URI is interpreted as is. Specifying that the URI is conceptually
1658 equivalent to a directory can break some ambiguities when
1659 interpreting the last element of a path.
1661 Returns
1662 -------
1663 modified : `~urllib.parse.ParseResult`
1664 Update result if a URI is being handled.
1665 dirLike : `bool`
1666 `True` if given parsed URI has a trailing separator or
1667 ``forceDirectory`` is `True`. Otherwise returns the given value
1668 of ``forceDirectory``.
1670 Notes
1671 -----
1672 Relative paths are explicitly not supported by RFC8089 but `urllib`
1673 does accept URIs of the form ``file:relative/path.ext``. They need
1674 to be turned into absolute paths before they can be used. This is
1675 always done regardless of the ``forceAbsolute`` parameter.
1677 AWS S3 differentiates between keys with trailing POSIX separators (i.e
1678 ``/dir`` and ``/dir/``) whereas POSIX does not necessarily.
1680 Scheme-less paths are normalized.
1681 """
1682 return cls._fixDirectorySep(parsed, forceDirectory)
1684 def transfer_from(
1685 self,
1686 src: ResourcePath,
1687 transfer: str,
1688 overwrite: bool = False,
1689 transaction: TransactionProtocol | None = None,
1690 multithreaded: bool = True,
1691 ) -> None:
1692 """Transfer to this URI from another.
1694 Parameters
1695 ----------
1696 src : `ResourcePath`
1697 Source URI.
1698 transfer : `str`
1699 Mode to use for transferring the resource. Generically there are
1700 many standard options: copy, link, symlink, hardlink, relsymlink.
1701 Not all URIs support all modes.
1702 overwrite : `bool`, optional
1703 Allow an existing file to be overwritten. Defaults to `False`.
1704 transaction : `~lsst.resources.utils.TransactionProtocol`, optional
1705 A transaction object that can (depending on implementation)
1706 rollback transfers on error. Not guaranteed to be implemented.
1707 multithreaded : `bool`, optional
1708 If `True` the transfer will be allowed to attempt to improve
1709 throughput by using parallel download streams. This may of no
1710 effect if the URI scheme does not support parallel streams or
1711 if a global override has been applied. If `False` parallel
1712 streams will be disabled.
1714 Notes
1715 -----
1716 Conceptually this is hard to scale as the number of URI schemes
1717 grow. The destination URI is more important than the source URI
1718 since that is where all the transfer modes are relevant (with the
1719 complication that "move" deletes the source).
1721 Local file to local file is the fundamental use case but every
1722 other scheme has to support "copy" to local file (with implicit
1723 support for "move") and copy from local file.
1724 All the "link" options tend to be specific to local file systems.
1726 "move" is a "copy" where the remote resource is deleted at the end.
1727 Whether this works depends on the source URI rather than the
1728 destination URI. Reverting a move on transaction rollback is
1729 expected to be problematic if a remote resource was involved.
1730 """
1731 raise NotImplementedError(f"No transfer modes supported by URI scheme {self.scheme}")
1733 def walk(
1734 self, file_filter: str | re.Pattern | None = None
1735 ) -> Iterator[list | tuple[ResourcePath, list[str], list[str]]]:
1736 """Walk the directory tree returning matching files and directories.
1738 Parameters
1739 ----------
1740 file_filter : `str` or `re.Pattern`, optional
1741 Regex to filter out files from the list before it is returned.
1743 Yields
1744 ------
1745 dirpath : `ResourcePath`
1746 Current directory being examined.
1747 dirnames : `list` of `str`
1748 Names of subdirectories within dirpath.
1749 filenames : `list` of `str`
1750 Names of all the files within dirpath.
1751 """
1752 raise NotImplementedError()
1754 @overload
1755 @classmethod
1756 def findFileResources( 1756 ↛ exitline 1756 didn't return from function 'findFileResources' because
1757 cls,
1758 candidates: Iterable[ResourcePathExpression],
1759 file_filter: str | re.Pattern | None,
1760 grouped: Literal[True],
1761 ) -> Iterator[Iterator[ResourcePath]]: ...
1763 @overload
1764 @classmethod
1765 def findFileResources( 1765 ↛ exitline 1765 didn't return from function 'findFileResources' because
1766 cls,
1767 candidates: Iterable[ResourcePathExpression],
1768 *,
1769 grouped: Literal[True],
1770 ) -> Iterator[Iterator[ResourcePath]]: ...
1772 @overload
1773 @classmethod
1774 def findFileResources( 1774 ↛ exitline 1774 didn't return from function 'findFileResources' because
1775 cls,
1776 candidates: Iterable[ResourcePathExpression],
1777 file_filter: str | re.Pattern | None = None,
1778 grouped: Literal[False] = False,
1779 ) -> Iterator[ResourcePath]: ...
1781 @classmethod
1782 def findFileResources(
1783 cls,
1784 candidates: Iterable[ResourcePathExpression],
1785 file_filter: str | re.Pattern | None = None,
1786 grouped: bool = False,
1787 ) -> Iterator[ResourcePath | Iterator[ResourcePath]]:
1788 """Get all the files from a list of values.
1790 Parameters
1791 ----------
1792 candidates : iterable [`str` or `ResourcePath`]
1793 The files to return and directories in which to look for files to
1794 return.
1795 file_filter : `str` or `re.Pattern`, optional
1796 The regex to use when searching for files within directories.
1797 By default returns all the found files.
1798 grouped : `bool`, optional
1799 If `True` the results will be grouped by directory and each
1800 yielded value will be an iterator over URIs. If `False` each
1801 URI will be returned separately.
1803 Yields
1804 ------
1805 found_file: `ResourcePath`
1806 The passed-in URIs and URIs found in passed-in directories.
1807 If grouping is enabled, each of the yielded values will be an
1808 iterator yielding members of the group. Files given explicitly
1809 will be returned as a single group at the end.
1811 Notes
1812 -----
1813 If a value is a file it is yielded immediately without checking that it
1814 exists. If a value is a directory, all the files in the directory
1815 (recursively) that match the regex will be yielded in turn.
1816 """
1817 fileRegex = None if file_filter is None else re.compile(file_filter)
1819 singles = []
1821 # Find all the files of interest
1822 for location in candidates:
1823 uri = ResourcePath(location)
1824 if uri.isdir():
1825 for found in uri.walk(fileRegex):
1826 if not found: 1826 ↛ 1829line 1826 didn't jump to line 1829 because the condition on line 1826 was never true
1827 # This means the uri does not exist and by
1828 # convention we ignore it
1829 continue
1830 root, dirs, files = found
1831 if not files:
1832 continue
1833 if grouped:
1834 yield (root.join(name) for name in files)
1835 else:
1836 for name in files:
1837 yield root.join(name)
1838 else:
1839 if grouped:
1840 singles.append(uri)
1841 else:
1842 yield uri
1844 # Finally, return any explicitly given files in one group
1845 if grouped and singles:
1846 yield iter(singles)
1848 @contextlib.contextmanager
1849 def open(
1850 self,
1851 mode: str = "r",
1852 *,
1853 encoding: str | None = None,
1854 prefer_file_temporary: bool = False,
1855 ) -> Generator[ResourceHandleProtocol]:
1856 """Return a context manager that wraps an object that behaves like an
1857 open file at the location of the URI.
1859 Parameters
1860 ----------
1861 mode : `str`
1862 String indicating the mode in which to open the file. Values are
1863 the same as those accepted by `open`, though intrinsically
1864 read-only URI types may only support read modes, and
1865 `io.IOBase.seekable` is not guaranteed to be `True` on the returned
1866 object.
1867 encoding : `str`, optional
1868 Unicode encoding for text IO; ignored for binary IO. Defaults to
1869 ``locale.getpreferredencoding(False)``, just as `open`
1870 does.
1871 prefer_file_temporary : `bool`, optional
1872 If `True`, for implementations that require transfers from a remote
1873 system to temporary local storage and/or back, use a temporary file
1874 instead of an in-memory buffer; this is generally slower, but it
1875 may be necessary to avoid excessive memory usage by large files.
1876 Ignored by implementations that do not require a temporary.
1878 Yields
1879 ------
1880 cm : `~contextlib.AbstractContextManager`
1881 A context manager that wraps a `ResourceHandleProtocol` file-like
1882 object.
1884 Notes
1885 -----
1886 The default implementation of this method uses a local temporary buffer
1887 (in-memory or file, depending on ``prefer_file_temporary``) with calls
1888 to `read`, `write`, `as_local`, and `transfer_from` as necessary to
1889 read and write from/to remote systems. Remote writes thus occur only
1890 when the context manager is exited. `ResourcePath` implementations
1891 that can return a more efficient native buffer should do so whenever
1892 possible (as is guaranteed for local files). `ResourcePath`
1893 implementations for which `as_local` does not return a temporary are
1894 required to reimplement `open`, though they may delegate to `super`
1895 when ``prefer_file_temporary`` is `False`.
1896 """
1897 if self.isdir():
1898 raise IsADirectoryError(f"Directory-like URI {self} cannot be opened.")
1899 if "x" in mode and self.exists():
1900 raise FileExistsError(f"File at {self} already exists.")
1901 if prefer_file_temporary:
1902 if "r" in mode or "a" in mode:
1903 local_cm = self.as_local()
1904 else:
1905 local_cm = self.temporary_uri(suffix=self.getExtension())
1906 with local_cm as local_uri:
1907 assert local_uri.isTemporary, (
1908 "ResourcePath implementations for which as_local is not "
1909 "a temporary must reimplement `open`."
1910 )
1911 # An encoding is ignored for binary IO, so do not let the
1912 # builtin open() reject it.
1913 encoding_arg = None if "b" in mode else encoding
1914 with open(local_uri.ospath, mode=mode, encoding=encoding_arg) as file_buffer:
1915 if "a" in mode:
1916 file_buffer.seek(0, io.SEEK_END)
1917 yield file_buffer
1918 if "r" not in mode or "+" in mode:
1919 self.transfer_from(local_uri, transfer="copy", overwrite=("x" not in mode))
1920 else:
1921 with self._openImpl(mode, encoding=encoding) as handle:
1922 yield handle
1924 @contextlib.contextmanager
1925 def _openImpl(self, mode: str = "r", *, encoding: str | None = None) -> Generator[ResourceHandleProtocol]:
1926 """Implement opening of a resource handle.
1928 This private method may be overridden by specific `ResourcePath`
1929 implementations to provide a customized handle like interface.
1931 Parameters
1932 ----------
1933 mode : `str`
1934 The mode the handle should be opened with
1935 encoding : `str`, optional
1936 The byte encoding of any binary text
1938 Yields
1939 ------
1940 handle : `~._resourceHandles.BaseResourceHandle`
1941 A handle that conforms to the
1942 `~._resourceHandles.BaseResourceHandle` interface
1944 Notes
1945 -----
1946 The base implementation of a file handle reads in a files entire
1947 contents into a buffer for manipulation, and then writes it back out
1948 upon close. Subclasses of this class may offer more fine grained
1949 control.
1950 """
1951 in_bytes = self.read() if "r" in mode or "a" in mode else b""
1952 if "b" in mode: 1952 ↛ 1960line 1952 didn't jump to line 1960 because the condition on line 1952 was always true
1953 bytes_buffer = io.BytesIO(in_bytes)
1954 bytes_buffer.name = str(self)
1955 if "a" in mode: 1955 ↛ 1956line 1955 didn't jump to line 1956 because the condition on line 1955 was never true
1956 bytes_buffer.seek(0, io.SEEK_END)
1957 yield bytes_buffer
1958 out_bytes = bytes_buffer.getvalue()
1959 else:
1960 if encoding is None:
1961 encoding = locale.getpreferredencoding(False)
1962 str_buffer = io.StringIO(in_bytes.decode(encoding))
1963 str_buffer.name = str(self)
1964 if "a" in mode:
1965 str_buffer.seek(0, io.SEEK_END)
1966 yield str_buffer
1967 out_bytes = str_buffer.getvalue().encode(encoding)
1968 if "r" not in mode or "+" in mode: 1968 ↛ 1969line 1968 didn't jump to line 1969 because the condition on line 1968 was never true
1969 self.write(out_bytes, overwrite=("x" not in mode))
1971 def generate_presigned_get_url(self, *, expiration_time_seconds: int) -> str:
1972 """Return a pre-signed URL that can be used to retrieve this resource
1973 using an HTTP GET without supplying any access credentials.
1975 Parameters
1976 ----------
1977 expiration_time_seconds : `int`
1978 Number of seconds until the generated URL is no longer valid.
1980 Returns
1981 -------
1982 url : `str`
1983 HTTP URL signed for GET.
1984 """
1985 raise NotImplementedError(f"URL signing is not supported for '{self.scheme}'")
1987 def generate_presigned_put_url(self, *, expiration_time_seconds: int) -> str:
1988 """Return a pre-signed URL that can be used to upload a file to this
1989 path using an HTTP PUT without supplying any access credentials.
1991 Parameters
1992 ----------
1993 expiration_time_seconds : `int`
1994 Number of seconds until the generated URL is no longer valid.
1996 Returns
1997 -------
1998 url : `str`
1999 HTTP URL signed for PUT.
2000 """
2001 raise NotImplementedError(f"URL signing is not supported for '{self.scheme}'")
2003 def _copy_extra_attributes(self, original_uri: ResourcePath) -> None:
2004 # May be overridden by subclasses to transfer attributes when a
2005 # ResourcePath is constructed using the "clone" version of the
2006 # ResourcePath constructor by passing in a ResourcePath object.
2007 pass
2009 def get_info(self) -> ResourceInfo:
2010 """Return lightweight metadata about this resource.
2012 Returns
2013 -------
2014 info : `ResourceInfo`
2015 The information about this resource that can be obtained from
2016 the backend. Will not read the file contents.
2017 """
2018 raise NotImplementedError("")
2021ResourcePathExpression = str | urllib.parse.ParseResult | ResourcePath | Path
2022"""Type-annotation alias for objects that can be coerced to ResourcePath.
2023"""