Coverage for python/lsst/resources/_resourcePath.py: 92%

597 statements  

« prev     ^ index     » next       coverage.py v7.16.1, created at 2026-09-22 09:28 +0000

1# This file is part of lsst-resources. 

2# 

3# Developed for the LSST Data Management System. 

4# This product includes software developed by the LSST Project 

5# (https://www.lsst.org). 

6# See the COPYRIGHT file at the top-level directory of this distribution 

7# for details of code ownership. 

8# 

9# Use of this source code is governed by a 3-clause BSD-style 

10# license that can be found in the LICENSE file. 

11 

12from __future__ import annotations 

13 

14__all__ = ("ResourceInfo", "ResourcePath", "ResourcePathExpression") 

15 

16import concurrent.futures 

17import contextlib 

18import copy 

19import dataclasses 

20import datetime 

21import io 

22import locale 

23import logging 

24import os 

25import posixpath 

26import re 

27import sys 

28import urllib.parse 

29from collections import defaultdict 

30from pathlib import Path, PurePath, PurePosixPath 

31from random import Random 

32from typing import TYPE_CHECKING 

33 

34try: 

35 import fsspec 

36 from fsspec.spec import AbstractFileSystem 

37except ImportError: 

38 # Hidden from type checkers so that the names above keep the types they 

39 # have when fsspec is installed. 

40 if not TYPE_CHECKING: 

41 fsspec = None 

42 AbstractFileSystem = type 

43 

44from collections.abc import Generator, Iterable, Iterator 

45from typing import Any, Literal, NamedTuple, TypeVar, overload 

46 

47from lsst.utils.iteration import chunk_iterable 

48 

49from ._resourceHandles._baseResourceHandle import ResourceHandleProtocol 

50from .utils import MAX_WORKERS, _get_num_workers, get_tempdir 

51 

52if TYPE_CHECKING: 

53 from .utils import TransactionProtocol 

54 

55 

56log = logging.getLogger(__name__) 

57 

58# Regex for looking for URI escapes 

59ESCAPES_RE = re.compile(r"%[A-F0-9]{2}") 

60 

61# Precomputed escaped hash 

62ESCAPED_HASH = urllib.parse.quote("#") 

63 

64_T = TypeVar("_T") 

65 

66 

67class MBulkResult(NamedTuple): 

68 """Report on a bulk operation.""" 

69 

70 success: bool 

71 exception: Exception | None 

72 

73 

74@dataclasses.dataclass(frozen=True) 

75class ResourceInfo: 

76 """Information about this resource.""" 

77 

78 uri: str 

79 """URI in string form of the resource from which this information is 

80 derived. 

81 """ 

82 is_file: bool 

83 """Indicate whether the resource is a file or a directory.""" 

84 size: int 

85 """Size of the file in bytes. A directory or a URI that has no concept 

86 of size returns 0.""" 

87 last_modified: datetime.datetime | None 

88 """Modification date of the resource, if known.""" 

89 checksums: dict[str, Any] 

90 """Checksums for this file. Supported checksum implementations are 

91 backend dependent. 

92 """ 

93 

94 

95class ResourcePath: # numpydoc ignore=PR02 

96 """Convenience wrapper around URI parsers. 

97 

98 Provides access to URI components and can convert file 

99 paths into absolute path URIs. Scheme-less URIs are treated as if 

100 they are local file system paths and are converted to absolute URIs. 

101 

102 A specialist subclass is created for each supported URI scheme. 

103 

104 Parameters 

105 ---------- 

106 uri : `str`, `pathlib.Path`, `urllib.parse.ParseResult`, or `ResourcePath` 

107 URI in string form. Can be scheme-less if referring to a relative 

108 path or an absolute path on the local file system. 

109 root : `str` or `ResourcePath`, optional 

110 When fixing up a relative path in a ``file`` scheme or if scheme-less, 

111 use this as the root. Must be absolute. If `None` the current 

112 working directory will be used. Can be any supported URI scheme. 

113 Not used if ``forceAbsolute`` is `False`. 

114 forceAbsolute : `bool`, optional 

115 If `True`, scheme-less relative URI will be converted to an absolute 

116 path using a ``file`` scheme. If `False` scheme-less URI will remain 

117 scheme-less and will not be updated to ``file`` or absolute path unless 

118 it is already an absolute path, in which case it will be updated to 

119 a ``file`` scheme. 

120 forceDirectory : `bool` or `None`, optional 

121 If `True` forces the URI to end with a separator. If `False` the URI 

122 is interpreted as a file-like entity. Default, `None`, is that the 

123 given URI is interpreted as a directory if there is a trailing ``/`` or 

124 for some schemes the system will check to see if it is a file or a 

125 directory. 

126 isTemporary : `bool`, optional 

127 If `True` indicates that this URI points to a temporary resource. 

128 The default is `False`, unless ``uri`` is already a `ResourcePath` 

129 instance and ``uri.isTemporary is True``. 

130 

131 Notes 

132 ----- 

133 A non-standard URI of the form ``file:dir/file.txt`` is always converted 

134 to an absolute ``file`` URI. 

135 """ 

136 

137 _pathLib: type[PurePath] = PurePosixPath 

138 """Path library to use for this scheme.""" 

139 

140 _pathModule = posixpath 

141 """Path module to use for this scheme.""" 

142 

143 transferModes: tuple[str, ...] = ("copy", "auto", "move") 

144 """Transfer modes supported by this implementation. 

145 

146 Move is special in that it is generally a copy followed by an unlink. 

147 Whether that unlink works depends critically on whether the source URI 

148 implements unlink. If it does not the move will be reported as a failure. 

149 """ 

150 

151 transferDefault: str = "copy" 

152 """Default mode to use for transferring if ``auto`` is specified.""" 

153 

154 quotePaths = True 

155 """True if path-like elements modifying a URI should be quoted. 

156 

157 All non-schemeless URIs have to internally use quoted paths. Therefore 

158 if a new file name is given (e.g. to updatedFile or join) a decision must 

159 be made whether to quote it to be consistent. 

160 """ 

161 

162 isLocal = False 

163 """If `True` this URI refers to a local file.""" 

164 

165 _max_workers: int = MAX_WORKERS 

166 """Upper bound on workers for parallel operations on this scheme. 

167 

168 Schemes backed by a connection pool keep this modest because the pool is 

169 sized to match it; schemes with no pool can raise it. 

170 """ 

171 

172 _chunk_size: int = 1 

173 """Number of URIs given to each worker by `mexists` and `mremove`. 

174 

175 A batch that fits in a single chunk is handled in the calling thread 

176 rather than being sent to a pool. The default suits a scheme where every 

177 operation is a network round trip, which is expensive enough that handing 

178 over one URI at a time costs nothing and gives the pool the freedom to 

179 balance itself. A scheme whose operations are cheap should raise it until 

180 a chunk is worth handing over. 

181 """ 

182 

183 _transfer_chunk_size: int = 1 

184 """Number of files given to each worker by `mtransfer`. 

185 

186 Kept separate from ``_chunk_size`` because a transfer costs orders of 

187 magnitude more than an existence check on the same scheme, and its cost 

188 scales with a file size the caller does not know in advance, so chunks 

189 have to stay small enough for the pool queue to balance them. 

190 """ 

191 

192 # This is not an ABC with abstract methods because the __new__ being 

193 # a factory confuses mypy such that it assumes that every constructor 

194 # returns a ResourcePath and then determines that all the abstract methods 

195 # are still abstract. If they are not marked abstract but just raise 

196 # mypy is fine with it. 

197 

198 # mypy is confused without these 

199 _uri: urllib.parse.ParseResult 

200 isTemporary: bool 

201 dirLike: bool | None 

202 """Whether the resource looks like a directory resource. `None` means that 

203 the status is uncertain.""" 

204 

205 def __new__( 

206 cls, 

207 uri: ResourcePathExpression, 

208 root: str | ResourcePath | None = None, 

209 forceAbsolute: bool = True, 

210 forceDirectory: bool | None = None, 

211 isTemporary: bool | None = None, 

212 ) -> ResourcePath: 

213 """Create and return new specialist ResourcePath subclass.""" 

214 parsed: urllib.parse.ParseResult 

215 dirLike: bool | None = forceDirectory 

216 subclass: type[ResourcePath] | None = None 

217 

218 # Force root to be a ResourcePath -- this simplifies downstream 

219 # code. 

220 if root is None: 

221 root_uri = None 

222 elif isinstance(root, str): 

223 root_uri = ResourcePath(root, forceDirectory=True, forceAbsolute=True) 

224 else: 

225 root_uri = root 

226 

227 if isinstance(uri, os.PathLike): 

228 uri = str(uri) 

229 

230 # Record if we need to post process the URI components 

231 # or if the instance is already fully configured 

232 if isinstance(uri, str): 

233 # Since local file names can have special characters in them 

234 # we need to quote them for the parser but we can unquote 

235 # later. Assume that all other URI schemes are quoted. 

236 # Since sometimes people write file:/a/b and not file:///a/b 

237 # we should not quote in the explicit case of file: 

238 if "://" not in uri and not uri.startswith("file:"): 

239 if ESCAPES_RE.search(uri): 

240 log.warning("Possible double encoding of %s", uri) 

241 else: 

242 # Fragments are generally not encoded so we must search 

243 # for the fragment boundary ourselves. This is making 

244 # an assumption that the filename does not include a "#" 

245 # and also that there is no "/" in the fragment itself. 

246 to_encode = uri 

247 fragment = "" 

248 if "#" in uri: 

249 dirpos = uri.rfind("/") 

250 trailing = uri[dirpos + 1 :] 

251 hashpos = trailing.rfind("#") 

252 if hashpos != -1: 

253 fragment = trailing[hashpos:] 

254 to_encode = uri[: dirpos + hashpos + 1] 

255 

256 uri = urllib.parse.quote(to_encode) + fragment 

257 

258 parsed = urllib.parse.urlparse(uri) 

259 elif isinstance(uri, urllib.parse.ParseResult): 

260 parsed = copy.copy(uri) 

261 # If we are being instantiated with a subclass, rather than 

262 # ResourcePath, ensure that that subclass is used directly. 

263 # This could lead to inconsistencies if this constructor 

264 # is used externally outside of the ResourcePath.replace() method. 

265 # S3ResourcePath(urllib.parse.urlparse("file://a/b.txt")) 

266 # will be a problem. 

267 # This is needed to prevent a schemeless absolute URI become 

268 # a file URI unexpectedly when calling updatedFile or 

269 # updatedExtension 

270 if cls is not ResourcePath: 

271 parsed, dirLike = cls._fixDirectorySep(parsed, forceDirectory) 

272 subclass = cls 

273 

274 elif isinstance(uri, ResourcePath): 

275 # Since ResourcePath is immutable we can return the argument 

276 # unchanged if it already agrees with forceDirectory, isTemporary, 

277 # and forceAbsolute. 

278 # We invoke __new__ again with str(self) to add a scheme for 

279 # forceAbsolute, but for the others that seems more likely to paper 

280 # over logic errors than do something useful, so we just raise. 

281 if forceDirectory is not None and uri.dirLike is not None and forceDirectory is not uri.dirLike: 

282 # Can not force a file-like URI to become a dir-like one or 

283 # vice versa. 

284 raise RuntimeError( 

285 f"{uri} can not be forced to change directory vs file state when previously declared." 

286 ) 

287 if isTemporary is not None and isTemporary is not uri.isTemporary: 

288 raise RuntimeError( 

289 f"{uri} is already a {'temporary' if uri.isTemporary else 'permanent'} " 

290 f"ResourcePath; cannot make it {'temporary' if isTemporary else 'permanent'}." 

291 ) 

292 

293 if forceAbsolute and not uri.scheme: 

294 # Create new absolute from relative. 

295 return ResourcePath( 

296 str(uri), 

297 root=root, 

298 forceAbsolute=forceAbsolute, 

299 forceDirectory=forceDirectory or uri.dirLike, 

300 isTemporary=uri.isTemporary, 

301 ) 

302 elif forceDirectory is not None and uri.dirLike is None: 

303 # Clone but with a new dirLike status. 

304 return uri.replace(forceDirectory=forceDirectory) 

305 return uri 

306 else: 

307 raise ValueError( 

308 f"Supplied URI must be string, Path, ResourcePath, or ParseResult but got '{uri!r}'" 

309 ) 

310 

311 if subclass is None: 

312 # Work out the subclass from the URI scheme 

313 if not parsed.scheme: 

314 # Root may be specified as a ResourcePath that overrides 

315 # the schemeless determination. 

316 if ( 

317 root_uri is not None 

318 and root_uri.scheme != "file" # file scheme has different code path 

319 and not parsed.path.startswith("/") # Not already absolute path 

320 ): 

321 if root_uri.dirLike is False: 

322 raise ValueError( 

323 f"Root URI ({root}) was not a directory so can not be joined with" 

324 f" path {parsed.path!r}" 

325 ) 

326 # If root is temporary or this schemeless is temporary we 

327 # assume this URI is temporary. 

328 isTemporary = isTemporary or root_uri.isTemporary 

329 joined = root_uri.join( 

330 parsed.path, forceDirectory=forceDirectory, isTemporary=isTemporary 

331 ) 

332 

333 # Rather than returning this new ResourcePath directly we 

334 # instead extract the path and the scheme and adjust the 

335 # URI we were given -- we need to do this to preserve 

336 # fragments since join() will drop them. 

337 parsed = parsed._replace(scheme=joined.scheme, path=joined.path, netloc=joined.netloc) 

338 subclass = type(joined) 

339 

340 # Clear the root parameter to indicate that it has 

341 # been applied already. 

342 root_uri = None 

343 else: 

344 from .schemeless import SchemelessResourcePath 

345 

346 subclass = SchemelessResourcePath 

347 elif parsed.scheme == "file": 

348 from .file import FileResourcePath 

349 

350 subclass = FileResourcePath 

351 elif parsed.scheme == "s3": 

352 from .s3 import S3ResourcePath 

353 

354 subclass = S3ResourcePath 

355 elif parsed.scheme.startswith("http"): 

356 from .http import HttpResourcePath 

357 

358 subclass = HttpResourcePath 

359 elif parsed.scheme in {"dav", "davs"}: 

360 from .dav import DavResourcePath 

361 

362 subclass = DavResourcePath 

363 elif parsed.scheme == "gs": 

364 from .gs import GSResourcePath 

365 

366 subclass = GSResourcePath 

367 elif parsed.scheme == "resource": 

368 # Rules for scheme names disallow pkg_resource 

369 from .packageresource import PackageResourcePath 

370 

371 subclass = PackageResourcePath 

372 elif parsed.scheme == "mem": 

373 # in-memory datastore object 

374 from .mem import InMemoryResourcePath 

375 

376 subclass = InMemoryResourcePath 

377 elif parsed.scheme == "eups": 

378 # EUPS package root. 

379 from .eups import EupsResourcePath 

380 

381 subclass = EupsResourcePath 

382 elif parsed.scheme == "remote-test": 

383 # EUPS package root. 

384 from .remote_test import RemoteTestResourcePath 

385 

386 subclass = RemoteTestResourcePath 

387 else: 

388 raise NotImplementedError( 

389 f"No URI support for scheme: '{parsed.scheme}' in {parsed.geturl()}" 

390 ) 

391 

392 parsed, dirLike = subclass._fixupPathUri( 

393 parsed, root=root_uri, forceAbsolute=forceAbsolute, forceDirectory=forceDirectory 

394 ) 

395 

396 # It is possible for the class to change from schemeless 

397 # to file or eups so handle that 

398 if parsed.scheme == "file": 

399 from .file import FileResourcePath 

400 

401 subclass = FileResourcePath 

402 elif parsed.scheme == "eups": 

403 from .eups import EupsResourcePath 

404 

405 subclass = EupsResourcePath 

406 

407 # Now create an instance of the correct subclass and set the 

408 # attributes directly 

409 self = object.__new__(subclass) 

410 self._uri = parsed 

411 self.dirLike = dirLike 

412 if isTemporary is None: 

413 isTemporary = False 

414 self.isTemporary = isTemporary 

415 self._set_proxy() 

416 return self 

417 

418 def _set_proxy(self) -> None: 

419 """Calculate internal proxy for externally visible resource path.""" 

420 pass 

421 

422 @property 

423 def scheme(self) -> str: 

424 """Return the URI scheme. 

425 

426 Notes 

427 ----- 

428 (``://`` is not part of the scheme). 

429 """ 

430 return self._uri.scheme 

431 

432 @property 

433 def netloc(self) -> str: 

434 """Return the URI network location.""" 

435 return self._uri.netloc 

436 

437 @property 

438 def path(self) -> str: 

439 """Return the path component of the URI.""" 

440 return self._uri.path 

441 

442 @property 

443 def unquoted_path(self) -> str: 

444 """Return path component of the URI with any URI quoting reversed.""" 

445 return urllib.parse.unquote(self._uri.path) 

446 

447 @property 

448 def ospath(self) -> str: 

449 """Return the path component of the URI localized to current OS.""" 

450 raise AttributeError(f"Non-file URI ({self}) has no local OS path.") 

451 

452 @property 

453 def relativeToPathRoot(self) -> str: 

454 """Return path relative to network location. 

455 

456 This is the path property with posix separator stripped 

457 from the left hand side of the path. 

458 

459 Always unquotes. 

460 """ 

461 relToRoot = self.path.lstrip("/") 

462 if relToRoot == "": 

463 return "./" 

464 return urllib.parse.unquote(relToRoot) 

465 

466 @property 

467 def is_root(self) -> bool: 

468 """Return whether this URI points to the root of the network location. 

469 

470 This means that the path components refers to the top level. 

471 """ 

472 relpath = self.relativeToPathRoot 

473 if relpath == "./": 

474 return True 

475 return False 

476 

477 @property 

478 def fragment(self) -> str: 

479 """Return the fragment component of the URI. May be quoted.""" 

480 return self._uri.fragment 

481 

482 @property 

483 def unquoted_fragment(self) -> str: 

484 """Return unquoted fragment.""" 

485 return urllib.parse.unquote(self.fragment) 

486 

487 @property 

488 def params(self) -> str: 

489 """Return any parameters included in the URI.""" 

490 return self._uri.params 

491 

492 @property 

493 def query(self) -> str: 

494 """Return any query strings included in the URI.""" 

495 return self._uri.query 

496 

497 def geturl(self) -> str: 

498 """Return the URI in string form. 

499 

500 Returns 

501 ------- 

502 url : `str` 

503 String form of URI. 

504 """ 

505 return self._uri.geturl() 

506 

507 def to_fsspec(self) -> tuple[AbstractFileSystem, str]: 

508 """Return an abstract file system and path that can be used by fsspec. 

509 

510 Returns 

511 ------- 

512 fs : `fsspec.spec.AbstractFileSystem` 

513 A file system object suitable for use with the returned path. 

514 path : `str` 

515 A path that can be opened by the file system object. 

516 """ 

517 if fsspec is None: 

518 raise ImportError("fsspec is not available") 

519 # By default give the URL to fsspec and hope. 

520 return fsspec.url_to_fs(self.geturl()) 

521 

522 def root_uri(self) -> ResourcePath: 

523 """Return the base root URI. 

524 

525 Returns 

526 ------- 

527 uri : `ResourcePath` 

528 Root URI. 

529 """ 

530 return self.replace(path="", query="", fragment="", params="", forceDirectory=True) 

531 

532 def split(self) -> tuple[ResourcePath, str]: 

533 """Split URI into head and tail. 

534 

535 Returns 

536 ------- 

537 head: `ResourcePath` 

538 Everything leading up to tail, expanded and normalized as per 

539 ResourcePath rules. 

540 tail : `str` 

541 Last path component. Tail will be empty if path ends on a 

542 separator or if the URI is known to be associated with a directory. 

543 Tail will never contain separators. It will be unquoted. 

544 

545 Notes 

546 ----- 

547 Equivalent to `os.path.split` where head preserves the URI 

548 components. In some cases this method can result in a file system 

549 check to verify whether the URI is a directory or not (only if 

550 ``forceDirectory`` was `None` during construction). For a scheme-less 

551 URI this can mean that the result might change depending on current 

552 working directory. 

553 """ 

554 if self.isdir(): 

555 # This is known to be a directory so must return itself and 

556 # the empty string. 

557 return self, "" 

558 

559 head, tail = self._pathModule.split(self.path) 

560 headuri = self._uri._replace(path=head, fragment="", query="", params="") 

561 

562 # The file part should never include quoted metacharacters 

563 tail = urllib.parse.unquote(tail) 

564 

565 # Schemeless is special in that it can be a relative path. 

566 # We need to ensure that it stays that way. All other URIs will 

567 # be absolute already. 

568 forceAbsolute = self.isabs() 

569 return ResourcePath(headuri, forceDirectory=True, forceAbsolute=forceAbsolute), tail 

570 

571 def basename(self) -> str: 

572 """Return the base name, last element of path, of the URI. 

573 

574 Returns 

575 ------- 

576 tail : `str` 

577 Last part of the path attribute. Trail will be empty if path ends 

578 on a separator. 

579 

580 Notes 

581 ----- 

582 If URI ends on a slash returns an empty string. This is the second 

583 element returned by `split()`. 

584 

585 Equivalent of `os.path.basename`. 

586 """ 

587 return self.split()[1] 

588 

589 def dirname(self) -> ResourcePath: 

590 """Return the directory component of the path as a new `ResourcePath`. 

591 

592 Returns 

593 ------- 

594 head : `ResourcePath` 

595 Everything except the tail of path attribute, expanded and 

596 normalized as per ResourcePath rules. 

597 

598 Notes 

599 ----- 

600 Equivalent of `os.path.dirname`. If this is a directory URI it will 

601 be returned unchanged. If the parent directory is always required 

602 use `parent`. 

603 """ 

604 return self.split()[0] 

605 

606 def parent(self) -> ResourcePath: 

607 """Return a `ResourcePath` of the parent directory. 

608 

609 Returns 

610 ------- 

611 head : `ResourcePath` 

612 Everything except the tail of path attribute, expanded and 

613 normalized as per `ResourcePath` rules. 

614 

615 Notes 

616 ----- 

617 For a file-like URI this will be the same as calling `dirname`. 

618 For a directory-like URI this will always return the parent directory 

619 whereas `dirname()` will return the original URI. This is consistent 

620 with `os.path.dirname` compared to the `pathlib.Path` property 

621 ``parent``. 

622 """ 

623 if self.dirLike is False: 

624 # os.path.split() is slightly faster than calling Path().parent. 

625 return self.dirname() 

626 # When self is dir-like, returns its parent directory, 

627 # regardless of the presence of a trailing separator 

628 originalPath = self._pathLib(self.path) 

629 parentPath = originalPath.parent 

630 return self.replace(path=str(parentPath), forceDirectory=True, fragment="", query="", params="") 

631 

632 def replace( 

633 self, forceDirectory: bool | None = None, isTemporary: bool = False, **kwargs: Any 

634 ) -> ResourcePath: 

635 """Return new `ResourcePath` with specified components replaced. 

636 

637 Parameters 

638 ---------- 

639 forceDirectory : `bool` or `None`, optional 

640 Parameter passed to ResourcePath constructor to force this 

641 new URI to be dir-like or file-like. 

642 isTemporary : `bool`, optional 

643 Indicate that the resulting URI is temporary resource. 

644 **kwargs 

645 Components of a `urllib.parse.ParseResult` that should be 

646 modified for the newly-created `ResourcePath`. 

647 

648 Returns 

649 ------- 

650 new : `ResourcePath` 

651 New `ResourcePath` object with updated values. 

652 

653 Notes 

654 ----- 

655 Does not, for now, allow a change in URI scheme. 

656 """ 

657 # Disallow a change in scheme 

658 if "scheme" in kwargs: 

659 raise ValueError(f"Can not use replace() method to change URI scheme for {self}") 

660 result = self.__class__( 

661 self._uri._replace(**kwargs), forceDirectory=forceDirectory, isTemporary=isTemporary 

662 ) 

663 result._copy_extra_attributes(self) 

664 return result 

665 

666 def updatedFile(self, newfile: str) -> ResourcePath: 

667 """Return new URI with an updated final component of the path. 

668 

669 Parameters 

670 ---------- 

671 newfile : `str` 

672 File name with no path component. 

673 

674 Returns 

675 ------- 

676 updated : `ResourcePath` 

677 Updated `ResourcePath` with new updated final component. 

678 

679 Notes 

680 ----- 

681 Forces the ``ResourcePath.dirLike`` attribute to be false. The new file 

682 path will be quoted if necessary. If the current URI is known to 

683 refer to a directory, the new file will be joined to the current file. 

684 It is recommended that this behavior no longer be used and a call 

685 to `isdir` by the caller should be used to decide whether to join or 

686 replace. In the future this method may be modified to always replace 

687 the final element of the path. 

688 """ 

689 if self.dirLike: 

690 return self.join(newfile, forceDirectory=False) 

691 return self.parent().join(newfile, forceDirectory=False) 

692 

693 def updatedExtension(self, ext: str | None) -> ResourcePath: 

694 """Return a new `ResourcePath` with updated file extension. 

695 

696 All file extensions are replaced. 

697 

698 Parameters 

699 ---------- 

700 ext : `str` or `None` 

701 New extension. If an empty string is given any extension will 

702 be removed. If `None` is given there will be no change. 

703 

704 Returns 

705 ------- 

706 updated : `ResourcePath` 

707 URI with the specified extension. Can return itself if 

708 no extension was specified. 

709 """ 

710 if ext is None: 

711 return self 

712 

713 # Get the extension 

714 current = self.getExtension() 

715 

716 # Nothing to do if the extension already matches 

717 if current == ext: 

718 return self 

719 

720 # Remove the current extension from the path 

721 # .fits.gz counts as one extension do not use os.path.splitext 

722 path = self.path 

723 if current: 

724 path = path.removesuffix(current) 

725 

726 # Ensure that we have a leading "." on file extension (and we do not 

727 # try to modify the empty string) 

728 if ext and not ext.startswith("."): 

729 ext = "." + ext 

730 

731 return self.replace(path=path + ext, forceDirectory=False) 

732 

733 def getExtension(self) -> str: 

734 """Return the extension(s) associated with this URI path. 

735 

736 Returns 

737 ------- 

738 ext : `str` 

739 The file extension (including the ``.``). Can be empty string 

740 if there is no file extension. Usually returns only the last 

741 file extension unless there is a special extension modifier 

742 indicating file compression, in which case the combined 

743 extension (e.g. ``.fits.gz``) will be returned. 

744 

745 Notes 

746 ----- 

747 Does not distinguish between file and directory URIs when determining 

748 a suffix. An extension is only determined from the final component 

749 of the path. 

750 """ 

751 special = {".gz", ".bz2", ".xz", ".fz"} 

752 

753 # path lib will ignore any "." in directories. 

754 # path lib works well: 

755 # extensions = self._pathLib(self.path).suffixes 

756 # But the constructor is slow. Therefore write our own implementation. 

757 # Strip trailing separator if present, do not care if this is a 

758 # directory or not. 

759 parts = self.path.rstrip("/").rsplit(self._pathModule.sep, 1) 

760 _, *extensions = parts[-1].split(".") 

761 

762 if not extensions: 

763 return "" 

764 extensions = ["." + x for x in extensions] 

765 

766 ext = extensions.pop() 

767 

768 # Multiple extensions, decide whether to include the final two 

769 if extensions and ext in special: 

770 ext = f"{extensions[-1]}{ext}" 

771 

772 return ext 

773 

774 def join( 

775 self, path: str | ResourcePath, isTemporary: bool | None = None, forceDirectory: bool | None = None 

776 ) -> ResourcePath: 

777 """Return new `ResourcePath` with additional path components. 

778 

779 Parameters 

780 ---------- 

781 path : `str`, `ResourcePath` 

782 Additional file components to append to the current URI. Will be 

783 quoted depending on the associated URI scheme. If the path looks 

784 like a URI referring to an absolute location, it will be returned 

785 directly (matching the behavior of `os.path.join`). It can 

786 also be a `ResourcePath`. Fragments are propagated. 

787 isTemporary : `bool`, optional 

788 Indicate that the resulting URI represents a temporary resource. 

789 Default is ``self.isTemporary``. 

790 forceDirectory : `bool` or `None`, optional 

791 If `True` forces the URI to end with a separator. If `False` the 

792 resultant URI is declared to refer to a file. `None` indicates 

793 that the file directory status is unknown. 

794 

795 Returns 

796 ------- 

797 new : `ResourcePath` 

798 New URI with the path appended. 

799 

800 Notes 

801 ----- 

802 Schemeless URIs assume local path separator but all other URIs assume 

803 POSIX separator if the supplied path has directory structure. It 

804 may be this never becomes a problem but datastore templates assume 

805 POSIX separator is being used. 

806 

807 If an absolute `ResourcePath` is given for ``path`` is is assumed that 

808 this should be returned directly. Giving a ``path`` of an absolute 

809 scheme-less URI is not allowed for safety reasons as it may indicate 

810 a mistake in the calling code. 

811 

812 It is an error to attempt to join to something that is known to 

813 refer to a file. Use `updatedFile` if the file is to be 

814 replaced. 

815 

816 If an unquoted ``#`` is included in the path it is assumed to be 

817 referring to a fragment and not part of the file name. 

818 

819 Raises 

820 ------ 

821 ValueError 

822 Raised if the given path object refers to a directory but the 

823 ``forceDirectory`` parameter insists the outcome should be a file, 

824 and vice versa. Also raised if the URI being joined with is known 

825 to refer to a file. 

826 RuntimeError 

827 Raised if this attempts to join a temporary URI to a non-temporary 

828 URI. 

829 """ 

830 if self.dirLike is False: 

831 raise ValueError("Can not join a new path component to a file.") 

832 if isTemporary is None: 

833 isTemporary = self.isTemporary 

834 elif not isTemporary and self.isTemporary: 

835 raise RuntimeError("Cannot join temporary URI to non-temporary URI.") 

836 # If we have a full URI in path we will use it directly 

837 # but without forcing to absolute so that we can trap the 

838 # expected option of relative path. 

839 path_uri = ResourcePath( 

840 path, forceAbsolute=False, forceDirectory=forceDirectory, isTemporary=isTemporary 

841 ) 

842 if forceDirectory is not None and path_uri.dirLike is not forceDirectory: 842 ↛ 843line 842 didn't jump to line 843 because the condition on line 842 was never true

843 raise ValueError( 

844 "The supplied path URI to join has inconsistent directory state " 

845 f"with forceDirectory parameter: {path_uri.dirLike} vs {forceDirectory}" 

846 ) 

847 forceDirectory = path_uri.dirLike 

848 

849 if path_uri.isabs(): 

850 # Absolute URI so return it directly. 

851 return path_uri 

852 

853 # We want to propagate fragments to the joined path and we rely on 

854 # the ResourcePath parser to find these fragments for us even in plain 

855 # strings. Must assume there are no `#` characters in filenames. 

856 if not isinstance(path, str) or path_uri.fragment: 

857 path = path_uri.unquoted_path 

858 

859 # Might need to quote the path. 

860 if self.quotePaths: 

861 path = urllib.parse.quote(path) 

862 

863 newpath = self._pathModule.normpath(self._pathModule.join(self.path, path)) 

864 

865 # normpath can strip trailing / so we force directory if the supplied 

866 # path ended with a / 

867 has_dir_sep = path.endswith(self._pathModule.sep) 

868 if forceDirectory is None and has_dir_sep: 868 ↛ 869line 868 didn't jump to line 869 because the condition on line 868 was never true

869 forceDirectory = True 

870 elif forceDirectory is False and has_dir_sep: 870 ↛ 871line 870 didn't jump to line 871 because the condition on line 870 was never true

871 raise ValueError("Path to join has trailing / but is being forced to be a file.") 

872 return self.replace( 

873 path=newpath, 

874 forceDirectory=forceDirectory, 

875 isTemporary=isTemporary, 

876 fragment=path_uri.fragment, 

877 query=path_uri.query, 

878 params=path_uri.params, 

879 ) 

880 

881 def relative_to(self, other: ResourcePath, walk_up: bool = False) -> str | None: 

882 """Return the relative path from this URI to the other URI. 

883 

884 Parameters 

885 ---------- 

886 other : `ResourcePath` 

887 URI to use to calculate the relative path. Must be a parent 

888 of this URI. 

889 walk_up : `bool`, optional 

890 Control whether "``..``" can be used to resolve a relative path. 

891 Default is `False`. Can not be `True` on Python version 3.11. 

892 

893 Returns 

894 ------- 

895 subpath : `str` 

896 The sub path of this URI relative to the supplied other URI. 

897 Returns `None` if there is no parent child relationship. 

898 Scheme and netloc must match. 

899 """ 

900 # Scheme-less self is handled elsewhere. 

901 if self.scheme != other.scheme: 

902 return None 

903 if self.netloc != other.netloc: 

904 # Special case for localhost vs empty string. 

905 # There can be many variants of localhost. 

906 local_netlocs = {"", "localhost", "localhost.localdomain", "127.0.0.1"} 

907 if not {self.netloc, other.netloc}.issubset(local_netlocs): 

908 return None 

909 

910 # Rather than trying to guess a failure reason from the TypeError 

911 # explicitly check for python 3.11. Doing this will simplify the 

912 # rediscovery of a useless python version check when we set a new 

913 # minimum version. 

914 kwargs = {} 

915 if walk_up: 

916 if sys.version_info < (3, 12, 0): 916 ↛ 917line 916 didn't jump to line 917 because the condition on line 916 was never true

917 raise TypeError("walk_up parameter can not be true in python 3.11 and older") 

918 

919 kwargs["walk_up"] = True 

920 

921 enclosed_path = self._pathLib(self.relativeToPathRoot) 

922 parent_path = other.relativeToPathRoot 

923 subpath: str | None 

924 try: 

925 subpath = str(enclosed_path.relative_to(parent_path, **kwargs)) 

926 except ValueError: 

927 subpath = None 

928 else: 

929 subpath = urllib.parse.unquote(subpath) 

930 return subpath 

931 

932 def exists(self) -> bool: 

933 """Indicate that the resource is available. 

934 

935 Returns 

936 ------- 

937 exists : `bool` 

938 `True` if the resource exists. 

939 """ 

940 raise NotImplementedError() 

941 

942 @classmethod 

943 def _group_uris(cls, uris: Iterable[ResourcePath]) -> dict[type[ResourcePath], list[ResourcePath]]: 

944 """Group URIs by class/scheme.""" 

945 grouped: dict[type[ResourcePath], list[ResourcePath]] = defaultdict(list) 

946 for uri in uris: 

947 grouped[uri.__class__].append(uri) 

948 return grouped 

949 

950 @classmethod 

951 def mexists( 

952 cls, uris: Iterable[ResourcePath], *, num_workers: int | None = None 

953 ) -> dict[ResourcePath, bool]: 

954 """Check for existence of multiple URIs at once. 

955 

956 Parameters 

957 ---------- 

958 uris : iterable of `ResourcePath` 

959 The URIs to test. 

960 num_workers : `int` or `None`, optional 

961 The number of parallel workers to use when checking for existence. 

962 If `None`, the default value will be taken from the environment 

963 and bounded by the limit for this scheme. 

964 

965 Returns 

966 ------- 

967 existence : `dict` of [`ResourcePath`, `bool`] 

968 Mapping of original URI to boolean indicating existence. 

969 """ 

970 existence: dict[ResourcePath, bool] = {} 

971 for uri_class, group in cls._group_uris(uris).items(): 

972 existence.update(uri_class._mexists(group, num_workers=num_workers)) 

973 

974 return existence 

975 

976 @classmethod 

977 def _mexists( 

978 cls, uris: Iterable[ResourcePath], *, num_workers: int | None = None 

979 ) -> dict[ResourcePath, bool]: 

980 """Check for existence of multiple URIs at once. 

981 

982 Implementation helper method for `mexists`. 

983 

984 Parameters 

985 ---------- 

986 uris : iterable of `ResourcePath` 

987 The URIs to test. 

988 num_workers : `int` or `None`, optional 

989 The number of parallel workers to use when checking for existence 

990 If `None`, the default value will be taken from the environment. 

991 

992 Returns 

993 ------- 

994 existence : `dict` of [`ResourcePath`, `bool`] 

995 Mapping of original URI to boolean indicating existence. 

996 """ 

997 uri_list = list(uris) 

998 max_workers = num_workers if num_workers is not None else _get_num_workers(cls._max_workers) 

999 chunks = cls._chunk_work(uri_list, cls._chunk_size) 

1000 if not chunks: 1000 ↛ 1001line 1000 didn't jump to line 1001 because the condition on line 1000 was never true

1001 return {} 

1002 if len(chunks) == 1: 

1003 # Not enough work to be worth handing to another thread. 

1004 return cls._exists_chunk(chunks[0]) 

1005 

1006 results: dict[ResourcePath, bool] = {} 

1007 with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as exists_executor: 

1008 future_exists = {exists_executor.submit(cls._exists_chunk, chunk): chunk for chunk in chunks} 

1009 for future in concurrent.futures.as_completed(future_exists): 

1010 try: 

1011 results.update(future.result()) 

1012 except Exception: 

1013 # The chunk failed as a whole, for example because the 

1014 # pool could not start a thread. 

1015 for uri in future_exists[future]: 

1016 results[uri] = False 

1017 return results 

1018 

1019 @classmethod 

1020 def _exists_chunk(cls, uris: tuple[ResourcePath, ...]) -> dict[ResourcePath, bool]: 

1021 """Check a batch of URIs for existence. 

1022 

1023 Parameters 

1024 ---------- 

1025 uris : `tuple` [ `ResourcePath`, ... ] 

1026 The URIs to check. 

1027 

1028 Returns 

1029 ------- 

1030 results : `dict` [ `ResourcePath`, `bool` ] 

1031 An entry for every URI in ``uris``. A URI that cannot be checked 

1032 is reported as absent, and does not prevent the URIs after it in 

1033 the batch from being checked. 

1034 """ 

1035 results: dict[ResourcePath, bool] = {} 

1036 for uri in uris: 

1037 try: 

1038 results[uri] = uri.exists() 

1039 except Exception: 

1040 results[uri] = False 

1041 return results 

1042 

1043 @classmethod 

1044 def mtransfer( 

1045 cls, 

1046 transfer: str, 

1047 from_to: Iterable[tuple[ResourcePath, ResourcePath]], 

1048 overwrite: bool = False, 

1049 transaction: TransactionProtocol | None = None, 

1050 do_raise: bool = True, 

1051 ) -> dict[ResourcePath, MBulkResult]: 

1052 """Transfer many files in bulk. 

1053 

1054 Parameters 

1055 ---------- 

1056 transfer : `str` 

1057 Mode to use for transferring the resource. Generically there are 

1058 many standard options: copy, link, symlink, hardlink, relsymlink. 

1059 Not all URIs support all modes. 

1060 from_to : `list` [ `tuple` [ `ResourcePath`, `ResourcePath` ] ] 

1061 A sequence of the source URIs and the target URIs. 

1062 overwrite : `bool`, optional 

1063 Allow an existing file to be overwritten. Defaults to `False`. 

1064 transaction : `~lsst.resources.utils.TransactionProtocol`, optional 

1065 A transaction object that can (depending on implementation) 

1066 rollback transfers on error. Not guaranteed to be implemented. 

1067 The transaction object must be thread safe. 

1068 do_raise : `bool`, optional 

1069 If `True` an `ExceptionGroup` will be raised containing any 

1070 exceptions raised by the individual transfers. If `False`, or if 

1071 there were no exceptions, a dict reporting the status of each 

1072 `ResourcePath` will be returned. 

1073 

1074 Returns 

1075 ------- 

1076 copy_status : `dict` [ `ResourcePath`, `MBulkResult` ] 

1077 A dict of all the transfer attempts with a value indicating 

1078 whether the transfer succeeded for the target URI. If ``do_raise`` 

1079 is `True`, this will only be returned if there are no errors. 

1080 """ 

1081 # A transfer is driven by the target, so group by the target scheme 

1082 # and let each scheme decide how many workers to use. 

1083 grouped: dict[type[ResourcePath], list[tuple[ResourcePath, ResourcePath]]] = defaultdict(list) 

1084 for from_uri, to_uri in from_to: 

1085 grouped[to_uri.__class__].append((from_uri, to_uri)) 

1086 

1087 results: dict[ResourcePath, MBulkResult] = {} 

1088 for uri_class, group in grouped.items(): 

1089 results.update( 

1090 uri_class._mtransfer(transfer, group, overwrite=overwrite, transaction=transaction) 

1091 ) 

1092 

1093 if do_raise and any(not res.success for res in results.values()): 

1094 raise ExceptionGroup( 

1095 f"Errors transferring {len(results)} artifacts", 

1096 tuple(res.exception for res in results.values() if res.exception is not None), 

1097 ) 

1098 

1099 return results 

1100 

1101 @classmethod 

1102 def _mtransfer( 

1103 cls, 

1104 transfer: str, 

1105 from_to: Iterable[tuple[ResourcePath, ResourcePath]], 

1106 *, 

1107 overwrite: bool = False, 

1108 transaction: TransactionProtocol | None = None, 

1109 ) -> dict[ResourcePath, MBulkResult]: 

1110 """Transfer many files in bulk to targets of this scheme. 

1111 

1112 Implementation helper method for `mtransfer`. 

1113 

1114 Parameters 

1115 ---------- 

1116 transfer : `str` 

1117 Mode to use for transferring the resource. 

1118 from_to : iterable [ `tuple` [ `ResourcePath`, `ResourcePath` ] ] 

1119 A sequence of the source URIs and the target URIs. 

1120 overwrite : `bool`, optional 

1121 Allow an existing file to be overwritten. Defaults to `False`. 

1122 transaction : `~lsst.resources.utils.TransactionProtocol`, optional 

1123 A transaction object that can (depending on implementation) 

1124 rollback transfers on error. Not guaranteed to be implemented. 

1125 The transaction object must be thread safe. 

1126 

1127 Returns 

1128 ------- 

1129 copy_status : `dict` [ `ResourcePath`, `MBulkResult` ] 

1130 A dict of all the transfer attempts with a value indicating 

1131 whether the transfer succeeded for the target URI. 

1132 """ 

1133 pairs = list(from_to) 

1134 max_workers = _get_num_workers(cls._max_workers) 

1135 chunks = cls._chunk_work(pairs, cls._transfer_chunk_size) 

1136 if not chunks: 1136 ↛ 1137line 1136 didn't jump to line 1137 because the condition on line 1136 was never true

1137 return {} 

1138 if len(chunks) == 1: 

1139 # Not enough work to be worth handing to another thread. 

1140 return cls._transfer_chunk(chunks[0], transfer, overwrite, transaction) 

1141 

1142 results: dict[ResourcePath, MBulkResult] = {} 

1143 with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as transfer_executor: 

1144 future_transfers = { 

1145 transfer_executor.submit(cls._transfer_chunk, chunk, transfer, overwrite, transaction): chunk 

1146 for chunk in chunks 

1147 } 

1148 for future in concurrent.futures.as_completed(future_transfers): 

1149 try: 

1150 results.update(future.result()) 

1151 except Exception as e: 

1152 # The chunk failed as a whole, for example because the 

1153 # pool could not start a thread. 

1154 for _, to_uri in future_transfers[future]: 

1155 results[to_uri] = MBulkResult(False, e) 

1156 return results 

1157 

1158 @classmethod 

1159 def _transfer_chunk( 

1160 cls, 

1161 from_to: tuple[tuple[ResourcePath, ResourcePath], ...], 

1162 transfer: str, 

1163 overwrite: bool, 

1164 transaction: TransactionProtocol | None, 

1165 ) -> dict[ResourcePath, MBulkResult]: 

1166 """Transfer a batch of files, reporting each result independently. 

1167 

1168 Parameters 

1169 ---------- 

1170 from_to : `tuple` [ `tuple` [ `ResourcePath`, `ResourcePath` ], ... ] 

1171 The source and target URIs to transfer. 

1172 transfer : `str` 

1173 Mode to use for transferring the resource. 

1174 overwrite : `bool` 

1175 Allow an existing file to be overwritten. 

1176 transaction : `~lsst.resources.utils.TransactionProtocol` or `None` 

1177 A transaction object that can (depending on implementation) 

1178 rollback transfers on error. 

1179 

1180 Returns 

1181 ------- 

1182 results : `dict` [ `ResourcePath`, `MBulkResult` ] 

1183 An entry for every target URI in ``from_to``. A transfer that 

1184 fails does not prevent the transfers after it in the batch. 

1185 """ 

1186 results: dict[ResourcePath, MBulkResult] = {} 

1187 for from_uri, to_uri in from_to: 

1188 try: 

1189 to_uri.transfer_from( 

1190 from_uri, 

1191 transfer=transfer, 

1192 overwrite=overwrite, 

1193 transaction=transaction, 

1194 multithreaded=False, 

1195 ) 

1196 except Exception as e: 

1197 results[to_uri] = MBulkResult(False, e) 

1198 else: 

1199 results[to_uri] = MBulkResult(True, None) 

1200 return results 

1201 

1202 def remove(self) -> None: 

1203 """Remove the resource.""" 

1204 raise NotImplementedError() 

1205 

1206 @classmethod 

1207 def mremove( 

1208 cls, uris: Iterable[ResourcePath], *, do_raise: bool = True 

1209 ) -> dict[ResourcePath, MBulkResult]: 

1210 """Remove multiple URIs at once. 

1211 

1212 Parameters 

1213 ---------- 

1214 uris : iterable of `ResourcePath` 

1215 URIs to remove. 

1216 do_raise : `bool`, optional 

1217 If `True` an `ExceptionGroup` will be raised containing any 

1218 exceptions raised by the individual transfers. If `False`, or if 

1219 there were no exceptions, a dict reporting the status of each 

1220 `ResourcePath` will be returned. 

1221 

1222 Returns 

1223 ------- 

1224 results : `dict` [ `ResourcePath`, `MBulkResult` ] 

1225 Dictionary mapping each URI to a result object indicating whether 

1226 the removal succeeded or resulted in an exception. If ``do_raise`` 

1227 is `True` this will only be returned if everything succeeded. 

1228 """ 

1229 # Group URIs by scheme since some URI schemes support native bulk 

1230 # APIs. 

1231 results: dict[ResourcePath, MBulkResult] = {} 

1232 for uri_class, group in cls._group_uris(uris).items(): 

1233 results.update(uri_class._mremove(group)) 

1234 if do_raise: 

1235 failed = any(not r.success for r in results.values()) 

1236 if failed: 1236 ↛ 1237line 1236 didn't jump to line 1237 because the condition on line 1236 was never true

1237 s = "s" if len(results) != 1 else "" 

1238 raise ExceptionGroup( 

1239 f"Error{s} removing {len(results)} artifact{s}", 

1240 tuple(res.exception for res in results.values() if res.exception is not None), 

1241 ) 

1242 

1243 return results 

1244 

1245 @staticmethod 

1246 def _chunk_work(items: list[_T], chunk_size: int) -> list[tuple[_T, ...]]: 

1247 """Split work items into batches of a fixed size. 

1248 

1249 Parameters 

1250 ---------- 

1251 items : `list` 

1252 The work items to split. 

1253 chunk_size : `int` 

1254 Number of items to put in each batch. 

1255 

1256 Returns 

1257 ------- 

1258 chunks : `list` [ `tuple` ] 

1259 The batches. Empty if ``items`` is empty. A single batch means the 

1260 work is not worth spreading, and callers run it directly. 

1261 

1262 Notes 

1263 ----- 

1264 The batch size does not depend on how many items there are. Sizing it 

1265 from the total would make a batch grow without bound as the total 

1266 grows, and a batch that draws a run of slow URIs then stalls a worker 

1267 for the rest of the operation with no way to rebalance. Asking for 

1268 more batches than there are workers is harmless, since a pool only 

1269 starts a thread when there is a batch waiting for it. 

1270 """ 

1271 if not items: 

1272 return [] 

1273 return list(chunk_iterable(items, chunk_size=chunk_size)) 

1274 

1275 @classmethod 

1276 def _remove_chunk(cls, uris: tuple[ResourcePath, ...]) -> dict[ResourcePath, MBulkResult]: 

1277 """Remove a batch of URIs, reporting each result independently. 

1278 

1279 Parameters 

1280 ---------- 

1281 uris : `tuple` [ `ResourcePath`, ... ] 

1282 The URIs to remove. 

1283 

1284 Returns 

1285 ------- 

1286 results : `dict` [ `ResourcePath`, `MBulkResult` ] 

1287 An entry for every URI in ``uris``. A URI that cannot be removed 

1288 does not prevent the removal of the URIs after it. 

1289 """ 

1290 results: dict[ResourcePath, MBulkResult] = {} 

1291 for uri in uris: 

1292 try: 

1293 uri.remove() 

1294 except Exception as e: 

1295 results[uri] = MBulkResult(False, e) 

1296 else: 

1297 results[uri] = MBulkResult(True, None) 

1298 return results 

1299 

1300 @classmethod 

1301 def _mremove(cls, uris: Iterable[ResourcePath]) -> dict[ResourcePath, MBulkResult]: 

1302 """Remove multiple URIs using threads. 

1303 

1304 Implementation helper method for `mremove`. 

1305 

1306 Parameters 

1307 ---------- 

1308 uris : iterable of `ResourcePath` 

1309 The URIs to remove. 

1310 

1311 Returns 

1312 ------- 

1313 removal : `dict` of [`ResourcePath`, `MBulkResult`] 

1314 Mapping of original URI to the result of removing it. 

1315 """ 

1316 uri_list = list(uris) 

1317 max_workers = _get_num_workers(cls._max_workers) 

1318 chunks = cls._chunk_work(uri_list, cls._chunk_size) 

1319 if not chunks: 1319 ↛ 1320line 1319 didn't jump to line 1320 because the condition on line 1319 was never true

1320 return {} 

1321 if len(chunks) == 1: 

1322 # Not enough work to be worth handing to another thread. 

1323 return cls._remove_chunk(chunks[0]) 

1324 

1325 results: dict[ResourcePath, MBulkResult] = {} 

1326 with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as remove_executor: 

1327 future_remove = {remove_executor.submit(cls._remove_chunk, chunk): chunk for chunk in chunks} 

1328 for future in concurrent.futures.as_completed(future_remove): 

1329 try: 

1330 results.update(future.result()) 

1331 except Exception as e: 

1332 # The chunk failed as a whole, for example because the 

1333 # pool could not start a thread. 

1334 for uri in future_remove[future]: 

1335 results[uri] = MBulkResult(False, e) 

1336 return results 

1337 

1338 def isabs(self) -> bool: 

1339 """Indicate that the resource is fully specified. 

1340 

1341 For non-schemeless URIs this is always true. 

1342 

1343 Returns 

1344 ------- 

1345 isabs : `bool` 

1346 `True` in all cases except schemeless URI. 

1347 """ 

1348 return True 

1349 

1350 def abspath(self) -> ResourcePath: 

1351 """Return URI using an absolute path. 

1352 

1353 Returns 

1354 ------- 

1355 abs : `ResourcePath` 

1356 Absolute URI. For non-schemeless URIs this always returns itself. 

1357 Schemeless URIs are upgraded to file URIs. 

1358 """ 

1359 return self 

1360 

1361 @contextlib.contextmanager 

1362 def _as_local( 

1363 self, multithreaded: bool = True, tmpdir: ResourcePath | None = None 

1364 ) -> Generator[ResourcePath]: 

1365 """Return the location of the (possibly remote) resource as local file. 

1366 

1367 This is a helper function for `as_local` context manager. 

1368 

1369 Parameters 

1370 ---------- 

1371 multithreaded : `bool`, optional 

1372 If `True` the transfer will be allowed to attempt to improve 

1373 throughput by using parallel download streams. This may of no 

1374 effect if the URI scheme does not support parallel streams or 

1375 if a global override has been applied. If `False` parallel 

1376 streams will be disabled. 

1377 tmpdir : `ResourcePath` or `None`, optional 

1378 Explicit override of the temporary directory to use for remote 

1379 downloads. 

1380 

1381 Returns 

1382 ------- 

1383 local_uri : `ResourcePath` 

1384 A URI to a local POSIX file. This can either be the same resource 

1385 or a local downloaded copy of the resource. 

1386 """ 

1387 raise NotImplementedError() 

1388 

1389 @contextlib.contextmanager 

1390 def as_local( 

1391 self, multithreaded: bool = True, tmpdir: ResourcePathExpression | None = None 

1392 ) -> Generator[ResourcePath]: 

1393 """Return the location of the (possibly remote) resource as local file. 

1394 

1395 Parameters 

1396 ---------- 

1397 multithreaded : `bool`, optional 

1398 If `True` the transfer will be allowed to attempt to improve 

1399 throughput by using parallel download streams. This may of no 

1400 effect if the URI scheme does not support parallel streams or 

1401 if a global override has been applied. If `False` parallel 

1402 streams will be disabled. 

1403 tmpdir : `lsst.resources.ResourcePathExpression` or `None`, optional 

1404 Explicit override of the temporary directory to use for remote 

1405 downloads. This directory must be a local POSIX directory and 

1406 must exist. 

1407 

1408 Yields 

1409 ------ 

1410 local : `ResourcePath` 

1411 If this is a remote resource, it will be a copy of the resource 

1412 on the local file system, probably in a temporary directory. 

1413 For a local resource this should be the actual path to the 

1414 resource. 

1415 

1416 Notes 

1417 ----- 

1418 The context manager will automatically delete any local temporary 

1419 file. 

1420 

1421 Examples 

1422 -------- 

1423 Should be used as a context manager: 

1424 

1425 .. code-block:: py 

1426 

1427 with uri.as_local() as local: 

1428 ospath = local.ospath 

1429 """ 

1430 if self.isdir(): 

1431 raise IsADirectoryError(f"Directory-like URI {self} cannot be fetched as local.") 

1432 temp_dir = ResourcePath(tmpdir, forceDirectory=True) if tmpdir is not None else None 

1433 if temp_dir is not None and not temp_dir.isLocal: 

1434 raise ValueError(f"Temporary directory for as_local must be local resource not {temp_dir}") 

1435 with self._as_local(multithreaded=multithreaded, tmpdir=temp_dir) as local_uri: 

1436 yield local_uri 

1437 

1438 @classmethod 

1439 @contextlib.contextmanager 

1440 def temporary_uri( 

1441 cls, 

1442 prefix: ResourcePath | None = None, 

1443 suffix: str | None = None, 

1444 delete: bool = True, 

1445 ) -> Generator[ResourcePath]: 

1446 """Create a temporary file-like URI. 

1447 

1448 Parameters 

1449 ---------- 

1450 prefix : `ResourcePath`, optional 

1451 Temporary directory to use (can be any scheme). Without this the 

1452 path will be formed as a local file URI in a temporary directory 

1453 obtained from `lsst.resources.utils.get_tempdir`. Ensuring that the 

1454 prefix location exists is the responsibility of the caller. 

1455 suffix : `str`, optional 

1456 A file suffix to be used. The ``.`` should be included in this 

1457 suffix. 

1458 delete : `bool`, optional 

1459 By default the resource will be deleted when the context manager 

1460 is exited. Setting this flag to `False` will leave the resource 

1461 alone. 

1462 

1463 Yields 

1464 ------ 

1465 uri : `ResourcePath` 

1466 The temporary URI. Will be removed when the context is completed. 

1467 """ 

1468 if prefix is None: 

1469 prefix = ResourcePath(get_tempdir(), forceDirectory=True) 

1470 

1471 # Need to create a randomized file name. For consistency do not 

1472 # use mkstemp for local and something else for remote. Additionally 

1473 # this method does not create the file to prevent name clashes. 

1474 characters = "abcdefghijklmnopqrstuvwxyz0123456789_" 

1475 rng = Random() 

1476 tempname = "".join(rng.choice(characters) for _ in range(16)) 

1477 if suffix: 

1478 tempname += suffix 

1479 temporary_uri = prefix.join(tempname, isTemporary=True) 

1480 if temporary_uri.isdir(): 

1481 # If we had a safe way to clean up a remote temporary directory, we 

1482 # could support this. 

1483 raise NotImplementedError("temporary_uri cannot be used to create a temporary directory.") 

1484 try: 

1485 yield temporary_uri 

1486 finally: 

1487 if delete: 

1488 with contextlib.suppress(FileNotFoundError): 

1489 # It's okay if this does not work because the user 

1490 # removed the file. 

1491 temporary_uri.remove() 

1492 

1493 def read(self, size: int = -1) -> bytes: 

1494 """Open the resource and return the contents in bytes. 

1495 

1496 Parameters 

1497 ---------- 

1498 size : `int`, optional 

1499 The number of bytes to read. Negative or omitted indicates 

1500 that all data should be read. 

1501 """ 

1502 raise NotImplementedError() 

1503 

1504 def write(self, data: bytes, overwrite: bool = True) -> None: 

1505 """Write the supplied bytes to the new resource. 

1506 

1507 Parameters 

1508 ---------- 

1509 data : `bytes` 

1510 The bytes to write to the resource. The entire contents of the 

1511 resource will be replaced. 

1512 overwrite : `bool`, optional 

1513 If `True` the resource will be overwritten if it exists. Otherwise 

1514 the write will fail. 

1515 """ 

1516 raise NotImplementedError() 

1517 

1518 def mkdir(self) -> None: 

1519 """For a dir-like URI, create the directory resource if needed.""" 

1520 raise NotImplementedError() 

1521 

1522 def isdir(self) -> bool: 

1523 """Return True if this URI looks like a directory, else False.""" 

1524 return bool(self.dirLike) 

1525 

1526 def size(self) -> int: 

1527 """For non-dir-like URI, return the size of the resource. 

1528 

1529 Returns 

1530 ------- 

1531 sz : `int` 

1532 The size in bytes of the resource associated with this URI. 

1533 Returns 0 if dir-like. 

1534 """ 

1535 raise NotImplementedError() 

1536 

1537 def __str__(self) -> str: 

1538 """Convert the URI to its native string form.""" 

1539 return self.geturl() 

1540 

1541 def __repr__(self) -> str: 

1542 """Return string representation suitable for evaluation.""" 

1543 return f'ResourcePath("{self.geturl()}")' 

1544 

1545 def __eq__(self, other: Any) -> bool: 

1546 """Compare supplied object with this `ResourcePath`.""" 

1547 if not isinstance(other, ResourcePath): 

1548 return NotImplemented 

1549 return self.geturl() == other.geturl() 

1550 

1551 def __hash__(self) -> int: 

1552 """Return hash of this object.""" 

1553 return hash(str(self)) 

1554 

1555 def __lt__(self, other: ResourcePath) -> bool: 

1556 return self.geturl() < other.geturl() 

1557 

1558 def __le__(self, other: ResourcePath) -> bool: 

1559 return self.geturl() <= other.geturl() 

1560 

1561 def __gt__(self, other: ResourcePath) -> bool: 

1562 return self.geturl() > other.geturl() 

1563 

1564 def __ge__(self, other: ResourcePath) -> bool: 

1565 return self.geturl() >= other.geturl() 

1566 

1567 def __copy__(self) -> ResourcePath: 

1568 """Copy constructor. 

1569 

1570 Object is immutable so copy can return itself. 

1571 """ 

1572 # Implement here because the __new__ method confuses things 

1573 return self 

1574 

1575 def __deepcopy__(self, memo: Any) -> ResourcePath: 

1576 """Deepcopy the object. 

1577 

1578 Object is immutable so copy can return itself. 

1579 """ 

1580 # Implement here because the __new__ method confuses things 

1581 return self 

1582 

1583 def __getnewargs__(self) -> tuple: 

1584 """Support pickling.""" 

1585 return (str(self),) 

1586 

1587 @classmethod 

1588 def _fixDirectorySep( 

1589 cls, parsed: urllib.parse.ParseResult, forceDirectory: bool | None = None 

1590 ) -> tuple[urllib.parse.ParseResult, bool | None]: 

1591 """Ensure that a path separator is present on directory paths. 

1592 

1593 Parameters 

1594 ---------- 

1595 parsed : `~urllib.parse.ParseResult` 

1596 The result from parsing a URI using `urllib.parse`. 

1597 forceDirectory : `bool` or `None`, optional 

1598 If `True` forces the URI to end with a separator, otherwise given 

1599 URI is interpreted as is. Specifying that the URI is conceptually 

1600 equivalent to a directory can break some ambiguities when 

1601 interpreting the last element of a path. 

1602 

1603 Returns 

1604 ------- 

1605 modified : `~urllib.parse.ParseResult` 

1606 Update result if a URI is being handled. 

1607 dirLike : `bool` or `None` 

1608 `True` if given parsed URI has a trailing separator or 

1609 ``forceDirectory`` is `True`. Otherwise returns the given value of 

1610 ``forceDirectory``. 

1611 """ 

1612 # Assume the forceDirectory flag can give us a clue. 

1613 dirLike = forceDirectory 

1614 

1615 # Directory separator 

1616 sep = cls._pathModule.sep 

1617 

1618 # URI is dir-like if explicitly stated or if it ends on a separator 

1619 endsOnSep = parsed.path.endswith(sep) 

1620 

1621 if forceDirectory is False and endsOnSep: 

1622 raise ValueError( 

1623 f"URI {parsed.geturl()} ends with {sep} but " 

1624 "forceDirectory parameter declares it to be a file." 

1625 ) 

1626 

1627 if forceDirectory or endsOnSep: 

1628 dirLike = True 

1629 # only add the separator if it's not already there 

1630 if not endsOnSep: 

1631 parsed = parsed._replace(path=parsed.path + sep) 

1632 

1633 return parsed, dirLike 

1634 

1635 @classmethod 

1636 def _fixupPathUri( 

1637 cls, 

1638 parsed: urllib.parse.ParseResult, 

1639 root: ResourcePath | None = None, 

1640 forceAbsolute: bool = False, 

1641 forceDirectory: bool | None = None, 

1642 ) -> tuple[urllib.parse.ParseResult, bool | None]: 

1643 """Correct any issues with the supplied URI. 

1644 

1645 Parameters 

1646 ---------- 

1647 parsed : `~urllib.parse.ParseResult` 

1648 The result from parsing a URI using `urllib.parse`. 

1649 root : `ResourcePath`, ignored 

1650 Not used by the this implementation since all URIs are 

1651 absolute except for those representing the local file system. 

1652 forceAbsolute : `bool`, ignored. 

1653 Not used by this implementation. URIs are generally always 

1654 absolute. 

1655 forceDirectory : `bool` or `None`, optional 

1656 If `True` forces the URI to end with a separator, otherwise given 

1657 URI is interpreted as is. Specifying that the URI is conceptually 

1658 equivalent to a directory can break some ambiguities when 

1659 interpreting the last element of a path. 

1660 

1661 Returns 

1662 ------- 

1663 modified : `~urllib.parse.ParseResult` 

1664 Update result if a URI is being handled. 

1665 dirLike : `bool` 

1666 `True` if given parsed URI has a trailing separator or 

1667 ``forceDirectory`` is `True`. Otherwise returns the given value 

1668 of ``forceDirectory``. 

1669 

1670 Notes 

1671 ----- 

1672 Relative paths are explicitly not supported by RFC8089 but `urllib` 

1673 does accept URIs of the form ``file:relative/path.ext``. They need 

1674 to be turned into absolute paths before they can be used. This is 

1675 always done regardless of the ``forceAbsolute`` parameter. 

1676 

1677 AWS S3 differentiates between keys with trailing POSIX separators (i.e 

1678 ``/dir`` and ``/dir/``) whereas POSIX does not necessarily. 

1679 

1680 Scheme-less paths are normalized. 

1681 """ 

1682 return cls._fixDirectorySep(parsed, forceDirectory) 

1683 

1684 def transfer_from( 

1685 self, 

1686 src: ResourcePath, 

1687 transfer: str, 

1688 overwrite: bool = False, 

1689 transaction: TransactionProtocol | None = None, 

1690 multithreaded: bool = True, 

1691 ) -> None: 

1692 """Transfer to this URI from another. 

1693 

1694 Parameters 

1695 ---------- 

1696 src : `ResourcePath` 

1697 Source URI. 

1698 transfer : `str` 

1699 Mode to use for transferring the resource. Generically there are 

1700 many standard options: copy, link, symlink, hardlink, relsymlink. 

1701 Not all URIs support all modes. 

1702 overwrite : `bool`, optional 

1703 Allow an existing file to be overwritten. Defaults to `False`. 

1704 transaction : `~lsst.resources.utils.TransactionProtocol`, optional 

1705 A transaction object that can (depending on implementation) 

1706 rollback transfers on error. Not guaranteed to be implemented. 

1707 multithreaded : `bool`, optional 

1708 If `True` the transfer will be allowed to attempt to improve 

1709 throughput by using parallel download streams. This may of no 

1710 effect if the URI scheme does not support parallel streams or 

1711 if a global override has been applied. If `False` parallel 

1712 streams will be disabled. 

1713 

1714 Notes 

1715 ----- 

1716 Conceptually this is hard to scale as the number of URI schemes 

1717 grow. The destination URI is more important than the source URI 

1718 since that is where all the transfer modes are relevant (with the 

1719 complication that "move" deletes the source). 

1720 

1721 Local file to local file is the fundamental use case but every 

1722 other scheme has to support "copy" to local file (with implicit 

1723 support for "move") and copy from local file. 

1724 All the "link" options tend to be specific to local file systems. 

1725 

1726 "move" is a "copy" where the remote resource is deleted at the end. 

1727 Whether this works depends on the source URI rather than the 

1728 destination URI. Reverting a move on transaction rollback is 

1729 expected to be problematic if a remote resource was involved. 

1730 """ 

1731 raise NotImplementedError(f"No transfer modes supported by URI scheme {self.scheme}") 

1732 

1733 def walk( 

1734 self, file_filter: str | re.Pattern | None = None 

1735 ) -> Iterator[list | tuple[ResourcePath, list[str], list[str]]]: 

1736 """Walk the directory tree returning matching files and directories. 

1737 

1738 Parameters 

1739 ---------- 

1740 file_filter : `str` or `re.Pattern`, optional 

1741 Regex to filter out files from the list before it is returned. 

1742 

1743 Yields 

1744 ------ 

1745 dirpath : `ResourcePath` 

1746 Current directory being examined. 

1747 dirnames : `list` of `str` 

1748 Names of subdirectories within dirpath. 

1749 filenames : `list` of `str` 

1750 Names of all the files within dirpath. 

1751 """ 

1752 raise NotImplementedError() 

1753 

1754 @overload 

1755 @classmethod 

1756 def findFileResources( 1756 ↛ exitline 1756 didn't return from function 'findFileResources' because

1757 cls, 

1758 candidates: Iterable[ResourcePathExpression], 

1759 file_filter: str | re.Pattern | None, 

1760 grouped: Literal[True], 

1761 ) -> Iterator[Iterator[ResourcePath]]: ... 

1762 

1763 @overload 

1764 @classmethod 

1765 def findFileResources( 1765 ↛ exitline 1765 didn't return from function 'findFileResources' because

1766 cls, 

1767 candidates: Iterable[ResourcePathExpression], 

1768 *, 

1769 grouped: Literal[True], 

1770 ) -> Iterator[Iterator[ResourcePath]]: ... 

1771 

1772 @overload 

1773 @classmethod 

1774 def findFileResources( 1774 ↛ exitline 1774 didn't return from function 'findFileResources' because

1775 cls, 

1776 candidates: Iterable[ResourcePathExpression], 

1777 file_filter: str | re.Pattern | None = None, 

1778 grouped: Literal[False] = False, 

1779 ) -> Iterator[ResourcePath]: ... 

1780 

1781 @classmethod 

1782 def findFileResources( 

1783 cls, 

1784 candidates: Iterable[ResourcePathExpression], 

1785 file_filter: str | re.Pattern | None = None, 

1786 grouped: bool = False, 

1787 ) -> Iterator[ResourcePath | Iterator[ResourcePath]]: 

1788 """Get all the files from a list of values. 

1789 

1790 Parameters 

1791 ---------- 

1792 candidates : iterable [`str` or `ResourcePath`] 

1793 The files to return and directories in which to look for files to 

1794 return. 

1795 file_filter : `str` or `re.Pattern`, optional 

1796 The regex to use when searching for files within directories. 

1797 By default returns all the found files. 

1798 grouped : `bool`, optional 

1799 If `True` the results will be grouped by directory and each 

1800 yielded value will be an iterator over URIs. If `False` each 

1801 URI will be returned separately. 

1802 

1803 Yields 

1804 ------ 

1805 found_file: `ResourcePath` 

1806 The passed-in URIs and URIs found in passed-in directories. 

1807 If grouping is enabled, each of the yielded values will be an 

1808 iterator yielding members of the group. Files given explicitly 

1809 will be returned as a single group at the end. 

1810 

1811 Notes 

1812 ----- 

1813 If a value is a file it is yielded immediately without checking that it 

1814 exists. If a value is a directory, all the files in the directory 

1815 (recursively) that match the regex will be yielded in turn. 

1816 """ 

1817 fileRegex = None if file_filter is None else re.compile(file_filter) 

1818 

1819 singles = [] 

1820 

1821 # Find all the files of interest 

1822 for location in candidates: 

1823 uri = ResourcePath(location) 

1824 if uri.isdir(): 

1825 for found in uri.walk(fileRegex): 

1826 if not found: 1826 ↛ 1829line 1826 didn't jump to line 1829 because the condition on line 1826 was never true

1827 # This means the uri does not exist and by 

1828 # convention we ignore it 

1829 continue 

1830 root, dirs, files = found 

1831 if not files: 

1832 continue 

1833 if grouped: 

1834 yield (root.join(name) for name in files) 

1835 else: 

1836 for name in files: 

1837 yield root.join(name) 

1838 else: 

1839 if grouped: 

1840 singles.append(uri) 

1841 else: 

1842 yield uri 

1843 

1844 # Finally, return any explicitly given files in one group 

1845 if grouped and singles: 

1846 yield iter(singles) 

1847 

1848 @contextlib.contextmanager 

1849 def open( 

1850 self, 

1851 mode: str = "r", 

1852 *, 

1853 encoding: str | None = None, 

1854 prefer_file_temporary: bool = False, 

1855 ) -> Generator[ResourceHandleProtocol]: 

1856 """Return a context manager that wraps an object that behaves like an 

1857 open file at the location of the URI. 

1858 

1859 Parameters 

1860 ---------- 

1861 mode : `str` 

1862 String indicating the mode in which to open the file. Values are 

1863 the same as those accepted by `open`, though intrinsically 

1864 read-only URI types may only support read modes, and 

1865 `io.IOBase.seekable` is not guaranteed to be `True` on the returned 

1866 object. 

1867 encoding : `str`, optional 

1868 Unicode encoding for text IO; ignored for binary IO. Defaults to 

1869 ``locale.getpreferredencoding(False)``, just as `open` 

1870 does. 

1871 prefer_file_temporary : `bool`, optional 

1872 If `True`, for implementations that require transfers from a remote 

1873 system to temporary local storage and/or back, use a temporary file 

1874 instead of an in-memory buffer; this is generally slower, but it 

1875 may be necessary to avoid excessive memory usage by large files. 

1876 Ignored by implementations that do not require a temporary. 

1877 

1878 Yields 

1879 ------ 

1880 cm : `~contextlib.AbstractContextManager` 

1881 A context manager that wraps a `ResourceHandleProtocol` file-like 

1882 object. 

1883 

1884 Notes 

1885 ----- 

1886 The default implementation of this method uses a local temporary buffer 

1887 (in-memory or file, depending on ``prefer_file_temporary``) with calls 

1888 to `read`, `write`, `as_local`, and `transfer_from` as necessary to 

1889 read and write from/to remote systems. Remote writes thus occur only 

1890 when the context manager is exited. `ResourcePath` implementations 

1891 that can return a more efficient native buffer should do so whenever 

1892 possible (as is guaranteed for local files). `ResourcePath` 

1893 implementations for which `as_local` does not return a temporary are 

1894 required to reimplement `open`, though they may delegate to `super` 

1895 when ``prefer_file_temporary`` is `False`. 

1896 """ 

1897 if self.isdir(): 

1898 raise IsADirectoryError(f"Directory-like URI {self} cannot be opened.") 

1899 if "x" in mode and self.exists(): 

1900 raise FileExistsError(f"File at {self} already exists.") 

1901 if prefer_file_temporary: 

1902 if "r" in mode or "a" in mode: 

1903 local_cm = self.as_local() 

1904 else: 

1905 local_cm = self.temporary_uri(suffix=self.getExtension()) 

1906 with local_cm as local_uri: 

1907 assert local_uri.isTemporary, ( 

1908 "ResourcePath implementations for which as_local is not " 

1909 "a temporary must reimplement `open`." 

1910 ) 

1911 # An encoding is ignored for binary IO, so do not let the 

1912 # builtin open() reject it. 

1913 encoding_arg = None if "b" in mode else encoding 

1914 with open(local_uri.ospath, mode=mode, encoding=encoding_arg) as file_buffer: 

1915 if "a" in mode: 

1916 file_buffer.seek(0, io.SEEK_END) 

1917 yield file_buffer 

1918 if "r" not in mode or "+" in mode: 

1919 self.transfer_from(local_uri, transfer="copy", overwrite=("x" not in mode)) 

1920 else: 

1921 with self._openImpl(mode, encoding=encoding) as handle: 

1922 yield handle 

1923 

1924 @contextlib.contextmanager 

1925 def _openImpl(self, mode: str = "r", *, encoding: str | None = None) -> Generator[ResourceHandleProtocol]: 

1926 """Implement opening of a resource handle. 

1927 

1928 This private method may be overridden by specific `ResourcePath` 

1929 implementations to provide a customized handle like interface. 

1930 

1931 Parameters 

1932 ---------- 

1933 mode : `str` 

1934 The mode the handle should be opened with 

1935 encoding : `str`, optional 

1936 The byte encoding of any binary text 

1937 

1938 Yields 

1939 ------ 

1940 handle : `~._resourceHandles.BaseResourceHandle` 

1941 A handle that conforms to the 

1942 `~._resourceHandles.BaseResourceHandle` interface 

1943 

1944 Notes 

1945 ----- 

1946 The base implementation of a file handle reads in a files entire 

1947 contents into a buffer for manipulation, and then writes it back out 

1948 upon close. Subclasses of this class may offer more fine grained 

1949 control. 

1950 """ 

1951 in_bytes = self.read() if "r" in mode or "a" in mode else b"" 

1952 if "b" in mode: 1952 ↛ 1960line 1952 didn't jump to line 1960 because the condition on line 1952 was always true

1953 bytes_buffer = io.BytesIO(in_bytes) 

1954 bytes_buffer.name = str(self) 

1955 if "a" in mode: 1955 ↛ 1956line 1955 didn't jump to line 1956 because the condition on line 1955 was never true

1956 bytes_buffer.seek(0, io.SEEK_END) 

1957 yield bytes_buffer 

1958 out_bytes = bytes_buffer.getvalue() 

1959 else: 

1960 if encoding is None: 

1961 encoding = locale.getpreferredencoding(False) 

1962 str_buffer = io.StringIO(in_bytes.decode(encoding)) 

1963 str_buffer.name = str(self) 

1964 if "a" in mode: 

1965 str_buffer.seek(0, io.SEEK_END) 

1966 yield str_buffer 

1967 out_bytes = str_buffer.getvalue().encode(encoding) 

1968 if "r" not in mode or "+" in mode: 1968 ↛ 1969line 1968 didn't jump to line 1969 because the condition on line 1968 was never true

1969 self.write(out_bytes, overwrite=("x" not in mode)) 

1970 

1971 def generate_presigned_get_url(self, *, expiration_time_seconds: int) -> str: 

1972 """Return a pre-signed URL that can be used to retrieve this resource 

1973 using an HTTP GET without supplying any access credentials. 

1974 

1975 Parameters 

1976 ---------- 

1977 expiration_time_seconds : `int` 

1978 Number of seconds until the generated URL is no longer valid. 

1979 

1980 Returns 

1981 ------- 

1982 url : `str` 

1983 HTTP URL signed for GET. 

1984 """ 

1985 raise NotImplementedError(f"URL signing is not supported for '{self.scheme}'") 

1986 

1987 def generate_presigned_put_url(self, *, expiration_time_seconds: int) -> str: 

1988 """Return a pre-signed URL that can be used to upload a file to this 

1989 path using an HTTP PUT without supplying any access credentials. 

1990 

1991 Parameters 

1992 ---------- 

1993 expiration_time_seconds : `int` 

1994 Number of seconds until the generated URL is no longer valid. 

1995 

1996 Returns 

1997 ------- 

1998 url : `str` 

1999 HTTP URL signed for PUT. 

2000 """ 

2001 raise NotImplementedError(f"URL signing is not supported for '{self.scheme}'") 

2002 

2003 def _copy_extra_attributes(self, original_uri: ResourcePath) -> None: 

2004 # May be overridden by subclasses to transfer attributes when a 

2005 # ResourcePath is constructed using the "clone" version of the 

2006 # ResourcePath constructor by passing in a ResourcePath object. 

2007 pass 

2008 

2009 def get_info(self) -> ResourceInfo: 

2010 """Return lightweight metadata about this resource. 

2011 

2012 Returns 

2013 ------- 

2014 info : `ResourceInfo` 

2015 The information about this resource that can be obtained from 

2016 the backend. Will not read the file contents. 

2017 """ 

2018 raise NotImplementedError("") 

2019 

2020 

2021ResourcePathExpression = str | urllib.parse.ParseResult | ResourcePath | Path 

2022"""Type-annotation alias for objects that can be coerced to ResourcePath. 

2023"""