Coverage for python/lsst/resources/_resourceHandles/_httpResourceHandle.py: 70%
131 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-23 09:29 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-23 09:29 +0000
1# This file is part of lsst-resources.
2#
3# Developed for the LSST Data Management System.
4# This product includes software developed by the LSST Project
5# (https://www.lsst.org).
6# See the COPYRIGHT file at the top-level directory of this distribution
7# for details of code ownership.
8#
9# Use of this source code is governed by a 3-clause BSD-style
10# license that can be found in the LICENSE file.
12from __future__ import annotations
14__all__ = ("HttpReadResourceHandle",)
16import io
17import logging
18import re
19from collections.abc import Callable, Iterable
20from typing import TYPE_CHECKING, AnyStr, NamedTuple
22import requests
24from lsst.utils.timer import time_this
26from ._baseResourceHandle import BaseResourceHandle, CloseStatus
28if TYPE_CHECKING:
29 from ..http import HttpResourcePath
32# Prevent circular import by copying this code. Can be removed as soon
33# as separate dav implementation is implemented.
34def _dav_to_http(url: str) -> str:
35 """Convert dav scheme in URL to http scheme."""
36 if url.startswith("dav"): 36 ↛ 37line 36 didn't jump to line 37 because the condition on line 36 was never true
37 url = "http" + url.removeprefix("dav")
38 return url
41class HttpReadResourceHandle(BaseResourceHandle[bytes]):
42 """HTTP-based specialization of `.BaseResourceHandle`.
44 Parameters
45 ----------
46 mode : `str`
47 Handle modes as described in the python `io` module.
48 log : `~logging.Logger`
49 Logger to used when writing messages.
50 uri : `lsst.resources.http.HttpResourcePath`
51 URI of remote resource.
52 timeout : `tuple` [`int`, `int`]
53 Timeout to use for connections: connection timeout and read timeout
54 in a tuple.
55 newline : `str` or `None`, optional
56 When doing multiline operations, break the stream on given character.
57 Defaults to newline. If a file is opened in binary mode, this argument
58 is not used, as binary files will only split lines on the binary
59 newline representation.
60 size : `int` or `None`, optional
61 Total size of the remote resource in bytes, if it is already known.
62 Saves a request to the server the first time the size is needed, for
63 example when seeking relative to the end of the resource.
64 """
66 def __init__(
67 self,
68 mode: str,
69 log: logging.Logger,
70 uri: HttpResourcePath,
71 *,
72 timeout: tuple[float, float] | None = None,
73 newline: AnyStr | None = None,
74 size: int | None = None,
75 ) -> None:
76 super().__init__(mode, log, uri, newline=newline)
77 self._url = uri.geturl()
78 self._session = uri.data_session
80 if timeout is None: 80 ↛ 81line 80 didn't jump to line 81 because the condition on line 80 was never true
81 raise ValueError("timeout must be specified when constructing this object")
82 self._timeout = timeout
84 self._completeBuffer: io.BytesIO | None = None
86 self._closed = CloseStatus.OPEN
87 self._current_position = 0
88 self._eof = False
89 self._total_size = -1 if size is None else size # -1 means unknown
91 def close(self) -> None:
92 self._closed = CloseStatus.CLOSED
93 self._completeBuffer = None
94 self._eof = True
96 @property
97 def closed(self) -> bool:
98 return self._closed == CloseStatus.CLOSED
100 def fileno(self) -> int:
101 raise io.UnsupportedOperation("HttpReadResourceHandle does not have a file number")
103 def flush(self) -> None:
104 modes = set(self._mode)
105 if {"w", "x", "a", "+"} & modes:
106 raise io.UnsupportedOperation("HttpReadResourceHandles are read only")
108 @property
109 def isatty(self) -> bool | Callable[[], bool]:
110 return False
112 def readable(self) -> bool:
113 return True
115 def readline(self, size: int = -1) -> bytes:
116 raise io.UnsupportedOperation("HttpReadResourceHandles Do not support line by line reading")
118 def readlines(self, hint: int = -1) -> Iterable[bytes]:
119 raise io.UnsupportedOperation("HttpReadResourceHandles Do not support line by line reading")
121 def _size(self) -> int:
122 if self._total_size == -1: 122 ↛ 123line 122 didn't jump to line 123 because the condition on line 122 was never true
123 self._total_size = self._uri.size()
124 return self._total_size
126 def seek(self, offset: int, whence: int = io.SEEK_SET) -> int:
127 self._eof = False
128 if whence == io.SEEK_CUR and (self._current_position + offset) >= 0: 128 ↛ 129line 128 didn't jump to line 129 because the condition on line 128 was never true
129 self._current_position += offset
130 elif whence == io.SEEK_SET and offset >= 0: 130 ↛ 131line 130 didn't jump to line 131 because the condition on line 130 was never true
131 self._current_position = offset
132 elif whence == io.SEEK_END: 132 ↛ 135line 132 didn't jump to line 135 because the condition on line 132 was always true
133 self._current_position = self._size() + offset
134 else:
135 raise io.UnsupportedOperation("Seek value is incorrect, or whence mode is unsupported")
137 # handle if the complete file has be read already
138 if self._completeBuffer is not None: 138 ↛ 139line 138 didn't jump to line 139 because the condition on line 138 was never true
139 self._completeBuffer.seek(self._current_position, whence)
140 return self._current_position
142 def seekable(self) -> bool:
143 return True
145 def tell(self) -> int:
146 return self._current_position
148 def truncate(self, size: int | None = None) -> int:
149 raise io.UnsupportedOperation("HttpReadResourceHandles Do not support truncation")
151 def writable(self) -> bool:
152 return False
154 def write(self, b: bytes, /) -> int:
155 raise io.UnsupportedOperation("HttpReadResourceHandles are read only")
157 def writelines(self, b: Iterable[bytes], /) -> None:
158 raise io.UnsupportedOperation("HttpReadResourceHandles are read only")
160 def read(self, size: int = -1) -> bytes:
161 if self._eof: 161 ↛ 163line 161 didn't jump to line 163 because the condition on line 161 was never true
162 # At EOF so always return an empty byte string.
163 return b""
165 # branch for if the complete file has been read before
166 if self._completeBuffer is not None: 166 ↛ 167line 166 didn't jump to line 167 because the condition on line 166 was never true
167 result = self._completeBuffer.read(size)
168 self._current_position += len(result)
169 return result
171 if self._completeBuffer is None and size == -1 and self._current_position == 0:
172 # The whole file has been requested, read it into a buffer and
173 # return the result
174 self._completeBuffer = io.BytesIO()
175 with time_this(self._log, msg="Read from remote resource %s", args=(self._url,)):
176 with self._session as session:
177 resp = session.get(_dav_to_http(self._url), stream=False, timeout=self._timeout)
179 if (code := resp.status_code) not in (requests.codes.ok, requests.codes.partial): 179 ↛ 180line 179 didn't jump to line 180 because the condition on line 179 was never true
180 raise FileNotFoundError(f"Unable to read resource {self._url}; status code: {code}")
181 self._completeBuffer.write(resp.content)
182 self._current_position = self._completeBuffer.tell()
184 return self._completeBuffer.getbuffer().tobytes()
186 # A partial read is required, either because a size has been specified,
187 # or a read has previously been done. Any time we specify a byte range
188 # we must disable the gzip compression on the server since we want
189 # to address ranges in the uncompressed file. If we send ranges that
190 # are interpreted by the server as offsets into the compressed file
191 # then that is at least confusing and also there is no guarantee that
192 # the bytes can be uncompressed.
194 end_pos = self._current_position + (size - 1) if size >= 0 else ""
195 headers = {"Range": f"bytes={self._current_position}-{end_pos}", "Accept-Encoding": "identity"}
197 with time_this(
198 self._log, msg="Read from remote resource %s using headers %s", args=(self._url, headers)
199 ):
200 with self._session as session:
201 resp = session.get(
202 _dav_to_http(self._url), stream=False, timeout=self._timeout, headers=headers
203 )
205 if resp.status_code == requests.codes.range_not_satisfiable: 205 ↛ 209line 205 didn't jump to line 209 because the condition on line 205 was never true
206 # Must have run off the end of the file. A standard file handle
207 # will treat this as EOF so be consistent with that. Do not change
208 # the current position.
209 self._eof = True
210 return b""
212 if (code := resp.status_code) not in (requests.codes.ok, requests.codes.partial): 212 ↛ 213line 212 didn't jump to line 213 because the condition on line 212 was never true
213 raise FileNotFoundError(
214 f"Unable to read resource {self._url}, or bytes are out of range; status code: {code}"
215 )
217 # The response header should tell us the total number of bytes
218 # in the file and also the current position we have got to in the
219 # server.
220 if "Content-Range" in resp.headers: 220 ↛ 235line 220 didn't jump to line 235 because the condition on line 220 was always true
221 content_range = parse_content_range_header(resp.headers["Content-Range"])
222 if content_range.total is not None: 222 ↛ 225line 222 didn't jump to line 225 because the condition on line 222 was always true
223 # Store in case we need this later.
224 self._total_size = content_range.total
225 if ( 225 ↛ 235line 225 didn't jump to line 235 because the condition on line 225 was always true
226 content_range.total is not None
227 and content_range.range_end is not None
228 and content_range.range_end >= content_range.total - 1
229 ):
230 self._eof = True
232 # Try to guess that we overran the end. This will not help if we
233 # read exactly the number of bytes to get us to the end and so we
234 # will need to do one more read and get a 416.
235 len_content = len(resp.content)
236 if len_content < size: 236 ↛ 237line 236 didn't jump to line 237 because the condition on line 236 was never true
237 self._eof = True
239 self._current_position += len_content
240 return resp.content
243class ContentRange(NamedTuple):
244 """Represents the data in an HTTP Content-Range header."""
246 range_start: int | None
247 """First byte of the zero-indexed, inclusive range returned by this
248 response. `None` if the range was not available in the header.
249 """
250 range_end: int | None
251 """Last byte of the zero-indexed, inclusive range returned by this
252 response. `None` if the range was not available in the header.
253 """
254 total: int | None
255 """Total size of the file in bytes. `None` if the file size was not
256 available in the header.
257 """
260def parse_content_range_header(header: str) -> ContentRange:
261 """Parse an HTTP 'Content-Range' header.
263 Parameters
264 ----------
265 header : `str`
266 Value of an HTTP Content-Range header to be parsed.
268 Returns
269 -------
270 content_range : `ContentRange`
271 The byte range included in the response and the total file size.
273 Raises
274 ------
275 ValueError
276 If the header was not in the expected format.
277 """
278 # There are three possible formats for Content-Range. All of them start
279 # with optional whitespace and a unit, which for our purposes should always
280 # be "bytes".
281 prefix = r"^\s*bytes\s+"
283 # Content-Range: <unit> <range-start>-<range-end>/<size>
284 if (case1 := re.match(prefix + r"(\d+)-(\d+)/(\d+)", header)) is not None:
285 return ContentRange(
286 range_start=int(case1.group(1)), range_end=int(case1.group(2)), total=int(case1.group(3))
287 )
289 # Content-Range: <unit> <range-start>-<range-end>/*
290 if (case2 := re.match(prefix + r"(\d+)-(\d+)/\*", header)) is not None:
291 return ContentRange(range_start=int(case2.group(1)), range_end=int(case2.group(2)), total=None)
293 # Content-Range: <unit> */<size>
294 if (case3 := re.match(prefix + r"\*/(\d+)", header)) is not None:
295 return ContentRange(range_start=None, range_end=None, total=int(case3.group(1)))
297 raise ValueError(f"Content-Range header in unexpected format: '{header}'")