Coverage for python/lsst/resources/_resourceHandles/_httpResourceHandle.py: 70%

131 statements  

« prev     ^ index     » next       coverage.py v7.16.0, created at 2026-09-19 09:04 +0000

1# This file is part of lsst-resources. 

2# 

3# Developed for the LSST Data Management System. 

4# This product includes software developed by the LSST Project 

5# (https://www.lsst.org). 

6# See the COPYRIGHT file at the top-level directory of this distribution 

7# for details of code ownership. 

8# 

9# Use of this source code is governed by a 3-clause BSD-style 

10# license that can be found in the LICENSE file. 

11 

12from __future__ import annotations 

13 

14__all__ = ("HttpReadResourceHandle",) 

15 

16import io 

17import logging 

18import re 

19from collections.abc import Callable, Iterable 

20from typing import TYPE_CHECKING, AnyStr, NamedTuple 

21 

22import requests 

23 

24from lsst.utils.timer import time_this 

25 

26from ._baseResourceHandle import BaseResourceHandle, CloseStatus 

27 

28if TYPE_CHECKING: 

29 from ..http import HttpResourcePath 

30 

31 

32# Prevent circular import by copying this code. Can be removed as soon 

33# as separate dav implementation is implemented. 

34def _dav_to_http(url: str) -> str: 

35 """Convert dav scheme in URL to http scheme.""" 

36 if url.startswith("dav"): 36 ↛ 37line 36 didn't jump to line 37 because the condition on line 36 was never true

37 url = "http" + url.removeprefix("dav") 

38 return url 

39 

40 

41class HttpReadResourceHandle(BaseResourceHandle[bytes]): 

42 """HTTP-based specialization of `.BaseResourceHandle`. 

43 

44 Parameters 

45 ---------- 

46 mode : `str` 

47 Handle modes as described in the python `io` module. 

48 log : `~logging.Logger` 

49 Logger to used when writing messages. 

50 uri : `lsst.resources.http.HttpResourcePath` 

51 URI of remote resource. 

52 timeout : `tuple` [`int`, `int`] 

53 Timeout to use for connections: connection timeout and read timeout 

54 in a tuple. 

55 newline : `str` or `None`, optional 

56 When doing multiline operations, break the stream on given character. 

57 Defaults to newline. If a file is opened in binary mode, this argument 

58 is not used, as binary files will only split lines on the binary 

59 newline representation. 

60 size : `int` or `None`, optional 

61 Total size of the remote resource in bytes, if it is already known. 

62 Saves a request to the server the first time the size is needed, for 

63 example when seeking relative to the end of the resource. 

64 """ 

65 

66 def __init__( 

67 self, 

68 mode: str, 

69 log: logging.Logger, 

70 uri: HttpResourcePath, 

71 *, 

72 timeout: tuple[float, float] | None = None, 

73 newline: AnyStr | None = None, 

74 size: int | None = None, 

75 ) -> None: 

76 super().__init__(mode, log, uri, newline=newline) 

77 self._url = uri.geturl() 

78 self._session = uri.data_session 

79 

80 if timeout is None: 80 ↛ 81line 80 didn't jump to line 81 because the condition on line 80 was never true

81 raise ValueError("timeout must be specified when constructing this object") 

82 self._timeout = timeout 

83 

84 self._completeBuffer: io.BytesIO | None = None 

85 

86 self._closed = CloseStatus.OPEN 

87 self._current_position = 0 

88 self._eof = False 

89 self._total_size = -1 if size is None else size # -1 means unknown 

90 

91 def close(self) -> None: 

92 self._closed = CloseStatus.CLOSED 

93 self._completeBuffer = None 

94 self._eof = True 

95 

96 @property 

97 def closed(self) -> bool: 

98 return self._closed == CloseStatus.CLOSED 

99 

100 def fileno(self) -> int: 

101 raise io.UnsupportedOperation("HttpReadResourceHandle does not have a file number") 

102 

103 def flush(self) -> None: 

104 modes = set(self._mode) 

105 if {"w", "x", "a", "+"} & modes: 

106 raise io.UnsupportedOperation("HttpReadResourceHandles are read only") 

107 

108 @property 

109 def isatty(self) -> bool | Callable[[], bool]: 

110 return False 

111 

112 def readable(self) -> bool: 

113 return True 

114 

115 def readline(self, size: int = -1) -> bytes: 

116 raise io.UnsupportedOperation("HttpReadResourceHandles Do not support line by line reading") 

117 

118 def readlines(self, hint: int = -1) -> Iterable[bytes]: 

119 raise io.UnsupportedOperation("HttpReadResourceHandles Do not support line by line reading") 

120 

121 def _size(self) -> int: 

122 if self._total_size == -1: 122 ↛ 123line 122 didn't jump to line 123 because the condition on line 122 was never true

123 self._total_size = self._uri.size() 

124 return self._total_size 

125 

126 def seek(self, offset: int, whence: int = io.SEEK_SET) -> int: 

127 self._eof = False 

128 if whence == io.SEEK_CUR and (self._current_position + offset) >= 0: 128 ↛ 129line 128 didn't jump to line 129 because the condition on line 128 was never true

129 self._current_position += offset 

130 elif whence == io.SEEK_SET and offset >= 0: 130 ↛ 131line 130 didn't jump to line 131 because the condition on line 130 was never true

131 self._current_position = offset 

132 elif whence == io.SEEK_END: 132 ↛ 135line 132 didn't jump to line 135 because the condition on line 132 was always true

133 self._current_position = self._size() + offset 

134 else: 

135 raise io.UnsupportedOperation("Seek value is incorrect, or whence mode is unsupported") 

136 

137 # handle if the complete file has be read already 

138 if self._completeBuffer is not None: 138 ↛ 139line 138 didn't jump to line 139 because the condition on line 138 was never true

139 self._completeBuffer.seek(self._current_position, whence) 

140 return self._current_position 

141 

142 def seekable(self) -> bool: 

143 return True 

144 

145 def tell(self) -> int: 

146 return self._current_position 

147 

148 def truncate(self, size: int | None = None) -> int: 

149 raise io.UnsupportedOperation("HttpReadResourceHandles Do not support truncation") 

150 

151 def writable(self) -> bool: 

152 return False 

153 

154 def write(self, b: bytes, /) -> int: 

155 raise io.UnsupportedOperation("HttpReadResourceHandles are read only") 

156 

157 def writelines(self, b: Iterable[bytes], /) -> None: 

158 raise io.UnsupportedOperation("HttpReadResourceHandles are read only") 

159 

160 def read(self, size: int = -1) -> bytes: 

161 if self._eof: 161 ↛ 163line 161 didn't jump to line 163 because the condition on line 161 was never true

162 # At EOF so always return an empty byte string. 

163 return b"" 

164 

165 # branch for if the complete file has been read before 

166 if self._completeBuffer is not None: 166 ↛ 167line 166 didn't jump to line 167 because the condition on line 166 was never true

167 result = self._completeBuffer.read(size) 

168 self._current_position += len(result) 

169 return result 

170 

171 if self._completeBuffer is None and size == -1 and self._current_position == 0: 

172 # The whole file has been requested, read it into a buffer and 

173 # return the result 

174 self._completeBuffer = io.BytesIO() 

175 with time_this(self._log, msg="Read from remote resource %s", args=(self._url,)): 

176 with self._session as session: 

177 resp = session.get(_dav_to_http(self._url), stream=False, timeout=self._timeout) 

178 

179 if (code := resp.status_code) not in (requests.codes.ok, requests.codes.partial): 179 ↛ 180line 179 didn't jump to line 180 because the condition on line 179 was never true

180 raise FileNotFoundError(f"Unable to read resource {self._url}; status code: {code}") 

181 self._completeBuffer.write(resp.content) 

182 self._current_position = self._completeBuffer.tell() 

183 

184 return self._completeBuffer.getbuffer().tobytes() 

185 

186 # A partial read is required, either because a size has been specified, 

187 # or a read has previously been done. Any time we specify a byte range 

188 # we must disable the gzip compression on the server since we want 

189 # to address ranges in the uncompressed file. If we send ranges that 

190 # are interpreted by the server as offsets into the compressed file 

191 # then that is at least confusing and also there is no guarantee that 

192 # the bytes can be uncompressed. 

193 

194 end_pos = self._current_position + (size - 1) if size >= 0 else "" 

195 headers = {"Range": f"bytes={self._current_position}-{end_pos}", "Accept-Encoding": "identity"} 

196 

197 with time_this( 

198 self._log, msg="Read from remote resource %s using headers %s", args=(self._url, headers) 

199 ): 

200 with self._session as session: 

201 resp = session.get( 

202 _dav_to_http(self._url), stream=False, timeout=self._timeout, headers=headers 

203 ) 

204 

205 if resp.status_code == requests.codes.range_not_satisfiable: 205 ↛ 209line 205 didn't jump to line 209 because the condition on line 205 was never true

206 # Must have run off the end of the file. A standard file handle 

207 # will treat this as EOF so be consistent with that. Do not change 

208 # the current position. 

209 self._eof = True 

210 return b"" 

211 

212 if (code := resp.status_code) not in (requests.codes.ok, requests.codes.partial): 212 ↛ 213line 212 didn't jump to line 213 because the condition on line 212 was never true

213 raise FileNotFoundError( 

214 f"Unable to read resource {self._url}, or bytes are out of range; status code: {code}" 

215 ) 

216 

217 # The response header should tell us the total number of bytes 

218 # in the file and also the current position we have got to in the 

219 # server. 

220 if "Content-Range" in resp.headers: 220 ↛ 235line 220 didn't jump to line 235 because the condition on line 220 was always true

221 content_range = parse_content_range_header(resp.headers["Content-Range"]) 

222 if content_range.total is not None: 222 ↛ 225line 222 didn't jump to line 225 because the condition on line 222 was always true

223 # Store in case we need this later. 

224 self._total_size = content_range.total 

225 if ( 225 ↛ 235line 225 didn't jump to line 235 because the condition on line 225 was always true

226 content_range.total is not None 

227 and content_range.range_end is not None 

228 and content_range.range_end >= content_range.total - 1 

229 ): 

230 self._eof = True 

231 

232 # Try to guess that we overran the end. This will not help if we 

233 # read exactly the number of bytes to get us to the end and so we 

234 # will need to do one more read and get a 416. 

235 len_content = len(resp.content) 

236 if len_content < size: 236 ↛ 237line 236 didn't jump to line 237 because the condition on line 236 was never true

237 self._eof = True 

238 

239 self._current_position += len_content 

240 return resp.content 

241 

242 

243class ContentRange(NamedTuple): 

244 """Represents the data in an HTTP Content-Range header.""" 

245 

246 range_start: int | None 

247 """First byte of the zero-indexed, inclusive range returned by this 

248 response. `None` if the range was not available in the header. 

249 """ 

250 range_end: int | None 

251 """Last byte of the zero-indexed, inclusive range returned by this 

252 response. `None` if the range was not available in the header. 

253 """ 

254 total: int | None 

255 """Total size of the file in bytes. `None` if the file size was not 

256 available in the header. 

257 """ 

258 

259 

260def parse_content_range_header(header: str) -> ContentRange: 

261 """Parse an HTTP 'Content-Range' header. 

262 

263 Parameters 

264 ---------- 

265 header : `str` 

266 Value of an HTTP Content-Range header to be parsed. 

267 

268 Returns 

269 ------- 

270 content_range : `ContentRange` 

271 The byte range included in the response and the total file size. 

272 

273 Raises 

274 ------ 

275 ValueError 

276 If the header was not in the expected format. 

277 """ 

278 # There are three possible formats for Content-Range. All of them start 

279 # with optional whitespace and a unit, which for our purposes should always 

280 # be "bytes". 

281 prefix = r"^\s*bytes\s+" 

282 

283 # Content-Range: <unit> <range-start>-<range-end>/<size> 

284 if (case1 := re.match(prefix + r"(\d+)-(\d+)/(\d+)", header)) is not None: 

285 return ContentRange( 

286 range_start=int(case1.group(1)), range_end=int(case1.group(2)), total=int(case1.group(3)) 

287 ) 

288 

289 # Content-Range: <unit> <range-start>-<range-end>/* 

290 if (case2 := re.match(prefix + r"(\d+)-(\d+)/\*", header)) is not None: 

291 return ContentRange(range_start=int(case2.group(1)), range_end=int(case2.group(2)), total=None) 

292 

293 # Content-Range: <unit> */<size> 

294 if (case3 := re.match(prefix + r"\*/(\d+)", header)) is not None: 

295 return ContentRange(range_start=None, range_end=None, total=int(case3.group(1))) 

296 

297 raise ValueError(f"Content-Range header in unexpected format: '{header}'")