Coverage for gws-app/gws/lib/net/__init__.py: 87%

255 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-05 13:35 +0200

1"""URL utilities and an HTTP client. 

2 

3URLs: 

4 

5``parse_url`` splits a URL into a ``Url`` object (scheme, host, port, path, query parameters and so on), 

6``make_url`` builds a URL string from such an object or a dict. ``make_qs``, ``add_params``, 

7``extract_params`` and the ``quote_*`` functions handle query strings and URL quoting. 

8 

9HTTP: 

10 

11``http_request`` sends a request with ``requests``, using one session per process, so that 

12connections to the same host are reused. It does not raise on failures; instead, it returns an 

13``HTTPResponse`` with ``ok`` set to ``False``. Connection errors, timeouts and other request 

14errors get the pseudo status codes 900, 901 and 999. ``HTTPResponse.raise_if_failed`` turns a 

15failed response into an ``HTTPError`` subclass. The response text is decoded with the charset 

16from the content type header, or as UTF-8 or ISO-8859-1 if there is none. 

17 

18Example:: 

19 

20 u = gws.lib.net.parse_url('https://example.com/wms?SERVICE=WMS') 

21 url = gws.lib.net.add_params(u.url, REQUEST='GetCapabilities') 

22 

23 res = gws.lib.net.http_request(url, timeout=10) 

24 res.raise_if_failed() 

25 xml = res.text 

26""" 

27 

28from typing import Optional 

29 

30import re 

31import requests 

32import requests.adapters 

33import urllib.parse 

34import certifi 

35 

36import gws 

37import gws.lib.osx 

38 

39 

40class Error(gws.Error): 

41 """Network error.""" 

42 

43 pass 

44 

45 

46class HTTPError(Error): 

47 """HTTP request failed.""" 

48 

49 pass 

50 

51 

52class Timeout(HTTPError): 

53 """HTTP request timed out.""" 

54 

55 pass 

56 

57 

58class ConnectionError(HTTPError): 

59 """HTTP connection failed.""" 

60 

61 pass 

62 

63 

64class GenericError(HTTPError): 

65 """HTTP request failed for another reason.""" 

66 

67 pass 

68 

69 

70_STATUS_CONNECTION_ERROR = 900 

71_STATUS_TIMEOUT = 901 

72_STATUS_GENERIC_ERROR = 999 

73 

74 

75class Url(gws.Data): 

76 """Components of a URL, as returned by ``parse_url``.""" 

77 

78 fragment: str 

79 """Fragment, without ``#``.""" 

80 hostname: str 

81 """Host name.""" 

82 netloc: str 

83 """Network location, including user name, password and port.""" 

84 params: dict 

85 """Query parameters. For repeated parameters, the first value is used.""" 

86 password: str 

87 """Password, unquoted.""" 

88 path: str 

89 """Path.""" 

90 pathparts: gws.lib.osx.ParsePathResult 

91 """Components of the path.""" 

92 port: int 

93 """Port, ``0`` if not given.""" 

94 qsl: list 

95 """Query parameters as a list of ``(name, value)`` pairs.""" 

96 query: str 

97 """Query string, without ``?``.""" 

98 scheme: str 

99 """Scheme, e.g. ``https``.""" 

100 url: str 

101 """The parsed URL; with a ``//`` prefix if the original URL had no scheme.""" 

102 username: str 

103 """User name, unquoted.""" 

104 

105 

106def parse_url(url: str, **kwargs) -> Url: 

107 """Parse a URL. 

108 

109 A URL without ``//`` is treated as starting with a host name. 

110 

111 Args: 

112 url: URL string. 

113 **kwargs: Values to set on the ``Url`` object after parsing. 

114 

115 Returns: 

116 A Url object. 

117 """ 

118 

119 if not is_abs_url(url): 

120 url = '//' + url 

121 

122 us = urllib.parse.urlsplit(url) 

123 

124 u = Url( 

125 fragment=us.fragment or '', 

126 hostname=us.hostname or '', 

127 netloc=us.netloc or '', 

128 params={}, 

129 password=us.password or '', 

130 path=us.path or '', 

131 pathparts=gws.lib.osx.ParsePathResult(), 

132 port=0, 

133 qsl=[], 

134 query=us.query or '', 

135 scheme=us.scheme or '', 

136 url=url, 

137 username=us.username or '', 

138 ) 

139 

140 if us.port: 

141 try: 

142 u.port = int(us.port) 

143 except ValueError: 

144 pass 

145 

146 if u.path: 

147 u.pathparts = gws.lib.osx.parse_path(u.path) 

148 

149 if u.query: 

150 u.qsl = urllib.parse.parse_qsl(u.query) 

151 for k, v in u.qsl: 

152 u.params.setdefault(k, v) 

153 

154 if u.username: 

155 u.username = unquote(u.username) 

156 u.password = unquote(u.get('password') or '') 

157 

158 u.update(**kwargs) 

159 return u 

160 

161 

162_DEFAULT_PORTS = {'http': '80', 'https': '443'} 

163 

164 

165def make_url(u: Optional[Url | dict] = None, **kwargs) -> str: 

166 """Build a URL string. 

167 

168 Uses ``scheme``, ``hostname``, ``port``, ``username``, ``password``, ``path``, ``params`` 

169 and ``fragment``; other keys, like ``query`` or ``url``, are ignored. The port is omitted 

170 if it is the default port of the scheme. 

171 

172 Args: 

173 u: URL components, a Url object or a dict. 

174 **kwargs: Components that override the values from ``u``. 

175 

176 Returns: 

177 The URL. 

178 """ 

179 

180 p = gws.u.merge({}, u, kwargs) 

181 

182 s = '' 

183 

184 scheme = p.get('scheme', '').lower() 

185 if scheme: 

186 s += scheme + ':' 

187 

188 host = p.get('hostname', '') 

189 port = p.get('port', '') 

190 path = p.get('path', '') 

191 

192 if scheme or host: 

193 s += '//' 

194 

195 if host: 

196 username = p.get('username', '') 

197 if username: 

198 s += quote_param(username) + ':' + quote_param(p.get('password', '')) + '@' 

199 

200 s += host 

201 if port and str(port) != _DEFAULT_PORTS.get(scheme): 

202 s += ':' + str(port) 

203 

204 if path: 

205 s += '/' + quote_path(path.lstrip('/')) 

206 

207 params = p.get('params') 

208 if params: 

209 s += '?' + make_qs(params) 

210 

211 fragment = p.get('fragment', '') 

212 if fragment: 

213 s += '#' + fragment.lstrip('#') 

214 

215 return s 

216 

217 

218def parse_qs(x) -> dict: 

219 """Parse a query string. 

220 

221 Args: 

222 x: Query string. 

223 

224 Returns: 

225 A dict that maps each parameter name to a list of its values. 

226 """ 

227 

228 return urllib.parse.parse_qs(x) 

229 

230 

231def make_qs(x) -> str: 

232 """Convert a dict or a list of pairs to a query string. 

233 

234 Values are encoded as UTF-8 and quoted. Booleans become ``true`` and ``false``, 

235 lists and other iterables are joined with commas, other values are converted with ``str``. 

236 

237 Args: 

238 x: A dict, an object that can be converted to a dict, or a list of ``(name, value)`` pairs. 

239 

240 Returns: 

241 The query string, without ``?``. 

242 """ 

243 

244 p = [] 

245 items = x if isinstance(x, (list, tuple)) else gws.u.to_dict(x).items() 

246 

247 def _value(v): 

248 if isinstance(v, (bytes, bytearray)): 

249 return v 

250 if isinstance(v, str): 

251 return v.encode('utf8') 

252 if v is True: 

253 return b'true' 

254 if v is False: 

255 return b'false' 

256 try: 

257 return b','.join(_value(y) for y in v) 

258 except TypeError: 

259 return str(v).encode('utf8') 

260 

261 for k, v in items: 

262 k = urllib.parse.quote_from_bytes(_value(k)) 

263 v = urllib.parse.quote_from_bytes(_value(v)) 

264 p.append(k + '=' + v) 

265 

266 return '&'.join(p) 

267 

268 

269def quote_param(s: str) -> str: 

270 """Quote a string for use in a URL, including slashes. 

271 

272 Args: 

273 s: A string. 

274 

275 Returns: 

276 The quoted string. 

277 """ 

278 

279 return urllib.parse.quote(s, safe='') 

280 

281 

282def quote_path(s: str) -> str: 

283 """Quote a URL path, leaving slashes as they are. 

284 

285 Args: 

286 s: A path. 

287 

288 Returns: 

289 The quoted path. 

290 """ 

291 

292 return urllib.parse.quote(s, safe='/') 

293 

294 

295def unquote(s: str) -> str: 

296 """Unquote a URL-quoted string. 

297 

298 Args: 

299 s: A quoted string. 

300 

301 Returns: 

302 The unquoted string. 

303 """ 

304 

305 return urllib.parse.unquote(s) 

306 

307 

308def add_params(url: str, params: dict = None, **kwargs) -> str: 

309 """Add query parameters to a URL. 

310 

311 Existing parameters with the same names are replaced. 

312 

313 Args: 

314 url: A URL. 

315 params: Parameters to add. 

316 **kwargs: More parameters to add. 

317 

318 Returns: 

319 The new URL. 

320 """ 

321 

322 u = parse_url(url) 

323 if params: 

324 u.params.update(params) 

325 u.params.update(kwargs) 

326 return make_url(u) 

327 

328 

329def make_relative_url(path: str, params: dict = None, **kwargs) -> str: 

330 """Build a URL from a path and query parameters, without a scheme and host. 

331 

332 Args: 

333 path: URL path; a leading slash is added if needed. 

334 params: Query parameters. 

335 **kwargs: More query parameters. 

336 

337 Returns: 

338 The URL. 

339 """ 

340 

341 s = '/' + quote_path(path.lstrip('/')) 

342 p = {} 

343 if params: 

344 p.update(params) 

345 p.update(kwargs) 

346 if p: 

347 s += '?' + make_qs(p) 

348 return s 

349 

350 

351def extract_params(url: str) -> tuple[str, dict]: 

352 """Split the query parameters off a URL. 

353 

354 Args: 

355 url: A URL. 

356 

357 Returns: 

358 The URL without the query string, and the query parameters. 

359 """ 

360 

361 u = parse_url(url) 

362 params = u.params 

363 u.params = None 

364 return make_url(u), params 

365 

366 

367def is_abs_url(url): 

368 """Check if a URL has a host part, i.e. starts with ``//`` or a lowercase scheme and ``//``. 

369 

370 Args: 

371 url: A URL. 

372 

373 Returns: 

374 A truthy match object if the URL is absolute, ``None`` otherwise. 

375 """ 

376 

377 return re.match(r'^([a-z]+:|)//', url) 

378 

379 

380## 

381 

382 

383class HTTPResponse: 

384 """Response of ``http_request``. 

385 

386 Attributes: 

387 ok: ``True`` if the request succeeded with a 2xx status. 

388 url: Request URL. 

389 content: Response body. 

390 content_type: Content type, without parameters. 

391 content_encoding: Charset from the content type header, or ``None``. 

392 status_code: HTTP status code, or a pseudo status code for requests that failed without a response. 

393 """ 

394 

395 def __init__( 

396 self, 

397 ok: bool, 

398 url: str, 

399 res: requests.Response = None, 

400 text: str = None, 

401 status_code=0, 

402 ): 

403 """Create a response. 

404 

405 Args: 

406 ok: ``True`` if the request succeeded. 

407 url: Request URL. 

408 res: The ``requests`` response, if there is one. 

409 text: Content of a response without ``res``, e.g. an error message. It is stored as UTF-8 plain text. 

410 status_code: Status code of a response without ``res``. 

411 """ 

412 self.ok = ok 

413 self.url = url 

414 if res is not None: 

415 self.content_type, self.content_encoding = _parse_content_type(res.headers) 

416 self.content = res.content 

417 self.status_code = res.status_code 

418 else: 

419 self.content_type, self.content_encoding = 'text/plain', 'utf8' 

420 self.content = text.encode('utf8') if text is not None else b'' 

421 self.status_code = status_code 

422 

423 @property 

424 def text(self) -> str: 

425 """Response body as text, decoded on first access. 

426 

427 Returns: 

428 The decoded body. 

429 """ 

430 

431 if not hasattr(self, '_text'): 

432 setattr(self, '_text', _get_text(self.content, self.content_encoding)) 

433 return getattr(self, '_text') 

434 

435 def raise_if_failed(self): 

436 """Raise an error if the request failed. 

437 

438 Raises: 

439 ``ConnectionError``: If the connection failed. 

440 ``Timeout``: If the request timed out. 

441 ``GenericError``: If the request failed for another reason, without a response. 

442 ``HTTPError``: If the response status is not 2xx. 

443 """ 

444 

445 if self.ok: 

446 return 

447 if self.status_code == _STATUS_CONNECTION_ERROR: 

448 raise ConnectionError(self.text) 

449 if self.status_code == _STATUS_TIMEOUT: 

450 raise Timeout(self.text) 

451 if self.status_code == _STATUS_GENERIC_ERROR: 

452 raise GenericError(self.text) 

453 raise HTTPError(f'HTTP error: {self.status_code}') 

454 

455 

456def _get_text(content, encoding) -> str: 

457 """Decode content with the given encoding, falling back to UTF-8 and ISO-8859-1.""" 

458 

459 if encoding: 

460 try: 

461 return str(content, encoding=encoding, errors='strict') 

462 except UnicodeDecodeError: 

463 pass 

464 

465 # some folks serve utf8 content without a header, in which case requests thinks it's ISO-8859-1 

466 # (see http://docs.python-requests.org/en/master/user/advanced/#encodings) 

467 # 

468 # 'apparent_encoding' is not always reliable 

469 # 

470 # therefore when there's no header, we try utf8 first, and then ISO-8859-1 

471 

472 try: 

473 return str(content, encoding='utf8', errors='strict') 

474 except UnicodeDecodeError: 

475 pass 

476 

477 try: 

478 return str(content, encoding='ISO-8859-1', errors='strict') 

479 except UnicodeDecodeError: 

480 pass 

481 

482 # both failed, do utf8 with replace 

483 

484 gws.log.warning(f'decode failed') 

485 return str(content, encoding='utf8', errors='replace') 

486 

487 

488def _parse_content_type_header(header): 

489 """Split a content type header into the lowercase type and a dict of parameters.""" 

490 

491 parts = header.split(';') 

492 ctype = parts[0].strip().lower() 

493 params: dict[str, str] = {} 

494 

495 for part in parts[1:]: 

496 part = part.strip() 

497 if '=' in part: 

498 k, v = part.split('=', 1) 

499 k = k.strip().lower() 

500 v = v.strip() 

501 # strip matched quotes (single or double) 

502 if len(v) >= 2 and v[0] == v[-1] and v[0] in ('"', "'"): 

503 v = v[1:-1] 

504 params[k] = v 

505 

506 return ctype, params 

507 

508 

509def _parse_content_type(headers): 

510 """Return the content type and the valid charset, or ``None``, from response headers.""" 

511 

512 # copied from requests.utils.get_encoding_from_headers, but with no ISO-8859-1 default 

513 

514 header = headers.get('content-type') 

515 if not header: 

516 # https://www.w3.org/Protocols/rfc2616/rfc2616-sec7.html#sec7.2.1 

517 return 'application/octet-stream', None 

518 

519 ctype, params = _parse_content_type_header(header) 

520 if 'charset' not in params: 

521 return ctype, None 

522 

523 # make sure this is a valid python encoding 

524 enc = params['charset'] 

525 try: 

526 str(b'.', encoding=enc, errors='strict') 

527 except LookupError: 

528 gws.log.warning(f'invalid content-type encoding {enc!r}') 

529 return ctype, None 

530 

531 return ctype, enc 

532 

533 

534## 

535 

536# @TODO locking for caches 

537 

538 

539def http_request(url, **kwargs) -> HTTPResponse: 

540 """Send an HTTP request. 

541 

542 Failures are logged and returned as a response with ``ok`` set to ``False``, they do not raise. 

543 By default, the connect and read timeouts are 60 seconds, certificates are verified with ``certifi``, 

544 and a GBD WebSuite ``User-Agent`` header is sent. 

545 

546 Args: 

547 url: Request URL. 

548 **kwargs: Options for ``requests.Session.request``. ``method`` is the HTTP method (``GET`` by default), 

549 ``params`` are added to the URL, a numeric ``timeout`` applies to both connect and read. 

550 

551 Returns: 

552 The response. 

553 """ 

554 

555 kwargs = dict(kwargs) 

556 

557 if 'params' in kwargs: 

558 url = add_params(url, kwargs.pop('params')) 

559 

560 method = kwargs.pop('method', 'GET').upper() 

561 

562 gws.debug.time_start(f'HTTP_{method}={url!r}') 

563 res = _http_request(method, url, kwargs) 

564 gws.debug.time_end() 

565 

566 return res 

567 

568 

569_DEFAULT_CONNECT_TIMEOUT = 60 

570_DEFAULT_READ_TIMEOUT = 60 

571 

572_USER_AGENT = f'GBD WebSuite (https://gbd-websuite.de)' 

573 

574_POOL_SIZE = 16 

575"""Max. keep-alive connections per host and process.""" 

576 

577_session: requests.Session | None = None 

578 

579 

580def _get_session() -> requests.Session: 

581 """Return the HTTP session of this process, creating it if needed.""" 

582 # one session per process: reuses TCP/TLS connections to the same host across requests 

583 global _session 

584 if _session is None: 

585 s = requests.Session() 

586 adapter = requests.adapters.HTTPAdapter(pool_connections=_POOL_SIZE, pool_maxsize=_POOL_SIZE) 

587 s.mount('http://', adapter) 

588 s.mount('https://', adapter) 

589 _session = s 

590 return _session 

591 

592 

593def _http_request(method, url, kwargs) -> HTTPResponse: 

594 """Send a request with default options and wrap the result in an ``HTTPResponse``.""" 

595 

596 kwargs['stream'] = False 

597 

598 if 'verify' not in kwargs: 

599 kwargs['verify'] = certifi.where() 

600 

601 timeout = kwargs.get('timeout', (_DEFAULT_CONNECT_TIMEOUT, _DEFAULT_READ_TIMEOUT)) 

602 if isinstance(timeout, (int, float)): 

603 timeout = int(timeout), int(timeout) 

604 kwargs['timeout'] = timeout 

605 

606 if 'headers' not in kwargs: 

607 kwargs['headers'] = {} 

608 kwargs['headers'].setdefault('User-Agent', _USER_AGENT) 

609 

610 try: 

611 res = _get_session().request(method, url, **kwargs) 

612 if 200 <= res.status_code < 300: 

613 gws.log.debug(f'HTTP_OK_{method}: url={url!r} status={res.status_code!r}') 

614 return HTTPResponse(ok=True, url=url, res=res) 

615 gws.log.error(f'HTTP_FAILED_{method}: ({res.status_code!r}) url={url!r}') 

616 return HTTPResponse(ok=False, url=url, res=res) 

617 except requests.ConnectionError as exc: 

618 gws.log.error(f'HTTP_FAILED_{method}: (ConnectionError) url={url!r}') 

619 return HTTPResponse(ok=False, url=url, text=repr(exc), status_code=_STATUS_CONNECTION_ERROR) 

620 except requests.Timeout as exc: 

621 gws.log.error(f'HTTP_FAILED_{method}: (Timeout) url={url!r}') 

622 return HTTPResponse(ok=False, url=url, text=repr(exc), status_code=_STATUS_TIMEOUT) 

623 except requests.RequestException as exc: 

624 gws.log.error(f'HTTP_FAILED_{method}: (Generic: {exc!r}) url={url!r}') 

625 return HTTPResponse(ok=False, url=url, text=repr(exc), status_code=_STATUS_GENERIC_ERROR)