Script 'mail_helper' called by obssrc Hello community, here is the log from the commit of package python-WebOb for openSUSE:Factory checked in at 2026-09-23 14:32:25 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ Comparing /work/SRC/openSUSE:Factory/python-WebOb (Old) and /work/SRC/openSUSE:Factory/.python-WebOb.new.383539 (New) ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Package is "python-WebOb" Wed Sep 23 14:32:25 2026 rev:44 rq:1379547 version:1.8.11 Changes: -------- --- /work/SRC/openSUSE:Factory/python-WebOb/python-WebOb.changes 2026-06-16 18:29:27.389665299 +0200 +++ /work/SRC/openSUSE:Factory/.python-WebOb.new.383539/python-WebOb.changes 2026-09-23 14:32:55.444253892 +0200 @@ -1,0 +2,9 @@ +Mon Sep 21 23:30:20 UTC 2026 - Matej Cepl <[email protected]> + +- Update to 1.8.11: + - CVE-2026-54770: WebOb no longer uses urllib.parse.urljoin and + instead ships its own RFC 3986 urljoin implementation to + prevent open redirects via Location normalization smuggling + (bsc#1275917, GHSA-6hx8-3wjj-gr8g). + +------------------------------------------------------------------- Old: ---- webob-1.8.10.tar.gz New: ---- webob-1.8.11.tar.gz ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ Other differences: ------------------ ++++++ python-WebOb.spec ++++++ --- /var/tmp/diff_new_pack.UCy9DT/_old 2026-09-23 14:32:56.900314763 +0200 +++ /var/tmp/diff_new_pack.UCy9DT/_new 2026-09-23 14:32:56.902314847 +0200 @@ -18,25 +18,25 @@ %{?sle15_python_module_pythons} Name: python-WebOb -Version: 1.8.10 +Version: 1.8.11 Release: 0 Summary: WSGI request and response object License: MIT -URL: http://webob.org/ +URL: https://webob.org/ Source: https://files.pythonhosted.org/packages/source/w/webob/webob-%{version}.tar.gz BuildRequires: %{python_module legacy-cgi if %python-base >= 3.13} BuildRequires: %{python_module pip} BuildRequires: %{python_module pytest} BuildRequires: %{python_module setuptools} BuildRequires: %{python_module wheel} -BuildRequires: python-rpm-macros # Documentation requirements: BuildRequires: fdupes +BuildRequires: python-rpm-macros BuildRequires: python3-Sphinx +BuildArch: noarch %if %{python_version_nodots} >= 313 Requires: python-legacy-cgi >= 2.6 %endif -BuildArch: noarch %python_subpackages %description ++++++ webob-1.8.10.tar.gz -> webob-1.8.11.tar.gz ++++++ diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/webob-1.8.10/CHANGES.txt new/webob-1.8.11/CHANGES.txt --- old/webob-1.8.10/CHANGES.txt 2026-06-02 21:54:18.000000000 +0200 +++ new/webob-1.8.11/CHANGES.txt 2026-08-02 08:24:38.000000000 +0200 @@ -1,3 +1,28 @@ +1.8.11 (2026-08-02) +------------------- + +Security Fix +~~~~~~~~~~~~ + +- The fixes for CVE-2024-42353 and GHSA-fh3h-vg37-cc95 were still + incomplete: besides removing tab, CR, and LF, ``urllib.parse.urljoin`` + also strips leading and trailing C0 control and space characters from a + URL before parsing it. A Location value such as + ``" //www.example.com/test"`` could therefore still be interpreted as a + protocol-relative URL (and ``" https://www.example.com/test"`` as an + absolute one), allowing an open redirect. + + WebOb no longer uses ``urllib.parse.urljoin`` and instead ships its own + implementation of the RFC 3986 reference resolution algorithm, + ``webob.util.urljoin``, which resolves the URL exactly as given without + removing any characters. It is now used to make the Location header + absolute, by ``Request.relative_url``, and by the ``_HTTPMove`` based + HTTP exceptions, which normalize the Location header through the same + code path as the Response object (so a protocol-relative location + passed to e.g. ``HTTPFound`` no longer redirects off-host either). + + See https://github.com/Pylons/webob/security/advisories/GHSA-6hx8-3wjj-gr8g + 1.8.10 (2026-06-02) ------------------- diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/webob-1.8.10/PKG-INFO new/webob-1.8.11/PKG-INFO --- old/webob-1.8.10/PKG-INFO 2026-06-02 21:55:46.617048000 +0200 +++ new/webob-1.8.11/PKG-INFO 2026-08-02 08:25:38.252923500 +0200 @@ -1,6 +1,6 @@ Metadata-Version: 2.4 Name: WebOb -Version: 1.8.10 +Version: 1.8.11 Summary: WSGI request and response object Home-page: http://webob.org/ Author: Ian Bicking @@ -84,6 +84,31 @@ WebOb was authored by Ian Bicking and is currently maintained by the `Pylons Project <https://pylonsproject.org/>`_ and a team of contributors. +1.8.11 (2026-08-02) +------------------- + +Security Fix +~~~~~~~~~~~~ + +- The fixes for CVE-2024-42353 and GHSA-fh3h-vg37-cc95 were still + incomplete: besides removing tab, CR, and LF, ``urllib.parse.urljoin`` + also strips leading and trailing C0 control and space characters from a + URL before parsing it. A Location value such as + ``" //www.example.com/test"`` could therefore still be interpreted as a + protocol-relative URL (and ``" https://www.example.com/test"`` as an + absolute one), allowing an open redirect. + + WebOb no longer uses ``urllib.parse.urljoin`` and instead ships its own + implementation of the RFC 3986 reference resolution algorithm, + ``webob.util.urljoin``, which resolves the URL exactly as given without + removing any characters. It is now used to make the Location header + absolute, by ``Request.relative_url``, and by the ``_HTTPMove`` based + HTTP exceptions, which normalize the Location header through the same + code path as the Response object (so a protocol-relative location + passed to e.g. ``HTTPFound`` no longer redirects off-host either). + + See https://github.com/Pylons/webob/security/advisories/GHSA-6hx8-3wjj-gr8g + 1.8.10 (2026-06-02) ------------------- diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/webob-1.8.10/setup.py new/webob-1.8.11/setup.py --- old/webob-1.8.10/setup.py 2026-06-02 21:54:39.000000000 +0200 +++ new/webob-1.8.11/setup.py 2026-08-02 08:24:38.000000000 +0200 @@ -26,7 +26,7 @@ setup( name='WebOb', - version='1.8.10', + version='1.8.11', description="WSGI request and response object", long_description=README + '\n\n' + CHANGES, classifiers=[ diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/webob-1.8.10/src/WebOb.egg-info/PKG-INFO new/webob-1.8.11/src/WebOb.egg-info/PKG-INFO --- old/webob-1.8.10/src/WebOb.egg-info/PKG-INFO 2026-06-02 21:55:46.000000000 +0200 +++ new/webob-1.8.11/src/WebOb.egg-info/PKG-INFO 2026-08-02 08:25:38.000000000 +0200 @@ -1,6 +1,6 @@ Metadata-Version: 2.4 Name: WebOb -Version: 1.8.10 +Version: 1.8.11 Summary: WSGI request and response object Home-page: http://webob.org/ Author: Ian Bicking @@ -84,6 +84,31 @@ WebOb was authored by Ian Bicking and is currently maintained by the `Pylons Project <https://pylonsproject.org/>`_ and a team of contributors. +1.8.11 (2026-08-02) +------------------- + +Security Fix +~~~~~~~~~~~~ + +- The fixes for CVE-2024-42353 and GHSA-fh3h-vg37-cc95 were still + incomplete: besides removing tab, CR, and LF, ``urllib.parse.urljoin`` + also strips leading and trailing C0 control and space characters from a + URL before parsing it. A Location value such as + ``" //www.example.com/test"`` could therefore still be interpreted as a + protocol-relative URL (and ``" https://www.example.com/test"`` as an + absolute one), allowing an open redirect. + + WebOb no longer uses ``urllib.parse.urljoin`` and instead ships its own + implementation of the RFC 3986 reference resolution algorithm, + ``webob.util.urljoin``, which resolves the URL exactly as given without + removing any characters. It is now used to make the Location header + absolute, by ``Request.relative_url``, and by the ``_HTTPMove`` based + HTTP exceptions, which normalize the Location header through the same + code path as the Response object (so a protocol-relative location + passed to e.g. ``HTTPFound`` no longer redirects off-host either). + + See https://github.com/Pylons/webob/security/advisories/GHSA-6hx8-3wjj-gr8g + 1.8.10 (2026-06-02) ------------------- diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/webob-1.8.10/src/webob/exc.py new/webob-1.8.11/src/webob/exc.py --- old/webob-1.8.10/src/webob/exc.py 2026-05-06 08:46:29.000000000 +0200 +++ new/webob-1.8.11/src/webob/exc.py 2026-08-02 08:24:38.000000000 +0200 @@ -175,7 +175,6 @@ class_types, text_, text_type, - urlparse, ) from webob.request import Request from webob.response import Response @@ -530,7 +529,14 @@ if req.environ.get('QUERY_STRING'): url += '?' + req.environ['QUERY_STRING'] self.location = url - self.location = urlparse.urljoin(req.path_url, self.location) + if self.location: + # Normalize the location through the same code path used for + # the Location header so that a relative (or protocol-relative) + # location cannot turn into an open redirect. See + # CVE-2024-42353, GHSA-fh3h-vg37-cc95, and GHSA-6hx8-3wjj-gr8g. + self.location = self._make_location_absolute(environ, self.location) + else: + self.location = req.path_url return super(_HTTPMove, self).__call__( environ, start_response) diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/webob-1.8.10/src/webob/request.py new/webob-1.8.11/src/webob/request.py --- old/webob-1.8.10/src/webob/request.py 2026-05-06 08:46:29.000000000 +0200 +++ new/webob-1.8.11/src/webob/request.py 2026-08-02 08:24:38.000000000 +0200 @@ -67,6 +67,8 @@ from webob.headers import EnvironHeaders +from webob.util import urljoin + from webob.multidict import ( NestedMultiDict, MultiDict, @@ -511,7 +513,10 @@ url += '/' else: url = self.path_url - return urlparse.urljoin(url, other_url) + # Use WebOb's own RFC 3986 urljoin() rather than + # urllib.parse.urljoin(), which removes ASCII tab/CR/LF and strips + # leading/trailing C0 control and space characters before parsing. + return urljoin(url, other_url) def path_info_pop(self, pattern=None): """ diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/webob-1.8.10/src/webob/response.py new/webob-1.8.11/src/webob/response.py --- old/webob-1.8.10/src/webob/response.py 2026-05-06 08:46:29.000000000 +0200 +++ new/webob-1.8.11/src/webob/response.py 2026-08-02 08:24:38.000000000 +0200 @@ -14,7 +14,6 @@ string_types, text_type, url_quote, - urlparse, ) from webob.cookies import Cookie, make_cookie from webob.datetime_utils import ( @@ -41,7 +40,12 @@ ) from webob.headers import ResponseHeaders from webob.request import BaseRequest -from webob.util import status_generic_reasons, status_reasons, warn_deprecation +from webob.util import ( + status_generic_reasons, + status_reasons, + urljoin, + warn_deprecation, +) try: import simplejson as json @@ -1281,10 +1285,9 @@ @staticmethod def _make_location_absolute(environ, value): - # urllib.parse.urlsplit() (called internally by urljoin) strips - # ASCII tab, CR, and LF from the URL on Python 3.10+. Strip them - # ourselves first so they cannot be used to bypass the SCHEME_RE - # or protocol-relative ("//") checks below. See CVE-2024-42353, + # Strip ASCII tab, CR, and LF so they cannot be used to smuggle a + # protocol-relative URL past the checks below (user agents remove + # them when parsing a URL). See CVE-2024-42353, # https://github.com/Pylons/webob/security/advisories/GHSA-mg3v-6m49-jhp3, # and the follow-up advisory GHSA-fh3h-vg37-cc95. value = value.replace("\t", "").replace("\r", "").replace("\n", "") @@ -1294,7 +1297,15 @@ if value.startswith("//"): value = "/%2f{}".format(value[2:]) - new_location = urlparse.urljoin(_request_uri(environ), value) + + # urllib.parse.urljoin() removes ASCII tab/CR/LF anywhere in the + # URL and strips leading and trailing C0 control and space + # characters before parsing (Python 3.10+). That turns values such + # as " //evil.example" into protocol-relative URLs, bypassing the + # checks above. Use WebOb's own RFC 3986 urljoin(), which resolves + # the value exactly as given. See GHSA-6hx8-3wjj-gr8g. + new_location = urljoin(_request_uri(environ), value) + return new_location def _abs_headerlist(self, environ): diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/webob-1.8.10/src/webob/util.py new/webob-1.8.11/src/webob/util.py --- old/webob-1.8.10/src/webob/util.py 2026-05-06 08:46:29.000000000 +0200 +++ new/webob-1.8.11/src/webob/util.py 2026-08-02 08:24:38.000000000 +0200 @@ -1,3 +1,4 @@ +import re import warnings from webob.compat import ( @@ -35,6 +36,189 @@ s = s.encode('ascii', 'xmlcharrefreplace') return text_(s) + +# RFC 3986 section 3.1: scheme = ALPHA *( ALPHA / DIGIT / "+" / "-" / "." ) +_URI_SCHEME_RE = re.compile(r"^[A-Za-z][A-Za-z0-9+\-.]*$") + + +def _split_uri_reference(uri): + """Split a URI reference into its five components. + + Returns a ``(scheme, authority, path, query, fragment)`` tuple, + following the grammar from RFC 3986 (see appendix B). Components that + are not present in the reference are ``None``. The path is always + present, but may be the empty string. + + Unlike ``urllib.parse.urlsplit()``, no characters are ever removed + from the reference: ASCII tab/CR/LF and leading or trailing C0 + control and space characters are treated like any other character. + """ + scheme = authority = query = fragment = None + + rest, sep, token = uri.partition("#") + + if sep: + fragment = token + + rest, sep, token = rest.partition("?") + + if sep: + query = token + + token, sep, candidate = rest.partition(":") + + if sep and _URI_SCHEME_RE.match(token): + scheme = token + rest = candidate + + if rest.startswith("//"): + end = rest.find("/", 2) + + if end == -1: + authority, rest = rest[2:], "" + else: + authority, rest = rest[2:end], rest[end:] + + return scheme, authority, rest, query, fragment + + +def _remove_dot_segments(path): + """Remove ``.`` and ``..`` segments from a path (RFC 3986 5.2.4).""" + output = [] + + while path: + if path.startswith("../"): + path = path[3:] + elif path.startswith("./"): + path = path[2:] + elif path.startswith("/./"): + path = "/" + path[3:] + elif path == "/.": + path = "/" + elif path.startswith("/../"): + path = "/" + path[4:] + + if output: + output.pop() + elif path == "/..": + path = "/" + + if output: + output.pop() + elif path in (".", ".."): + path = "" + else: + end = path.find("/", 1) if path.startswith("/") else path.find("/") + + if end == -1: + output.append(path) + path = "" + else: + output.append(path[:end]) + path = path[end:] + + return "".join(output) + + +def _merge_paths(base_authority, base_path, path): + """Merge a relative-path reference with the base path (RFC 3986 5.2.3).""" + + if base_authority is not None and base_path == "": + return "/" + path + + if "/" in base_path: + return base_path[: base_path.rfind("/") + 1] + path + + return path + + +def urljoin(base, url): + """Resolve a URI reference relative to a base URI (RFC 3986 section 5). + + A replacement for ``urllib.parse.urljoin()``. The standard library + implementation follows the WHATWG URL living standard (on Python + 3.10+) by removing ASCII tab, CR, and LF anywhere in the URL and + stripping leading and trailing C0 control and space characters + before parsing. Those transformations can silently turn an otherwise + harmless relative reference such as ``" //evil.example"`` into a + protocol-relative or absolute URL, which has repeatedly led to open + redirect issues when normalizing the ``Location`` header (see + CVE-2024-42353/GHSA-mg3v-6m49-jhp3, GHSA-fh3h-vg37-cc95, and + GHSA-6hx8-3wjj-gr8g). + + This implementation resolves the reference exactly as given, + character for character, with no whitespace removal whatsoever. + """ + + # Mirror urllib.parse.urljoin()'s short-circuits for degenerate input + # (such as a reference of None or the empty string), which callers of + # Request.relative_url() may rely on. + + if not base: + return url + + if not url: + return base + + b_scheme, b_authority, b_path, b_query, b_fragment = _split_uri_reference(base) + r_scheme, r_authority, r_path, r_query, r_fragment = _split_uri_reference(url) + + # Like urllib.parse.urljoin(), use the non-strict variant of the + # resolution algorithm (RFC 3986 5.2.2): a reference whose scheme + # matches the base scheme is treated as a relative reference. + + if ( + r_scheme is not None + and b_scheme is not None + and r_scheme.lower() == b_scheme.lower() + ): + r_scheme = None + + if r_scheme is not None: + scheme = r_scheme + authority = r_authority + path = _remove_dot_segments(r_path) + query = r_query + elif r_authority is not None: + scheme = b_scheme + authority = r_authority + path = _remove_dot_segments(r_path) + query = r_query + elif r_path == "": + scheme = b_scheme + authority = b_authority + path = b_path + query = r_query if r_query is not None else b_query + else: + scheme = b_scheme + authority = b_authority + + if r_path.startswith("/"): + path = _remove_dot_segments(r_path) + else: + path = _remove_dot_segments(_merge_paths(b_authority, b_path, r_path)) + query = r_query + fragment = r_fragment + + # Recompose the components (RFC 3986 5.3) + result = [] + + if scheme is not None: + result.append(scheme + ":") + + if authority is not None: + result.append("//" + authority) + result.append(path) + + if query is not None: + result.append("?" + query) + + if fragment is not None: + result.append("#" + fragment) + + return "".join(result) + + def header_docstring(header, rfc_section): if header.isupper(): header = _trans_key(header) diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/webob-1.8.10/tests/test_exc.py new/webob-1.8.11/tests/test_exc.py --- old/webob-1.8.10/tests/test_exc.py 2026-05-06 08:46:29.000000000 +0200 +++ new/webob-1.8.11/tests/test_exc.py 2026-08-02 08:24:38.000000000 +0200 @@ -402,6 +402,64 @@ environ['PATH_INFO'] = '/' assert m( environ, start_response ) == [] [email protected]( + "location, expected", + [ + # protocol-relative URLs must not redirect off-host, just like + # Response.location (CVE-2024-42353) + ("//www.example.com/test", "http://localhost/%2fwww.example.com/test"), + # whitespace must not be usable to smuggle a protocol-relative or + # absolute URL past the checks (GHSA-fh3h-vg37-cc95 and + # GHSA-6hx8-3wjj-gr8g) + ("/\t/www.example.com/test", "http://localhost/%2fwww.example.com/test"), + (" //www.example.com/test", "http://localhost/ //www.example.com/test"), + ( + " https://www.example.com/test", + "http://localhost/ https://www.example.com/test", + ), + # relative and absolute locations keep working + ("test", "http://localhost/test"), + ("/test", "http://localhost/test"), + ("https://www.example.com/test", "https://www.example.com/test"), + ], +) +def test_HTTPMove_location_no_open_redirect(location, expected): + headers = [] + + def start_response(status, response_headers, exc_info=None): + headers[:] = response_headers + + environ = { + "wsgi.url_scheme": "http", + "SERVER_NAME": "localhost", + "SERVER_PORT": "80", + "REQUEST_METHOD": "HEAD", + "PATH_INFO": "/", + } + m = webob_exc._HTTPMove(location=location) + assert m(environ, start_response) == [] + assert dict(headers)["Location"] == expected + assert m.location == expected + + +def test_HTTPMove_location_none_defaults_to_request_url(): + headers = [] + + def start_response(status, response_headers, exc_info=None): + headers[:] = response_headers + + environ = { + "wsgi.url_scheme": "http", + "SERVER_NAME": "localhost", + "SERVER_PORT": "80", + "REQUEST_METHOD": "HEAD", + "PATH_INFO": "/", + } + m = webob_exc._HTTPMove() + assert m(environ, start_response) == [] + assert dict(headers)["Location"] == "http://localhost/" + + def test_HTTPFound_unused_environ_variable(): class Crashy(object): def __str__(self): diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/webob-1.8.10/tests/test_response.py new/webob-1.8.11/tests/test_response.py --- old/webob-1.8.10/tests/test_response.py 2026-05-06 08:46:29.000000000 +0200 +++ new/webob-1.8.11/tests/test_response.py 2026-08-02 08:24:38.000000000 +0200 @@ -1088,6 +1088,54 @@ assert result == "http://example.com/%2fwww.example.com/test" [email protected]( + "payload, expected", + [ + ( + " //www.example.com/test", + "http://localhost/ //www.example.com/test", + ), + ( + "\x00//www.example.com/test", + "http://localhost/\x00//www.example.com/test", + ), + ( + "\x1f//www.example.com/test", + "http://localhost/\x1f//www.example.com/test", + ), + ( + " \t//www.example.com/test", + "http://localhost/ //www.example.com/test", + ), + ( + " //www.example.com/test ", + "http://localhost/ //www.example.com/test ", + ), + ( + " https://www.example.com/test", + "http://localhost/ https://www.example.com/test", + ), + ( + "\x00https://www.example.com/test", + "http://localhost/\x00https://www.example.com/test", + ), + ], +) +def test_location_no_open_redirect_c0_control_or_space(payload, expected): + # Follow-up to GHSA-fh3h-vg37-cc95. urllib.parse.urljoin() also strips + # leading/trailing C0 control and space characters before parsing, so + # a Location value such as " //www.example.com/test" was still parsed + # as protocol-relative (and " https://www.example.com/test" as + # absolute), bypassing the SCHEME_RE and "//" checks. WebOb now uses + # its own RFC 3986 urljoin() that resolves the value exactly as given, + # keeping the redirect on the request's host. See GHSA-6hx8-3wjj-gr8g. + res = Response() + res.status = "301" + res.location = payload + req = Request.blank("/") + assert req.get_response(res).location == expected + + @pytest.mark.xfail(sys.version_info < (3,0), reason="Python 2.x unicode != str, WSGI requires str. Test " "added due to https://github.com/Pylons/webob/issues/247. " diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/webob-1.8.10/tests/test_util.py new/webob-1.8.11/tests/test_util.py --- old/webob-1.8.10/tests/test_util.py 2026-05-06 08:46:29.000000000 +0200 +++ new/webob-1.8.11/tests/test_util.py 2026-08-02 08:24:38.000000000 +0200 @@ -1,5 +1,13 @@ import unittest + +import pytest + from webob.response import Response +from webob.util import ( + _merge_paths, + _remove_dot_segments, + urljoin, +) class Test_warn_deprecation(unittest.TestCase): def setUp(self): @@ -92,3 +100,160 @@ self.assertTrue(dummy_compare.called) self.assertTrue(result) + +RFC3986_BASE = "http://a/b/c/d;p?q" + + [email protected]( + "reference, expected", + [ + # RFC 3986 section 5.4.1, normal examples + ("g:h", "g:h"), + ("g", "http://a/b/c/g"), + ("./g", "http://a/b/c/g"), + ("g/", "http://a/b/c/g/"), + ("/g", "http://a/g"), + ("//g", "http://g"), + ("?y", "http://a/b/c/d;p?y"), + ("g?y", "http://a/b/c/g?y"), + ("#s", "http://a/b/c/d;p?q#s"), + ("g#s", "http://a/b/c/g#s"), + ("g?y#s", "http://a/b/c/g?y#s"), + (";x", "http://a/b/c/;x"), + ("g;x", "http://a/b/c/g;x"), + ("g;x?y#s", "http://a/b/c/g;x?y#s"), + ("", "http://a/b/c/d;p?q"), + (".", "http://a/b/c/"), + ("./", "http://a/b/c/"), + ("..", "http://a/b/"), + ("../", "http://a/b/"), + ("../g", "http://a/b/g"), + ("../..", "http://a/"), + ("../../", "http://a/"), + ("../../g", "http://a/g"), + # RFC 3986 section 5.4.2, abnormal examples + ("../../../g", "http://a/g"), + ("../../../../g", "http://a/g"), + ("/./g", "http://a/g"), + ("/../g", "http://a/g"), + ("g.", "http://a/b/c/g."), + (".g", "http://a/b/c/.g"), + ("g..", "http://a/b/c/g.."), + ("..g", "http://a/b/c/..g"), + ("./../g", "http://a/b/g"), + ("./g/.", "http://a/b/c/g/"), + ("g/./h", "http://a/b/c/g/h"), + ("g/../h", "http://a/b/c/h"), + ("g;x=1/./y", "http://a/b/c/g;x=1/y"), + ("g;x=1/../y", "http://a/b/c/y"), + ("g#s/./x", "http://a/b/c/g#s/./x"), + ("g#s/../x", "http://a/b/c/g#s/../x"), + # Like urllib.parse.urljoin(), use the non-strict (backwards + # compatible) variant of RFC 3986 5.2.2: a reference with the same + # scheme as the base is treated as relative. A strict parser would + # return "http:g" here. + ("http:g", "http://a/b/c/g"), + ("HTTP:g", "http://a/b/c/g"), + ], +) +def test_urljoin_rfc3986_reference_resolution(reference, expected): + # The examples from RFC 3986 section 5.4 + assert urljoin(RFC3986_BASE, reference) == expected + + [email protected]( + "reference, expected", + [ + # urllib.parse.urljoin() removes ASCII tab/CR/LF anywhere in the + # URL and strips leading/trailing C0 control and space characters + # before parsing, which can promote a relative path to a + # protocol-relative or absolute URL. WebOb's urljoin() must treat + # those characters like any other character. See + # GHSA-6hx8-3wjj-gr8g, GHSA-fh3h-vg37-cc95, and CVE-2024-42353. + (" //www.example.com/test", "http://a/b/c/ //www.example.com/test"), + ("\t//www.example.com/test", "http://a/b/c/\t//www.example.com/test"), + ("\n//www.example.com/test", "http://a/b/c/\n//www.example.com/test"), + ("\r//www.example.com/test", "http://a/b/c/\r//www.example.com/test"), + ("\x00//www.example.com/test", "http://a/b/c/\x00//www.example.com/test"), + ("\x1f//www.example.com/test", "http://a/b/c/\x1f//www.example.com/test"), + ("/\t/www.example.com/test", "http://a/\t/www.example.com/test"), + ("/\r\n/www.example.com/test", "http://a/\r\n/www.example.com/test"), + (" http://www.example.com/test", "http://a/b/c/ http://www.example.com/test"), + ( + "\thttps://www.example.com/test", + "http://a/b/c/\thttps://www.example.com/test", + ), + ( + "https\t://www.example.com/test", + "http://a/b/c/https\t://www.example.com/test", + ), + ], +) +def test_urljoin_does_not_strip_whitespace(reference, expected): + assert urljoin(RFC3986_BASE, reference) == expected + + [email protected]( + "base, reference, expected", + [ + # authority with an empty path + ("http://a", "g", "http://a/g"), + # empty authority + ("http://a/b", "///g", "http:///g"), + # base without an authority + ( + "mailto:[email protected]", + "[email protected]", + "mailto:[email protected]", + ), + # base without a scheme + ("//a/b/c", "g", "//a/b/g"), + # reference with an authority and a path + ("http://a/b/c", "//g/x/../y?q#f", "http://g/y?q#f"), + # a colon in the first segment of a relative path is parsed as a + # scheme, as in urllib.parse.urljoin(); use "./" to avoid that + ("http://a/b/c", "g:x/y", "g:x/y"), + ("http://a/b/c", "./g:x/y", "http://a/b/g:x/y"), + # invalid scheme (must start with ALPHA) means a relative path + ("http://a/b/c", "0http://evil.example", "http://a/b/0http://evil.example"), + # empty fragment and query are preserved + ("http://a/b/c", "g?", "http://a/b/g?"), + ("http://a/b/c", "g#", "http://a/b/g#"), + # base query/fragment are dropped when the reference has a path + ("http://a/b/c?q#f", "g", "http://a/b/g"), + # degenerate input short-circuits, like urllib.parse.urljoin() + ("http://a/b/c#f", "", "http://a/b/c#f"), + ("http://a/b/c", None, "http://a/b/c"), + ("", "g", "g"), + (None, "g", "g"), + ], +) +def test_urljoin_component_edge_cases(base, reference, expected): + assert urljoin(base, reference) == expected + + [email protected]( + "path, expected", + [ + ("", ""), + (".", ""), + ("..", ""), + ("./g", "g"), + ("../g", "g"), + ("/.", "/"), + ("/..", "/"), + ("/./g", "/g"), + ("/../g", "/g"), + ("/a/b/c/./../../g", "/a/g"), + ("mid/content=5/../6", "mid/6"), + ], +) +def test_remove_dot_segments(path, expected): + # The examples from RFC 3986 section 5.2.4, plus relative-path edge + # cases that urljoin() itself cannot reach with an absolute base URI + assert _remove_dot_segments(path) == expected + + +def test_merge_paths_relative_base_without_slash(): + # only reachable through urljoin() with a relative, rootless base + assert _merge_paths(None, "x", "y") == "y"
