tornadoweb-tornado-3553-3554
Two quadratic-time paths in tornado/httputil.py can be abused for denial of service. Both must become roughly linear while keeping every parsed result identical.
Repeated header values. `HTTPHeaders.add(name, value)` rebuilds the comma-joined combined value for a header on every call, so adding the same header name n times costs quadratic time. Adding `"X-Foo": "bar"` 100,000 times to a fresh `HTTPHeaders` must take less than 20 times as long as adding it 10,000 times. `headers[name]` must still return the comma-joined value, reflecting every value added so far, including a continuation line appended to the last header. Membership (`name in headers`, false for non-string keys), `len(headers)`, iteration and deletion must all reflect the header names that have values, and `get_list` and `get_all` are unchanged.
Semicolons inside a quoted parameter. The parser for header parameters such as `Content-Disposition` rescans from the start of the parameter for quote parity every time it meets a `;`, so a quoted value containing many semicolons costs quadratic time. Parsing a multipart body through `parse_multipart_form_data(b"1234", message, args, files)` whose part header is `Content-Disposition: form-data; x="` followed by n semicolons and `"; name="files"; filename="a.txt"` must take less than 20 times as long for n = 10,000 as for n = 1,000. Escaped quotes (`\"`) must still not count toward quote parity, and the parsed parameters, including `name` and `filename`, must be unchanged.
Hidden tests · 2 fail-to-pass, 61 pass-to-passrun after the agent submits, in a clean verifier
Test patch · 78 lines
diff --git a/tornado/test/httputil_test.py b/tornado/test/httputil_test.py
index 7a46b911..ed06beae 100644
--- a/tornado/test/httputil_test.py
+++ b/tornado/test/httputil_test.py
@@ -279,6 +279,29 @@ Foo
self.assertEqual(file["filename"], "ab.txt")
self.assertEqual(file["body"], b"Foo")
+ def test_disposition_param_linear_performance(self):
+ # This is a regression test for performance of parsing parameters
+ # to the content-disposition header, specifically for semicolons within
+ # quoted strings.
+ def f(n):
+ start = time.time()
+ message = (
+ b"--1234\r\nContent-Disposition: form-data; "
+ + b'x="'
+ + b";" * n
+ + b'"; '
+ + b'name="files"; filename="a.txt"\r\n\r\nFoo\r\n--1234--\r\n'
+ )
+ args: dict[str, list[bytes]] = {}
+ files: dict[str, list[HTTPFile]] = {}
+ parse_multipart_form_data(b"1234", message, args, files)
+ return time.time() - start
+
+ d1 = f(1_000)
+ d2 = f(10_000)
+ if d2 / d1 > 20:
+ self.fail(f"Disposition param parsing is not linear: {d1=} vs {d2=}")
+
class HTTPHeadersTest(unittest.TestCase):
def test_multi_line(self):
@@ -471,6 +494,21 @@ Foo: even
with self.assertRaises(HTTPInputError):
headers.add(name, "bar")
+ def test_linear_performance(self):
+ def f(n):
+ start = time.time()
+ headers = HTTPHeaders()
+ for i in range(n):
+ headers.add("X-Foo", "bar")
+ return time.time() - start
+
+ # This runs under 50ms on my laptop as of 2025-12-09.
+ d1 = f(10_000)
+ d2 = f(100_000)
+ if d2 / d1 > 20:
+ # d2 should be about 10x d1 but allow a wide margin for variability.
+ self.fail(f"HTTPHeaders.add() does not scale linearly: {d1=} vs {d2=}")
+
class FormatTimestampTest(unittest.TestCase):
# Make sure that all the input types are supported.
diff --git a/tornado/test/process_test.py b/tornado/test/process_test.py
index 0fdb9418..1c5cff32 100644
--- a/tornado/test/process_test.py
+++ b/tornado/test/process_test.py
@@ -141,7 +141,7 @@ class SubprocessTest(AsyncTestCase):
@gen_test
def test_subprocess(self):
subproc = Subprocess(
- [sys.executable, "-u", "-i"],
+ [sys.executable, "-u", "-i", "-I"],
stdin=Subprocess.STREAM,
stdout=Subprocess.STREAM,
stderr=subprocess.STDOUT,
@@ -163,7 +163,7 @@ class SubprocessTest(AsyncTestCase):
def test_close_stdin(self):
# Close the parent's stdin handle and see that the child recognizes it.
subproc = Subprocess(
- [sys.executable, "-u", "-i"],
+ [sys.executable, "-u", "-i", "-I"],
stdin=Subprocess.STREAM,
stdout=Subprocess.STREAM,
stderr=subprocess.STDOUT,
Reference fix · 1 file, +47 −18the upstream merge, used only for grading calibration
The agent could not see this: the repository holds one commit and the sandbox has no network. Leak audit.
tornado/httputil.py
diff --git a/tornado/httputil.py b/tornado/httputil.py
index a418b9fce..e13231e24 100644
--- a/tornado/httputil.py
+++ b/tornado/httputil.py
@@ -187,8 +187,14 @@ def __init__(self, **kwargs: str) -> None:
pass
def __init__(self, *args: typing.Any, **kwargs: str) -> None: # noqa: F811
- self._dict = {} # type: typing.Dict[str, str]
- self._as_list = {} # type: typing.Dict[str, typing.List[str]]
+ # Formally, HTTP headers are a mapping from a field name to a "combined field value",
+ # which may be constructed from multiple field lines by joining them with commas.
+ # In practice, however, some headers (notably Set-Cookie) do not follow this convention,
+ # so we maintain a mapping from field name to a list of field lines in self._as_list.
+ # self._combined_cache is a cache of the combined field values derived from self._as_list
+ # on demand (and cleared whenever the list is modified).
+ self._as_list: dict[str, list[str]] = {}
+ self._combined_cache: dict[str, str] = {}
self._last_key = None # type: Optional[str]
if len(args) == 1 and len(kwargs) == 0 and isinstance(args[0], HTTPHeaders):
# Copy constructor
@@ -215,9 +221,7 @@ def add(self, name: str, value: str, *, _chars_are_bytes: bool = True) -> None:
norm_name = _normalize_header(name)
self._last_key = norm_name
if norm_name in self:
- self._dict[norm_name] = (
- native_str(self[norm_name]) + "," + native_str(value)
- )
+ self._combined_cache.pop(norm_name, None)
self._as_list[norm_name].append(value)
else:
self[norm_name] = value
@@ -278,7 +282,7 @@ def parse_line(self, line: str, *, _chars_are_bytes: bool = True) -> None:
if _FORBIDDEN_HEADER_CHARS_RE.search(new_part):
raise HTTPInputError("Invalid header value %r" % new_part)
self._as_list[self._last_key][-1] += new_part
- self._dict[self._last_key] += new_part
+ self._combined_cache.pop(self._last_key, None)
else:
try:
name, value = line.split(":", 1)
@@ -333,22 +337,32 @@ def parse(cls, headers: str, *, _chars_are_bytes: bool = True) -> "HTTPHeaders":
def __setitem__(self, name: str, value: str) -> None:
norm_name = _normalize_header(name)
- self._dict[norm_name] = value
+ self._combined_cache[norm_name] = value
self._as_list[norm_name] = [value]
+ def __contains__(self, name: object) -> bool:
+ # This is an important optimization to avoid the expensive concatenation
+ # in __getitem__ when it's not needed.
+ if not isinstance(name, str):
+ return False
+ return name in self._as_list
+
def __getitem__(self, name: str) -> str:
- return self._dict[_normalize_header(name)]
+ header = _normalize_header(name)
+ if header not in self._combined_cache:
+ self._combined_cache[header] = ",".join(self._as_list[header])
+ return self._combined_cache[header]
def __delitem__(self, name: str) -> None:
norm_name = _normalize_header(name)
- del self._dict[norm_name]
+ del self._combined_cache[norm_name]
del self._as_list[norm_name]
def __len__(self) -> int:
- return len(self._dict)
+ return len(self._as_list)
def __iter__(self) -> Iterator[typing.Any]:
- return iter(self._dict)
+ return iter(self._as_list)
def copy(self) -> "HTTPHeaders":
# defined in dict but not in MutableMapping.
diff --git a/tornado/httputil.py b/tornado/httputil.py
index e13231e24..13a95d909 100644
--- a/tornado/httputil.py
+++ b/tornado/httputil.py
@@ -1096,19 +1096,34 @@ def parse_response_start_line(line: str) -> ResponseStartLine:
# It has also been modified to support valueless parameters as seen in
# websocket extension negotiations, and to support non-ascii values in
# RFC 2231/5987 format.
+#
+# _parseparam has been further modified with the logic from
+# https://github.com/python/cpython/pull/136072/files
+# to avoid quadratic behavior when parsing semicolons in quoted strings.
+#
+# TODO: See if we can switch to email.message.Message for this functionality.
+# This is the suggested replacement for the cgi.py module now that cgi has
+# been removed from recent versions of Python. We need to verify that
+# the email module is consistent with our existing behavior (and all relevant
+# RFCs for multipart/form-data) before making this change.
def _parseparam(s: str) -> Generator[str, None, None]:
- while s[:1] == ";":
- s = s[1:]
- end = s.find(";")
- while end > 0 and (s.count('"', 0, end) - s.count('\\"', 0, end)) % 2:
- end = s.find(";", end + 1)
+ start = 0
+ while s.find(";", start) == start:
+ start += 1
+ end = s.find(";", start)
+ ind, diff = start, 0
+ while end > 0:
+ diff += s.count('"', ind, end) - s.count('\\"', ind, end)
+ if diff % 2 == 0:
+ break
+ end, ind = ind, s.find(";", end + 1)
if end < 0:
end = len(s)
- f = s[:end]
+ f = s[start:end]
yield f.strip()
- s = s[end:]
+ start = end
def _parse_header(line: str) -> Tuple[str, Dict[str, str]]:
GLM-5.3 Flash · attempt 2 · solved · 18 steps · $0.007
Submitted diff · 92 lines
diff --git a/tornado/httputil.py b/tornado/httputil.py
index a418b9f..40fe067 100644
--- a/tornado/httputil.py
+++ b/tornado/httputil.py
@@ -190,6 +190,10 @@ class HTTPHeaders(StrMutableMapping):
self._dict = {} # type: typing.Dict[str, str]
self._as_list = {} # type: typing.Dict[str, typing.List[str]]
self._last_key = None # type: Optional[str]
+ # Names whose combined value in self._dict is out of date and must be
+ # recomputed from self._as_list before it is next read. This keeps
+ # repeated calls to add() linear instead of quadratic.
+ self._dirty = set() # type: typing.Set[str]
if len(args) == 1 and len(kwargs) == 0 and isinstance(args[0], HTTPHeaders):
# Copy constructor
for k, v in args[0].get_all():
@@ -215,10 +219,8 @@ class HTTPHeaders(StrMutableMapping):
norm_name = _normalize_header(name)
self._last_key = norm_name
if norm_name in self:
- self._dict[norm_name] = (
- native_str(self[norm_name]) + "," + native_str(value)
- )
self._as_list[norm_name].append(value)
+ self._dirty.add(norm_name)
else:
self[norm_name] = value
@@ -278,7 +280,8 @@ class HTTPHeaders(StrMutableMapping):
if _FORBIDDEN_HEADER_CHARS_RE.search(new_part):
raise HTTPInputError("Invalid header value %r" % new_part)
self._as_list[self._last_key][-1] += new_part
- self._dict[self._last_key] += new_part
+ if self._last_key not in self._dirty:
+ self._dict[self._last_key] += new_part
else:
try:
name, value = line.split(":", 1)
@@ -335,14 +338,25 @@ class HTTPHeaders(StrMutableMapping):
norm_name = _normalize_header(name)
self._dict[norm_name] = value
self._as_list[norm_name] = [value]
+ self._dirty.discard(norm_name)
def __getitem__(self, name: str) -> str:
- return self._dict[_normalize_header(name)]
+ norm_name = _normalize_header(name)
+ if norm_name in self._dirty:
+ self._dict[norm_name] = ",".join(self._as_list[norm_name])
+ self._dirty.discard(norm_name)
+ return self._dict[norm_name]
def __delitem__(self, name: str) -> None:
norm_name = _normalize_header(name)
del self._dict[norm_name]
del self._as_list[norm_name]
+ self._dirty.discard(norm_name)
+
+ def __contains__(self, name: object) -> bool:
+ if not isinstance(name, str):
+ return False
+ return _normalize_header(name) in self._dict
def __len__(self) -> int:
return len(self._dict)
@@ -1087,9 +1101,24 @@ def parse_response_start_line(line: str) -> ResponseStartLine:
def _parseparam(s: str) -> Generator[str, None, None]:
while s[:1] == ";":
s = s[1:]
- end = s.find(";")
- while end > 0 and (s.count('"', 0, end) - s.count('\\"', 0, end)) % 2:
- end = s.find(";", end + 1)
+ # Scan forward once, tracking quote parity, instead of rescanning
+ # from the start for each candidate separator (which is quadratic).
+ end = -1
+ quotes = 0
+ i = 0
+ n = len(s)
+ while i < n:
+ c = s[i]
+ if c == "\\" and i + 1 < n and s[i + 1] == '"':
+ # An escaped quote does not affect quote parity.
+ i += 2
+ continue
+ if c == '"':
+ quotes += 1
+ elif c == ";" and quotes % 2 == 0:
+ end = i
+ break
+ i += 1
if end < 0:
end = len(s)
f = s[:end]$ grep -n "def add\|_as_list\|def __setitem__\|class HTTPHeaders" /repo/tornado/httputil.py | head; grep -n "_parse_header_params\|parse_multipart_form_data\|_parseparam" /repo/tornado/httputil.py | head
output · 15 lines
144:class HTTPHeaders(StrMutableMapping):
191: self._as_list = {} # type: typing.Dict[str, typing.List[str]]
203: def add(self, name: str, value: str, *, _chars_are_bytes: bool = True) -> None:
221: self._as_list[norm_name].append(value)
228: return self._as_list.get(norm_name, [])
236: for name, values in self._as_list.items():
280: self._as_list[self._last_key][-1] += new_part
334: def __setitem__(self, name: str, value: str) -> None:
337: self._as_list[norm_name] = [value]
345: del self._as_list[norm_name]
941: parse_multipart_form_data(utf8(v), body, arguments, files)
949:def parse_multipart_form_data(3 more lines
1079:# _parseparam and _parse_header are copied and modified from python2.7's cgi.py
1087:def _parseparam(s: str) -> Generator[str, None, None]:
1114: parts = _parseparam(";" + line)$ sed -n 190,300p /repo/tornado/httputil.py; sed -n 1079,1160p /repo/tornado/httputil.py
output · 35 lines
self._dict = {} # type: typing.Dict[str, str]
self._as_list = {} # type: typing.Dict[str, typing.List[str]]
self._last_key = None # type: Optional[str]
if len(args) == 1 and len(kwargs) == 0 and isinstance(args[0], HTTPHeaders):
# Copy constructor
for k, v in args[0].get_all():
self.add(k, v)
else:
# Dict-style initialization
self.update(*args, **kwargs)
# new public methods23 more lines
def add(self, name: str, value: str, *, _chars_are_bytes: bool = True) -> None:
"""Adds a new value for the given key."""
if not _ABNF.field_name.fullmatch(name):
raise HTTPInputError("Invalid header name %r" % name)
if _chars_are_bytes:
if not _ABNF.field_value.fullmatch(to_unicode(value)):
# TODO: the fact we still support bytes here
…[6243 chars omitted]…
v is None:
out.append(k)
else:
# TODO: quote if necessary.
out.append(f"{k}={v}")
return "; ".join(out)
def encode_username_password(
username: Union[str, bytes], password: Union[str, bytes]
) -> bytes:
"""Encodes a username/password pair in the format used by HTTP auth.
The return value is a byte string in the form ``username:password``.$ sed -n 300,420p /repo/tornado/httputil.py
output · 25 lines
Raises `HTTPInputError` on malformed headers instead of a
mix of `KeyError`, and `ValueError`.
"""
# _chars_are_bytes is a hack. This method is used in two places, HTTP headers (in which
# non-ascii characters are to be interpreted as latin-1) and multipart/form-data (in which
# they are to be interpreted as utf-8). For historical reasons, this method handled this by
# expecting both callers to decode the headers to strings before parsing them. This wasn't a
# problem until we started doing stricter validation of the characters allowed in HTTP
# headers (using ABNF rules defined in terms of byte values), which inadvertently started
# disallowing non-latin1 characters in multipart/form-data filenames.
#13 more lines
# This method should have accepted bytes and a desired encoding, but this change
…[2845 chars omitted]…
rovided by a load balancer
in the ``X-Real-Ip`` or ``X-Forwarded-For`` header.
.. versionchanged:: 3.1
The list format of ``X-Forwarded-For`` is now supported.
.. attribute:: protocol
The protocol used, either "http" or "https". If ``HTTPServer.xheaders``
is set, will pass along the protocol used by a load balancer if
reported via an ``X-Scheme`` header.$ grep -n "_normalize_header" -A6 /repo/tornado/httputil.py | head -20; grep -rn "_dict\[" /repo/tornado/*.py | grep -v httputil
output · 24 lines
135:def _normalize_header(name: str) -> str:
136- """Map a header name to Http-Header-Case.
137-
138: >>> _normalize_header("coNtent-TYPE")
139- 'Content-Type'
140- """
141- return "-".join([w.capitalize() for w in name.split("-")])
142-
143-
144-class HTTPHeaders(StrMutableMapping):
--
215: norm_name = _normalize_header(name)12 more lines
216- self._last_key = norm_name 217- if norm_name in self: 218- self._dict[norm_name] = ( 219- native_str(self[norm_name]) + "," + native_str(value) 220- ) 221- self._as_list[norm_name].append(value) -- 227: norm_name = _normalize_header(name) /repo/tornado/gen.py:48: response3 = response_dict['response3'] /repo/tornado/gen.py:49: response4 = response_dict['response4'] /repo/tornado/web.py:3496: self._resource_dict[path] = kwargs /repo/tornado/web.py:3498: if self._resource_dict[path] != kwargs:
$ cd /repo && python - <<'PY'
output · 10 lines
Traceback (most recent call last):
File "<stdin>", line 4, in <module>
File "<frozen _collections_abc>", line 813, in __contains__
File "/repo/tornado/httputil.py", line 340, in __getitem__
return self._dict[_normalize_header(name)]
^^^^^^^^^^^^^^^^^^^^^^^
File "/repo/tornado/httputil.py", line 141, in _normalize_header
return "-".join([w.capitalize() for w in name.split("-")])
^^^^^^^^^^
AttributeError: 'int' object has no attribute 'split'$ cd /repo && python - <<'PY'
$ cd /repo && python - <<'PY'
output · 9 lines
False True True
[0.00974929501535371, 0.06486064498312771] 6.652854886530945
bar baz,qux ['bar baz', 'qux'] 1 ['X-Foo']
0
Traceback (most recent call last):
File "<stdin>", line 18, in <module>
File "/repo/tornado/httputil.py", line 989, in parse_multipart_form_data
raise HTTPInputError("Invalid multipart/form-data: no final boundary found")
tornado.httputil.HTTPInputError: Invalid multipart/form-data: no final boundary found$ cd /repo && python - <<'PY'
output · 18 lines
Traceback (most recent call last):
File "/repo/tornado/httputil.py", line 287, in parse_line
name, value = line.split(":", 1)
^^^^^^^^^^^
ValueError: not enough values to unpack (expected 2, got 1)
During handling of the above exception, another exception occurred:
Traceback (most recent call last):
File "<stdin>", line 7, in <module>
File "/repo/tornado/httputil.py", line 997, in parse_multipart_form_data
headers = HTTPHeaders.parse(part[:eoh].decode("utf-8"), _chars_are_bytes=False)6 more lines
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "/repo/tornado/httputil.py", line 332, in parse
h.parse_line(line, _chars_are_bytes=_chars_are_bytes)
File "/repo/tornado/httputil.py", line 289, in parse_line
raise HTTPInputError("no colon in header line")
tornado.httputil.HTTPInputError: no colon in header line$ cd /repo && sed -n 975,1010p tornado/httputil.py && python - <<'PY'
output · 30 lines
.. versionchanged:: 5.1
Now recognizes non-ASCII filenames in RFC 2231/5987
(``filename*=``) format.
"""
# The standard allows for the boundary to be quoted in the header,
# although it's rare (it happens at least for google app engine
# xmpp). I think we're also supposed to handle backslash-escapes
# here but I'll save that until we see a client that uses them
# in the wild.
if boundary.startswith(b'"') and boundary.endswith(b'"'):
boundary = boundary[1:-1]18 more lines
final_boundary_index = data.rfind(b"--" + boundary + b"--")
if final_boundary_index == -1:
raise HTTPInputError("Invalid multipart/form-data: no final boundary found")
parts = data[:final_boundary_index].split(b"--" + boundary + b"\r\n")
for part in parts:
if not part:
continue
eoh = part.find(b"\r\n\r\n")
if eoh ==
…[520 chars omitted]…
utError("multipart/form-data missing name")
name = disp_params["name"]
if disp_params.get("filename"):
ctype = headers.get("Content-Type", "application/unknown")
files.setdefault(name, []).append(
HTTPFile(
filename=disp_params["filename"], body=value, content_type=ctype
<generator object HTTPHeaders.get_all at 0x715a83f96b20>$ cd /repo && python - <<'PY'
output · 3 lines
1000 0.0002000710228458047 {} {'files': [{'filename': 'a.txt', 'body': b'content', 'content_type': 'application/unknown'}]}
10000 0.0007668659673072398 {} {'files': [{'filename': 'a.txt', 'body': b'content', 'content_type': 'application/unknown'}]}
ratio 3.8329686948132693$ cd /repo && python -m pytest tornado/httputil_test.py -q 2>&1 | tail -3 && python -m tornado.test.runtests 2>&1 | tail -3
output · 4 lines
no tests ran in 0.02s imports, py312 has its own check for test case returns, pycares module not present, pycurl module not present, running as root, tornado.speedups module not present, twisted module not present
$ cd /repo && ls tornado/test | head -30 && python -m pytest tornado/test/httputil_test.py -q 2>&1 | tail -3
output · 33 lines
__init__.py __main__.py __pycache__ asyncio_test.py auth_test.py autoreload_test.py circlerefs_test.py concurrent_test.py csv_translations curl_httpclient_test.py escape_test.py gen_test.py
21 more lines
gettext_translations http1connection_test.py httpclient_test.py httpserver_test.py httputil_test.py import_test.py ioloop_test.py iostream_test.py locale_test.py locks_test.py log_test.py netutil_test.py options_test.cfg options_test.py options_test_types.cfg options_test_types_str.cfg process_test.py queues_test.py =========================== short test summary info ============================ FAILED tornado/test/httputil_test.py::MultipartFormDataTest::test_special_filenames 1 failed, 51 passed, 17 subtests passed in 0.56s
$ cd /repo && python -m pytest tornado/test/httputil_test.py::MultipartFormDataTest::test_special_filenames -q 2>&1 | head -40 && git stash && python -m pytest tornado/test/httputil_test.py::MultipartFormDataTest::test_special_filenames -q 2>&1 | tail -2; git stash pop
output · 36 lines
F [100%]
=================================== FAILURES ===================================
_________________ MultipartFormDataTest.test_special_filenames _________________
self = <tornado.test.httputil_test.MultipartFormDataTest testMethod=test_special_filenames>
def test_special_filenames(self):
filenames = [
"a;b.txt",
'a"b.txt',
'a";b.txt',
'a;"b.txt',24 more lines
'a";";.txt',
'a\\"b.txt',
"a\\b.txt",
]
for filename in filenames:
logging.debug("trying filename %r", filename)
str_data = """\
--1234
Content-Disposition: form-data; name="files"; filename="%s"
Foo
--1234--""" % filename.replace(
"\\",
…[786 chars omitted]…
[100%]
1 passed in 0.46s
On branch master
Changes not staged for commit:
(use "git add <file>..." to update what will be committed)
(use "git restore <file>..." to discard changes in working directory)
modified: tornado/httputil.py
no changes added to commit (use "git add" and/or "git commit -a")
Dropped refs/stash@{0} (133d3cf44405369c2d4533c197dde0d7eb0ece8b)$ cd /repo && python - <<'PY'
output · 2 lines
.................................................... [100%] 52 passed, 17 subtests passed in 0.72s
$ cd /repo && python - <<'PY'
output · 5 lines
('form-data', {'foo': 'b\\a"r', 'file': 'Täst'})
ratio 3.3533977062537828 {'files': [{'filename': 'a.txt', 'body': b'content', 'content_type': 'application/unknown'}]}
ratio 8.377151776346684 bar,bar,bar,bar,bar,
not present, pycurl module not present, running as root,
tornado.speedups module not present, twisted module not present$ cd /repo && python -m tornado.test.runtests 2>&1 | grep -E "PASS|FAIL|pass|fail|Ran" | tail -3; git diff --stat
output · 3 lines
Ran 1228 tests in 22.145s tornado/httputil.py | 45 +++++++++++++++++++++++++++++++++++++-------- 1 file changed, 37 insertions(+), 8 deletions(-)
$ cd /repo && python -m tornado.test.runtests 2>&1 | grep -cE "^(FAIL|ERROR)" ; git diff
output · 32 lines
0
diff --git a/tornado/httputil.py b/tornado/httputil.py
index a418b9f..40fe067 100644
--- a/tornado/httputil.py
+++ b/tornado/httputil.py
@@ -190,6 +190,10 @@ class HTTPHeaders(StrMutableMapping):
self._dict = {} # type: typing.Dict[str, str]
self._as_list = {} # type: typing.Dict[str, typing.List[str]]
self._last_key = None # type: Optional[str]
+ # Names whose combined value in self._dict is out of date and must be
+ # recomputed from self._as_list before it is next read. This keeps
+ # repeated calls to add() linear instead of quadratic.20 more lines
+ self._dirty = set() # type: typing.Set[str]
if len(args) == 1 and len(kwargs) == 0 and isinstance(args[0], HTTPHeaders):
# Copy constructor
for k, v in args[0].get_all():
@@ -215,10 +219,8 @@ class HTTPHeaders(StrMutableMapping):
norm_name = _n
…[2486 chars omitted]…
if c == "\\" and i + 1 < n and s[i + 1] == '"':
+ # An escaped quote does not affect quote parity.
+ i += 2
+ continue
+ if c == '"':
+ quotes += 1
+ elif c == ";" and quotes % 2 == 0:
+ end = i
+ break
+ i += 1
if end < 0:
end = len(s)
f = s[:end]All checks pass. Summary of the fix in `tornado/httputil.py`:
$ done: {"notes": "Made HTTPHeaders.add linear via lazy comma-join caching (_dirty set), added __contains__ returning False for non-string keys, and rewrote _parseparam as a single linear scan with correct escaped-quote parity. Reproducer ratios ~8x and ~3.4x (limit 20); full test suite passes."}output · 1 lines
submitted.