Skip to content

Commit

Permalink
Replace quadratic algo in LineDecoder
Browse files Browse the repository at this point in the history
Leading to enormous speedups when doing things such as
Response(...).iter_lines() as described on issue #2422
  • Loading branch information
giannitedesco committed Nov 21, 2022
1 parent 9e97d7d commit ee687ae
Show file tree
Hide file tree
Showing 2 changed files with 24 additions and 57 deletions.
61 changes: 14 additions & 47 deletions httpx/_decoders.py
Expand Up @@ -266,57 +266,24 @@ def __init__(self) -> None:
self.buffer = ""

def decode(self, text: str) -> typing.List[str]:
lines = []

if text and self.buffer and self.buffer[-1] == "\r":
if text.startswith("\n"):
# Handle the case where we have an "\r\n" split across
# our previous input, and our new chunk.
lines.append(self.buffer[:-1] + "\n")
self.buffer = ""
text = text[1:]
else:
# Handle the case where we have "\r" at the end of our
# previous input.
lines.append(self.buffer[:-1] + "\n")
self.buffer = ""

while text:
num_chars = len(text)
for idx in range(num_chars):
char = text[idx]
next_char = None if idx + 1 == num_chars else text[idx + 1]
if char == "\n":
lines.append(self.buffer + text[: idx + 1])
self.buffer = ""
text = text[idx + 1 :]
break
elif char == "\r" and next_char == "\n":
lines.append(self.buffer + text[:idx] + "\n")
self.buffer = ""
text = text[idx + 2 :]
break
elif char == "\r" and next_char is not None:
lines.append(self.buffer + text[:idx] + "\n")
self.buffer = ""
text = text[idx + 1 :]
break
elif next_char is None:
self.buffer += text
text = ""
break
if self.buffer:
text = self.buffer + text

if not text:
return []

lines = text.splitlines(True)
if text.endswith("\n"):
self.buffer = ""
else:
remainder = lines.pop()
self.buffer = remainder

return lines

def flush(self) -> typing.List[str]:
if self.buffer.endswith("\r"):
# Handle the case where we had a trailing '\r', which could have
# been a '\r\n' pair.
lines = [self.buffer[:-1] + "\n"]
elif self.buffer:
lines = [self.buffer]
else:
lines = []
# this handles stripping any trailing "\r"
lines = self.buffer.splitlines(True)
self.buffer = ""
return lines

Expand Down
20 changes: 10 additions & 10 deletions tests/test_decoders.py
Expand Up @@ -257,48 +257,48 @@ def test_line_decoder_nl():
def test_line_decoder_cr():
decoder = LineDecoder()
assert decoder.decode("") == []
assert decoder.decode("a\r\rb\rc") == ["a\n", "\n", "b\n"]
assert decoder.decode("a\r\rb\rc") == ["a\r", "\r", "b\r"]
assert decoder.flush() == ["c"]

decoder = LineDecoder()
assert decoder.decode("") == []
assert decoder.decode("a\r\rb\rc\r") == ["a\n", "\n", "b\n"]
assert decoder.flush() == ["c\n"]
assert decoder.decode("a\r\rb\rc\r") == ["a\r", "\r", "b\r"]
assert decoder.flush() == ["c\r"]

# Issue #1033
decoder = LineDecoder()
assert decoder.decode("") == []
assert decoder.decode("12345\r") == []
assert decoder.decode("foo ") == ["12345\n"]
assert decoder.decode("foo ") == ["12345\r"]
assert decoder.decode("bar ") == []
assert decoder.decode("baz\r") == []
assert decoder.flush() == ["foo bar baz\n"]
assert decoder.flush() == ["foo bar baz\r"]


def test_line_decoder_crnl():
decoder = LineDecoder()
assert decoder.decode("") == []
assert decoder.decode("a\r\n\r\nb\r\nc") == ["a\n", "\n", "b\n"]
assert decoder.decode("a\r\n\r\nb\r\nc") == ["a\r\n", "\r\n", "b\r\n"]
assert decoder.flush() == ["c"]

decoder = LineDecoder()
assert decoder.decode("") == []
assert decoder.decode("a\r\n\r\nb\r\nc\r\n") == ["a\n", "\n", "b\n", "c\n"]
assert decoder.decode("a\r\n\r\nb\r\nc\r\n") == ["a\r\n", "\r\n", "b\r\n", "c\r\n"]
assert decoder.flush() == []

decoder = LineDecoder()
assert decoder.decode("") == []
assert decoder.decode("a\r") == []
assert decoder.decode("\n\r\nb\r\nc") == ["a\n", "\n", "b\n"]
assert decoder.decode("\n\r\nb\r\nc") == ["a\r\n", "\r\n", "b\r\n"]
assert decoder.flush() == ["c"]

# Issue #1033
decoder = LineDecoder()
assert decoder.decode("") == []
assert decoder.decode("12345\r\n") == ["12345\n"]
assert decoder.decode("12345\r\n") == ["12345\r\n"]
assert decoder.decode("foo ") == []
assert decoder.decode("bar ") == []
assert decoder.decode("baz\r\n") == ["foo bar baz\n"]
assert decoder.decode("baz\r\n") == ["foo bar baz\r\n"]
assert decoder.flush() == []


Expand Down

0 comments on commit ee687ae

Please sign in to comment.