diff --git a/doc/source/changes.rst b/doc/source/changes.rst index 5b809bd9e..beb14f6c2 100644 --- a/doc/source/changes.rst +++ b/doc/source/changes.rst @@ -8,6 +8,7 @@ Changelog Security fixes for * https://github.com/gitpython-developers/GitPython/security/advisories/GHSA-w8jc-g24h-crhw +* https://github.com/gitpython-developers/GitPython/security/advisories/GHSA-m64x-33q8-m5h7 3.2.0 ===== diff --git a/git/objects/util.py b/git/objects/util.py index a68d701f5..64cdf71fd 100644 --- a/git/objects/util.py +++ b/git/objects/util.py @@ -318,9 +318,10 @@ def parse_date(string_date: Union[str, datetime]) -> Tuple[int, int]: # END handle exceptions -# Precompiled regexes -_re_actor_epoch = re.compile(r"^.+? (.*) (\d+) ([+-]\d+).*$") -_re_only_actor = re.compile(r"^.+? (.*)$") +# Check the line ending once, before parsing fields, to avoid repeated backtracking. +# Keep the field name from consuming spaces belonging to the actor. +_re_actor_epoch = re.compile(r"^(?=[^\n]*$).[^ \n]* (.*) (\d+) ([+-]\d+)") +_re_only_actor = re.compile(r"^.[^ \n]* (.*)$") def parse_actor_and_date(line: str) -> Tuple[Actor, int, int]: diff --git a/test/test_util.py b/test/test_util.py index cf4299d5a..c6e68cf51 100644 --- a/test/test_util.py +++ b/test/test_util.py @@ -22,6 +22,7 @@ from git.objects.util import ( altz_to_utctz_str, from_timestamp, + parse_actor_and_date, parse_date, tzoffset, utctz_to_altz, @@ -571,6 +572,54 @@ def test_actor_from_string(self): Actor("name last another", "some-very-long-email@example.com"), ) + @ddt.data( + ("", Actor("", None), 0, 0), + ("author", Actor("author", None), 0, 0), + ("author Name 42 -0700", Actor("Name", "email"), 42, 25200), + ("committer Name 42 +0530\n", Actor("Name", "email"), 42, -19800), + ("tagger Name 42 +0000\r\n", Actor("Name", "email"), 42, 0), + ("author Name 42 -0700 trailing", Actor("Name", "email"), 42, 25200), + ("author Name 1 +0 42 -0700", Actor("Name", "email"), 42, 25200), + ("author Name invalid -0700", Actor("Name", "email"), 0, 0), + ("author Name 42 invalid", Actor("Name", "email"), 0, 0), + ("author 42 -0700", Actor("42 -0700", None), 0, 0), + ("author 42 -0700", Actor("", None), 42, 25200), + (" author Name 42 -0700", Actor("Name", "email"), 42, 25200), + ("author Näme ١ +٠١٣٠", Actor("Näme", "email"), 1, -5400), + ("author\nName 42 -0700", Actor("author\nName 42 -0700", None), 0, 0), + ("author Name 42 -0700\nextra", Actor("author Name", "email"), 0, 0), + ) + @ddt.unpack + def test_parse_actor_and_date(self, line, actor, epoch, offset): + self.assertEqual(parse_actor_and_date(line), (actor, epoch, offset)) + + def test_parse_actor_and_date_long_malformed_lines(self): + padding = " " * 64_000 + for field in ("author", "committer", "tagger"): + for tail in ("", " 42 -0700"), + (Actor(name.rstrip(), "email"), 42, 25200), + ) + + def test_parse_actor_and_date_long_multiline_input(self): + for field in ("author", "committer", "tagger"): + line = f"{field} Name 42 +" + "0" * 64_000 + "\nextra" + start = time.process_time() + result = parse_actor_and_date(line) + elapsed = time.process_time() - start + self.assertLess(elapsed, 1.0, field) + self.assertEqual(result, (Actor(f"{field} Name", "email"), 0, 0)) + @ddt.data( ("name", ""), ("name", "prefix_"),