From 8290c98d35465a38090cdcc1b17be69967092a15 Mon Sep 17 00:00:00 2001 From: Byron Date: Sun, 27 Sep 2026 19:31:47 +0200 Subject: [PATCH] fix: bound actor and date parsing Address `GHSA-m64x-33q8-m5h7` in the shared `parse_actor_and_date()` helper used for commit authors, committers, and annotated taggers. Malformed metadata could make its regular expressions retry overlapping field boundaries and consume excessive CPU time before returning. Bound the leading field name at its first separator and check the line ending once, before extracting the actor and date. The date expression then needs no trailing wildcard or end assertion. Apply the same field boundary to the actor-only fallback. This prevents repeated scans while preserving accepted suffixes, Unicode digits, final newlines, and the existing zero-date fallback for malformed input. Add compatibility cases and CPU-time regressions for long malformed metadata and multiline input, covering all three field names and long valid names. Both timing regressions failed before the fix and pass afterward. A separate comparison preserved capture groups in 3,240 cases. Git reference: `git/git@d38352cd43ab9745686d697872408bc3249a153f`, `ident.c:split_ident_line()` and `t/t4212-log-corrupt.sh`, which scan identity delimiters directly and cover tolerant handling of invalid dates. Retain GitPython's existing return values for malformed input. Assisted-by: GPT 6.0 Co-authored-by: GPT 6.0 --- doc/source/changes.rst | 1 + git/objects/util.py | 7 +++--- test/test_util.py | 49 ++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 54 insertions(+), 3 deletions(-) diff --git a/doc/source/changes.rst b/doc/source/changes.rst index 5b809bd9e..beb14f6c2 100644 --- a/doc/source/changes.rst +++ b/doc/source/changes.rst @@ -8,6 +8,7 @@ Changelog Security fixes for * https://github.com/gitpython-developers/GitPython/security/advisories/GHSA-w8jc-g24h-crhw +* https://github.com/gitpython-developers/GitPython/security/advisories/GHSA-m64x-33q8-m5h7 3.2.0 ===== diff --git a/git/objects/util.py b/git/objects/util.py index a68d701f5..64cdf71fd 100644 --- a/git/objects/util.py +++ b/git/objects/util.py @@ -318,9 +318,10 @@ def parse_date(string_date: Union[str, datetime]) -> Tuple[int, int]: # END handle exceptions -# Precompiled regexes -_re_actor_epoch = re.compile(r"^.+? (.*) (\d+) ([+-]\d+).*$") -_re_only_actor = re.compile(r"^.+? (.*)$") +# Check the line ending once, before parsing fields, to avoid repeated backtracking. +# Keep the field name from consuming spaces belonging to the actor. +_re_actor_epoch = re.compile(r"^(?=[^\n]*$).[^ \n]* (.*) (\d+) ([+-]\d+)") +_re_only_actor = re.compile(r"^.[^ \n]* (.*)$") def parse_actor_and_date(line: str) -> Tuple[Actor, int, int]: diff --git a/test/test_util.py b/test/test_util.py index cf4299d5a..c6e68cf51 100644 --- a/test/test_util.py +++ b/test/test_util.py @@ -22,6 +22,7 @@ from git.objects.util import ( altz_to_utctz_str, from_timestamp, + parse_actor_and_date, parse_date, tzoffset, utctz_to_altz, @@ -571,6 +572,54 @@ def test_actor_from_string(self): Actor("name last another", "some-very-long-email@example.com"), ) + @ddt.data( + ("", Actor("", None), 0, 0), + ("author", Actor("author", None), 0, 0), + ("author Name 42 -0700", Actor("Name", "email"), 42, 25200), + ("committer Name 42 +0530\n", Actor("Name", "email"), 42, -19800), + ("tagger Name 42 +0000\r\n", Actor("Name", "email"), 42, 0), + ("author Name 42 -0700 trailing", Actor("Name", "email"), 42, 25200), + ("author Name 1 +0 42 -0700", Actor("Name", "email"), 42, 25200), + ("author Name invalid -0700", Actor("Name", "email"), 0, 0), + ("author Name 42 invalid", Actor("Name", "email"), 0, 0), + ("author 42 -0700", Actor("42 -0700", None), 0, 0), + ("author 42 -0700", Actor("", None), 42, 25200), + (" author Name 42 -0700", Actor("Name", "email"), 42, 25200), + ("author Näme ١ +٠١٣٠", Actor("Näme", "email"), 1, -5400), + ("author\nName 42 -0700", Actor("author\nName 42 -0700", None), 0, 0), + ("author Name 42 -0700\nextra", Actor("author Name", "email"), 0, 0), + ) + @ddt.unpack + def test_parse_actor_and_date(self, line, actor, epoch, offset): + self.assertEqual(parse_actor_and_date(line), (actor, epoch, offset)) + + def test_parse_actor_and_date_long_malformed_lines(self): + padding = " " * 64_000 + for field in ("author", "committer", "tagger"): + for tail in ("", " 42 -0700"), + (Actor(name.rstrip(), "email"), 42, 25200), + ) + + def test_parse_actor_and_date_long_multiline_input(self): + for field in ("author", "committer", "tagger"): + line = f"{field} Name 42 +" + "0" * 64_000 + "\nextra" + start = time.process_time() + result = parse_actor_and_date(line) + elapsed = time.process_time() - start + self.assertLess(elapsed, 1.0, field) + self.assertEqual(result, (Actor(f"{field} Name", "email"), 0, 0)) + @ddt.data( ("name", ""), ("name", "prefix_"),