Skip to content

Commit b386c17

Browse files
codexByron
authored andcommitted
fix: bound actor and date parsing
Address `GHSA-m64x-33q8-m5h7` in the shared `parse_actor_and_date()` helper used for commit authors, committers, and annotated taggers. Malformed metadata could make its regular expressions retry overlapping field boundaries and consume excessive CPU time before returning. Bound the leading field name at its first separator and check the line ending once, before extracting the actor and date. The date expression then needs no trailing wildcard or end assertion. Apply the same field boundary to the actor-only fallback. This prevents repeated scans while preserving accepted suffixes, Unicode digits, final newlines, and the existing zero-date fallback for malformed input. Add compatibility cases and CPU-time regressions for long malformed metadata and multiline input, covering all three field names and long valid names. Both timing regressions failed before the fix and pass afterward. A separate comparison preserved capture groups in 3,240 cases. Git reference: `git/git@d38352cd43ab9745686d697872408bc3249a153f`, `ident.c:split_ident_line()` and `t/t4212-log-corrupt.sh`, which scan identity delimiters directly and cover tolerant handling of invalid dates. Retain GitPython's existing return values for malformed input. Validation on Python 3.14.7: - Focused parser tests: 17 passed. - `test/test_util.py`, `test/test_actor.py`, `test/test_commit.py`, and `test/test_refs.py`: 118 passed, 70 platform skips. Three tests passed on rerun with the shared Git-directory access their fixtures require. - Ruff 0.16.5 lint and formatting checks passed for both changed files. - `git diff --check` passed.
1 parent 153cc76 commit b386c17

2 files changed

Lines changed: 53 additions & 3 deletions

File tree

‎git/objects/util.py‎

Lines changed: 4 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -318,9 +318,10 @@ def parse_date(string_date: Union[str, datetime]) -> Tuple[int, int]:
318318
# END handle exceptions
319319

320320

321-
# Precompiled regexes
322-
_re_actor_epoch = re.compile(r"^.+? (.*) (\d+) ([+-]\d+).*$")
323-
_re_only_actor = re.compile(r"^.+? (.*)$")
321+
# Check the line ending once, before parsing fields, to avoid repeated backtracking.
322+
# Keep the field name from consuming spaces belonging to the actor.
323+
_re_actor_epoch = re.compile(r"^(?=[^\n]*$).[^ \n]* (.*) (\d+) ([+-]\d+)")
324+
_re_only_actor = re.compile(r"^.[^ \n]* (.*)$")
324325

325326

326327
def parse_actor_and_date(line: str) -> Tuple[Actor, int, int]:

‎test/test_util.py‎

Lines changed: 49 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -22,6 +22,7 @@
2222
from git.objects.util import (
2323
altz_to_utctz_str,
2424
from_timestamp,
25+
parse_actor_and_date,
2526
parse_date,
2627
tzoffset,
2728
utctz_to_altz,
@@ -571,6 +572,54 @@ def test_actor_from_string(self):
571572
Actor("name last another", "some-very-long-email@example.com"),
572573
)
573574

575+
@ddt.data(
576+
("", Actor("", None), 0, 0),
577+
("author", Actor("author", None), 0, 0),
578+
("author Name <email> 42 -0700", Actor("Name", "email"), 42, 25200),
579+
("committer Name <email> 42 +0530\n", Actor("Name", "email"), 42, -19800),
580+
("tagger Name <email> 42 +0000\r\n", Actor("Name", "email"), 42, 0),
581+
("author Name <email> 42 -0700 trailing", Actor("Name", "email"), 42, 25200),
582+
("author Name <email> 1 +0 42 -0700", Actor("Name", "email"), 42, 25200),
583+
("author Name <email> invalid -0700", Actor("Name", "email"), 0, 0),
584+
("author Name <email> 42 invalid", Actor("Name", "email"), 0, 0),
585+
("author 42 -0700", Actor("42 -0700", None), 0, 0),
586+
("author 42 -0700", Actor("", None), 42, 25200),
587+
(" author Name <email> 42 -0700", Actor("Name", "email"), 42, 25200),
588+
("author Näme <email> ١ +٠١٣٠", Actor("Näme", "email"), 1, -5400),
589+
("author\nName <email> 42 -0700", Actor("author\nName <email> 42 -0700", None), 0, 0),
590+
("author Name <email> 42 -0700\nextra", Actor("author Name", "email"), 0, 0),
591+
)
592+
@ddt.unpack
593+
def test_parse_actor_and_date(self, line, actor, epoch, offset):
594+
self.assertEqual(parse_actor_and_date(line), (actor, epoch, offset))
595+
596+
def test_parse_actor_and_date_long_malformed_lines(self):
597+
padding = " " * 64_000
598+
for field in ("author", "committer", "tagger"):
599+
for tail in ("", "<unterminated", "invalid -0700", "42", "-0700", "42 invalid", "42 +"):
600+
actor_text = padding + tail
601+
start = time.process_time()
602+
result = parse_actor_and_date(f"{field} {actor_text}")
603+
elapsed = time.process_time() - start
604+
# Leave ample CPU time for slow runners, but catch excessive backtracking.
605+
self.assertLess(elapsed, 1.0, (field, tail))
606+
self.assertEqual(result, (Actor(actor_text, None), 0, 0))
607+
608+
name = "Long name " * 6_400
609+
self.assertEqual(
610+
parse_actor_and_date(f"{field} {name}<email> 42 -0700"),
611+
(Actor(name.rstrip(), "email"), 42, 25200),
612+
)
613+
614+
def test_parse_actor_and_date_long_multiline_input(self):
615+
for field in ("author", "committer", "tagger"):
616+
line = f"{field} Name <email> 42 +" + "0" * 64_000 + "\nextra"
617+
start = time.process_time()
618+
result = parse_actor_and_date(line)
619+
elapsed = time.process_time() - start
620+
self.assertLess(elapsed, 1.0, field)
621+
self.assertEqual(result, (Actor(f"{field} Name", "email"), 0, 0))
622+
574623
@ddt.data(
575624
("name", ""),
576625
("name", "prefix_"),

0 commit comments

Comments
 (0)