Make pre-commit scripts more OS-agnostic (#6724)

# Description of Changes
Fix #6723
This commit is contained in:
James Brunton
2026-06-23 08:42:18 +00:00
committed by GitHub
parent 1816bad1ba
commit f2b65f4a77
5 changed files with 145 additions and 105 deletions
+20 -6
View File
@@ -3,11 +3,14 @@
Replaces the end-of-file-fixer / trailing-whitespace pre-commit hooks, which
have no read-only mode. Run via `task pre-commit` (check) and `task
pre-commit:fix`; Task selects the files (with `git ls-files`) and passes them
as arguments.
pre-commit:fix`.
python scripts/whitespace.py <files>... # check: report, exit 1 if any need fixing
python scripts/whitespace.py --fix <files>... # fix: rewrite in place
Takes git pathspecs (not a file list) and runs `git ls-files` itself, so the
matched files never hit the command line - on Windows that list can be ~66KB
and exceed the ~32KB CreateProcess argv limit.
python whitespace.py <pathspec>... # check: report, exit 1 if any need fixing
python whitespace.py --fix <pathspec>... # fix: rewrite in place
Operates on bytes and only ever touches trailing spaces/tabs and the final
newline, so it never mangles content or line endings. Binary files (those with
@@ -16,10 +19,21 @@ a NUL byte) are skipped.
from __future__ import annotations
import subprocess
import sys
from pathlib import Path
def tracked_files(pathspecs: list[str]) -> list[str]:
result = subprocess.run(
["git", "ls-files", "-z", *pathspecs],
check=True,
capture_output=True,
text=True,
)
return [path for path in result.stdout.split("\0") if path]
def normalise(data: bytes) -> bytes:
# Strip trailing spaces/tabs from each line (leave \r so CRLF survives).
lines = [line.rstrip(b" \t") for line in data.split(b"\n")]
@@ -32,10 +46,10 @@ def normalise(data: bytes) -> bytes:
def main() -> int:
args = sys.argv[1:]
fix = "--fix" in args
paths = [a for a in args if a != "--fix"]
pathspecs = [a for a in args if a != "--fix"]
offenders: list[str] = []
for path in paths:
for path in tracked_files(pathspecs):
data = Path(path).read_bytes()
if b"\0" in data:
continue