Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions .claude/hooks/reflow-check.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
#!/usr/bin/env bash
# Advisory check, run before an agent commits: flag docstring and comment lines
# that break mid-phrase, looking only at the lines being added.
# Never blocks -- the check is a heuristic, and the agent judges each hit.
set -uo pipefail
cd "${CLAUDE_PROJECT_DIR:-.}" || exit 0
command -v python3 >/dev/null 2>&1 || exit 0
python3 .claude/hooks/reflow_check.py --staged 2>/dev/null || true
exit 0
135 changes: 135 additions & 0 deletions .claude/hooks/reflow_check.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,135 @@
"""Flag docstring and comment lines that break in the middle of a phrase.

The repo convention (see .github/instructions/docstrings.instructions.md) is that
each physical line of a docstring or comment ends at punctuation,
so that review comments and text search stay stable.
The mechanical form of that rule:
inside a multi-line docstring or comment block, every line but the last ends in punctuation.

This is a heuristic, and deliberately an advisory one --
docstrings here also hold OpenAPI YAML, shell commands and bullet lists,
none of which are prose and none of which end in punctuation.
The agent reading the output is expected to judge each hit rather than obey it.

Usage:
python reflow_check.py <file>... # whole files
python reflow_check.py --staged # only lines added in the git index
"""

from __future__ import annotations

import ast
import re
import subprocess
import sys
from pathlib import Path

PUNCTUATION = (".", ",", ";", ":", "!", "?", ")", "]", "}", "-", "—", "–")

#: Lines that are not prose, and so are not expected to end in punctuation.
NOT_PROSE = re.compile(
r"""^(
\s*[-*+]\s # bullet list item
| \s*\.\.\s # RST directive
| \s*>>> # doctest
| \s*\$ # shell prompt
| \s*\w[\w\s-]*:\s*\S # "key: value", i.e. embedded YAML
| \s*[\[{] # start of an embedded JSON/dict literal
)""",
re.VERBOSE,
)


def _offending_lines(block: list[str], first_line: int) -> list[tuple[int, str]]:
"""Lines of one block that end mid-phrase, ignoring the last line of the block."""
found = []
body = [(i, ln) for i, ln in enumerate(block) if ln.strip()]
for i, line in body[:-1]:
text = re.sub(r"^\s*#:?\s?", "", line).strip().strip('"').strip("'").strip()
if not text or text.endswith("\\") or NOT_PROSE.match(line):
continue
if not text.endswith(PUNCTUATION):
found.append((first_line + i, line.strip()[:110]))
return found


def check(path: Path) -> list[tuple[int, str]]:
"""Every mid-phrase line break in one Python file."""
try:
source = path.read_text()
tree = ast.parse(source)
except (SyntaxError, UnicodeDecodeError, OSError):
return []

found: list[tuple[int, str]] = []
for node in ast.walk(tree):
if isinstance(
node, (ast.Module, ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)
):
docstring = ast.get_docstring(node, clean=False)
if docstring and "\n" in docstring:
first = node.body[0].lineno if node.body else 1
found += _offending_lines(docstring.split("\n"), first)

block: list[str] = []
start = 0
for number, line in enumerate(source.split("\n"), start=1):
if line.strip().startswith("#"):
start = start or number
block.append(line)
continue
if len(block) > 1:
found += _offending_lines(block, start)
block, start = [], 0
if len(block) > 1:
found += _offending_lines(block, start)
return found


def staged_lines() -> dict[str, set[int]]:
"""Line numbers added to each staged Python file."""
diff = subprocess.run(
["git", "diff", "--cached", "-U0", "--", "*.py"],
capture_output=True,
text=True,
).stdout
added: dict[str, set[int]] = {}
current = None
for line in diff.split("\n"):
if line.startswith("+++ b/"):
current = line[6:]
added.setdefault(current, set())
header = re.match(r"^@@ -\S+ \+(\d+)(?:,(\d+))?", line)
if header and current:
first = int(header.group(1))
added[current].update(range(first, first + int(header.group(2) or 1)))
return added


def main() -> int:
if "--staged" in sys.argv:
targets = staged_lines()
else:
targets = {argument: None for argument in sys.argv[1:]}

hits = []
for name, wanted in targets.items():
path = Path(name)
if not path.exists():
continue
for number, text in sorted(set(check(path))):
if wanted is None or number in wanted:
hits.append(f"{name}:{number}: {text}")

if hits:
print("Possible mid-phrase line breaks in docstrings or comments:")
print("\n".join(f" {hit}" for hit in hits))
print(
"\nEach line above should end at punctuation, or be reflowed so it does."
"\nIgnore any that are not prose (embedded YAML, shell commands, list items)."
)
return 0 # advisory: never blocks the commit


if __name__ == "__main__":
sys.exit(main())
8 changes: 7 additions & 1 deletion .claude/settings.json
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,12 @@
{
"matcher": "Bash",
"hooks": [
{
"type": "command",
"command": "bash \"$CLAUDE_PROJECT_DIR/.claude/hooks/reflow-check.sh\"",
"if": "Bash(git commit:*)",
"timeout": 30
},
{
"type": "command",
"command": "bash \"$CLAUDE_PROJECT_DIR/.claude/hooks/pre-commit-check.sh\"",
Expand All @@ -56,4 +62,4 @@
}
]
}
}
}
Loading