davideisinger.com

My personal website
Log | Files | Refs | README

check-staged-spelling (3164B)


      1 #!/usr/bin/env python3
      2 """Collect unknown words from staged Markdown and require dictionary review."""
      3 
      4 from pathlib import Path
      5 import os
      6 import subprocess
      7 import sys
      8 import unicodedata
      9 
     10 
     11 def git(*args):
     12     return subprocess.check_output(["git", *args])
     13 
     14 
     15 def main():
     16     os.chdir(os.fsdecode(git("rev-parse", "--show-toplevel")).rstrip("\n"))
     17     paths = git("diff", "--cached", "--name-only", "--diff-filter=ACMR", "-z")
     18     markdown = [os.fsdecode(path) for path in paths.split(b"\0")
     19                 if path.lower().endswith((b".md", b".markdown"))]
     20     if not markdown:
     21         return 0
     22 
     23     dictionary = Path(".dictionary")
     24     original = dictionary.read_bytes() if dictionary.exists() else b""
     25     words = set(original.decode("utf-8").splitlines())
     26     for path in markdown:
     27         print(f"Spellchecking staged {path}", file=sys.stderr)
     28         result = subprocess.run(
     29             ["npx", "--yes", "cspell", "--words-only", "--unique", "--no-progress",
     30              "--no-summary", "--config", ".cspell.json",
     31              f"stdin://{path}"],
     32             input=git("show", f":{path}"), stdout=subprocess.PIPE, stderr=subprocess.PIPE,
     33         )
     34         # CSpell uses exit 1 for both spelling issues and configuration errors.
     35         # Only accept that status when it produced words and no diagnostics.
     36         if (result.returncode not in (0, 1) or result.stderr
     37                 or (result.returncode == 1 and not result.stdout.strip())):
     38             sys.stderr.buffer.write(result.stderr)
     39             print("Spellcheck failed; .dictionary was not updated.", file=sys.stderr)
     40             return 1
     41         reported = result.stdout.decode("utf-8").splitlines()
     42         if any(not word or not any(char.isalpha() for char in word)
     43                or any(not (char.isalnum() or char in "'’-_"
     44                           or unicodedata.category(char).startswith("M"))
     45                       for char in word)
     46                for word in reported):
     47             print("Spellcheck returned unexpected output; .dictionary was not updated:\n"
     48                   + result.stdout.decode("utf-8"), file=sys.stderr)
     49             return 1
     50         words.update(reported)
     51 
     52     # Match the user's shell sort, including its locale-specific collation.
     53     updated = subprocess.check_output(
     54         ["sort", "-u"],
     55         input="".join(f"{word}\n" for word in words if word).encode("utf-8"),
     56     )
     57     if updated != original:
     58         dictionary.write_bytes(updated)
     59 
     60     staged = subprocess.run(["git", "show", ":.dictionary"],
     61                             stdout=subprocess.PIPE, stderr=subprocess.DEVNULL)
     62     if updated != original or staged.returncode or staged.stdout != updated:
     63         print("Commit blocked: review .dictionary (git diff -- .dictionary).\n"
     64               "Remove actual typos from it and fix them in your Markdown, then\n"
     65               "stage the corrected files and accepted dictionary words and retry.",
     66               file=sys.stderr)
     67         return 1
     68     return 0
     69 
     70 
     71 if __name__ == "__main__":
     72     try:
     73         sys.exit(main())
     74     except (OSError, UnicodeError, subprocess.CalledProcessError) as error:
     75         sys.exit(f"Spellcheck failed: {error}")