#!/usr/bin/env python3
"""Check a file or text for the habits in a Dead Giveaway rules file. Standard library only.

  python check.py draft.md                     # uses rules.json next to this script
  python check.py page.html --rules claude.json
  cat reply.txt | python check.py -            # read stdin
Prints one line per hit (line number, habit, what matched, link to the evidence) and exits 1 when anything is found.
Only habits with a mechanical check are tested. Rhythm, padding and similar habits need a human read; the rules text covers them.
"""
import argparse, json, re, sys
from pathlib import Path

QUOTED = re.compile(r'"[^"\n]*"|“[^”]*”|`[^`\n]*`')
def visible_text(markup):
    """Text a reader sees on an HTML page: scripts and styles removed, tags dropped, one block per line."""
    markup = re.sub(r"(?is)<(script|style)\b.*?</\1>", " ", markup)
    markup = re.sub(r"(?i)<br\s*/?>|</(p|div|li|h[1-6]|section|tr|td|th|blockquote|figcaption|button|a)>", "\n", markup)
    text = re.sub(r"<[^>]+>", " ", markup)
    text = text.replace("&nbsp;", " ").replace("&amp;", "&").replace("&mdash;", "\u2014").replace("&ndash;", "\u2013")
    return "\n".join(re.sub(r"[ \t]+", " ", l).strip() for l in text.splitlines() if l.strip())


TARGETS = {"COMMIT_EDITMSG": "commit", ".md": "text", ".txt": "text", ".rst": "text", ".html": "html", ".htm": "html", ".css": "css", ".scss": "css"}


def kinds_for(name):
    ext = Path(name).suffix.lower()
    t = TARGETS.get(Path(name).name) or TARGETS.get(ext)
    if t == "html":
        return {"html", "css", "text"}
    if t:
        return {t}
    return {"code", "text"} if ext else {"text"}


def check(text, rules, kinds):
    hits = []
    source = text
    if "html" in kinds:               # prose rules read what a visitor sees, markup rules read the source
        text = visible_text(source)
    lines = text.splitlines() or [text]
    src_lines = source.splitlines() or [source]
    for h in rules["habits"]:
        for c in h.get("checks", []):
            if c["target"] not in kinds:
                continue
            rx = re.compile(c["regex"], re.I if "i" in c.get("flags", "") else 0)
            found = []
            if c["target"] == "commit":      # commit checks look at the whole message
                found = [(1, m.group(0)) for m in rx.finditer(text)]
            for no, line in enumerate((lines if c["target"] == "text" else src_lines) if c["target"] != "commit" else [], 1):
                probe = QUOTED.sub("Q", line) if c["target"] == "text" else line   # quoted text is a mention, not a use
                found += [(no, m.group(0)) for m in rx.finditer(probe)]
            if len(found) < c.get("min", 1):    # habits about frequency count only when they repeat
                continue
            hits += [{"line": no, "habit": h["name"], "key": h["key"], "what": c["what"], "match": g[:60], "evidence": h["card_url"]} for no, g in found]
    return hits


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("file")
    here = Path(__file__).resolve().parent
    ap.add_argument("--rules", default=str(next((p for p in (here / "rules.json", here.parent / "rules.json") if p.exists()), here / "rules.json")))
    a = ap.parse_args()
    rules = json.loads(Path(a.rules).read_text())
    text = sys.stdin.read() if a.file == "-" else Path(a.file).read_text(errors="replace")
    hits = check(text, rules, kinds_for("x.txt" if a.file == "-" else a.file))
    for h in hits:
        print(f"{a.file}:{h['line']}: {h['habit']} ({h['what']}): {h['match']!r}  {h['evidence']}")
    print(f"{len(hits)} hits against {rules['vendor']} rules of {rules['generated'][:10]}", file=sys.stderr)
    sys.exit(1 if hits else 0)


if __name__ == "__main__":
    main()
