#!/usr/bin/env python3
"""Check a clause review file against its source document.

Usage: python3 clause-check.py SOURCE REVIEW

Checks that:
  * every numbered clause in SOURCE appears exactly once in REVIEW
    as a line "Clause N: ASK" or "Clause N: OK";
  * REVIEW names no clause number that SOURCE lacks;
  * every ASK entry has a Quote: line and a Question: line;
  * every OK entry is a single line with nothing else attached;
  * every quoted passage appears word for word in SOURCE, inside the
    clause it is filed under.

Prints each problem found. Exits 1 if any check fails, 2 on usage or
read errors, and 0 if everything passes.
"""

import re
import sys

CLAUSE_RE = re.compile(r"^\s*(\d+)\.\s")
ENTRY_RE = re.compile(r"^Clause (\d+): (ASK|OK)\s*$")
LOOSE_ENTRY_RE = re.compile(r"^\s*Clause\s+(\d+)\b", re.IGNORECASE)
QUOTE_RE = re.compile(r'^Quote: "(.+)"\s*$')


def norm(text):
    """Collapse runs of whitespace so line wrapping does not matter."""
    return " ".join(text.split())


def read(path):
    try:
        with open(path, encoding="utf-8") as f:
            return f.read()
    except OSError as e:
        print(f"ERROR: cannot read {path}: {e}")
        sys.exit(2)


def source_clauses(text):
    """Return {number: clause text} for each numbered clause, in order."""
    clauses = {}
    duplicates = []
    current = None
    for line in text.splitlines():
        m = CLAUSE_RE.match(line)
        if m:
            current = int(m.group(1))
            if current in clauses:
                duplicates.append(current)
            clauses[current] = line
        elif current is not None and line.strip():
            clauses[current] += " " + line
    return {n: norm(t) for n, t in clauses.items()}, duplicates


def review_entries(text):
    """Return a list of (number, status, body_lines, line_no) and format problems."""
    entries = []
    problems = []
    for line_no, line in enumerate(text.splitlines(), 1):
        m = ENTRY_RE.match(line)
        if m:
            entries.append((int(m.group(1)), m.group(2), [], line_no))
        elif LOOSE_ENTRY_RE.match(line):
            problems.append(
                f"line {line_no}: malformed entry header {line.strip()!r}; "
                "expected 'Clause N: ASK' or 'Clause N: OK'"
            )
        elif entries and line.strip():
            entries[-1][2].append((line_no, line))
    return entries, problems


def main(argv):
    if len(argv) != 3:
        print("usage: python3 clause-check.py SOURCE REVIEW")
        return 2
    source_path, review_path = argv[1], argv[2]
    source_text = read(source_path)
    review_text = read(review_path)
    flat_source = norm(source_text)

    clauses, dup_source = source_clauses(source_text)
    entries, problems = review_entries(review_text)

    if not clauses:
        problems.append(f"no numbered clauses found in {source_path}")
    for n in dup_source:
        problems.append(f"clause {n} is numbered more than once in {source_path}")

    counts = {}
    for n, _, _, _ in entries:
        counts[n] = counts.get(n, 0) + 1
    for n in clauses:
        c = counts.get(n, 0)
        if c == 0:
            problems.append(f"clause {n} is missing from {review_path}")
        elif c > 1:
            problems.append(f"clause {n} appears {c} times in {review_path}")
    for n in sorted(counts):
        if n not in clauses:
            problems.append(
                f"{review_path} lists clause {n}, which does not exist in {source_path}"
            )

    for n, status, body, line_no in entries:
        where = f"clause {n} (line {line_no})"
        if status == "OK":
            if body:
                problems.append(
                    f"{where}: OK entry should be one line but has extra text "
                    f"on line {body[0][0]}"
                )
            continue
        quotes = [(ln, l) for ln, l in body if l.startswith("Quote:")]
        questions = [(ln, l) for ln, l in body if l.startswith("Question:")]
        if not quotes:
            problems.append(f"{where}: ASK entry has no Quote: line")
        if not questions:
            problems.append(f"{where}: ASK entry has no Question: line")
        for ln, l in questions:
            if not l[len("Question:"):].strip():
                problems.append(f"line {ln}: Question: line is empty")
        for ln, l in quotes:
            m = QUOTE_RE.match(l)
            if not m:
                problems.append(
                    f'line {ln}: Quote: line must be Quote: "exact words" '
                    "in double quotes"
                )
                continue
            passage = norm(m.group(1))
            if passage not in flat_source:
                problems.append(
                    f"line {ln}: quoted passage not found word for word in "
                    f"{source_path}: \"{passage}\""
                )
            elif n in clauses and passage not in clauses[n]:
                problems.append(
                    f"line {ln}: quoted passage is in {source_path} but not "
                    f"in clause {n}: \"{passage}\""
                )

    if problems:
        print(f"FAIL: {len(problems)} problem(s) found")
        for p in problems:
            print(f"  - {p}")
        return 1
    asks = sum(1 for e in entries if e[1] == "ASK")
    print(
        f"PASS: all {len(clauses)} clauses appear exactly once "
        f"({asks} ASK, {len(entries) - asks} OK); every quote matches the source"
    )
    return 0


if __name__ == "__main__":
    sys.exit(main(sys.argv))
