"""
Extract spec sections from a construction spec PDF, plus pages in other
sections that reference them.

Works on specs that print the section number on every page, such as
"SECTION 26 28 00.00 28 Page 5" (UFGS) or "SECTION 26 05 00 - 3".

Examples
  All of Division 26:
    python extract_sections.py spec.pdf div26.pdf --sections 26

  Division 26 plus pages elsewhere that reference it:
    python extract_sections.py spec.pdf div26.pdf --sections 26 --refs

  Selected Division 01 sections, skipping one:
    python extract_sections.py spec.pdf div01.pdf --sections "01 14 00" "01 32 01" --exclude "01 14 00.90"

  Only the references, to review scope boundaries on their own:
    python extract_sections.py spec.pdf div26_refs.pdf --sections 26 --refs-only

A section is matched by prefix, so "26" matches every Division 26 section
and "01 14 00" matches both "01 14 00.10 28" and "01 14 00.90 28".

Requires: pip install pypdf
"""

import argparse
import re
import sys

from pypdf import PdfReader, PdfWriter

# A section number: 6 digits in pairs, optionally followed by a UFGS suffix like ".00 28"
SECTION_NUM = r"\d{2}\s*\d{2}\s*\d{2}(?:\s*\.\s*\d{2}\s*\d{2})?"
# The page's own header or footer: "SECTION 26 28 00.00 28 Page 5" or "SECTION 26 05 00 - 3"
OWN_HEADER = re.compile(r"SECTION\s+(" + SECTION_NUM + r")\s*(?:Page|-)\s*\d+")
# Any reference to a section in the text, any case
ANY_REF = re.compile(r"section\s+(" + SECTION_NUM + r")", re.IGNORECASE)


def norm(num):
    """Normalize a section number to single-spaced pairs, e.g. '262800.0028' -> '26 28 00.00 28'."""
    digits = re.sub(r"[^\d.]", "", num)
    main, _, suffix = digits.partition(".")
    main = " ".join(main[i:i + 2] for i in range(0, len(main), 2))
    if suffix:
        suffix = " ".join(suffix[i:i + 2] for i in range(0, len(suffix), 2))
        return f"{main}.{suffix}"
    return main


def matches(section, prefixes):
    return any(section.startswith(p) for p in prefixes)


def main():
    parser = argparse.ArgumentParser(description="Extract spec sections and the pages that reference them.")
    parser.add_argument("input", help="spec PDF to read")
    parser.add_argument("output", help="PDF to write")
    parser.add_argument("--sections", nargs="+", required=True,
                        help='section numbers or prefixes to keep, e.g. 26 or "01 14 00"')
    parser.add_argument("--exclude", nargs="*", default=[],
                        help="section numbers or prefixes to leave out")
    group = parser.add_mutually_exclusive_group()
    group.add_argument("--refs", action="store_true",
                       help="also keep pages from other sections that reference the kept sections")
    group.add_argument("--refs-only", action="store_true",
                       help="keep only the referencing pages from other sections")
    args = parser.parse_args()

    wanted = [norm(s) for s in args.sections]
    excluded = [norm(s) for s in args.exclude]

    reader = PdfReader(args.input)
    writer = PdfWriter()
    kept = []
    unlabeled = 0

    for i, page in enumerate(reader.pages, start=1):
        text = page.extract_text() or ""
        header = OWN_HEADER.search(text)
        own = norm(header.group(1)) if header else None
        if own is None:
            unlabeled += 1

        is_own = own is not None and matches(own, wanted) and not matches(own, excluded)

        refs = []
        if own is None or not matches(own, wanted):
            for m in ANY_REF.finditer(text):
                ref = norm(m.group(1))
                if matches(ref, wanted) and not matches(ref, excluded) and ref not in refs:
                    refs.append(ref)
        is_ref = bool(refs) and own is not None and not matches(own, excluded)

        keep = (is_own and not args.refs_only) or (is_ref and (args.refs or args.refs_only))
        if keep:
            writer.add_page(page)
            kept.append((i, own or "unlabeled", "section" if is_own else "reference", refs))

    if not kept:
        print("No pages matched. Check how the section number prints on the pages.")
        sys.exit(1)

    writer.write(args.output)

    by_section = {}
    for _, own, kind, _ in kept:
        if kind == "section":
            by_section[own] = by_section.get(own, 0) + 1

    print(f"Wrote {len(kept)} pages to {args.output}\n")
    if by_section:
        print("Section pages:")
        for sec, n in sorted(by_section.items()):
            print(f"  {sec}: {n}")
    ref_pages = [k for k in kept if k[2] == "reference"]
    if ref_pages:
        print("\nReferencing pages from other sections:")
        for page_num, own, _, refs in ref_pages:
            print(f"  PDF page {page_num} ({own}) references {', '.join(refs)}")
    if unlabeled:
        print(f"\n{unlabeled} pages had no section header (covers, tables of contents) and were skipped.")


if __name__ == "__main__":
    main()