#!/usr/bin/env python3
"""
Readability measurement for the Clarity dimension of the Data Human Index.

Why this exists rather than a web tool: the score has to be reproducible by
anyone who disagrees with it, and a third-party editor that silently reparses
pasted text is not reproducible. Hemingway treats a PDF dump's hard line breaks
as sentence ends, which inflates difficulty; a different tool forgives them and
deflates it. Same document, opposite conclusions, and neither is auditable.

Everything here is public arithmetic with no dependencies.

    python3 scripts/readability.py --url https://example.com/privacy
    python3 scripts/readability.py --file policy.txt

Sentence counting is the one judgment call. A privacy policy is mostly bullets,
and a bullet with no full stop is still a sentence a person has to read. So each
block element counts as at least one sentence, and blocks split further on
terminal punctuation. Treating line breaks as sentence ends, or ignoring
unpunctuated bullets, are both wrong in ways that move the answer by grades.
"""

import argparse
import json
import re
import sys
from html.parser import HTMLParser
from urllib.request import Request, urlopen

BLOCK = {
    'p', 'li', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6',
    'td', 'th', 'dd', 'dt', 'blockquote', 'div', 'section', 'article', 'br',
}
DROP = {'script', 'style', 'noscript', 'nav', 'header', 'footer', 'svg', 'form'}


class Extract(HTMLParser):
    """Pull block-level text out of a page, keeping the block boundaries."""

    def __init__(self):
        super().__init__(convert_charrefs=True)
        self.blocks = []
        self.buf = []
        self.skip = 0

    def handle_starttag(self, tag, attrs):
        if tag in DROP:
            self.skip += 1
        elif tag in BLOCK:
            self.flush()

    def handle_endtag(self, tag):
        if tag in DROP and self.skip:
            self.skip -= 1
        elif tag in BLOCK:
            self.flush()

    def handle_data(self, data):
        if not self.skip:
            self.buf.append(data)

    def flush(self):
        text = re.sub(r'\s+', ' ', ''.join(self.buf)).strip()
        if text:
            self.blocks.append(text)
        self.buf = []

    def close(self):
        super().close()
        self.flush()


BULLET = re.compile(r'^\s*(?:[•●▪·*\-–—]|\(?[a-z0-9]{1,3}[.)])\s+')
PAGE_NUMBER = re.compile(r'^\s*(?:page\s+)?\d{1,3}\s*$', re.I)


def plain_blocks(raw):
    """Rebuild paragraphs from extracted plain text.

    This is the function that matters. A PDF converter wraps prose at the page
    margin, so a single sentence arrives as four lines. Treat each line as a
    sentence and the average sentence length collapses, which drags every
    grade-level formula down with it. That is exactly how a nine-thousand-word
    bank policy can be made to look like a children's book, and it is the error
    this whole exercise started with.

    So: blank lines separate blocks, wrapped lines are rejoined, and a new block
    begins only at a bullet or a numbered item.
    """
    blocks = []
    for chunk in re.split(r'\n\s*\n', raw):
        current = []
        for line in chunk.split('\n'):
            if not line.strip() or PAGE_NUMBER.match(line):
                continue
            if BULLET.match(line) and current:
                blocks.append(' '.join(current))
                current = [line.strip()]
            else:
                current.append(line.strip())
        if current:
            blocks.append(' '.join(current))
    return [re.sub(r'\s+', ' ', b).strip() for b in blocks if b.strip()]


VOWELS = 'aeiouy'


def syllables(word):
    """The standard heuristic: vowel groups, minus a silent trailing e."""
    w = re.sub(r'[^a-z]', '', word.lower())
    if not w:
        return 0
    groups = re.findall(r'[aeiouy]+', w)
    n = len(groups)
    if w.endswith('e') and not w.endswith(('le', 'ee')) and n > 1:
        n -= 1
    return max(1, n)


def analyse(blocks):
    sentences = []
    for block in blocks:
        # Split on terminal punctuation followed by a space and a capital, so
        # "Inc." and "e.g." do not each become a sentence.
        parts = re.split(r'(?<=[.!?])\s+(?=[A-Z"“])', block)
        parts = [p.strip() for p in parts if p.strip()]
        # A bullet with no full stop is still something a person has to read.
        sentences.extend(parts or [block])

    words = []
    for s in sentences:
        words.extend(re.findall(r"[A-Za-z][A-Za-z'\-]*", s))

    n_words = len(words)
    n_sents = len(sentences)
    if not n_words or not n_sents:
        raise SystemExit('No readable text found.')

    syl = [syllables(w) for w in words]
    n_syl = sum(syl)
    polysyll = sum(1 for s in syl if s >= 3)
    letters = sum(len(re.sub(r"[^A-Za-z]", '', w)) for w in words)

    wps = n_words / n_sents
    spw = n_syl / n_words

    flesch = 206.835 - 1.015 * wps - 84.6 * spw
    fk = 0.39 * wps + 11.8 * spw - 15.59
    fog = 0.4 * (wps + 100 * polysyll / n_words)
    smog = 1.0430 * ((polysyll * (30 / n_sents)) ** 0.5) + 3.1291
    coleman = 0.0588 * (letters / n_words * 100) - 0.296 * (n_sents / n_words * 100) - 15.8
    ari = 4.71 * (letters / n_words) + 0.5 * wps - 21.43

    grades = sorted([fk, fog, smog, coleman, ari])
    consensus = (grades[2] + grades[2]) / 2  # median of five

    return {
        'words': n_words,
        'sentences': n_sents,
        'words_per_sentence': round(wps, 1),
        'polysyllabic_pct': round(100 * polysyll / n_words, 1),
        'flesch_reading_ease': round(flesch, 1),
        'flesch_kincaid_grade': round(fk, 1),
        'gunning_fog': round(fog, 1),
        'smog': round(smog, 1),
        'coleman_liau': round(coleman, 1),
        'automated_readability': round(ari, 1),
        'consensus_grade': round(consensus, 1),
    }


def main():
    ap = argparse.ArgumentParser()
    src = ap.add_mutually_exclusive_group(required=True)
    src.add_argument('--url')
    src.add_argument('--file')
    ap.add_argument('--json', action='store_true')
    args = ap.parse_args()

    if args.url:
        req = Request(args.url, headers={'User-Agent': 'Mozilla/5.0 (DataHumanIndex readability check)'})
        html = urlopen(req, timeout=30).read().decode('utf-8', 'replace')
        p = Extract()
        p.feed(html)
        p.close()
        blocks = p.blocks
    else:
        raw = open(args.file, encoding='utf-8').read()
        if '<' in raw and '>' in raw:
            p = Extract()
            p.feed(raw)
            p.close()
            blocks = p.blocks
        else:
            blocks = plain_blocks(raw)

    r = analyse(blocks)

    if args.json:
        print(json.dumps(r, indent=2))
        return

    print('  words                %d' % r['words'])
    print('  sentences            %d' % r['sentences'])
    print('  words per sentence   %.1f' % r['words_per_sentence'])
    print('  polysyllabic         %.1f%%' % r['polysyllabic_pct'])
    print()
    print('  Flesch Reading Ease  %.1f   (higher is easier; under 50 is difficult)' % r['flesch_reading_ease'])
    print('  Flesch-Kincaid       %.1f' % r['flesch_kincaid_grade'])
    print('  Gunning Fog          %.1f' % r['gunning_fog'])
    print('  SMOG                 %.1f' % r['smog'])
    print('  Coleman-Liau         %.1f' % r['coleman_liau'])
    print('  Automated Readability %.1f' % r['automated_readability'])
    print()
    print('  CONSENSUS GRADE      %.1f' % r['consensus_grade'])


if __name__ == '__main__':
    main()
