X-S2-03 Readability report

Field Value
Purpose Print a text file’s Grade Level and Reading Ease scores against target bands, each clearly labelled, and never fail on its own.
Usage python3 scripts/s2/readability_report.py –help In a shell: python3 scripts/s2/readability_report.py TEXT.md [–grade-low N] [–grade-high N] [–re-low N] [–re-high N]
Dependencies stdlib
Writes files no
License CC0-1.0
Inputs One Markdown or plain-text file.
Outputs Two lines, one per readability scale, each followed by in-band or out-of-band, printed to standard output.
Used in S2.3 and S2.4 Readability polish and six-pass revision
Source scripts/s2/readability_report.py

Source code

"""
ID: X-S2-03
Title: Readability report
Stage: S2
Purpose: Print a text file's Grade Level and Reading Ease scores against
    target bands, each clearly labelled, and never fail on its own.
Usage: python3 scripts/s2/readability_report.py --help
    In a shell: python3 scripts/s2/readability_report.py TEXT.md
    [--grade-low N] [--grade-high N] [--re-low N] [--re-high N]
Dependencies: stdlib
Writes files: no
License: CC0-1.0
Inputs: One Markdown or plain-text file.
Outputs: Two lines, one per readability scale, each followed by
    in-band or out-of-band, printed to standard output.

This script only reads and reports. in-band and out-of-band never
change the exit code, and this script never passes or fails anything
by itself: it exits 0 whenever it can compute both scores, and 2 only
for a usage or input error, such as a missing file, a file with no
words or no sentences to score, or a target band whose low end is
above its high end.
"""

import argparse
import re
import sys
from pathlib import Path

sys.dont_write_bytecode = True
if sys.version_info < (3, 10):
    print(
        "readability_report.py: this script needs Python 3.10 or newer, "
        f"but this is {sys.version_info.major}.{sys.version_info.minor}. "
        "Run it with a newer python3.",
        file=sys.stderr,
    )
    sys.exit(2)

MAX_BYTES = 5_000_000
DEFAULT_GRADE_LOW = 9.0
DEFAULT_GRADE_HIGH = 11.0
DEFAULT_RE_LOW = 60.0
DEFAULT_RE_HIGH = 70.0

WORD_RE = re.compile(r"[A-Za-z]+(?:'[A-Za-z]+)?")
SENTENCE_SPLIT_RE = re.compile(r"[.!?]+(?:\s+|$)")
VOWEL_GROUP_RE = re.compile(r"[aeiouy]+")
NON_LETTER_RE = re.compile(r"[^a-z]")


def load_text(path: Path) -> str:
    """Read the input file as UTF-8; never follows a symlink."""
    if path.is_symlink():
        raise OSError(f"refusing to read a symlink: {path}")
    size = path.stat().st_size
    if size > MAX_BYTES:
        raise ValueError(f"{path} is over {MAX_BYTES} bytes; skipping")
    return path.read_text(encoding="utf-8", errors="replace")


def split_sentences(text: str) -> list[str]:
    """Split text on '.', '!' or '?' runs; keep only pieces with a word."""
    pieces = SENTENCE_SPLIT_RE.split(text)
    return [piece for piece in pieces if WORD_RE.search(piece)]


def count_syllables(word: str) -> int:
    """Approximate one word's syllable count from its vowel groups."""
    cleaned = NON_LETTER_RE.sub("", word.lower())
    if not cleaned:
        return 0
    groups = VOWEL_GROUP_RE.findall(cleaned)
    count = len(groups)
    if cleaned.endswith("e") and not cleaned.endswith("le") and count > 1:
        count -= 1
    return max(count, 1)


def compute_scores(text: str) -> tuple[float, float]:
    """Return (grade_level, reading_ease) for the given text.

    Uses the standard Flesch-Kincaid Grade Level and Flesch Reading Ease
    formulas, with word, sentence and syllable counts from simple rules
    (no third-party text-statistics package).
    """
    words = WORD_RE.findall(text)
    sentences = split_sentences(text)
    if not words:
        raise ValueError("the text has no words to score")
    if not sentences:
        raise ValueError("the text has no sentences to score")
    word_count = len(words)
    sentence_count = len(sentences)
    syllable_count = sum(count_syllables(word) for word in words)
    words_per_sentence = word_count / sentence_count
    syllables_per_word = syllable_count / word_count
    grade_level = 0.39 * words_per_sentence + 11.8 * syllables_per_word - 15.59
    reading_ease = 206.835 - 1.015 * words_per_sentence - 84.6 * syllables_per_word
    return grade_level, reading_ease


def band_status(value: float, low: float, high: float) -> str:
    """'in-band' when low <= value <= high, else 'out-of-band'."""
    return "in-band" if low <= value <= high else "out-of-band"


def format_bound(value: float) -> str:
    """Print a whole-number bound without a trailing '.0'."""
    if value == int(value):
        return str(int(value))
    return f"{value:g}"


def format_grade_line(grade_level: float, low: float, high: float) -> str:
    """One labelled Grade Level line, ending in in-band or out-of-band."""
    status = band_status(grade_level, low, high)
    bounds = f"{format_bound(low)}-{format_bound(high)}"
    return f"grade-level {grade_level:.2f} (target {bounds}) {status}"


def format_reading_ease_line(reading_ease: float, low: float, high: float) -> str:
    """One labelled Reading Ease line, ending in in-band or out-of-band."""
    status = band_status(reading_ease, low, high)
    bounds = f"{format_bound(low)}-{format_bound(high)}"
    return (
        f"reading-ease {reading_ease:.2f} (target {bounds}, "
        f"higher is easier) {status}"
    )


def build_parser() -> argparse.ArgumentParser:
    """Build the argument parser."""
    parser = argparse.ArgumentParser(
        prog="readability_report.py",
        description=(
            "Print a text file's Grade Level and Reading Ease scores "
            "against target bands, each clearly labelled. Writes no "
            "files, and never fails on its own: in-band and out-of-band "
            "never change the exit code."
        ),
    )
    parser.add_argument("text", metavar="TEXT", help="path to the text file")
    parser.add_argument(
        "--grade-low",
        type=float,
        default=DEFAULT_GRADE_LOW,
        metavar="N",
        help=f"low end of the Grade Level target band (default: "
        f"{format_bound(DEFAULT_GRADE_LOW)})",
    )
    parser.add_argument(
        "--grade-high",
        type=float,
        default=DEFAULT_GRADE_HIGH,
        metavar="N",
        help=f"high end of the Grade Level target band (default: "
        f"{format_bound(DEFAULT_GRADE_HIGH)})",
    )
    parser.add_argument(
        "--re-low",
        type=float,
        default=DEFAULT_RE_LOW,
        metavar="N",
        help=f"low end of the Reading Ease target band (default: "
        f"{format_bound(DEFAULT_RE_LOW)})",
    )
    parser.add_argument(
        "--re-high",
        type=float,
        default=DEFAULT_RE_HIGH,
        metavar="N",
        help=f"high end of the Reading Ease target band (default: "
        f"{format_bound(DEFAULT_RE_HIGH)})",
    )
    return parser


def main(argv: list[str] | None = None) -> int:
    """Command line entry point; prints the report, returns an exit code."""
    parser = build_parser()
    args = parser.parse_args(argv)
    try:
        if args.grade_low > args.grade_high:
            raise ValueError("--grade-low is greater than --grade-high")
        if args.re_low > args.re_high:
            raise ValueError("--re-low is greater than --re-high")
        text = load_text(Path(args.text))
        grade_level, reading_ease = compute_scores(text)
    except (OSError, UnicodeError, ValueError, RecursionError) as exc:
        print(f"readability_report.py: error: {exc}", file=sys.stderr)
        return 2
    print(format_grade_line(grade_level, args.grade_low, args.grade_high))
    print(format_reading_ease_line(reading_ease, args.re_low, args.re_high))
    return 0


if __name__ == "__main__":
    sys.exit(main())

To the extent possible under law, copyright and related rights in this work are waived under CC0 1.0 Universal.

This site uses Just the Docs, a documentation theme for Jekyll.