feat(statusbar): word count + reading time + Flesch-Kincaid grade

- src/utils/writing-stats.js — pure module: stripMarkdown() drops fenced
  code blocks, inline code, image URLs, link URLs, headings, blockquote
  markers, list bullets, and emphasis markers so the word count reflects
  the prose a reader actually consumes (the convention used by Hemingway,
  iA Writer). splitSentences / countSyllables use Flesch's standard
  heuristics. computeStats returns wordCount, sentenceCount,
  syllableCount, readingTimeMinutes (default 220 wpm),
  fleschKincaidGrade, charCount.
- src/renderer.js — _updateStatusBar now calls computeStats and updates
  three new status items: ~X min read, Grade N.N, and the existing Words
  + Chars counters use the stripped count.
- src/index.html — two new status items: #reading-time and #grade-level.

Tests (23 new, tests/writing-stats.test.js): markdown stripping (fenced,
inline, images, links, wikilinks, headings, lists, emphasis, HTML),
sentence splitter edge cases, syllable heuristic (short words, silent e,
empty), and computeStats with custom wpm + null-safe inputs.

Full suite: 68 suites, 789 tests, lint+format clean.

Amit Haridas
This commit is contained in:
2026-09-14 08:54:28 +05:30
parent da97369b44
commit 7397e7f618
4 changed files with 347 additions and 7 deletions
+147
View File
@@ -0,0 +1,147 @@
/**
* Writing stats — word count, reading time, Flesch-Kincaid grade level.
*
* Pure, dependency-free module. The renderer pulls this into a status-bar
* widget that updates on every keystroke. Counts exclude code fences and
* link/image URLs (the prose a reader actually consumes), which is what
* serious writing tools (Hemingway, iA Writer) do.
*
* Flesch-Kincaid grade level is the standard US-grade readability score:
grade = 0.39 * (words/sentences) + 11.8 * (syllables/words) - 15.59
* Used widely in education and journalism to gauge reading difficulty.
*
* @module writing-stats
*/
/**
* Strip markdown chrome so we count the prose a reader actually sees.
* - fenced code blocks ```…```
* - inline code `…`
* - image markup ![alt](url)
* - link markup [text](url) → keep "text" only
* - emphasis markers *, _, **, __
* - heading hashes at line start
* - blockquote markers
* - list bullets at line start (- * +)
* - HTML comments <!-- … -->
*/
function stripMarkdown(content) {
if (typeof content !== 'string') return '';
let s = content;
// Fenced code blocks: drop the whole block including the fences
s = s.replace(/```[\s\S]*?```/g, ' ');
s = s.replace(/~~~[\s\S]*?~~~/g, ' ');
// Inline code
s = s.replace(/`[^`\n]*`/g, ' ');
// Images: drop entirely (the URL is not prose)
s = s.replace(/!\[[^\]]*\]\([^)]*\)/g, ' ');
// Links: keep the text, drop the URL part
s = s.replace(/\[([^\]]*)\]\([^)]*\)/g, '$1');
// Wikilinks: keep the visible label
s = s.replace(/\[\[([^\]|]+)(\|([^\]]+))?\]\]/g, (_, a, _b, c) => (c || a));
// Headings at line start
s = s.replace(/^\s{0,3}#{1,6}\s+/gm, '');
// Blockquote markers
s = s.replace(/^\s{0,3}>\s?/gm, '');
// List bullets / numbers at line start
s = s.replace(/^\s{0,3}([-*+]|\d+\.)\s+/gm, '');
// Emphasis markers
s = s.replace(/(\*\*|__)(.*?)\1/g, '$2');
s = s.replace(/(\*|_)(.*?)\1/g, '$2');
// Strikethrough
s = s.replace(/~~(.*?)~~/g, '$1');
// HTML comments
s = s.replace(/<!--[\s\S]*?-->/g, ' ');
// Horizontal rules
s = s.replace(/^\s*([-*_])\s*\1\s*\1[\s\S]*?$/gm, ' ');
return s;
}
/** Split prose into sentences. Robust enough for ASCII prose + common punctuation. */
function splitSentences(prose) {
if (!prose) return [];
// Split on . ! ? followed by whitespace or end. Keep abbreviations rough.
const parts = prose.split(/(?<=[.!?])\s+/);
return parts
.map((s) => s.trim())
.filter((s) => /[A-Za-z0-9À-ɏ]/.test(s) && s.length >= 2);
}
/**
* Approximate syllable count for an English word. The standard regex-based
* heuristic (Flesch's original) — accurate to within ±1 syllable for ~80%
* of common English words, which is good enough for grade-level scoring.
*/
function countSyllablesInWord(word) {
if (!word) return 0;
const w = word.toLowerCase().replace(/[^a-z]/g, '');
if (w.length === 0) return 0;
if (w.length <= 3) return 1;
// Drop trailing silent e, es, ed
let stripped = w.replace(/(?:[^laeiouy]es|ed|[^laeiouy]e)$/, '');
stripped = stripped.replace(/^y/, '');
const matches = stripped.match(/[aeiouy]{1,2}/g);
return matches ? matches.length : 1;
}
function countSyllables(prose) {
const words = (prose.match(/\b[A-Za-z'À-ɏ]+\b/g) || []);
let total = 0;
for (const w of words) total += countSyllablesInWord(w);
return total;
}
function countWords(prose) {
if (!prose) return 0;
const matches = prose.match(/\b[\w'À-ɏ]+\b/g);
return matches ? matches.length : 0;
}
/**
* Compute the full writing-stats report for a markdown document.
*
* @param {string} content raw markdown
* @param {object} [opts]
* @param {number} [opts.wordsPerMinute=220] average reading speed for English prose
* @returns {{
* wordCount:number,
* sentenceCount:number,
* syllableCount:number,
* readingTimeMinutes:number,
* fleschKincaidGrade:number|null,
* charCount:number
* }}
*/
function computeStats(content, opts = {}) {
const prose = stripMarkdown(content);
const words = countWords(prose);
const sentences = splitSentences(prose).length;
const syllables = countSyllables(prose);
const wpm = typeof opts.wordsPerMinute === 'number' && opts.wordsPerMinute > 0 ? opts.wordsPerMinute : 220;
const readingTimeMinutes = words > 0 ? words / wpm : 0;
let grade = null;
if (sentences > 0 && words > 0) {
grade = 0.39 * (words / sentences) + 11.8 * (syllables / words) - 15.59;
// Round to one decimal for status-bar display; round-trips cleanly.
grade = Math.round(grade * 10) / 10;
}
return {
wordCount: words,
sentenceCount: sentences,
syllableCount: syllables,
readingTimeMinutes,
fleschKincaidGrade: grade,
charCount: prose.replace(/\s/g, '').length,
};
}
module.exports = {
stripMarkdown,
splitSentences,
countSyllablesInWord,
countSyllables,
countWords,
computeStats,
};