feat(pkm): daily notes + workspace search + doc-aware Q&A

Three more features from the brainstorm menu, built on a shared search
algorithm so the codebase stays small.

- src/main/DailyNotes.js — Zettelkasten-style helper. One YYYY-MM-DD.md
  per local date under <userData>/notes/daily/; loads skeleton from
  <userData>/notes/templates/daily.md when present (built-in default
  otherwise). openOrCreate never clobbers existing content.
- src/main/WorkspaceSearch.js — tag/wikilink-aware content search.
  Pure module, injectable-IO tested. Query grammar: bare words,
  #tag, @wikilink, "quoted phrases". Facets weight +3 each; prose
  terms +1/occurrence capped at 5; edits within 7 days get a recency
  nudge. Returns ranked results with snippets.
- src/main/DocQA.js — chunk-level Q&A wrapper over WorkspaceSearch.
  cleanQuestion strips question words (what/how/why/...) and verb
  noise (write/read/show/tell/...) so they don't drown the ranking.
  Returns top-K passages instead of whole-file hits — multiple chunks
  from the same file can appear in the answer.
- src/main.js — IPC: daily-notes:open-today, daily-notes:list,
  workspace-search:query, doc-qa:ask. Path validation through the
  existing validatePath gate; a global Ctrl+Alt+D shortcut creates
  today's daily note from anywhere.
- src/preload.js — all four channels added to ALLOWED_SEND_CHANNELS.

Tests (56 new across the three modules):
- tests/main/DailyNotes.test.js (15): dateKey formatting, pathFor,
  template load + {date}/{weekday} substitution, openOrCreate +
  no-clobber, nested-dir creation, listExisting filtering,
  isValidDir rejects NUL/non-string.
- tests/main/WorkspaceSearch.test.js (25): parseQuery grammar,
  hasTag/hasWikilink word boundaries, scoreDocument scoring,
  per-term spam cap, recency nudge, search ranking + limit +
  empty-query short-circuit, bad-input safety.
- tests/main/DocQA.test.js (16): cleanQuestion stripping + facet
  preservation, chunkDocument paragraph + hard-split, ask()
  top-K, recency tiebreaker, missing-files fallback.

Full suite: 748 tests pass, 66 suites, lint+format clean.

Amit Haridas
This commit is contained in:
2026-09-14 00:02:58 +05:30
parent cd0050d988
commit 3f856bc857
9 changed files with 1275 additions and 2 deletions
+130
View File
@@ -0,0 +1,130 @@
/**
* Daily notes — Zettelkasten/journal entry helper.
*
* Convention: a folder of YYYY-MM-DD.md files (PKM/Obsidian/Logseq style).
* `pathFor(date, dir)` returns the absolute path for a given date. The
* template loader returns a starting skeleton the user can fill in; if the
* file already exists, we return its current content instead (no clobber).
*
* Implementation is pure (no fs/IPC coupling) — main.js wires the IO and
* template path. Tests inject a virtual fs + path util.
*/
const DEFAULT_TEMPLATE_NAME = 'daily.md';
/**
* Format a Date as YYYY-MM-DD in local time (matches the on-disk filename).
* Returns 'YYYY-MM-DD' string. Today (no arg) defaults to new Date().
*/
function dateKey(date = new Date()) {
const y = date.getFullYear();
const m = String(date.getMonth() + 1).padStart(2, '0');
const d = String(date.getDate()).padStart(2, '0');
return `${y}-${m}-${d}`;
}
/**
* Absolute path for the daily note of `date` inside `dir`. Pure.
*/
function pathFor(date, dir, pathUtil) {
return pathUtil.join(dir, `${dateKey(date)}.md`);
}
/**
* Read the daily note skeleton. Looks in `templateDir` for a file named
* `templateName` (default 'daily.md'); if missing, returns a built-in
* default so the feature works out-of-the-box without setup.
*
* Returns the template string with any {date}/{weekday} placeholders
* substituted.
*/
function loadTemplate({
date,
templateDir,
templateName = DEFAULT_TEMPLATE_NAME,
fs,
pathUtil,
now = new Date(),
}) {
let body = `# ${dateKey(date)}\n\n## Notes\n\n`;
if (templateDir && fs) {
const templatePath = pathUtil.join(templateDir, templateName);
try {
body = fs.readFileSync(templatePath, 'utf-8');
} catch {
/* fall through to default */
}
}
return body
.replace(/\{date\}/g, dateKey(date))
.replace(/\{weekday\}/g, now.toLocaleDateString('en-US', { weekday: 'long' }));
}
/**
* Open or create the daily note for `date` inside `dir`. Returns
* { path, content, created: boolean }.
*
* Behavior:
* - If <dir>/<date>.md exists, return its content (created: false).
* - Otherwise load the template, write it, and return it (created: true).
*
* Caller (main.js IPC handler) decides what to do with `created` —
* typically: open in the existing tab if any, else createNewTab.
*/
function openOrCreate({ date, dir, templateDir, fs, pathUtil, now = new Date() }) {
if (!dir) throw new Error('DailyNotes: dir is required');
fs.mkdirSync(dir, { recursive: true });
const notePath = pathFor(date, dir, pathUtil);
let content;
let created = false;
try {
content = fs.readFileSync(notePath, 'utf-8');
} catch (err) {
if (err.code !== 'ENOENT') throw err;
content = loadTemplate({ date, templateDir, fs, pathUtil, now });
fs.writeFileSync(notePath, content, 'utf-8');
created = true;
}
return { path: notePath, content, created };
}
/**
* List existing daily-note filenames in `dir`, newest first.
* Returns ['2026-09-13.md', '2026-09-12.md', ...] — only entries that
* match the YYYY-MM-DD.md shape so a stray readme.md in the folder is
* ignored.
*/
function listExisting({ dir, fs }) {
let entries;
try {
entries = fs.readdirSync(dir);
} catch (err) {
if (err.code === 'ENOENT') return [];
throw err;
}
// Strict-enough date pattern: year-month-day with month 01-12 and day 01-31.
// We don't enforce real-calendar validity (Feb 30 still matches) — that's
// the user's problem to fix, not ours to silently drop.
const pattern = /^\d{4}-(0[1-9]|1[0-2])-(0[1-9]|[12]\d|3[01])\.md$/;
return entries.filter((name) => pattern.test(name)).sort((a, b) => (a < b ? 1 : a > b ? -1 : 0));
}
/**
* Validate a `dir` candidate for daily notes. Used by the IPC handler so a
* renderer compromise can't point this at /etc or other sensitive paths —
* any path outside userData/templates is rejected by main.js's existing
* validatePath() before we get here.
*/
function isValidDir(dir) {
return typeof dir === 'string' && dir.length > 0 && !dir.includes('\0');
}
module.exports = {
dateKey,
pathFor,
loadTemplate,
openOrCreate,
listExisting,
isValidDir,
DEFAULT_TEMPLATE_NAME,
};
+231
View File
@@ -0,0 +1,231 @@
/**
* Doc-aware Q&A — thin wrapper over WorkspaceSearch tuned for questions.
*
* User asks "what did I write about rust async last week?" and gets back the
* top-K chunks from their workspace ranked by relevance. No neural model, no
* API key — the same TF-idf-style ranking WorkspaceSearch uses, with a few
* QA-specific tweaks:
*
* - Question words (what/how/why/when/where/who/which/does/is/are/...) are
* stripped before ranking — they don't carry meaning, just grammar
* - The top chunks are split out as independent results so the renderer
* can show "3 passages from 2 files" instead of one giant result
* - Recency weight is doubled: "last week" implies the user wants fresh
* content, not old material
*
* Future upgrade path: swap `WorkspaceSearch.search` for a vector-similarity
* call (transformers.js in the renderer, or a sidecar process). The public
* shape — {question, chunks:[{filePath, snippet, score}]} — stays the same.
*
* @module DocQA
*/
const WorkspaceSearch = require('./WorkspaceSearch');
const QUESTION_WORDS = new Set([
'what',
'when',
'where',
'who',
'whom',
'whose',
'why',
'how',
'which',
'whether',
'does',
'do',
'did',
'is',
'are',
'was',
'were',
'be',
'been',
'being',
'have',
'has',
'had',
'can',
'could',
'would',
'should',
'will',
'shall',
'may',
'might',
'i',
'me',
'my',
'we',
'our',
'you',
'your',
'they',
'them',
'their',
'a',
'an',
'the',
'and',
'or',
'but',
'so',
'of',
'to',
'in',
'on',
'at',
'for',
'with',
'about',
'into',
'from',
'by',
'as',
'this',
'that',
'these',
'those',
'it',
'its',
'write',
'wrote',
'written',
'read',
'think',
'know',
'find',
'show',
'tell',
'say',
'see',
'use',
'used',
]);
/**
* Strip question words from a raw question so WorkspaceSearch's bare-term
* matching isn't drowned in grammar noise. Quoted phrases, #tags, and
* @wikilinks pass through unchanged.
*/
function cleanQuestion(raw) {
if (typeof raw !== 'string') return '';
// First pull out facets and phrases so we don't strip them
const facets = [];
const working = raw.replace(/[#@][A-Za-z0-9_-]+|"[^"]+"/g, (m) => {
facets.push(m);
return ' ';
});
const words = working.split(/\s+/).filter((w) => {
const lower = w.toLowerCase().replace(/[^a-z0-9-]/g, '');
return lower.length >= 2 && !QUESTION_WORDS.has(lower);
});
return [...words, ...facets].join(' ').trim();
}
/**
* Split a document into rough passages (~800 chars or paragraph break,
* whichever comes first). Returns an array of {start, end, text} so the
* caller can build snippets. Pure.
*/
function chunkDocument(content, maxChunkChars = 800) {
if (typeof content !== 'string' || content.length === 0) return [];
const chunks = [];
const paragraphs = content.split(/\n\s*\n/);
let buffer = '';
let startOffset = 0;
for (const para of paragraphs) {
// If a single paragraph exceeds the cap, hard-split it by character.
if (para.length > maxChunkChars) {
// Flush whatever was buffered first.
if (buffer) {
chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
startOffset += buffer.length + 2;
buffer = '';
}
for (let i = 0; i < para.length; i += maxChunkChars) {
const slice = para.slice(i, i + maxChunkChars);
chunks.push({ start: startOffset, end: startOffset + slice.length, text: slice });
startOffset += slice.length;
}
continue;
}
const tentative = buffer ? `${buffer}\n\n${para}` : para;
if (tentative.length > maxChunkChars && buffer) {
chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
startOffset += buffer.length + 2; // account for the "\n\n" we split on
buffer = para;
} else {
buffer = tentative;
}
}
if (buffer) chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
return chunks;
}
/**
* Answer a natural-language question by ranking the workspace's chunks.
*
* @param {object} args
* @param {string} args.question
* @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files
* @param {number} [args.topK=5] number of chunks to return
* @param {number} [args.nowMs=Date.now()]
* @returns {{question:string, chunks:Array<{filePath:string, snippet:string, score:number, mtimeMs:number}>}}
*/
function ask({ question, files, topK = 5, nowMs = Date.now() }) {
const cleaned = cleanQuestion(question);
if (!cleaned || !Array.isArray(files) || files.length === 0) {
return { question: String(question || ''), chunks: [] };
}
// First, find the docs that match at all (cheap, broad pass).
const docHits = WorkspaceSearch.search({ query: cleaned, files, limit: 20, nowMs });
// Then re-rank at the chunk level within those docs.
const chunkCorpus = [];
for (const hit of docHits) {
const file = files.find((f) => f.path === hit.filePath);
if (!file || typeof file.content !== 'string') continue;
const chunks = chunkDocument(file.content);
for (const chunk of chunks) {
chunkCorpus.push({
path: `${hit.filePath}#${chunk.start}`,
content: chunk.text,
mtimeMs: file.mtimeMs,
});
}
}
const chunkHits = WorkspaceSearch.search({
query: cleaned,
files: chunkCorpus,
limit: topK,
nowMs,
});
// Translate the per-chunk hits back into the public shape. The fake path
// "<file>#<offset>" carries the chunk start; the renderer's existing
// file-opened handler can split it on '#' if it wants to deep-link.
const chunks = chunkHits.map((h) => {
const offsetMatch = /#(\d+)$/.exec(h.filePath);
const offset = offsetMatch ? Number(offsetMatch[1]) : 0;
return {
filePath: h.filePath.replace(/#\d+$/, ''),
offset,
snippet: h.snippet,
score: h.score,
mtimeMs: chunkCorpus.find((c) => c.path === h.filePath)?.mtimeMs || 0,
};
});
return { question: String(question), chunks };
}
module.exports = {
cleanQuestion,
chunkDocument,
ask,
QUESTION_WORDS,
};
+203
View File
@@ -0,0 +1,203 @@
/**
* Workspace content search — the simplest useful thing that scales.
*
* Given a list of file paths and a query, returns ranked matches across
* them. Designed for personal-workspace sizes (hundreds, maybe thousands of
* notes) — runs synchronously on the main process with injectable IO so it
* can be unit-tested without touching the filesystem.
*
* Query grammar (all optional, combinable):
* - Bare words → term matches (case-insensitive substring)
* - Words prefixed with `#` → require a matching `#tag` in the doc
* - Words prefixed with `@` → require a matching `[[wikilink]]` target
* - Quoted phrases → require the exact substring
*
* Scoring:
* - Each matched term adds +1 per occurrence (capped at 5/term to dampen
* repetition spam)
* - Each tag hit (`#foo`) or wikilink hit (`[[foo]]`) adds +3 if the query
* asked for it (facets weigh more than prose)
* - Recent edits (mtime within 7 days) get +0.5 to nudge "what I was
* working on yesterday" upward
*
* Returns [{ filePath, score, snippet, matchedTerms, matchedTags, matchedLinks }]
* sorted by score desc, filePath asc as tiebreaker.
*
* @module WorkspaceSearch
*/
/** Tokenize a query into a structured form. Pure. */
function parseQuery(raw) {
if (typeof raw !== 'string') return { terms: [], phrases: [], tags: [], links: [] };
const terms = [];
const phrases = [];
const tags = [];
const links = [];
// Phrases first — match "double quoted text" as a unit
const phraseRe = /"([^"]+)"/g;
let remaining = raw.replace(phraseRe, (_, inner) => {
if (inner.trim()) phrases.push(inner.toLowerCase());
return ' ';
});
// Tags and links: #foo, @bar
const facetRe = /[#@]([A-Za-z0-9_-]+)/g;
remaining = remaining.replace(facetRe, (m, name) => {
if (m.startsWith('#')) tags.push(name.toLowerCase());
else links.push(name.toLowerCase());
return ' ';
});
// Bare words (≥2 chars to avoid noise)
for (const w of remaining.split(/\s+/)) {
if (w.length >= 2) terms.push(w.toLowerCase());
}
return { terms, phrases, tags, links };
}
/** Extract a one-line snippet around the first match for `term`. Pure. */
function snippetAround(content, term, radius = 60) {
const haystack = content.toLowerCase();
const idx = haystack.indexOf(term);
if (idx < 0) return '';
const start = Math.max(0, idx - radius);
const end = Math.min(content.length, idx + term.length + radius);
let s = content.slice(start, end).replace(/\s+/g, ' ').trim();
if (start > 0) s = '…' + s;
if (end < content.length) s = s + '…';
return s;
}
/** Count occurrences of `needle` (case-insensitive substring) in `haystack`. */
function countOccurrences(haystack, needle) {
if (!needle) return 0;
const hay = haystack.toLowerCase();
const ndl = needle.toLowerCase();
let count = 0;
let idx = 0;
while ((idx = hay.indexOf(ndl, idx)) !== -1) {
count++;
idx += ndl.length;
}
return count;
}
/** Does the content contain the `#tag`? Looks for word-boundary `#tag`. */
function hasTag(content, tag) {
const re = new RegExp(`(^|\\s)#${escapeRegex(tag)}\\b`, 'i');
return re.test(content);
}
/** Does the content contain `[[wikilink]]` or `[[wikilink|alias]]`? */
function hasWikilink(content, link) {
const re = new RegExp(`\\[\\[${escapeRegex(link)}(\\|[^\\]]+)?\\]\\]`, 'i');
return re.test(content);
}
function escapeRegex(s) {
return String(s).replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
}
/**
* Run a parsed query against one document's content. Returns
* { score, matchedTerms, matchedTags, matchedLinks, snippet } or null if
* nothing matched (so the caller can filter cheaply).
*/
function scoreDocument({ content, parsed, mtimeMs = 0, nowMs = Date.now() }) {
let score = 0;
const matchedTerms = [];
const matchedTags = [];
const matchedLinks = [];
const lower = content.toLowerCase();
for (const term of parsed.terms) {
const n = countOccurrences(content, term);
if (n > 0) {
score += Math.min(n, 5);
matchedTerms.push(term);
}
}
for (const phrase of parsed.phrases) {
if (lower.includes(phrase)) {
score += 3; // phrase hits weigh more
matchedTerms.push(phrase);
}
}
for (const tag of parsed.tags) {
if (hasTag(content, tag)) {
score += 3;
matchedTags.push(tag);
}
}
for (const link of parsed.links) {
if (hasWikilink(content, link)) {
score += 3;
matchedLinks.push(link);
}
}
if (score === 0) return null;
// Recency nudge: edits within the last 7 days
const ageDays = (nowMs - mtimeMs) / (24 * 60 * 60 * 1000);
if (Number.isFinite(ageDays) && ageDays >= 0 && ageDays <= 7) {
score += 0.5 * (1 - ageDays / 7);
}
// Build a snippet around the first matched term (or phrase/tag) for the UI
let snippet = '';
const firstTerm = matchedTerms[0] || (parsed.tags[0] ? `#${parsed.tags[0]}` : null);
if (firstTerm) {
snippet = snippetAround(content, firstTerm);
} else {
snippet = content.slice(0, 120).replace(/\s+/g, ' ').trim() + (content.length > 120 ? '…' : '');
}
return { score, matchedTerms, matchedTags, matchedLinks, snippet };
}
/**
* Search across many files. Files whose content matches the query are
* scored and returned sorted by score.
*
* @param {object} args
* @param {string} args.query raw query string
* @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files
* @param {number} [args.limit=50]
* @returns {Array<{filePath:string, score:number, snippet:string, matchedTerms:string[], matchedTags:string[], matchedLinks:string[]}>}
*/
function search({ query, files, limit = 50, nowMs = Date.now() }) {
const parsed = parseQuery(query);
// Nothing to search for — short-circuit so callers don't pay the file loop
if (
parsed.terms.length === 0 &&
parsed.phrases.length === 0 &&
parsed.tags.length === 0 &&
parsed.links.length === 0
) {
return [];
}
const out = [];
for (const file of files) {
if (!file || typeof file.content !== 'string') continue;
const r = scoreDocument({ content: file.content, parsed, mtimeMs: file.mtimeMs, nowMs });
if (!r) continue;
out.push({ filePath: file.path, ...r });
}
out.sort(
(a, b) => b.score - a.score || (a.filePath < b.filePath ? -1 : a.filePath > b.filePath ? 1 : 0)
);
return out.slice(0, limit);
}
module.exports = {
parseQuery,
snippetAround,
countOccurrences,
hasTag,
hasWikilink,
scoreDocument,
search,
};