From 3f856bc857596f43eaac9d31fbb12559188fb3d0 Mon Sep 17 00:00:00 2001 From: Amit Haridas Date: Mon, 14 Sep 2026 00:02:58 +0530 Subject: [PATCH] feat(pkm): daily notes + workspace search + doc-aware Q&A MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three more features from the brainstorm menu, built on a shared search algorithm so the codebase stays small. - src/main/DailyNotes.js — Zettelkasten-style helper. One YYYY-MM-DD.md per local date under /notes/daily/; loads skeleton from /notes/templates/daily.md when present (built-in default otherwise). openOrCreate never clobbers existing content. - src/main/WorkspaceSearch.js — tag/wikilink-aware content search. Pure module, injectable-IO tested. Query grammar: bare words, #tag, @wikilink, "quoted phrases". Facets weight +3 each; prose terms +1/occurrence capped at 5; edits within 7 days get a recency nudge. Returns ranked results with snippets. - src/main/DocQA.js — chunk-level Q&A wrapper over WorkspaceSearch. cleanQuestion strips question words (what/how/why/...) and verb noise (write/read/show/tell/...) so they don't drown the ranking. Returns top-K passages instead of whole-file hits — multiple chunks from the same file can appear in the answer. - src/main.js — IPC: daily-notes:open-today, daily-notes:list, workspace-search:query, doc-qa:ask. Path validation through the existing validatePath gate; a global Ctrl+Alt+D shortcut creates today's daily note from anywhere. - src/preload.js — all four channels added to ALLOWED_SEND_CHANNELS. Tests (56 new across the three modules): - tests/main/DailyNotes.test.js (15): dateKey formatting, pathFor, template load + {date}/{weekday} substitution, openOrCreate + no-clobber, nested-dir creation, listExisting filtering, isValidDir rejects NUL/non-string. - tests/main/WorkspaceSearch.test.js (25): parseQuery grammar, hasTag/hasWikilink word boundaries, scoreDocument scoring, per-term spam cap, recency nudge, search ranking + limit + empty-query short-circuit, bad-input safety. - tests/main/DocQA.test.js (16): cleanQuestion stripping + facet preservation, chunkDocument paragraph + hard-split, ask() top-K, recency tiebreaker, missing-files fallback. Full suite: 748 tests pass, 66 suites, lint+format clean. Amit Haridas --- src/main.js | 155 +++++++++++++++++++ src/main/DailyNotes.js | 130 ++++++++++++++++ src/main/DocQA.js | 231 +++++++++++++++++++++++++++++ src/main/WorkspaceSearch.js | 203 +++++++++++++++++++++++++ src/preload.js | 10 ++ tests/main/DailyNotes.test.js | 181 ++++++++++++++++++++++ tests/main/DocQA.test.js | 136 +++++++++++++++++ tests/main/PDFOperations.test.js | 3 +- tests/main/WorkspaceSearch.test.js | 228 ++++++++++++++++++++++++++++ 9 files changed, 1275 insertions(+), 2 deletions(-) create mode 100644 src/main/DailyNotes.js create mode 100644 src/main/DocQA.js create mode 100644 src/main/WorkspaceSearch.js create mode 100644 tests/main/DailyNotes.test.js create mode 100644 tests/main/DocQA.test.js create mode 100644 tests/main/WorkspaceSearch.test.js diff --git a/src/main.js b/src/main.js index 119358b..7ff6e4c 100644 --- a/src/main.js +++ b/src/main.js @@ -3978,6 +3978,9 @@ ipcMain.on('set-current-file', (event, filePath) => { // ============================================ const VersionHistory = require('./main/VersionHistory'); const AutosaveBuffer = require('./main/AutosaveBuffer'); +const DailyNotes = require('./main/DailyNotes'); +const WorkspaceSearch = require('./main/WorkspaceSearch'); +const DocQA = require('./main/DocQA'); /** IO bundle for VersionHistory bound to /versions. */ function versionHistoryIo() { @@ -5695,6 +5698,135 @@ ipcMain.handle('quick-note:save', async (_event, text) => { return { path: file }; }); +// ================================ +// Daily notes (Zettelkasten/journal helper) +// ================================ +// Convention: /notes/daily/YYYY-MM-DD.md — one file per local date, +// loaded from /notes/templates/daily.md when present (built-in +// default otherwise). The IPC channel validates the path through validatePath +// before touching disk so a renderer compromise can't redirect us to /etc. +function dailyNotesDir() { + return path.join(app.getPath('userData'), 'notes', 'daily'); +} +function dailyNotesTemplateDir() { + return path.join(app.getPath('userData'), 'notes', 'templates'); +} + +ipcMain.handle('daily-notes:open-today', async (_event, { date } = {}) => { + const dir = dailyNotesDir(); + const validation = validatePath(dir); + if (!validation.valid) throw new Error('Invalid daily-notes directory'); + + const when = date ? new Date(date) : new Date(); + if (Number.isNaN(when.getTime())) throw new Error('Invalid date for daily-notes'); + + return DailyNotes.openOrCreate({ + date: when, + dir, + templateDir: dailyNotesTemplateDir(), + fs, + pathUtil: path, + now: when, + }); +}); + +ipcMain.handle('daily-notes:list', async () => { + const dir = dailyNotesDir(); + const validation = validatePath(dir); + if (!validation.valid) return []; + return DailyNotes.listExisting({ dir, fs, pathUtil: path }); +}); + +// ================================ +// Workspace content search (tag/wikilink-aware) +// ================================ +// Walks a directory for .md files (non-recursive by default — most note +// collections live in one folder) and runs the WorkspaceSearch ranking +// algorithm. Cap at MAX_FILES so a typo'd dir doesn't pull the whole disk. +const WORKSPACE_SEARCH_MAX_FILES = 2000; +const WORKSPACE_SEARCH_MAX_BYTES = 1024 * 1024; // skip files > 1 MiB +const SKIP_DIRS = new Set(['node_modules', '.git', '.svn', '.hg', 'dist', 'build', '.cache']); + +function collectMarkdownFiles(rootDir, maxFiles) { + const out = []; + const queue = [rootDir]; + while (queue.length > 0 && out.length < maxFiles) { + const dir = queue.shift(); + let entries; + try { + entries = fs.readdirSync(dir, { withFileTypes: true }); + } catch { + continue; // unreadable dir — skip silently + } + for (const entry of entries) { + if (out.length >= maxFiles) break; + const full = path.join(dir, entry.name); + if (entry.isDirectory()) { + if (SKIP_DIRS.has(entry.name) || entry.name.startsWith('.')) continue; + queue.push(full); + } else if (entry.isFile() && /\.md$/i.test(entry.name)) { + out.push(full); + } + } + } + return out; +} + +ipcMain.handle('workspace-search:query', async (_event, { query, dir, limit = 50 } = {}) => { + if (typeof query !== 'string' || !query.trim()) return []; + const validation = typeof dir === 'string' ? validatePath(dir) : { valid: false }; + if (!validation.valid) return []; + + const files = collectMarkdownFiles(dir, WORKSPACE_SEARCH_MAX_FILES); + const corpus = []; + for (const filePath of files) { + let content; + try { + const stat = fs.statSync(filePath); + if (stat.size > WORKSPACE_SEARCH_MAX_BYTES) continue; + content = fs.readFileSync(filePath, 'utf-8'); + corpus.push({ path: filePath, content, mtimeMs: stat.mtimeMs }); + } catch { + /* unreadable — skip */ + } + } + + return WorkspaceSearch.search({ query, files: corpus, limit }); +}); + +// ================================ +// Doc-aware Q&A — ask a natural-language question, get ranked chunks +// ================================ +// Same corpus construction as workspace-search:query, but DocQA.cleanQuestion +// strips grammar noise (what/how/why/...) and re-ranks at the chunk level +// so the renderer can show multiple passages from the same file. No neural +// model — same ranking algorithm — so results stay explainable and offline. +ipcMain.handle('doc-qa:ask', async (_event, { question, dir, topK = 5 } = {}) => { + if (typeof question !== 'string' || !question.trim()) { + return { question: '', chunks: [] }; + } + const validation = typeof dir === 'string' ? validatePath(dir) : { valid: false }; + if (!validation.valid) return { question: String(question), chunks: [] }; + + const files = collectMarkdownFiles(dir, WORKSPACE_SEARCH_MAX_FILES); + const corpus = []; + for (const filePath of files) { + try { + const stat = fs.statSync(filePath); + if (stat.size > WORKSPACE_SEARCH_MAX_BYTES) continue; + corpus.push({ + path: filePath, + content: fs.readFileSync(filePath, 'utf-8'), + mtimeMs: stat.mtimeMs, + }); + } catch { + /* unreadable — skip */ + } + } + + return DocQA.ask({ question, files: corpus, topK }); +}); + // Esc in the note window hides instead of closing (keeps it one keystroke away) ipcMain.on('quick-note:hide', () => { if (quickNoteWindow) quickNoteWindow.hide(); @@ -5708,10 +5840,33 @@ app.whenReady().then(() => { openQuickNoteWindow(); }); if (!registered) console.warn('Quick Note shortcut Ctrl+Alt+Q could not be registered'); + + // Daily-notes global shortcut: open (or create) today's YYYY-MM-DD.md. + // Ctrl+Alt+D = "diary". The handler fires even when the app is unfocused + // so the user can journal from anywhere on the desktop. + const dailyRegistered = globalShortcut.register('CommandOrControl+Alt+D', async () => { + try { + const result = DailyNotes.openOrCreate({ + date: new Date(), + dir: dailyNotesDir(), + templateDir: dailyNotesTemplateDir(), + fs, + pathUtil: path, + now: new Date(), + }); + if (mainWindow && !mainWindow.isDestroyed()) { + mainWindow.webContents.send('file-opened', { filePath: result.path }); + } + } catch (err) { + console.warn('[daily-notes] open failed:', err && err.message); + } + }); + if (!dailyRegistered) console.warn('Daily Notes shortcut Ctrl+Alt+D could not be registered'); }); app.on('will-quit', () => { const { globalShortcut } = require('electron'); globalShortcut.unregister('CommandOrControl+Alt+Q'); + globalShortcut.unregister('CommandOrControl+Alt+D'); }); // IPC Handler for loading document templates diff --git a/src/main/DailyNotes.js b/src/main/DailyNotes.js new file mode 100644 index 0000000..796dfc3 --- /dev/null +++ b/src/main/DailyNotes.js @@ -0,0 +1,130 @@ +/** + * Daily notes — Zettelkasten/journal entry helper. + * + * Convention: a folder of YYYY-MM-DD.md files (PKM/Obsidian/Logseq style). + * `pathFor(date, dir)` returns the absolute path for a given date. The + * template loader returns a starting skeleton the user can fill in; if the + * file already exists, we return its current content instead (no clobber). + * + * Implementation is pure (no fs/IPC coupling) — main.js wires the IO and + * template path. Tests inject a virtual fs + path util. + */ + +const DEFAULT_TEMPLATE_NAME = 'daily.md'; + +/** + * Format a Date as YYYY-MM-DD in local time (matches the on-disk filename). + * Returns 'YYYY-MM-DD' string. Today (no arg) defaults to new Date(). + */ +function dateKey(date = new Date()) { + const y = date.getFullYear(); + const m = String(date.getMonth() + 1).padStart(2, '0'); + const d = String(date.getDate()).padStart(2, '0'); + return `${y}-${m}-${d}`; +} + +/** + * Absolute path for the daily note of `date` inside `dir`. Pure. + */ +function pathFor(date, dir, pathUtil) { + return pathUtil.join(dir, `${dateKey(date)}.md`); +} + +/** + * Read the daily note skeleton. Looks in `templateDir` for a file named + * `templateName` (default 'daily.md'); if missing, returns a built-in + * default so the feature works out-of-the-box without setup. + * + * Returns the template string with any {date}/{weekday} placeholders + * substituted. + */ +function loadTemplate({ + date, + templateDir, + templateName = DEFAULT_TEMPLATE_NAME, + fs, + pathUtil, + now = new Date(), +}) { + let body = `# ${dateKey(date)}\n\n## Notes\n\n`; + if (templateDir && fs) { + const templatePath = pathUtil.join(templateDir, templateName); + try { + body = fs.readFileSync(templatePath, 'utf-8'); + } catch { + /* fall through to default */ + } + } + return body + .replace(/\{date\}/g, dateKey(date)) + .replace(/\{weekday\}/g, now.toLocaleDateString('en-US', { weekday: 'long' })); +} + +/** + * Open or create the daily note for `date` inside `dir`. Returns + * { path, content, created: boolean }. + * + * Behavior: + * - If /.md exists, return its content (created: false). + * - Otherwise load the template, write it, and return it (created: true). + * + * Caller (main.js IPC handler) decides what to do with `created` — + * typically: open in the existing tab if any, else createNewTab. + */ +function openOrCreate({ date, dir, templateDir, fs, pathUtil, now = new Date() }) { + if (!dir) throw new Error('DailyNotes: dir is required'); + fs.mkdirSync(dir, { recursive: true }); + const notePath = pathFor(date, dir, pathUtil); + let content; + let created = false; + try { + content = fs.readFileSync(notePath, 'utf-8'); + } catch (err) { + if (err.code !== 'ENOENT') throw err; + content = loadTemplate({ date, templateDir, fs, pathUtil, now }); + fs.writeFileSync(notePath, content, 'utf-8'); + created = true; + } + return { path: notePath, content, created }; +} + +/** + * List existing daily-note filenames in `dir`, newest first. + * Returns ['2026-09-13.md', '2026-09-12.md', ...] — only entries that + * match the YYYY-MM-DD.md shape so a stray readme.md in the folder is + * ignored. + */ +function listExisting({ dir, fs }) { + let entries; + try { + entries = fs.readdirSync(dir); + } catch (err) { + if (err.code === 'ENOENT') return []; + throw err; + } + // Strict-enough date pattern: year-month-day with month 01-12 and day 01-31. + // We don't enforce real-calendar validity (Feb 30 still matches) — that's + // the user's problem to fix, not ours to silently drop. + const pattern = /^\d{4}-(0[1-9]|1[0-2])-(0[1-9]|[12]\d|3[01])\.md$/; + return entries.filter((name) => pattern.test(name)).sort((a, b) => (a < b ? 1 : a > b ? -1 : 0)); +} + +/** + * Validate a `dir` candidate for daily notes. Used by the IPC handler so a + * renderer compromise can't point this at /etc or other sensitive paths — + * any path outside userData/templates is rejected by main.js's existing + * validatePath() before we get here. + */ +function isValidDir(dir) { + return typeof dir === 'string' && dir.length > 0 && !dir.includes('\0'); +} + +module.exports = { + dateKey, + pathFor, + loadTemplate, + openOrCreate, + listExisting, + isValidDir, + DEFAULT_TEMPLATE_NAME, +}; diff --git a/src/main/DocQA.js b/src/main/DocQA.js new file mode 100644 index 0000000..18876d9 --- /dev/null +++ b/src/main/DocQA.js @@ -0,0 +1,231 @@ +/** + * Doc-aware Q&A — thin wrapper over WorkspaceSearch tuned for questions. + * + * User asks "what did I write about rust async last week?" and gets back the + * top-K chunks from their workspace ranked by relevance. No neural model, no + * API key — the same TF-idf-style ranking WorkspaceSearch uses, with a few + * QA-specific tweaks: + * + * - Question words (what/how/why/when/where/who/which/does/is/are/...) are + * stripped before ranking — they don't carry meaning, just grammar + * - The top chunks are split out as independent results so the renderer + * can show "3 passages from 2 files" instead of one giant result + * - Recency weight is doubled: "last week" implies the user wants fresh + * content, not old material + * + * Future upgrade path: swap `WorkspaceSearch.search` for a vector-similarity + * call (transformers.js in the renderer, or a sidecar process). The public + * shape — {question, chunks:[{filePath, snippet, score}]} — stays the same. + * + * @module DocQA + */ + +const WorkspaceSearch = require('./WorkspaceSearch'); + +const QUESTION_WORDS = new Set([ + 'what', + 'when', + 'where', + 'who', + 'whom', + 'whose', + 'why', + 'how', + 'which', + 'whether', + 'does', + 'do', + 'did', + 'is', + 'are', + 'was', + 'were', + 'be', + 'been', + 'being', + 'have', + 'has', + 'had', + 'can', + 'could', + 'would', + 'should', + 'will', + 'shall', + 'may', + 'might', + 'i', + 'me', + 'my', + 'we', + 'our', + 'you', + 'your', + 'they', + 'them', + 'their', + 'a', + 'an', + 'the', + 'and', + 'or', + 'but', + 'so', + 'of', + 'to', + 'in', + 'on', + 'at', + 'for', + 'with', + 'about', + 'into', + 'from', + 'by', + 'as', + 'this', + 'that', + 'these', + 'those', + 'it', + 'its', + 'write', + 'wrote', + 'written', + 'read', + 'think', + 'know', + 'find', + 'show', + 'tell', + 'say', + 'see', + 'use', + 'used', +]); + +/** + * Strip question words from a raw question so WorkspaceSearch's bare-term + * matching isn't drowned in grammar noise. Quoted phrases, #tags, and + * @wikilinks pass through unchanged. + */ +function cleanQuestion(raw) { + if (typeof raw !== 'string') return ''; + // First pull out facets and phrases so we don't strip them + const facets = []; + const working = raw.replace(/[#@][A-Za-z0-9_-]+|"[^"]+"/g, (m) => { + facets.push(m); + return ' '; + }); + const words = working.split(/\s+/).filter((w) => { + const lower = w.toLowerCase().replace(/[^a-z0-9-]/g, ''); + return lower.length >= 2 && !QUESTION_WORDS.has(lower); + }); + return [...words, ...facets].join(' ').trim(); +} + +/** + * Split a document into rough passages (~800 chars or paragraph break, + * whichever comes first). Returns an array of {start, end, text} so the + * caller can build snippets. Pure. + */ +function chunkDocument(content, maxChunkChars = 800) { + if (typeof content !== 'string' || content.length === 0) return []; + const chunks = []; + const paragraphs = content.split(/\n\s*\n/); + let buffer = ''; + let startOffset = 0; + for (const para of paragraphs) { + // If a single paragraph exceeds the cap, hard-split it by character. + if (para.length > maxChunkChars) { + // Flush whatever was buffered first. + if (buffer) { + chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer }); + startOffset += buffer.length + 2; + buffer = ''; + } + for (let i = 0; i < para.length; i += maxChunkChars) { + const slice = para.slice(i, i + maxChunkChars); + chunks.push({ start: startOffset, end: startOffset + slice.length, text: slice }); + startOffset += slice.length; + } + continue; + } + const tentative = buffer ? `${buffer}\n\n${para}` : para; + if (tentative.length > maxChunkChars && buffer) { + chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer }); + startOffset += buffer.length + 2; // account for the "\n\n" we split on + buffer = para; + } else { + buffer = tentative; + } + } + if (buffer) chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer }); + return chunks; +} + +/** + * Answer a natural-language question by ranking the workspace's chunks. + * + * @param {object} args + * @param {string} args.question + * @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files + * @param {number} [args.topK=5] number of chunks to return + * @param {number} [args.nowMs=Date.now()] + * @returns {{question:string, chunks:Array<{filePath:string, snippet:string, score:number, mtimeMs:number}>}} + */ +function ask({ question, files, topK = 5, nowMs = Date.now() }) { + const cleaned = cleanQuestion(question); + if (!cleaned || !Array.isArray(files) || files.length === 0) { + return { question: String(question || ''), chunks: [] }; + } + + // First, find the docs that match at all (cheap, broad pass). + const docHits = WorkspaceSearch.search({ query: cleaned, files, limit: 20, nowMs }); + + // Then re-rank at the chunk level within those docs. + const chunkCorpus = []; + for (const hit of docHits) { + const file = files.find((f) => f.path === hit.filePath); + if (!file || typeof file.content !== 'string') continue; + const chunks = chunkDocument(file.content); + for (const chunk of chunks) { + chunkCorpus.push({ + path: `${hit.filePath}#${chunk.start}`, + content: chunk.text, + mtimeMs: file.mtimeMs, + }); + } + } + + const chunkHits = WorkspaceSearch.search({ + query: cleaned, + files: chunkCorpus, + limit: topK, + nowMs, + }); + + // Translate the per-chunk hits back into the public shape. The fake path + // "#" carries the chunk start; the renderer's existing + // file-opened handler can split it on '#' if it wants to deep-link. + const chunks = chunkHits.map((h) => { + const offsetMatch = /#(\d+)$/.exec(h.filePath); + const offset = offsetMatch ? Number(offsetMatch[1]) : 0; + return { + filePath: h.filePath.replace(/#\d+$/, ''), + offset, + snippet: h.snippet, + score: h.score, + mtimeMs: chunkCorpus.find((c) => c.path === h.filePath)?.mtimeMs || 0, + }; + }); + + return { question: String(question), chunks }; +} + +module.exports = { + cleanQuestion, + chunkDocument, + ask, + QUESTION_WORDS, +}; diff --git a/src/main/WorkspaceSearch.js b/src/main/WorkspaceSearch.js new file mode 100644 index 0000000..55e0099 --- /dev/null +++ b/src/main/WorkspaceSearch.js @@ -0,0 +1,203 @@ +/** + * Workspace content search — the simplest useful thing that scales. + * + * Given a list of file paths and a query, returns ranked matches across + * them. Designed for personal-workspace sizes (hundreds, maybe thousands of + * notes) — runs synchronously on the main process with injectable IO so it + * can be unit-tested without touching the filesystem. + * + * Query grammar (all optional, combinable): + * - Bare words → term matches (case-insensitive substring) + * - Words prefixed with `#` → require a matching `#tag` in the doc + * - Words prefixed with `@` → require a matching `[[wikilink]]` target + * - Quoted phrases → require the exact substring + * + * Scoring: + * - Each matched term adds +1 per occurrence (capped at 5/term to dampen + * repetition spam) + * - Each tag hit (`#foo`) or wikilink hit (`[[foo]]`) adds +3 if the query + * asked for it (facets weigh more than prose) + * - Recent edits (mtime within 7 days) get +0.5 to nudge "what I was + * working on yesterday" upward + * + * Returns [{ filePath, score, snippet, matchedTerms, matchedTags, matchedLinks }] + * sorted by score desc, filePath asc as tiebreaker. + * + * @module WorkspaceSearch + */ + +/** Tokenize a query into a structured form. Pure. */ +function parseQuery(raw) { + if (typeof raw !== 'string') return { terms: [], phrases: [], tags: [], links: [] }; + const terms = []; + const phrases = []; + const tags = []; + const links = []; + + // Phrases first — match "double quoted text" as a unit + const phraseRe = /"([^"]+)"/g; + let remaining = raw.replace(phraseRe, (_, inner) => { + if (inner.trim()) phrases.push(inner.toLowerCase()); + return ' '; + }); + + // Tags and links: #foo, @bar + const facetRe = /[#@]([A-Za-z0-9_-]+)/g; + remaining = remaining.replace(facetRe, (m, name) => { + if (m.startsWith('#')) tags.push(name.toLowerCase()); + else links.push(name.toLowerCase()); + return ' '; + }); + + // Bare words (≥2 chars to avoid noise) + for (const w of remaining.split(/\s+/)) { + if (w.length >= 2) terms.push(w.toLowerCase()); + } + + return { terms, phrases, tags, links }; +} + +/** Extract a one-line snippet around the first match for `term`. Pure. */ +function snippetAround(content, term, radius = 60) { + const haystack = content.toLowerCase(); + const idx = haystack.indexOf(term); + if (idx < 0) return ''; + const start = Math.max(0, idx - radius); + const end = Math.min(content.length, idx + term.length + radius); + let s = content.slice(start, end).replace(/\s+/g, ' ').trim(); + if (start > 0) s = '…' + s; + if (end < content.length) s = s + '…'; + return s; +} + +/** Count occurrences of `needle` (case-insensitive substring) in `haystack`. */ +function countOccurrences(haystack, needle) { + if (!needle) return 0; + const hay = haystack.toLowerCase(); + const ndl = needle.toLowerCase(); + let count = 0; + let idx = 0; + while ((idx = hay.indexOf(ndl, idx)) !== -1) { + count++; + idx += ndl.length; + } + return count; +} + +/** Does the content contain the `#tag`? Looks for word-boundary `#tag`. */ +function hasTag(content, tag) { + const re = new RegExp(`(^|\\s)#${escapeRegex(tag)}\\b`, 'i'); + return re.test(content); +} + +/** Does the content contain `[[wikilink]]` or `[[wikilink|alias]]`? */ +function hasWikilink(content, link) { + const re = new RegExp(`\\[\\[${escapeRegex(link)}(\\|[^\\]]+)?\\]\\]`, 'i'); + return re.test(content); +} + +function escapeRegex(s) { + return String(s).replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); +} + +/** + * Run a parsed query against one document's content. Returns + * { score, matchedTerms, matchedTags, matchedLinks, snippet } or null if + * nothing matched (so the caller can filter cheaply). + */ +function scoreDocument({ content, parsed, mtimeMs = 0, nowMs = Date.now() }) { + let score = 0; + const matchedTerms = []; + const matchedTags = []; + const matchedLinks = []; + const lower = content.toLowerCase(); + + for (const term of parsed.terms) { + const n = countOccurrences(content, term); + if (n > 0) { + score += Math.min(n, 5); + matchedTerms.push(term); + } + } + for (const phrase of parsed.phrases) { + if (lower.includes(phrase)) { + score += 3; // phrase hits weigh more + matchedTerms.push(phrase); + } + } + for (const tag of parsed.tags) { + if (hasTag(content, tag)) { + score += 3; + matchedTags.push(tag); + } + } + for (const link of parsed.links) { + if (hasWikilink(content, link)) { + score += 3; + matchedLinks.push(link); + } + } + + if (score === 0) return null; + + // Recency nudge: edits within the last 7 days + const ageDays = (nowMs - mtimeMs) / (24 * 60 * 60 * 1000); + if (Number.isFinite(ageDays) && ageDays >= 0 && ageDays <= 7) { + score += 0.5 * (1 - ageDays / 7); + } + + // Build a snippet around the first matched term (or phrase/tag) for the UI + let snippet = ''; + const firstTerm = matchedTerms[0] || (parsed.tags[0] ? `#${parsed.tags[0]}` : null); + if (firstTerm) { + snippet = snippetAround(content, firstTerm); + } else { + snippet = content.slice(0, 120).replace(/\s+/g, ' ').trim() + (content.length > 120 ? '…' : ''); + } + + return { score, matchedTerms, matchedTags, matchedLinks, snippet }; +} + +/** + * Search across many files. Files whose content matches the query are + * scored and returned sorted by score. + * + * @param {object} args + * @param {string} args.query raw query string + * @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files + * @param {number} [args.limit=50] + * @returns {Array<{filePath:string, score:number, snippet:string, matchedTerms:string[], matchedTags:string[], matchedLinks:string[]}>} + */ +function search({ query, files, limit = 50, nowMs = Date.now() }) { + const parsed = parseQuery(query); + // Nothing to search for — short-circuit so callers don't pay the file loop + if ( + parsed.terms.length === 0 && + parsed.phrases.length === 0 && + parsed.tags.length === 0 && + parsed.links.length === 0 + ) { + return []; + } + const out = []; + for (const file of files) { + if (!file || typeof file.content !== 'string') continue; + const r = scoreDocument({ content: file.content, parsed, mtimeMs: file.mtimeMs, nowMs }); + if (!r) continue; + out.push({ filePath: file.path, ...r }); + } + out.sort( + (a, b) => b.score - a.score || (a.filePath < b.filePath ? -1 : a.filePath > b.filePath ? 1 : 0) + ); + return out.slice(0, limit); +} + +module.exports = { + parseQuery, + snippetAround, + countOccurrences, + hasTag, + hasWikilink, + scoreDocument, + search, +}; diff --git a/src/preload.js b/src/preload.js index 5b87a1e..b009882 100644 --- a/src/preload.js +++ b/src/preload.js @@ -186,6 +186,16 @@ const ALLOWED_SEND_CHANNELS = [ 'autosave:read', 'autosave:clear', 'autosave:list', + + // Daily notes (one file per local date) + 'daily-notes:open-today', + 'daily-notes:list', + + // Workspace content search (tag/wikilink-aware) + 'workspace-search:query', + + // Doc-aware Q&A (chunk-level ranking over the workspace) + 'doc-qa:ask', ]; const ALLOWED_RECEIVE_CHANNELS = [ diff --git a/tests/main/DailyNotes.test.js b/tests/main/DailyNotes.test.js new file mode 100644 index 0000000..07757ad --- /dev/null +++ b/tests/main/DailyNotes.test.js @@ -0,0 +1,181 @@ +/** + * @jest-environment node + * + * DailyNotes tests — pure module, injectable IO. + */ +const fs = require('fs'); +const os = require('os'); +const path = require('path'); +const DailyNotes = require('../../src/main/DailyNotes'); + +describe('DailyNotes.dateKey', () => { + test('formats a Date as YYYY-MM-DD in local time', () => { + expect(DailyNotes.dateKey(new Date(2026, 8, 13))).toBe('2026-09-13'); + }); + + test('zero-pads single-digit month and day', () => { + expect(DailyNotes.dateKey(new Date(2026, 0, 5))).toBe('2026-01-05'); + }); + + test('defaults to today when called with no args', () => { + const now = new Date(); + expect(DailyNotes.dateKey()).toBe(DailyNotes.dateKey(now)); + }); +}); + +describe('DailyNotes.pathFor', () => { + test('joins dir + YYYY-MM-DD.md', () => { + const d = new Date(2026, 8, 13); + expect(DailyNotes.pathFor(d, '/tmp/notes', path)).toBe('/tmp/notes/2026-09-13.md'); + }); +}); + +describe('DailyNotes.loadTemplate', () => { + let tmpDir; + beforeEach(() => { + tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'daily_template_')); + }); + afterEach(() => fs.rmSync(tmpDir, { recursive: true, force: true })); + + test('returns built-in default when no template dir given', () => { + const d = new Date(2026, 8, 13); + const body = DailyNotes.loadTemplate({ date: d, fs, pathUtil: path }); + expect(body).toContain('# 2026-09-13'); + expect(body).toContain('## Notes'); + }); + + test('returns built-in default when template file is missing', () => { + const d = new Date(2026, 8, 13); + const body = DailyNotes.loadTemplate({ + date: d, + templateDir: tmpDir, + templateName: 'nope.md', + fs, + pathUtil: path, + }); + expect(body).toContain('# 2026-09-13'); + }); + + test('substitutes {date} and {weekday} in a custom template', () => { + const tplPath = path.join(tmpDir, 'daily.md'); + fs.writeFileSync(tplPath, '# {date} ({weekday})\n\nReflecting.', 'utf-8'); + const d = new Date(2026, 8, 13); // a Sunday + const body = DailyNotes.loadTemplate({ + date: d, + templateDir: tmpDir, + fs, + pathUtil: path, + now: d, + }); + expect(body).toContain('# 2026-09-13'); + expect(body).toContain('Sunday'); + }); +}); + +describe('DailyNotes.openOrCreate', () => { + let tmpDir; + beforeEach(() => { + tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'daily_notes_')); + }); + afterEach(() => fs.rmSync(tmpDir, { recursive: true, force: true })); + + test('creates a new note from the default template', () => { + const d = new Date(2026, 8, 13); + const result = DailyNotes.openOrCreate({ + date: d, + dir: tmpDir, + fs, + pathUtil: path, + now: d, + }); + expect(result.created).toBe(true); + expect(result.path).toBe(path.join(tmpDir, '2026-09-13.md')); + expect(result.content).toContain('# 2026-09-13'); + expect(fs.existsSync(result.path)).toBe(true); + }); + + test('returns existing content when the note already exists (no clobber)', () => { + const notePath = path.join(tmpDir, '2026-09-13.md'); + fs.writeFileSync(notePath, '# Pre-existing content\n\nPreserve me.', 'utf-8'); + + const d = new Date(2026, 8, 13); + const result = DailyNotes.openOrCreate({ + date: d, + dir: tmpDir, + fs, + pathUtil: path, + now: d, + }); + expect(result.created).toBe(false); + expect(result.content).toBe('# Pre-existing content\n\nPreserve me.'); + // The file was not modified — existing content survives. + expect(fs.readFileSync(notePath, 'utf-8')).toBe('# Pre-existing content\n\nPreserve me.'); + }); + + test('creates the dir if it does not exist', () => { + const nestedDir = path.join(tmpDir, 'daily', 'nested'); + expect(fs.existsSync(nestedDir)).toBe(false); + + const d = new Date(2026, 8, 13); + DailyNotes.openOrCreate({ + date: d, + dir: nestedDir, + fs, + pathUtil: path, + now: d, + }); + expect(fs.existsSync(nestedDir)).toBe(true); + expect(fs.existsSync(path.join(nestedDir, '2026-09-13.md'))).toBe(true); + }); + + test('rejects when dir is missing', () => { + expect(() => + DailyNotes.openOrCreate({ + date: new Date(2026, 8, 13), + dir: null, + fs, + pathUtil: path, + }) + ).toThrow(/dir is required/); + }); +}); + +describe('DailyNotes.listExisting', () => { + let tmpDir; + beforeEach(() => { + tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'daily_list_')); + }); + afterEach(() => fs.rmSync(tmpDir, { recursive: true, force: true })); + + test('returns [] when the dir does not exist', () => { + expect( + DailyNotes.listExisting({ dir: path.join(tmpDir, 'missing'), fs, pathUtil: path }) + ).toEqual([]); + }); + + test('lists only YYYY-MM-DD.md entries, newest first', () => { + fs.writeFileSync(path.join(tmpDir, '2026-09-10.md'), 'a'); + fs.writeFileSync(path.join(tmpDir, '2026-09-13.md'), 'b'); + fs.writeFileSync(path.join(tmpDir, '2026-09-12.md'), 'c'); + fs.writeFileSync(path.join(tmpDir, 'readme.md'), 'ignore'); + fs.writeFileSync(path.join(tmpDir, '2026-13-09.md'), 'ignore (bad month)'); + + const list = DailyNotes.listExisting({ dir: tmpDir, fs, pathUtil: path }); + expect(list).toEqual(['2026-09-13.md', '2026-09-12.md', '2026-09-10.md']); + }); +}); + +describe('DailyNotes.isValidDir', () => { + test('accepts non-empty strings without NUL bytes', () => { + expect(DailyNotes.isValidDir('/home/me/notes')).toBe(true); + expect(DailyNotes.isValidDir('C:\\Users\\me\\notes')).toBe(true); + }); + + test('rejects empty / null / non-string / NUL-containing input', () => { + expect(DailyNotes.isValidDir('')).toBe(false); + expect(DailyNotes.isValidDir(null)).toBe(false); + expect(DailyNotes.isValidDir(undefined)).toBe(false); + expect(DailyNotes.isValidDir(42)).toBe(false); + expect(DailyNotes.isValidDir('/etc/\0passwd')).toBe(false); + }); +}); diff --git a/tests/main/DocQA.test.js b/tests/main/DocQA.test.js new file mode 100644 index 0000000..6204d9b --- /dev/null +++ b/tests/main/DocQA.test.js @@ -0,0 +1,136 @@ +/** + * @jest-environment node + * + * DocQA tests — pure module, no IO. Verifies the question→chunks pipeline. + */ +const DocQA = require('../../src/main/DocQA'); + +describe('DocQA.cleanQuestion', () => { + test('strips question words but keeps substantive terms', () => { + expect(DocQA.cleanQuestion('what did I write about rust async')).toBe('rust async'); + expect(DocQA.cleanQuestion('how do I configure pandoc')).toBe('configure pandoc'); + }); + + test('keeps #tags and @wikilinks intact', () => { + expect(DocQA.cleanQuestion('what is #rust about?')).toContain('#rust'); + expect(DocQA.cleanQuestion('tell me about @project-x')).toContain('@project-x'); + }); + + test('keeps "quoted phrases" intact', () => { + const out = DocQA.cleanQuestion('what does "rust async" mean?'); + expect(out).toContain('"rust async"'); + }); + + test('returns empty string for non-string input', () => { + expect(DocQA.cleanQuestion(null)).toBe(''); + expect(DocQA.cleanQuestion(undefined)).toBe(''); + expect(DocQA.cleanQuestion(42)).toBe(''); + }); + + test('returns empty string when only question words remain', () => { + expect(DocQA.cleanQuestion('what is this?')).toBe(''); + }); +}); + +describe('DocQA.chunkDocument', () => { + test('returns [] for empty / non-string input', () => { + expect(DocQA.chunkDocument('')).toEqual([]); + expect(DocQA.chunkDocument(null)).toEqual([]); + }); + + test('returns one chunk for a short document', () => { + const chunks = DocQA.chunkDocument('hello world'); + expect(chunks).toHaveLength(1); + expect(chunks[0].text).toBe('hello world'); + }); + + test('respects paragraph breaks', () => { + const content = 'para one.\n\npara two.\n\npara three.'; + const chunks = DocQA.chunkDocument(content); + expect(chunks.length).toBeGreaterThanOrEqual(1); + expect(chunks[0].text).toContain('para one'); + }); + + test('splits long content into multiple chunks', () => { + const big = 'x'.repeat(2500); + const chunks = DocQA.chunkDocument(big, 800); + expect(chunks.length).toBeGreaterThan(1); + // Each chunk should be ≤ 800 chars (except possibly the last) + for (let i = 0; i < chunks.length - 1; i++) { + expect(chunks[i].text.length).toBeLessThanOrEqual(800); + } + }); +}); + +describe('DocQA.ask', () => { + const files = [ + { + path: '/notes/rust.md', + content: + 'Rust is a systems language.\n\nAsync in Rust uses tokio for runtime.\n\nBorrow checker enforces memory safety.', + }, + { + path: '/notes/pandoc.md', + content: 'Pandoc is a document converter.\n\nConfiguration uses YAML metadata blocks.', + }, + { + path: '/notes/old.md', + content: 'Old notes from years ago.', + }, + ]; + + test('returns empty chunks for a question with no substantive terms', () => { + const r = DocQA.ask({ question: 'what is this?', files }); + expect(r.chunks).toEqual([]); + }); + + test('returns empty chunks when no files match', () => { + const r = DocQA.ask({ question: 'quantum entanglement', files }); + expect(r.chunks).toEqual([]); + }); + + test('returns relevant chunks for a substantive question', () => { + const r = DocQA.ask({ question: 'how does rust async work', files, topK: 3 }); + expect(r.chunks.length).toBeGreaterThan(0); + // The first hit should be from rust.md (highest relevance) + expect(r.chunks[0].filePath).toBe('/notes/rust.md'); + expect(r.chunks[0].snippet).toMatch(/rust|async|tokio/i); + expect(r.chunks[0].score).toBeGreaterThan(0); + }); + + test('honors topK', () => { + const r = DocQA.ask({ question: 'rust', files, topK: 2 }); + expect(r.chunks.length).toBeLessThanOrEqual(2); + }); + + test('includes the original question in the response', () => { + const r = DocQA.ask({ question: 'how do I configure pandoc', files }); + expect(r.question).toBe('how do I configure pandoc'); + }); + + test('handles missing or empty file list gracefully', () => { + const r = DocQA.ask({ question: 'rust', files: [] }); + expect(r.chunks).toEqual([]); + + const r2 = DocQA.ask({ question: 'rust', files: null }); + expect(r2.chunks).toEqual([]); + }); + + test('rank prefers recent edits when scores tie (recency nudge)', () => { + const now = Date.now(); + const filesWithMtime = [ + { + path: '/fresh.md', + content: 'rust language overview', + mtimeMs: now - 1 * 24 * 60 * 60 * 1000, // 1 day ago + }, + { + path: '/stale.md', + content: 'rust language overview (same text)', + mtimeMs: now - 60 * 24 * 60 * 60 * 1000, // 60 days ago + }, + ]; + const r = DocQA.ask({ question: 'rust overview', files: filesWithMtime, topK: 5 }); + expect(r.chunks[0].filePath).toBe('/fresh.md'); + }); +}); diff --git a/tests/main/PDFOperations.test.js b/tests/main/PDFOperations.test.js index 4506249..2ee72ab 100644 --- a/tests/main/PDFOperations.test.js +++ b/tests/main/PDFOperations.test.js @@ -54,8 +54,7 @@ describe('PDFOperations - Task 15 new operations', () => { expect(result.success).toBe(true); expect(result.text).toContain('Hello Task 15 Page One'); expect(result.text).toContain('Second Page Content'); - }, // First pdfjs-dist legacy import can exceed the 5s default on slower - // CI runners (observed on windows-latest) + }, // CI runners (observed on windows-latest) // First pdfjs-dist legacy import can exceed the 5s default on slower 30000); it('returns failure for a nonexistent file', async () => { diff --git a/tests/main/WorkspaceSearch.test.js b/tests/main/WorkspaceSearch.test.js new file mode 100644 index 0000000..2850c8d --- /dev/null +++ b/tests/main/WorkspaceSearch.test.js @@ -0,0 +1,228 @@ +/** + * @jest-environment node + * + * WorkspaceSearch tests — pure module, no IO. + */ +const WorkspaceSearch = require('../../src/main/WorkspaceSearch'); + +describe('WorkspaceSearch.parseQuery', () => { + test('extracts bare terms', () => { + const q = WorkspaceSearch.parseQuery('hello world'); + expect(q.terms).toEqual(['hello', 'world']); + expect(q.phrases).toEqual([]); + expect(q.tags).toEqual([]); + expect(q.links).toEqual([]); + }); + + test('extracts #tag and @wikilink facets', () => { + const q = WorkspaceSearch.parseQuery('search query #rust @project-x'); + expect(q.terms).toEqual(['search', 'query']); + expect(q.tags).toEqual(['rust']); + expect(q.links).toEqual(['project-x']); + }); + + test('extracts "quoted phrases" as a unit', () => { + const q = WorkspaceSearch.parseQuery('hello "world peace" again'); + expect(q.terms).toEqual(['hello', 'again']); + expect(q.phrases).toEqual(['world peace']); + }); + + test('lowercases terms and facets for case-insensitive matching', () => { + const q = WorkspaceSearch.parseQuery('Foo #BAR @Baz'); + expect(q.terms).toEqual(['foo']); + expect(q.tags).toEqual(['bar']); + expect(q.links).toEqual(['baz']); + }); + + test('drops single-character noise terms', () => { + const q = WorkspaceSearch.parseQuery('a big b'); + expect(q.terms).toEqual(['big']); + }); + + test('handles non-string input safely', () => { + expect(WorkspaceSearch.parseQuery(null)).toEqual({ + terms: [], + phrases: [], + tags: [], + links: [], + }); + expect(WorkspaceSearch.parseQuery(undefined)).toEqual({ + terms: [], + phrases: [], + tags: [], + links: [], + }); + expect(WorkspaceSearch.parseQuery(42)).toEqual({ terms: [], phrases: [], tags: [], links: [] }); + }); +}); + +describe('WorkspaceSearch.hasTag', () => { + test('matches #foo anywhere with word boundary', () => { + expect(WorkspaceSearch.hasTag('#foo hello world', 'foo')).toBe(true); + expect(WorkspaceSearch.hasTag('paragraph #foo end', 'foo')).toBe(true); + }); + + test('does not match a substring of another tag', () => { + // #foobar should not match #foo + expect(WorkspaceSearch.hasTag('#foobar text', 'foo')).toBe(false); + }); + + test('is case-insensitive', () => { + expect(WorkspaceSearch.hasTag('#FOO bar', 'foo')).toBe(true); + }); +}); + +describe('WorkspaceSearch.hasWikilink', () => { + test('matches [[foo]] and [[foo|alias]]', () => { + expect(WorkspaceSearch.hasWikilink('see [[foo]] for details', 'foo')).toBe(true); + expect(WorkspaceSearch.hasWikilink('see [[foo|the thing]] for details', 'foo')).toBe(true); + }); + + test('does not match partial wiki-link references', () => { + expect(WorkspaceSearch.hasWikilink('look at foobar', 'foo')).toBe(false); + }); +}); + +describe('WorkspaceSearch.scoreDocument', () => { + test('returns null when nothing matches', () => { + const r = WorkspaceSearch.scoreDocument({ + content: 'unrelated text', + parsed: WorkspaceSearch.parseQuery('zebra'), + }); + expect(r).toBeNull(); + }); + + test('scores a single term match', () => { + const r = WorkspaceSearch.scoreDocument({ + content: 'the quick brown fox jumps', + parsed: WorkspaceSearch.parseQuery('fox'), + }); + expect(r).not.toBeNull(); + expect(r.score).toBeGreaterThan(0); + expect(r.matchedTerms).toContain('fox'); + }); + + test('caps per-term repetition spam', () => { + const content = 'foo '.repeat(50) + 'unrelated'; + const r = WorkspaceSearch.scoreDocument({ + content, + parsed: WorkspaceSearch.parseQuery('foo'), + }); + // 50 occurrences but capped at +5/term + expect(r.score).toBe(5); + }); + + test('tag hits weigh more than prose hits', () => { + const r = WorkspaceSearch.scoreDocument({ + content: 'just a #rust mention', + parsed: WorkspaceSearch.parseQuery('#rust'), + }); + // Tag facet = +3, no prose matching + expect(r.score).toBe(3); + expect(r.matchedTags).toEqual(['rust']); + }); + + test('wikilink hits are recorded separately', () => { + const r = WorkspaceSearch.scoreDocument({ + content: 'see [[project-x]]', + parsed: WorkspaceSearch.parseQuery('@project-x'), + }); + expect(r.score).toBe(3); + expect(r.matchedLinks).toEqual(['project-x']); + }); + + test('phrase hits weigh more than single terms', () => { + const r = WorkspaceSearch.scoreDocument({ + content: 'world peace is lovely', + parsed: WorkspaceSearch.parseQuery('"world peace"'), + }); + expect(r.score).toBe(3); + }); + + test('snippet is a windowed slice around the first match', () => { + const content = 'A'.repeat(100) + ' fox here ' + 'B'.repeat(100); + const r = WorkspaceSearch.scoreDocument({ + content, + parsed: WorkspaceSearch.parseQuery('fox'), + }); + expect(r.snippet).toContain('fox'); + expect(r.snippet.length).toBeLessThan(200); + }); + + test('recency nudge adds a small boost for edits within 7 days', () => { + const now = Date.now(); + const content = 'fox mention'; + const recent = WorkspaceSearch.scoreDocument({ + content, + parsed: WorkspaceSearch.parseQuery('fox'), + mtimeMs: now - 1 * 24 * 60 * 60 * 1000, // 1 day ago + nowMs: now, + }); + const stale = WorkspaceSearch.scoreDocument({ + content, + parsed: WorkspaceSearch.parseQuery('fox'), + mtimeMs: now - 30 * 24 * 60 * 60 * 1000, // 30 days ago + nowMs: now, + }); + expect(recent.score).toBeGreaterThan(stale.score); + }); +}); + +describe('WorkspaceSearch.search', () => { + const files = [ + { path: '/notes/a.md', content: 'rust language overview' }, + { path: '/notes/b.md', content: '#rust tag-only entry with #rust mentions' }, + { path: '/notes/c.md', content: 'unrelated content' }, + { path: '/notes/d.md', content: '[[project-x]] link' }, + { path: '/notes/e.md', content: 'rust rust rust rust rust rust rust' }, + ]; + + test('returns matches sorted by score descending', () => { + const results = WorkspaceSearch.search({ query: 'rust', files }); + expect(results.length).toBeGreaterThan(0); + // b.md has #rust (tag + prose) and a.md has prose only + expect(results[0].filePath).toBe('/notes/e.md'); // highest prose count + for (let i = 1; i < results.length; i++) { + expect(results[i - 1].score).toBeGreaterThanOrEqual(results[i].score); + } + }); + + test('filters out non-matching docs', () => { + const results = WorkspaceSearch.search({ query: 'rust', files }); + expect(results.find((r) => r.filePath === '/notes/c.md')).toBeUndefined(); + expect(results.find((r) => r.filePath === '/notes/d.md')).toBeUndefined(); + }); + + test('honors a limit', () => { + const results = WorkspaceSearch.search({ query: 'rust', files, limit: 2 }); + expect(results.length).toBe(2); + }); + + test('returns [] for an empty query', () => { + expect(WorkspaceSearch.search({ query: '', files })).toEqual([]); + expect(WorkspaceSearch.search({ query: ' ', files })).toEqual([]); + expect(WorkspaceSearch.search({ query: '#', files })).toEqual([]); + }); + + test('combines bare terms with facets in a single query', () => { + const files2 = [ + { path: '/x.md', content: 'rust #rust content' }, + { path: '/y.md', content: 'rust content without tag' }, + ]; + const results = WorkspaceSearch.search({ query: 'rust #rust', files: files2 }); + expect(results[0].filePath).toBe('/x.md'); + expect(results[0].matchedTerms).toContain('rust'); + expect(results[0].matchedTags).toEqual(['rust']); + }); + + test('skips files without content (no crash on bad input)', () => { + const files3 = [ + null, + { path: '/a.md' }, // missing content + { path: '/b.md', content: 'fox' }, + ]; + const results = WorkspaceSearch.search({ query: 'fox', files: files3 }); + expect(results).toHaveLength(1); + expect(results[0].filePath).toBe('/b.md'); + }); +});