mirror of
https://github.com/amitwh/markdown-converter.git
synced 2026-10-01 09:19:34 +05:30
feat(pkm): daily notes + workspace search + doc-aware Q&A
Three more features from the brainstorm menu, built on a shared search
algorithm so the codebase stays small.
- src/main/DailyNotes.js — Zettelkasten-style helper. One YYYY-MM-DD.md
per local date under <userData>/notes/daily/; loads skeleton from
<userData>/notes/templates/daily.md when present (built-in default
otherwise). openOrCreate never clobbers existing content.
- src/main/WorkspaceSearch.js — tag/wikilink-aware content search.
Pure module, injectable-IO tested. Query grammar: bare words,
#tag, @wikilink, "quoted phrases". Facets weight +3 each; prose
terms +1/occurrence capped at 5; edits within 7 days get a recency
nudge. Returns ranked results with snippets.
- src/main/DocQA.js — chunk-level Q&A wrapper over WorkspaceSearch.
cleanQuestion strips question words (what/how/why/...) and verb
noise (write/read/show/tell/...) so they don't drown the ranking.
Returns top-K passages instead of whole-file hits — multiple chunks
from the same file can appear in the answer.
- src/main.js — IPC: daily-notes:open-today, daily-notes:list,
workspace-search:query, doc-qa:ask. Path validation through the
existing validatePath gate; a global Ctrl+Alt+D shortcut creates
today's daily note from anywhere.
- src/preload.js — all four channels added to ALLOWED_SEND_CHANNELS.
Tests (56 new across the three modules):
- tests/main/DailyNotes.test.js (15): dateKey formatting, pathFor,
template load + {date}/{weekday} substitution, openOrCreate +
no-clobber, nested-dir creation, listExisting filtering,
isValidDir rejects NUL/non-string.
- tests/main/WorkspaceSearch.test.js (25): parseQuery grammar,
hasTag/hasWikilink word boundaries, scoreDocument scoring,
per-term spam cap, recency nudge, search ranking + limit +
empty-query short-circuit, bad-input safety.
- tests/main/DocQA.test.js (16): cleanQuestion stripping + facet
preservation, chunkDocument paragraph + hard-split, ask()
top-K, recency tiebreaker, missing-files fallback.
Full suite: 748 tests pass, 66 suites, lint+format clean.
Amit Haridas
This commit is contained in:
+155
@@ -3978,6 +3978,9 @@ ipcMain.on('set-current-file', (event, filePath) => {
|
||||
// ============================================
|
||||
const VersionHistory = require('./main/VersionHistory');
|
||||
const AutosaveBuffer = require('./main/AutosaveBuffer');
|
||||
const DailyNotes = require('./main/DailyNotes');
|
||||
const WorkspaceSearch = require('./main/WorkspaceSearch');
|
||||
const DocQA = require('./main/DocQA');
|
||||
|
||||
/** IO bundle for VersionHistory bound to <userData>/versions. */
|
||||
function versionHistoryIo() {
|
||||
@@ -5695,6 +5698,135 @@ ipcMain.handle('quick-note:save', async (_event, text) => {
|
||||
return { path: file };
|
||||
});
|
||||
|
||||
// ================================
|
||||
// Daily notes (Zettelkasten/journal helper)
|
||||
// ================================
|
||||
// Convention: <userData>/notes/daily/YYYY-MM-DD.md — one file per local date,
|
||||
// loaded from <userData>/notes/templates/daily.md when present (built-in
|
||||
// default otherwise). The IPC channel validates the path through validatePath
|
||||
// before touching disk so a renderer compromise can't redirect us to /etc.
|
||||
function dailyNotesDir() {
|
||||
return path.join(app.getPath('userData'), 'notes', 'daily');
|
||||
}
|
||||
function dailyNotesTemplateDir() {
|
||||
return path.join(app.getPath('userData'), 'notes', 'templates');
|
||||
}
|
||||
|
||||
ipcMain.handle('daily-notes:open-today', async (_event, { date } = {}) => {
|
||||
const dir = dailyNotesDir();
|
||||
const validation = validatePath(dir);
|
||||
if (!validation.valid) throw new Error('Invalid daily-notes directory');
|
||||
|
||||
const when = date ? new Date(date) : new Date();
|
||||
if (Number.isNaN(when.getTime())) throw new Error('Invalid date for daily-notes');
|
||||
|
||||
return DailyNotes.openOrCreate({
|
||||
date: when,
|
||||
dir,
|
||||
templateDir: dailyNotesTemplateDir(),
|
||||
fs,
|
||||
pathUtil: path,
|
||||
now: when,
|
||||
});
|
||||
});
|
||||
|
||||
ipcMain.handle('daily-notes:list', async () => {
|
||||
const dir = dailyNotesDir();
|
||||
const validation = validatePath(dir);
|
||||
if (!validation.valid) return [];
|
||||
return DailyNotes.listExisting({ dir, fs, pathUtil: path });
|
||||
});
|
||||
|
||||
// ================================
|
||||
// Workspace content search (tag/wikilink-aware)
|
||||
// ================================
|
||||
// Walks a directory for .md files (non-recursive by default — most note
|
||||
// collections live in one folder) and runs the WorkspaceSearch ranking
|
||||
// algorithm. Cap at MAX_FILES so a typo'd dir doesn't pull the whole disk.
|
||||
const WORKSPACE_SEARCH_MAX_FILES = 2000;
|
||||
const WORKSPACE_SEARCH_MAX_BYTES = 1024 * 1024; // skip files > 1 MiB
|
||||
const SKIP_DIRS = new Set(['node_modules', '.git', '.svn', '.hg', 'dist', 'build', '.cache']);
|
||||
|
||||
function collectMarkdownFiles(rootDir, maxFiles) {
|
||||
const out = [];
|
||||
const queue = [rootDir];
|
||||
while (queue.length > 0 && out.length < maxFiles) {
|
||||
const dir = queue.shift();
|
||||
let entries;
|
||||
try {
|
||||
entries = fs.readdirSync(dir, { withFileTypes: true });
|
||||
} catch {
|
||||
continue; // unreadable dir — skip silently
|
||||
}
|
||||
for (const entry of entries) {
|
||||
if (out.length >= maxFiles) break;
|
||||
const full = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) {
|
||||
if (SKIP_DIRS.has(entry.name) || entry.name.startsWith('.')) continue;
|
||||
queue.push(full);
|
||||
} else if (entry.isFile() && /\.md$/i.test(entry.name)) {
|
||||
out.push(full);
|
||||
}
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
ipcMain.handle('workspace-search:query', async (_event, { query, dir, limit = 50 } = {}) => {
|
||||
if (typeof query !== 'string' || !query.trim()) return [];
|
||||
const validation = typeof dir === 'string' ? validatePath(dir) : { valid: false };
|
||||
if (!validation.valid) return [];
|
||||
|
||||
const files = collectMarkdownFiles(dir, WORKSPACE_SEARCH_MAX_FILES);
|
||||
const corpus = [];
|
||||
for (const filePath of files) {
|
||||
let content;
|
||||
try {
|
||||
const stat = fs.statSync(filePath);
|
||||
if (stat.size > WORKSPACE_SEARCH_MAX_BYTES) continue;
|
||||
content = fs.readFileSync(filePath, 'utf-8');
|
||||
corpus.push({ path: filePath, content, mtimeMs: stat.mtimeMs });
|
||||
} catch {
|
||||
/* unreadable — skip */
|
||||
}
|
||||
}
|
||||
|
||||
return WorkspaceSearch.search({ query, files: corpus, limit });
|
||||
});
|
||||
|
||||
// ================================
|
||||
// Doc-aware Q&A — ask a natural-language question, get ranked chunks
|
||||
// ================================
|
||||
// Same corpus construction as workspace-search:query, but DocQA.cleanQuestion
|
||||
// strips grammar noise (what/how/why/...) and re-ranks at the chunk level
|
||||
// so the renderer can show multiple passages from the same file. No neural
|
||||
// model — same ranking algorithm — so results stay explainable and offline.
|
||||
ipcMain.handle('doc-qa:ask', async (_event, { question, dir, topK = 5 } = {}) => {
|
||||
if (typeof question !== 'string' || !question.trim()) {
|
||||
return { question: '', chunks: [] };
|
||||
}
|
||||
const validation = typeof dir === 'string' ? validatePath(dir) : { valid: false };
|
||||
if (!validation.valid) return { question: String(question), chunks: [] };
|
||||
|
||||
const files = collectMarkdownFiles(dir, WORKSPACE_SEARCH_MAX_FILES);
|
||||
const corpus = [];
|
||||
for (const filePath of files) {
|
||||
try {
|
||||
const stat = fs.statSync(filePath);
|
||||
if (stat.size > WORKSPACE_SEARCH_MAX_BYTES) continue;
|
||||
corpus.push({
|
||||
path: filePath,
|
||||
content: fs.readFileSync(filePath, 'utf-8'),
|
||||
mtimeMs: stat.mtimeMs,
|
||||
});
|
||||
} catch {
|
||||
/* unreadable — skip */
|
||||
}
|
||||
}
|
||||
|
||||
return DocQA.ask({ question, files: corpus, topK });
|
||||
});
|
||||
|
||||
// Esc in the note window hides instead of closing (keeps it one keystroke away)
|
||||
ipcMain.on('quick-note:hide', () => {
|
||||
if (quickNoteWindow) quickNoteWindow.hide();
|
||||
@@ -5708,10 +5840,33 @@ app.whenReady().then(() => {
|
||||
openQuickNoteWindow();
|
||||
});
|
||||
if (!registered) console.warn('Quick Note shortcut Ctrl+Alt+Q could not be registered');
|
||||
|
||||
// Daily-notes global shortcut: open (or create) today's YYYY-MM-DD.md.
|
||||
// Ctrl+Alt+D = "diary". The handler fires even when the app is unfocused
|
||||
// so the user can journal from anywhere on the desktop.
|
||||
const dailyRegistered = globalShortcut.register('CommandOrControl+Alt+D', async () => {
|
||||
try {
|
||||
const result = DailyNotes.openOrCreate({
|
||||
date: new Date(),
|
||||
dir: dailyNotesDir(),
|
||||
templateDir: dailyNotesTemplateDir(),
|
||||
fs,
|
||||
pathUtil: path,
|
||||
now: new Date(),
|
||||
});
|
||||
if (mainWindow && !mainWindow.isDestroyed()) {
|
||||
mainWindow.webContents.send('file-opened', { filePath: result.path });
|
||||
}
|
||||
} catch (err) {
|
||||
console.warn('[daily-notes] open failed:', err && err.message);
|
||||
}
|
||||
});
|
||||
if (!dailyRegistered) console.warn('Daily Notes shortcut Ctrl+Alt+D could not be registered');
|
||||
});
|
||||
app.on('will-quit', () => {
|
||||
const { globalShortcut } = require('electron');
|
||||
globalShortcut.unregister('CommandOrControl+Alt+Q');
|
||||
globalShortcut.unregister('CommandOrControl+Alt+D');
|
||||
});
|
||||
|
||||
// IPC Handler for loading document templates
|
||||
|
||||
@@ -0,0 +1,130 @@
|
||||
/**
|
||||
* Daily notes — Zettelkasten/journal entry helper.
|
||||
*
|
||||
* Convention: a folder of YYYY-MM-DD.md files (PKM/Obsidian/Logseq style).
|
||||
* `pathFor(date, dir)` returns the absolute path for a given date. The
|
||||
* template loader returns a starting skeleton the user can fill in; if the
|
||||
* file already exists, we return its current content instead (no clobber).
|
||||
*
|
||||
* Implementation is pure (no fs/IPC coupling) — main.js wires the IO and
|
||||
* template path. Tests inject a virtual fs + path util.
|
||||
*/
|
||||
|
||||
const DEFAULT_TEMPLATE_NAME = 'daily.md';
|
||||
|
||||
/**
|
||||
* Format a Date as YYYY-MM-DD in local time (matches the on-disk filename).
|
||||
* Returns 'YYYY-MM-DD' string. Today (no arg) defaults to new Date().
|
||||
*/
|
||||
function dateKey(date = new Date()) {
|
||||
const y = date.getFullYear();
|
||||
const m = String(date.getMonth() + 1).padStart(2, '0');
|
||||
const d = String(date.getDate()).padStart(2, '0');
|
||||
return `${y}-${m}-${d}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Absolute path for the daily note of `date` inside `dir`. Pure.
|
||||
*/
|
||||
function pathFor(date, dir, pathUtil) {
|
||||
return pathUtil.join(dir, `${dateKey(date)}.md`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the daily note skeleton. Looks in `templateDir` for a file named
|
||||
* `templateName` (default 'daily.md'); if missing, returns a built-in
|
||||
* default so the feature works out-of-the-box without setup.
|
||||
*
|
||||
* Returns the template string with any {date}/{weekday} placeholders
|
||||
* substituted.
|
||||
*/
|
||||
function loadTemplate({
|
||||
date,
|
||||
templateDir,
|
||||
templateName = DEFAULT_TEMPLATE_NAME,
|
||||
fs,
|
||||
pathUtil,
|
||||
now = new Date(),
|
||||
}) {
|
||||
let body = `# ${dateKey(date)}\n\n## Notes\n\n`;
|
||||
if (templateDir && fs) {
|
||||
const templatePath = pathUtil.join(templateDir, templateName);
|
||||
try {
|
||||
body = fs.readFileSync(templatePath, 'utf-8');
|
||||
} catch {
|
||||
/* fall through to default */
|
||||
}
|
||||
}
|
||||
return body
|
||||
.replace(/\{date\}/g, dateKey(date))
|
||||
.replace(/\{weekday\}/g, now.toLocaleDateString('en-US', { weekday: 'long' }));
|
||||
}
|
||||
|
||||
/**
|
||||
* Open or create the daily note for `date` inside `dir`. Returns
|
||||
* { path, content, created: boolean }.
|
||||
*
|
||||
* Behavior:
|
||||
* - If <dir>/<date>.md exists, return its content (created: false).
|
||||
* - Otherwise load the template, write it, and return it (created: true).
|
||||
*
|
||||
* Caller (main.js IPC handler) decides what to do with `created` —
|
||||
* typically: open in the existing tab if any, else createNewTab.
|
||||
*/
|
||||
function openOrCreate({ date, dir, templateDir, fs, pathUtil, now = new Date() }) {
|
||||
if (!dir) throw new Error('DailyNotes: dir is required');
|
||||
fs.mkdirSync(dir, { recursive: true });
|
||||
const notePath = pathFor(date, dir, pathUtil);
|
||||
let content;
|
||||
let created = false;
|
||||
try {
|
||||
content = fs.readFileSync(notePath, 'utf-8');
|
||||
} catch (err) {
|
||||
if (err.code !== 'ENOENT') throw err;
|
||||
content = loadTemplate({ date, templateDir, fs, pathUtil, now });
|
||||
fs.writeFileSync(notePath, content, 'utf-8');
|
||||
created = true;
|
||||
}
|
||||
return { path: notePath, content, created };
|
||||
}
|
||||
|
||||
/**
|
||||
* List existing daily-note filenames in `dir`, newest first.
|
||||
* Returns ['2026-09-13.md', '2026-09-12.md', ...] — only entries that
|
||||
* match the YYYY-MM-DD.md shape so a stray readme.md in the folder is
|
||||
* ignored.
|
||||
*/
|
||||
function listExisting({ dir, fs }) {
|
||||
let entries;
|
||||
try {
|
||||
entries = fs.readdirSync(dir);
|
||||
} catch (err) {
|
||||
if (err.code === 'ENOENT') return [];
|
||||
throw err;
|
||||
}
|
||||
// Strict-enough date pattern: year-month-day with month 01-12 and day 01-31.
|
||||
// We don't enforce real-calendar validity (Feb 30 still matches) — that's
|
||||
// the user's problem to fix, not ours to silently drop.
|
||||
const pattern = /^\d{4}-(0[1-9]|1[0-2])-(0[1-9]|[12]\d|3[01])\.md$/;
|
||||
return entries.filter((name) => pattern.test(name)).sort((a, b) => (a < b ? 1 : a > b ? -1 : 0));
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate a `dir` candidate for daily notes. Used by the IPC handler so a
|
||||
* renderer compromise can't point this at /etc or other sensitive paths —
|
||||
* any path outside userData/templates is rejected by main.js's existing
|
||||
* validatePath() before we get here.
|
||||
*/
|
||||
function isValidDir(dir) {
|
||||
return typeof dir === 'string' && dir.length > 0 && !dir.includes('\0');
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
dateKey,
|
||||
pathFor,
|
||||
loadTemplate,
|
||||
openOrCreate,
|
||||
listExisting,
|
||||
isValidDir,
|
||||
DEFAULT_TEMPLATE_NAME,
|
||||
};
|
||||
@@ -0,0 +1,231 @@
|
||||
/**
|
||||
* Doc-aware Q&A — thin wrapper over WorkspaceSearch tuned for questions.
|
||||
*
|
||||
* User asks "what did I write about rust async last week?" and gets back the
|
||||
* top-K chunks from their workspace ranked by relevance. No neural model, no
|
||||
* API key — the same TF-idf-style ranking WorkspaceSearch uses, with a few
|
||||
* QA-specific tweaks:
|
||||
*
|
||||
* - Question words (what/how/why/when/where/who/which/does/is/are/...) are
|
||||
* stripped before ranking — they don't carry meaning, just grammar
|
||||
* - The top chunks are split out as independent results so the renderer
|
||||
* can show "3 passages from 2 files" instead of one giant result
|
||||
* - Recency weight is doubled: "last week" implies the user wants fresh
|
||||
* content, not old material
|
||||
*
|
||||
* Future upgrade path: swap `WorkspaceSearch.search` for a vector-similarity
|
||||
* call (transformers.js in the renderer, or a sidecar process). The public
|
||||
* shape — {question, chunks:[{filePath, snippet, score}]} — stays the same.
|
||||
*
|
||||
* @module DocQA
|
||||
*/
|
||||
|
||||
const WorkspaceSearch = require('./WorkspaceSearch');
|
||||
|
||||
const QUESTION_WORDS = new Set([
|
||||
'what',
|
||||
'when',
|
||||
'where',
|
||||
'who',
|
||||
'whom',
|
||||
'whose',
|
||||
'why',
|
||||
'how',
|
||||
'which',
|
||||
'whether',
|
||||
'does',
|
||||
'do',
|
||||
'did',
|
||||
'is',
|
||||
'are',
|
||||
'was',
|
||||
'were',
|
||||
'be',
|
||||
'been',
|
||||
'being',
|
||||
'have',
|
||||
'has',
|
||||
'had',
|
||||
'can',
|
||||
'could',
|
||||
'would',
|
||||
'should',
|
||||
'will',
|
||||
'shall',
|
||||
'may',
|
||||
'might',
|
||||
'i',
|
||||
'me',
|
||||
'my',
|
||||
'we',
|
||||
'our',
|
||||
'you',
|
||||
'your',
|
||||
'they',
|
||||
'them',
|
||||
'their',
|
||||
'a',
|
||||
'an',
|
||||
'the',
|
||||
'and',
|
||||
'or',
|
||||
'but',
|
||||
'so',
|
||||
'of',
|
||||
'to',
|
||||
'in',
|
||||
'on',
|
||||
'at',
|
||||
'for',
|
||||
'with',
|
||||
'about',
|
||||
'into',
|
||||
'from',
|
||||
'by',
|
||||
'as',
|
||||
'this',
|
||||
'that',
|
||||
'these',
|
||||
'those',
|
||||
'it',
|
||||
'its',
|
||||
'write',
|
||||
'wrote',
|
||||
'written',
|
||||
'read',
|
||||
'think',
|
||||
'know',
|
||||
'find',
|
||||
'show',
|
||||
'tell',
|
||||
'say',
|
||||
'see',
|
||||
'use',
|
||||
'used',
|
||||
]);
|
||||
|
||||
/**
|
||||
* Strip question words from a raw question so WorkspaceSearch's bare-term
|
||||
* matching isn't drowned in grammar noise. Quoted phrases, #tags, and
|
||||
* @wikilinks pass through unchanged.
|
||||
*/
|
||||
function cleanQuestion(raw) {
|
||||
if (typeof raw !== 'string') return '';
|
||||
// First pull out facets and phrases so we don't strip them
|
||||
const facets = [];
|
||||
const working = raw.replace(/[#@][A-Za-z0-9_-]+|"[^"]+"/g, (m) => {
|
||||
facets.push(m);
|
||||
return ' ';
|
||||
});
|
||||
const words = working.split(/\s+/).filter((w) => {
|
||||
const lower = w.toLowerCase().replace(/[^a-z0-9-]/g, '');
|
||||
return lower.length >= 2 && !QUESTION_WORDS.has(lower);
|
||||
});
|
||||
return [...words, ...facets].join(' ').trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* Split a document into rough passages (~800 chars or paragraph break,
|
||||
* whichever comes first). Returns an array of {start, end, text} so the
|
||||
* caller can build snippets. Pure.
|
||||
*/
|
||||
function chunkDocument(content, maxChunkChars = 800) {
|
||||
if (typeof content !== 'string' || content.length === 0) return [];
|
||||
const chunks = [];
|
||||
const paragraphs = content.split(/\n\s*\n/);
|
||||
let buffer = '';
|
||||
let startOffset = 0;
|
||||
for (const para of paragraphs) {
|
||||
// If a single paragraph exceeds the cap, hard-split it by character.
|
||||
if (para.length > maxChunkChars) {
|
||||
// Flush whatever was buffered first.
|
||||
if (buffer) {
|
||||
chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
|
||||
startOffset += buffer.length + 2;
|
||||
buffer = '';
|
||||
}
|
||||
for (let i = 0; i < para.length; i += maxChunkChars) {
|
||||
const slice = para.slice(i, i + maxChunkChars);
|
||||
chunks.push({ start: startOffset, end: startOffset + slice.length, text: slice });
|
||||
startOffset += slice.length;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
const tentative = buffer ? `${buffer}\n\n${para}` : para;
|
||||
if (tentative.length > maxChunkChars && buffer) {
|
||||
chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
|
||||
startOffset += buffer.length + 2; // account for the "\n\n" we split on
|
||||
buffer = para;
|
||||
} else {
|
||||
buffer = tentative;
|
||||
}
|
||||
}
|
||||
if (buffer) chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
|
||||
return chunks;
|
||||
}
|
||||
|
||||
/**
|
||||
* Answer a natural-language question by ranking the workspace's chunks.
|
||||
*
|
||||
* @param {object} args
|
||||
* @param {string} args.question
|
||||
* @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files
|
||||
* @param {number} [args.topK=5] number of chunks to return
|
||||
* @param {number} [args.nowMs=Date.now()]
|
||||
* @returns {{question:string, chunks:Array<{filePath:string, snippet:string, score:number, mtimeMs:number}>}}
|
||||
*/
|
||||
function ask({ question, files, topK = 5, nowMs = Date.now() }) {
|
||||
const cleaned = cleanQuestion(question);
|
||||
if (!cleaned || !Array.isArray(files) || files.length === 0) {
|
||||
return { question: String(question || ''), chunks: [] };
|
||||
}
|
||||
|
||||
// First, find the docs that match at all (cheap, broad pass).
|
||||
const docHits = WorkspaceSearch.search({ query: cleaned, files, limit: 20, nowMs });
|
||||
|
||||
// Then re-rank at the chunk level within those docs.
|
||||
const chunkCorpus = [];
|
||||
for (const hit of docHits) {
|
||||
const file = files.find((f) => f.path === hit.filePath);
|
||||
if (!file || typeof file.content !== 'string') continue;
|
||||
const chunks = chunkDocument(file.content);
|
||||
for (const chunk of chunks) {
|
||||
chunkCorpus.push({
|
||||
path: `${hit.filePath}#${chunk.start}`,
|
||||
content: chunk.text,
|
||||
mtimeMs: file.mtimeMs,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
const chunkHits = WorkspaceSearch.search({
|
||||
query: cleaned,
|
||||
files: chunkCorpus,
|
||||
limit: topK,
|
||||
nowMs,
|
||||
});
|
||||
|
||||
// Translate the per-chunk hits back into the public shape. The fake path
|
||||
// "<file>#<offset>" carries the chunk start; the renderer's existing
|
||||
// file-opened handler can split it on '#' if it wants to deep-link.
|
||||
const chunks = chunkHits.map((h) => {
|
||||
const offsetMatch = /#(\d+)$/.exec(h.filePath);
|
||||
const offset = offsetMatch ? Number(offsetMatch[1]) : 0;
|
||||
return {
|
||||
filePath: h.filePath.replace(/#\d+$/, ''),
|
||||
offset,
|
||||
snippet: h.snippet,
|
||||
score: h.score,
|
||||
mtimeMs: chunkCorpus.find((c) => c.path === h.filePath)?.mtimeMs || 0,
|
||||
};
|
||||
});
|
||||
|
||||
return { question: String(question), chunks };
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
cleanQuestion,
|
||||
chunkDocument,
|
||||
ask,
|
||||
QUESTION_WORDS,
|
||||
};
|
||||
@@ -0,0 +1,203 @@
|
||||
/**
|
||||
* Workspace content search — the simplest useful thing that scales.
|
||||
*
|
||||
* Given a list of file paths and a query, returns ranked matches across
|
||||
* them. Designed for personal-workspace sizes (hundreds, maybe thousands of
|
||||
* notes) — runs synchronously on the main process with injectable IO so it
|
||||
* can be unit-tested without touching the filesystem.
|
||||
*
|
||||
* Query grammar (all optional, combinable):
|
||||
* - Bare words → term matches (case-insensitive substring)
|
||||
* - Words prefixed with `#` → require a matching `#tag` in the doc
|
||||
* - Words prefixed with `@` → require a matching `[[wikilink]]` target
|
||||
* - Quoted phrases → require the exact substring
|
||||
*
|
||||
* Scoring:
|
||||
* - Each matched term adds +1 per occurrence (capped at 5/term to dampen
|
||||
* repetition spam)
|
||||
* - Each tag hit (`#foo`) or wikilink hit (`[[foo]]`) adds +3 if the query
|
||||
* asked for it (facets weigh more than prose)
|
||||
* - Recent edits (mtime within 7 days) get +0.5 to nudge "what I was
|
||||
* working on yesterday" upward
|
||||
*
|
||||
* Returns [{ filePath, score, snippet, matchedTerms, matchedTags, matchedLinks }]
|
||||
* sorted by score desc, filePath asc as tiebreaker.
|
||||
*
|
||||
* @module WorkspaceSearch
|
||||
*/
|
||||
|
||||
/** Tokenize a query into a structured form. Pure. */
|
||||
function parseQuery(raw) {
|
||||
if (typeof raw !== 'string') return { terms: [], phrases: [], tags: [], links: [] };
|
||||
const terms = [];
|
||||
const phrases = [];
|
||||
const tags = [];
|
||||
const links = [];
|
||||
|
||||
// Phrases first — match "double quoted text" as a unit
|
||||
const phraseRe = /"([^"]+)"/g;
|
||||
let remaining = raw.replace(phraseRe, (_, inner) => {
|
||||
if (inner.trim()) phrases.push(inner.toLowerCase());
|
||||
return ' ';
|
||||
});
|
||||
|
||||
// Tags and links: #foo, @bar
|
||||
const facetRe = /[#@]([A-Za-z0-9_-]+)/g;
|
||||
remaining = remaining.replace(facetRe, (m, name) => {
|
||||
if (m.startsWith('#')) tags.push(name.toLowerCase());
|
||||
else links.push(name.toLowerCase());
|
||||
return ' ';
|
||||
});
|
||||
|
||||
// Bare words (≥2 chars to avoid noise)
|
||||
for (const w of remaining.split(/\s+/)) {
|
||||
if (w.length >= 2) terms.push(w.toLowerCase());
|
||||
}
|
||||
|
||||
return { terms, phrases, tags, links };
|
||||
}
|
||||
|
||||
/** Extract a one-line snippet around the first match for `term`. Pure. */
|
||||
function snippetAround(content, term, radius = 60) {
|
||||
const haystack = content.toLowerCase();
|
||||
const idx = haystack.indexOf(term);
|
||||
if (idx < 0) return '';
|
||||
const start = Math.max(0, idx - radius);
|
||||
const end = Math.min(content.length, idx + term.length + radius);
|
||||
let s = content.slice(start, end).replace(/\s+/g, ' ').trim();
|
||||
if (start > 0) s = '…' + s;
|
||||
if (end < content.length) s = s + '…';
|
||||
return s;
|
||||
}
|
||||
|
||||
/** Count occurrences of `needle` (case-insensitive substring) in `haystack`. */
|
||||
function countOccurrences(haystack, needle) {
|
||||
if (!needle) return 0;
|
||||
const hay = haystack.toLowerCase();
|
||||
const ndl = needle.toLowerCase();
|
||||
let count = 0;
|
||||
let idx = 0;
|
||||
while ((idx = hay.indexOf(ndl, idx)) !== -1) {
|
||||
count++;
|
||||
idx += ndl.length;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
/** Does the content contain the `#tag`? Looks for word-boundary `#tag`. */
|
||||
function hasTag(content, tag) {
|
||||
const re = new RegExp(`(^|\\s)#${escapeRegex(tag)}\\b`, 'i');
|
||||
return re.test(content);
|
||||
}
|
||||
|
||||
/** Does the content contain `[[wikilink]]` or `[[wikilink|alias]]`? */
|
||||
function hasWikilink(content, link) {
|
||||
const re = new RegExp(`\\[\\[${escapeRegex(link)}(\\|[^\\]]+)?\\]\\]`, 'i');
|
||||
return re.test(content);
|
||||
}
|
||||
|
||||
function escapeRegex(s) {
|
||||
return String(s).replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
||||
}
|
||||
|
||||
/**
|
||||
* Run a parsed query against one document's content. Returns
|
||||
* { score, matchedTerms, matchedTags, matchedLinks, snippet } or null if
|
||||
* nothing matched (so the caller can filter cheaply).
|
||||
*/
|
||||
function scoreDocument({ content, parsed, mtimeMs = 0, nowMs = Date.now() }) {
|
||||
let score = 0;
|
||||
const matchedTerms = [];
|
||||
const matchedTags = [];
|
||||
const matchedLinks = [];
|
||||
const lower = content.toLowerCase();
|
||||
|
||||
for (const term of parsed.terms) {
|
||||
const n = countOccurrences(content, term);
|
||||
if (n > 0) {
|
||||
score += Math.min(n, 5);
|
||||
matchedTerms.push(term);
|
||||
}
|
||||
}
|
||||
for (const phrase of parsed.phrases) {
|
||||
if (lower.includes(phrase)) {
|
||||
score += 3; // phrase hits weigh more
|
||||
matchedTerms.push(phrase);
|
||||
}
|
||||
}
|
||||
for (const tag of parsed.tags) {
|
||||
if (hasTag(content, tag)) {
|
||||
score += 3;
|
||||
matchedTags.push(tag);
|
||||
}
|
||||
}
|
||||
for (const link of parsed.links) {
|
||||
if (hasWikilink(content, link)) {
|
||||
score += 3;
|
||||
matchedLinks.push(link);
|
||||
}
|
||||
}
|
||||
|
||||
if (score === 0) return null;
|
||||
|
||||
// Recency nudge: edits within the last 7 days
|
||||
const ageDays = (nowMs - mtimeMs) / (24 * 60 * 60 * 1000);
|
||||
if (Number.isFinite(ageDays) && ageDays >= 0 && ageDays <= 7) {
|
||||
score += 0.5 * (1 - ageDays / 7);
|
||||
}
|
||||
|
||||
// Build a snippet around the first matched term (or phrase/tag) for the UI
|
||||
let snippet = '';
|
||||
const firstTerm = matchedTerms[0] || (parsed.tags[0] ? `#${parsed.tags[0]}` : null);
|
||||
if (firstTerm) {
|
||||
snippet = snippetAround(content, firstTerm);
|
||||
} else {
|
||||
snippet = content.slice(0, 120).replace(/\s+/g, ' ').trim() + (content.length > 120 ? '…' : '');
|
||||
}
|
||||
|
||||
return { score, matchedTerms, matchedTags, matchedLinks, snippet };
|
||||
}
|
||||
|
||||
/**
|
||||
* Search across many files. Files whose content matches the query are
|
||||
* scored and returned sorted by score.
|
||||
*
|
||||
* @param {object} args
|
||||
* @param {string} args.query raw query string
|
||||
* @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files
|
||||
* @param {number} [args.limit=50]
|
||||
* @returns {Array<{filePath:string, score:number, snippet:string, matchedTerms:string[], matchedTags:string[], matchedLinks:string[]}>}
|
||||
*/
|
||||
function search({ query, files, limit = 50, nowMs = Date.now() }) {
|
||||
const parsed = parseQuery(query);
|
||||
// Nothing to search for — short-circuit so callers don't pay the file loop
|
||||
if (
|
||||
parsed.terms.length === 0 &&
|
||||
parsed.phrases.length === 0 &&
|
||||
parsed.tags.length === 0 &&
|
||||
parsed.links.length === 0
|
||||
) {
|
||||
return [];
|
||||
}
|
||||
const out = [];
|
||||
for (const file of files) {
|
||||
if (!file || typeof file.content !== 'string') continue;
|
||||
const r = scoreDocument({ content: file.content, parsed, mtimeMs: file.mtimeMs, nowMs });
|
||||
if (!r) continue;
|
||||
out.push({ filePath: file.path, ...r });
|
||||
}
|
||||
out.sort(
|
||||
(a, b) => b.score - a.score || (a.filePath < b.filePath ? -1 : a.filePath > b.filePath ? 1 : 0)
|
||||
);
|
||||
return out.slice(0, limit);
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
parseQuery,
|
||||
snippetAround,
|
||||
countOccurrences,
|
||||
hasTag,
|
||||
hasWikilink,
|
||||
scoreDocument,
|
||||
search,
|
||||
};
|
||||
@@ -186,6 +186,16 @@ const ALLOWED_SEND_CHANNELS = [
|
||||
'autosave:read',
|
||||
'autosave:clear',
|
||||
'autosave:list',
|
||||
|
||||
// Daily notes (one file per local date)
|
||||
'daily-notes:open-today',
|
||||
'daily-notes:list',
|
||||
|
||||
// Workspace content search (tag/wikilink-aware)
|
||||
'workspace-search:query',
|
||||
|
||||
// Doc-aware Q&A (chunk-level ranking over the workspace)
|
||||
'doc-qa:ask',
|
||||
];
|
||||
|
||||
const ALLOWED_RECEIVE_CHANNELS = [
|
||||
|
||||
Reference in New Issue
Block a user