mirror of
https://github.com/amitwh/markdown-converter.git
synced 2026-10-01 17:29:29 +05:30
feat(pkm): daily notes + workspace search + doc-aware Q&A
Three more features from the brainstorm menu, built on a shared search
algorithm so the codebase stays small.
- src/main/DailyNotes.js — Zettelkasten-style helper. One YYYY-MM-DD.md
per local date under <userData>/notes/daily/; loads skeleton from
<userData>/notes/templates/daily.md when present (built-in default
otherwise). openOrCreate never clobbers existing content.
- src/main/WorkspaceSearch.js — tag/wikilink-aware content search.
Pure module, injectable-IO tested. Query grammar: bare words,
#tag, @wikilink, "quoted phrases". Facets weight +3 each; prose
terms +1/occurrence capped at 5; edits within 7 days get a recency
nudge. Returns ranked results with snippets.
- src/main/DocQA.js — chunk-level Q&A wrapper over WorkspaceSearch.
cleanQuestion strips question words (what/how/why/...) and verb
noise (write/read/show/tell/...) so they don't drown the ranking.
Returns top-K passages instead of whole-file hits — multiple chunks
from the same file can appear in the answer.
- src/main.js — IPC: daily-notes:open-today, daily-notes:list,
workspace-search:query, doc-qa:ask. Path validation through the
existing validatePath gate; a global Ctrl+Alt+D shortcut creates
today's daily note from anywhere.
- src/preload.js — all four channels added to ALLOWED_SEND_CHANNELS.
Tests (56 new across the three modules):
- tests/main/DailyNotes.test.js (15): dateKey formatting, pathFor,
template load + {date}/{weekday} substitution, openOrCreate +
no-clobber, nested-dir creation, listExisting filtering,
isValidDir rejects NUL/non-string.
- tests/main/WorkspaceSearch.test.js (25): parseQuery grammar,
hasTag/hasWikilink word boundaries, scoreDocument scoring,
per-term spam cap, recency nudge, search ranking + limit +
empty-query short-circuit, bad-input safety.
- tests/main/DocQA.test.js (16): cleanQuestion stripping + facet
preservation, chunkDocument paragraph + hard-split, ask()
top-K, recency tiebreaker, missing-files fallback.
Full suite: 748 tests pass, 66 suites, lint+format clean.
Amit Haridas
This commit is contained in:
@@ -0,0 +1,231 @@
|
||||
/**
|
||||
* Doc-aware Q&A — thin wrapper over WorkspaceSearch tuned for questions.
|
||||
*
|
||||
* User asks "what did I write about rust async last week?" and gets back the
|
||||
* top-K chunks from their workspace ranked by relevance. No neural model, no
|
||||
* API key — the same TF-idf-style ranking WorkspaceSearch uses, with a few
|
||||
* QA-specific tweaks:
|
||||
*
|
||||
* - Question words (what/how/why/when/where/who/which/does/is/are/...) are
|
||||
* stripped before ranking — they don't carry meaning, just grammar
|
||||
* - The top chunks are split out as independent results so the renderer
|
||||
* can show "3 passages from 2 files" instead of one giant result
|
||||
* - Recency weight is doubled: "last week" implies the user wants fresh
|
||||
* content, not old material
|
||||
*
|
||||
* Future upgrade path: swap `WorkspaceSearch.search` for a vector-similarity
|
||||
* call (transformers.js in the renderer, or a sidecar process). The public
|
||||
* shape — {question, chunks:[{filePath, snippet, score}]} — stays the same.
|
||||
*
|
||||
* @module DocQA
|
||||
*/
|
||||
|
||||
const WorkspaceSearch = require('./WorkspaceSearch');
|
||||
|
||||
const QUESTION_WORDS = new Set([
|
||||
'what',
|
||||
'when',
|
||||
'where',
|
||||
'who',
|
||||
'whom',
|
||||
'whose',
|
||||
'why',
|
||||
'how',
|
||||
'which',
|
||||
'whether',
|
||||
'does',
|
||||
'do',
|
||||
'did',
|
||||
'is',
|
||||
'are',
|
||||
'was',
|
||||
'were',
|
||||
'be',
|
||||
'been',
|
||||
'being',
|
||||
'have',
|
||||
'has',
|
||||
'had',
|
||||
'can',
|
||||
'could',
|
||||
'would',
|
||||
'should',
|
||||
'will',
|
||||
'shall',
|
||||
'may',
|
||||
'might',
|
||||
'i',
|
||||
'me',
|
||||
'my',
|
||||
'we',
|
||||
'our',
|
||||
'you',
|
||||
'your',
|
||||
'they',
|
||||
'them',
|
||||
'their',
|
||||
'a',
|
||||
'an',
|
||||
'the',
|
||||
'and',
|
||||
'or',
|
||||
'but',
|
||||
'so',
|
||||
'of',
|
||||
'to',
|
||||
'in',
|
||||
'on',
|
||||
'at',
|
||||
'for',
|
||||
'with',
|
||||
'about',
|
||||
'into',
|
||||
'from',
|
||||
'by',
|
||||
'as',
|
||||
'this',
|
||||
'that',
|
||||
'these',
|
||||
'those',
|
||||
'it',
|
||||
'its',
|
||||
'write',
|
||||
'wrote',
|
||||
'written',
|
||||
'read',
|
||||
'think',
|
||||
'know',
|
||||
'find',
|
||||
'show',
|
||||
'tell',
|
||||
'say',
|
||||
'see',
|
||||
'use',
|
||||
'used',
|
||||
]);
|
||||
|
||||
/**
|
||||
* Strip question words from a raw question so WorkspaceSearch's bare-term
|
||||
* matching isn't drowned in grammar noise. Quoted phrases, #tags, and
|
||||
* @wikilinks pass through unchanged.
|
||||
*/
|
||||
function cleanQuestion(raw) {
|
||||
if (typeof raw !== 'string') return '';
|
||||
// First pull out facets and phrases so we don't strip them
|
||||
const facets = [];
|
||||
const working = raw.replace(/[#@][A-Za-z0-9_-]+|"[^"]+"/g, (m) => {
|
||||
facets.push(m);
|
||||
return ' ';
|
||||
});
|
||||
const words = working.split(/\s+/).filter((w) => {
|
||||
const lower = w.toLowerCase().replace(/[^a-z0-9-]/g, '');
|
||||
return lower.length >= 2 && !QUESTION_WORDS.has(lower);
|
||||
});
|
||||
return [...words, ...facets].join(' ').trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* Split a document into rough passages (~800 chars or paragraph break,
|
||||
* whichever comes first). Returns an array of {start, end, text} so the
|
||||
* caller can build snippets. Pure.
|
||||
*/
|
||||
function chunkDocument(content, maxChunkChars = 800) {
|
||||
if (typeof content !== 'string' || content.length === 0) return [];
|
||||
const chunks = [];
|
||||
const paragraphs = content.split(/\n\s*\n/);
|
||||
let buffer = '';
|
||||
let startOffset = 0;
|
||||
for (const para of paragraphs) {
|
||||
// If a single paragraph exceeds the cap, hard-split it by character.
|
||||
if (para.length > maxChunkChars) {
|
||||
// Flush whatever was buffered first.
|
||||
if (buffer) {
|
||||
chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
|
||||
startOffset += buffer.length + 2;
|
||||
buffer = '';
|
||||
}
|
||||
for (let i = 0; i < para.length; i += maxChunkChars) {
|
||||
const slice = para.slice(i, i + maxChunkChars);
|
||||
chunks.push({ start: startOffset, end: startOffset + slice.length, text: slice });
|
||||
startOffset += slice.length;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
const tentative = buffer ? `${buffer}\n\n${para}` : para;
|
||||
if (tentative.length > maxChunkChars && buffer) {
|
||||
chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
|
||||
startOffset += buffer.length + 2; // account for the "\n\n" we split on
|
||||
buffer = para;
|
||||
} else {
|
||||
buffer = tentative;
|
||||
}
|
||||
}
|
||||
if (buffer) chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
|
||||
return chunks;
|
||||
}
|
||||
|
||||
/**
|
||||
* Answer a natural-language question by ranking the workspace's chunks.
|
||||
*
|
||||
* @param {object} args
|
||||
* @param {string} args.question
|
||||
* @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files
|
||||
* @param {number} [args.topK=5] number of chunks to return
|
||||
* @param {number} [args.nowMs=Date.now()]
|
||||
* @returns {{question:string, chunks:Array<{filePath:string, snippet:string, score:number, mtimeMs:number}>}}
|
||||
*/
|
||||
function ask({ question, files, topK = 5, nowMs = Date.now() }) {
|
||||
const cleaned = cleanQuestion(question);
|
||||
if (!cleaned || !Array.isArray(files) || files.length === 0) {
|
||||
return { question: String(question || ''), chunks: [] };
|
||||
}
|
||||
|
||||
// First, find the docs that match at all (cheap, broad pass).
|
||||
const docHits = WorkspaceSearch.search({ query: cleaned, files, limit: 20, nowMs });
|
||||
|
||||
// Then re-rank at the chunk level within those docs.
|
||||
const chunkCorpus = [];
|
||||
for (const hit of docHits) {
|
||||
const file = files.find((f) => f.path === hit.filePath);
|
||||
if (!file || typeof file.content !== 'string') continue;
|
||||
const chunks = chunkDocument(file.content);
|
||||
for (const chunk of chunks) {
|
||||
chunkCorpus.push({
|
||||
path: `${hit.filePath}#${chunk.start}`,
|
||||
content: chunk.text,
|
||||
mtimeMs: file.mtimeMs,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
const chunkHits = WorkspaceSearch.search({
|
||||
query: cleaned,
|
||||
files: chunkCorpus,
|
||||
limit: topK,
|
||||
nowMs,
|
||||
});
|
||||
|
||||
// Translate the per-chunk hits back into the public shape. The fake path
|
||||
// "<file>#<offset>" carries the chunk start; the renderer's existing
|
||||
// file-opened handler can split it on '#' if it wants to deep-link.
|
||||
const chunks = chunkHits.map((h) => {
|
||||
const offsetMatch = /#(\d+)$/.exec(h.filePath);
|
||||
const offset = offsetMatch ? Number(offsetMatch[1]) : 0;
|
||||
return {
|
||||
filePath: h.filePath.replace(/#\d+$/, ''),
|
||||
offset,
|
||||
snippet: h.snippet,
|
||||
score: h.score,
|
||||
mtimeMs: chunkCorpus.find((c) => c.path === h.filePath)?.mtimeMs || 0,
|
||||
};
|
||||
});
|
||||
|
||||
return { question: String(question), chunks };
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
cleanQuestion,
|
||||
chunkDocument,
|
||||
ask,
|
||||
QUESTION_WORDS,
|
||||
};
|
||||
Reference in New Issue
Block a user