mirror of
https://github.com/amitwh/markdown-converter.git
synced 2026-10-01 09:19:34 +05:30
12 commits on master since v4.7.1 baseline, taking tests 524 → 914 across 27 new suites. This commit: - Bumps package.json + README version to v4.8.0 (additive features → semver minor) - Applies Prettier formatting to all files touched during the overnight session so release build is reproducible from a clean repo New in v4.8.0: - PDF encryption close-out (D1) - KaTeX rendering test coverage - Autosave buffer + crash recovery banner - Daily notes + workspace search + doc-aware Q&A - Search sidebar panel + Ask mode - Status bar: word count, reading time, Flesch-Kincaid grade - Footnote hover preview - Smart paste (URL → link, CSV/TSV → table) - Daily notes menu item + sidebar panel + Q&A deep-link - Daily templates gallery - Pluggable DocQA engine (TF-idF default + lazy neural embeddings) Amit Haridas
251 lines
6.8 KiB
JavaScript
251 lines
6.8 KiB
JavaScript
/**
|
|
* Doc-aware Q&A — thin wrapper over WorkspaceSearch tuned for questions.
|
|
*
|
|
* User asks "what did I write about rust async last week?" and gets back the
|
|
* top-K chunks from their workspace ranked by relevance. No neural model, no
|
|
* API key — the same TF-idf-style ranking WorkspaceSearch uses, with a few
|
|
* QA-specific tweaks:
|
|
*
|
|
* - Question words (what/how/why/when/where/who/which/does/is/are/...) are
|
|
* stripped before ranking — they don't carry meaning, just grammar
|
|
* - The top chunks are split out as independent results so the renderer
|
|
* can show "3 passages from 2 files" instead of one giant result
|
|
* - Recency weight is doubled: "last week" implies the user wants fresh
|
|
* content, not old material
|
|
*
|
|
* Future upgrade path: swap `WorkspaceSearch.search` for a vector-similarity
|
|
* call (transformers.js in the renderer, or a sidecar process). The public
|
|
* shape — {question, chunks:[{filePath, snippet, score}]} — stays the same.
|
|
*
|
|
* @module DocQA
|
|
*/
|
|
|
|
const WorkspaceSearch = require('./WorkspaceSearch');
|
|
const SemanticEngine = require('./SemanticEngine');
|
|
|
|
const QUESTION_WORDS = new Set([
|
|
'what',
|
|
'when',
|
|
'where',
|
|
'who',
|
|
'whom',
|
|
'whose',
|
|
'why',
|
|
'how',
|
|
'which',
|
|
'whether',
|
|
'does',
|
|
'do',
|
|
'did',
|
|
'is',
|
|
'are',
|
|
'was',
|
|
'were',
|
|
'be',
|
|
'been',
|
|
'being',
|
|
'have',
|
|
'has',
|
|
'had',
|
|
'can',
|
|
'could',
|
|
'would',
|
|
'should',
|
|
'will',
|
|
'shall',
|
|
'may',
|
|
'might',
|
|
'i',
|
|
'me',
|
|
'my',
|
|
'we',
|
|
'our',
|
|
'you',
|
|
'your',
|
|
'they',
|
|
'them',
|
|
'their',
|
|
'a',
|
|
'an',
|
|
'the',
|
|
'and',
|
|
'or',
|
|
'but',
|
|
'so',
|
|
'of',
|
|
'to',
|
|
'in',
|
|
'on',
|
|
'at',
|
|
'for',
|
|
'with',
|
|
'about',
|
|
'into',
|
|
'from',
|
|
'by',
|
|
'as',
|
|
'this',
|
|
'that',
|
|
'these',
|
|
'those',
|
|
'it',
|
|
'its',
|
|
'write',
|
|
'wrote',
|
|
'written',
|
|
'read',
|
|
'think',
|
|
'know',
|
|
'find',
|
|
'show',
|
|
'tell',
|
|
'say',
|
|
'see',
|
|
'use',
|
|
'used',
|
|
]);
|
|
|
|
/**
|
|
* Strip question words from a raw question so WorkspaceSearch's bare-term
|
|
* matching isn't drowned in grammar noise. Quoted phrases, #tags, and
|
|
* @wikilinks pass through unchanged.
|
|
*/
|
|
function cleanQuestion(raw) {
|
|
if (typeof raw !== 'string') return '';
|
|
// First pull out facets and phrases so we don't strip them
|
|
const facets = [];
|
|
const working = raw.replace(/[#@][A-Za-z0-9_-]+|"[^"]+"/g, (m) => {
|
|
facets.push(m);
|
|
return ' ';
|
|
});
|
|
const words = working.split(/\s+/).filter((w) => {
|
|
const lower = w.toLowerCase().replace(/[^a-z0-9-]/g, '');
|
|
return lower.length >= 2 && !QUESTION_WORDS.has(lower);
|
|
});
|
|
return [...words, ...facets].join(' ').trim();
|
|
}
|
|
|
|
/**
|
|
* Split a document into rough passages (~800 chars or paragraph break,
|
|
* whichever comes first). Returns an array of {start, end, text} so the
|
|
* caller can build snippets. Pure.
|
|
*/
|
|
function chunkDocument(content, maxChunkChars = 800) {
|
|
if (typeof content !== 'string' || content.length === 0) return [];
|
|
const chunks = [];
|
|
const paragraphs = content.split(/\n\s*\n/);
|
|
let buffer = '';
|
|
let startOffset = 0;
|
|
for (const para of paragraphs) {
|
|
// If a single paragraph exceeds the cap, hard-split it by character.
|
|
if (para.length > maxChunkChars) {
|
|
// Flush whatever was buffered first.
|
|
if (buffer) {
|
|
chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
|
|
startOffset += buffer.length + 2;
|
|
buffer = '';
|
|
}
|
|
for (let i = 0; i < para.length; i += maxChunkChars) {
|
|
const slice = para.slice(i, i + maxChunkChars);
|
|
chunks.push({ start: startOffset, end: startOffset + slice.length, text: slice });
|
|
startOffset += slice.length;
|
|
}
|
|
continue;
|
|
}
|
|
const tentative = buffer ? `${buffer}\n\n${para}` : para;
|
|
if (tentative.length > maxChunkChars && buffer) {
|
|
chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
|
|
startOffset += buffer.length + 2; // account for the "\n\n" we split on
|
|
buffer = para;
|
|
} else {
|
|
buffer = tentative;
|
|
}
|
|
}
|
|
if (buffer) chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
|
|
return chunks;
|
|
}
|
|
|
|
/**
|
|
* Answer a natural-language question by ranking the workspace's chunks.
|
|
*
|
|
* @param {object} args
|
|
* @param {string} args.question
|
|
* @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files
|
|
* @param {number} [args.topK=5] number of chunks to return
|
|
* @param {number} [args.nowMs=Date.now()]
|
|
* @param {object} [args.engine] Optional SemanticEngine instance. When
|
|
* omitted, the default TF-idF engine is used. Pass a neural engine to
|
|
* swap in semantic embeddings.
|
|
* @returns {{question:string, chunks:Array<{filePath:string, snippet:string, score:number, mtimeMs:number}>}}
|
|
*/
|
|
async function ask({ question, files, topK = 5, nowMs = Date.now(), engine = null }) {
|
|
const cleaned = cleanQuestion(question);
|
|
if (!cleaned || !Array.isArray(files) || files.length === 0) {
|
|
return { question: String(question || ''), chunks: [] };
|
|
}
|
|
|
|
// Chunk the corpus up front — both default and neural engines rank at
|
|
// chunk granularity.
|
|
const chunkCorpus = [];
|
|
for (const file of files) {
|
|
if (!file || typeof file.content !== 'string') continue;
|
|
const chunks = chunkDocument(file.content);
|
|
for (const chunk of chunks) {
|
|
chunkCorpus.push({
|
|
path: `${file.path}#${chunk.start}`,
|
|
content: chunk.text,
|
|
offset: chunk.start,
|
|
mtimeMs: file.mtimeMs,
|
|
});
|
|
}
|
|
}
|
|
if (chunkCorpus.length === 0) {
|
|
return { question: String(question), chunks: [] };
|
|
}
|
|
|
|
// Resolve engine (default = tf-idf)
|
|
const eng = engine || SemanticEngine.defaultEngine();
|
|
|
|
let chunkHits;
|
|
if (eng.isNeural) {
|
|
// Neural: rank directly on the question against the chunk corpus.
|
|
chunkHits = await eng.rank(cleaned, chunkCorpus);
|
|
} else {
|
|
// TF-idF: broad doc pass first (caps the chunk corpus), then chunk rank.
|
|
const docHits = WorkspaceSearch.search({ query: cleaned, files, limit: 20, nowMs });
|
|
const docPaths = new Set(docHits.map((h) => h.filePath));
|
|
const filtered = chunkCorpus.filter((c) => {
|
|
const filePath = c.path.replace(/#\d+$/, '');
|
|
return docPaths.has(filePath);
|
|
});
|
|
chunkHits = WorkspaceSearch.search({
|
|
query: cleaned,
|
|
files: filtered.length > 0 ? filtered : chunkCorpus,
|
|
limit: topK,
|
|
nowMs,
|
|
});
|
|
}
|
|
|
|
// Translate the per-chunk hits back into the public shape.
|
|
const chunks = chunkHits.slice(0, topK).map((h) => {
|
|
const offsetMatch = /#(\d+)$/.exec(h.filePath);
|
|
const offset = offsetMatch ? Number(offsetMatch[1]) : h.offset || 0;
|
|
return {
|
|
filePath: h.filePath.replace(/#\d+$/, ''),
|
|
offset,
|
|
snippet: h.snippet,
|
|
score: h.score,
|
|
mtimeMs: chunkCorpus.find((c) => c.path === h.filePath)?.mtimeMs || h.mtimeMs || 0,
|
|
};
|
|
});
|
|
|
|
return { question: String(question), chunks };
|
|
}
|
|
|
|
module.exports = {
|
|
cleanQuestion,
|
|
chunkDocument,
|
|
ask,
|
|
QUESTION_WORDS,
|
|
};
|