Files
markdown-converter/src/main/DocQA.js
T
amitwh c3fe72bcae feat(templates,qa): template gallery + pluggable DocQA engine
Two more features from the deferred menu:

Daily-note template gallery:
- src/main/DailyNotesTemplates.js — pure module: listTemplates() /
  saveTemplate() / deleteTemplate() / labelFor() with injectable IO.
- src/main/DailyNotes.js — openOrCreate() now accepts seedContent so a
  non-default template can seed a NEW note (existing notes never get
  clobbered).
- src/main.js — IPC channels daily-templates:list / save / delete /
  apply. apply renders the chosen template (with {date}/{weekday}
  substitution) and pipes through DailyNotes.openOrCreate.
- src/sidebar/daily-templates-panel.js — gallery UI: list, +New
  (prompt for name + content), Use (applies to today's note),
  delete (refuses to remove the last template so the default survives).
- src/renderer.js — registers the panel.
- src/index.html — icon (already added).

Pluggable DocQA engine (semantic search hook):
- src/main/SemanticEngine.js — engine interface with defaultEngine() (TF-idF,
  always available) and neuralEngine() (lazy @xenova/transformers,
  falls back gracefully when the dep is missing). getEngine(name)
  resolves either.
- src/main/DocQA.js — ask() is now async and accepts an engine arg.
  TF-idF path unchanged; neural path calls engine.rank(question, chunks)
  directly. The chunk corpus is built up front regardless of engine so
  ranking is consistent.
- src/main.js — doc-qa:ask IPC resolves the engine via SemanticEngine.getEngine(name)
  before calling DocQA.ask. The renderer can pass {engine: 'transformers'}
  to opt in once @xenova/transformers is installed.

Tests (51 new across this batch):
- tests/main/DailyNotesTemplates.test.js (18): labelFor separators /
  edge cases / non-string safety, listTemplates empty / present / sort,
  saveTemplate nested dir + .md extension + validation + null content,
  deleteTemplate success / missing / validation.
- tests/daily-templates-panel.test.js (12): mount + empty state + list +
  XSS safety, Use button (apply + error path), Delete button (success +
  last-template guard), New template (save + cancel), refresh.
- tests/main/SemanticEngine.test.js (8): default engine shape + rank
  matches WorkspaceSearch, getEngine for tf-idf / unknown / transformers
  (graceful fallback when @xenova/transformers missing), parity check.
- DocQA: 5 new tests for engine arg (custom engine.rank called, default
  fallback, neural hit shape translation); existing tests updated to
  await the now-async ask().

Full suite: 76 suites, 914 tests, lint+format clean.

Activation for the neural engine:
  npm install @xenova/transformers
  (heavy; ~50 MiB with deps) — then 'transformers' is selectable in
  doc-qa:ask. Until then, all calls use TF-idF transparently.

Amit Haridas
2026-09-14 11:57:32 +05:30

258 lines
6.9 KiB
JavaScript

/**
* Doc-aware Q&A — thin wrapper over WorkspaceSearch tuned for questions.
*
* User asks "what did I write about rust async last week?" and gets back the
* top-K chunks from their workspace ranked by relevance. No neural model, no
* API key — the same TF-idf-style ranking WorkspaceSearch uses, with a few
* QA-specific tweaks:
*
* - Question words (what/how/why/when/where/who/which/does/is/are/...) are
* stripped before ranking — they don't carry meaning, just grammar
* - The top chunks are split out as independent results so the renderer
* can show "3 passages from 2 files" instead of one giant result
* - Recency weight is doubled: "last week" implies the user wants fresh
* content, not old material
*
* Future upgrade path: swap `WorkspaceSearch.search` for a vector-similarity
* call (transformers.js in the renderer, or a sidecar process). The public
* shape — {question, chunks:[{filePath, snippet, score}]} — stays the same.
*
* @module DocQA
*/
const WorkspaceSearch = require('./WorkspaceSearch');
const SemanticEngine = require('./SemanticEngine');
const QUESTION_WORDS = new Set([
'what',
'when',
'where',
'who',
'whom',
'whose',
'why',
'how',
'which',
'whether',
'does',
'do',
'did',
'is',
'are',
'was',
'were',
'be',
'been',
'being',
'have',
'has',
'had',
'can',
'could',
'would',
'should',
'will',
'shall',
'may',
'might',
'i',
'me',
'my',
'we',
'our',
'you',
'your',
'they',
'them',
'their',
'a',
'an',
'the',
'and',
'or',
'but',
'so',
'of',
'to',
'in',
'on',
'at',
'for',
'with',
'about',
'into',
'from',
'by',
'as',
'this',
'that',
'these',
'those',
'it',
'its',
'write',
'wrote',
'written',
'read',
'think',
'know',
'find',
'show',
'tell',
'say',
'see',
'use',
'used',
]);
/**
* Strip question words from a raw question so WorkspaceSearch's bare-term
* matching isn't drowned in grammar noise. Quoted phrases, #tags, and
* @wikilinks pass through unchanged.
*/
function cleanQuestion(raw) {
if (typeof raw !== 'string') return '';
// First pull out facets and phrases so we don't strip them
const facets = [];
const working = raw.replace(/[#@][A-Za-z0-9_-]+|"[^"]+"/g, (m) => {
facets.push(m);
return ' ';
});
const words = working.split(/\s+/).filter((w) => {
const lower = w.toLowerCase().replace(/[^a-z0-9-]/g, '');
return lower.length >= 2 && !QUESTION_WORDS.has(lower);
});
return [...words, ...facets].join(' ').trim();
}
/**
* Split a document into rough passages (~800 chars or paragraph break,
* whichever comes first). Returns an array of {start, end, text} so the
* caller can build snippets. Pure.
*/
function chunkDocument(content, maxChunkChars = 800) {
if (typeof content !== 'string' || content.length === 0) return [];
const chunks = [];
const paragraphs = content.split(/\n\s*\n/);
let buffer = '';
let startOffset = 0;
for (const para of paragraphs) {
// If a single paragraph exceeds the cap, hard-split it by character.
if (para.length > maxChunkChars) {
// Flush whatever was buffered first.
if (buffer) {
chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
startOffset += buffer.length + 2;
buffer = '';
}
for (let i = 0; i < para.length; i += maxChunkChars) {
const slice = para.slice(i, i + maxChunkChars);
chunks.push({ start: startOffset, end: startOffset + slice.length, text: slice });
startOffset += slice.length;
}
continue;
}
const tentative = buffer ? `${buffer}\n\n${para}` : para;
if (tentative.length > maxChunkChars && buffer) {
chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
startOffset += buffer.length + 2; // account for the "\n\n" we split on
buffer = para;
} else {
buffer = tentative;
}
}
if (buffer) chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
return chunks;
}
/**
* Answer a natural-language question by ranking the workspace's chunks.
*
* @param {object} args
* @param {string} args.question
* @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files
* @param {number} [args.topK=5] number of chunks to return
* @param {number} [args.nowMs=Date.now()]
* @param {object} [args.engine] Optional SemanticEngine instance. When
* omitted, the default TF-idF engine is used. Pass a neural engine to
* swap in semantic embeddings.
* @returns {{question:string, chunks:Array<{filePath:string, snippet:string, score:number, mtimeMs:number}>}}
*/
async function ask({
question,
files,
topK = 5,
nowMs = Date.now(),
engine = null,
}) {
const cleaned = cleanQuestion(question);
if (!cleaned || !Array.isArray(files) || files.length === 0) {
return { question: String(question || ''), chunks: [] };
}
// Chunk the corpus up front — both default and neural engines rank at
// chunk granularity.
const chunkCorpus = [];
for (const file of files) {
if (!file || typeof file.content !== 'string') continue;
const chunks = chunkDocument(file.content);
for (const chunk of chunks) {
chunkCorpus.push({
path: `${file.path}#${chunk.start}`,
content: chunk.text,
offset: chunk.start,
mtimeMs: file.mtimeMs,
});
}
}
if (chunkCorpus.length === 0) {
return { question: String(question), chunks: [] };
}
// Resolve engine (default = tf-idf)
const eng = engine || SemanticEngine.defaultEngine();
let chunkHits;
if (eng.isNeural) {
// Neural: rank directly on the question against the chunk corpus.
chunkHits = await eng.rank(cleaned, chunkCorpus);
} else {
// TF-idF: broad doc pass first (caps the chunk corpus), then chunk rank.
const docHits = WorkspaceSearch.search({ query: cleaned, files, limit: 20, nowMs });
const docPaths = new Set(docHits.map((h) => h.filePath));
const filtered = chunkCorpus.filter((c) => {
const filePath = c.path.replace(/#\d+$/, '');
return docPaths.has(filePath);
});
chunkHits = WorkspaceSearch.search({
query: cleaned,
files: filtered.length > 0 ? filtered : chunkCorpus,
limit: topK,
nowMs,
});
}
// Translate the per-chunk hits back into the public shape.
const chunks = chunkHits.slice(0, topK).map((h) => {
const offsetMatch = /#(\d+)$/.exec(h.filePath);
const offset = offsetMatch ? Number(offsetMatch[1]) : h.offset || 0;
return {
filePath: h.filePath.replace(/#\d+$/, ''),
offset,
snippet: h.snippet,
score: h.score,
mtimeMs:
chunkCorpus.find((c) => c.path === h.filePath)?.mtimeMs || h.mtimeMs || 0,
};
});
return { question: String(question), chunks };
}
module.exports = {
cleanQuestion,
chunkDocument,
ask,
QUESTION_WORDS,
};