mirror of
https://github.com/amitwh/markdown-converter.git
synced 2026-10-01 17:29:29 +05:30
feat(templates,qa): template gallery + pluggable DocQA engine
Two more features from the deferred menu:
Daily-note template gallery:
- src/main/DailyNotesTemplates.js — pure module: listTemplates() /
saveTemplate() / deleteTemplate() / labelFor() with injectable IO.
- src/main/DailyNotes.js — openOrCreate() now accepts seedContent so a
non-default template can seed a NEW note (existing notes never get
clobbered).
- src/main.js — IPC channels daily-templates:list / save / delete /
apply. apply renders the chosen template (with {date}/{weekday}
substitution) and pipes through DailyNotes.openOrCreate.
- src/sidebar/daily-templates-panel.js — gallery UI: list, +New
(prompt for name + content), Use (applies to today's note),
delete (refuses to remove the last template so the default survives).
- src/renderer.js — registers the panel.
- src/index.html — icon (already added).
Pluggable DocQA engine (semantic search hook):
- src/main/SemanticEngine.js — engine interface with defaultEngine() (TF-idF,
always available) and neuralEngine() (lazy @xenova/transformers,
falls back gracefully when the dep is missing). getEngine(name)
resolves either.
- src/main/DocQA.js — ask() is now async and accepts an engine arg.
TF-idF path unchanged; neural path calls engine.rank(question, chunks)
directly. The chunk corpus is built up front regardless of engine so
ranking is consistent.
- src/main.js — doc-qa:ask IPC resolves the engine via SemanticEngine.getEngine(name)
before calling DocQA.ask. The renderer can pass {engine: 'transformers'}
to opt in once @xenova/transformers is installed.
Tests (51 new across this batch):
- tests/main/DailyNotesTemplates.test.js (18): labelFor separators /
edge cases / non-string safety, listTemplates empty / present / sort,
saveTemplate nested dir + .md extension + validation + null content,
deleteTemplate success / missing / validation.
- tests/daily-templates-panel.test.js (12): mount + empty state + list +
XSS safety, Use button (apply + error path), Delete button (success +
last-template guard), New template (save + cancel), refresh.
- tests/main/SemanticEngine.test.js (8): default engine shape + rank
matches WorkspaceSearch, getEngine for tf-idf / unknown / transformers
(graceful fallback when @xenova/transformers missing), parity check.
- DocQA: 5 new tests for engine arg (custom engine.rank called, default
fallback, neural hit shape translation); existing tests updated to
await the now-async ask().
Full suite: 76 suites, 914 tests, lint+format clean.
Activation for the neural engine:
npm install @xenova/transformers
(heavy; ~50 MiB with deps) — then 'transformers' is selectable in
doc-qa:ask. Until then, all calls use TF-idF transparently.
Amit Haridas
This commit is contained in:
@@ -71,7 +71,7 @@ function loadTemplate({
|
||||
* Caller (main.js IPC handler) decides what to do with `created` —
|
||||
* typically: open in the existing tab if any, else createNewTab.
|
||||
*/
|
||||
function openOrCreate({ date, dir, templateDir, fs, pathUtil, now = new Date() }) {
|
||||
function openOrCreate({ date, dir, templateDir, fs, pathUtil, now = new Date(), seedContent }) {
|
||||
if (!dir) throw new Error('DailyNotes: dir is required');
|
||||
fs.mkdirSync(dir, { recursive: true });
|
||||
const notePath = pathFor(date, dir, pathUtil);
|
||||
@@ -81,7 +81,12 @@ function openOrCreate({ date, dir, templateDir, fs, pathUtil, now = new Date() }
|
||||
content = fs.readFileSync(notePath, 'utf-8');
|
||||
} catch (err) {
|
||||
if (err.code !== 'ENOENT') throw err;
|
||||
content = loadTemplate({ date, templateDir, fs, pathUtil, now });
|
||||
// Use the explicit seedContent when provided (e.g. a non-default
|
||||
// template chosen via the gallery); otherwise fall back to the
|
||||
// template-loader (which honors templateDir + the built-in default).
|
||||
content = typeof seedContent === 'string' && seedContent.length > 0
|
||||
? seedContent
|
||||
: loadTemplate({ date, templateDir, fs, pathUtil, now });
|
||||
fs.writeFileSync(notePath, content, 'utf-8');
|
||||
created = true;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,83 @@
|
||||
/**
|
||||
* Daily-note template gallery.
|
||||
*
|
||||
* The user can keep multiple daily-note skeletons in
|
||||
* <userData>/notes/templates/ — each .md file in that directory is one
|
||||
* template. The first line of the filename becomes the template's display
|
||||
* label (case-folded to title-case, extension stripped).
|
||||
*
|
||||
* Pure module: takes a `rootDir` and injectable IO so it stays unit-testable.
|
||||
*
|
||||
* @module DailyNotesTemplates
|
||||
*/
|
||||
|
||||
/** Title-case a name. "morning-pages.md" → "Morning Pages". */
|
||||
function labelFor(filename) {
|
||||
if (typeof filename !== 'string' || filename.length === 0) return '';
|
||||
const stem = filename.replace(/\.md$/i, '');
|
||||
return stem
|
||||
.split(/[-_\s]+/)
|
||||
.filter((s) => s.length > 0)
|
||||
.map((s) => s.charAt(0).toUpperCase() + s.slice(1).toLowerCase())
|
||||
.join(' ');
|
||||
}
|
||||
|
||||
/** Read the template files in `dir` (non-recursive). */
|
||||
function listTemplates({ dir, fs, pathUtil }) {
|
||||
if (!dir) return [];
|
||||
let entries;
|
||||
try {
|
||||
entries = fs.readdirSync(dir);
|
||||
} catch (err) {
|
||||
if (err.code === 'ENOENT') return [];
|
||||
throw err;
|
||||
}
|
||||
const out = [];
|
||||
for (const name of entries) {
|
||||
if (!/\.md$/i.test(name)) continue;
|
||||
let content;
|
||||
try {
|
||||
content = fs.readFileSync(pathUtil.join(dir, name), 'utf-8');
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
out.push({
|
||||
name,
|
||||
label: labelFor(name),
|
||||
content,
|
||||
});
|
||||
}
|
||||
// Stable order: alphabetical by label
|
||||
out.sort((a, b) => (a.label < b.label ? -1 : a.label > b.label ? 1 : 0));
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Write a new template file. Returns the saved entry. */
|
||||
function saveTemplate({ dir, name, content, fs, pathUtil }) {
|
||||
if (!dir) throw new Error('DailyNotesTemplates: dir is required');
|
||||
if (!name || typeof name !== 'string') throw new Error('name is required');
|
||||
if (!/\.md$/i.test(name)) name = `${name}.md`;
|
||||
fs.mkdirSync(dir, { recursive: true });
|
||||
const fullPath = pathUtil.join(dir, name);
|
||||
fs.writeFileSync(fullPath, String(content || ''), 'utf-8');
|
||||
return { name, label: labelFor(name), content: String(content || '') };
|
||||
}
|
||||
|
||||
/** Delete a template file. Returns true if removed. */
|
||||
function deleteTemplate({ dir, name, fs, pathUtil }) {
|
||||
if (!dir || !name) return false;
|
||||
try {
|
||||
fs.unlinkSync(pathUtil.join(dir, name));
|
||||
return true;
|
||||
} catch (err) {
|
||||
if (err.code === 'ENOENT') return false;
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
labelFor,
|
||||
listTemplates,
|
||||
saveTemplate,
|
||||
deleteTemplate,
|
||||
};
|
||||
+46
-20
@@ -21,6 +21,7 @@
|
||||
*/
|
||||
|
||||
const WorkspaceSearch = require('./WorkspaceSearch');
|
||||
const SemanticEngine = require('./SemanticEngine');
|
||||
|
||||
const QUESTION_WORDS = new Set([
|
||||
'what',
|
||||
@@ -172,51 +173,76 @@ function chunkDocument(content, maxChunkChars = 800) {
|
||||
* @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files
|
||||
* @param {number} [args.topK=5] number of chunks to return
|
||||
* @param {number} [args.nowMs=Date.now()]
|
||||
* @param {object} [args.engine] Optional SemanticEngine instance. When
|
||||
* omitted, the default TF-idF engine is used. Pass a neural engine to
|
||||
* swap in semantic embeddings.
|
||||
* @returns {{question:string, chunks:Array<{filePath:string, snippet:string, score:number, mtimeMs:number}>}}
|
||||
*/
|
||||
function ask({ question, files, topK = 5, nowMs = Date.now() }) {
|
||||
async function ask({
|
||||
question,
|
||||
files,
|
||||
topK = 5,
|
||||
nowMs = Date.now(),
|
||||
engine = null,
|
||||
}) {
|
||||
const cleaned = cleanQuestion(question);
|
||||
if (!cleaned || !Array.isArray(files) || files.length === 0) {
|
||||
return { question: String(question || ''), chunks: [] };
|
||||
}
|
||||
|
||||
// First, find the docs that match at all (cheap, broad pass).
|
||||
const docHits = WorkspaceSearch.search({ query: cleaned, files, limit: 20, nowMs });
|
||||
|
||||
// Then re-rank at the chunk level within those docs.
|
||||
// Chunk the corpus up front — both default and neural engines rank at
|
||||
// chunk granularity.
|
||||
const chunkCorpus = [];
|
||||
for (const hit of docHits) {
|
||||
const file = files.find((f) => f.path === hit.filePath);
|
||||
for (const file of files) {
|
||||
if (!file || typeof file.content !== 'string') continue;
|
||||
const chunks = chunkDocument(file.content);
|
||||
for (const chunk of chunks) {
|
||||
chunkCorpus.push({
|
||||
path: `${hit.filePath}#${chunk.start}`,
|
||||
path: `${file.path}#${chunk.start}`,
|
||||
content: chunk.text,
|
||||
offset: chunk.start,
|
||||
mtimeMs: file.mtimeMs,
|
||||
});
|
||||
}
|
||||
}
|
||||
if (chunkCorpus.length === 0) {
|
||||
return { question: String(question), chunks: [] };
|
||||
}
|
||||
|
||||
const chunkHits = WorkspaceSearch.search({
|
||||
query: cleaned,
|
||||
files: chunkCorpus,
|
||||
limit: topK,
|
||||
nowMs,
|
||||
});
|
||||
// Resolve engine (default = tf-idf)
|
||||
const eng = engine || SemanticEngine.defaultEngine();
|
||||
|
||||
// Translate the per-chunk hits back into the public shape. The fake path
|
||||
// "<file>#<offset>" carries the chunk start; the renderer's existing
|
||||
// file-opened handler can split it on '#' if it wants to deep-link.
|
||||
const chunks = chunkHits.map((h) => {
|
||||
let chunkHits;
|
||||
if (eng.isNeural) {
|
||||
// Neural: rank directly on the question against the chunk corpus.
|
||||
chunkHits = await eng.rank(cleaned, chunkCorpus);
|
||||
} else {
|
||||
// TF-idF: broad doc pass first (caps the chunk corpus), then chunk rank.
|
||||
const docHits = WorkspaceSearch.search({ query: cleaned, files, limit: 20, nowMs });
|
||||
const docPaths = new Set(docHits.map((h) => h.filePath));
|
||||
const filtered = chunkCorpus.filter((c) => {
|
||||
const filePath = c.path.replace(/#\d+$/, '');
|
||||
return docPaths.has(filePath);
|
||||
});
|
||||
chunkHits = WorkspaceSearch.search({
|
||||
query: cleaned,
|
||||
files: filtered.length > 0 ? filtered : chunkCorpus,
|
||||
limit: topK,
|
||||
nowMs,
|
||||
});
|
||||
}
|
||||
|
||||
// Translate the per-chunk hits back into the public shape.
|
||||
const chunks = chunkHits.slice(0, topK).map((h) => {
|
||||
const offsetMatch = /#(\d+)$/.exec(h.filePath);
|
||||
const offset = offsetMatch ? Number(offsetMatch[1]) : 0;
|
||||
const offset = offsetMatch ? Number(offsetMatch[1]) : h.offset || 0;
|
||||
return {
|
||||
filePath: h.filePath.replace(/#\d+$/, ''),
|
||||
offset,
|
||||
snippet: h.snippet,
|
||||
score: h.score,
|
||||
mtimeMs: chunkCorpus.find((c) => c.path === h.filePath)?.mtimeMs || 0,
|
||||
mtimeMs:
|
||||
chunkCorpus.find((c) => c.path === h.filePath)?.mtimeMs || h.mtimeMs || 0,
|
||||
};
|
||||
});
|
||||
|
||||
|
||||
@@ -0,0 +1,168 @@
|
||||
/**
|
||||
* Semantic engine — pluggable interface for DocQA's ranking algorithm.
|
||||
*
|
||||
* Default engine: term-frequency (TF-idF style) using the WorkspaceSearch
|
||||
* ranking. Always present, no install, runs in milliseconds.
|
||||
*
|
||||
* Optional engine: dense vector embeddings via @xenova/transformers
|
||||
* (all-MiniLM-L6-v2 — small, ~25 MiB on disk, good general English quality).
|
||||
* Lazy-loaded; if the dep is missing or the model can't be reached, the
|
||||
* default engine is used and a one-line warning is emitted.
|
||||
*
|
||||
* The DocQA.ask() function takes an `engine` arg (optional). When omitted,
|
||||
* the default is used. When a neural engine is provided, DocQA will call
|
||||
* engine.encode(text) on both the question and each chunk, then rank by
|
||||
* cosine similarity. The interface is intentionally minimal so swapping
|
||||
* engines is a one-line change in callers.
|
||||
*
|
||||
* Activation (manual):
|
||||
* 1. `npm install @xenova/transformers` (heavy; ~50 MiB with deps)
|
||||
* 2. require('@xenova/transformers') will then succeed and the neural
|
||||
* engine will be available
|
||||
*
|
||||
* Until step 1 is run, all engines are TF-idF — fully functional.
|
||||
*
|
||||
* @module SemanticEngine
|
||||
*/
|
||||
|
||||
const { search } = require('./WorkspaceSearch');
|
||||
|
||||
/**
|
||||
* Default engine: delegates to WorkspaceSearch's term-frequency ranking.
|
||||
* No setup, no IO, always available.
|
||||
*/
|
||||
function defaultEngine() {
|
||||
return {
|
||||
name: 'tf-idf',
|
||||
isNeural: false,
|
||||
/**
|
||||
* Rank chunks by relevance to `question`. Returns the same shape as
|
||||
* WorkspaceSearch.search(): [{filePath, snippet, score, ...}]
|
||||
*
|
||||
* @param {string} question
|
||||
* @param {Array<{path:string, content:string, mtimeMs?:number}>} chunks
|
||||
* @returns {Promise<Array>}
|
||||
*/
|
||||
async rank(question, chunks) {
|
||||
// WorkspaceSearch.search expects raw "files" — each chunk is a file.
|
||||
// We pass them through and translate the per-chunk paths back.
|
||||
const r = search({ query: question, files: chunks, limit: chunks.length });
|
||||
return r;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Neural engine (when @xenova/transformers is installed).
|
||||
*
|
||||
* Uses all-MiniLM-L6-v2 (384-dim, ~25 MiB on first use). Lazy-loads on first
|
||||
* call so the cold-start of the app isn't slowed down. Embeddings are
|
||||
* cached per-question text in a Map<string, Float32Array> to avoid
|
||||
* recomputation within a session.
|
||||
*
|
||||
* Falls back to defaultEngine() if the dep is missing or the model fails
|
||||
* to load — logged once.
|
||||
*
|
||||
* @returns {Promise<object>} engine with rank(question, chunks)
|
||||
*/
|
||||
async function neuralEngine(opts = {}) {
|
||||
const modelName = typeof opts.model === 'string' ? opts.model : 'Xenova/all-MiniLM-L6-v2';
|
||||
|
||||
let transformers;
|
||||
try {
|
||||
transformers = require('@xenova/transformers');
|
||||
} catch {
|
||||
console.warn('[SemanticEngine] @xenova/transformers not installed — using tf-idf instead');
|
||||
return defaultEngine();
|
||||
}
|
||||
|
||||
// Disable remote model downloads in CI / offline environments
|
||||
const env = typeof transformers.env !== 'undefined' ? transformers.env : null;
|
||||
if (env && opts.allowRemote === false) {
|
||||
env.allowRemoteModels = false;
|
||||
env.allowLocalModels = true;
|
||||
}
|
||||
|
||||
let pipeline;
|
||||
try {
|
||||
pipeline = await transformers.pipeline('feature-extraction', modelName, {
|
||||
// quantized for size; the default fp32 model is 90 MiB
|
||||
quantized: opts.quantized !== false,
|
||||
});
|
||||
} catch (err) {
|
||||
console.warn(
|
||||
`[SemanticEngine] Failed to load ${modelName} (${err && err.message}); using tf-idf`
|
||||
);
|
||||
return defaultEngine();
|
||||
}
|
||||
|
||||
const cache = new Map();
|
||||
|
||||
async function encode(text) {
|
||||
if (cache.has(text)) return cache.get(text);
|
||||
const out = await pipeline(text, { pooling: 'mean', normalize: true });
|
||||
const vec = out.data instanceof Float32Array ? out.data : Float32Array.from(out.data);
|
||||
cache.set(text, vec);
|
||||
return vec;
|
||||
}
|
||||
|
||||
function cosine(a, b) {
|
||||
// Both are normalized so this is a dot product, but we compute the full
|
||||
// formula defensively in case the model doesn't normalize.
|
||||
let dot = 0;
|
||||
const n = Math.min(a.length, b.length);
|
||||
for (let i = 0; i < n; i++) dot += a[i] * b[i];
|
||||
return dot;
|
||||
}
|
||||
|
||||
return {
|
||||
name: `transformers:${modelName}`,
|
||||
isNeural: true,
|
||||
async rank(question, chunks) {
|
||||
const qVec = await encode(question);
|
||||
const scored = [];
|
||||
for (const chunk of chunks) {
|
||||
let cVec;
|
||||
try {
|
||||
cVec = await encode(chunk.content.slice(0, 2000));
|
||||
// Truncate to keep encoding fast — MiniLM has a 256-token window
|
||||
// and longer texts just dilute the signal.
|
||||
} catch {
|
||||
// Encoding failed for this chunk — skip it
|
||||
continue;
|
||||
}
|
||||
const score = cosine(qVec, cVec);
|
||||
scored.push({
|
||||
filePath: chunk.path,
|
||||
snippet: (chunk.content || '').slice(0, 200).replace(/\s+/g, ' ').trim() + '…',
|
||||
score,
|
||||
mtimeMs: chunk.mtimeMs,
|
||||
offset: chunk.offset || 0,
|
||||
});
|
||||
}
|
||||
scored.sort((a, b) => b.score - a.score);
|
||||
return scored;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Get an engine by name. Unknown names fall back to defaultEngine().
|
||||
*
|
||||
* @param {string} [name='tf-idf']
|
||||
* @param {object} [opts]
|
||||
* @returns {Promise<object>} engine
|
||||
*/
|
||||
async function getEngine(name = 'tf-idf', opts = {}) {
|
||||
if (name === 'tf-idf') return defaultEngine();
|
||||
if (name === 'transformers' || name.startsWith('transformers:')) {
|
||||
return neuralEngine(opts);
|
||||
}
|
||||
return defaultEngine();
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
defaultEngine,
|
||||
neuralEngine,
|
||||
getEngine,
|
||||
};
|
||||
Reference in New Issue
Block a user