mirror of
https://github.com/amitwh/markdown-converter.git
synced 2026-10-01 17:29:29 +05:30
feat(pkm): daily notes + workspace search + doc-aware Q&A
Three more features from the brainstorm menu, built on a shared search
algorithm so the codebase stays small.
- src/main/DailyNotes.js — Zettelkasten-style helper. One YYYY-MM-DD.md
per local date under <userData>/notes/daily/; loads skeleton from
<userData>/notes/templates/daily.md when present (built-in default
otherwise). openOrCreate never clobbers existing content.
- src/main/WorkspaceSearch.js — tag/wikilink-aware content search.
Pure module, injectable-IO tested. Query grammar: bare words,
#tag, @wikilink, "quoted phrases". Facets weight +3 each; prose
terms +1/occurrence capped at 5; edits within 7 days get a recency
nudge. Returns ranked results with snippets.
- src/main/DocQA.js — chunk-level Q&A wrapper over WorkspaceSearch.
cleanQuestion strips question words (what/how/why/...) and verb
noise (write/read/show/tell/...) so they don't drown the ranking.
Returns top-K passages instead of whole-file hits — multiple chunks
from the same file can appear in the answer.
- src/main.js — IPC: daily-notes:open-today, daily-notes:list,
workspace-search:query, doc-qa:ask. Path validation through the
existing validatePath gate; a global Ctrl+Alt+D shortcut creates
today's daily note from anywhere.
- src/preload.js — all four channels added to ALLOWED_SEND_CHANNELS.
Tests (56 new across the three modules):
- tests/main/DailyNotes.test.js (15): dateKey formatting, pathFor,
template load + {date}/{weekday} substitution, openOrCreate +
no-clobber, nested-dir creation, listExisting filtering,
isValidDir rejects NUL/non-string.
- tests/main/WorkspaceSearch.test.js (25): parseQuery grammar,
hasTag/hasWikilink word boundaries, scoreDocument scoring,
per-term spam cap, recency nudge, search ranking + limit +
empty-query short-circuit, bad-input safety.
- tests/main/DocQA.test.js (16): cleanQuestion stripping + facet
preservation, chunkDocument paragraph + hard-split, ask()
top-K, recency tiebreaker, missing-files fallback.
Full suite: 748 tests pass, 66 suites, lint+format clean.
Amit Haridas
This commit is contained in:
@@ -0,0 +1,130 @@
|
||||
/**
|
||||
* Daily notes — Zettelkasten/journal entry helper.
|
||||
*
|
||||
* Convention: a folder of YYYY-MM-DD.md files (PKM/Obsidian/Logseq style).
|
||||
* `pathFor(date, dir)` returns the absolute path for a given date. The
|
||||
* template loader returns a starting skeleton the user can fill in; if the
|
||||
* file already exists, we return its current content instead (no clobber).
|
||||
*
|
||||
* Implementation is pure (no fs/IPC coupling) — main.js wires the IO and
|
||||
* template path. Tests inject a virtual fs + path util.
|
||||
*/
|
||||
|
||||
const DEFAULT_TEMPLATE_NAME = 'daily.md';
|
||||
|
||||
/**
|
||||
* Format a Date as YYYY-MM-DD in local time (matches the on-disk filename).
|
||||
* Returns 'YYYY-MM-DD' string. Today (no arg) defaults to new Date().
|
||||
*/
|
||||
function dateKey(date = new Date()) {
|
||||
const y = date.getFullYear();
|
||||
const m = String(date.getMonth() + 1).padStart(2, '0');
|
||||
const d = String(date.getDate()).padStart(2, '0');
|
||||
return `${y}-${m}-${d}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Absolute path for the daily note of `date` inside `dir`. Pure.
|
||||
*/
|
||||
function pathFor(date, dir, pathUtil) {
|
||||
return pathUtil.join(dir, `${dateKey(date)}.md`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the daily note skeleton. Looks in `templateDir` for a file named
|
||||
* `templateName` (default 'daily.md'); if missing, returns a built-in
|
||||
* default so the feature works out-of-the-box without setup.
|
||||
*
|
||||
* Returns the template string with any {date}/{weekday} placeholders
|
||||
* substituted.
|
||||
*/
|
||||
function loadTemplate({
|
||||
date,
|
||||
templateDir,
|
||||
templateName = DEFAULT_TEMPLATE_NAME,
|
||||
fs,
|
||||
pathUtil,
|
||||
now = new Date(),
|
||||
}) {
|
||||
let body = `# ${dateKey(date)}\n\n## Notes\n\n`;
|
||||
if (templateDir && fs) {
|
||||
const templatePath = pathUtil.join(templateDir, templateName);
|
||||
try {
|
||||
body = fs.readFileSync(templatePath, 'utf-8');
|
||||
} catch {
|
||||
/* fall through to default */
|
||||
}
|
||||
}
|
||||
return body
|
||||
.replace(/\{date\}/g, dateKey(date))
|
||||
.replace(/\{weekday\}/g, now.toLocaleDateString('en-US', { weekday: 'long' }));
|
||||
}
|
||||
|
||||
/**
|
||||
* Open or create the daily note for `date` inside `dir`. Returns
|
||||
* { path, content, created: boolean }.
|
||||
*
|
||||
* Behavior:
|
||||
* - If <dir>/<date>.md exists, return its content (created: false).
|
||||
* - Otherwise load the template, write it, and return it (created: true).
|
||||
*
|
||||
* Caller (main.js IPC handler) decides what to do with `created` —
|
||||
* typically: open in the existing tab if any, else createNewTab.
|
||||
*/
|
||||
function openOrCreate({ date, dir, templateDir, fs, pathUtil, now = new Date() }) {
|
||||
if (!dir) throw new Error('DailyNotes: dir is required');
|
||||
fs.mkdirSync(dir, { recursive: true });
|
||||
const notePath = pathFor(date, dir, pathUtil);
|
||||
let content;
|
||||
let created = false;
|
||||
try {
|
||||
content = fs.readFileSync(notePath, 'utf-8');
|
||||
} catch (err) {
|
||||
if (err.code !== 'ENOENT') throw err;
|
||||
content = loadTemplate({ date, templateDir, fs, pathUtil, now });
|
||||
fs.writeFileSync(notePath, content, 'utf-8');
|
||||
created = true;
|
||||
}
|
||||
return { path: notePath, content, created };
|
||||
}
|
||||
|
||||
/**
|
||||
* List existing daily-note filenames in `dir`, newest first.
|
||||
* Returns ['2026-09-13.md', '2026-09-12.md', ...] — only entries that
|
||||
* match the YYYY-MM-DD.md shape so a stray readme.md in the folder is
|
||||
* ignored.
|
||||
*/
|
||||
function listExisting({ dir, fs }) {
|
||||
let entries;
|
||||
try {
|
||||
entries = fs.readdirSync(dir);
|
||||
} catch (err) {
|
||||
if (err.code === 'ENOENT') return [];
|
||||
throw err;
|
||||
}
|
||||
// Strict-enough date pattern: year-month-day with month 01-12 and day 01-31.
|
||||
// We don't enforce real-calendar validity (Feb 30 still matches) — that's
|
||||
// the user's problem to fix, not ours to silently drop.
|
||||
const pattern = /^\d{4}-(0[1-9]|1[0-2])-(0[1-9]|[12]\d|3[01])\.md$/;
|
||||
return entries.filter((name) => pattern.test(name)).sort((a, b) => (a < b ? 1 : a > b ? -1 : 0));
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate a `dir` candidate for daily notes. Used by the IPC handler so a
|
||||
* renderer compromise can't point this at /etc or other sensitive paths —
|
||||
* any path outside userData/templates is rejected by main.js's existing
|
||||
* validatePath() before we get here.
|
||||
*/
|
||||
function isValidDir(dir) {
|
||||
return typeof dir === 'string' && dir.length > 0 && !dir.includes('\0');
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
dateKey,
|
||||
pathFor,
|
||||
loadTemplate,
|
||||
openOrCreate,
|
||||
listExisting,
|
||||
isValidDir,
|
||||
DEFAULT_TEMPLATE_NAME,
|
||||
};
|
||||
@@ -0,0 +1,231 @@
|
||||
/**
|
||||
* Doc-aware Q&A — thin wrapper over WorkspaceSearch tuned for questions.
|
||||
*
|
||||
* User asks "what did I write about rust async last week?" and gets back the
|
||||
* top-K chunks from their workspace ranked by relevance. No neural model, no
|
||||
* API key — the same TF-idf-style ranking WorkspaceSearch uses, with a few
|
||||
* QA-specific tweaks:
|
||||
*
|
||||
* - Question words (what/how/why/when/where/who/which/does/is/are/...) are
|
||||
* stripped before ranking — they don't carry meaning, just grammar
|
||||
* - The top chunks are split out as independent results so the renderer
|
||||
* can show "3 passages from 2 files" instead of one giant result
|
||||
* - Recency weight is doubled: "last week" implies the user wants fresh
|
||||
* content, not old material
|
||||
*
|
||||
* Future upgrade path: swap `WorkspaceSearch.search` for a vector-similarity
|
||||
* call (transformers.js in the renderer, or a sidecar process). The public
|
||||
* shape — {question, chunks:[{filePath, snippet, score}]} — stays the same.
|
||||
*
|
||||
* @module DocQA
|
||||
*/
|
||||
|
||||
const WorkspaceSearch = require('./WorkspaceSearch');
|
||||
|
||||
const QUESTION_WORDS = new Set([
|
||||
'what',
|
||||
'when',
|
||||
'where',
|
||||
'who',
|
||||
'whom',
|
||||
'whose',
|
||||
'why',
|
||||
'how',
|
||||
'which',
|
||||
'whether',
|
||||
'does',
|
||||
'do',
|
||||
'did',
|
||||
'is',
|
||||
'are',
|
||||
'was',
|
||||
'were',
|
||||
'be',
|
||||
'been',
|
||||
'being',
|
||||
'have',
|
||||
'has',
|
||||
'had',
|
||||
'can',
|
||||
'could',
|
||||
'would',
|
||||
'should',
|
||||
'will',
|
||||
'shall',
|
||||
'may',
|
||||
'might',
|
||||
'i',
|
||||
'me',
|
||||
'my',
|
||||
'we',
|
||||
'our',
|
||||
'you',
|
||||
'your',
|
||||
'they',
|
||||
'them',
|
||||
'their',
|
||||
'a',
|
||||
'an',
|
||||
'the',
|
||||
'and',
|
||||
'or',
|
||||
'but',
|
||||
'so',
|
||||
'of',
|
||||
'to',
|
||||
'in',
|
||||
'on',
|
||||
'at',
|
||||
'for',
|
||||
'with',
|
||||
'about',
|
||||
'into',
|
||||
'from',
|
||||
'by',
|
||||
'as',
|
||||
'this',
|
||||
'that',
|
||||
'these',
|
||||
'those',
|
||||
'it',
|
||||
'its',
|
||||
'write',
|
||||
'wrote',
|
||||
'written',
|
||||
'read',
|
||||
'think',
|
||||
'know',
|
||||
'find',
|
||||
'show',
|
||||
'tell',
|
||||
'say',
|
||||
'see',
|
||||
'use',
|
||||
'used',
|
||||
]);
|
||||
|
||||
/**
|
||||
* Strip question words from a raw question so WorkspaceSearch's bare-term
|
||||
* matching isn't drowned in grammar noise. Quoted phrases, #tags, and
|
||||
* @wikilinks pass through unchanged.
|
||||
*/
|
||||
function cleanQuestion(raw) {
|
||||
if (typeof raw !== 'string') return '';
|
||||
// First pull out facets and phrases so we don't strip them
|
||||
const facets = [];
|
||||
const working = raw.replace(/[#@][A-Za-z0-9_-]+|"[^"]+"/g, (m) => {
|
||||
facets.push(m);
|
||||
return ' ';
|
||||
});
|
||||
const words = working.split(/\s+/).filter((w) => {
|
||||
const lower = w.toLowerCase().replace(/[^a-z0-9-]/g, '');
|
||||
return lower.length >= 2 && !QUESTION_WORDS.has(lower);
|
||||
});
|
||||
return [...words, ...facets].join(' ').trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* Split a document into rough passages (~800 chars or paragraph break,
|
||||
* whichever comes first). Returns an array of {start, end, text} so the
|
||||
* caller can build snippets. Pure.
|
||||
*/
|
||||
function chunkDocument(content, maxChunkChars = 800) {
|
||||
if (typeof content !== 'string' || content.length === 0) return [];
|
||||
const chunks = [];
|
||||
const paragraphs = content.split(/\n\s*\n/);
|
||||
let buffer = '';
|
||||
let startOffset = 0;
|
||||
for (const para of paragraphs) {
|
||||
// If a single paragraph exceeds the cap, hard-split it by character.
|
||||
if (para.length > maxChunkChars) {
|
||||
// Flush whatever was buffered first.
|
||||
if (buffer) {
|
||||
chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
|
||||
startOffset += buffer.length + 2;
|
||||
buffer = '';
|
||||
}
|
||||
for (let i = 0; i < para.length; i += maxChunkChars) {
|
||||
const slice = para.slice(i, i + maxChunkChars);
|
||||
chunks.push({ start: startOffset, end: startOffset + slice.length, text: slice });
|
||||
startOffset += slice.length;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
const tentative = buffer ? `${buffer}\n\n${para}` : para;
|
||||
if (tentative.length > maxChunkChars && buffer) {
|
||||
chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
|
||||
startOffset += buffer.length + 2; // account for the "\n\n" we split on
|
||||
buffer = para;
|
||||
} else {
|
||||
buffer = tentative;
|
||||
}
|
||||
}
|
||||
if (buffer) chunks.push({ start: startOffset, end: startOffset + buffer.length, text: buffer });
|
||||
return chunks;
|
||||
}
|
||||
|
||||
/**
|
||||
* Answer a natural-language question by ranking the workspace's chunks.
|
||||
*
|
||||
* @param {object} args
|
||||
* @param {string} args.question
|
||||
* @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files
|
||||
* @param {number} [args.topK=5] number of chunks to return
|
||||
* @param {number} [args.nowMs=Date.now()]
|
||||
* @returns {{question:string, chunks:Array<{filePath:string, snippet:string, score:number, mtimeMs:number}>}}
|
||||
*/
|
||||
function ask({ question, files, topK = 5, nowMs = Date.now() }) {
|
||||
const cleaned = cleanQuestion(question);
|
||||
if (!cleaned || !Array.isArray(files) || files.length === 0) {
|
||||
return { question: String(question || ''), chunks: [] };
|
||||
}
|
||||
|
||||
// First, find the docs that match at all (cheap, broad pass).
|
||||
const docHits = WorkspaceSearch.search({ query: cleaned, files, limit: 20, nowMs });
|
||||
|
||||
// Then re-rank at the chunk level within those docs.
|
||||
const chunkCorpus = [];
|
||||
for (const hit of docHits) {
|
||||
const file = files.find((f) => f.path === hit.filePath);
|
||||
if (!file || typeof file.content !== 'string') continue;
|
||||
const chunks = chunkDocument(file.content);
|
||||
for (const chunk of chunks) {
|
||||
chunkCorpus.push({
|
||||
path: `${hit.filePath}#${chunk.start}`,
|
||||
content: chunk.text,
|
||||
mtimeMs: file.mtimeMs,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
const chunkHits = WorkspaceSearch.search({
|
||||
query: cleaned,
|
||||
files: chunkCorpus,
|
||||
limit: topK,
|
||||
nowMs,
|
||||
});
|
||||
|
||||
// Translate the per-chunk hits back into the public shape. The fake path
|
||||
// "<file>#<offset>" carries the chunk start; the renderer's existing
|
||||
// file-opened handler can split it on '#' if it wants to deep-link.
|
||||
const chunks = chunkHits.map((h) => {
|
||||
const offsetMatch = /#(\d+)$/.exec(h.filePath);
|
||||
const offset = offsetMatch ? Number(offsetMatch[1]) : 0;
|
||||
return {
|
||||
filePath: h.filePath.replace(/#\d+$/, ''),
|
||||
offset,
|
||||
snippet: h.snippet,
|
||||
score: h.score,
|
||||
mtimeMs: chunkCorpus.find((c) => c.path === h.filePath)?.mtimeMs || 0,
|
||||
};
|
||||
});
|
||||
|
||||
return { question: String(question), chunks };
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
cleanQuestion,
|
||||
chunkDocument,
|
||||
ask,
|
||||
QUESTION_WORDS,
|
||||
};
|
||||
@@ -0,0 +1,203 @@
|
||||
/**
|
||||
* Workspace content search — the simplest useful thing that scales.
|
||||
*
|
||||
* Given a list of file paths and a query, returns ranked matches across
|
||||
* them. Designed for personal-workspace sizes (hundreds, maybe thousands of
|
||||
* notes) — runs synchronously on the main process with injectable IO so it
|
||||
* can be unit-tested without touching the filesystem.
|
||||
*
|
||||
* Query grammar (all optional, combinable):
|
||||
* - Bare words → term matches (case-insensitive substring)
|
||||
* - Words prefixed with `#` → require a matching `#tag` in the doc
|
||||
* - Words prefixed with `@` → require a matching `[[wikilink]]` target
|
||||
* - Quoted phrases → require the exact substring
|
||||
*
|
||||
* Scoring:
|
||||
* - Each matched term adds +1 per occurrence (capped at 5/term to dampen
|
||||
* repetition spam)
|
||||
* - Each tag hit (`#foo`) or wikilink hit (`[[foo]]`) adds +3 if the query
|
||||
* asked for it (facets weigh more than prose)
|
||||
* - Recent edits (mtime within 7 days) get +0.5 to nudge "what I was
|
||||
* working on yesterday" upward
|
||||
*
|
||||
* Returns [{ filePath, score, snippet, matchedTerms, matchedTags, matchedLinks }]
|
||||
* sorted by score desc, filePath asc as tiebreaker.
|
||||
*
|
||||
* @module WorkspaceSearch
|
||||
*/
|
||||
|
||||
/** Tokenize a query into a structured form. Pure. */
|
||||
function parseQuery(raw) {
|
||||
if (typeof raw !== 'string') return { terms: [], phrases: [], tags: [], links: [] };
|
||||
const terms = [];
|
||||
const phrases = [];
|
||||
const tags = [];
|
||||
const links = [];
|
||||
|
||||
// Phrases first — match "double quoted text" as a unit
|
||||
const phraseRe = /"([^"]+)"/g;
|
||||
let remaining = raw.replace(phraseRe, (_, inner) => {
|
||||
if (inner.trim()) phrases.push(inner.toLowerCase());
|
||||
return ' ';
|
||||
});
|
||||
|
||||
// Tags and links: #foo, @bar
|
||||
const facetRe = /[#@]([A-Za-z0-9_-]+)/g;
|
||||
remaining = remaining.replace(facetRe, (m, name) => {
|
||||
if (m.startsWith('#')) tags.push(name.toLowerCase());
|
||||
else links.push(name.toLowerCase());
|
||||
return ' ';
|
||||
});
|
||||
|
||||
// Bare words (≥2 chars to avoid noise)
|
||||
for (const w of remaining.split(/\s+/)) {
|
||||
if (w.length >= 2) terms.push(w.toLowerCase());
|
||||
}
|
||||
|
||||
return { terms, phrases, tags, links };
|
||||
}
|
||||
|
||||
/** Extract a one-line snippet around the first match for `term`. Pure. */
|
||||
function snippetAround(content, term, radius = 60) {
|
||||
const haystack = content.toLowerCase();
|
||||
const idx = haystack.indexOf(term);
|
||||
if (idx < 0) return '';
|
||||
const start = Math.max(0, idx - radius);
|
||||
const end = Math.min(content.length, idx + term.length + radius);
|
||||
let s = content.slice(start, end).replace(/\s+/g, ' ').trim();
|
||||
if (start > 0) s = '…' + s;
|
||||
if (end < content.length) s = s + '…';
|
||||
return s;
|
||||
}
|
||||
|
||||
/** Count occurrences of `needle` (case-insensitive substring) in `haystack`. */
|
||||
function countOccurrences(haystack, needle) {
|
||||
if (!needle) return 0;
|
||||
const hay = haystack.toLowerCase();
|
||||
const ndl = needle.toLowerCase();
|
||||
let count = 0;
|
||||
let idx = 0;
|
||||
while ((idx = hay.indexOf(ndl, idx)) !== -1) {
|
||||
count++;
|
||||
idx += ndl.length;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
/** Does the content contain the `#tag`? Looks for word-boundary `#tag`. */
|
||||
function hasTag(content, tag) {
|
||||
const re = new RegExp(`(^|\\s)#${escapeRegex(tag)}\\b`, 'i');
|
||||
return re.test(content);
|
||||
}
|
||||
|
||||
/** Does the content contain `[[wikilink]]` or `[[wikilink|alias]]`? */
|
||||
function hasWikilink(content, link) {
|
||||
const re = new RegExp(`\\[\\[${escapeRegex(link)}(\\|[^\\]]+)?\\]\\]`, 'i');
|
||||
return re.test(content);
|
||||
}
|
||||
|
||||
function escapeRegex(s) {
|
||||
return String(s).replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
||||
}
|
||||
|
||||
/**
|
||||
* Run a parsed query against one document's content. Returns
|
||||
* { score, matchedTerms, matchedTags, matchedLinks, snippet } or null if
|
||||
* nothing matched (so the caller can filter cheaply).
|
||||
*/
|
||||
function scoreDocument({ content, parsed, mtimeMs = 0, nowMs = Date.now() }) {
|
||||
let score = 0;
|
||||
const matchedTerms = [];
|
||||
const matchedTags = [];
|
||||
const matchedLinks = [];
|
||||
const lower = content.toLowerCase();
|
||||
|
||||
for (const term of parsed.terms) {
|
||||
const n = countOccurrences(content, term);
|
||||
if (n > 0) {
|
||||
score += Math.min(n, 5);
|
||||
matchedTerms.push(term);
|
||||
}
|
||||
}
|
||||
for (const phrase of parsed.phrases) {
|
||||
if (lower.includes(phrase)) {
|
||||
score += 3; // phrase hits weigh more
|
||||
matchedTerms.push(phrase);
|
||||
}
|
||||
}
|
||||
for (const tag of parsed.tags) {
|
||||
if (hasTag(content, tag)) {
|
||||
score += 3;
|
||||
matchedTags.push(tag);
|
||||
}
|
||||
}
|
||||
for (const link of parsed.links) {
|
||||
if (hasWikilink(content, link)) {
|
||||
score += 3;
|
||||
matchedLinks.push(link);
|
||||
}
|
||||
}
|
||||
|
||||
if (score === 0) return null;
|
||||
|
||||
// Recency nudge: edits within the last 7 days
|
||||
const ageDays = (nowMs - mtimeMs) / (24 * 60 * 60 * 1000);
|
||||
if (Number.isFinite(ageDays) && ageDays >= 0 && ageDays <= 7) {
|
||||
score += 0.5 * (1 - ageDays / 7);
|
||||
}
|
||||
|
||||
// Build a snippet around the first matched term (or phrase/tag) for the UI
|
||||
let snippet = '';
|
||||
const firstTerm = matchedTerms[0] || (parsed.tags[0] ? `#${parsed.tags[0]}` : null);
|
||||
if (firstTerm) {
|
||||
snippet = snippetAround(content, firstTerm);
|
||||
} else {
|
||||
snippet = content.slice(0, 120).replace(/\s+/g, ' ').trim() + (content.length > 120 ? '…' : '');
|
||||
}
|
||||
|
||||
return { score, matchedTerms, matchedTags, matchedLinks, snippet };
|
||||
}
|
||||
|
||||
/**
|
||||
* Search across many files. Files whose content matches the query are
|
||||
* scored and returned sorted by score.
|
||||
*
|
||||
* @param {object} args
|
||||
* @param {string} args.query raw query string
|
||||
* @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files
|
||||
* @param {number} [args.limit=50]
|
||||
* @returns {Array<{filePath:string, score:number, snippet:string, matchedTerms:string[], matchedTags:string[], matchedLinks:string[]}>}
|
||||
*/
|
||||
function search({ query, files, limit = 50, nowMs = Date.now() }) {
|
||||
const parsed = parseQuery(query);
|
||||
// Nothing to search for — short-circuit so callers don't pay the file loop
|
||||
if (
|
||||
parsed.terms.length === 0 &&
|
||||
parsed.phrases.length === 0 &&
|
||||
parsed.tags.length === 0 &&
|
||||
parsed.links.length === 0
|
||||
) {
|
||||
return [];
|
||||
}
|
||||
const out = [];
|
||||
for (const file of files) {
|
||||
if (!file || typeof file.content !== 'string') continue;
|
||||
const r = scoreDocument({ content: file.content, parsed, mtimeMs: file.mtimeMs, nowMs });
|
||||
if (!r) continue;
|
||||
out.push({ filePath: file.path, ...r });
|
||||
}
|
||||
out.sort(
|
||||
(a, b) => b.score - a.score || (a.filePath < b.filePath ? -1 : a.filePath > b.filePath ? 1 : 0)
|
||||
);
|
||||
return out.slice(0, limit);
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
parseQuery,
|
||||
snippetAround,
|
||||
countOccurrences,
|
||||
hasTag,
|
||||
hasWikilink,
|
||||
scoreDocument,
|
||||
search,
|
||||
};
|
||||
Reference in New Issue
Block a user