Files
markdown-converter/src/main/WorkspaceSearch.js
T
amitwh 3f856bc857 feat(pkm): daily notes + workspace search + doc-aware Q&A
Three more features from the brainstorm menu, built on a shared search
algorithm so the codebase stays small.

- src/main/DailyNotes.js — Zettelkasten-style helper. One YYYY-MM-DD.md
  per local date under <userData>/notes/daily/; loads skeleton from
  <userData>/notes/templates/daily.md when present (built-in default
  otherwise). openOrCreate never clobbers existing content.
- src/main/WorkspaceSearch.js — tag/wikilink-aware content search.
  Pure module, injectable-IO tested. Query grammar: bare words,
  #tag, @wikilink, "quoted phrases". Facets weight +3 each; prose
  terms +1/occurrence capped at 5; edits within 7 days get a recency
  nudge. Returns ranked results with snippets.
- src/main/DocQA.js — chunk-level Q&A wrapper over WorkspaceSearch.
  cleanQuestion strips question words (what/how/why/...) and verb
  noise (write/read/show/tell/...) so they don't drown the ranking.
  Returns top-K passages instead of whole-file hits — multiple chunks
  from the same file can appear in the answer.
- src/main.js — IPC: daily-notes:open-today, daily-notes:list,
  workspace-search:query, doc-qa:ask. Path validation through the
  existing validatePath gate; a global Ctrl+Alt+D shortcut creates
  today's daily note from anywhere.
- src/preload.js — all four channels added to ALLOWED_SEND_CHANNELS.

Tests (56 new across the three modules):
- tests/main/DailyNotes.test.js (15): dateKey formatting, pathFor,
  template load + {date}/{weekday} substitution, openOrCreate +
  no-clobber, nested-dir creation, listExisting filtering,
  isValidDir rejects NUL/non-string.
- tests/main/WorkspaceSearch.test.js (25): parseQuery grammar,
  hasTag/hasWikilink word boundaries, scoreDocument scoring,
  per-term spam cap, recency nudge, search ranking + limit +
  empty-query short-circuit, bad-input safety.
- tests/main/DocQA.test.js (16): cleanQuestion stripping + facet
  preservation, chunkDocument paragraph + hard-split, ask()
  top-K, recency tiebreaker, missing-files fallback.

Full suite: 748 tests pass, 66 suites, lint+format clean.

Amit Haridas
2026-09-14 00:02:58 +05:30

204 lines
6.4 KiB
JavaScript

/**
* Workspace content search — the simplest useful thing that scales.
*
* Given a list of file paths and a query, returns ranked matches across
* them. Designed for personal-workspace sizes (hundreds, maybe thousands of
* notes) — runs synchronously on the main process with injectable IO so it
* can be unit-tested without touching the filesystem.
*
* Query grammar (all optional, combinable):
* - Bare words → term matches (case-insensitive substring)
* - Words prefixed with `#` → require a matching `#tag` in the doc
* - Words prefixed with `@` → require a matching `[[wikilink]]` target
* - Quoted phrases → require the exact substring
*
* Scoring:
* - Each matched term adds +1 per occurrence (capped at 5/term to dampen
* repetition spam)
* - Each tag hit (`#foo`) or wikilink hit (`[[foo]]`) adds +3 if the query
* asked for it (facets weigh more than prose)
* - Recent edits (mtime within 7 days) get +0.5 to nudge "what I was
* working on yesterday" upward
*
* Returns [{ filePath, score, snippet, matchedTerms, matchedTags, matchedLinks }]
* sorted by score desc, filePath asc as tiebreaker.
*
* @module WorkspaceSearch
*/
/** Tokenize a query into a structured form. Pure. */
function parseQuery(raw) {
if (typeof raw !== 'string') return { terms: [], phrases: [], tags: [], links: [] };
const terms = [];
const phrases = [];
const tags = [];
const links = [];
// Phrases first — match "double quoted text" as a unit
const phraseRe = /"([^"]+)"/g;
let remaining = raw.replace(phraseRe, (_, inner) => {
if (inner.trim()) phrases.push(inner.toLowerCase());
return ' ';
});
// Tags and links: #foo, @bar
const facetRe = /[#@]([A-Za-z0-9_-]+)/g;
remaining = remaining.replace(facetRe, (m, name) => {
if (m.startsWith('#')) tags.push(name.toLowerCase());
else links.push(name.toLowerCase());
return ' ';
});
// Bare words (≥2 chars to avoid noise)
for (const w of remaining.split(/\s+/)) {
if (w.length >= 2) terms.push(w.toLowerCase());
}
return { terms, phrases, tags, links };
}
/** Extract a one-line snippet around the first match for `term`. Pure. */
function snippetAround(content, term, radius = 60) {
const haystack = content.toLowerCase();
const idx = haystack.indexOf(term);
if (idx < 0) return '';
const start = Math.max(0, idx - radius);
const end = Math.min(content.length, idx + term.length + radius);
let s = content.slice(start, end).replace(/\s+/g, ' ').trim();
if (start > 0) s = '…' + s;
if (end < content.length) s = s + '…';
return s;
}
/** Count occurrences of `needle` (case-insensitive substring) in `haystack`. */
function countOccurrences(haystack, needle) {
if (!needle) return 0;
const hay = haystack.toLowerCase();
const ndl = needle.toLowerCase();
let count = 0;
let idx = 0;
while ((idx = hay.indexOf(ndl, idx)) !== -1) {
count++;
idx += ndl.length;
}
return count;
}
/** Does the content contain the `#tag`? Looks for word-boundary `#tag`. */
function hasTag(content, tag) {
const re = new RegExp(`(^|\\s)#${escapeRegex(tag)}\\b`, 'i');
return re.test(content);
}
/** Does the content contain `[[wikilink]]` or `[[wikilink|alias]]`? */
function hasWikilink(content, link) {
const re = new RegExp(`\\[\\[${escapeRegex(link)}(\\|[^\\]]+)?\\]\\]`, 'i');
return re.test(content);
}
function escapeRegex(s) {
return String(s).replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
}
/**
* Run a parsed query against one document's content. Returns
* { score, matchedTerms, matchedTags, matchedLinks, snippet } or null if
* nothing matched (so the caller can filter cheaply).
*/
function scoreDocument({ content, parsed, mtimeMs = 0, nowMs = Date.now() }) {
let score = 0;
const matchedTerms = [];
const matchedTags = [];
const matchedLinks = [];
const lower = content.toLowerCase();
for (const term of parsed.terms) {
const n = countOccurrences(content, term);
if (n > 0) {
score += Math.min(n, 5);
matchedTerms.push(term);
}
}
for (const phrase of parsed.phrases) {
if (lower.includes(phrase)) {
score += 3; // phrase hits weigh more
matchedTerms.push(phrase);
}
}
for (const tag of parsed.tags) {
if (hasTag(content, tag)) {
score += 3;
matchedTags.push(tag);
}
}
for (const link of parsed.links) {
if (hasWikilink(content, link)) {
score += 3;
matchedLinks.push(link);
}
}
if (score === 0) return null;
// Recency nudge: edits within the last 7 days
const ageDays = (nowMs - mtimeMs) / (24 * 60 * 60 * 1000);
if (Number.isFinite(ageDays) && ageDays >= 0 && ageDays <= 7) {
score += 0.5 * (1 - ageDays / 7);
}
// Build a snippet around the first matched term (or phrase/tag) for the UI
let snippet = '';
const firstTerm = matchedTerms[0] || (parsed.tags[0] ? `#${parsed.tags[0]}` : null);
if (firstTerm) {
snippet = snippetAround(content, firstTerm);
} else {
snippet = content.slice(0, 120).replace(/\s+/g, ' ').trim() + (content.length > 120 ? '…' : '');
}
return { score, matchedTerms, matchedTags, matchedLinks, snippet };
}
/**
* Search across many files. Files whose content matches the query are
* scored and returned sorted by score.
*
* @param {object} args
* @param {string} args.query raw query string
* @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files
* @param {number} [args.limit=50]
* @returns {Array<{filePath:string, score:number, snippet:string, matchedTerms:string[], matchedTags:string[], matchedLinks:string[]}>}
*/
function search({ query, files, limit = 50, nowMs = Date.now() }) {
const parsed = parseQuery(query);
// Nothing to search for — short-circuit so callers don't pay the file loop
if (
parsed.terms.length === 0 &&
parsed.phrases.length === 0 &&
parsed.tags.length === 0 &&
parsed.links.length === 0
) {
return [];
}
const out = [];
for (const file of files) {
if (!file || typeof file.content !== 'string') continue;
const r = scoreDocument({ content: file.content, parsed, mtimeMs: file.mtimeMs, nowMs });
if (!r) continue;
out.push({ filePath: file.path, ...r });
}
out.sort(
(a, b) => b.score - a.score || (a.filePath < b.filePath ? -1 : a.filePath > b.filePath ? 1 : 0)
);
return out.slice(0, limit);
}
module.exports = {
parseQuery,
snippetAround,
countOccurrences,
hasTag,
hasWikilink,
scoreDocument,
search,
};