mirror of
https://github.com/amitwh/markdown-converter.git
synced 2026-10-01 17:29:29 +05:30
Three more features from the brainstorm menu, built on a shared search
algorithm so the codebase stays small.
- src/main/DailyNotes.js — Zettelkasten-style helper. One YYYY-MM-DD.md
per local date under <userData>/notes/daily/; loads skeleton from
<userData>/notes/templates/daily.md when present (built-in default
otherwise). openOrCreate never clobbers existing content.
- src/main/WorkspaceSearch.js — tag/wikilink-aware content search.
Pure module, injectable-IO tested. Query grammar: bare words,
#tag, @wikilink, "quoted phrases". Facets weight +3 each; prose
terms +1/occurrence capped at 5; edits within 7 days get a recency
nudge. Returns ranked results with snippets.
- src/main/DocQA.js — chunk-level Q&A wrapper over WorkspaceSearch.
cleanQuestion strips question words (what/how/why/...) and verb
noise (write/read/show/tell/...) so they don't drown the ranking.
Returns top-K passages instead of whole-file hits — multiple chunks
from the same file can appear in the answer.
- src/main.js — IPC: daily-notes:open-today, daily-notes:list,
workspace-search:query, doc-qa:ask. Path validation through the
existing validatePath gate; a global Ctrl+Alt+D shortcut creates
today's daily note from anywhere.
- src/preload.js — all four channels added to ALLOWED_SEND_CHANNELS.
Tests (56 new across the three modules):
- tests/main/DailyNotes.test.js (15): dateKey formatting, pathFor,
template load + {date}/{weekday} substitution, openOrCreate +
no-clobber, nested-dir creation, listExisting filtering,
isValidDir rejects NUL/non-string.
- tests/main/WorkspaceSearch.test.js (25): parseQuery grammar,
hasTag/hasWikilink word boundaries, scoreDocument scoring,
per-term spam cap, recency nudge, search ranking + limit +
empty-query short-circuit, bad-input safety.
- tests/main/DocQA.test.js (16): cleanQuestion stripping + facet
preservation, chunkDocument paragraph + hard-split, ask()
top-K, recency tiebreaker, missing-files fallback.
Full suite: 748 tests pass, 66 suites, lint+format clean.
Amit Haridas
204 lines
6.4 KiB
JavaScript
204 lines
6.4 KiB
JavaScript
/**
|
|
* Workspace content search — the simplest useful thing that scales.
|
|
*
|
|
* Given a list of file paths and a query, returns ranked matches across
|
|
* them. Designed for personal-workspace sizes (hundreds, maybe thousands of
|
|
* notes) — runs synchronously on the main process with injectable IO so it
|
|
* can be unit-tested without touching the filesystem.
|
|
*
|
|
* Query grammar (all optional, combinable):
|
|
* - Bare words → term matches (case-insensitive substring)
|
|
* - Words prefixed with `#` → require a matching `#tag` in the doc
|
|
* - Words prefixed with `@` → require a matching `[[wikilink]]` target
|
|
* - Quoted phrases → require the exact substring
|
|
*
|
|
* Scoring:
|
|
* - Each matched term adds +1 per occurrence (capped at 5/term to dampen
|
|
* repetition spam)
|
|
* - Each tag hit (`#foo`) or wikilink hit (`[[foo]]`) adds +3 if the query
|
|
* asked for it (facets weigh more than prose)
|
|
* - Recent edits (mtime within 7 days) get +0.5 to nudge "what I was
|
|
* working on yesterday" upward
|
|
*
|
|
* Returns [{ filePath, score, snippet, matchedTerms, matchedTags, matchedLinks }]
|
|
* sorted by score desc, filePath asc as tiebreaker.
|
|
*
|
|
* @module WorkspaceSearch
|
|
*/
|
|
|
|
/** Tokenize a query into a structured form. Pure. */
|
|
function parseQuery(raw) {
|
|
if (typeof raw !== 'string') return { terms: [], phrases: [], tags: [], links: [] };
|
|
const terms = [];
|
|
const phrases = [];
|
|
const tags = [];
|
|
const links = [];
|
|
|
|
// Phrases first — match "double quoted text" as a unit
|
|
const phraseRe = /"([^"]+)"/g;
|
|
let remaining = raw.replace(phraseRe, (_, inner) => {
|
|
if (inner.trim()) phrases.push(inner.toLowerCase());
|
|
return ' ';
|
|
});
|
|
|
|
// Tags and links: #foo, @bar
|
|
const facetRe = /[#@]([A-Za-z0-9_-]+)/g;
|
|
remaining = remaining.replace(facetRe, (m, name) => {
|
|
if (m.startsWith('#')) tags.push(name.toLowerCase());
|
|
else links.push(name.toLowerCase());
|
|
return ' ';
|
|
});
|
|
|
|
// Bare words (≥2 chars to avoid noise)
|
|
for (const w of remaining.split(/\s+/)) {
|
|
if (w.length >= 2) terms.push(w.toLowerCase());
|
|
}
|
|
|
|
return { terms, phrases, tags, links };
|
|
}
|
|
|
|
/** Extract a one-line snippet around the first match for `term`. Pure. */
|
|
function snippetAround(content, term, radius = 60) {
|
|
const haystack = content.toLowerCase();
|
|
const idx = haystack.indexOf(term);
|
|
if (idx < 0) return '';
|
|
const start = Math.max(0, idx - radius);
|
|
const end = Math.min(content.length, idx + term.length + radius);
|
|
let s = content.slice(start, end).replace(/\s+/g, ' ').trim();
|
|
if (start > 0) s = '…' + s;
|
|
if (end < content.length) s = s + '…';
|
|
return s;
|
|
}
|
|
|
|
/** Count occurrences of `needle` (case-insensitive substring) in `haystack`. */
|
|
function countOccurrences(haystack, needle) {
|
|
if (!needle) return 0;
|
|
const hay = haystack.toLowerCase();
|
|
const ndl = needle.toLowerCase();
|
|
let count = 0;
|
|
let idx = 0;
|
|
while ((idx = hay.indexOf(ndl, idx)) !== -1) {
|
|
count++;
|
|
idx += ndl.length;
|
|
}
|
|
return count;
|
|
}
|
|
|
|
/** Does the content contain the `#tag`? Looks for word-boundary `#tag`. */
|
|
function hasTag(content, tag) {
|
|
const re = new RegExp(`(^|\\s)#${escapeRegex(tag)}\\b`, 'i');
|
|
return re.test(content);
|
|
}
|
|
|
|
/** Does the content contain `[[wikilink]]` or `[[wikilink|alias]]`? */
|
|
function hasWikilink(content, link) {
|
|
const re = new RegExp(`\\[\\[${escapeRegex(link)}(\\|[^\\]]+)?\\]\\]`, 'i');
|
|
return re.test(content);
|
|
}
|
|
|
|
function escapeRegex(s) {
|
|
return String(s).replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
}
|
|
|
|
/**
|
|
* Run a parsed query against one document's content. Returns
|
|
* { score, matchedTerms, matchedTags, matchedLinks, snippet } or null if
|
|
* nothing matched (so the caller can filter cheaply).
|
|
*/
|
|
function scoreDocument({ content, parsed, mtimeMs = 0, nowMs = Date.now() }) {
|
|
let score = 0;
|
|
const matchedTerms = [];
|
|
const matchedTags = [];
|
|
const matchedLinks = [];
|
|
const lower = content.toLowerCase();
|
|
|
|
for (const term of parsed.terms) {
|
|
const n = countOccurrences(content, term);
|
|
if (n > 0) {
|
|
score += Math.min(n, 5);
|
|
matchedTerms.push(term);
|
|
}
|
|
}
|
|
for (const phrase of parsed.phrases) {
|
|
if (lower.includes(phrase)) {
|
|
score += 3; // phrase hits weigh more
|
|
matchedTerms.push(phrase);
|
|
}
|
|
}
|
|
for (const tag of parsed.tags) {
|
|
if (hasTag(content, tag)) {
|
|
score += 3;
|
|
matchedTags.push(tag);
|
|
}
|
|
}
|
|
for (const link of parsed.links) {
|
|
if (hasWikilink(content, link)) {
|
|
score += 3;
|
|
matchedLinks.push(link);
|
|
}
|
|
}
|
|
|
|
if (score === 0) return null;
|
|
|
|
// Recency nudge: edits within the last 7 days
|
|
const ageDays = (nowMs - mtimeMs) / (24 * 60 * 60 * 1000);
|
|
if (Number.isFinite(ageDays) && ageDays >= 0 && ageDays <= 7) {
|
|
score += 0.5 * (1 - ageDays / 7);
|
|
}
|
|
|
|
// Build a snippet around the first matched term (or phrase/tag) for the UI
|
|
let snippet = '';
|
|
const firstTerm = matchedTerms[0] || (parsed.tags[0] ? `#${parsed.tags[0]}` : null);
|
|
if (firstTerm) {
|
|
snippet = snippetAround(content, firstTerm);
|
|
} else {
|
|
snippet = content.slice(0, 120).replace(/\s+/g, ' ').trim() + (content.length > 120 ? '…' : '');
|
|
}
|
|
|
|
return { score, matchedTerms, matchedTags, matchedLinks, snippet };
|
|
}
|
|
|
|
/**
|
|
* Search across many files. Files whose content matches the query are
|
|
* scored and returned sorted by score.
|
|
*
|
|
* @param {object} args
|
|
* @param {string} args.query raw query string
|
|
* @param {Array<{path:string, content:string, mtimeMs?:number}>} args.files
|
|
* @param {number} [args.limit=50]
|
|
* @returns {Array<{filePath:string, score:number, snippet:string, matchedTerms:string[], matchedTags:string[], matchedLinks:string[]}>}
|
|
*/
|
|
function search({ query, files, limit = 50, nowMs = Date.now() }) {
|
|
const parsed = parseQuery(query);
|
|
// Nothing to search for — short-circuit so callers don't pay the file loop
|
|
if (
|
|
parsed.terms.length === 0 &&
|
|
parsed.phrases.length === 0 &&
|
|
parsed.tags.length === 0 &&
|
|
parsed.links.length === 0
|
|
) {
|
|
return [];
|
|
}
|
|
const out = [];
|
|
for (const file of files) {
|
|
if (!file || typeof file.content !== 'string') continue;
|
|
const r = scoreDocument({ content: file.content, parsed, mtimeMs: file.mtimeMs, nowMs });
|
|
if (!r) continue;
|
|
out.push({ filePath: file.path, ...r });
|
|
}
|
|
out.sort(
|
|
(a, b) => b.score - a.score || (a.filePath < b.filePath ? -1 : a.filePath > b.filePath ? 1 : 0)
|
|
);
|
|
return out.slice(0, limit);
|
|
}
|
|
|
|
module.exports = {
|
|
parseQuery,
|
|
snippetAround,
|
|
countOccurrences,
|
|
hasTag,
|
|
hasWikilink,
|
|
scoreDocument,
|
|
search,
|
|
};
|