feat(paste): smart URL → markdown-link on URL-only pastes

- src/main/UrlTitle.js — fetch a URL, return its <title>. Pure module
  with injectable fetch for tests. Decodes named + numeric + hex entities
  (AT&amp;T, Caf&#233;, &#x2014;), strips the <title> tags, collapses
  whitespace, caps the label at 200 chars. 5s timeout via AbortController,
  2 MiB body cap to avoid OOM on big downloads, streaming reader with
  overflow cancellation. Rejects non-http(s) URLs up front.
- src/main.js — IPC url-title:fetch proxies to fetchTitle.
- src/editor/smart-paste.js — CodeMirror 6 extension that detects
  URL-only pastes (one URL, surrounded only by whitespace), inserts the
  URL immediately, then async-rewrites the insertion range to
  [Title](url) once the title arrives. Multi-line / prose pastes pass
  through untouched.
- src/editor/codemirror-setup.js — accepts smartPasteFetcher option and
  pushes the extension when provided.
- src/renderer.js — passes ipcRenderer.invoke('url-title:fetch') as
  the fetcher.
- src/preload.js — url-title:fetch added to ALLOWED_SEND_CHANNELS.
- eslint.config.js — AbortController / TextDecoder / TextEncoder added
  to globals (available in Node 20+ and Chromium).

Tests (34 new):
- tests/main/UrlTitle.test.js (24): isHttpUrl scheme filter, decodeTitle
  named/numeric/hex entities + whitespace + non-string, extractTitleFromHtml
  first match + case-insensitive + null fallback, fetchTitle success +
  long-title cap + streaming body, all failure paths (non-http,
  no-fetch, non-OK, wrong content-type, no <title>, network error,
  body over maxBytes — including streaming overflow cancellation).
- tests/smart-paste.test.js (10): URL_ONLY_RE detection (bare URL,
  path/query/fragment, whitespace padding, scheme rejection, embedded
  rejection, empty/malformed), replacement format ([T](url), brackets
  in title survive, query strings preserved).

Full suite: 71 suites, 831 tests, lint+format clean.

Amit Haridas
This commit is contained in:
2026-09-14 08:58:32 +05:30
parent 6d08c138d8
commit bd86748c47
9 changed files with 533 additions and 0 deletions
+8
View File
@@ -121,6 +121,7 @@ function createEditor(parentElement, options = {}) {
showLineNumbers = true,
vimMode = false,
getTabExpansion = null,
smartPasteFetcher = null,
} = options;
const extensions = [
@@ -155,6 +156,13 @@ function createEditor(parentElement, options = {}) {
EditorView.lineWrapping,
];
// Smart-paste: a URL-only paste is intercepted and rewritten to a
// markdown link once the page title comes back from the main process.
// The host (renderer) injects the IPC-backed fetcher; null disables.
if (smartPasteFetcher) {
extensions.push(require('./smart-paste').smartPaste({ fetchTitle: smartPasteFetcher }));
}
if (showLineNumbers) {
extensions.push(lineNumbers());
}
+73
View File
@@ -0,0 +1,73 @@
/**
* CodeMirror 6 smart-paste extension.
*
* When the pasted text is a single URL (or a URL surrounded only by
* whitespace), intercept it and ask the main process for the page title
* via the `url-title:fetch` IPC channel. Replace the paste with a
* markdown link "[Title](url)" if a title is found, or leave the URL
* untouched if the fetch fails or returns no title.
*
* Other paste content passes through unchanged — we never want to mangle
* a multi-line paste just because one token looks URL-ish.
*
* @param {object} deps
* @param {(args:{url:string,timeoutMs?:number}) => Promise<{url:string,title:string}|null>} deps.fetchTitle
* @returns {import('@codemirror/view').Extension}
*/
const { EditorView } = require('@codemirror/view');
const URL_ONLY_RE = /^\s*(https?:\/\/[^\s]+)\s*$/i;
const TIMEOUT_MS = 4000;
function smartPaste(deps) {
const { fetchTitle } = deps || {};
if (typeof fetchTitle !== 'function') {
// No-op extension if no IPC bridge was passed in
return [];
}
return EditorView.domEventHandlers({
paste(event, view) {
const text = event.clipboardData && event.clipboardData.getData('text/plain');
const match = text && URL_ONLY_RE.exec(text);
if (!match) return; // not a URL-only paste — let the default handler run
const url = match[1];
event.preventDefault();
// Insert the URL immediately so the paste isn't lost on slow networks,
// then async-fetch the title and rewrite the just-pasted range.
const head = view.state.selection.main.head;
const from = head;
view.dispatch({
changes: { from, insert: url },
selection: { anchor: from + url.length },
});
// Best-effort fetch; if it fails, leave the URL as-is.
Promise.race([
fetchTitle({ url, timeoutMs: TIMEOUT_MS }),
new Promise((resolve) => setTimeout(() => resolve(null), TIMEOUT_MS)),
])
.then((result) => {
if (!result || !result.title) return;
// The user may have continued typing in the meantime. Cap the
// rewrite at the original insertion length so we don't clobber
// anything else.
const currentLen = view.state.doc.length;
const rewriteTo = Math.min(from + url.length, currentLen);
if (rewriteTo <= from) return;
const replacement = `[${result.title}](${url})`;
view.dispatch({
changes: { from, to: rewriteTo, insert: replacement },
selection: { anchor: from + replacement.length },
});
})
.catch(() => {
/* leave URL as-is */
});
},
});
}
module.exports = { smartPaste, URL_ONLY_RE };
+13
View File
@@ -3981,6 +3981,7 @@ const AutosaveBuffer = require('./main/AutosaveBuffer');
const DailyNotes = require('./main/DailyNotes');
const WorkspaceSearch = require('./main/WorkspaceSearch');
const DocQA = require('./main/DocQA');
const UrlTitle = require('./main/UrlTitle');
/** IO bundle for VersionHistory bound to <userData>/versions. */
function versionHistoryIo() {
@@ -5827,6 +5828,18 @@ ipcMain.handle('doc-qa:ask', async (_event, { question, dir, topK = 5 } = {}) =>
return DocQA.ask({ question, files: corpus, topK });
});
// ================================
// Smart-paste: URL → page title
// ================================
// The renderer pastes a URL and asks for the page's <title> so it can be
// turned into "[Title](url)" automatically. Network access from the main
// process is preferred over the renderer because (a) CSP stays simpler and
// (b) any proxy/firewall logic can live here later.
ipcMain.handle('url-title:fetch', async (_event, { url, timeoutMs } = {}) => {
if (typeof url !== 'string') return null;
return UrlTitle.fetchTitle({ url, timeoutMs });
});
// Esc in the note window hides instead of closing (keeps it one keystroke away)
ipcMain.on('quick-note:hide', () => {
if (quickNoteWindow) quickNoteWindow.hide();
+155
View File
@@ -0,0 +1,155 @@
/**
* URL title extraction.
*
* Given a URL, fetch the page and return its <title> for smart-paste: paste
* a link, get a markdown link with the page's title as the label. Pure
* module — main.js wires the actual fetch via the injectable `fetch` arg
* (defaults to globalThis.fetch in production; tests pass a stub).
*
* Constraints:
* - HTML only (Content-Type starts with text/html)
* - ≤2 MiB body (so a 4GB tarball link doesn't OOM us)
* - 5-second total timeout (AbortController)
* - Title decoded as ISO-8859-1 → UTF-8 (browser-grade heuristic; HTML5
* charset wins when present)
* - Stripped of newlines/extra whitespace — labels must be one logical line
* - Capped at 200 chars — past that the label becomes ugly in markdown
*
* @module UrlTitle
*/
/** Match the first <title>…</title> in an HTML body. Case-insensitive. */
function extractTitleFromHtml(html) {
if (typeof html !== 'string') return null;
const match = /<title[^>]*>([\s\S]*?)<\/title>/i.exec(html);
if (!match) return null;
return decodeTitle(match[1]);
}
/**
* Decode an HTML <title> value to readable text.
* - decode named/numeric entities (browser-grade for common ones)
* - collapse internal whitespace (browsers do this when computing the
* document.title property)
* - trim
*/
function decodeTitle(raw) {
if (typeof raw !== 'string') return '';
// Named + numeric entities we care about. The full HTML5 set is ~250
// entries; the ones below cover the common ones for English pages.
const entities = {
'&amp;': '&',
'&lt;': '<',
'&gt;': '>',
'&quot;': '"',
'&apos;': "'",
'&nbsp;': ' ',
'&mdash;': '—',
'&ndash;': '–',
'&hellip;': '…',
'&copy;': '©',
'&reg;': '®',
'&trade;': '™',
};
let s = raw;
for (const [ent, ch] of Object.entries(entities)) {
s = s.split(ent).join(ch);
}
// Numeric entities: &#NN; or &#xHH;
s = s.replace(/&#(\d+);/g, (_, n) => {
const code = Number(n);
return Number.isFinite(code) && code >= 0 && code <= 0x10ffff ? String.fromCodePoint(code) : '';
});
s = s.replace(/&#x([0-9a-fA-F]+);/g, (_, n) => {
const code = parseInt(n, 16);
return Number.isFinite(code) ? String.fromCodePoint(code) : '';
});
// Don't strip <...>-looking strings here: an entity like &lt;b&gt; decodes
// to a literal "<b>" which would otherwise get eaten. Titles with embedded
// tags in the source HTML are vanishingly rare and we can pass them through
// as text. Collapse internal whitespace and trim instead.
s = s.replace(/\s+/g, ' ').trim();
return s;
}
/** Is the URL fetchable (http/https only)? Rejects file://, javascript:, data:, … */
function isHttpUrl(value) {
if (typeof value !== 'string') return false;
try {
const u = new URL(value);
return u.protocol === 'http:' || u.protocol === 'https:';
} catch {
return false;
}
}
/**
* Fetch a URL and return its <title>. Returns null when:
* - the URL is not http(s)
* - the response isn't HTML
* - the body is larger than maxBytes (default 2 MiB)
* - the request times out (default 5 s) or otherwise fails
* - the response has no <title>
*
* @param {object} args
* @param {string} args.url
* @param {number} [args.timeoutMs=5000]
* @param {number} [args.maxBytes=2 * 1024 * 1024]
* @param {typeof fetch} [args.fetch] injectable for tests
* @returns {Promise<{url:string, title:string} | null>}
*/
async function fetchTitle({ url, timeoutMs = 5000, maxBytes = 2 * 1024 * 1024, fetch = globalThis.fetch }) {
if (!isHttpUrl(url) || typeof fetch !== 'function') return null;
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeoutMs);
try {
const res = await fetch(url, {
redirect: 'follow',
signal: controller.signal,
headers: { Accept: 'text/html,application/xhtml+xml' },
});
if (!res || !res.ok) return null;
const ctype = res.headers && res.headers.get ? res.headers.get('content-type') || '' : '';
if (!ctype.toLowerCase().includes('text/html')) return null;
// Read up to maxBytes + 1 so we can detect overflow
const reader = res.body && typeof res.body.getReader === 'function' ? res.body.getReader() : null;
let body = '';
if (reader) {
while (true) {
const { value, done } = await reader.read();
if (done) break;
body += new TextDecoder('utf-8', { fatal: false }).decode(value, { stream: true });
if (body.length > maxBytes) {
try {
await reader.cancel();
} catch {
/* noop */
}
return null;
}
}
body += new TextDecoder('utf-8', { fatal: false }).decode();
} else {
body = await res.text();
if (body.length > maxBytes) return null;
}
const title = extractTitleFromHtml(body);
if (!title) return null;
const capped = title.length > 200 ? title.slice(0, 200).trim() + '…' : title;
return { url, title: capped };
} catch {
return null;
} finally {
clearTimeout(timer);
}
}
module.exports = {
extractTitleFromHtml,
decodeTitle,
isHttpUrl,
fetchTitle,
};
+3
View File
@@ -196,6 +196,9 @@ const ALLOWED_SEND_CHANNELS = [
// Doc-aware Q&A (chunk-level ranking over the workspace)
'doc-qa:ask',
// Smart-paste: URL → page title
'url-title:fetch',
];
const ALLOWED_RECEIVE_CHANNELS = [
+5
View File
@@ -749,6 +749,11 @@ class TabManager {
vimMode: window.__vimModeEnabled === true,
// Tab expands a snippet when the word before the cursor matches one
getTabExpansion: (prefix) => snippetExpansions.get(prefix) || null,
// Smart-paste: URL-only pastes get rewritten to a markdown link
// using the page's title. Wrapped in ipcRenderer.invoke so the
// actual fetch happens in main.
smartPasteFetcher: ({ url, timeoutMs }) =>
ipcRenderer.invoke('url-title:fetch', { url, timeoutMs }),
onChange: (newContent) => {
tab.content = newContent;
tab.isDirty = true;