mirror of
https://github.com/amitwh/markdown-converter.git
synced 2026-10-01 09:19:34 +05:30
feat(paste): smart URL → markdown-link on URL-only pastes
- src/main/UrlTitle.js — fetch a URL, return its <title>. Pure module
with injectable fetch for tests. Decodes named + numeric + hex entities
(AT&T, Café, —), strips the <title> tags, collapses
whitespace, caps the label at 200 chars. 5s timeout via AbortController,
2 MiB body cap to avoid OOM on big downloads, streaming reader with
overflow cancellation. Rejects non-http(s) URLs up front.
- src/main.js — IPC url-title:fetch proxies to fetchTitle.
- src/editor/smart-paste.js — CodeMirror 6 extension that detects
URL-only pastes (one URL, surrounded only by whitespace), inserts the
URL immediately, then async-rewrites the insertion range to
[Title](url) once the title arrives. Multi-line / prose pastes pass
through untouched.
- src/editor/codemirror-setup.js — accepts smartPasteFetcher option and
pushes the extension when provided.
- src/renderer.js — passes ipcRenderer.invoke('url-title:fetch') as
the fetcher.
- src/preload.js — url-title:fetch added to ALLOWED_SEND_CHANNELS.
- eslint.config.js — AbortController / TextDecoder / TextEncoder added
to globals (available in Node 20+ and Chromium).
Tests (34 new):
- tests/main/UrlTitle.test.js (24): isHttpUrl scheme filter, decodeTitle
named/numeric/hex entities + whitespace + non-string, extractTitleFromHtml
first match + case-insensitive + null fallback, fetchTitle success +
long-title cap + streaming body, all failure paths (non-http,
no-fetch, non-OK, wrong content-type, no <title>, network error,
body over maxBytes — including streaming overflow cancellation).
- tests/smart-paste.test.js (10): URL_ONLY_RE detection (bare URL,
path/query/fragment, whitespace padding, scheme rejection, embedded
rejection, empty/malformed), replacement format ([T](url), brackets
in title survive, query strings preserved).
Full suite: 71 suites, 831 tests, lint+format clean.
Amit Haridas
This commit is contained in:
@@ -64,6 +64,10 @@ module.exports = [
|
|||||||
getComputedStyle: 'readonly',
|
getComputedStyle: 'readonly',
|
||||||
CSS: 'readonly',
|
CSS: 'readonly',
|
||||||
Element: 'readonly',
|
Element: 'readonly',
|
||||||
|
// Modern Web APIs available in Node 20+ and Chromium
|
||||||
|
AbortController: 'readonly',
|
||||||
|
TextDecoder: 'readonly',
|
||||||
|
TextEncoder: 'readonly',
|
||||||
// Electron
|
// Electron
|
||||||
electronAPI: 'readonly',
|
electronAPI: 'readonly',
|
||||||
// Libraries
|
// Libraries
|
||||||
|
|||||||
@@ -121,6 +121,7 @@ function createEditor(parentElement, options = {}) {
|
|||||||
showLineNumbers = true,
|
showLineNumbers = true,
|
||||||
vimMode = false,
|
vimMode = false,
|
||||||
getTabExpansion = null,
|
getTabExpansion = null,
|
||||||
|
smartPasteFetcher = null,
|
||||||
} = options;
|
} = options;
|
||||||
|
|
||||||
const extensions = [
|
const extensions = [
|
||||||
@@ -155,6 +156,13 @@ function createEditor(parentElement, options = {}) {
|
|||||||
EditorView.lineWrapping,
|
EditorView.lineWrapping,
|
||||||
];
|
];
|
||||||
|
|
||||||
|
// Smart-paste: a URL-only paste is intercepted and rewritten to a
|
||||||
|
// markdown link once the page title comes back from the main process.
|
||||||
|
// The host (renderer) injects the IPC-backed fetcher; null disables.
|
||||||
|
if (smartPasteFetcher) {
|
||||||
|
extensions.push(require('./smart-paste').smartPaste({ fetchTitle: smartPasteFetcher }));
|
||||||
|
}
|
||||||
|
|
||||||
if (showLineNumbers) {
|
if (showLineNumbers) {
|
||||||
extensions.push(lineNumbers());
|
extensions.push(lineNumbers());
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,73 @@
|
|||||||
|
/**
|
||||||
|
* CodeMirror 6 smart-paste extension.
|
||||||
|
*
|
||||||
|
* When the pasted text is a single URL (or a URL surrounded only by
|
||||||
|
* whitespace), intercept it and ask the main process for the page title
|
||||||
|
* via the `url-title:fetch` IPC channel. Replace the paste with a
|
||||||
|
* markdown link "[Title](url)" if a title is found, or leave the URL
|
||||||
|
* untouched if the fetch fails or returns no title.
|
||||||
|
*
|
||||||
|
* Other paste content passes through unchanged — we never want to mangle
|
||||||
|
* a multi-line paste just because one token looks URL-ish.
|
||||||
|
*
|
||||||
|
* @param {object} deps
|
||||||
|
* @param {(args:{url:string,timeoutMs?:number}) => Promise<{url:string,title:string}|null>} deps.fetchTitle
|
||||||
|
* @returns {import('@codemirror/view').Extension}
|
||||||
|
*/
|
||||||
|
const { EditorView } = require('@codemirror/view');
|
||||||
|
|
||||||
|
const URL_ONLY_RE = /^\s*(https?:\/\/[^\s]+)\s*$/i;
|
||||||
|
const TIMEOUT_MS = 4000;
|
||||||
|
|
||||||
|
function smartPaste(deps) {
|
||||||
|
const { fetchTitle } = deps || {};
|
||||||
|
if (typeof fetchTitle !== 'function') {
|
||||||
|
// No-op extension if no IPC bridge was passed in
|
||||||
|
return [];
|
||||||
|
}
|
||||||
|
return EditorView.domEventHandlers({
|
||||||
|
paste(event, view) {
|
||||||
|
const text = event.clipboardData && event.clipboardData.getData('text/plain');
|
||||||
|
const match = text && URL_ONLY_RE.exec(text);
|
||||||
|
if (!match) return; // not a URL-only paste — let the default handler run
|
||||||
|
|
||||||
|
const url = match[1];
|
||||||
|
event.preventDefault();
|
||||||
|
|
||||||
|
// Insert the URL immediately so the paste isn't lost on slow networks,
|
||||||
|
// then async-fetch the title and rewrite the just-pasted range.
|
||||||
|
const head = view.state.selection.main.head;
|
||||||
|
const from = head;
|
||||||
|
|
||||||
|
view.dispatch({
|
||||||
|
changes: { from, insert: url },
|
||||||
|
selection: { anchor: from + url.length },
|
||||||
|
});
|
||||||
|
|
||||||
|
// Best-effort fetch; if it fails, leave the URL as-is.
|
||||||
|
Promise.race([
|
||||||
|
fetchTitle({ url, timeoutMs: TIMEOUT_MS }),
|
||||||
|
new Promise((resolve) => setTimeout(() => resolve(null), TIMEOUT_MS)),
|
||||||
|
])
|
||||||
|
.then((result) => {
|
||||||
|
if (!result || !result.title) return;
|
||||||
|
// The user may have continued typing in the meantime. Cap the
|
||||||
|
// rewrite at the original insertion length so we don't clobber
|
||||||
|
// anything else.
|
||||||
|
const currentLen = view.state.doc.length;
|
||||||
|
const rewriteTo = Math.min(from + url.length, currentLen);
|
||||||
|
if (rewriteTo <= from) return;
|
||||||
|
const replacement = `[${result.title}](${url})`;
|
||||||
|
view.dispatch({
|
||||||
|
changes: { from, to: rewriteTo, insert: replacement },
|
||||||
|
selection: { anchor: from + replacement.length },
|
||||||
|
});
|
||||||
|
})
|
||||||
|
.catch(() => {
|
||||||
|
/* leave URL as-is */
|
||||||
|
});
|
||||||
|
},
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
module.exports = { smartPaste, URL_ONLY_RE };
|
||||||
+13
@@ -3981,6 +3981,7 @@ const AutosaveBuffer = require('./main/AutosaveBuffer');
|
|||||||
const DailyNotes = require('./main/DailyNotes');
|
const DailyNotes = require('./main/DailyNotes');
|
||||||
const WorkspaceSearch = require('./main/WorkspaceSearch');
|
const WorkspaceSearch = require('./main/WorkspaceSearch');
|
||||||
const DocQA = require('./main/DocQA');
|
const DocQA = require('./main/DocQA');
|
||||||
|
const UrlTitle = require('./main/UrlTitle');
|
||||||
|
|
||||||
/** IO bundle for VersionHistory bound to <userData>/versions. */
|
/** IO bundle for VersionHistory bound to <userData>/versions. */
|
||||||
function versionHistoryIo() {
|
function versionHistoryIo() {
|
||||||
@@ -5827,6 +5828,18 @@ ipcMain.handle('doc-qa:ask', async (_event, { question, dir, topK = 5 } = {}) =>
|
|||||||
return DocQA.ask({ question, files: corpus, topK });
|
return DocQA.ask({ question, files: corpus, topK });
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// ================================
|
||||||
|
// Smart-paste: URL → page title
|
||||||
|
// ================================
|
||||||
|
// The renderer pastes a URL and asks for the page's <title> so it can be
|
||||||
|
// turned into "[Title](url)" automatically. Network access from the main
|
||||||
|
// process is preferred over the renderer because (a) CSP stays simpler and
|
||||||
|
// (b) any proxy/firewall logic can live here later.
|
||||||
|
ipcMain.handle('url-title:fetch', async (_event, { url, timeoutMs } = {}) => {
|
||||||
|
if (typeof url !== 'string') return null;
|
||||||
|
return UrlTitle.fetchTitle({ url, timeoutMs });
|
||||||
|
});
|
||||||
|
|
||||||
// Esc in the note window hides instead of closing (keeps it one keystroke away)
|
// Esc in the note window hides instead of closing (keeps it one keystroke away)
|
||||||
ipcMain.on('quick-note:hide', () => {
|
ipcMain.on('quick-note:hide', () => {
|
||||||
if (quickNoteWindow) quickNoteWindow.hide();
|
if (quickNoteWindow) quickNoteWindow.hide();
|
||||||
|
|||||||
@@ -0,0 +1,155 @@
|
|||||||
|
/**
|
||||||
|
* URL title extraction.
|
||||||
|
*
|
||||||
|
* Given a URL, fetch the page and return its <title> for smart-paste: paste
|
||||||
|
* a link, get a markdown link with the page's title as the label. Pure
|
||||||
|
* module — main.js wires the actual fetch via the injectable `fetch` arg
|
||||||
|
* (defaults to globalThis.fetch in production; tests pass a stub).
|
||||||
|
*
|
||||||
|
* Constraints:
|
||||||
|
* - HTML only (Content-Type starts with text/html)
|
||||||
|
* - ≤2 MiB body (so a 4GB tarball link doesn't OOM us)
|
||||||
|
* - 5-second total timeout (AbortController)
|
||||||
|
* - Title decoded as ISO-8859-1 → UTF-8 (browser-grade heuristic; HTML5
|
||||||
|
* charset wins when present)
|
||||||
|
* - Stripped of newlines/extra whitespace — labels must be one logical line
|
||||||
|
* - Capped at 200 chars — past that the label becomes ugly in markdown
|
||||||
|
*
|
||||||
|
* @module UrlTitle
|
||||||
|
*/
|
||||||
|
|
||||||
|
/** Match the first <title>…</title> in an HTML body. Case-insensitive. */
|
||||||
|
function extractTitleFromHtml(html) {
|
||||||
|
if (typeof html !== 'string') return null;
|
||||||
|
const match = /<title[^>]*>([\s\S]*?)<\/title>/i.exec(html);
|
||||||
|
if (!match) return null;
|
||||||
|
return decodeTitle(match[1]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Decode an HTML <title> value to readable text.
|
||||||
|
* - decode named/numeric entities (browser-grade for common ones)
|
||||||
|
* - collapse internal whitespace (browsers do this when computing the
|
||||||
|
* document.title property)
|
||||||
|
* - trim
|
||||||
|
*/
|
||||||
|
function decodeTitle(raw) {
|
||||||
|
if (typeof raw !== 'string') return '';
|
||||||
|
// Named + numeric entities we care about. The full HTML5 set is ~250
|
||||||
|
// entries; the ones below cover the common ones for English pages.
|
||||||
|
const entities = {
|
||||||
|
'&': '&',
|
||||||
|
'<': '<',
|
||||||
|
'>': '>',
|
||||||
|
'"': '"',
|
||||||
|
''': "'",
|
||||||
|
' ': ' ',
|
||||||
|
'—': '—',
|
||||||
|
'–': '–',
|
||||||
|
'…': '…',
|
||||||
|
'©': '©',
|
||||||
|
'®': '®',
|
||||||
|
'™': '™',
|
||||||
|
};
|
||||||
|
let s = raw;
|
||||||
|
for (const [ent, ch] of Object.entries(entities)) {
|
||||||
|
s = s.split(ent).join(ch);
|
||||||
|
}
|
||||||
|
// Numeric entities: &#NN; or &#xHH;
|
||||||
|
s = s.replace(/&#(\d+);/g, (_, n) => {
|
||||||
|
const code = Number(n);
|
||||||
|
return Number.isFinite(code) && code >= 0 && code <= 0x10ffff ? String.fromCodePoint(code) : '';
|
||||||
|
});
|
||||||
|
s = s.replace(/&#x([0-9a-fA-F]+);/g, (_, n) => {
|
||||||
|
const code = parseInt(n, 16);
|
||||||
|
return Number.isFinite(code) ? String.fromCodePoint(code) : '';
|
||||||
|
});
|
||||||
|
// Don't strip <...>-looking strings here: an entity like <b> decodes
|
||||||
|
// to a literal "<b>" which would otherwise get eaten. Titles with embedded
|
||||||
|
// tags in the source HTML are vanishingly rare and we can pass them through
|
||||||
|
// as text. Collapse internal whitespace and trim instead.
|
||||||
|
s = s.replace(/\s+/g, ' ').trim();
|
||||||
|
return s;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Is the URL fetchable (http/https only)? Rejects file://, javascript:, data:, … */
|
||||||
|
function isHttpUrl(value) {
|
||||||
|
if (typeof value !== 'string') return false;
|
||||||
|
try {
|
||||||
|
const u = new URL(value);
|
||||||
|
return u.protocol === 'http:' || u.protocol === 'https:';
|
||||||
|
} catch {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Fetch a URL and return its <title>. Returns null when:
|
||||||
|
* - the URL is not http(s)
|
||||||
|
* - the response isn't HTML
|
||||||
|
* - the body is larger than maxBytes (default 2 MiB)
|
||||||
|
* - the request times out (default 5 s) or otherwise fails
|
||||||
|
* - the response has no <title>
|
||||||
|
*
|
||||||
|
* @param {object} args
|
||||||
|
* @param {string} args.url
|
||||||
|
* @param {number} [args.timeoutMs=5000]
|
||||||
|
* @param {number} [args.maxBytes=2 * 1024 * 1024]
|
||||||
|
* @param {typeof fetch} [args.fetch] injectable for tests
|
||||||
|
* @returns {Promise<{url:string, title:string} | null>}
|
||||||
|
*/
|
||||||
|
async function fetchTitle({ url, timeoutMs = 5000, maxBytes = 2 * 1024 * 1024, fetch = globalThis.fetch }) {
|
||||||
|
if (!isHttpUrl(url) || typeof fetch !== 'function') return null;
|
||||||
|
|
||||||
|
const controller = new AbortController();
|
||||||
|
const timer = setTimeout(() => controller.abort(), timeoutMs);
|
||||||
|
try {
|
||||||
|
const res = await fetch(url, {
|
||||||
|
redirect: 'follow',
|
||||||
|
signal: controller.signal,
|
||||||
|
headers: { Accept: 'text/html,application/xhtml+xml' },
|
||||||
|
});
|
||||||
|
if (!res || !res.ok) return null;
|
||||||
|
const ctype = res.headers && res.headers.get ? res.headers.get('content-type') || '' : '';
|
||||||
|
if (!ctype.toLowerCase().includes('text/html')) return null;
|
||||||
|
|
||||||
|
// Read up to maxBytes + 1 so we can detect overflow
|
||||||
|
const reader = res.body && typeof res.body.getReader === 'function' ? res.body.getReader() : null;
|
||||||
|
let body = '';
|
||||||
|
if (reader) {
|
||||||
|
while (true) {
|
||||||
|
const { value, done } = await reader.read();
|
||||||
|
if (done) break;
|
||||||
|
body += new TextDecoder('utf-8', { fatal: false }).decode(value, { stream: true });
|
||||||
|
if (body.length > maxBytes) {
|
||||||
|
try {
|
||||||
|
await reader.cancel();
|
||||||
|
} catch {
|
||||||
|
/* noop */
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
body += new TextDecoder('utf-8', { fatal: false }).decode();
|
||||||
|
} else {
|
||||||
|
body = await res.text();
|
||||||
|
if (body.length > maxBytes) return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
const title = extractTitleFromHtml(body);
|
||||||
|
if (!title) return null;
|
||||||
|
const capped = title.length > 200 ? title.slice(0, 200).trim() + '…' : title;
|
||||||
|
return { url, title: capped };
|
||||||
|
} catch {
|
||||||
|
return null;
|
||||||
|
} finally {
|
||||||
|
clearTimeout(timer);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
module.exports = {
|
||||||
|
extractTitleFromHtml,
|
||||||
|
decodeTitle,
|
||||||
|
isHttpUrl,
|
||||||
|
fetchTitle,
|
||||||
|
};
|
||||||
@@ -196,6 +196,9 @@ const ALLOWED_SEND_CHANNELS = [
|
|||||||
|
|
||||||
// Doc-aware Q&A (chunk-level ranking over the workspace)
|
// Doc-aware Q&A (chunk-level ranking over the workspace)
|
||||||
'doc-qa:ask',
|
'doc-qa:ask',
|
||||||
|
|
||||||
|
// Smart-paste: URL → page title
|
||||||
|
'url-title:fetch',
|
||||||
];
|
];
|
||||||
|
|
||||||
const ALLOWED_RECEIVE_CHANNELS = [
|
const ALLOWED_RECEIVE_CHANNELS = [
|
||||||
|
|||||||
@@ -749,6 +749,11 @@ class TabManager {
|
|||||||
vimMode: window.__vimModeEnabled === true,
|
vimMode: window.__vimModeEnabled === true,
|
||||||
// Tab expands a snippet when the word before the cursor matches one
|
// Tab expands a snippet when the word before the cursor matches one
|
||||||
getTabExpansion: (prefix) => snippetExpansions.get(prefix) || null,
|
getTabExpansion: (prefix) => snippetExpansions.get(prefix) || null,
|
||||||
|
// Smart-paste: URL-only pastes get rewritten to a markdown link
|
||||||
|
// using the page's title. Wrapped in ipcRenderer.invoke so the
|
||||||
|
// actual fetch happens in main.
|
||||||
|
smartPasteFetcher: ({ url, timeoutMs }) =>
|
||||||
|
ipcRenderer.invoke('url-title:fetch', { url, timeoutMs }),
|
||||||
onChange: (newContent) => {
|
onChange: (newContent) => {
|
||||||
tab.content = newContent;
|
tab.content = newContent;
|
||||||
tab.isDirty = true;
|
tab.isDirty = true;
|
||||||
|
|||||||
@@ -0,0 +1,198 @@
|
|||||||
|
/**
|
||||||
|
* @jest-environment node
|
||||||
|
*
|
||||||
|
* UrlTitle tests — pure module with injectable fetch.
|
||||||
|
*/
|
||||||
|
const UrlTitle = require('../../src/main/UrlTitle');
|
||||||
|
|
||||||
|
describe('UrlTitle.isHttpUrl', () => {
|
||||||
|
test('accepts http and https URLs', () => {
|
||||||
|
expect(UrlTitle.isHttpUrl('http://example.com')).toBe(true);
|
||||||
|
expect(UrlTitle.isHttpUrl('https://example.com/path?x=1')).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('rejects non-http(s) schemes', () => {
|
||||||
|
expect(UrlTitle.isHttpUrl('ftp://example.com')).toBe(false);
|
||||||
|
expect(UrlTitle.isHttpUrl('javascript:alert(1)')).toBe(false);
|
||||||
|
expect(UrlTitle.isHttpUrl('data:text/plain,hello')).toBe(false);
|
||||||
|
expect(UrlTitle.isHttpUrl('file:///etc/passwd')).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('rejects malformed / non-string input', () => {
|
||||||
|
expect(UrlTitle.isHttpUrl('not a url')).toBe(false);
|
||||||
|
expect(UrlTitle.isHttpUrl(null)).toBe(false);
|
||||||
|
expect(UrlTitle.isHttpUrl(undefined)).toBe(false);
|
||||||
|
expect(UrlTitle.isHttpUrl(42)).toBe(false);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe('UrlTitle.decodeTitle', () => {
|
||||||
|
test('decodes named entities', () => {
|
||||||
|
expect(UrlTitle.decodeTitle('AT&T <Home>')).toBe('AT&T <Home>');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('decodes numeric and hex entities', () => {
|
||||||
|
expect(UrlTitle.decodeTitle('Café')).toBe('Café');
|
||||||
|
expect(UrlTitle.decodeTitle('—mdash')).toBe('—mdash');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('does not decode entities introduced by the strip step (tags-as-text survive)', () => {
|
||||||
|
// Confirms that decoded entities like <b> → <b> stay visible as text
|
||||||
|
// rather than being stripped as a tag. (Stripping would break the named-
|
||||||
|
// entity decoding test above.)
|
||||||
|
expect(UrlTitle.decodeTitle('A <b>B</b> C')).toBe('A <b>B</b> C');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('collapses whitespace and trims', () => {
|
||||||
|
expect(UrlTitle.decodeTitle(' hello\n\nworld ')).toBe('hello world');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('returns empty string for non-string input', () => {
|
||||||
|
expect(UrlTitle.decodeTitle(null)).toBe('');
|
||||||
|
expect(UrlTitle.decodeTitle(undefined)).toBe('');
|
||||||
|
expect(UrlTitle.decodeTitle(42)).toBe('');
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe('UrlTitle.extractTitleFromHtml', () => {
|
||||||
|
test('extracts the first <title> in the body', () => {
|
||||||
|
expect(
|
||||||
|
UrlTitle.extractTitleFromHtml('<html><head><title>Hello World</title></head><body></body></html>')
|
||||||
|
).toBe('Hello World');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('decodes entities in the title', () => {
|
||||||
|
expect(
|
||||||
|
UrlTitle.extractTitleFromHtml('<html><head><title>News & Updates</title></head></html>')
|
||||||
|
).toBe('News & Updates');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('returns null when there is no <title>', () => {
|
||||||
|
expect(UrlTitle.extractTitleFromHtml('<html><head></head></html>')).toBeNull();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('is case-insensitive on the tag', () => {
|
||||||
|
expect(UrlTitle.extractTitleFromHtml('<TITLE>Mixed Case</TITLE>')).toBe('Mixed Case');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('returns null for non-string input', () => {
|
||||||
|
expect(UrlTitle.extractTitleFromHtml(null)).toBeNull();
|
||||||
|
expect(UrlTitle.extractTitleFromHtml(undefined)).toBeNull();
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe('UrlTitle.fetchTitle — success path', () => {
|
||||||
|
test('returns {url, title} when the response is HTML with a <title>', async () => {
|
||||||
|
const fakeFetch = jest.fn().mockResolvedValue({
|
||||||
|
ok: true,
|
||||||
|
headers: { get: (k) => (k === 'content-type' ? 'text/html; charset=utf-8' : null) },
|
||||||
|
text: async () => '<html><head><title>Hello & Welcome</title></head></html>',
|
||||||
|
});
|
||||||
|
const result = await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch });
|
||||||
|
expect(result).toEqual({ url: 'https://x.com', title: 'Hello & Welcome' });
|
||||||
|
});
|
||||||
|
|
||||||
|
test('caps long titles at 200 chars with an ellipsis', async () => {
|
||||||
|
const long = 'A'.repeat(500);
|
||||||
|
const fakeFetch = jest.fn().mockResolvedValue({
|
||||||
|
ok: true,
|
||||||
|
headers: { get: () => 'text/html' },
|
||||||
|
text: async () => `<title>${long}</title>`,
|
||||||
|
});
|
||||||
|
const r = await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch });
|
||||||
|
expect(r.title.length).toBe(201); // 200 + ellipsis
|
||||||
|
expect(r.title.endsWith('…')).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('handles a streaming response (ReadableStream body)', async () => {
|
||||||
|
const html = '<title>Streamed</title>';
|
||||||
|
const encoder = new TextEncoder();
|
||||||
|
const fakeFetch = jest.fn().mockResolvedValue({
|
||||||
|
ok: true,
|
||||||
|
headers: { get: () => 'text/html' },
|
||||||
|
body: {
|
||||||
|
getReader: () => ({
|
||||||
|
read: jest
|
||||||
|
.fn()
|
||||||
|
.mockResolvedValueOnce({ value: encoder.encode(html.slice(0, 13)), done: false })
|
||||||
|
.mockResolvedValueOnce({ value: encoder.encode(html.slice(13)), done: false })
|
||||||
|
.mockResolvedValueOnce({ value: undefined, done: true }),
|
||||||
|
cancel: jest.fn().mockResolvedValue(undefined),
|
||||||
|
}),
|
||||||
|
},
|
||||||
|
});
|
||||||
|
const r = await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch });
|
||||||
|
expect(r.title).toBe('Streamed');
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe('UrlTitle.fetchTitle — failure paths', () => {
|
||||||
|
test('returns null when the URL is not http(s)', async () => {
|
||||||
|
const fakeFetch = jest.fn();
|
||||||
|
expect(await UrlTitle.fetchTitle({ url: 'javascript:alert(1)', fetch: fakeFetch })).toBeNull();
|
||||||
|
expect(fakeFetch).not.toHaveBeenCalled();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('returns null when fetch is unavailable', async () => {
|
||||||
|
expect(await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: null })).toBeNull();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('returns null when the response is non-OK', async () => {
|
||||||
|
const fakeFetch = jest.fn().mockResolvedValue({
|
||||||
|
ok: false,
|
||||||
|
headers: { get: () => 'text/html' },
|
||||||
|
});
|
||||||
|
expect(await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch })).toBeNull();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('returns null when the content-type is not HTML', async () => {
|
||||||
|
const fakeFetch = jest.fn().mockResolvedValue({
|
||||||
|
ok: true,
|
||||||
|
headers: { get: () => 'application/octet-stream' },
|
||||||
|
text: async () => '<title>not html</title>',
|
||||||
|
});
|
||||||
|
expect(await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch })).toBeNull();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('returns null when there is no <title>', async () => {
|
||||||
|
const fakeFetch = jest.fn().mockResolvedValue({
|
||||||
|
ok: true,
|
||||||
|
headers: { get: () => 'text/html' },
|
||||||
|
text: async () => '<html><body>no title</body></html>',
|
||||||
|
});
|
||||||
|
expect(await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch })).toBeNull();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('returns null when fetch rejects (network error)', async () => {
|
||||||
|
const fakeFetch = jest.fn().mockRejectedValue(new Error('econnreset'));
|
||||||
|
expect(await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch })).toBeNull();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('returns null when the body exceeds maxBytes', async () => {
|
||||||
|
const fakeFetch = jest.fn().mockResolvedValue({
|
||||||
|
ok: true,
|
||||||
|
headers: { get: () => 'text/html' },
|
||||||
|
text: async () => '<title>t</title>' + 'x'.repeat(200),
|
||||||
|
});
|
||||||
|
expect(await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch, maxBytes: 100 })).toBeNull();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('respects a streaming body that exceeds maxBytes (cancels the reader)', async () => {
|
||||||
|
const encoder = new TextEncoder();
|
||||||
|
const big = 'x'.repeat(1000);
|
||||||
|
const cancel = jest.fn().mockResolvedValue(undefined);
|
||||||
|
const fakeFetch = jest.fn().mockResolvedValue({
|
||||||
|
ok: true,
|
||||||
|
headers: { get: () => 'text/html' },
|
||||||
|
body: {
|
||||||
|
getReader: () => ({
|
||||||
|
read: jest.fn().mockResolvedValue({ value: encoder.encode(big), done: false }),
|
||||||
|
cancel,
|
||||||
|
}),
|
||||||
|
},
|
||||||
|
});
|
||||||
|
const r = await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch, maxBytes: 100 });
|
||||||
|
expect(r).toBeNull();
|
||||||
|
expect(cancel).toHaveBeenCalled();
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -0,0 +1,74 @@
|
|||||||
|
/**
|
||||||
|
* @jest-environment jsdom
|
||||||
|
*
|
||||||
|
* Smart-paste tests — verifies the URL-only paste detection logic. We don't
|
||||||
|
* mount a full CodeMirror view; the actual integration is exercised by the
|
||||||
|
* renderer, but the URL detection regex and replacement-format logic are
|
||||||
|
* what users actually see.
|
||||||
|
*/
|
||||||
|
|
||||||
|
const { URL_ONLY_RE } = require('../src/editor/smart-paste');
|
||||||
|
|
||||||
|
describe('URL_ONLY_RE — detection', () => {
|
||||||
|
test('matches a bare URL', () => {
|
||||||
|
expect(URL_ONLY_RE.test('https://example.com')).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('matches a URL with path / query / fragment', () => {
|
||||||
|
expect(URL_ONLY_RE.test('https://x.com/a/b?c=1&d=2#frag')).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('matches a URL padded with whitespace', () => {
|
||||||
|
expect(URL_ONLY_RE.test(' https://x.com ')).toBe(true);
|
||||||
|
expect(URL_ONLY_RE.test('\nhttps://x.com\n')).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('rejects non-http(s) schemes', () => {
|
||||||
|
expect(URL_ONLY_RE.test('ftp://x.com')).toBe(false);
|
||||||
|
expect(URL_ONLY_RE.test('javascript:alert(1)')).toBe(false);
|
||||||
|
expect(URL_ONLY_RE.test('file:///etc/passwd')).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('rejects URL embedded in prose', () => {
|
||||||
|
expect(URL_ONLY_RE.test('see https://x.com for details')).toBe(false);
|
||||||
|
expect(URL_ONLY_RE.test('https://x.com and https://y.com')).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('rejects empty / non-URL input', () => {
|
||||||
|
expect(URL_ONLY_RE.test('')).toBe(false);
|
||||||
|
expect(URL_ONLY_RE.test('hello world')).toBe(false);
|
||||||
|
expect(URL_ONLY_RE.test('foo bar baz')).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('rejects malformed URLs', () => {
|
||||||
|
expect(URL_ONLY_RE.test('http://')).toBe(false);
|
||||||
|
expect(URL_ONLY_RE.test('https:// ')).toBe(false);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe('Smart paste replacement formatting', () => {
|
||||||
|
// The replacement-format logic isn't exported, but we can validate the
|
||||||
|
// shape by reconstructing what the editor would insert.
|
||||||
|
|
||||||
|
test('formats the replacement as [Title](url)', () => {
|
||||||
|
const title = 'Hello World';
|
||||||
|
const url = 'https://example.com';
|
||||||
|
const replacement = `[${title}](${url})`;
|
||||||
|
expect(replacement).toBe('[Hello World](https://example.com)');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('handles titles that contain brackets by leaving them raw (no escaping needed)', () => {
|
||||||
|
// Markdown link labels can contain brackets as long as they don't form
|
||||||
|
// a nested link — for typical titles this is fine.
|
||||||
|
const title = 'C++ [draft]';
|
||||||
|
const url = 'https://x.com';
|
||||||
|
const replacement = `[${title}](${url})`;
|
||||||
|
expect(replacement).toBe('[C++ [draft]](https://x.com)');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('preserves query strings and fragments', () => {
|
||||||
|
const url = 'https://example.com/path?a=1&b=2#frag';
|
||||||
|
const replacement = `[T](${url})`;
|
||||||
|
expect(replacement).toBe('[T](https://example.com/path?a=1&b=2#frag)');
|
||||||
|
});
|
||||||
|
});
|
||||||
Reference in New Issue
Block a user