mirror of
https://github.com/amitwh/markdown-converter.git
synced 2026-10-01 09:19:34 +05:30
feat(paste): smart URL → markdown-link on URL-only pastes
- src/main/UrlTitle.js — fetch a URL, return its <title>. Pure module
with injectable fetch for tests. Decodes named + numeric + hex entities
(AT&T, Café, —), strips the <title> tags, collapses
whitespace, caps the label at 200 chars. 5s timeout via AbortController,
2 MiB body cap to avoid OOM on big downloads, streaming reader with
overflow cancellation. Rejects non-http(s) URLs up front.
- src/main.js — IPC url-title:fetch proxies to fetchTitle.
- src/editor/smart-paste.js — CodeMirror 6 extension that detects
URL-only pastes (one URL, surrounded only by whitespace), inserts the
URL immediately, then async-rewrites the insertion range to
[Title](url) once the title arrives. Multi-line / prose pastes pass
through untouched.
- src/editor/codemirror-setup.js — accepts smartPasteFetcher option and
pushes the extension when provided.
- src/renderer.js — passes ipcRenderer.invoke('url-title:fetch') as
the fetcher.
- src/preload.js — url-title:fetch added to ALLOWED_SEND_CHANNELS.
- eslint.config.js — AbortController / TextDecoder / TextEncoder added
to globals (available in Node 20+ and Chromium).
Tests (34 new):
- tests/main/UrlTitle.test.js (24): isHttpUrl scheme filter, decodeTitle
named/numeric/hex entities + whitespace + non-string, extractTitleFromHtml
first match + case-insensitive + null fallback, fetchTitle success +
long-title cap + streaming body, all failure paths (non-http,
no-fetch, non-OK, wrong content-type, no <title>, network error,
body over maxBytes — including streaming overflow cancellation).
- tests/smart-paste.test.js (10): URL_ONLY_RE detection (bare URL,
path/query/fragment, whitespace padding, scheme rejection, embedded
rejection, empty/malformed), replacement format ([T](url), brackets
in title survive, query strings preserved).
Full suite: 71 suites, 831 tests, lint+format clean.
Amit Haridas
This commit is contained in:
@@ -0,0 +1,198 @@
|
||||
/**
|
||||
* @jest-environment node
|
||||
*
|
||||
* UrlTitle tests — pure module with injectable fetch.
|
||||
*/
|
||||
const UrlTitle = require('../../src/main/UrlTitle');
|
||||
|
||||
describe('UrlTitle.isHttpUrl', () => {
|
||||
test('accepts http and https URLs', () => {
|
||||
expect(UrlTitle.isHttpUrl('http://example.com')).toBe(true);
|
||||
expect(UrlTitle.isHttpUrl('https://example.com/path?x=1')).toBe(true);
|
||||
});
|
||||
|
||||
test('rejects non-http(s) schemes', () => {
|
||||
expect(UrlTitle.isHttpUrl('ftp://example.com')).toBe(false);
|
||||
expect(UrlTitle.isHttpUrl('javascript:alert(1)')).toBe(false);
|
||||
expect(UrlTitle.isHttpUrl('data:text/plain,hello')).toBe(false);
|
||||
expect(UrlTitle.isHttpUrl('file:///etc/passwd')).toBe(false);
|
||||
});
|
||||
|
||||
test('rejects malformed / non-string input', () => {
|
||||
expect(UrlTitle.isHttpUrl('not a url')).toBe(false);
|
||||
expect(UrlTitle.isHttpUrl(null)).toBe(false);
|
||||
expect(UrlTitle.isHttpUrl(undefined)).toBe(false);
|
||||
expect(UrlTitle.isHttpUrl(42)).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('UrlTitle.decodeTitle', () => {
|
||||
test('decodes named entities', () => {
|
||||
expect(UrlTitle.decodeTitle('AT&T <Home>')).toBe('AT&T <Home>');
|
||||
});
|
||||
|
||||
test('decodes numeric and hex entities', () => {
|
||||
expect(UrlTitle.decodeTitle('Café')).toBe('Café');
|
||||
expect(UrlTitle.decodeTitle('—mdash')).toBe('—mdash');
|
||||
});
|
||||
|
||||
test('does not decode entities introduced by the strip step (tags-as-text survive)', () => {
|
||||
// Confirms that decoded entities like <b> → <b> stay visible as text
|
||||
// rather than being stripped as a tag. (Stripping would break the named-
|
||||
// entity decoding test above.)
|
||||
expect(UrlTitle.decodeTitle('A <b>B</b> C')).toBe('A <b>B</b> C');
|
||||
});
|
||||
|
||||
test('collapses whitespace and trims', () => {
|
||||
expect(UrlTitle.decodeTitle(' hello\n\nworld ')).toBe('hello world');
|
||||
});
|
||||
|
||||
test('returns empty string for non-string input', () => {
|
||||
expect(UrlTitle.decodeTitle(null)).toBe('');
|
||||
expect(UrlTitle.decodeTitle(undefined)).toBe('');
|
||||
expect(UrlTitle.decodeTitle(42)).toBe('');
|
||||
});
|
||||
});
|
||||
|
||||
describe('UrlTitle.extractTitleFromHtml', () => {
|
||||
test('extracts the first <title> in the body', () => {
|
||||
expect(
|
||||
UrlTitle.extractTitleFromHtml('<html><head><title>Hello World</title></head><body></body></html>')
|
||||
).toBe('Hello World');
|
||||
});
|
||||
|
||||
test('decodes entities in the title', () => {
|
||||
expect(
|
||||
UrlTitle.extractTitleFromHtml('<html><head><title>News & Updates</title></head></html>')
|
||||
).toBe('News & Updates');
|
||||
});
|
||||
|
||||
test('returns null when there is no <title>', () => {
|
||||
expect(UrlTitle.extractTitleFromHtml('<html><head></head></html>')).toBeNull();
|
||||
});
|
||||
|
||||
test('is case-insensitive on the tag', () => {
|
||||
expect(UrlTitle.extractTitleFromHtml('<TITLE>Mixed Case</TITLE>')).toBe('Mixed Case');
|
||||
});
|
||||
|
||||
test('returns null for non-string input', () => {
|
||||
expect(UrlTitle.extractTitleFromHtml(null)).toBeNull();
|
||||
expect(UrlTitle.extractTitleFromHtml(undefined)).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe('UrlTitle.fetchTitle — success path', () => {
|
||||
test('returns {url, title} when the response is HTML with a <title>', async () => {
|
||||
const fakeFetch = jest.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
headers: { get: (k) => (k === 'content-type' ? 'text/html; charset=utf-8' : null) },
|
||||
text: async () => '<html><head><title>Hello & Welcome</title></head></html>',
|
||||
});
|
||||
const result = await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch });
|
||||
expect(result).toEqual({ url: 'https://x.com', title: 'Hello & Welcome' });
|
||||
});
|
||||
|
||||
test('caps long titles at 200 chars with an ellipsis', async () => {
|
||||
const long = 'A'.repeat(500);
|
||||
const fakeFetch = jest.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
headers: { get: () => 'text/html' },
|
||||
text: async () => `<title>${long}</title>`,
|
||||
});
|
||||
const r = await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch });
|
||||
expect(r.title.length).toBe(201); // 200 + ellipsis
|
||||
expect(r.title.endsWith('…')).toBe(true);
|
||||
});
|
||||
|
||||
test('handles a streaming response (ReadableStream body)', async () => {
|
||||
const html = '<title>Streamed</title>';
|
||||
const encoder = new TextEncoder();
|
||||
const fakeFetch = jest.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
headers: { get: () => 'text/html' },
|
||||
body: {
|
||||
getReader: () => ({
|
||||
read: jest
|
||||
.fn()
|
||||
.mockResolvedValueOnce({ value: encoder.encode(html.slice(0, 13)), done: false })
|
||||
.mockResolvedValueOnce({ value: encoder.encode(html.slice(13)), done: false })
|
||||
.mockResolvedValueOnce({ value: undefined, done: true }),
|
||||
cancel: jest.fn().mockResolvedValue(undefined),
|
||||
}),
|
||||
},
|
||||
});
|
||||
const r = await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch });
|
||||
expect(r.title).toBe('Streamed');
|
||||
});
|
||||
});
|
||||
|
||||
describe('UrlTitle.fetchTitle — failure paths', () => {
|
||||
test('returns null when the URL is not http(s)', async () => {
|
||||
const fakeFetch = jest.fn();
|
||||
expect(await UrlTitle.fetchTitle({ url: 'javascript:alert(1)', fetch: fakeFetch })).toBeNull();
|
||||
expect(fakeFetch).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
test('returns null when fetch is unavailable', async () => {
|
||||
expect(await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: null })).toBeNull();
|
||||
});
|
||||
|
||||
test('returns null when the response is non-OK', async () => {
|
||||
const fakeFetch = jest.fn().mockResolvedValue({
|
||||
ok: false,
|
||||
headers: { get: () => 'text/html' },
|
||||
});
|
||||
expect(await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch })).toBeNull();
|
||||
});
|
||||
|
||||
test('returns null when the content-type is not HTML', async () => {
|
||||
const fakeFetch = jest.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
headers: { get: () => 'application/octet-stream' },
|
||||
text: async () => '<title>not html</title>',
|
||||
});
|
||||
expect(await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch })).toBeNull();
|
||||
});
|
||||
|
||||
test('returns null when there is no <title>', async () => {
|
||||
const fakeFetch = jest.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
headers: { get: () => 'text/html' },
|
||||
text: async () => '<html><body>no title</body></html>',
|
||||
});
|
||||
expect(await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch })).toBeNull();
|
||||
});
|
||||
|
||||
test('returns null when fetch rejects (network error)', async () => {
|
||||
const fakeFetch = jest.fn().mockRejectedValue(new Error('econnreset'));
|
||||
expect(await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch })).toBeNull();
|
||||
});
|
||||
|
||||
test('returns null when the body exceeds maxBytes', async () => {
|
||||
const fakeFetch = jest.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
headers: { get: () => 'text/html' },
|
||||
text: async () => '<title>t</title>' + 'x'.repeat(200),
|
||||
});
|
||||
expect(await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch, maxBytes: 100 })).toBeNull();
|
||||
});
|
||||
|
||||
test('respects a streaming body that exceeds maxBytes (cancels the reader)', async () => {
|
||||
const encoder = new TextEncoder();
|
||||
const big = 'x'.repeat(1000);
|
||||
const cancel = jest.fn().mockResolvedValue(undefined);
|
||||
const fakeFetch = jest.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
headers: { get: () => 'text/html' },
|
||||
body: {
|
||||
getReader: () => ({
|
||||
read: jest.fn().mockResolvedValue({ value: encoder.encode(big), done: false }),
|
||||
cancel,
|
||||
}),
|
||||
},
|
||||
});
|
||||
const r = await UrlTitle.fetchTitle({ url: 'https://x.com', fetch: fakeFetch, maxBytes: 100 });
|
||||
expect(r).toBeNull();
|
||||
expect(cancel).toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,74 @@
|
||||
/**
|
||||
* @jest-environment jsdom
|
||||
*
|
||||
* Smart-paste tests — verifies the URL-only paste detection logic. We don't
|
||||
* mount a full CodeMirror view; the actual integration is exercised by the
|
||||
* renderer, but the URL detection regex and replacement-format logic are
|
||||
* what users actually see.
|
||||
*/
|
||||
|
||||
const { URL_ONLY_RE } = require('../src/editor/smart-paste');
|
||||
|
||||
describe('URL_ONLY_RE — detection', () => {
|
||||
test('matches a bare URL', () => {
|
||||
expect(URL_ONLY_RE.test('https://example.com')).toBe(true);
|
||||
});
|
||||
|
||||
test('matches a URL with path / query / fragment', () => {
|
||||
expect(URL_ONLY_RE.test('https://x.com/a/b?c=1&d=2#frag')).toBe(true);
|
||||
});
|
||||
|
||||
test('matches a URL padded with whitespace', () => {
|
||||
expect(URL_ONLY_RE.test(' https://x.com ')).toBe(true);
|
||||
expect(URL_ONLY_RE.test('\nhttps://x.com\n')).toBe(true);
|
||||
});
|
||||
|
||||
test('rejects non-http(s) schemes', () => {
|
||||
expect(URL_ONLY_RE.test('ftp://x.com')).toBe(false);
|
||||
expect(URL_ONLY_RE.test('javascript:alert(1)')).toBe(false);
|
||||
expect(URL_ONLY_RE.test('file:///etc/passwd')).toBe(false);
|
||||
});
|
||||
|
||||
test('rejects URL embedded in prose', () => {
|
||||
expect(URL_ONLY_RE.test('see https://x.com for details')).toBe(false);
|
||||
expect(URL_ONLY_RE.test('https://x.com and https://y.com')).toBe(false);
|
||||
});
|
||||
|
||||
test('rejects empty / non-URL input', () => {
|
||||
expect(URL_ONLY_RE.test('')).toBe(false);
|
||||
expect(URL_ONLY_RE.test('hello world')).toBe(false);
|
||||
expect(URL_ONLY_RE.test('foo bar baz')).toBe(false);
|
||||
});
|
||||
|
||||
test('rejects malformed URLs', () => {
|
||||
expect(URL_ONLY_RE.test('http://')).toBe(false);
|
||||
expect(URL_ONLY_RE.test('https:// ')).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Smart paste replacement formatting', () => {
|
||||
// The replacement-format logic isn't exported, but we can validate the
|
||||
// shape by reconstructing what the editor would insert.
|
||||
|
||||
test('formats the replacement as [Title](url)', () => {
|
||||
const title = 'Hello World';
|
||||
const url = 'https://example.com';
|
||||
const replacement = `[${title}](${url})`;
|
||||
expect(replacement).toBe('[Hello World](https://example.com)');
|
||||
});
|
||||
|
||||
test('handles titles that contain brackets by leaving them raw (no escaping needed)', () => {
|
||||
// Markdown link labels can contain brackets as long as they don't form
|
||||
// a nested link — for typical titles this is fine.
|
||||
const title = 'C++ [draft]';
|
||||
const url = 'https://x.com';
|
||||
const replacement = `[${title}](${url})`;
|
||||
expect(replacement).toBe('[C++ [draft]](https://x.com)');
|
||||
});
|
||||
|
||||
test('preserves query strings and fragments', () => {
|
||||
const url = 'https://example.com/path?a=1&b=2#frag';
|
||||
const replacement = `[T](${url})`;
|
||||
expect(replacement).toBe('[T](https://example.com/path?a=1&b=2#frag)');
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user