src/text.js (3199 bytes)
1 // Pure text helpers: titles, tags, bullet splitting, and cheap duplicate detection. 2 3 const STOPWORDS = new Set( 4 ('a an and any are as at be but by can could do does for from get got has have how i if in into is it its ' + 5 'just like make me more my of on or our out so some than that the their them then there these they this ' + 6 'to too up use using via was way we what when where which while who why will with would you your') 7 .split(' '), 8 ); 9 10 /** First non-empty line, whitespace-collapsed, cut at a word boundary. */ 11 export function deriveTitle(text, max = 72) { 12 const first = String(text).split(/\r?\n/).map((s) => s.trim()).find(Boolean) || ''; 13 const clean = first 14 .replace(/^(?:[-*•]|\d+[.)])\s+/, '') 15 .replace(/(?:^|\s)#[a-z][\w-]{0,31}(?=\s|$)/gi, ' ') 16 .replace(/\s+/g, ' ') 17 .trim() || first; 18 if (clean.length <= max) return clean; 19 const cut = clean.slice(0, max - 1); 20 const space = cut.lastIndexOf(' '); 21 return (space > max * 0.6 ? cut.slice(0, space) : cut).replace(/[\s,.;:-]+$/, '') + '…'; 22 } 23 24 /** `#tag` words (must start with a letter, so `#12` and `C#` are not tags). */ 25 export function extractTags(text) { 26 const tags = new Set(); 27 for (const m of String(text).matchAll(/(?:^|\s)#([a-z][\w-]{0,31})/gi)) tags.add(m[1].toLowerCase()); 28 return [...tags]; 29 } 30 31 export function normalizeTags(tags) { 32 if (!Array.isArray(tags)) return []; 33 return [...new Set(tags.map((t) => String(t).trim().replace(/^#/, '').toLowerCase()).filter(Boolean))]; 34 } 35 36 const BULLET = /^(\s{0,3})(?:[-*•]|\d+[.)])\s+/; 37 38 /** 39 * A pasted bulleted list becomes one idea per bullet; anything else is one idea. 40 * Indented lines under a bullet are kept with that bullet. 41 */ 42 export function splitIdeas(text) { 43 const lines = String(text).split(/\r?\n/); 44 const items = []; 45 let pureList = true; 46 for (const line of lines) { 47 if (!line.trim()) continue; 48 if (BULLET.test(line)) items.push(line.replace(BULLET, '').trim()); 49 else if (items.length && /^\s+/.test(line)) items[items.length - 1] += '\n' + line.trim(); 50 else { 51 pureList = false; 52 break; 53 } 54 } 55 if (pureList && items.length >= 2) return items.filter(Boolean); 56 const whole = String(text).trim(); 57 return whole ? [whole] : []; 58 } 59 60 export function wordSet(text) { 61 const words = String(text) 62 .toLowerCase() 63 .normalize('NFKD') 64 .replace(/\p{M}/gu, '') 65 .replace(/[^\p{L}\p{N}\s]/gu, ' ') 66 .split(/\s+/) 67 .filter((w) => w.length > 2 && !STOPWORDS.has(w)) 68 .map((w) => (w.length > 4 && w.endsWith('s') && !w.endsWith('ss') ? w.slice(0, -1) : w)); 69 return new Set(words); 70 } 71 72 /** Jaccard similarity of two word sets, plus the shared-word count. */ 73 export function similarity(a, b) { 74 let shared = 0; 75 for (const w of a) if (b.has(w)) shared++; 76 const union = a.size + b.size - shared; 77 return { score: union ? shared / union : 0, shared }; 78 } 79 80 export function isSimilar(a, b) { 81 const { score, shared } = similarity(a, b); 82 return (score >= 0.55 && shared >= 2) || score >= 0.8; 83 } 84 85 export function clip(text, max) { 86 const s = String(text ?? '').replace(/\s+/g, ' ').trim(); 87 return s.length <= max ? s : s.slice(0, max - 1).trimEnd() + '…'; 88 }