diff --git a/src/__tests__/content.test.ts b/src/__tests__/content.test.ts index 48a610e..7a80eba 100644 --- a/src/__tests__/content.test.ts +++ b/src/__tests__/content.test.ts @@ -142,6 +142,18 @@ describe('parseDoc', () => { expect(searchDoc(page, 'needle')?.excerpt).toContain('😀…'); }); + it('keeps a grapheme cluster intact at the end of a search excerpt', () => { + const page = parseDoc('search/grapheme.md', `# X\n\nmatch ${'a'.repeat(171)}👩‍💻 tail`); + + expect(searchDoc(page, 'match')?.excerpt).toBe(`X match ${'a'.repeat(171)}…`); + }); + + it('keeps a grapheme cluster intact when truncating a summary', () => { + const page = parseDoc('search/summary.md', `${'a'.repeat(159)}👩‍💻 tail`); + + expect(page.summary).toBe(`${'a'.repeat(159)}…`); + }); + it('maps a match after NFKC-expanded ligatures back to the source excerpt', () => { const page = parseDoc('search/ligature.md', `# X\n\n${'fi'.repeat(100)} needle tail`); @@ -169,6 +181,12 @@ describe('parseDoc', () => { expect(searchDoc(page, 'needle')?.excerpt).toContain('needle'); }); + it('maps a match that follows many expanded and composed characters', () => { + const page = parseDoc('search/long-mixed.md', `# X\n\n${'İ fi ㄱㅏ '.repeat(300)}needle tail`); + + expect(searchDoc(page, 'needle')?.excerpt).toBe(`…${'ㄱㅏ İ fi '.repeat(8)}ㄱㅏ needle tail`); + }); + it('truncates a query longer than the excerpt window from the match start', () => { const query = 'q'.repeat(240); const page = parseDoc('search/long-query.md', `# X\n\n${query} tail`); diff --git a/src/client.ts b/src/client.ts index 34e9c04..e5b3b7b 100644 --- a/src/client.ts +++ b/src/client.ts @@ -248,11 +248,9 @@ function cancelUnusedResponseBody(response: Response | undefined): void { if (!response || response.bodyUsed) return; const body = response.body; if (!body) return; - void body.cancel().catch(() => { - // A transport may close or lock the body while the request settles. - }); + void body.cancel().catch(() => undefined); } catch { - // Cleanup is best-effort and must not override the request result. + // Cleanup must not override the request result. } } @@ -341,8 +339,7 @@ export function createDocsClient(options: DocsClientOptions = {}): DocsClient { return await consume(response); } finally { clearTimeout(timer); - // Cleanup must not delay a stale-cache result if a custom transport's - // cancel implementation never settles. + // Not awaited: a custom transport's cancel may never settle. cancelUnusedResponseBody(response); } } diff --git a/src/content.ts b/src/content.ts index 8d76139..7c17d02 100644 --- a/src/content.ts +++ b/src/content.ts @@ -3,6 +3,10 @@ import type { DocComponent, DocPage, DocsSearchResult } from './types.js'; const SUMMARY_LENGTH = 160; const EXCERPT_LENGTH = 180; const GRAPHEME_SEGMENTER = new Intl.Segmenter(undefined, { granularity: 'grapheme' }); +const LIST_ITEM = /^( {0,3}(?:[-+*]|\d{1,9}[.)]))([ \t]+)/; +const NORMALIZATION_CHUNK = 512; +// NFKC never joins these characters to what precedes them, so chunks split before them. +const NORMALIZATION_BOUNDARY = /[ -~\u4e00-\u9fff]/g; interface ParsedSource { body: string; @@ -106,8 +110,7 @@ function transitionFence(line: string, current: MarkdownFence | undefined): Fenc candidate = candidate.slice(Math.min(indentation, current.listIndent)); } } else { - const listPrefix = /^ {0,3}(?:(?:[-+*]|\d{1,9}[.)]))[ \t]+/.exec(candidate)?.[0] ?? ''; - listIndent = listPrefix.length; + listIndent = LIST_ITEM.exec(candidate)?.[0].length ?? 0; candidate = candidate.slice(listIndent); } const match = /^ {0,3}(`{3,}|~{3,})(.*)$/.exec(candidate); @@ -165,11 +168,12 @@ function proseLines(body: string): (string | undefined)[] { inCode = width - listIndent >= 4 && (afterBreak || inCode); afterBreak = false; if (inCode) return undefined; - const item = /^ {0,3}(?:[-+*]|\d{1,9}[.)])([ \t]+)(.*)$/.exec(candidate); + const item = LIST_ITEM.exec(candidate); if (!item) return line; - const padding = item[1] ?? ''; - const marker = candidate.length - (item[2] ?? '').length - padding.length; - if (padding.includes('\t') || padding.length >= 5 || /^(?: {4}|\t)/.test(item[2] ?? '')) { + const marker = item[1]?.length ?? 0; + const padding = item[2] ?? ''; + const rest = candidate.slice(item[0].length); + if (padding.includes('\t') || padding.length >= 5 || /^(?: {4}|\t)/.test(rest)) { listIndent = marker + 1; inCode = true; return undefined; @@ -189,10 +193,14 @@ function extractTitle(body: string): string | undefined { } function truncate(value: string, length: number): string { - const characters: string[] = []; - for (const character of value) characters.push(character); - if (characters.length <= length) return value; - return `${characters.slice(0, length).join('').trimEnd()}…`; + let end = 0; + let codePoints = 0; + for (const { index, segment } of GRAPHEME_SEGMENTER.segment(value)) { + codePoints += Array.from(segment).length; + if (codePoints > length) return `${value.slice(0, end).trimEnd()}…`; + end = index + segment.length; + } + return value; } function extractSummary(body: string): string { @@ -355,76 +363,94 @@ export function parseDoc(path: string, content: string): DocPage { } function normalize(value: string): string { - // JavaScript lowercasing preserves Greek final sigma (ς), while a - // case-insensitive search for Σ produces σ. Fold the positional variants to - // one form after NFKC/lowercase so substring matching is position agnostic. + // Lowercasing keeps final sigma (ς) where a search for Σ yields σ. return value .normalize('NFKC') .toLowerCase() .replace(/\u03c2/g, '\u03c3'); } -interface NormalizedIndex { - boundaries: number[]; - codePoints: number[]; +interface NormalizedText { + offsets: number[]; + sources: number[]; value: string; } -function indexNormalizedText(value: string): NormalizedIndex { - const boundaries: number[] = []; - const codePoints: number[] = []; - for (const part of GRAPHEME_SEGMENTER.segment(value)) { - boundaries.push(part.index); - codePoints.push(Array.from(part.segment).length); +interface Grapheme { + codePoints: number; + end: number; + start: number; +} + +function normalizeChunks(text: string): NormalizedText { + const offsets: number[] = []; + const sources: number[] = []; + const parts: string[] = []; + let length = 0; + for (let start = 0; start < text.length;) { + NORMALIZATION_BOUNDARY.lastIndex = start + NORMALIZATION_CHUNK; + const end = NORMALIZATION_BOUNDARY.exec(text)?.index ?? text.length; + const part = normalize(text.slice(start, end)); + offsets.push(length); + sources.push(start); + parts.push(part); + length += part.length; + start = end; } - boundaries.push(value.length); - return { boundaries, codePoints, value: normalize(value) }; + return { offsets, sources, value: parts.join('') }; } -function sourceBoundaryForNormalizedOffset( - text: string, - indexed: NormalizedIndex, - normalizedOffset: number, -): number { - // NFKC may compose across grapheme boundaries (for example compatibility - // Jamo ㄱ + ㅏ -> 가), so independently normalizing each grapheme cannot - // produce a correct offset map. Normalized prefix lengths are monotonic: - // composition may create a plateau but cannot remove prior output. Binary - // search those true whole-prefix lengths instead. +function lastIndexAtMost(count: number, target: number, valueAt: (index: number) => number) { let low = 0; - let high = indexed.boundaries.length - 1; - const lengths = new Map([ - [0, 0], - [high, indexed.value.length], - ]); - const prefixLength = (boundary: number): number => { - const cached = lengths.get(boundary); - if (cached !== undefined) return cached; - const length = normalize(text.slice(0, indexed.boundaries[boundary])).length; - lengths.set(boundary, length); - return length; - }; + let high = count - 1; while (low < high) { const middle = Math.ceil((low + high) / 2); - if (prefixLength(middle) <= normalizedOffset) low = middle; + if (valueAt(middle) <= target) low = middle; else high = middle - 1; } return low; } -function sourceBoundaryForNormalizedEnd( +function graphemeAt(graphemes: Intl.Segments, position: number): Grapheme { + const part = graphemes.containing(position); + if (!part) return { codePoints: 0, end: position, start: position }; + return { + codePoints: Array.from(part.segment).length, + end: part.index + part.segment.length, + start: part.index, + }; +} + +function normalizedLength(text: string, normalized: NormalizedText, position: number): number { + const { offsets, sources } = normalized; + const chunk = lastIndexAtMost(sources.length, position, (index) => sources[index] ?? 0); + const start = sources[chunk] ?? 0; + return (offsets[chunk] ?? 0) + normalize(text.slice(start, position)).length; +} + +function sourceBoundary( text: string, - indexed: NormalizedIndex, - normalizedOffset: number, -): number { - const floor = sourceBoundaryForNormalizedOffset(text, indexed, normalizedOffset); - const floorLength = normalize(text.slice(0, indexed.boundaries[floor])).length; - // An offset inside an expanded source grapheme (İ -> i + dot, fi -> fi) - // needs the following boundary. Exact offsets use the end of their plateau, - // which also captures composition across boundaries (ㄱ + ㅏ -> 가). - return floorLength < normalizedOffset - ? Math.min(floor + 1, indexed.boundaries.length - 1) - : floor; + normalized: NormalizedText, + graphemes: Intl.Segments, + offset: number, +): { length: number; position: number } { + const { offsets, sources } = normalized; + const chunk = lastIndexAtMost(offsets.length, offset, (index) => offsets[index] ?? 0); + const chunkEnd = sources[chunk + 1] ?? text.length; + const boundaries = [graphemeAt(graphemes, sources[chunk] ?? 0).start]; + for (let position = boundaries[0] ?? 0; position < chunkEnd;) { + position = graphemeAt(graphemes, position).end; + boundaries.push(position); + } + const lengths = new Map(); + const lengthAt = (index: number): number => { + const position = boundaries[index] ?? 0; + const length = lengths.get(position) ?? normalizedLength(text, normalized, position); + lengths.set(position, length); + return length; + }; + const index = lastIndexAtMost(boundaries.length, offset, lengthAt); + return { length: lengthAt(index), position: boundaries[index] ?? 0 }; } function countMatches(value: string, term: string): number { @@ -441,15 +467,14 @@ function countMatches(value: string, term: string): number { function excerpt(text: string, query: string, terms: string[]): string { if (!text) return ''; - const indexed = indexNormalizedText(text); - const normalized = indexed.value; - const exactIndex = normalized.indexOf(query); + const normalized = normalizeChunks(text); + const exactIndex = normalized.value.indexOf(query); let matchIndex = exactIndex; let matchLength = query.length; if (matchIndex < 0) { matchIndex = Number.POSITIVE_INFINITY; for (const term of terms) { - const index = normalized.indexOf(term); + const index = normalized.value.indexOf(term); if (index >= 0 && index < matchIndex) { matchIndex = index; matchLength = term.length; @@ -458,45 +483,37 @@ function excerpt(text: string, query: string, terms: string[]): string { } if (!Number.isFinite(matchIndex)) return truncate(text, EXCERPT_LENGTH); - const matchStartGrapheme = sourceBoundaryForNormalizedOffset(text, indexed, matchIndex); - const matchEndGrapheme = sourceBoundaryForNormalizedEnd( - text, - indexed, - Math.min(normalized.length, matchIndex + matchLength), - ); - let matchCodePoints = 0; - for (let index = matchStartGrapheme; index < matchEndGrapheme; index += 1) { - matchCodePoints += indexed.codePoints[index] ?? 0; - } + const graphemes = GRAPHEME_SEGMENTER.segment(text); + const matchStart = sourceBoundary(text, normalized, graphemes, matchIndex).position; + const endOffset = Math.min(normalized.value.length, matchIndex + matchLength); + const floor = sourceBoundary(text, normalized, graphemes, endOffset); + const matchEnd = + floor.length < endOffset ? graphemeAt(graphemes, floor.position).end : floor.position; + const matchCodePoints = Array.from(text.slice(matchStart, matchEnd)).length; - // Anchor the window to the source match rather than subtracting from its - // normalized offset: NFKC can contract one source grapheme (for example a - // three-code-point Hangul Jamo sequence) to one indexed character. Reserve - // room for the complete source match when it fits; overlong matches are - // sensibly shown from their beginning and truncated to the normal window. const contextLimit = Math.min( Math.floor(EXCERPT_LENGTH / 3), Math.max(0, EXCERPT_LENGTH - matchCodePoints), ); - let startGrapheme = matchStartGrapheme; + let start = matchStart; let contextCodePoints = 0; - while (startGrapheme > 0) { - const previousLength = indexed.codePoints[startGrapheme - 1] ?? 0; - if (contextCodePoints + previousLength > contextLimit) break; - contextCodePoints += previousLength; - startGrapheme -= 1; + while (start > 0) { + const previous = graphemeAt(graphemes, start - 1); + if (contextCodePoints + previous.codePoints > contextLimit) break; + contextCodePoints += previous.codePoints; + start = previous.start; } - const start = indexed.boundaries[startGrapheme] ?? 0; - const prefix = start > 0 ? '…' : ''; - const characters: string[] = []; - for (const character of text.slice(start)) { - if (characters.length >= EXCERPT_LENGTH) break; - characters.push(character); + let end = start; + let selectedCodePoints = 0; + while (end < text.length) { + const next = graphemeAt(graphemes, end); + if (end > start && selectedCodePoints + next.codePoints > EXCERPT_LENGTH) break; + selectedCodePoints += next.codePoints; + end = next.end; } - const selected = characters.join(''); - const value = selected.trim(); - const suffix = start + selected.length < text.length ? '…' : ''; - return `${prefix}${value}${suffix}`; + const prefix = start > 0 ? '…' : ''; + const suffix = end < text.length ? '…' : ''; + return `${prefix}${text.slice(start, end).trim()}${suffix}`; } export function searchDoc(page: DocPage, query: string): DocsSearchResult | null {