Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 18 additions & 0 deletions src/__tests__/content.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -142,6 +142,18 @@ describe('parseDoc', () => {
expect(searchDoc(page, 'needle')?.excerpt).toContain('πŸ˜€β€¦');
});

it('keeps a grapheme cluster intact at the end of a search excerpt', () => {
const page = parseDoc('search/grapheme.md', `# X\n\nmatch ${'a'.repeat(171)}πŸ‘©β€πŸ’» tail`);

expect(searchDoc(page, 'match')?.excerpt).toBe(`X match ${'a'.repeat(171)}…`);
});

it('keeps a grapheme cluster intact when truncating a summary', () => {
const page = parseDoc('search/summary.md', `${'a'.repeat(159)}πŸ‘©β€πŸ’» tail`);

expect(page.summary).toBe(`${'a'.repeat(159)}…`);
});

it('maps a match after NFKC-expanded ligatures back to the source excerpt', () => {
const page = parseDoc('search/ligature.md', `# X\n\n${'fi'.repeat(100)} needle tail`);

Expand Down Expand Up @@ -169,6 +181,12 @@ describe('parseDoc', () => {
expect(searchDoc(page, 'needle')?.excerpt).toContain('needle');
});

it('maps a match that follows many expanded and composed characters', () => {
const page = parseDoc('search/long-mixed.md', `# X\n\n${'Δ° fi ㄱㅏ '.repeat(300)}needle tail`);

expect(searchDoc(page, 'needle')?.excerpt).toBe(`…${'ㄱㅏ Δ° fi '.repeat(8)}ㄱㅏ needle tail`);
});

it('truncates a query longer than the excerpt window from the match start', () => {
const query = 'q'.repeat(240);
const page = parseDoc('search/long-query.md', `# X\n\n${query} tail`);
Expand Down
9 changes: 3 additions & 6 deletions src/client.ts
Original file line number Diff line number Diff line change
Expand Up @@ -248,11 +248,9 @@ function cancelUnusedResponseBody(response: Response | undefined): void {
if (!response || response.bodyUsed) return;
const body = response.body;
if (!body) return;
void body.cancel().catch(() => {
// A transport may close or lock the body while the request settles.
});
void body.cancel().catch(() => undefined);
} catch {
// Cleanup is best-effort and must not override the request result.
// Cleanup must not override the request result.
}
}

Expand Down Expand Up @@ -341,8 +339,7 @@ export function createDocsClient(options: DocsClientOptions = {}): DocsClient {
return await consume(response);
} finally {
clearTimeout(timer);
// Cleanup must not delay a stale-cache result if a custom transport's
// cancel implementation never settles.
// Not awaited: a custom transport's cancel may never settle.
cancelUnusedResponseBody(response);
}
}
Expand Down
205 changes: 111 additions & 94 deletions src/content.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,10 @@ import type { DocComponent, DocPage, DocsSearchResult } from './types.js';
const SUMMARY_LENGTH = 160;
const EXCERPT_LENGTH = 180;
const GRAPHEME_SEGMENTER = new Intl.Segmenter(undefined, { granularity: 'grapheme' });
const LIST_ITEM = /^( {0,3}(?:[-+*]|\d{1,9}[.)]))([ \t]+)/;
const NORMALIZATION_CHUNK = 512;
// NFKC never joins these characters to what precedes them, so chunks split before them.
const NORMALIZATION_BOUNDARY = /[ -~\u4e00-\u9fff]/g;

interface ParsedSource {
body: string;
Expand Down Expand Up @@ -106,8 +110,7 @@ function transitionFence(line: string, current: MarkdownFence | undefined): Fenc
candidate = candidate.slice(Math.min(indentation, current.listIndent));
}
} else {
const listPrefix = /^ {0,3}(?:(?:[-+*]|\d{1,9}[.)]))[ \t]+/.exec(candidate)?.[0] ?? '';
listIndent = listPrefix.length;
listIndent = LIST_ITEM.exec(candidate)?.[0].length ?? 0;
candidate = candidate.slice(listIndent);
}
const match = /^ {0,3}(`{3,}|~{3,})(.*)$/.exec(candidate);
Expand Down Expand Up @@ -165,11 +168,12 @@ function proseLines(body: string): (string | undefined)[] {
inCode = width - listIndent >= 4 && (afterBreak || inCode);
afterBreak = false;
if (inCode) return undefined;
const item = /^ {0,3}(?:[-+*]|\d{1,9}[.)])([ \t]+)(.*)$/.exec(candidate);
const item = LIST_ITEM.exec(candidate);
if (!item) return line;
const padding = item[1] ?? '';
const marker = candidate.length - (item[2] ?? '').length - padding.length;
if (padding.includes('\t') || padding.length >= 5 || /^(?: {4}|\t)/.test(item[2] ?? '')) {
const marker = item[1]?.length ?? 0;
const padding = item[2] ?? '';
const rest = candidate.slice(item[0].length);
if (padding.includes('\t') || padding.length >= 5 || /^(?: {4}|\t)/.test(rest)) {
listIndent = marker + 1;
inCode = true;
return undefined;
Expand All @@ -189,10 +193,14 @@ function extractTitle(body: string): string | undefined {
}

function truncate(value: string, length: number): string {
const characters: string[] = [];
for (const character of value) characters.push(character);
if (characters.length <= length) return value;
return `${characters.slice(0, length).join('').trimEnd()}…`;
let end = 0;
let codePoints = 0;
for (const { index, segment } of GRAPHEME_SEGMENTER.segment(value)) {
codePoints += Array.from(segment).length;
if (codePoints > length) return `${value.slice(0, end).trimEnd()}…`;
end = index + segment.length;
}
return value;
}

function extractSummary(body: string): string {
Expand Down Expand Up @@ -355,76 +363,94 @@ export function parseDoc(path: string, content: string): DocPage {
}

function normalize(value: string): string {
// JavaScript lowercasing preserves Greek final sigma (Ο‚), while a
// case-insensitive search for Ξ£ produces Οƒ. Fold the positional variants to
// one form after NFKC/lowercase so substring matching is position agnostic.
// Lowercasing keeps final sigma (Ο‚) where a search for Ξ£ yields Οƒ.
return value
.normalize('NFKC')
.toLowerCase()
.replace(/\u03c2/g, '\u03c3');
}

interface NormalizedIndex {
boundaries: number[];
codePoints: number[];
interface NormalizedText {
offsets: number[];
sources: number[];
value: string;
}

function indexNormalizedText(value: string): NormalizedIndex {
const boundaries: number[] = [];
const codePoints: number[] = [];
for (const part of GRAPHEME_SEGMENTER.segment(value)) {
boundaries.push(part.index);
codePoints.push(Array.from(part.segment).length);
interface Grapheme {
codePoints: number;
end: number;
start: number;
}

function normalizeChunks(text: string): NormalizedText {
const offsets: number[] = [];
const sources: number[] = [];
const parts: string[] = [];
let length = 0;
for (let start = 0; start < text.length;) {
NORMALIZATION_BOUNDARY.lastIndex = start + NORMALIZATION_CHUNK;
const end = NORMALIZATION_BOUNDARY.exec(text)?.index ?? text.length;
const part = normalize(text.slice(start, end));
offsets.push(length);
sources.push(start);
parts.push(part);
length += part.length;
start = end;
}
boundaries.push(value.length);
return { boundaries, codePoints, value: normalize(value) };
return { offsets, sources, value: parts.join('') };
}

function sourceBoundaryForNormalizedOffset(
text: string,
indexed: NormalizedIndex,
normalizedOffset: number,
): number {
// NFKC may compose across grapheme boundaries (for example compatibility
// Jamo γ„± + ㅏ -> κ°€), so independently normalizing each grapheme cannot
// produce a correct offset map. Normalized prefix lengths are monotonic:
// composition may create a plateau but cannot remove prior output. Binary
// search those true whole-prefix lengths instead.
function lastIndexAtMost(count: number, target: number, valueAt: (index: number) => number) {
let low = 0;
let high = indexed.boundaries.length - 1;
const lengths = new Map<number, number>([
[0, 0],
[high, indexed.value.length],
]);
const prefixLength = (boundary: number): number => {
const cached = lengths.get(boundary);
if (cached !== undefined) return cached;
const length = normalize(text.slice(0, indexed.boundaries[boundary])).length;
lengths.set(boundary, length);
return length;
};
let high = count - 1;
while (low < high) {
const middle = Math.ceil((low + high) / 2);
if (prefixLength(middle) <= normalizedOffset) low = middle;
if (valueAt(middle) <= target) low = middle;
else high = middle - 1;
}
return low;
}

function sourceBoundaryForNormalizedEnd(
function graphemeAt(graphemes: Intl.Segments, position: number): Grapheme {
const part = graphemes.containing(position);
if (!part) return { codePoints: 0, end: position, start: position };
return {
codePoints: Array.from(part.segment).length,
end: part.index + part.segment.length,
start: part.index,
};
}

function normalizedLength(text: string, normalized: NormalizedText, position: number): number {
const { offsets, sources } = normalized;
const chunk = lastIndexAtMost(sources.length, position, (index) => sources[index] ?? 0);
const start = sources[chunk] ?? 0;
return (offsets[chunk] ?? 0) + normalize(text.slice(start, position)).length;
}

function sourceBoundary(
text: string,
indexed: NormalizedIndex,
normalizedOffset: number,
): number {
const floor = sourceBoundaryForNormalizedOffset(text, indexed, normalizedOffset);
const floorLength = normalize(text.slice(0, indexed.boundaries[floor])).length;
// An offset inside an expanded source grapheme (İ -> i + dot, fi -> fi)
// needs the following boundary. Exact offsets use the end of their plateau,
// which also captures composition across boundaries (γ„± + ㅏ -> κ°€).
return floorLength < normalizedOffset
? Math.min(floor + 1, indexed.boundaries.length - 1)
: floor;
normalized: NormalizedText,
graphemes: Intl.Segments,
offset: number,
): { length: number; position: number } {
const { offsets, sources } = normalized;
const chunk = lastIndexAtMost(offsets.length, offset, (index) => offsets[index] ?? 0);
const chunkEnd = sources[chunk + 1] ?? text.length;
const boundaries = [graphemeAt(graphemes, sources[chunk] ?? 0).start];
for (let position = boundaries[0] ?? 0; position < chunkEnd;) {
position = graphemeAt(graphemes, position).end;
boundaries.push(position);
}
const lengths = new Map<number, number>();
const lengthAt = (index: number): number => {
const position = boundaries[index] ?? 0;
const length = lengths.get(position) ?? normalizedLength(text, normalized, position);
lengths.set(position, length);
return length;
};
const index = lastIndexAtMost(boundaries.length, offset, lengthAt);
return { length: lengthAt(index), position: boundaries[index] ?? 0 };
}

function countMatches(value: string, term: string): number {
Expand All @@ -441,15 +467,14 @@ function countMatches(value: string, term: string): number {

function excerpt(text: string, query: string, terms: string[]): string {
if (!text) return '';
const indexed = indexNormalizedText(text);
const normalized = indexed.value;
const exactIndex = normalized.indexOf(query);
const normalized = normalizeChunks(text);
const exactIndex = normalized.value.indexOf(query);
let matchIndex = exactIndex;
let matchLength = query.length;
if (matchIndex < 0) {
matchIndex = Number.POSITIVE_INFINITY;
for (const term of terms) {
const index = normalized.indexOf(term);
const index = normalized.value.indexOf(term);
if (index >= 0 && index < matchIndex) {
matchIndex = index;
matchLength = term.length;
Expand All @@ -458,45 +483,37 @@ function excerpt(text: string, query: string, terms: string[]): string {
}
if (!Number.isFinite(matchIndex)) return truncate(text, EXCERPT_LENGTH);

const matchStartGrapheme = sourceBoundaryForNormalizedOffset(text, indexed, matchIndex);
const matchEndGrapheme = sourceBoundaryForNormalizedEnd(
text,
indexed,
Math.min(normalized.length, matchIndex + matchLength),
);
let matchCodePoints = 0;
for (let index = matchStartGrapheme; index < matchEndGrapheme; index += 1) {
matchCodePoints += indexed.codePoints[index] ?? 0;
}
const graphemes = GRAPHEME_SEGMENTER.segment(text);
const matchStart = sourceBoundary(text, normalized, graphemes, matchIndex).position;
const endOffset = Math.min(normalized.value.length, matchIndex + matchLength);
const floor = sourceBoundary(text, normalized, graphemes, endOffset);
const matchEnd =
floor.length < endOffset ? graphemeAt(graphemes, floor.position).end : floor.position;
const matchCodePoints = Array.from(text.slice(matchStart, matchEnd)).length;

// Anchor the window to the source match rather than subtracting from its
// normalized offset: NFKC can contract one source grapheme (for example a
// three-code-point Hangul Jamo sequence) to one indexed character. Reserve
// room for the complete source match when it fits; overlong matches are
// sensibly shown from their beginning and truncated to the normal window.
const contextLimit = Math.min(
Math.floor(EXCERPT_LENGTH / 3),
Math.max(0, EXCERPT_LENGTH - matchCodePoints),
);
let startGrapheme = matchStartGrapheme;
let start = matchStart;
let contextCodePoints = 0;
while (startGrapheme > 0) {
const previousLength = indexed.codePoints[startGrapheme - 1] ?? 0;
if (contextCodePoints + previousLength > contextLimit) break;
contextCodePoints += previousLength;
startGrapheme -= 1;
while (start > 0) {
const previous = graphemeAt(graphemes, start - 1);
if (contextCodePoints + previous.codePoints > contextLimit) break;
contextCodePoints += previous.codePoints;
start = previous.start;
}
const start = indexed.boundaries[startGrapheme] ?? 0;
const prefix = start > 0 ? '…' : '';
const characters: string[] = [];
for (const character of text.slice(start)) {
if (characters.length >= EXCERPT_LENGTH) break;
characters.push(character);
let end = start;
let selectedCodePoints = 0;
while (end < text.length) {
const next = graphemeAt(graphemes, end);
if (end > start && selectedCodePoints + next.codePoints > EXCERPT_LENGTH) break;
selectedCodePoints += next.codePoints;
end = next.end;
}
const selected = characters.join('');
const value = selected.trim();
const suffix = start + selected.length < text.length ? '…' : '';
return `${prefix}${value}${suffix}`;
const prefix = start > 0 ? '…' : '';
const suffix = end < text.length ? '…' : '';
return `${prefix}${text.slice(start, end).trim()}${suffix}`;
}

export function searchDoc(page: DocPage, query: string): DocsSearchResult | null {
Expand Down
Loading