diff --git a/CHANGELOG.md b/CHANGELOG.md index c5a77b5..ddb78d8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +## [0.4.0] - 2026-09-06 + +### Added + +- Upgraded email extraction to a multi-value envelope: one email now yields every finding (verification code, each classified link), an expiry hint (e.g. "10 分钟内有效"), and SPF/DKIM/DMARC sender verdicts from Authentication-Results. Chinese verification keywords (验证码/校验码/动态码…), hyphen-joined codes, code-before-keyword order, and RFC 8058 List-Unsubscribe headers are recognized; bare digit runs without a nearby code keyword are rejected, killing price/date false positives. The AI fallback returns a validated JSON array and merges all findings. The API keeps the legacy `extraction` field (best finding) and adds `extractions`, `expiresHint`, and `authSummary`; stored rows from before the upgrade render unchanged. The message detail panel shows secondary findings, the expiry hint, and sender-auth badges. + ## [0.3.0] - 2026-09-05 ### Added diff --git a/apps/docs/package.json b/apps/docs/package.json index 087a742..bef5d66 100644 --- a/apps/docs/package.json +++ b/apps/docs/package.json @@ -1,6 +1,6 @@ { "name": "@wemail/docs", - "version": "0.3.0", + "version": "0.4.0", "private": true, "type": "module", "scripts": { diff --git a/apps/web/package.json b/apps/web/package.json index fe590f0..139dc3b 100644 --- a/apps/web/package.json +++ b/apps/web/package.json @@ -1,6 +1,6 @@ { "name": "@wemail/web", - "version": "0.3.0", + "version": "0.4.0", "private": true, "type": "module", "scripts": { diff --git a/apps/web/src/features/inbox/MessageDetailPanel.tsx b/apps/web/src/features/inbox/MessageDetailPanel.tsx index 8660a1c..785c174 100644 --- a/apps/web/src/features/inbox/MessageDetailPanel.tsx +++ b/apps/web/src/features/inbox/MessageDetailPanel.tsx @@ -132,6 +132,27 @@ function getRemoteImageBlocks(bodyText: string) { }); } +const authVerdictLabels: Record = { + pass: "通过", + fail: "失败", + softfail: "软失败", + none: "无", + unknown: "未知" +}; + +type DetailViewModel = NonNullable>; + +function buildAuthSummaryBadges(authSummary: NonNullable) { + return [ + { key: "SPF", verdict: authSummary.spf }, + { key: "DKIM", verdict: authSummary.dkim }, + { key: "DMARC", verdict: authSummary.dmarc } + ].map((entry) => ({ + ...entry, + label: authVerdictLabels[entry.verdict] ?? entry.verdict + })); +} + function getExtractionInsight(viewModel: NonNullable>) { if (viewModel.extraction.type === "auth_code" && viewModel.extraction.value.trim()) { return { @@ -248,6 +269,9 @@ export function MessageDetailPanel({ errorMessage = null, isLoading = false, onR const hasExtractionValue = viewModel.extraction.value.trim().length > 0; const copyLabel = viewModel.extraction.type === "auth_code" ? "复制验证码" : "复制提取值"; const extractionInsight = getExtractionInsight(viewModel); + const secondaryFindings = (viewModel.extractions ?? []).filter( + (item) => item.type !== viewModel.extraction.type || item.value !== viewModel.extraction.value + ); const ExtractionInsightIcon = extractionInsight.Icon as LucideIcon; const retentionLabel = formatRetentionLabel(viewModel.expiresAt); const linkRisk = extractionInsight.kind === "link" ? analyzeExtractionLink(viewModel.extraction.value) : null; @@ -307,6 +331,16 @@ export function MessageDetailPanel({ errorMessage = null, isLoading = false, onR {retentionLabel} + {viewModel.authSummary ? ( +
+ 发件人验证 + {buildAuthSummaryBadges(viewModel.authSummary).map((badge) => ( + + {badge.key} {badge.label} + + ))} +
+ ) : null}

@@ -326,6 +360,25 @@ export function MessageDetailPanel({ errorMessage = null, isLoading = false, onR

+ {viewModel.expiresHint ? ( +

+

+ ) : null} + {secondaryFindings.length > 0 ? ( +
+

同时识别到

+
    + {secondaryFindings.map((item) => ( +
  • + {item.label} + {item.value} +
  • + ))} +
+
+ ) : null}
{linkRisk ? (
diff --git a/apps/web/src/shared/styles/index.css b/apps/web/src/shared/styles/index.css index ffe5e63..8673748 100644 --- a/apps/web/src/shared/styles/index.css +++ b/apps/web/src/shared/styles/index.css @@ -3646,6 +3646,104 @@ button:disabled { grid-column: 1 / -1; } +/* Extraction envelope additions: expiry hint, secondary findings, sender auth. */ +.extraction-card-expiry { + display: flex; + align-items: center; + gap: 6px; + margin: 0; + color: var(--text-muted); + font-size: 0.86rem; +} + +.extraction-card-secondary { + display: grid; + gap: 6px; + border-top: 1px solid var(--border); + padding-top: 10px; +} + +.extraction-card-secondary > p { + margin: 0; + color: var(--text-soft); + font-size: 0.78rem; + letter-spacing: 0.04em; +} + +.extraction-card-secondary ul { + display: grid; + gap: 4px; + margin: 0; + padding: 0; + list-style: none; +} + +.extraction-card-secondary li { + display: flex; + align-items: baseline; + gap: 8px; + min-width: 0; + font-size: 0.86rem; +} + +.extraction-card-secondary li span { + flex-shrink: 0; + color: var(--text-muted); +} + +.extraction-card-secondary li strong { + min-width: 0; + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; + direction: rtl; + text-align: left; +} + +.auth-summary-row { + display: flex; + align-items: center; + flex-wrap: wrap; + gap: 8px; + font-size: 0.84rem; +} + +.auth-summary-title { + color: var(--text-soft); +} + +.auth-verdict { + display: inline-flex; + align-items: center; + padding: 2px 8px; + border-radius: var(--radius-pill, 999px); + border: 1px solid var(--border); + color: var(--text-muted); + white-space: nowrap; +} + +.auth-verdict-pass { + border-color: color-mix(in srgb, #20c997 42%, var(--border)); + color: #158f6b; + background: var(--success-soft, transparent); +} + +.auth-verdict-fail, +.auth-verdict-softfail { + border-color: color-mix(in srgb, #e03131 42%, var(--border)); + color: #c92a2a; + background: var(--warning-soft, transparent); +} + +:root[data-theme="dark"] .auth-verdict-pass { + color: #4ddbb1; +} + +:root[data-theme="dark"] .auth-verdict-fail, +:root[data-theme="dark"] .auth-verdict-softfail { + color: #ff8787; +} + .inbox-detail-panel .extraction-card { grid-template-columns: minmax(0, 1fr) minmax(132px, 0.28fr); align-items: center; diff --git a/apps/web/src/test/integration/inbox-page.test.tsx b/apps/web/src/test/integration/inbox-page.test.tsx index 4269916..1e315e7 100644 --- a/apps/web/src/test/integration/inbox-page.test.tsx +++ b/apps/web/src/test/integration/inbox-page.test.tsx @@ -92,6 +92,12 @@ const mockInboxMessages: MessageSummary[] = [ previewText: "Open the login link", bodyText: "Open the login link", extraction: { method: "regex", type: "auth_link", value: "https://contoso.test/magic", label: "登录链接" }, + extractions: [ + { method: "regex", type: "auth_link", value: "https://contoso.test/magic", label: "登录链接", source: "body" }, + { method: "regex", type: "subscription_link", value: "https://contoso.test/unsubscribe", label: "Unsubscribe (header)", source: "header" } + ], + expiresHint: "10 分钟内有效", + authSummary: { spf: "pass", dkim: "pass", dmarc: "pass", raw: null }, oversizeStatus: null, attachmentCount: 1, attachments: [ @@ -723,6 +729,11 @@ describe("mail list integration", () => { const extractedLink = within(extractionCard).getByText("https://contoso.test/magic"); expect(within(extractionCard).getByText(/^识别到链接$/i)).toBeInTheDocument(); + expect(within(extractionCard).getByText(/^同时识别到$/i)).toBeInTheDocument(); + expect(within(extractionCard).getByText(/contoso.test\/unsubscribe/)).toBeInTheDocument(); + expect(within(extractionCard).getByText(/验证码 10 分钟内有效/)).toBeInTheDocument(); + expect(within(screen.getByRole("region", { name: /^阅读与提取详情$/i })).getByText("SPF 通过")).toBeInTheDocument(); + expect(within(screen.getByRole("region", { name: /^阅读与提取详情$/i })).getByText("DMARC 通过")).toBeInTheDocument(); expect(extractedLink).toHaveClass("extraction-card-value-link"); expect(extractedLink).toHaveAttribute("title", "https://contoso.test/magic"); expect(within(extractionCard).queryByText(/^登录链接$/i)).not.toBeInTheDocument(); diff --git a/apps/worker/package.json b/apps/worker/package.json index 86b6685..a38da31 100644 --- a/apps/worker/package.json +++ b/apps/worker/package.json @@ -1,6 +1,6 @@ { "name": "@wemail/worker", - "version": "0.3.0", + "version": "0.4.0", "private": true, "type": "module", "scripts": { diff --git a/apps/worker/src/app/runtime.ts b/apps/worker/src/app/runtime.ts index 2ac7340..094a040 100644 --- a/apps/worker/src/app/runtime.ts +++ b/apps/worker/src/app/runtime.ts @@ -1,7 +1,7 @@ import { parseAccountPolicyRecord } from "@wemail/shared"; import type { AppBindings, AppStore, MailboxRecord, PersistedMessageRecord } from "../core/bindings"; -import { buildExtraction, createPreview, maybeRunAiFallback, parseRawEmail } from "../shared/mail"; +import { buildMessageExtraction, createPreview, maybeRunAiFallback, parseRawEmail } from "../shared/mail"; import { recordAudit } from "./services/audit-service"; import { defaultFeatureToggles } from "./services/config-service"; import { getRuntimeSettings } from "./services/runtime-settings-service"; @@ -117,7 +117,7 @@ async function saveInboundMessage( text: string; attachments: Array<{ filename: string; contentType: string; data: Uint8Array; size: number }>; }; - extraction: ReturnType; + extraction: ReturnType; } ) { const duplicate = await findRecentDuplicateMessage(store, input); @@ -169,13 +169,20 @@ async function processInboundForMailbox( toAddress: string, parsed: { messageId?: string | null; + listUnsubscribe?: string | null; + authenticationResults?: string | null; fromAddress: string; subject: string; text: string; attachments: Array<{ filename: string; contentType: string; data: Uint8Array; size: number }>; } ) { - let extraction = buildExtraction(parsed.subject, parsed.text); + let extraction = buildMessageExtraction({ + subject: parsed.subject, + bodyText: parsed.text, + listUnsubscribe: parsed.listUnsubscribe, + authenticationResults: parsed.authenticationResults + }); const featureToggles = await getFeatureToggles(store, env); const settings = await getRuntimeSettings(store, env); const aiUsageToday = await store.audit.countByActorSince( @@ -184,9 +191,9 @@ async function processInboundForMailbox( `${new Date().toISOString().slice(0, 10)}T00:00:00.000Z` ); - if (featureToggles.aiEnabled && extraction.type === "none" && aiUsageToday < settings.ai.fallbackLimit) { - extraction = (await maybeRunAiFallback(env, extraction, parsed.text)) as typeof extraction; - if (extraction.method === "ai") { + if (featureToggles.aiEnabled && extraction.primary.type === "none" && aiUsageToday < settings.ai.fallbackLimit) { + extraction = await maybeRunAiFallback(env, extraction, parsed.text); + if (extraction.primary.method === "ai") { await recordAudit(store, "user", mailbox.userId, "ai-fallback", { mailboxId: mailbox.id }); } } @@ -223,21 +230,26 @@ async function processInboundForMailbox( subject: parsed.subject }); - if (extraction.type !== "none") { + if (extraction.primary.type !== "none") { await sendTelegramNotification( { store, env, featureToggles }, { userId: mailbox.userId, eventId: "message.extraction.detected", - text: `Extracted result for ${mailbox.address}\n${extraction.label}: ${extraction.value}\nSubject: ${parsed.subject}`, - metadata: { mailboxId: mailbox.id, messageId: message.id, extractionType: extraction.type } + text: `Extracted result for ${mailbox.address}\n${extraction.primary.label}: ${extraction.primary.value}\nSubject: ${parsed.subject}`, + metadata: { mailboxId: mailbox.id, messageId: message.id, extractionType: extraction.primary.type } } ); + // extraction (primary) keeps the legacy single-result contract; the + // envelope fields carry every finding plus expiry and auth verdicts. await sendWebhookEventToUser(store, mailbox.userId, "message.extracted", { mailboxAddress: mailbox.address, mailboxId: mailbox.id, messageId: message.id, - extraction + extraction: extraction.primary, + extractions: extraction.items, + expiresHint: extraction.expiresHint, + authSummary: extraction.authSummary }); } @@ -264,7 +276,12 @@ export async function processInboundEmail( mailboxId: buildUnmatchedMailboxId(toAddress), toAddress, parsed, - extraction: buildExtraction(parsed.subject, parsed.text) + extraction: buildMessageExtraction({ + subject: parsed.subject, + bodyText: parsed.text, + listUnsubscribe: parsed.listUnsubscribe, + authenticationResults: parsed.authenticationResults + }) }); return unmatchedMessage; } diff --git a/apps/worker/src/infrastructure/persistence/d1/mail.ts b/apps/worker/src/infrastructure/persistence/d1/mail.ts index 6ecd6c1..3c600a0 100644 --- a/apps/worker/src/infrastructure/persistence/d1/mail.ts +++ b/apps/worker/src/infrastructure/persistence/d1/mail.ts @@ -326,11 +326,17 @@ export function createMailAggregate(db: D1Database): MailAggregate { bindings.push(searchLike, searchLike, searchLike, searchLike, searchLike, searchLike); } + // extraction_json holds the envelope ($.primary.type) or a legacy + // single result ($.type); COALESCE reads both shapes. if (query.filter === "code") { - whereConditions.push("json_extract(extraction_json, '$.type') = 'auth_code'"); + whereConditions.push( + "COALESCE(json_extract(extraction_json, '$.primary.type'), json_extract(extraction_json, '$.type')) = 'auth_code'" + ); } if (query.filter === "link") { - whereConditions.push("json_extract(extraction_json, '$.type') NOT IN ('auth_code', 'none')"); + whereConditions.push( + "COALESCE(json_extract(extraction_json, '$.primary.type'), json_extract(extraction_json, '$.type')) NOT IN ('auth_code', 'none')" + ); } if (query.filter === "attachment") { whereConditions.push("attachment_count > 0"); diff --git a/apps/worker/src/infrastructure/persistence/in-memory.ts b/apps/worker/src/infrastructure/persistence/in-memory.ts index 46d4190..6f8331d 100644 --- a/apps/worker/src/infrastructure/persistence/in-memory.ts +++ b/apps/worker/src/infrastructure/persistence/in-memory.ts @@ -9,7 +9,8 @@ import { type MailDomainSummary, type MessageFilter, type MessageListSummary, - type OutboundListStatus + type OutboundListStatus, + parseMessageExtraction as parseSharedMessageExtraction } from "@wemail/shared"; import type { @@ -88,7 +89,8 @@ function getInactiveDays(value: number | undefined) { } function parseMessageExtraction(record: PersistedMessageRecord) { - return JSON.parse(record.extractionJson) as { type?: string; value?: string; label?: string }; + // Shared normalizer handles both the envelope and legacy single results. + return parseSharedMessageExtraction(record.extractionJson); } const apiKeyScopeIds = new Set(API_KEY_SCOPE_DEFINITIONS.map((scope) => scope.id)); @@ -99,7 +101,7 @@ function normalizeApiKeyScopes(scopes: unknown[] | undefined): ApiKeyScope[] { } function matchesMessageFilter(record: PersistedMessageRecord, filter: MessageFilter = "all") { - const extraction = parseMessageExtraction(record); + const extraction = parseMessageExtraction(record).primary; if (filter === "code") return extraction.type === "auth_code"; if (filter === "link") return extraction.type !== "auth_code" && extraction.type !== "none"; if (filter === "attachment") return record.attachmentCount > 0; @@ -127,8 +129,8 @@ function matchesMessageSearch(record: PersistedMessageRecord, searchValue?: stri record.subject, record.previewText, record.bodyText, - extraction.value ?? "", - extraction.label ?? "" + // Search covers every finding, not just the primary one. + ...extraction.items.flatMap((item) => [item.value, item.label]) ] .join(" ") .toLowerCase() @@ -158,7 +160,8 @@ function matchesMessageAdvancedFilters(record: PersistedMessageRecord, query: { } if (query.extractionType) { - const extraction = parseMessageExtraction(record); + // Primary-only, matching the D1 json_extract filter semantics. + const extraction = parseMessageExtraction(record).primary; if (extraction.type !== query.extractionType) return false; } @@ -169,7 +172,7 @@ function summarizeMessageRecords(records: PersistedMessageRecord[]): MessageList return { messageCount: records.length, extractionCount: records.filter((record) => { - const extraction = parseMessageExtraction(record); + const extraction = parseMessageExtraction(record).primary; return extraction.type !== "none" && Boolean(extraction.value?.trim()); }).length, attachmentCount: records.reduce((sum, record) => sum + record.attachmentCount, 0) diff --git a/apps/worker/src/shared/mail.ts b/apps/worker/src/shared/mail.ts index 48390d2..e1a3c67 100644 --- a/apps/worker/src/shared/mail.ts +++ b/apps/worker/src/shared/mail.ts @@ -1,6 +1,12 @@ import PostalMime from "postal-mime"; -import { extractImportantInfo, type ExtractionResult } from "@wemail/shared"; +import { + buildMessageExtraction as buildSharedMessageExtraction, + mergeAiItems, + parseMessageExtraction, + type ExtractionResult, + type MessageExtraction +} from "@wemail/shared"; import type { AppBindings, AttachmentRecord, PersistedMessageRecord, ResendClient, TelegramApiClient } from "../core/bindings"; const htmlEntityMap: Record = { @@ -57,22 +63,39 @@ function htmlToReadableText(html: string) { return safeSrc ? `\nRemote image blocked: ${safeSrc}\n` : " "; } ); - const withLinks = withRemoteImageBlocks.replace( + // Quote chains (reply history) carry no signal for a disposable inbox and + // push the actual content below the fold — drop them before flattening. + const withoutQuotes = withRemoteImageBlocks + .replace(/]*>[\s\S]*?<\/blockquote>/gi, " ") + .replace(/]*class=(["']?)[^"']*gmail_quote[^"']*\1[^>]*>[\s\S]*?<\/div>/gi, " "); + const withLinks = withoutQuotes.replace( /]*\bhref=(["']?)([^"'\s>]+)\1[^>]*>([\s\S]*?)<\/a>/gi, (_match, _quote: string, href: string, label: string) => `${label} ${href}` ); - const text = withLinks + // Tables read as "col | col" rows so order confirmations and receipts keep + // their column relationships in the flattened text. + const withTables = withLinks .replace(/<\s*(script|style|head)\b[\s\S]*?<\/\s*\1\s*>/gi, " ") .replace(//g, " ") - .replace(/<\s*(br|\/p|\/div|\/h[1-6]|\/li|\/tr)\b[^>]*>/gi, "\n") - .replace(/<\s*(p|div|h[1-6]|li|tr|td|th)\b[^>]*>/gi, "\n") - .replace(/<[^>]+>/g, " "); + .replace(/<\s*t[dh]\b[^>]*>/gi, " ") + .replace(/<\s*\/t[dh]\b[^>]*>\s*(?!<\s*t[dh]\b|<\s*\/tr\b)/gi, " | ") + .replace(/<\s*\/tr\b[^>]*>/gi, "\n") + .replace(/<\s*(br|\/p|\/div|\/h[1-6]|\/li)\b[^>]*>/gi, "\n") + .replace(/<\s*(p|div|h[1-6]|li|tr)\b[^>]*>/gi, "\n"); + const text = withTables.replace(/<[^>]+>/g, " "); return normalizeTextLines(decodeHtmlEntities(text)); } function pickReadableBodyText(parsed: { text?: string; html?: string }) { const text = parsed.text?.trim(); - if (text) return parsed.text ?? ""; + if (text) { + // Drop reply-quoted lines so extraction reads the new content only. + const withoutQuotes = parsed.text + ?.split("\n") + .filter((line) => !/^\s*>/.test(line)) + .join("\n"); + return withoutQuotes ?? ""; + } return parsed.html ? htmlToReadableText(parsed.html) : ""; } @@ -110,27 +133,53 @@ export async function parseRawEmail(raw: ReadableStream) { }; }); + const headers = parsed.headers ?? []; + const findHeader = (name: string) => headers.find((header) => header.key === name)?.value ?? null; + return { fromAddress: parsed.from?.address ?? "unknown@sender.invalid", subject: parsed.subject ?? "(no subject)", // The RFC 5322 Message-ID survives redelivery unchanged, making it the // idempotency key for inbound processing. messageId: parsed.messageId?.trim() || null, + // RFC 8058 list-unsubscribe and the receiving mail server's auth verdicts + // feed the extraction envelope. + listUnsubscribe: findHeader("list-unsubscribe"), + authenticationResults: findHeader("authentication-results"), text: pickReadableBodyText(parsed), attachments: normalizedAttachments }; } -export function buildExtraction(subject: string, bodyText: string) { - return extractImportantInfo({ subject, text: bodyText }); +export function buildMessageExtraction(input: { + subject: string; + bodyText: string; + listUnsubscribe?: string | null; + authenticationResults?: string | null; +}): MessageExtraction { + return buildSharedMessageExtraction({ + subject: input.subject, + text: input.bodyText, + listUnsubscribe: input.listUnsubscribe ?? null, + authenticationResults: input.authenticationResults ?? null + }); +} + +const aiAllowedTypes = new Set(["auth_code", "auth_link", "service_link", "subscription_link", "other_link"]); + +// Llama 3.1 responds more reliably to JSON when fenced; strip fences before +// parsing so one formatting quirk doesn't discard the whole fallback. +function stripJsonFences(value: string) { + const fenced = value.match(/```(?:json)?\s*([\s\S]*?)```/i); + return (fenced?.[1] ?? value).trim(); } export async function maybeRunAiFallback( env: { AI?: AppBindings["AI"] }, - current: ExtractionResult, + current: MessageExtraction, content: string ) { - if (current.type !== "none" || !env.AI) return current; + if (current.primary.type !== "none" || !env.AI) return current; try { const result = await env.AI.run("@cf/meta/llama-3.1-8b-instruct" as any, { @@ -138,7 +187,7 @@ export async function maybeRunAiFallback( { role: "system", content: - "Extract exactly one useful auth code or auth link from the email. Return JSON with keys type, value, label." + "Extract every useful auth code and actionable link from the email. Respond with a JSON array; each element has keys type (one of auth_code, auth_link, service_link, subscription_link, other_link), value, label. No other text." }, { role: "user", content } ] @@ -148,15 +197,28 @@ export async function maybeRunAiFallback( typeof result === "object" && result && "response" in result ? (result.response as string) : null; if (!response) return current; - const parsed = JSON.parse(response) as { type?: string; value?: string; label?: string }; - if (!parsed.type || !parsed.value) return current; + const parsed = JSON.parse(stripJsonFences(response)) as unknown; + if (!Array.isArray(parsed)) return current; - return { - method: "ai", - type: parsed.type as ExtractionResult["type"], - value: parsed.value, - label: parsed.label ?? "AI result" - }; + const aiItems = parsed + .filter( + (item): item is { type: string; value: string; label?: string } => + typeof item === "object" && + item !== null && + typeof (item as { type?: unknown }).type === "string" && + typeof (item as { value?: unknown }).value === "string" && + aiAllowedTypes.has((item as { type: string }).type) + ) + .map((item) => ({ + method: "regex" as const, + type: item.type as ExtractionResult["type"], + value: item.value, + label: item.label ?? "AI result", + source: "body" as const + })); + if (aiItems.length === 0) return current; + + return mergeAiItems(current, aiItems); } catch { return current; } @@ -288,6 +350,9 @@ export function buildResendClient(apiKey: string | undefined): ResendClient | nu } export function toMessageJson(message: PersistedMessageRecord, attachments: AttachmentRecord[]) { + // extractionJson holds either the new envelope or a pre-0.4.0 single + // result; the shared normalizer returns the envelope shape for both. + const envelope = parseMessageExtraction(message.extractionJson); return { id: message.id, mailboxId: message.mailboxId, @@ -296,7 +361,10 @@ export function toMessageJson(message: PersistedMessageRecord, attachments: Atta subject: message.subject, previewText: message.previewText, bodyText: message.bodyText, - extraction: JSON.parse(message.extractionJson) as ExtractionResult, + extraction: envelope.primary, + extractions: envelope.items, + expiresHint: envelope.expiresHint, + authSummary: envelope.authSummary, oversizeStatus: message.oversizeStatus, attachmentCount: message.attachmentCount, attachments, diff --git a/apps/worker/tests/integration/message.integration.test.ts b/apps/worker/tests/integration/message.integration.test.ts index 8bf3cec..75af228 100644 --- a/apps/worker/tests/integration/message.integration.test.ts +++ b/apps/worker/tests/integration/message.integration.test.ts @@ -1,5 +1,6 @@ import { afterEach, describe, expect, it, vi } from "vitest"; +import type { AppBindings } from "../../src/core/bindings"; import { processInboundEmail } from "../../src/app/create-app"; import { createWorkerTestHarness } from "../helpers/test-env"; @@ -112,7 +113,7 @@ describe("worker message integration", () => { to: mailbox.address, raw: new Response(rawEmail).body! }); - const extraction = JSON.parse(message.extractionJson) as { type: string; value: string }; + const extraction = JSON.parse(message.extractionJson) as { primary: { type: string; value: string } }; expect(message.bodyText).toContain("NVIDIA email verification"); expect(message.bodyText).toContain("Your verification code will expire shortly."); @@ -120,13 +121,109 @@ describe("worker message integration", () => { expect(message.bodyText).not.toContain(" { + const { env, store, mailbox } = await registerMemberAndCreateMailbox(); + const rawEmail = [ + "From: 某某服务 ", + `To: ${mailbox.address}`, + "Subject: 【某某服务】登录验证", + "Message-ID: ", + "List-Unsubscribe: ", + "Authentication-Results: mx.example.com; spf=pass; dkim=pass; dmarc=pass", + "Content-Type: text/plain; charset=utf-8", + "", + "您的验证码为 583-914,10 分钟内有效。如非本人操作请忽略。" + ].join("\r\n"); + + const message = await processInboundEmail(env, store, { to: mailbox.address, raw: new Response(rawEmail).body! }); + const envelope = JSON.parse(message.extractionJson) as { + primary: { type: string; value: string }; + items: Array<{ type: string; source?: string }>; + expiresHint: string | null; + authSummary: { spf: string; dkim: string; dmarc: string } | null; + }; + + expect(envelope.primary).toMatchObject({ type: "auth_code", value: "583914" }); + expect(envelope.expiresHint).toBe("10 分钟内有效"); + expect(envelope.items.some((item) => item.type === "subscription_link" && item.source === "header")).toBe(true); + expect(envelope.authSummary).toMatchObject({ spf: "pass", dkim: "pass", dmarc: "pass" }); + }); + + it("exposes the extraction envelope on the message detail API", async () => { + const { app, env, store, cookie, mailbox } = await registerMemberAndCreateMailbox(); + const rawEmail = [ + "From: Shop ", + `To: ${mailbox.address}`, + "Subject: Order confirmation", + "Message-ID: ", + "Content-Type: text/plain; charset=utf-8", + "", + "Your verification code is 776655. Track your order at https://shop.example/track/99" + ].join("\r\n"); + const message = await processInboundEmail(env, store, { to: mailbox.address, raw: new Response(rawEmail).body! }); + + const response = await app.request( + `/api/mail/messages/${message.id}`, + { headers: { cookie } }, + env + ); + const payload = (await response.json()) as { + message: { + extraction: { type: string }; + extractions: Array<{ type: string }>; + expiresHint: string | null; + }; + }; + + expect(response.status).toBe(200); + expect(payload.message.extraction.type).toBe("auth_code"); + expect(payload.message.extractions.map((item) => item.type)).toEqual( + expect.arrayContaining(["auth_code", "other_link"]) + ); + expect(payload.message.expiresHint).toBeNull(); + }); + + it("merges multi-value AI fallback findings when regex finds nothing", async () => { + const { env, store, mailbox } = await registerMemberAndCreateMailbox(); + const rawEmail = [ + "From: Blurb ", + `To: ${mailbox.address}`, + "Subject: A note", + "Message-ID: ", + "Content-Type: text/plain; charset=utf-8", + "", + "Nothing machine-readable here at all, just prose." + ].join("\r\n"); + + const fakeAi = { + run: async () => ({ + response: JSON.stringify([ + { type: "auth_code", value: "445566", label: "AI code" }, + { type: "auth_link", value: "https://blurb.example/continue", label: "AI link" } + ]) + }) + }; + + const message = await processInboundEmail({ ...env, AI: fakeAi as unknown as AppBindings["AI"] }, store, { + to: mailbox.address, + raw: new Response(rawEmail).body! + }); + const envelope = JSON.parse(message.extractionJson) as { + primary: { type: string; method: string; value: string }; + items: Array<{ type: string; method: string }>; + }; + + expect(envelope.primary).toMatchObject({ type: "auth_code", method: "ai", value: "445566" }); + expect(envelope.items.filter((item) => item.method === "ai")).toHaveLength(2); + }); + it("does not send telegram notifications when the global telegram feature is disabled", async () => { const fetchMock = vi.fn(); vi.stubGlobal("fetch", fetchMock); diff --git a/docs/api-guide.md b/docs/api-guide.md index d464265..a303bef 100644 --- a/docs/api-guide.md +++ b/docs/api-guide.md @@ -34,6 +34,7 @@ WeMail 后端 API 已按管理后台左侧菜单分组,旧 `/auth`、`/admin` - 发件记录支持服务端分页、搜索和 `all/sent/failed` 状态筛选;详情接口会返回正文、实际发给 provider 的请求 payload、provider 响应和 message id,便于审计。 - 设置和治理数据按菜单拆分到 `account_settings`、`mail_settings`、`webhook_*`、`announcements`、`system_settings`。 - Webhook 端点支持 `channel` 字段:`webhook`(默认,投递完整 JSON 事件包并带签名头)、`slack` / `discord` / `feishu` / `wecom`(按对应平台的传入式 Webhook 消息格式投递)。通知规则可以按渠道定向(`target` 取渠道名)。 +- 邮件详情的 `extraction` 是最佳单项发现(兼容字段);`extractions` 返回全部发现(验证码、各分类链接),`expiresHint` 为验证码有效期提示,`authSummary` 为 SPF/DKIM/DMARC 判定。提取支持中文验证码关键词与 List-Unsubscribe 头。 ## 常用流程 diff --git a/docs/openapi.yaml b/docs/openapi.yaml index b07d3f2..2dd7448 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -2,7 +2,7 @@ "openapi": "3.1.0", "info": { "title": "WeMail 菜单化后端 API", - "version": "0.3.0", + "version": "0.4.0", "description": "WeMail 后端 API 已按管理后台左侧菜单分组。旧 /auth、/admin、/api/mailboxes、/api/messages、/api/outbound、/api/keys、/api/telegram 路径不再保留。" }, "servers": [ diff --git a/package.json b/package.json index 51363d1..966afa5 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "wemail", - "version": "0.3.0", + "version": "0.4.0", "private": true, "type": "module", "packageManager": "pnpm@10.18.2", diff --git a/packages/shared/package.json b/packages/shared/package.json index e512dfd..0f0ab3d 100644 --- a/packages/shared/package.json +++ b/packages/shared/package.json @@ -1,6 +1,6 @@ { "name": "@wemail/shared", - "version": "0.3.0", + "version": "0.4.0", "private": true, "type": "module", "exports": { diff --git a/packages/shared/src/extraction.ts b/packages/shared/src/extraction.ts index f3fd297..acf6425 100644 --- a/packages/shared/src/extraction.ts +++ b/packages/shared/src/extraction.ts @@ -1,15 +1,63 @@ -import type { ExtractionResult } from "./types"; +import type { AuthenticationSummary, ExtractionResult, ExtractionType, MessageExtraction } from "./types"; -const codePatterns = [ - /\b(?:code|otp|verification code|security code|passcode)[:\s-]*([A-Z0-9]{4,8})\b/i, - /\b([0-9]{4,8})\b/ +// --------------------------------------------------------------------------- +// Inbound extraction engine. +// +// One email can carry several findings: a verification code, an auth link, a +// subscription link from the List-Unsubscribe header, and so on. The engine +// collects them all into a MessageExtraction envelope; "primary" keeps the +// single best result so chips, webhooks, and legacy rows stay compatible. +// --------------------------------------------------------------------------- + +const linkPattern = /https?:\/\/[^\s)"'<>]+/gi; + +// Keyword-anchored captures (English and Chinese). A keyword directly before +// or after the candidate is the strongest signal a digit run is a code. +// Codes may arrive hyphen-joined ("836-592"); the captured run is validated +// and normalized afterwards, so the group allows one hyphen segment. +const codeCapture = "([A-Z0-9]{2,10}(?:-[A-Z0-9]{2,10})?|[A-Z0-9]{4,10})"; + +const codeAfterKeywordPatterns = [ + new RegExp("\\b(?:verification|security|one[-\\s]?time|auth(?:entication)?|login|access|pass)[- ]?(?:code|otp|password|pin)\\b[^A-Za-z0-9]{0,16}" + codeCapture + "\\b", "i"), + new RegExp("\\b(?:code|otp|passcode|pin)\\b[^A-Za-z0-9]{0,12}" + codeCapture + "\\b", "i"), + new RegExp("(?:验证码|校验码|动态码|动态密码|一次性密码|识别码|登入码|登录码|口令)[^A-Za-z0-9]{0,10}" + codeCapture) ]; -const linkPattern = /https?:\/\/[^\s)]+/gi; +const codeBeforeKeywordPatterns = [ + new RegExp("\\b" + codeCapture + "\\b[^A-Za-z0-9]{0,16}(?:is|为|是)?\\s*(?:your\\s+)?(?:verification|security|one[-\\s]?time|login|access)?\\s*(?:code|otp|passcode|验证码|校验码|动态码|动态密码)\\b", "i") +]; -function pickLink(links: string[], matcher: RegExp) { - return links.find((link) => matcher.test(link)); -} +// Window around a bare digit run that must contain a code keyword for the +// run to count. This is what rejects prices, dates, and order numbers. +const codeContextWindow = 40; +const codeContextKeywords = + /\b(?:code|otp|passcode|pin|password)\b|verification|verify|security|one[-\s]?time|登录|登入|验证|校验|动态|口令/i; + +const bareDigitPattern = /\b[0-9]{4,10}\b/g; + +const expiryPatterns: Array<{ pattern: RegExp; units: Record }> = [ + { + pattern: /(?:有效期|有效时间|过期时间)[^0-9]{0,8}(\d+)\s*(分钟|分鐘|小时|小時|天)/, + units: { "分钟": "分钟", "分鐘": "分钟", "小时": "小时", "小時": "小时", "天": "天" } + }, + { + // Reversed Chinese form: the window precedes 有效 ("10 分钟内有效"). + pattern: /(\d+)\s*(分钟|分鐘|小时|小時|天)\s*(?:之?内|內)?\s*有效/, + units: { "分钟": "分钟", "分鐘": "分钟", "小时": "小时", "小時": "小时", "天": "天" } + }, + { + pattern: /\b(?:valid for|expires? in|expire[sd]? within|expiry)[:\s]*(\d+)\s*(minutes?|mins?|hours?|hrs?|days?)\b/i, + units: { minute: "分钟", minutes: "分钟", min: "分钟", mins: "分钟", hour: "小时", hours: "小时", hr: "小时", hrs: "小时", day: "天", days: "天" } + } +]; + +const linkClassifiers: Array<{ type: ExtractionType; label: string; matcher: RegExp }> = [ + { type: "auth_link", label: "Verification link", matcher: /(verify|activate|confirm|signin|login|reset|auth)/i }, + { type: "service_link", label: "Service link", matcher: /(github|gitlab|deploy|issue|pull|commit|docs|dashboard)/i }, + { type: "subscription_link", label: "Subscription link", matcher: /(unsubscribe|opt-?out|preferences|subscription)/i } +]; + +const primaryTypeOrder: ExtractionType[] = ["auth_code", "auth_link", "service_link", "subscription_link", "other_link"]; function normalizeCodeCandidate(value: string) { return value.replace(/[\s-]/g, ""); @@ -17,74 +65,191 @@ function normalizeCodeCandidate(value: string) { function isLikelyCodeCandidate(value: string) { const normalized = normalizeCodeCandidate(value); - return /^[A-Z0-9]{4,8}$/i.test(normalized) && /\d/.test(normalized); + return /^[A-Z0-9]{4,10}$/i.test(normalized) && /\d/.test(normalized); } -export function extractImportantInfo(input: { - subject?: string; - text?: string; - html?: string; -}): ExtractionResult { - const subject = input.subject ?? ""; - const text = [subject, input.text ?? "", input.html ?? ""].filter(Boolean).join("\n"); +function collectCodeCandidates(text: string): string[] { + const found: string[] = []; - for (const pattern of codePatterns) { - const match = text.match(pattern); - if (match?.[1] && isLikelyCodeCandidate(match[1])) { - return { - method: "regex", - type: "auth_code", - value: normalizeCodeCandidate(match[1]), - label: "Verification code" - }; + for (const pattern of codeAfterKeywordPatterns) { + for (const match of text.matchAll(new RegExp(pattern.source, pattern.flags.includes("g") ? pattern.flags : pattern.flags + "g"))) { + if (match?.[1] && isLikelyCodeCandidate(match[1])) found.push(match[1]); + } + } + for (const pattern of codeBeforeKeywordPatterns) { + for (const match of text.matchAll(new RegExp(pattern.source, pattern.flags.includes("g") ? pattern.flags : pattern.flags + "g"))) { + if (match?.[1] && isLikelyCodeCandidate(match[1])) found.push(match[1]); } } + // Bare digit runs only count inside a code-keyword context window. + for (const match of text.matchAll(bareDigitPattern)) { + const candidate = match[0]; + if (!isLikelyCodeCandidate(candidate)) continue; + const start = Math.max(0, match.index - codeContextWindow); + const end = Math.min(text.length, match.index + candidate.length + codeContextWindow); + if (codeContextKeywords.test(text.slice(start, end))) found.push(candidate); + } + + const unique: string[] = []; + for (const value of found) { + const normalized = normalizeCodeCandidate(value); + if (!unique.some((existing) => normalizeCodeCandidate(existing) === normalized)) unique.push(value); + } + return unique; +} + +function collectLinkItems(text: string): ExtractionResult[] { const links = Array.from(new Set(text.match(linkPattern) ?? [])); + const items: ExtractionResult[] = []; - const authLink = pickLink(links, /(verify|activate|confirm|signin|login|reset)/i); - if (authLink) { - return { - method: "regex", - type: "auth_link", - value: authLink, - label: "Verification link" - }; + for (const classifier of linkClassifiers) { + const link = links.find((value) => classifier.matcher.test(value)); + if (link) items.push({ method: "regex", type: classifier.type, value: link, label: classifier.label, source: "body" }); } - const serviceLink = pickLink(links, /(github|gitlab|deploy|issue|pull|commit)/i); - if (serviceLink) { - return { - method: "regex", - type: "service_link", - value: serviceLink, - label: "Service link" - }; + const unmatched = links.find((value) => !linkClassifiers.some((classifier) => classifier.matcher.test(value))); + if (unmatched) { + items.push({ method: "regex", type: "other_link", value: unmatched, label: "Useful link", source: "body" }); } + return items; +} - const subscriptionLink = pickLink(links, /(unsubscribe|opt-?out|preferences)/i); - if (subscriptionLink) { - return { +function parseListUnsubscribe(header: string | null | undefined): ExtractionResult | null { + if (!header) return null; + // The header may hold several comma-separated values, each possibly in + // angle brackets; only http(s) URLs are actionable in the product UI. + const urls = Array.from(header.matchAll(/,\s]+)>?/gi)).map((match) => match[1]); + const url = Array.from(new Set(urls))[0]; + if (!url) return null; + return { method: "regex", type: "subscription_link", value: url, label: "Unsubscribe (header)", source: "header" }; +} + +function findExpiresHint(text: string): string | null { + for (const { pattern, units } of expiryPatterns) { + const match = text.match(pattern); + if (match?.[1] && match[2]) { + const unit = units[match[2].toLowerCase()] ?? units[match[2]]; + if (unit) return `${match[1]} ${unit}内有效`; + } + } + return null; +} + +const authVerdictPatterns: Array<{ key: keyof AuthenticationSummary; pattern: RegExp }> = [ + { key: "spf", pattern: /\bspf=(pass|fail|softfail|none)\b/i }, + { key: "dkim", pattern: /\bdkim=(pass|fail|none)\b/i }, + { key: "dmarc", pattern: /\bdmarc=(pass|fail|quarantine|reject|none)\b/i } +]; + +export function parseAuthenticationResults(raw: string | null | undefined): AuthenticationSummary | null { + if (!raw) return null; + const summary: AuthenticationSummary = { spf: "unknown", dkim: "unknown", dmarc: "unknown", raw: raw.slice(0, 500) }; + + for (const { key, pattern } of authVerdictPatterns) { + const match = raw.match(pattern); + if (match?.[1]) { + const verdict = match[1].toLowerCase(); + summary[key] = verdict === "quarantine" || verdict === "reject" ? "fail" : (verdict as AuthenticationSummary["spf"]); + } + } + return summary; +} + +export function pickPrimaryExtraction(items: ExtractionResult[]): ExtractionResult { + for (const type of primaryTypeOrder) { + // Header-derived findings outrank body guesses within the same type. + const match = items.find((item) => item.type === type && item.source === "header") ?? items.find((item) => item.type === type); + if (match) return match; + } + return { method: "none", type: "none", value: "", label: "", source: null }; +} + +function dedupeItems(items: ExtractionResult[]): ExtractionResult[] { + const seen = new Set(); + const unique: ExtractionResult[] = []; + for (const item of items) { + if (item.type === "none" || !item.value) continue; + const key = `${item.type}:${item.value}`; + if (seen.has(key)) continue; + seen.add(key); + unique.push(item); + } + return unique; +} + +export function buildMessageExtraction(input: { + subject?: string; + text?: string; + html?: string; + listUnsubscribe?: string | null; + authenticationResults?: string | null; +}): MessageExtraction { + const text = [input.subject ?? "", input.text ?? "", input.html ?? ""].filter(Boolean).join("\n"); + + const items = dedupeItems([ + ...collectCodeCandidates(text).map((value) => ({ method: "regex", - type: "subscription_link", - value: subscriptionLink, - label: "Subscription link" - }; + type: "auth_code", + value: normalizeCodeCandidate(value), + label: "Verification code", + source: "body" + })), + ...collectLinkItems(text), + ...(parseListUnsubscribe(input.listUnsubscribe) ? [parseListUnsubscribe(input.listUnsubscribe)!] : []) + ]); + + return { + primary: pickPrimaryExtraction(items), + items, + expiresHint: findExpiresHint(text), + authSummary: parseAuthenticationResults(input.authenticationResults) + }; +} + +// Legacy single-result API: returns the primary finding. +export function extractImportantInfo(input: { + subject?: string; + text?: string; + html?: string; +}): ExtractionResult { + return buildMessageExtraction(input).primary; +} + +// Normalizes whatever is stored in mail_messages.extraction_json: the new +// envelope, or a legacy single ExtractionResult written before 0.4.0. +export function parseMessageExtraction(json: string | null | undefined): MessageExtraction { + const none: MessageExtraction = { primary: { method: "none", type: "none", value: "", label: "", source: null }, items: [], expiresHint: null, authSummary: null }; + if (!json) return none; + + let parsed: unknown; + try { + parsed = JSON.parse(json); + } catch { + return none; } + if (typeof parsed !== "object" || parsed === null) return none; - if (links.length > 0) { + if ("items" in parsed && Array.isArray((parsed as MessageExtraction).items)) { + const envelope = parsed as MessageExtraction; + const items = dedupeItems(envelope.items ?? []); return { - method: "regex", - type: "other_link", - value: links[0], - label: "Useful link" + primary: envelope.primary ?? pickPrimaryExtraction(items), + items, + expiresHint: envelope.expiresHint ?? null, + authSummary: envelope.authSummary ?? null }; } - return { - method: "none", - type: "none", - value: "", - label: "" - }; + const legacy = parsed as ExtractionResult; + if (!legacy || typeof legacy !== "object" || !legacy.type) return none; + const legacyItems = legacy.type !== "none" && legacy.value ? [legacy] : []; + return { primary: legacy, items: legacyItems, expiresHint: null, authSummary: null }; +} + +// Merges AI-fallback findings into an envelope whose regex pass found +// nothing, then re-picks the primary. +export function mergeAiItems(current: MessageExtraction, aiItems: ExtractionResult[]): MessageExtraction { + const items = dedupeItems([...current.items, ...aiItems.map((item) => ({ ...item, method: "ai" as const }))]); + return { ...current, items, primary: pickPrimaryExtraction(items) }; } diff --git a/packages/shared/src/types.ts b/packages/shared/src/types.ts index 07200e6..960dbb5 100644 --- a/packages/shared/src/types.ts +++ b/packages/shared/src/types.ts @@ -17,6 +17,31 @@ export type ExtractionResult = { type: ExtractionType; value: string; label: string; + // Where the finding came from: "header" for RFC-defined sources such as + // List-Unsubscribe, "body" for content matches. + source?: "body" | "header" | null; +}; + +// Verdicts distilled from the Authentication-Results header so readers can +// judge sender trustworthiness without parsing the raw chain. +export type AuthVerdict = "pass" | "fail" | "softfail" | "none" | "unknown"; + +export type AuthenticationSummary = { + spf: AuthVerdict; + dkim: AuthVerdict; + dmarc: AuthVerdict; + raw: string | null; +}; + +// One email can carry several findings (a code plus several link kinds). +// The envelope is what gets stored in mail_messages.extraction_json and +// exposed on the API; "primary" keeps the single best result so every +// legacy consumer (chips, webhook payloads) keeps working unchanged. +export type MessageExtraction = { + primary: ExtractionResult; + items: ExtractionResult[]; + expiresHint: string | null; + authSummary: AuthenticationSummary | null; }; export type FeatureToggles = { @@ -286,7 +311,11 @@ export type MessageSummary = { subject: string; previewText: string; bodyText: string; + // Single best finding (legacy-compatible); see extractions for the rest. extraction: ExtractionResult; + extractions?: ExtractionResult[]; + expiresHint?: string | null; + authSummary?: AuthenticationSummary | null; oversizeStatus: string | null; attachmentCount: number; attachments: MessageAttachmentSummary[]; diff --git a/packages/shared/src/version.ts b/packages/shared/src/version.ts index 30393fa..1951e33 100644 --- a/packages/shared/src/version.ts +++ b/packages/shared/src/version.ts @@ -1,2 +1,2 @@ -export const WEMAIL_VERSION = "0.3.0"; +export const WEMAIL_VERSION = "0.4.0"; export const WEMAIL_VERSION_LABEL = `v${WEMAIL_VERSION}`; diff --git a/packages/shared/tests/extraction.test.ts b/packages/shared/tests/extraction.test.ts index 9bdeaa8..964ffbd 100644 --- a/packages/shared/tests/extraction.test.ts +++ b/packages/shared/tests/extraction.test.ts @@ -1,8 +1,16 @@ import { describe, expect, it } from "vitest"; -import { extractImportantInfo } from "../src/index"; +import { + buildMessageExtraction, + extractImportantInfo, + mergeAiItems, + parseAuthenticationResults, + parseMessageExtraction +} from "../src/index"; -describe("extractImportantInfo", () => { +const noneResult = { method: "none", type: "none", value: "", label: "", source: null }; + +describe("extractImportantInfo (legacy single-result API)", () => { it("prefers a verification code over links", () => { const result = extractImportantInfo({ subject: "Your verification code", @@ -13,7 +21,8 @@ describe("extractImportantInfo", () => { method: "regex", type: "auth_code", value: "123456", - label: "Verification code" + label: "Verification code", + source: "body" }); }); @@ -33,39 +42,154 @@ describe("extractImportantInfo", () => { text: "Your verification code will expire shortly." }); - expect(result).toEqual({ - method: "none", - type: "none", - value: "", - label: "" - }); + expect(result).toEqual(noneResult); }); it("keeps extracting alphanumeric codes that contain digits", () => { const result = extractImportantInfo({ - subject: "Your verification code", - text: "Use verification code A1B2C3 to continue." + subject: "Your security code", + text: "Security code: A7B2C9" }); expect(result).toEqual({ method: "regex", type: "auth_code", - value: "A1B2C3", - label: "Verification code" + value: "A7B2C9", + label: "Verification code", + source: "body" }); }); it("returns none when nothing useful is found", () => { - const result = extractImportantInfo({ - subject: "Newsletter", - text: "Welcome to our monthly update." + const result = extractImportantInfo({ subject: "Hello", text: "Just a friendly note." }); + expect(result).toEqual(noneResult); + }); +}); + +describe("buildMessageExtraction (multi-value envelope)", () => { + it("collects a code and classified links from the same email", () => { + const envelope = buildMessageExtraction({ + subject: "Your login code", + text: "验证码 482914,10 分钟内有效。管理面板 https://app.example.com/verify?token=x 退订 https://example.com/unsubscribe" }); - expect(result).toEqual({ - method: "none", - type: "none", - value: "", - label: "" + expect(envelope.primary.type).toBe("auth_code"); + expect(envelope.primary.value).toBe("482914"); + expect(envelope.expiresHint).toBe("10 分钟内有效"); + expect(envelope.items.map((item) => item.type)).toEqual( + expect.arrayContaining(["auth_code", "auth_link", "subscription_link"]) + ); + }); + + it("extracts Chinese verification codes with full-width punctuation", () => { + const envelope = buildMessageExtraction({ + subject: "【某某服务】登录验证", + text: "您的动态码为:836-592,请勿泄露。" }); + + expect(envelope.primary).toEqual({ + method: "regex", + type: "auth_code", + value: "836592", + label: "Verification code", + source: "body" + }); + }); + + it("rejects bare digits that sit outside a code-keyword context window", () => { + const envelope = buildMessageExtraction({ + subject: "Your monthly invoice", + text: "Total due: 49.00 USD. Order 3021546 shipped on 2026-09-05. Track via the app." + }); + + expect(envelope.primary.type).toBe("none"); + expect(envelope.items.filter((item) => item.type === "auth_code")).toHaveLength(0); + }); + + it("keeps a bare digit run that sits inside a code-keyword window", () => { + const envelope = buildMessageExtraction({ + subject: "Sign-in attempt", + text: "Someone tried signing in. Your one-time code is below.\n\n582014\n\nDidn't request it?" + }); + + expect(envelope.primary.type).toBe("auth_code"); + expect(envelope.primary.value).toBe("582014"); + }); + + it("captures a code written before the keyword", () => { + const envelope = buildMessageExtraction({ + subject: "Acme login", + text: "483920 为您的验证码,请在 5 分钟内输入。" + }); + + expect(envelope.primary.type).toBe("auth_code"); + expect(envelope.primary.value).toBe("483920"); + }); + + it("prefers the List-Unsubscribe header over body-guessed subscription links", () => { + const envelope = buildMessageExtraction({ + subject: "Weekly digest", + text: "Read more at https://news.example.com/digest", + listUnsubscribe: ", " + }); + + const subscription = envelope.items.find((item) => item.type === "subscription_link"); + expect(subscription?.source).toBe("header"); + expect(subscription?.value).toBe("https://news.example.com/u/abc"); + expect(envelope.primary.type).toBe("subscription_link"); + expect(envelope.primary.source).toBe("header"); + }); + + it("parses SPF, DKIM, and DMARC verdicts from Authentication-Results", () => { + const envelope = buildMessageExtraction({ + subject: "Bank notice", + text: "Your statement is ready.", + authenticationResults: + "spf=pass (sender IP is 192.0.2.1) smtp.mailfrom=bank.example; dkim=pass (2048-bit key) header.d=bank.example; dmarc=pass action=none header.from=bank.example" + }); + + expect(envelope.authSummary).toMatchObject({ spf: "pass", dkim: "pass", dmarc: "pass" }); + }); + + it("marks failed verdicts and maps DMARC quarantine to fail", () => { + const summary = parseAuthenticationResults("spf=fail; dkim=none; dmarc=quarantine"); + expect(summary).toMatchObject({ spf: "fail", dkim: "none", dmarc: "fail" }); + }); +}); + +describe("parseMessageExtraction (stored JSON compatibility)", () => { + it("reads legacy single-result rows", () => { + const envelope = parseMessageExtraction( + JSON.stringify({ method: "regex", type: "auth_code", value: "998877", label: "Verification code" }) + ); + + expect(envelope.primary.value).toBe("998877"); + expect(envelope.items).toHaveLength(1); + expect(envelope.expiresHint).toBeNull(); + }); + + it("reads the new envelope and tolerates malformed JSON", () => { + const stored = JSON.stringify( + buildMessageExtraction({ subject: "验证码 112233", text: "https://example.com/verify" }) + ); + const envelope = parseMessageExtraction(stored); + expect(envelope.primary.value).toBe("112233"); + expect(envelope.items.length).toBeGreaterThanOrEqual(2); + + expect(parseMessageExtraction("not json").primary.type).toBe("none"); + expect(parseMessageExtraction(null).primary.type).toBe("none"); + }); +}); + +describe("mergeAiItems", () => { + it("merges AI findings and re-picks the primary", () => { + const base = buildMessageExtraction({ subject: "Notice", text: "No links here." }); + expect(base.primary.type).toBe("none"); + + const merged = mergeAiItems(base, [ + { method: "regex", type: "auth_link", value: "https://ai.example.com/verify", label: "AI link" } + ]); + + expect(merged.primary).toMatchObject({ method: "ai", type: "auth_link", value: "https://ai.example.com/verify" }); }); });