From b0a21d3880f17790a522f3813235eedeed30b1bb Mon Sep 17 00:00:00 2001 From: fylorn <249551762+fylorn@users.noreply.github.com> Date: Fri, 9 Oct 2026 18:06:36 +0800 Subject: [PATCH] Mark prompt-cache breakpoints for Claude when the client marked none Anthropic caches only what a request marks with `cache_control`, and Bedrock only what it marks with `cachePoint`. Clients in OpenAI or Gemini formats (Codex, Chat clients) cannot mark anything, because those providers cache a repeated prefix on their own. So a converted request to Claude was never cached, and every turn paid full price for the whole conversation. When the IR has no breakpoints, the Anthropic encoder (any Anthropic-format upstream) and the Bedrock encoder (Claude model ids only) now mark, with the default 5-minute TTL: - the end of the tools - the end of the system prompt - the end of the last user turn and of the one before it, at most four in total The earlier user mark is the previous request's last breakpoint, so each turn reads back what the previous one wrote even when a step adds more blocks than Anthropic's ~20-block lookback. Thinking blocks are never marked. Requests that carry their own breakpoints (Claude Code) keep exactly those, and same-format passthrough is untouched. Cost accounting needed no change: usage parsing already reads `cache_creation_input_tokens` / `cache_read_input_tokens` and Bedrock's `cacheWriteInputTokens` / `cacheReadInputTokens`, the sniffer reads the upstream's own bytes for converted requests, and the recorder charges cache writes and reads at the price table's cache rates. New tests cover both. Documented in docs/config.md and docs/config.zh-CN.md. Co-Authored-By: Claude Opus 5.5 --- crates/tw-dialect/src/anthropic/request.rs | 45 +++ crates/tw-dialect/src/bedrock/request.rs | 30 ++ crates/tw-dialect/tests/auto_cache.rs | 288 ++++++++++++++++++++ crates/tw-dialect/tests/codex_compaction.rs | 29 +- crates/tw-gateway/tests/conversion.rs | 69 +++++ crates/tw-gateway/tests/harness.rs | 7 +- crates/tw-store/src/recorder.rs | 25 ++ docs/config.md | 13 + docs/config.zh-CN.md | 2 + 9 files changed, 505 insertions(+), 3 deletions(-) create mode 100644 crates/tw-dialect/tests/auto_cache.rs diff --git a/crates/tw-dialect/src/anthropic/request.rs b/crates/tw-dialect/src/anthropic/request.rs index 0651656b..b25bd6b7 100644 --- a/crates/tw-dialect/src/anthropic/request.rs +++ b/crates/tw-dialect/src/anthropic/request.rs @@ -209,6 +209,47 @@ fn cache_ttl(b: &Value) -> Option { }) } +/// 客户端没标断点时自动标的:**工具的末尾、系统提示的末尾、最后两条用户消息的末尾**, +/// 最多四个(Anthropic 的上限),5 分钟的默认 TTL。 +/// +/// 别的格式的客户端(Codex、Chat、Gemini 的客户端)没有这个写法:OpenAI 和 Gemini 自己 +/// 缓存开头相同的部分,转给 Anthropic 时不标就一点都不缓存,每一轮整段对话全价重算。 +/// +/// - 工具和系统提示:一段对话里几乎不变,各标一个,系统提示变了工具那段照样命中 +/// - 最后一条用户消息:这一轮写进缓存 +/// - 倒数第二条用户消息:**正是上一轮请求的最后一个断点**,这一轮从它读回上一轮写的缓存。 +/// Anthropic 只往回找约 20 块,一轮里工具调用多的时候光靠最后一个断点找不回去 +/// +/// 断点不算内容:上一轮标在更早位置的那个这一轮挪走了,缓存照样命中(Anthropic 文档里多轮 +/// 对话就是这么标的)。太短(不到模型的最小缓存长度)的断点上游直接忽略,不报错,所以不估 +/// 长度。思考块不能标,标在它前面的那块上 +fn auto_cache(out: &mut Map) { + fn mark(blocks: Option<&mut Value>) { + let Some(Value::Array(blocks)) = blocks else { + return; + }; + if let Some(b) = blocks + .iter_mut() + .rev() + .find(|b| !matches!(str_of(b, "type"), Some("thinking" | "redacted_thinking"))) + { + b["cache_control"] = cache_control(CacheTtl::Short); + } + } + mark(out.get_mut("tools")); + mark(out.get_mut("system")); + if let Some(Value::Array(messages)) = out.get_mut("messages") { + for m in messages + .iter_mut() + .rev() + .filter(|m| str_of(m, "role") == Some("user")) + .take(2) + { + mark(m.get_mut("content")); + } + } +} + /// 中间表示的断点 → `cache_control` fn cache_control(ttl: CacheTtl) -> Value { match ttl { @@ -396,6 +437,10 @@ pub fn encode_request(r: &Request, t: &Target, dropped: &mut Dropped) -> Value { .collect(); out.insert("tools".into(), Value::Array(tools)); } + // 客户端自己一个断点都没标:替它标(Claude Code 这类标了的原样不动) + if r.cache.is_empty() { + auto_cache(&mut out); + } let serial = r.parallel_tool_calls == Some(false); let choice = match (&r.tool_choice, serial) { diff --git a/crates/tw-dialect/src/bedrock/request.rs b/crates/tw-dialect/src/bedrock/request.rs index 3193ca6b..5cab8b27 100644 --- a/crates/tw-dialect/src/bedrock/request.rs +++ b/crates/tw-dialect/src/bedrock/request.rs @@ -80,6 +80,30 @@ fn caches(model: &str) -> bool { m.contains("anthropic.claude") || m.contains("amazon.nova") || m.starts_with("arn:") } +/// 客户端没标断点时自动标的,和转给 Anthropic 时同样的四处:工具的末尾、系统提示的末尾、 +/// 最后两条用户消息的末尾,各跟一个 5 分钟的 `cachePoint`(Converse 也最多四个) +fn auto_cache(out: &mut Map) { + fn mark(blocks: Option<&mut Value>) { + if let Some(Value::Array(blocks)) = blocks + && !blocks.is_empty() + { + blocks.push(cache_point(CacheTtl::Short)); + } + } + mark(out.get_mut("toolConfig").and_then(|c| c.get_mut("tools"))); + mark(out.get_mut("system")); + if let Some(Value::Array(messages)) = out.get_mut("messages") { + for m in messages + .iter_mut() + .rev() + .filter(|m| m.get("role").and_then(Value::as_str) == Some("user")) + .take(2) + { + mark(m.get_mut("content")); + } + } +} + /// 一个缓存断点块。`ttl` 缺省是 5 分钟 fn cache_point(ttl: CacheTtl) -> Value { match ttl { @@ -207,6 +231,12 @@ pub fn encode_request(r: &Request, t: &Target, dropped: &mut Dropped) -> Value { } else if r.tool_choice.is_some() { dropped.path("tool_choice"); } + // 客户端自己一个断点都没标,模型又是 Claude:替它标(见 `anthropic::request` 的 + // `auto_cache`)。Nova 和看不出是谁的推理配置 ARN 不自动标:不认的模型收到 + // `cachePoint` 会拒掉整个请求 + if r.cache.is_empty() && is_claude(&r.model) { + auto_cache(&mut out); + } // Converse 自己没有的旋钮 if r.seed.is_some() { diff --git a/crates/tw-dialect/tests/auto_cache.rs b/crates/tw-dialect/tests/auto_cache.rs new file mode 100644 index 00000000..ca0a7f58 --- /dev/null +++ b/crates/tw-dialect/tests/auto_cache.rs @@ -0,0 +1,288 @@ +//! 转给 Claude(Anthropic 格式、Bedrock 上的 Claude)时,客户端没标的提示缓存断点替它标上。 +//! +//! Codex 这类说 OpenAI 格式的客户端没有断点的写法:OpenAI 自己缓存开头相同的部分。转给 +//! Anthropic 时不标,就一点都不缓存,每一轮整段对话全价重算。标在工具末尾、系统提示末尾、 +//! 最后两条用户消息末尾:倒数第二条正是上一轮的最后一个断点,这一轮从那里读回上一轮写的 +//! 缓存。客户端自己标了的(Claude Code)原样不动。 +//! +//! 请求照 Codex 的 serde 类型写(`codex-rs/protocol/src/models.rs` 的 `ResponseItem`)。 + +// 整个请求写成一个 `json!`,嵌套得深 +#![recursion_limit = "512"] + +use serde_json::{Value, json}; +use tw_dialect::convert::{Prepared, decode, prepare}; +use tw_dialect::ir::*; + +fn tools() -> Value { + json!({"id": "at_1", "type": "additional_tools", "role": "developer", "tools": [ + {"type": "namespace", "name": "functions", "description": "", "tools": [ + {"type": "function", "name": "exec_command", "description": "Runs a command.", "strict": false, + "parameters": {"type": "object", "properties": {"cmd": {"type": "string"}}, "required": ["cmd"]}}, + {"type": "custom", "name": "apply_patch", "description": "Edit files.", + "format": {"type": "grammar", "syntax": "lark", "definition": "start: begin_patch hunk+ end_patch"}} + ]}, + {"type": "namespace", "name": "mcp__codex_apps__calendar", "description": "Plan events.", "tools": [ + {"type": "function", "name": "_create_event", "description": "Create an event.", "strict": false, + "parameters": {"type": "object", "properties": {"title": {"type": "string"}}}} + ]} + ]}) +} + +/// Codex 一轮里接连的三个请求:每一步多一次工具调用和它的结果,最后用户接着说 +fn turns() -> [Vec; 3] { + let first = vec![ + tools(), + json!({"id": "msg_b", "type": "message", "role": "developer", + "content": [{"type": "input_text", "text": "You are Codex, a coding agent."}]}), + json!({"type": "message", "role": "user", + "content": [{"type": "input_text", "text": "\n /repo\n"}]}), + json!({"type": "message", "role": "user", "content": [{"type": "input_text", "text": "Fix the failing parser test."}]}), + json!({"type": "message", "role": "assistant", "content": [{"type": "output_text", "text": "Running the tests."}], "phase": "commentary"}), + json!({"type": "function_call", "name": "exec_command", "namespace": "functions", "arguments": "{\"cmd\":\"cargo test\"}", "call_id": "call_1"}), + json!({"type": "function_call_output", "call_id": "call_1", "output": "parser::tests::nested FAILED"}), + ]; + let mut second = first.clone(); + second.extend([ + json!({"type": "custom_tool_call", "call_id": "call_2", "name": "apply_patch", "namespace": "functions", + "input": "*** Begin Patch\n*** Update File: src/parser.rs\n*** End Patch\n"}), + json!({"type": "custom_tool_call_output", "call_id": "call_2", "output": "Success."}), + ]); + let mut third = second.clone(); + third.extend([ + json!({"type": "message", "role": "assistant", "content": [{"type": "output_text", "text": "Fixed; the tests pass."}], "phase": "final_answer"}), + json!({"type": "message", "role": "user", "content": [{"type": "input_text", "text": "Now update the docs."}]}), + ]); + [first, second, third] +} + +fn request(input: Vec) -> Value { + json!({ + "model": "gpt-5.4", "stream": true, "input": input, + "tool_choice": "auto", "parallel_tool_calls": false, + "reasoning": {"effort": "medium", "summary": "auto"}, + "store": false, "include": ["reasoning.encrypted_content"] + }) +} + +fn target(d: Dialect) -> Target { + Target { + dialect: d, + official: false, + default_max_tokens: 8192, + } +} + +/// 网关的做法:解码一次,改成上游的模型名,再按上游的格式编码 +fn codex(input: Vec, upstream: Dialect, model: &str) -> Value { + let mut d = decode(Dialect::Responses, &request(input), "/v1/responses", None).unwrap(); + d.request.model = model.to_string(); + let p: Prepared = d.encode(&target(upstream)); + serde_json::from_slice(&p.body).unwrap() +} + +fn claude(input: Vec) -> Value { + codex(input, Dialect::Anthropic, "claude-opus-4-7") +} + +/// 标了 `cache_control` 的块在哪:("tools" | "system" | "messages", 第几条, 第几块) +fn marks(v: &Value) -> Vec<(&'static str, usize, usize)> { + let mut out = Vec::new(); + for (key, list) in [("tools", &v["tools"]), ("system", &v["system"])] { + for (i, b) in list.as_array().into_iter().flatten().enumerate() { + if b.get("cache_control").is_some() { + assert_eq!(b["cache_control"], json!({"type": "ephemeral"}), "{b}"); + out.push((key, i, 0)); + } + } + } + for (i, m) in v["messages"].as_array().unwrap().iter().enumerate() { + for (j, b) in m["content"].as_array().into_iter().flatten().enumerate() { + if b.get("cache_control").is_some() { + assert_eq!(b["cache_control"], json!({"type": "ephemeral"}), "{b}"); + out.push(("messages", i, j)); + } + } + } + out +} + +fn without_marks(mut v: Value) -> Value { + fn strip(v: &mut Value) { + match v { + Value::Object(o) => { + o.remove("cache_control"); + o.values_mut().for_each(strip); + } + Value::Array(a) => a.iter_mut().for_each(strip), + _ => {} + } + } + strip(&mut v); + v +} + +/// 一个请求从开头到 (第几条消息, 第几块) 为止的那段,按 Anthropic 算缓存的顺序:工具、 +/// 系统提示、消息 +fn prefix_to(v: &Value, message: usize, block: usize) -> String { + let mut messages: Vec = v["messages"].as_array().unwrap()[..=message].to_vec(); + let last = messages.last_mut().unwrap(); + let content = last["content"].as_array().unwrap()[..=block].to_vec(); + last["content"] = Value::Array(content); + json!([v["tools"], v["system"], messages]).to_string() +} + +#[test] +fn a_codex_request_for_claude_gets_the_four_usual_breakpoints() { + let [_, second, _] = turns(); + let v = claude(second); + let roles: Vec<&str> = v["messages"] + .as_array() + .unwrap() + .iter() + .map(|m| m["role"].as_str().unwrap()) + .collect(); + assert_eq!(roles, ["user", "assistant", "user", "assistant", "user"]); + assert_eq!( + marks(&v), + [ + // 工具末尾 + ("tools", 2, 0), + // 系统提示末尾:基础指令之后还有 MCP namespace 的说明 + ("system", 1, 0), + // 上一轮的最后一条:exec_command 的结果 + ("messages", 2, 0), + // 这一轮的最后一条:apply_patch 的结果 + ("messages", 4, 0), + ] + ); + assert_eq!(v.to_string().matches("cache_control").count(), 4); +} + +#[test] +fn each_turn_reads_back_what_the_turn_before_wrote() { + let [first, second, third] = turns(); + let bodies = [claude(first), claude(second), claude(third)]; + for pair in bodies.windows(2) { + let (earlier, later) = (&pair[0], &pair[1]); + // 前一个请求的最后一个断点 + let &(_, i, j) = marks(earlier).last().unwrap(); + // 后一个请求在同一块上也有一个:从这里读回前一个写进去的缓存 + assert!( + marks(later).contains(&("messages", i, j)), + "{:?} vs {:?}", + marks(earlier), + marks(later) + ); + // 到那一块为止一个字节都不差(断点本身不算内容:前一个请求标在更早位置的那个, + // 这一轮挪走了 —— Anthropic 文档里多轮对话就是这么标的) + assert_eq!( + prefix_to(&without_marks(earlier.clone()), i, j), + prefix_to(&without_marks(later.clone()), i, j) + ); + // 工具和系统提示连断点都一样 + assert_eq!(earlier["tools"], later["tools"]); + assert_eq!(earlier["system"], later["system"]); + } +} + +#[test] +fn a_client_that_marks_its_own_breakpoints_keeps_exactly_those() { + // Claude Code 的请求:自己标了系统提示和最后一条消息 + let claude_code = json!({ + "model": "claude-opus-4-7", "max_tokens": 1024, + "system": [{"type": "text", "text": "You are Claude Code.", "cache_control": {"type": "ephemeral"}}], + "tools": [{"name": "Read", "input_schema": {"type": "object"}}], + "messages": [ + {"role": "user", "content": "hi"}, + {"role": "assistant", "content": "hello"}, + {"role": "user", "content": [{"type": "text", "text": "read a.rs", "cache_control": {"type": "ephemeral", "ttl": "1h"}}]} + ] + }) + .to_string(); + // 转给 Anthropic 格式的另一种写法(经过中间表示) + let p = prepare( + Dialect::Anthropic, + claude_code.as_bytes(), + "/v1/messages", + None, + &target(Dialect::Anthropic), + ) + .unwrap(); + let v: Value = serde_json::from_slice(&p.body).unwrap(); + assert_eq!(marks_any(&v), 2, "{v}"); + assert!(v["tools"][0].get("cache_control").is_none()); + assert_eq!(v["messages"][2]["content"][0]["cache_control"]["ttl"], "1h"); + // 转给 Bedrock 上的 Claude + let mut d = decode( + Dialect::Anthropic, + &serde_json::from_str(&claude_code).unwrap(), + "/v1/messages", + None, + ) + .unwrap(); + d.request.model = "us.anthropic.claude-sonnet-4-5-20250929-v1:0".into(); + let v: Value = serde_json::from_slice(&d.encode(&target(Dialect::Bedrock)).body).unwrap(); + assert_eq!(v.to_string().matches("cachePoint").count(), 2, "{v}"); +} + +fn marks_any(v: &Value) -> usize { + v.to_string().matches("cache_control").count() +} + +#[test] +fn a_short_conversation_gets_only_the_breakpoints_it_has_room_for() { + // 没有工具、没有系统提示、只有一条用户消息:只标那一条 + let chat = json!({"model": "claude-opus-4-7", "messages": [{"role": "user", "content": "hi"}]}) + .to_string(); + let p = prepare( + Dialect::Chat, + chat.as_bytes(), + "/v1/chat/completions", + None, + &target(Dialect::Anthropic), + ) + .unwrap(); + let v: Value = serde_json::from_slice(&p.body).unwrap(); + assert_eq!(marks(&v), [("messages", 0, 0)]); + // 不认断点的格式什么都不加 + for d in [Dialect::Chat, Dialect::Gemini, Dialect::Responses] { + let [_, second, _] = turns(); + let v = codex(second, d, "m"); + assert!(!v.to_string().contains("cache_control"), "{d:?}"); + assert!(!v.to_string().contains("cachePoint"), "{d:?}"); + } +} + +#[test] +fn claude_on_bedrock_gets_the_same_four_cache_points() { + let [_, second, _] = turns(); + let v = codex( + second.clone(), + Dialect::Bedrock, + "us.anthropic.claude-sonnet-4-5-20250929-v1:0", + ); + let point = json!({"cachePoint": {"type": "default"}}); + let tools = v["toolConfig"]["tools"].as_array().unwrap(); + assert_eq!(tools.len(), 4, "{v}"); + assert_eq!(tools[3], point); + assert_eq!(v["system"].as_array().unwrap().last(), Some(&point)); + let messages = v["messages"].as_array().unwrap(); + for i in [2, 4] { + assert_eq!( + messages[i]["content"].as_array().unwrap().last(), + Some(&point), + "{i}" + ); + } + assert_eq!(v.to_string().matches("cachePoint").count(), 4); + // 别家的模型不自动标:不认 cachePoint 的会拒掉整个请求。看不出背后是谁的 ARN 也不标 + for model in [ + "meta.llama3-70b-instruct-v1:0", + "amazon.nova-pro-v1:0", + "arn:aws:bedrock:us-east-1:123:application-inference-profile/x", + ] { + let v = codex(second.clone(), Dialect::Bedrock, model); + assert!(!v.to_string().contains("cachePoint"), "{model}"); + } +} diff --git a/crates/tw-dialect/tests/codex_compaction.rs b/crates/tw-dialect/tests/codex_compaction.rs index b31f4aef..b7d5ac52 100644 --- a/crates/tw-dialect/tests/codex_compaction.rs +++ b/crates/tw-dialect/tests/codex_compaction.rs @@ -145,6 +145,26 @@ fn parts_of(d: Dialect, v: &Value) -> (Value, Value, Vec) { // ───────────────────────────────────────────────────────── 对话中途的 developer 消息 +/// 去掉缓存断点:Anthropic 的 `cache_control`、Converse 的 `cachePoint` 块 +fn unmarked(messages: &[Value]) -> Vec { + fn strip(v: &mut Value) { + match v { + Value::Object(o) => { + o.remove("cache_control"); + o.values_mut().for_each(strip); + } + Value::Array(a) => { + a.retain(|x| x.get("cachePoint").is_none()); + a.iter_mut().for_each(strip); + } + _ => {} + } + } + let mut out = messages.to_vec(); + out.iter_mut().for_each(strip); + out +} + #[test] fn a_developer_message_mid_conversation_leaves_the_prefix_alone() { // 这一轮:历史 + 新的一句 @@ -173,8 +193,13 @@ fn a_developer_message_mid_conversation_leaves_the_prefix_alone() { ); assert!(!sys.contains("collaboration_mode"), "{upstream:?}: {sys}"); assert!(!sys.contains("approval: never"), "{upstream:?}: {sys}"); - // 前一个请求的对话是后一个的开头 - assert_eq!(msgs_a[..], msgs_b[..msgs_a.len()], "{upstream:?}"); + // 前一个请求的对话是后一个的开头。缓存断点不算内容:自动标的那几个每一轮往后挪 + // (见 tests/auto_cache.rs) + assert_eq!( + unmarked(&msgs_a), + unmarked(&msgs_b[..msgs_a.len()]), + "{upstream:?}" + ); // 中途的 developer 消息在它原来的位置,标明是系统说的 let later = Value::Array(msgs_b[msgs_a.len()..].to_vec()).to_string(); assert!( diff --git a/crates/tw-gateway/tests/conversion.rs b/crates/tw-gateway/tests/conversion.rs index 7643dbdc..6a01c92f 100644 --- a/crates/tw-gateway/tests/conversion.rs +++ b/crates/tw-gateway/tests/conversion.rs @@ -809,3 +809,72 @@ async fn a_summary_written_on_a_claude_route_reaches_openai_as_a_message() { ); assert!(!got.to_string().contains("tw1.c.")); } + +#[tokio::test] +async fn a_codex_request_on_a_claude_route_marks_cache_breakpoints_and_records_cache_usage() { + // Claude 报了这一次写进缓存多少、从缓存读了多少 + let stream = [ + "event: message_start\ndata: {\"type\":\"message_start\",\"message\":{\"id\":\"msg_1\",\"model\":\"claude-opus-4-7\",\"usage\":{\"input_tokens\":30,\"cache_creation_input_tokens\":2000,\"cache_read_input_tokens\":9000,\"output_tokens\":1}}}\n\n", + "event: content_block_start\ndata: {\"type\":\"content_block_start\",\"index\":0,\"content_block\":{\"type\":\"text\",\"text\":\"\"}}\n\n", + "event: content_block_delta\ndata: {\"type\":\"content_block_delta\",\"index\":0,\"delta\":{\"type\":\"text_delta\",\"text\":\"ok\"}}\n\n", + "event: content_block_stop\ndata: {\"type\":\"content_block_stop\",\"index\":0}\n\n", + "event: message_delta\ndata: {\"type\":\"message_delta\",\"delta\":{\"stop_reason\":\"end_turn\"},\"usage\":{\"output_tokens\":12}}\n\n", + "event: message_stop\ndata: {\"type\":\"message_stop\"}\n\n", + ] + .concat(); + let (up, seen) = upstream(200, "text/event-stream", stream).await; + let (gw, mut rx) = gateway(provider(up, Protocol::Anthropic), SecurityMode::Observe).await; + let (status, _, body) = post( + gw, + "/v1/responses", + &[("authorization", "Bearer tw-k")], + codex_lite_request(), + ) + .await; + assert_eq!(status, 200, "{body}"); + + // Codex 不标断点:转给 Claude 时替它标在工具末尾、系统提示末尾、最后一条用户消息末尾 + let sent: Value = serde_json::from_slice(&seen.lock().unwrap().body).unwrap(); + let ephemeral = json!({"type": "ephemeral"}); + assert_eq!(sent["tools"][2]["cache_control"], ephemeral); + assert_eq!( + sent["system"].as_array().unwrap().last().unwrap()["cache_control"], + ephemeral + ); + assert_eq!( + sent["messages"][0]["content"][0]["cache_control"], + ephemeral + ); + assert_eq!(sent.to_string().matches("cache_control").count(), 3); + + // 缓存读写照上游报的记下,计费按价目表的缓存单价算 + let mut finished = None; + while let Ok(Ok(ev)) = tokio::time::timeout(Duration::from_secs(3), rx.recv()).await { + if let tw_api::Event::RequestFinished { usage, .. } = ev { + finished = usage; + break; + } + } + let u = finished.expect("没有用量"); + assert_eq!( + (u.input, u.cache_write, u.cache_read, u.output), + (30, 2000, 9000, 12) + ); + assert!(!u.cache_1h); +} + +#[tokio::test] +async fn a_claude_request_without_breakpoints_passes_straight_through_unchanged() { + // 同格式直通一个字节都不改:自动标断点只在转换时 + let (up, seen) = upstream(200, "text/event-stream", ANTHROPIC_STREAM.into()).await; + let (gw, _) = gateway(provider(up, Protocol::Anthropic), SecurityMode::Observe).await; + let sent = json!({ + "model": "claude-opus-4-7", "max_tokens": 100, "stream": true, + "system": "Be brief.", + "tools": [{"name": "Read", "input_schema": {"type": "object"}}], + "messages": [{"role": "user", "content": "hi"}] + }); + let (status, _, body) = post(gw, "/v1/messages", &[("x-api-key", "tw-k")], sent.clone()).await; + assert_eq!(status, 200, "{body}"); + assert_eq!(seen.lock().unwrap().body, sent.to_string().into_bytes()); +} diff --git a/crates/tw-gateway/tests/harness.rs b/crates/tw-gateway/tests/harness.rs index cca5da11..1e04cfd9 100644 --- a/crates/tw-gateway/tests/harness.rs +++ b/crates/tw-gateway/tests/harness.rs @@ -532,13 +532,18 @@ async fn chat_to_an_anthropic_upstream_is_cleaned_by_the_conversion() { assert_eq!(system, ["You are DeepSeek Harness."]); let messages = v["messages"].as_array().unwrap(); assert!(messages.iter().all(|m| m["role"] != "system")); + // 客户端没标缓存断点:最后两条用户消息的末尾各标一个 assert_eq!( messages[2]["content"], json!([ {"type": "text", "text": "\nThe project root is /work.\n"}, - {"type": "text", "text": "search it"} + {"type": "text", "text": "search it", "cache_control": {"type": "ephemeral"}} ]) ); + assert_eq!( + messages[0]["content"][0]["cache_control"], + json!({"type": "ephemeral"}) + ); assert_eq!(v["thinking"]["type"], "enabled"); let (bytes, translated) = events(&mut rx).await; diff --git a/crates/tw-store/src/recorder.rs b/crates/tw-store/src/recorder.rs index bf077d62..0ff511e4 100644 --- a/crates/tw-store/src/recorder.rs +++ b/crates/tw-store/src/recorder.rs @@ -2126,6 +2126,31 @@ mod cache_saving_tests { ); } + /// 缓存写和缓存读按价目表的缓存单价算,不按输入价:转给 Claude 时自动标的断点让 + /// Codex 这类客户端的请求也有了这两项 + #[test] + fn cache_writes_and_reads_are_charged_at_the_cache_prices() { + let (_d, mut r) = rec(); + r.on_event(&started(1, "claude-sonnet-4-5")); + r.on_event(&finished( + 1, + Some(UsageView { + input: 1000, + output: 500, + cache_read: 100_000, + cache_write: 10_000, + ..Default::default() + }), + )); + let row = r.db().get(1).unwrap().unwrap(); + // Sonnet 4.5:输入 $3、输出 $15、缓存读 $0.30、5 分钟缓存写 $3.75(每百万 token) + // 0.003 + 0.0075 + 0.03 + 0.0375 + assert_eq!(row.cost_micros, Some(78_000)); + assert!(!row.cost_estimated); + // 读省下 10 万 × $2.70,写多花 1 万 × $0.75 + assert_eq!(row.cache_saved_micros, Some(270_000 - 7_500)); + } + #[test] fn a_request_with_no_cache_hit_saved_a_real_zero() { let (_d, mut r) = rec(); diff --git a/docs/config.md b/docs/config.md index 00ce27c7..20119977 100644 --- a/docs/config.md +++ b/docs/config.md @@ -454,6 +454,19 @@ Identity fields that clients fill in themselves, such as Claude Code's `metadata.user_id`, are removed from the body. For an upstream that admits only certain clients, turn on `forward_client_identity`. +A request converted to Anthropic, or to Claude on Bedrock, marks where the +upstream may cache the prompt when the client marked nothing itself. Clients +in OpenAI or Gemini formats such as Codex cannot mark anything: those +providers cache a repeated prompt on their own, while Anthropic caches only +what is marked. The marks go at the end of the tools, at the end of the +system prompt and at the end of the last two user turns, at most four, each +kept for the default five minutes. The earlier of the two user marks is where +the previous request ended, so each turn reads back what the turn before +wrote and pays the cache price for it instead of the full input price. Cache +writes and reads are charged at the price table's cache prices. A request +that carries its own marks, as Claude Code's do, keeps exactly those, and a +request sent on in the upstream's own format is not changed. + A ChatGPT account upstream (`protocol: chatgpt`) takes only the credential the desktop app obtains by signing in; it cannot be written by hand. Claude and Google subscription sign-ins are not supported; use an API key. diff --git a/docs/config.zh-CN.md b/docs/config.zh-CN.md index d2efbc5d..57cd8312 100644 --- a/docs/config.zh-CN.md +++ b/docs/config.zh-CN.md @@ -333,6 +333,8 @@ providers: 每个请求只带请求本身和上游需要的请求头,客户端的其他信息一律不发:凭据和 `headers` 中写的请求头、ThinkWatch 自己的 `User-Agent`,以及客户端请求中该上游协议使用的请求头(Anthropic 为 `anthropic-*`,OpenAI 为 `Idempotency-Key` 和 `X-Client-Request-Id`,Gemini 没有)。客户端自动填写的身份字段(如 Claude Code 的 `metadata.user_id`)从请求体中去掉。只接受特定客户端的上游,打开 `forward_client_identity`。 +请求转换为 Anthropic 格式、或转给 Bedrock 上的 Claude 时,客户端自己没有标出缓存位置的,由网关标出可以缓存的位置。Codex 等使用 OpenAI 或 Gemini 格式的客户端无从标注:这两家自动缓存重复的提示,Anthropic 只缓存标出的部分。标注的位置是工具列表末尾、系统提示末尾和最后两条用户消息的末尾,最多四处,缓存时长为默认的五分钟。两条用户消息中靠前的那一处正是上一个请求结束的位置,因此每一轮都能读回上一轮写入的缓存,这部分按缓存价而不是全额输入价计费。缓存的写入和读取按价目表中的缓存单价计费。自带标注的请求(如 Claude Code 的)保持原样,按上游自身格式直接转发的请求不做改动。 + ChatGPT 账号上游(`protocol: chatgpt`)只接受桌面应用登录得到的凭据,不能手写。不支持 Claude 和 Google 的订阅登录,请使用 API 密钥。 有的中转站和账号同时只接受几个请求,多出来的直接拒绝。`max_concurrent` 让网关守住这个数:请求发出时占用这家的一个位置,回答完整交给客户端、或者客户端断开时归还。这家满了的时候,为复用提示缓存而留在这家的对话等空位,别的请求直接换下一家。最多等多久由 `failover.slot_wait_secs` 决定。等待不算失败,这家不会因此停用。只计算 token 数的请求不占位置。Responses 的 WebSocket 连接上,每个 `response.create` 从发出起占一个位置,直到它的回答结束,密钥的 `max_concurrent` 也一样;空闲的连接不占位置。