diff --git a/crates/tw-dialect/src/anthropic/request.rs b/crates/tw-dialect/src/anthropic/request.rs index 0651656b..b25bd6b7 100644 --- a/crates/tw-dialect/src/anthropic/request.rs +++ b/crates/tw-dialect/src/anthropic/request.rs @@ -209,6 +209,47 @@ fn cache_ttl(b: &Value) -> Option { }) } +/// 客户端没标断点时自动标的:**工具的末尾、系统提示的末尾、最后两条用户消息的末尾**, +/// 最多四个(Anthropic 的上限),5 分钟的默认 TTL。 +/// +/// 别的格式的客户端(Codex、Chat、Gemini 的客户端)没有这个写法:OpenAI 和 Gemini 自己 +/// 缓存开头相同的部分,转给 Anthropic 时不标就一点都不缓存,每一轮整段对话全价重算。 +/// +/// - 工具和系统提示:一段对话里几乎不变,各标一个,系统提示变了工具那段照样命中 +/// - 最后一条用户消息:这一轮写进缓存 +/// - 倒数第二条用户消息:**正是上一轮请求的最后一个断点**,这一轮从它读回上一轮写的缓存。 +/// Anthropic 只往回找约 20 块,一轮里工具调用多的时候光靠最后一个断点找不回去 +/// +/// 断点不算内容:上一轮标在更早位置的那个这一轮挪走了,缓存照样命中(Anthropic 文档里多轮 +/// 对话就是这么标的)。太短(不到模型的最小缓存长度)的断点上游直接忽略,不报错,所以不估 +/// 长度。思考块不能标,标在它前面的那块上 +fn auto_cache(out: &mut Map) { + fn mark(blocks: Option<&mut Value>) { + let Some(Value::Array(blocks)) = blocks else { + return; + }; + if let Some(b) = blocks + .iter_mut() + .rev() + .find(|b| !matches!(str_of(b, "type"), Some("thinking" | "redacted_thinking"))) + { + b["cache_control"] = cache_control(CacheTtl::Short); + } + } + mark(out.get_mut("tools")); + mark(out.get_mut("system")); + if let Some(Value::Array(messages)) = out.get_mut("messages") { + for m in messages + .iter_mut() + .rev() + .filter(|m| str_of(m, "role") == Some("user")) + .take(2) + { + mark(m.get_mut("content")); + } + } +} + /// 中间表示的断点 → `cache_control` fn cache_control(ttl: CacheTtl) -> Value { match ttl { @@ -396,6 +437,10 @@ pub fn encode_request(r: &Request, t: &Target, dropped: &mut Dropped) -> Value { .collect(); out.insert("tools".into(), Value::Array(tools)); } + // 客户端自己一个断点都没标:替它标(Claude Code 这类标了的原样不动) + if r.cache.is_empty() { + auto_cache(&mut out); + } let serial = r.parallel_tool_calls == Some(false); let choice = match (&r.tool_choice, serial) { diff --git a/crates/tw-dialect/src/bedrock/request.rs b/crates/tw-dialect/src/bedrock/request.rs index 3193ca6b..5cab8b27 100644 --- a/crates/tw-dialect/src/bedrock/request.rs +++ b/crates/tw-dialect/src/bedrock/request.rs @@ -80,6 +80,30 @@ fn caches(model: &str) -> bool { m.contains("anthropic.claude") || m.contains("amazon.nova") || m.starts_with("arn:") } +/// 客户端没标断点时自动标的,和转给 Anthropic 时同样的四处:工具的末尾、系统提示的末尾、 +/// 最后两条用户消息的末尾,各跟一个 5 分钟的 `cachePoint`(Converse 也最多四个) +fn auto_cache(out: &mut Map) { + fn mark(blocks: Option<&mut Value>) { + if let Some(Value::Array(blocks)) = blocks + && !blocks.is_empty() + { + blocks.push(cache_point(CacheTtl::Short)); + } + } + mark(out.get_mut("toolConfig").and_then(|c| c.get_mut("tools"))); + mark(out.get_mut("system")); + if let Some(Value::Array(messages)) = out.get_mut("messages") { + for m in messages + .iter_mut() + .rev() + .filter(|m| m.get("role").and_then(Value::as_str) == Some("user")) + .take(2) + { + mark(m.get_mut("content")); + } + } +} + /// 一个缓存断点块。`ttl` 缺省是 5 分钟 fn cache_point(ttl: CacheTtl) -> Value { match ttl { @@ -207,6 +231,12 @@ pub fn encode_request(r: &Request, t: &Target, dropped: &mut Dropped) -> Value { } else if r.tool_choice.is_some() { dropped.path("tool_choice"); } + // 客户端自己一个断点都没标,模型又是 Claude:替它标(见 `anthropic::request` 的 + // `auto_cache`)。Nova 和看不出是谁的推理配置 ARN 不自动标:不认的模型收到 + // `cachePoint` 会拒掉整个请求 + if r.cache.is_empty() && is_claude(&r.model) { + auto_cache(&mut out); + } // Converse 自己没有的旋钮 if r.seed.is_some() { diff --git a/crates/tw-dialect/tests/auto_cache.rs b/crates/tw-dialect/tests/auto_cache.rs new file mode 100644 index 00000000..ca0a7f58 --- /dev/null +++ b/crates/tw-dialect/tests/auto_cache.rs @@ -0,0 +1,288 @@ +//! 转给 Claude(Anthropic 格式、Bedrock 上的 Claude)时,客户端没标的提示缓存断点替它标上。 +//! +//! Codex 这类说 OpenAI 格式的客户端没有断点的写法:OpenAI 自己缓存开头相同的部分。转给 +//! Anthropic 时不标,就一点都不缓存,每一轮整段对话全价重算。标在工具末尾、系统提示末尾、 +//! 最后两条用户消息末尾:倒数第二条正是上一轮的最后一个断点,这一轮从那里读回上一轮写的 +//! 缓存。客户端自己标了的(Claude Code)原样不动。 +//! +//! 请求照 Codex 的 serde 类型写(`codex-rs/protocol/src/models.rs` 的 `ResponseItem`)。 + +// 整个请求写成一个 `json!`,嵌套得深 +#![recursion_limit = "512"] + +use serde_json::{Value, json}; +use tw_dialect::convert::{Prepared, decode, prepare}; +use tw_dialect::ir::*; + +fn tools() -> Value { + json!({"id": "at_1", "type": "additional_tools", "role": "developer", "tools": [ + {"type": "namespace", "name": "functions", "description": "", "tools": [ + {"type": "function", "name": "exec_command", "description": "Runs a command.", "strict": false, + "parameters": {"type": "object", "properties": {"cmd": {"type": "string"}}, "required": ["cmd"]}}, + {"type": "custom", "name": "apply_patch", "description": "Edit files.", + "format": {"type": "grammar", "syntax": "lark", "definition": "start: begin_patch hunk+ end_patch"}} + ]}, + {"type": "namespace", "name": "mcp__codex_apps__calendar", "description": "Plan events.", "tools": [ + {"type": "function", "name": "_create_event", "description": "Create an event.", "strict": false, + "parameters": {"type": "object", "properties": {"title": {"type": "string"}}}} + ]} + ]}) +} + +/// Codex 一轮里接连的三个请求:每一步多一次工具调用和它的结果,最后用户接着说 +fn turns() -> [Vec; 3] { + let first = vec![ + tools(), + json!({"id": "msg_b", "type": "message", "role": "developer", + "content": [{"type": "input_text", "text": "You are Codex, a coding agent."}]}), + json!({"type": "message", "role": "user", + "content": [{"type": "input_text", "text": "\n /repo\n"}]}), + json!({"type": "message", "role": "user", "content": [{"type": "input_text", "text": "Fix the failing parser test."}]}), + json!({"type": "message", "role": "assistant", "content": [{"type": "output_text", "text": "Running the tests."}], "phase": "commentary"}), + json!({"type": "function_call", "name": "exec_command", "namespace": "functions", "arguments": "{\"cmd\":\"cargo test\"}", "call_id": "call_1"}), + json!({"type": "function_call_output", "call_id": "call_1", "output": "parser::tests::nested FAILED"}), + ]; + let mut second = first.clone(); + second.extend([ + json!({"type": "custom_tool_call", "call_id": "call_2", "name": "apply_patch", "namespace": "functions", + "input": "*** Begin Patch\n*** Update File: src/parser.rs\n*** End Patch\n"}), + json!({"type": "custom_tool_call_output", "call_id": "call_2", "output": "Success."}), + ]); + let mut third = second.clone(); + third.extend([ + json!({"type": "message", "role": "assistant", "content": [{"type": "output_text", "text": "Fixed; the tests pass."}], "phase": "final_answer"}), + json!({"type": "message", "role": "user", "content": [{"type": "input_text", "text": "Now update the docs."}]}), + ]); + [first, second, third] +} + +fn request(input: Vec) -> Value { + json!({ + "model": "gpt-5.4", "stream": true, "input": input, + "tool_choice": "auto", "parallel_tool_calls": false, + "reasoning": {"effort": "medium", "summary": "auto"}, + "store": false, "include": ["reasoning.encrypted_content"] + }) +} + +fn target(d: Dialect) -> Target { + Target { + dialect: d, + official: false, + default_max_tokens: 8192, + } +} + +/// 网关的做法:解码一次,改成上游的模型名,再按上游的格式编码 +fn codex(input: Vec, upstream: Dialect, model: &str) -> Value { + let mut d = decode(Dialect::Responses, &request(input), "/v1/responses", None).unwrap(); + d.request.model = model.to_string(); + let p: Prepared = d.encode(&target(upstream)); + serde_json::from_slice(&p.body).unwrap() +} + +fn claude(input: Vec) -> Value { + codex(input, Dialect::Anthropic, "claude-opus-4-7") +} + +/// 标了 `cache_control` 的块在哪:("tools" | "system" | "messages", 第几条, 第几块) +fn marks(v: &Value) -> Vec<(&'static str, usize, usize)> { + let mut out = Vec::new(); + for (key, list) in [("tools", &v["tools"]), ("system", &v["system"])] { + for (i, b) in list.as_array().into_iter().flatten().enumerate() { + if b.get("cache_control").is_some() { + assert_eq!(b["cache_control"], json!({"type": "ephemeral"}), "{b}"); + out.push((key, i, 0)); + } + } + } + for (i, m) in v["messages"].as_array().unwrap().iter().enumerate() { + for (j, b) in m["content"].as_array().into_iter().flatten().enumerate() { + if b.get("cache_control").is_some() { + assert_eq!(b["cache_control"], json!({"type": "ephemeral"}), "{b}"); + out.push(("messages", i, j)); + } + } + } + out +} + +fn without_marks(mut v: Value) -> Value { + fn strip(v: &mut Value) { + match v { + Value::Object(o) => { + o.remove("cache_control"); + o.values_mut().for_each(strip); + } + Value::Array(a) => a.iter_mut().for_each(strip), + _ => {} + } + } + strip(&mut v); + v +} + +/// 一个请求从开头到 (第几条消息, 第几块) 为止的那段,按 Anthropic 算缓存的顺序:工具、 +/// 系统提示、消息 +fn prefix_to(v: &Value, message: usize, block: usize) -> String { + let mut messages: Vec = v["messages"].as_array().unwrap()[..=message].to_vec(); + let last = messages.last_mut().unwrap(); + let content = last["content"].as_array().unwrap()[..=block].to_vec(); + last["content"] = Value::Array(content); + json!([v["tools"], v["system"], messages]).to_string() +} + +#[test] +fn a_codex_request_for_claude_gets_the_four_usual_breakpoints() { + let [_, second, _] = turns(); + let v = claude(second); + let roles: Vec<&str> = v["messages"] + .as_array() + .unwrap() + .iter() + .map(|m| m["role"].as_str().unwrap()) + .collect(); + assert_eq!(roles, ["user", "assistant", "user", "assistant", "user"]); + assert_eq!( + marks(&v), + [ + // 工具末尾 + ("tools", 2, 0), + // 系统提示末尾:基础指令之后还有 MCP namespace 的说明 + ("system", 1, 0), + // 上一轮的最后一条:exec_command 的结果 + ("messages", 2, 0), + // 这一轮的最后一条:apply_patch 的结果 + ("messages", 4, 0), + ] + ); + assert_eq!(v.to_string().matches("cache_control").count(), 4); +} + +#[test] +fn each_turn_reads_back_what_the_turn_before_wrote() { + let [first, second, third] = turns(); + let bodies = [claude(first), claude(second), claude(third)]; + for pair in bodies.windows(2) { + let (earlier, later) = (&pair[0], &pair[1]); + // 前一个请求的最后一个断点 + let &(_, i, j) = marks(earlier).last().unwrap(); + // 后一个请求在同一块上也有一个:从这里读回前一个写进去的缓存 + assert!( + marks(later).contains(&("messages", i, j)), + "{:?} vs {:?}", + marks(earlier), + marks(later) + ); + // 到那一块为止一个字节都不差(断点本身不算内容:前一个请求标在更早位置的那个, + // 这一轮挪走了 —— Anthropic 文档里多轮对话就是这么标的) + assert_eq!( + prefix_to(&without_marks(earlier.clone()), i, j), + prefix_to(&without_marks(later.clone()), i, j) + ); + // 工具和系统提示连断点都一样 + assert_eq!(earlier["tools"], later["tools"]); + assert_eq!(earlier["system"], later["system"]); + } +} + +#[test] +fn a_client_that_marks_its_own_breakpoints_keeps_exactly_those() { + // Claude Code 的请求:自己标了系统提示和最后一条消息 + let claude_code = json!({ + "model": "claude-opus-4-7", "max_tokens": 1024, + "system": [{"type": "text", "text": "You are Claude Code.", "cache_control": {"type": "ephemeral"}}], + "tools": [{"name": "Read", "input_schema": {"type": "object"}}], + "messages": [ + {"role": "user", "content": "hi"}, + {"role": "assistant", "content": "hello"}, + {"role": "user", "content": [{"type": "text", "text": "read a.rs", "cache_control": {"type": "ephemeral", "ttl": "1h"}}]} + ] + }) + .to_string(); + // 转给 Anthropic 格式的另一种写法(经过中间表示) + let p = prepare( + Dialect::Anthropic, + claude_code.as_bytes(), + "/v1/messages", + None, + &target(Dialect::Anthropic), + ) + .unwrap(); + let v: Value = serde_json::from_slice(&p.body).unwrap(); + assert_eq!(marks_any(&v), 2, "{v}"); + assert!(v["tools"][0].get("cache_control").is_none()); + assert_eq!(v["messages"][2]["content"][0]["cache_control"]["ttl"], "1h"); + // 转给 Bedrock 上的 Claude + let mut d = decode( + Dialect::Anthropic, + &serde_json::from_str(&claude_code).unwrap(), + "/v1/messages", + None, + ) + .unwrap(); + d.request.model = "us.anthropic.claude-sonnet-4-5-20250929-v1:0".into(); + let v: Value = serde_json::from_slice(&d.encode(&target(Dialect::Bedrock)).body).unwrap(); + assert_eq!(v.to_string().matches("cachePoint").count(), 2, "{v}"); +} + +fn marks_any(v: &Value) -> usize { + v.to_string().matches("cache_control").count() +} + +#[test] +fn a_short_conversation_gets_only_the_breakpoints_it_has_room_for() { + // 没有工具、没有系统提示、只有一条用户消息:只标那一条 + let chat = json!({"model": "claude-opus-4-7", "messages": [{"role": "user", "content": "hi"}]}) + .to_string(); + let p = prepare( + Dialect::Chat, + chat.as_bytes(), + "/v1/chat/completions", + None, + &target(Dialect::Anthropic), + ) + .unwrap(); + let v: Value = serde_json::from_slice(&p.body).unwrap(); + assert_eq!(marks(&v), [("messages", 0, 0)]); + // 不认断点的格式什么都不加 + for d in [Dialect::Chat, Dialect::Gemini, Dialect::Responses] { + let [_, second, _] = turns(); + let v = codex(second, d, "m"); + assert!(!v.to_string().contains("cache_control"), "{d:?}"); + assert!(!v.to_string().contains("cachePoint"), "{d:?}"); + } +} + +#[test] +fn claude_on_bedrock_gets_the_same_four_cache_points() { + let [_, second, _] = turns(); + let v = codex( + second.clone(), + Dialect::Bedrock, + "us.anthropic.claude-sonnet-4-5-20250929-v1:0", + ); + let point = json!({"cachePoint": {"type": "default"}}); + let tools = v["toolConfig"]["tools"].as_array().unwrap(); + assert_eq!(tools.len(), 4, "{v}"); + assert_eq!(tools[3], point); + assert_eq!(v["system"].as_array().unwrap().last(), Some(&point)); + let messages = v["messages"].as_array().unwrap(); + for i in [2, 4] { + assert_eq!( + messages[i]["content"].as_array().unwrap().last(), + Some(&point), + "{i}" + ); + } + assert_eq!(v.to_string().matches("cachePoint").count(), 4); + // 别家的模型不自动标:不认 cachePoint 的会拒掉整个请求。看不出背后是谁的 ARN 也不标 + for model in [ + "meta.llama3-70b-instruct-v1:0", + "amazon.nova-pro-v1:0", + "arn:aws:bedrock:us-east-1:123:application-inference-profile/x", + ] { + let v = codex(second.clone(), Dialect::Bedrock, model); + assert!(!v.to_string().contains("cachePoint"), "{model}"); + } +} diff --git a/crates/tw-dialect/tests/codex_compaction.rs b/crates/tw-dialect/tests/codex_compaction.rs index b31f4aef..b7d5ac52 100644 --- a/crates/tw-dialect/tests/codex_compaction.rs +++ b/crates/tw-dialect/tests/codex_compaction.rs @@ -145,6 +145,26 @@ fn parts_of(d: Dialect, v: &Value) -> (Value, Value, Vec) { // ───────────────────────────────────────────────────────── 对话中途的 developer 消息 +/// 去掉缓存断点:Anthropic 的 `cache_control`、Converse 的 `cachePoint` 块 +fn unmarked(messages: &[Value]) -> Vec { + fn strip(v: &mut Value) { + match v { + Value::Object(o) => { + o.remove("cache_control"); + o.values_mut().for_each(strip); + } + Value::Array(a) => { + a.retain(|x| x.get("cachePoint").is_none()); + a.iter_mut().for_each(strip); + } + _ => {} + } + } + let mut out = messages.to_vec(); + out.iter_mut().for_each(strip); + out +} + #[test] fn a_developer_message_mid_conversation_leaves_the_prefix_alone() { // 这一轮:历史 + 新的一句 @@ -173,8 +193,13 @@ fn a_developer_message_mid_conversation_leaves_the_prefix_alone() { ); assert!(!sys.contains("collaboration_mode"), "{upstream:?}: {sys}"); assert!(!sys.contains("approval: never"), "{upstream:?}: {sys}"); - // 前一个请求的对话是后一个的开头 - assert_eq!(msgs_a[..], msgs_b[..msgs_a.len()], "{upstream:?}"); + // 前一个请求的对话是后一个的开头。缓存断点不算内容:自动标的那几个每一轮往后挪 + // (见 tests/auto_cache.rs) + assert_eq!( + unmarked(&msgs_a), + unmarked(&msgs_b[..msgs_a.len()]), + "{upstream:?}" + ); // 中途的 developer 消息在它原来的位置,标明是系统说的 let later = Value::Array(msgs_b[msgs_a.len()..].to_vec()).to_string(); assert!( diff --git a/crates/tw-gateway/tests/conversion.rs b/crates/tw-gateway/tests/conversion.rs index 7643dbdc..6a01c92f 100644 --- a/crates/tw-gateway/tests/conversion.rs +++ b/crates/tw-gateway/tests/conversion.rs @@ -809,3 +809,72 @@ async fn a_summary_written_on_a_claude_route_reaches_openai_as_a_message() { ); assert!(!got.to_string().contains("tw1.c.")); } + +#[tokio::test] +async fn a_codex_request_on_a_claude_route_marks_cache_breakpoints_and_records_cache_usage() { + // Claude 报了这一次写进缓存多少、从缓存读了多少 + let stream = [ + "event: message_start\ndata: {\"type\":\"message_start\",\"message\":{\"id\":\"msg_1\",\"model\":\"claude-opus-4-7\",\"usage\":{\"input_tokens\":30,\"cache_creation_input_tokens\":2000,\"cache_read_input_tokens\":9000,\"output_tokens\":1}}}\n\n", + "event: content_block_start\ndata: {\"type\":\"content_block_start\",\"index\":0,\"content_block\":{\"type\":\"text\",\"text\":\"\"}}\n\n", + "event: content_block_delta\ndata: {\"type\":\"content_block_delta\",\"index\":0,\"delta\":{\"type\":\"text_delta\",\"text\":\"ok\"}}\n\n", + "event: content_block_stop\ndata: {\"type\":\"content_block_stop\",\"index\":0}\n\n", + "event: message_delta\ndata: {\"type\":\"message_delta\",\"delta\":{\"stop_reason\":\"end_turn\"},\"usage\":{\"output_tokens\":12}}\n\n", + "event: message_stop\ndata: {\"type\":\"message_stop\"}\n\n", + ] + .concat(); + let (up, seen) = upstream(200, "text/event-stream", stream).await; + let (gw, mut rx) = gateway(provider(up, Protocol::Anthropic), SecurityMode::Observe).await; + let (status, _, body) = post( + gw, + "/v1/responses", + &[("authorization", "Bearer tw-k")], + codex_lite_request(), + ) + .await; + assert_eq!(status, 200, "{body}"); + + // Codex 不标断点:转给 Claude 时替它标在工具末尾、系统提示末尾、最后一条用户消息末尾 + let sent: Value = serde_json::from_slice(&seen.lock().unwrap().body).unwrap(); + let ephemeral = json!({"type": "ephemeral"}); + assert_eq!(sent["tools"][2]["cache_control"], ephemeral); + assert_eq!( + sent["system"].as_array().unwrap().last().unwrap()["cache_control"], + ephemeral + ); + assert_eq!( + sent["messages"][0]["content"][0]["cache_control"], + ephemeral + ); + assert_eq!(sent.to_string().matches("cache_control").count(), 3); + + // 缓存读写照上游报的记下,计费按价目表的缓存单价算 + let mut finished = None; + while let Ok(Ok(ev)) = tokio::time::timeout(Duration::from_secs(3), rx.recv()).await { + if let tw_api::Event::RequestFinished { usage, .. } = ev { + finished = usage; + break; + } + } + let u = finished.expect("没有用量"); + assert_eq!( + (u.input, u.cache_write, u.cache_read, u.output), + (30, 2000, 9000, 12) + ); + assert!(!u.cache_1h); +} + +#[tokio::test] +async fn a_claude_request_without_breakpoints_passes_straight_through_unchanged() { + // 同格式直通一个字节都不改:自动标断点只在转换时 + let (up, seen) = upstream(200, "text/event-stream", ANTHROPIC_STREAM.into()).await; + let (gw, _) = gateway(provider(up, Protocol::Anthropic), SecurityMode::Observe).await; + let sent = json!({ + "model": "claude-opus-4-7", "max_tokens": 100, "stream": true, + "system": "Be brief.", + "tools": [{"name": "Read", "input_schema": {"type": "object"}}], + "messages": [{"role": "user", "content": "hi"}] + }); + let (status, _, body) = post(gw, "/v1/messages", &[("x-api-key", "tw-k")], sent.clone()).await; + assert_eq!(status, 200, "{body}"); + assert_eq!(seen.lock().unwrap().body, sent.to_string().into_bytes()); +} diff --git a/crates/tw-gateway/tests/harness.rs b/crates/tw-gateway/tests/harness.rs index cca5da11..1e04cfd9 100644 --- a/crates/tw-gateway/tests/harness.rs +++ b/crates/tw-gateway/tests/harness.rs @@ -532,13 +532,18 @@ async fn chat_to_an_anthropic_upstream_is_cleaned_by_the_conversion() { assert_eq!(system, ["You are DeepSeek Harness."]); let messages = v["messages"].as_array().unwrap(); assert!(messages.iter().all(|m| m["role"] != "system")); + // 客户端没标缓存断点:最后两条用户消息的末尾各标一个 assert_eq!( messages[2]["content"], json!([ {"type": "text", "text": "\nThe project root is /work.\n"}, - {"type": "text", "text": "search it"} + {"type": "text", "text": "search it", "cache_control": {"type": "ephemeral"}} ]) ); + assert_eq!( + messages[0]["content"][0]["cache_control"], + json!({"type": "ephemeral"}) + ); assert_eq!(v["thinking"]["type"], "enabled"); let (bytes, translated) = events(&mut rx).await; diff --git a/crates/tw-store/src/recorder.rs b/crates/tw-store/src/recorder.rs index bf077d62..0ff511e4 100644 --- a/crates/tw-store/src/recorder.rs +++ b/crates/tw-store/src/recorder.rs @@ -2126,6 +2126,31 @@ mod cache_saving_tests { ); } + /// 缓存写和缓存读按价目表的缓存单价算,不按输入价:转给 Claude 时自动标的断点让 + /// Codex 这类客户端的请求也有了这两项 + #[test] + fn cache_writes_and_reads_are_charged_at_the_cache_prices() { + let (_d, mut r) = rec(); + r.on_event(&started(1, "claude-sonnet-4-5")); + r.on_event(&finished( + 1, + Some(UsageView { + input: 1000, + output: 500, + cache_read: 100_000, + cache_write: 10_000, + ..Default::default() + }), + )); + let row = r.db().get(1).unwrap().unwrap(); + // Sonnet 4.5:输入 $3、输出 $15、缓存读 $0.30、5 分钟缓存写 $3.75(每百万 token) + // 0.003 + 0.0075 + 0.03 + 0.0375 + assert_eq!(row.cost_micros, Some(78_000)); + assert!(!row.cost_estimated); + // 读省下 10 万 × $2.70,写多花 1 万 × $0.75 + assert_eq!(row.cache_saved_micros, Some(270_000 - 7_500)); + } + #[test] fn a_request_with_no_cache_hit_saved_a_real_zero() { let (_d, mut r) = rec(); diff --git a/docs/config.md b/docs/config.md index 00ce27c7..20119977 100644 --- a/docs/config.md +++ b/docs/config.md @@ -454,6 +454,19 @@ Identity fields that clients fill in themselves, such as Claude Code's `metadata.user_id`, are removed from the body. For an upstream that admits only certain clients, turn on `forward_client_identity`. +A request converted to Anthropic, or to Claude on Bedrock, marks where the +upstream may cache the prompt when the client marked nothing itself. Clients +in OpenAI or Gemini formats such as Codex cannot mark anything: those +providers cache a repeated prompt on their own, while Anthropic caches only +what is marked. The marks go at the end of the tools, at the end of the +system prompt and at the end of the last two user turns, at most four, each +kept for the default five minutes. The earlier of the two user marks is where +the previous request ended, so each turn reads back what the turn before +wrote and pays the cache price for it instead of the full input price. Cache +writes and reads are charged at the price table's cache prices. A request +that carries its own marks, as Claude Code's do, keeps exactly those, and a +request sent on in the upstream's own format is not changed. + A ChatGPT account upstream (`protocol: chatgpt`) takes only the credential the desktop app obtains by signing in; it cannot be written by hand. Claude and Google subscription sign-ins are not supported; use an API key. diff --git a/docs/config.zh-CN.md b/docs/config.zh-CN.md index d2efbc5d..57cd8312 100644 --- a/docs/config.zh-CN.md +++ b/docs/config.zh-CN.md @@ -333,6 +333,8 @@ providers: 每个请求只带请求本身和上游需要的请求头,客户端的其他信息一律不发:凭据和 `headers` 中写的请求头、ThinkWatch 自己的 `User-Agent`,以及客户端请求中该上游协议使用的请求头(Anthropic 为 `anthropic-*`,OpenAI 为 `Idempotency-Key` 和 `X-Client-Request-Id`,Gemini 没有)。客户端自动填写的身份字段(如 Claude Code 的 `metadata.user_id`)从请求体中去掉。只接受特定客户端的上游,打开 `forward_client_identity`。 +请求转换为 Anthropic 格式、或转给 Bedrock 上的 Claude 时,客户端自己没有标出缓存位置的,由网关标出可以缓存的位置。Codex 等使用 OpenAI 或 Gemini 格式的客户端无从标注:这两家自动缓存重复的提示,Anthropic 只缓存标出的部分。标注的位置是工具列表末尾、系统提示末尾和最后两条用户消息的末尾,最多四处,缓存时长为默认的五分钟。两条用户消息中靠前的那一处正是上一个请求结束的位置,因此每一轮都能读回上一轮写入的缓存,这部分按缓存价而不是全额输入价计费。缓存的写入和读取按价目表中的缓存单价计费。自带标注的请求(如 Claude Code 的)保持原样,按上游自身格式直接转发的请求不做改动。 + ChatGPT 账号上游(`protocol: chatgpt`)只接受桌面应用登录得到的凭据,不能手写。不支持 Claude 和 Google 的订阅登录,请使用 API 密钥。 有的中转站和账号同时只接受几个请求,多出来的直接拒绝。`max_concurrent` 让网关守住这个数:请求发出时占用这家的一个位置,回答完整交给客户端、或者客户端断开时归还。这家满了的时候,为复用提示缓存而留在这家的对话等空位,别的请求直接换下一家。最多等多久由 `failover.slot_wait_secs` 决定。等待不算失败,这家不会因此停用。只计算 token 数的请求不占位置。Responses 的 WebSocket 连接上,每个 `response.create` 从发出起占一个位置,直到它的回答结束,密钥的 `max_concurrent` 也一样;空闲的连接不占位置。