From 604ad4f73ef99921f795f71f76acb9624fdeeb52 Mon Sep 17 00:00:00 2001 From: Dmitry Prudnikov Date: Thu, 1 Oct 2026 04:01:03 +0300 Subject: [PATCH] perf(encode): tag Fast slots only where the tag pays - The branch probe (window_log >= 19, upstream zstd's useCmov boundary) runs bare: there the tag's extra live values spilled the scan's data base, hash shift and table base, about 13 more instructions per position at the same probe count, which made level 1 slower than v0.0.57 on frames from 512 KiB. The tagged branch-probe kernels are no longer built. - A copy-mode dictionary below 64 KiB leaves the table bare: its fill tagged small frames where bare slots ran faster. Upstream strips the tags when it copies a CDict table into the context. - The cmov probe keeps its tags, which pay 4-20% from 16 to 256 KiB. Part of #548 --- zstd/src/encoding/match_generator/tests.rs | 46 ++++++++++++++++++++++ zstd/src/encoding/simple/fast_matcher.rs | 38 ++++++++++++------ 2 files changed, 73 insertions(+), 11 deletions(-) diff --git a/zstd/src/encoding/match_generator/tests.rs b/zstd/src/encoding/match_generator/tests.rs index c1d580f2..7c58b90e 100644 --- a/zstd/src/encoding/match_generator/tests.rs +++ b/zstd/src/encoding/match_generator/tests.rs @@ -5479,6 +5479,52 @@ fn fast_slots_are_tagged_by_how_full_the_scan_leaves_the_table() { ); } +/// The two cases a filled table still leaves bare. A window of 2^19 or more +/// selects the branch probe, where the tag's live values spill the scan's +/// bases (a 256 KiB frame at level 1 has window 18 and is tagged, a 512 KiB one +/// has window 19 and is not). A copy-mode dictionary below +/// `COPY_MODE_DICT_TAG_MIN` fills a small frame's table, yet bare slots ran +/// faster there (a 1280-byte dictionary, as the dashboard's dictionary row +/// uses, leaves the frame bare; a 64 KiB one tags it). +#[test] +fn fast_slots_stay_bare_under_the_branch_probe_and_small_copy_dictionaries() { + use crate::encoding::workspace::{IngestPlan, Workspace, no_trailing}; + let tagged = |source: usize, dictionary: usize| { + let mut driver = MatchGeneratorDriver::new(1 << 17, 1); + if dictionary > 0 { + driver.set_dictionary_size_hint(crate::encoding::DictionarySizes::raw_content( + dictionary, + )); + } + driver.set_source_size_hint(source as u64); + let mut context = Workspace::new(); + context.begin_layout(source.min(1 << 17), no_trailing, IngestPlan::Stream); + driver.reset_in_workspace(CompressionLevel::Level(1), &mut context); + driver.simple_mut().slots_tagged() + }; + assert!( + tagged(256 * 1024, 0), + "level 1, 256 KiB: cmov probe, tagged" + ); + assert!( + !tagged(512 * 1024, 0), + "level 1, 512 KiB: branch probe, bare" + ); + assert!(!tagged(1 << 20, 0), "level 1, 1 MiB: branch probe, bare"); + assert!( + !tagged(10 * 1024, 1280), + "level 1, 10 KiB over a 1280-byte copy-mode dictionary: bare" + ); + assert!( + !tagged(10 * 1024, 32 * 1024), + "level 1, 10 KiB over a 32 KiB copy-mode dictionary: bare" + ); + assert!( + tagged(10 * 1024, 64 * 1024), + "level 1, 10 KiB over a 64 KiB copy-mode dictionary: tagged" + ); +} + /// i686 keeps the dfast tables bare below `DFAST_TAGGED_WINDOW_FLOOR`: its loop /// spills the three tags a tagged scan carries. Every other target tags every /// eligible window, and all tag a window past the floor. diff --git a/zstd/src/encoding/simple/fast_matcher.rs b/zstd/src/encoding/simple/fast_matcher.rs index a28512c1..d694729d 100644 --- a/zstd/src/encoding/simple/fast_matcher.rs +++ b/zstd/src/encoding/simple/fast_matcher.rs @@ -146,12 +146,20 @@ fn fast_slots_pay_for_tags( /// /// Measured on x86_64 (bare against tagged): a fill of 0.31 (10 KiB at levels /// 1 and -7) and 1.25 (20 KiB at -7, 10 KiB at -1) ran 3-11% faster bare; 2.0 -/// (32 KiB at -7) and up ran 7-16% faster tagged, and a 10 KiB frame at level 1 -/// over a 110 KiB copy-mode dictionary, whose fill takes the table past it, ran -/// 4% faster tagged. Targets without a measurement of their own take this one. +/// (32 KiB at -7) and up ran 7-16% faster tagged. Targets without a +/// measurement of their own take this one. #[cfg(not(target_arch = "x86"))] const FAST_TAG_MIN_FILL: (u128, u128) = (3, 2); +/// Dictionary length from which a copy-mode frame's table is tagged. The +/// dictionary fill alone takes a small frame's table past [`FAST_TAG_MIN_FILL`], +/// yet on 10 KiB frames at level 1 (x86_64, random and z000033 content) bare +/// slots ran 2-4.5% faster with dictionaries of 1-32 KiB, the two tied at +/// 48 KiB, and tags paid 1-3% from 64 KiB to 110 KiB. Upstream zstd strips the +/// tags from a CDict's table when it copies it into the context +/// (`ZSTD_copyCDictTableIntoCCtx`), so its copy-mode scan always runs bare. +const COPY_MODE_DICT_TAG_MIN: usize = 64 * 1024; + /// Table fill, as `(numerator, denominator)`, from which Fast slots are tagged. /// /// Measured on i686: at a fill of 1.25 (20 KiB at level -7, 10 KiB at -1) the @@ -695,7 +703,17 @@ impl FastKernelMatcher { .map_or(MAX_PRIMED_WINDOW_SIZE, |window| { window.min(MAX_PRIMED_WINDOW_SIZE) }); - let tagged = fast_slots_pay_for_tags(expected_input, dictionary_len, step_size, hash_log) + // Only under the cmov probe (`window_log < 19`, upstream zstd's + // `useCmov`). The branch probe keeps the data base, the hash shift and + // the table base in registers bare; the tag's extra live values push + // them to the stack, so a tagged branch probe ran about 13 more + // instructions per position (46 against 33) at the same probe count, + // and 4-10% slower on z000033 from 512 KiB at levels 1, -1 and -7. The + // cmov probe is register-bound bare as well, and there the tag's halved + // D1 misses pay: 4-17% faster tagged from 32 to 256 KiB. + let tagged = window_log < 19 + && (dictionary_len == 0 || dictionary_len >= COPY_MODE_DICT_TAG_MIN) + && fast_slots_pay_for_tags(expected_input, dictionary_len, step_size, hash_log) && carry != TableCarry::AdvanceEpoch && hash_log + TAG_BITS <= 32 && tagged_positions_fit(primed_window); @@ -2403,8 +2421,10 @@ fn run_fast_kernel_block( window_low, }; // Dispatch on (mls, use_cmov, tagged) — each triple monomorphises the - // kernel hot loop independently. `_` is unreachable: `FastHashTable::new` - // rejects mls outside 4..=8 at construction. + // kernel hot loop independently. A table is tagged only under the cmov + // probe (see `reset`), so the tagged branch-probe copies are never built. + // `_` is unreachable: `FastHashTable::new` rejects mls outside 4..=8 at + // construction. let tagged = hash_table.is_tagged(); macro_rules! run { ($mls:literal, $cmov:literal, $tagged:literal) => { @@ -2421,25 +2441,21 @@ fn run_fast_kernel_block( } let result = match (mls, use_cmov, tagged) { (4, false, false) => run!(4, false, false), - (4, false, true) => run!(4, false, true), (4, true, false) => run!(4, true, false), (4, true, true) => run!(4, true, true), (5, false, false) => run!(5, false, false), - (5, false, true) => run!(5, false, true), (5, true, false) => run!(5, true, false), (5, true, true) => run!(5, true, true), (6, false, false) => run!(6, false, false), - (6, false, true) => run!(6, false, true), (6, true, false) => run!(6, true, false), (6, true, true) => run!(6, true, true), (7, false, false) => run!(7, false, false), - (7, false, true) => run!(7, false, true), (7, true, false) => run!(7, true, false), (7, true, true) => run!(7, true, true), (8, false, false) => run!(8, false, false), - (8, false, true) => run!(8, false, true), (8, true, false) => run!(8, true, false), (8, true, true) => run!(8, true, true), + (_, false, true) => unreachable!("a tagged table under the branch probe"), _ => unreachable!( "FastHashTable construction rejects mls outside 4..=8 — \ got mls={mls} which means the table was bypassed",