From d03520eaf00da39e466006dfc73766f704a58815 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Wed, 16 Sep 2026 17:01:40 +0200 Subject: [PATCH 01/18] Implement reference chains for surviving live-heap samples Add the reference-chain engine for the live-heap profiler: - Track reference chains from at-risk/leak-tag candidate anchors via bounded descend walks, with retention-edge labels on every hop - Correlate leak-tagged allocations with anchor discoveries; cache and re-emit chains for repeated leak-tag targets - Tier anchor selection by class shape (leak/container/other), index root-attached anchors, and cover JNI-global and static-field roots - Guard self-edge chain demotion and contain FIFO floods with a per-class quota on the at-risk queue - Add a heap-wide time-to-OOM predictor to the fast-path leak gate - Make reference-chains debug logs runtime-configurable via rcDebugLevel, silent by default --- ddprof-lib/src/main/cpp/arguments.cpp | 113 + ddprof-lib/src/main/cpp/arguments.h | 140 + ddprof-lib/src/main/cpp/callTraceHashTable.h | 2 +- ddprof-lib/src/main/cpp/callTraceStorage.cpp | 1 + ddprof-lib/src/main/cpp/classTagAllocator.h | 68 + ddprof-lib/src/main/cpp/counters.h | 39 + ddprof-lib/src/main/cpp/event.h | 65 + ddprof-lib/src/main/cpp/flightRecorder.cpp | 192 +- ddprof-lib/src/main/cpp/flightRecorder.h | 35 + ddprof-lib/src/main/cpp/javaApi.cpp | 217 + ddprof-lib/src/main/cpp/jfrMetadata.cpp | 28 + ddprof-lib/src/main/cpp/jfrMetadata.h | 6 + ddprof-lib/src/main/cpp/livenessTracker.cpp | 1878 ++++- ddprof-lib/src/main/cpp/livenessTracker.h | 1372 +++- ddprof-lib/src/main/cpp/objectSampler.cpp | 22 +- ddprof-lib/src/main/cpp/os.h | 1 + ddprof-lib/src/main/cpp/os_linux.cpp | 68 + ddprof-lib/src/main/cpp/os_macos.cpp | 4 + ddprof-lib/src/main/cpp/painBudget.h | 92 + ddprof-lib/src/main/cpp/profiler.cpp | 189 + ddprof-lib/src/main/cpp/profiler.h | 18 +- ddprof-lib/src/main/cpp/rcDebugLevel.h | 65 + ddprof-lib/src/main/cpp/referenceChains.cpp | 7042 +++++++++++++++++ ddprof-lib/src/main/cpp/referenceChains.h | 3482 ++++++++ ddprof-lib/src/main/cpp/safeAccess.h | 2 +- ddprof-lib/src/main/cpp/stringDictionary.h | 9 + ddprof-lib/src/main/cpp/symbols_linux.cpp | 36 +- ddprof-lib/src/main/cpp/vmEntry.cpp | 13 +- .../com/datadoghq/profiler/JavaProfiler.java | 188 + 29 files changed, 15307 insertions(+), 80 deletions(-) create mode 100644 ddprof-lib/src/main/cpp/classTagAllocator.h create mode 100644 ddprof-lib/src/main/cpp/painBudget.h create mode 100644 ddprof-lib/src/main/cpp/rcDebugLevel.h create mode 100644 ddprof-lib/src/main/cpp/referenceChains.cpp create mode 100644 ddprof-lib/src/main/cpp/referenceChains.h diff --git a/ddprof-lib/src/main/cpp/arguments.cpp b/ddprof-lib/src/main/cpp/arguments.cpp index 93b562bca8..4ad910d4d9 100644 --- a/ddprof-lib/src/main/cpp/arguments.cpp +++ b/ddprof-lib/src/main/cpp/arguments.cpp @@ -18,6 +18,7 @@ #include "arguments.h" #include "vmEntry.h" +#include #include #include #include @@ -81,6 +82,26 @@ static const Multiplier UNIVERSAL[] = { // and keep the liveness track of 10% of the allocation // samples // generations - track surviving generations +// referencechains[=BOOL[:hops=N][:budget=N][:ttl=N][:framecap=N][:pausetarget=N][:painbudget=N][:firstpassbudget=N]] +// - (PROF-15341, off by default) tag/BFS-walk live-heap +// samples' referrer chains back toward a GC root. +// pausetarget=N (ms) is the pause-time-SLO ceiling +// ReferenceChainTracker::updatePacing() adapts the +// effective budget/cadence toward (pause-time pacing +// controller). painbudget=N (percent) bounds how much +// wall-clock time a *restarted* search (one begun +// after a prior search already completed/abandoned) +// may spend on average - see PainBudget (painBudget.h). +// firstpassbudget=N overrides just the search's +// one-shot, root-seeded first pass's edge budget +// (default 0 - auto-scales from budget=N instead, +// see ReferenceChainTracker::AUTO_FIRST_PASS_BUDGET_* +// in referenceChains.h) since that pass alone +// decides which GC roots ever enter the frontier at +// all, unlike every later pass's cheap, incremental +// per-node expansion. +// Sub-options are placeholders pending future tuning; +// see doc/architecture/LiveHeapReferenceChains*.md // lightweight[=BOOL] - enable lightweight profiling - events without // stacktraces (default: true) // remotesym[=BOOL] - enable remote symbolication for native frames @@ -444,6 +465,98 @@ Error Arguments::parse(const char *args) { _nativesocket = true; } + CASE("referencechains") + { + // Sub-options are colon-delimited key=value pairs after the boolean, + // e.g. "referencechains=true:hops=64:budget=2000". Parsed manually + // (not via strtok) because the outer arg loop above is itself mid + // strtok(..., ",") over the same buffer - a nested strtok call would + // clobber its saved state. + char *config = value ? strchr(value, ':') : nullptr; + if (config) { + *(config++) = 0; + } + if (value != NULL) { + switch (value[0]) { + case 'n': // no + case 'f': // false + case '0': // 0 + _reference_chains = false; + break; + default: + _reference_chains = true; + } + } else { + _reference_chains = true; + } + char *cursor = config; + while (cursor != NULL) { + char *next = strchr(cursor, ':'); + if (next) { + *(next++) = 0; + } + char *eq = strchr(cursor, '='); + if (eq) { + *(eq++) = 0; + // Floor every sub-option at the parse boundary rather than + // trusting a downstream cast/clamp to make an operator-supplied + // negative value safe: a negative hops value in particular gets + // compared as `depth >= (u32)ctx->hop_cap` (referenceChains.cpp), + // so an unclamped negative wraps to ~4e9 and silently disables + // the hop cap entirely - the opposite of the flag's intent, and + // it removes the one guard that otherwise bounds how long a + // single reference chain (and therefore its + // datadog.ReferenceChain JFR event) can grow. A negative budget + // similarly collapses ReferenceChainTracker::_effective_budget + // to 0 (updatePacing()'s own PID-clamp logic), which truncates + // every pass immediately and leaves the search RUNNING + // (re-walking the whole graph each cadence) until TTL instead of + // making progress. A negative framecap is handed straight to + // FrontierTable's constructor, which floors it to a + // zero-capacity table (that class's own std::max(max_cap, 0)), + // silently disabling tracking rather than erroring. ttl/ + // pausetarget/painbudget already have incidental downstream + // clamps (runPass()'s `_ttl_ms > 0` gate, this class's own + // PidController/PainBudget std::max(..., 0) calls) but are + // floored here too so every sub-option's validation lives at one + // boundary instead of being split between here and several + // unrelated call sites. hops/budget/framecap are also ceiling- + // clamped (MAX_REFERENCE_CHAINS_HOP_CAP/_BUDGET/_FRONTIER_CAP, + // arguments.h) for the same reason painbudget/firstpassbudget + // are below: an unbounded operator-supplied value would otherwise + // flow straight into a loop bound or FrontierTable's allocation. + if (strcasecmp(cursor, "hops") == 0) { + _reference_chains_hop_cap = + std::min(std::max(atoi(eq), 1), MAX_REFERENCE_CHAINS_HOP_CAP); + _reference_chains_tuned_mask |= REF_CHAINS_TUNED_HOP_CAP; + } else if (strcasecmp(cursor, "budget") == 0) { + _reference_chains_budget = + std::min(std::max(atoi(eq), 1), MAX_REFERENCE_CHAINS_BUDGET); + _reference_chains_tuned_mask |= REF_CHAINS_TUNED_BUDGET; + } else if (strcasecmp(cursor, "ttl") == 0) { + _reference_chains_ttl_ms = std::max(atol(eq), 0L); + _reference_chains_tuned_mask |= REF_CHAINS_TUNED_TTL; + } else if (strcasecmp(cursor, "framecap") == 0) { + _reference_chains_frontier_cap = std::min( + std::max(atoi(eq), 1), MAX_REFERENCE_CHAINS_FRONTIER_CAP); + _reference_chains_tuned_mask |= REF_CHAINS_TUNED_FRONTIER_CAP; + } else if (strcasecmp(cursor, "pausetarget") == 0) { + _reference_chains_pause_target_ms = std::max(atol(eq), 0L); + _reference_chains_tuned_mask |= REF_CHAINS_TUNED_PAUSE_TARGET; + } else if (strcasecmp(cursor, "painbudget") == 0) { + _reference_chains_pain_budget_percent = + std::min(std::max(atoi(eq), 0), 100); + _reference_chains_tuned_mask |= REF_CHAINS_TUNED_PAIN_BUDGET; + } else if (strcasecmp(cursor, "firstpassbudget") == 0) { + _reference_chains_first_pass_budget = std::min( + std::max(atoi(eq), 0), MAX_REFERENCE_CHAINS_FIRST_PASS_BUDGET); + _reference_chains_tuned_mask |= REF_CHAINS_TUNED_FIRST_PASS_BUDGET; + } + } + cursor = next; + } + } + DEFAULT() if (_unknown_arg == NULL) _unknown_arg = arg; diff --git a/ddprof-lib/src/main/cpp/arguments.h b/ddprof-lib/src/main/cpp/arguments.h index 08bd90e572..70a6ad97ba 100644 --- a/ddprof-lib/src/main/cpp/arguments.h +++ b/ddprof-lib/src/main/cpp/arguments.h @@ -23,12 +23,124 @@ #include #include +#include "arch.h" + const long DEFAULT_CPU_INTERVAL = 10 * 1000 * 1000; // 10 ms const long DEFAULT_WALL_INTERVAL = 50 * 1000 * 1000; // 50 ms const long DEFAULT_ALLOC_INTERVAL = 524287; // 512 KiB const int DEFAULT_WALL_THREADS_PER_TICK = 16; const int DEFAULT_JSTACKDEPTH = 2048; +// Every constant below is a provisional default pending empirical +// tuning (see doc/architecture/LiveHeapReferenceChains-ImplementationPlan.md) +// - none of these values are backed by a benchmark run against this +// codebase. Each is chosen conservatively from cited precedent or from the +// shape of an existing, already-tuned subsystem, per the rationale below; +// a future JMH/async-profiler benchmark matrix (see +// doc/architecture/LiveHeapReferenceChains-BenchmarkPlan.md) is the intended +// path to replacing them with measured values. +// +// Hop cap: mirrors HotSpot's own JFR leak-profiler chain cap (~200 hops, +// split 100/100 from leaf and from root), cited in +// doc/architecture/LiveHeapReferenceChains.md's "Approach B" section - the +// closest real-world precedent for "how many hops does a referrer-type +// chain typically need" that this codebase can cite without measuring it +// itself. +// Sub-option bitmask for the auto-tuner: tracks which referencechains +// sub-options were explicitly set by the operator, so the auto-tuner +// only overrides defaults that weren't. +constexpr u8 REF_CHAINS_TUNED_HOP_CAP = 1 << 0; +constexpr u8 REF_CHAINS_TUNED_BUDGET = 1 << 1; +constexpr u8 REF_CHAINS_TUNED_TTL = 1 << 2; +constexpr u8 REF_CHAINS_TUNED_FRONTIER_CAP = 1 << 3; +constexpr u8 REF_CHAINS_TUNED_PAUSE_TARGET = 1 << 4; +constexpr u8 REF_CHAINS_TUNED_PAIN_BUDGET = 1 << 5; +constexpr u8 REF_CHAINS_TUNED_FIRST_PASS_BUDGET = 1 << 6; + +const int DEFAULT_REFERENCE_CHAINS_HOP_CAP = 200; +// Per-pass edge budget: no cited precedent gives a number for this (JFR's +// leak profiler does not bound itself by a per-pass edge count - it runs to +// completion inside one already-scheduled GC pause). Chosen as a round, +// conservative middle value intended to keep a single FollowReferences- +// triggered safepoint short without so small a budget that a search needs +// an impractical number of passes to make progress. A future benchmark pass +// should measure per-pass wall-clock pause distribution at this value and adjust. +const int DEFAULT_REFERENCE_CHAINS_BUDGET = 1000; // edges expanded per BFS pass +// Per-search TTL: a conservative round number (one minute) chosen so a +// slow-moving or stalled search is bounded to a human-noticeable but not +// excessive lifetime, in the absence of any measured "passes needed to +// reach a target sample at various depths" data (a future benchmark's stated goal). +const long DEFAULT_REFERENCE_CHAINS_TTL_MS = 60000; // per-search wall-clock TTL +// Frontier-size cap: sized relative to LivenessTracker's own tuned ceiling +// (MAX_TRACKING_TABLE_SIZE = 262144, livenessTracker.h) rather than derived +// from any BFS-specific measurement - the design doc explicitly flags that +// LivenessTracker's allocation-sample-rate sizing formula does not transfer +// to a graph-search frontier (Open Question 2), so this only borrows the +// same order of magnitude, quartered as a conservative starting point since +// a FrontierEntry is smaller than a TrackingEntry but per-hop fan-out could +// still be large. Not a scaled/derived value - just a conservative guess +// pending a future frontier-table peak-occupancy measurement. +const int DEFAULT_REFERENCE_CHAINS_FRONTIER_CAP = 65536; // max live frontier entries per search +// Pause-time-SLO ceiling (pause-time pacing controller, doc/architecture/ +// LiveHeapReferenceChains-RemainingWorkPlan.md): target ceiling, per pass, on +// wall-clock time spent inside the safepoint-triggering +// FollowReferences/GetObjectsWithTags call +// (ReferenceChainTracker::updatePacing(), referenceChains.cpp). Like every +// other constant in this block this is a round, provisional default with no +// benchmark behind it - picking the real number is explicitly a future +// measurement question (design doc's Open Question 2), not a value to guess +// here; this only exists so the feedback loop this ceiling drives has +// something to target end-to-end before that measurement happens. +const long DEFAULT_REFERENCE_CHAINS_PAUSE_TARGET_MS = 50; // ms per pass +// Pain budget refill rate (ReferenceChainTracker::PainBudget, painBudget.h): +// the fraction of wall-clock time a *restarted* search is allowed to spend +// inside FollowReferences/GetObjectsWithTags safepoints, on average, before a +// later restart must wait for the debt from the previous search's cost to +// drain. Expressed as an integer percent (1 = 1%) for readability - see +// PainBudget's own header comment for why this single ratio needs no +// benchmark-derived tuning the way the per-pass constants above do, only a +// choice of how much background cost is acceptable. Round, provisional +// default like every other constant in this block. +const int DEFAULT_REFERENCE_CHAINS_PAIN_BUDGET_PERCENT = 1; +// First-pass edge budget override: the search's one-and-only root-seeded +// FollowReferences(0, nullptr, nullptr, ...) call (ReferenceChainTracker::runPass()'s +// !_search_started branch) enumerates every GC root in one JVMTI-controlled +// traversal order and stops admitting once this budget is spent - any root +// FollowReferences had not yet reached is excluded from the frontier for the +// rest of that search (every later pass only expands forward from already- +// admitted frontier entries, see expandFrontier()'s own comment). Unlike +// DEFAULT_REFERENCE_CHAINS_BUDGET, which bounds every pass including the many +// cheap, per-node expansion passes that follow, this only ever spends once +// per search, so a much larger one-time ceiling is affordable. 0 (the +// default) means "no override - use the same budget as every other pass", +// preserving prior behavior for anyone not setting this explicitly. +const int DEFAULT_REFERENCE_CHAINS_FIRST_PASS_BUDGET = 0; +// Upper clamp for an explicit firstpassbudget override: like painbudget just +// above, firstpassbudget was previously only floored at 0 with no ceiling. +// Unlike painbudget (a percentage, naturally bounded at 100), this is a raw +// edge count, so the ceiling is expressed relative to +// DEFAULT_REFERENCE_CHAINS_BUDGET (the per-pass budget every later pass is +// bounded by) rather than as its own standalone guess: a generous but finite +// multiple still lets the one-time root pass be far larger than a normal +// pass (its intended purpose) while keeping an operator from disabling the +// safepoint-pause-bounding mechanism entirely for that first FollowReferences +// call. +const int MAX_REFERENCE_CHAINS_FIRST_PASS_BUDGET = + DEFAULT_REFERENCE_CHAINS_BUDGET * 1000; +// Upper clamps for hops/budget/framecap: like MAX_REFERENCE_CHAINS_FIRST_PASS_BUDGET +// just above, these were previously only floored at 1 with no ceiling, so an +// operator typo (an extra digit) or a mistaken value flows straight into a +// loop bound (hops), a per-pass edge count (budget), or FrontierTable's +// capacity (framecap) unchecked. Same generous-but-finite-multiple-of-the- +// default approach as the first-pass budget clamp: large enough that no +// legitimate configuration should ever hit the ceiling, small enough to +// still fail a badly mistyped value safely instead of feeding it straight +// into an allocation or loop bound. +const int MAX_REFERENCE_CHAINS_HOP_CAP = DEFAULT_REFERENCE_CHAINS_HOP_CAP * 1000; +const int MAX_REFERENCE_CHAINS_BUDGET = DEFAULT_REFERENCE_CHAINS_BUDGET * 1000; +const int MAX_REFERENCE_CHAINS_FRONTIER_CAP = + DEFAULT_REFERENCE_CHAINS_FRONTIER_CAP * 1000; + const char *const EVENT_NOOP = "noop"; const char *const EVENT_CPU = "cpu"; const char *const EVENT_ALLOC = "alloc"; @@ -177,6 +289,25 @@ class Arguments { double _live_samples_ratio; bool _record_heap_usage; bool _gc_generations; + // Reference-chain tracking (PROF-15341 - see + // doc/architecture/LiveHeapReferenceChains-ImplementationPlan.md and + // -RemainingWorkPlan.md). Read by ReferenceChainTracker::start() + // (referenceChains.cpp) to size the frontier table and seed the per-search + // hop/budget/TTL tunables and the pause-time-SLO ceiling that + // updatePacing() adapts the effective budget/cadence toward. + bool _reference_chains; + int _reference_chains_hop_cap; + int _reference_chains_budget; + long _reference_chains_ttl_ms; + int _reference_chains_frontier_cap; + long _reference_chains_pause_target_ms; + int _reference_chains_pain_budget_percent; + int _reference_chains_first_pass_budget; + // Bitmask of REF_CHAINS_TUNED_*: which sub-options were explicitly + // set by the operator, so the auto-tuner knows which defaults it may + // override. 0 = all defaults, none explicitly set. + u8 _reference_chains_tuned_mask; + // Explicit opt-in for the legacy whole-graph JVMTI FollowReferences walk. long _nativemem; int _jstackdepth; int _safe_mode; @@ -219,6 +350,15 @@ class Arguments { _live_samples_ratio(0.1), // default to liveness-tracking 10% of the allocation samples _record_heap_usage(false), _gc_generations(false), + _reference_chains(false), + _reference_chains_hop_cap(DEFAULT_REFERENCE_CHAINS_HOP_CAP), + _reference_chains_budget(DEFAULT_REFERENCE_CHAINS_BUDGET), + _reference_chains_ttl_ms(DEFAULT_REFERENCE_CHAINS_TTL_MS), + _reference_chains_frontier_cap(DEFAULT_REFERENCE_CHAINS_FRONTIER_CAP), + _reference_chains_pause_target_ms(DEFAULT_REFERENCE_CHAINS_PAUSE_TARGET_MS), + _reference_chains_pain_budget_percent(DEFAULT_REFERENCE_CHAINS_PAIN_BUDGET_PERCENT), + _reference_chains_first_pass_budget(DEFAULT_REFERENCE_CHAINS_FIRST_PASS_BUDGET), + _reference_chains_tuned_mask(0), _nativemem(-1), _jstackdepth(DEFAULT_JSTACKDEPTH), _safe_mode(0), diff --git a/ddprof-lib/src/main/cpp/callTraceHashTable.h b/ddprof-lib/src/main/cpp/callTraceHashTable.h index bc2e2486ca..d36cde4c4f 100644 --- a/ddprof-lib/src/main/cpp/callTraceHashTable.h +++ b/ddprof-lib/src/main/cpp/callTraceHashTable.h @@ -104,7 +104,7 @@ class CallTraceHashTable { // - ACQUIRE loads in collect(), put(), and putWithExistingId() // Required for correct visibility on weakly-ordered architectures (aarch64). LongHashTable* _table; - + volatile u64 _overflow; u64 calcHash(int num_frames, ASGCT_CallFrame *frames, bool truncated); diff --git a/ddprof-lib/src/main/cpp/callTraceStorage.cpp b/ddprof-lib/src/main/cpp/callTraceStorage.cpp index e7a9d9f4a5..7bde3602e3 100644 --- a/ddprof-lib/src/main/cpp/callTraceStorage.cpp +++ b/ddprof-lib/src/main/cpp/callTraceStorage.cpp @@ -4,6 +4,7 @@ * SPDX-License-Identifier: Apache-2.0 */ +#include #include "callTraceStorage.h" #include "counters.h" #include "log.h" diff --git a/ddprof-lib/src/main/cpp/classTagAllocator.h b/ddprof-lib/src/main/cpp/classTagAllocator.h new file mode 100644 index 0000000000..90e60652bc --- /dev/null +++ b/ddprof-lib/src/main/cpp/classTagAllocator.h @@ -0,0 +1,68 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +#ifndef _CLASS_TAG_ALLOCATOR_H +#define _CLASS_TAG_ALLOCATOR_H + +#include "arch.h" +#include + +// Process-wide, negative JVMTI class-object tag allocator, shared by +// ReferenceChainTracker (which tags every loaded class's own jclass object +// via SetTag - see resolveLoadedClasses(), referenceChains.cpp) and +// LivenessTracker (which needs a stable per-class identifier independent of +// Profiler::classMap()'s dictionary id - see KlassPopulationEntry:: +// stable_class_tag's own comment, livenessTracker.h, for why: that +// dictionary can be compacted/regenerated, silently reassigning the same +// class a different id at different points in the process's life, breaking +// any attempt to correlate a klass_id LivenessTracker reports as growing +// against ReferenceChainTracker::FrontierEntry::referrer_klass values +// recorded at a different time). +// +// A single shared counter, not one independently owned by each subsystem, +// for two reasons, both load-bearing: +// 1. Two independent counters could otherwise hand out the SAME numeric +// value to TWO DIFFERENT classes (one minted by each subsystem for a +// class the other has not seen yet), making any cross-subsystem +// comparison meaningless. +// 2. Class tags must stay strictly NEGATIVE: +// ReferenceChainTracker::heapReferenceCallback() (referenceChains.cpp) +// uses `*tag_ptr < 0` to distinguish "this heap-walk-visited object is a +// pre-tagged class object" from an ordinary admitted instance (always +// tagged with a positive value via nextTag()). A class tagged by a +// counter that does not preserve this sign convention would be +// misidentified as an ordinary object and incorrectly admitted into the +// frontier table - a real correctness bug, not just a matching +// inconvenience. +// +// Deliberately a plain header-only function (Meyer's-singleton pattern, +// exactly like LivenessTracker::instance()/ReferenceChainTracker:: +// instance()'s own lazy-static singletons) rather than a member of either +// singleton class: ReferenceChainTracker already depends on LivenessTracker +// (referenceChains.cpp includes livenessTracker.h and calls into it), so +// putting this counter inside either one and having the other call into it +// would introduce a circular dependency between the two headers. +namespace ClassTagAllocator { + +inline volatile jlong &magnitude() { + static volatile jlong m = 1; + return m; +} + +// Hands out a fresh negative class tag - see this file's own header comment +// for why negative, and why this must be the only place in the process that +// mints one. +inline jlong next() { return -atomicIncRelaxed(magnitude(), (jlong)1); } + +// Test-only: resets the shared counter back to its starting value. Without +// this, gtest cases that assert on exact tag values (e.g. "the first class +// tagged gets -1") would see values keep climbing across every TEST_F in the +// same gtest binary, since this counter is genuinely process-wide (shared +// with LivenessTracker) rather than per-ReferenceChainTracker-instance. +inline void resetForTest() { magnitude() = 1; } + +} // namespace ClassTagAllocator + +#endif diff --git a/ddprof-lib/src/main/cpp/counters.h b/ddprof-lib/src/main/cpp/counters.h index b81819c222..d90d7e8faf 100644 --- a/ddprof-lib/src/main/cpp/counters.h +++ b/ddprof-lib/src/main/cpp/counters.h @@ -162,6 +162,45 @@ * signal for spotting a recurrence. */ \ X(METADATA_TREE_NULL_CHILD, "metadata_tree_null_child") \ X(METADATA_TREE_DEPTH_EXCEEDED, "metadata_tree_depth_exceeded") \ + /* A resolved datadog.ReferenceChain could not be cached in \ + * ReferenceChainTracker::_resolved_chains (referenceChains.h): a brand-new \ + * leak-candidate klass arrived with the cache already at \ + * MAX_RESOLVED_CHAINS, so its chain is dropped rather than evicting some \ + * other still-live sample's chain. See that constant's own comment. */ \ + X(REFERENCE_CHAIN_EVENTS_DROPPED, "reference_chain_events_dropped") \ + /* ReferenceChainTracker::releaseSearchTags() (referenceChains.cpp) failed \ + * to call GetObjectsWithTags() for at least one batch - the search's tag \ + * release is retried on a later call rather than proceeding, but this \ + * counts how often that retry path is taken. */ \ + X(REFERENCE_CHAIN_TAG_RELEASE_FAILED, "reference_chain_tag_release_failed") \ + /* Profiler::writeReferenceChain() (profiler.cpp) could not acquire a \ + * sample-record lock within its bounded retry budget and dropped the \ + * already-dequeued datadog.ReferenceChain event for this dump - not \ + * permanently lost, since ReferenceChainTracker::_resolved_chains (see \ + * REFERENCE_CHAIN_EVENTS_DROPPED above) keeps the resolved chain cached \ + * and re-emits it on a later dump while the leak candidate is still \ + * live. */ \ + X(REFERENCE_CHAIN_WRITE_DROPPED, "reference_chain_write_dropped") \ + /* FrontierTable's own calloc/realloc-backed storage (referenceChains.cpp) - \ + * outside NMT's visibility since it bypasses os::malloc, so this is the only \ + * way to attribute its native RSS contribution. */ \ + X(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, "reference_chain_frontier_table_bytes") \ + X(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, "reference_chain_frontier_table_capacity") \ + X(REFERENCE_CHAIN_CANDIDATE_COUNT, "reference_chain_candidate_count") \ + X(REFERENCE_CHAIN_CANDIDATES_FOUND, "reference_chain_candidates_found") \ + /* admitStaticFieldRoots() per-class non-static quota: non-STATIC_FIELD \ + * edges (CONSTANT_POOL, INTERFACE, SUPERCLASS, CLASS_LOADER, ...) that \ + * were dropped because the class already hit \ + * STATIC_FIELD_SWEEP_NON_STATIC_CAP_PER_CLASS. Total drops across all \ + * classes/laps — compare against kind_counts (k9 total) to gauge how \ + * much CP pressure the quota is absorbing. */ \ + X(REFERENCE_CHAIN_STATIC_SWEEP_NON_STATIC_DROPPED, "reference_chain_static_sweep_non_static_dropped") \ + /* Incremented once per class that hit the non-static cap at least once \ + * in a lap (on the first drop for that class). Distinguishes "a few fat \ + * outlier classes dropping many edges" from "systematic drops across \ + * almost all classes" — if this tracks the total class count per lap, \ + * the cap is too low; if it stays near zero, the cap is fine. */ \ + X(REFERENCE_CHAIN_STATIC_SWEEP_CLASSES_CAPPED, "reference_chain_static_sweep_classes_capped") \ DD_COUNTER_TABLE_FAULT_INJECTION(X) \ DD_COUNTER_TABLE_FI_DEBUG(X) \ DD_COUNTER_TABLE_DEBUG(X) diff --git a/ddprof-lib/src/main/cpp/event.h b/ddprof-lib/src/main/cpp/event.h index 67ff97b381..aafa3cc2c0 100644 --- a/ddprof-lib/src/main/cpp/event.h +++ b/ddprof-lib/src/main/cpp/event.h @@ -24,6 +24,8 @@ #include #include #include +#include +#include using namespace std; #define MAX_STRING_LEN 8191 @@ -87,9 +89,72 @@ class ObjectLivenessEvent : public Event { u64 _skipped; u64 _start_time; u64 _age; + int64_t leak_tag; // 0 = untagged; leak tag from LivenessTracker pool Context _ctx; }; +// Reporting surface for ReferenceChainTracker's bounded +// BFS (referenceChains.h/.cpp). `_target_tag` is the FrontierTable tag the +// chain was reconstructed for (FrontierTable::reconstructChain()); `_chain` +// holds the referrer-klass StringDictionary ids it returns, in the same +// leaf(target)-to-root order. `_depth` is the target entry's own +// FrontierEntry::depth (hop count from the search's root-side seed). +// `_root_kind` is the jvmtiHeapReferenceKind of whichever edge first +// admitted this chain into the frontier (FrontierEntry::root_kind, via +// FrontierTable::reconstructChain()'s out_root_kind) - labels *why* the +// chain is reachable at all (JNI global, thread stack, static field, ...), +// written out as a string (Recording::recordReferenceChain(), +// flightRecorder.cpp) rather than a synthetic node in `_chain` itself, +// since that array is a T_CLASS cpool array with no room for a +// non-class placeholder. +// Byte cap for one hop's retention-edge label in ReferenceChainEvent::_edges +// (fillHopEdgeLabels truncates to this; recordReferenceChain() reserves +// against it) - a shared constant so the collector and the writer cannot +// drift apart on the worst-case event size. +static constexpr size_t MAX_REFERENCE_CHAIN_EDGE_LABEL = 96; + +class ReferenceChainEvent : public Event { +public: + u64 _start_time; + u64 _target_tag; + u32 _depth; + u8 _root_kind; + std::vector _chain; + // Retention-edge label per hop, ALIGNED with _chain's leaf-to-root order + // (_edges[i] = the edge by which _chain[i] is retained - the field name of + // its parent hop for FIELD/STATIC_FIELD edges, the edge-kind label + // otherwise; ReferenceChainTracker::fillHopEdgeLabels). Empty for events + // built before the edge-label change or when label resolution is + // unavailable (partial mock environments). + std::vector _edges; + + ReferenceChainEvent() + : Event(), _start_time(0), _target_tag(0), _depth(0), _root_kind(0) {} +}; + +// Search-level abandonment signal (design doc's Termination section: +// "explicit reporting of abandoned searches ... no silent truncation"). +// Unlike ReferenceChainEvent this does not report any one object's chain - +// it reports why ReferenceChainTracker's current search stopped before +// every frontier entry could be resolved, using the same counters +// runPass()/expandFrontier() already maintain (referenceChains.h/.cpp). +class ReferenceChainAbandonedEvent : public Event { +public: + u64 _start_time; + u8 _reason; // SearchAbandonReason (referenceChains.h) + u32 _passes_run; + u32 _frontier_size; + int _hop_cap; + int _budget; + long _ttl_ms; + u64 _elapsed_ns; + + ReferenceChainAbandonedEvent() + : Event(), _start_time(0), _reason(0), _passes_run(0), + _frontier_size(0), _hop_cap(0), _budget(0), _ttl_ms(0), + _elapsed_ns(0) {} +}; + class MallocEvent : public Event { public: u64 _start_time; diff --git a/ddprof-lib/src/main/cpp/flightRecorder.cpp b/ddprof-lib/src/main/cpp/flightRecorder.cpp index 48ca267bb6..c72348172e 100644 --- a/ddprof-lib/src/main/cpp/flightRecorder.cpp +++ b/ddprof-lib/src/main/cpp/flightRecorder.cpp @@ -48,14 +48,18 @@ static const char *const SETTING_RING[] = {NULL, "kernel", "user", "any"}; static const char *const SETTING_CSTACK[] = {NULL, "no", "fp", "dwarf", "lbr"}; -// JVM spec SS4.7.3 caps a method's bytecode (code_length) at 65535 bytes (u2), -// so a well-formed LineNumberTable can never have more entries than that. -// Used to sanity-bound line_number_table_size before it drives the byte-count -// passed to SafeAccess::safeCopy(): if GetLineNumberTable() -// returns a corrupted pointer for a stale jmethodID (see the TOCTOU race -// documented in fillJavaMethodInfo below), the paired out-param size is just -// as likely to be corrupted, and an implausible size should be rejected -// before it is trusted to compute a byte range. +// JVM spec SS4.7.3 caps a method's bytecode (code_length) at 65535 bytes (u2). +// A LineNumberTable entry maps a bytecode offset to a source line, so a +// well-formed table can have at most one entry per bytecode offset -- the +// 65535 bound here is inherited indirectly through that one-entry-per-offset +// invariant, not a direct spec cap on entry count (the numeric equivalence +// with code_length's own u2 cap is coincidental). Used to sanity-bound +// line_number_table_size before it drives the byte-count passed to +// SafeAccess::safeCopy(): if GetLineNumberTable() returns a corrupted +// pointer for a stale jmethodID (see the TOCTOU race documented in +// fillJavaMethodInfo below), the paired out-param size is just as likely to +// be corrupted, and an implausible size should be rejected before it is +// trusted to compute a byte range. static const jint MAX_LINE_NUMBER_TABLE_ENTRIES = 65535; // Compute a non-negative event duration from TSC timestamps. Unsigned u64 @@ -1085,6 +1089,8 @@ void Recording::switchChunk(int fd) { _chunk_start = finishChunk(/*end_recording=*/true, /*do_cleanup=*/true); TEST_LOG("MethodMap: %zu methods after cleanup", _method_map.size()); + TEST_LOG("Recording::switchChunk copying [0, %lld) from _fd=%d to fd=%d", + (long long)_chunk_start, _fd, fd); _start_time = _stop_time; _start_ticks = _stop_ticks; @@ -1430,6 +1436,7 @@ void Recording::writeSettings(Buffer *buf, Arguments &args) { writeBoolSetting(buf, T_ALLOC, "enabled", args._record_allocations); writeBoolSetting(buf, T_HEAP_LIVE_OBJECT, "enabled", args._record_liveness); + writeBoolSetting(buf, T_REFERENCE_CHAIN, "enabled", args._reference_chains); writeBoolSetting(buf, T_MALLOC, "enabled", args._nativemem >= 0); if (args._nativemem >= 0) { writeIntSetting(buf, T_MALLOC, "nativemem", args._nativemem); @@ -2332,11 +2339,154 @@ void Recording::recordHeapLiveObject(Buffer *buf, int tid, u64 call_trace_id, ? ((event->_alloc._weight * event->_alloc._size) + event->_skipped) / event->_alloc._size : 0); + // leak_tag precedes the context snapshot to match the metadata field + // order (jfrMetadata.cpp: leakTag is declared between weight and spanId, + // before || contextAttributes) - a field written on the opposite side of + // writeContextSnapshot() than where the metadata declares it makes every + // parser read it from the first context-attribute byte. + buf->putVar64(event->leak_tag); writeContextSnapshot(buf, event->_ctx); writeEventSizePrefix(buf, start); flushIfNeeded(buf); } +// Maps a FrontierEntry::root_kind byte (a jvmtiHeapReferenceKind value, +// referenceChains.h/.cpp's heapReferenceCallback()) to a human-readable +// label for datadog.ReferenceChain's rootKind field - only the values +// heapReferenceCallback() can actually produce (a root reference's own +// jvmtiHeapReferenceKind, or JVMTI_HEAP_REFERENCE_STATIC_FIELD for the +// "referrer is a pre-tagged class" root-like case, see that method's own +// comment) have entries; anything else (including 0, root_kind's +// "not set" default) reports "unknown" rather than crashing on an +// out-of-range index. +// +// STACK_LOCAL (24) and JNI_LOCAL (25) are labeled "first_observed_via:..." +// rather than plain "stack_local"/"jni_local" (design doc's "Honest labeling +// in output", point 2): both are evidence this object was reachable from a +// live frame/local handle at the moment a pass observed it, not a durable +// retention reason - the frame can pop or the handle can be freed the +// instant the pass ends, so "rooted by" would overstate what is actually +// known. Every other kind here is durable enough for the plain "rooted by" +// framing this field's name already implies. +static const char *rootKindName(u8 root_kind) { + switch (root_kind) { + case 8: + return "static_field"; + case 21: + return "jni_global"; + case 22: + return "system_class"; + case 23: + return "monitor"; + case 24: + return "first_observed_via:stack_local"; + case 25: + return "first_observed_via:jni_local"; + case 26: + return "thread"; + case 27: + return "other"; + default: + return "unknown"; + } +} + +void Recording::recordReferenceChain(Buffer *buf, ReferenceChainEvent *event) { + // event->_chain's length is bounded only by FrontierTable::maxCapacity() + // (tens of thousands of entries, referenceChains.h) - NOT by + // MAX_JFR_EVENT_SIZE, so this event cannot use writeEventSizePrefix()'s + // single-byte size field (its assert(size < MAX_JFR_EVENT_SIZE) is + // compiled out in release builds, making an oversize chain a silent + // corrupt size byte rather than a caught bug) nor rely on the trailing + // flushIfNeeded(buf) every fixed-size event above uses (that only flushes + // *after* already writing past the buffer). Truncate to + // MAX_REFERENCE_CHAIN_EVENT_HOPS (that constant's own comment) and + // reserve room for the truncated worst case up front instead. + u32 chain_size = (u32)event->_chain.size(); + u32 emitted_size = chain_size < (u32)MAX_REFERENCE_CHAIN_EVENT_HOPS + ? chain_size + : (u32)MAX_REFERENCE_CHAIN_EVENT_HOPS; + + // rootKindName() never returns a string longer than + // "first_observed_via:stack_local" (31 bytes) - reserve a fixed, generous + // 32 bytes for its putUtf8() length prefix + payload rather than computing + // strlen() up front. + const char *root_kind_name = rootKindName(event->_root_kind); + // Per-hop edge labels: same emitted_size as the chain, each bounded by + // MAX_REFERENCE_CHAIN_EDGE_LABEL (event.h) + putUtf8's 1-byte encoding + // tag and up to 5-byte varint length prefix. + const u32 edge_count = + event->_edges.size() == emitted_size ? emitted_size : 0; + const char *edge_labels[MAX_REFERENCE_CHAIN_EVENT_HOPS]; + for (u32 i = 0; i < edge_count; i++) { + edge_labels[i] = event->_edges[i].c_str(); + } + flushIfNeeded( + buf, RECORDING_BUFFER_LIMIT - + (MAX_VAR32_LENGTH /* multi-byte size prefix, below */ + + 3 * MAX_VAR64_LENGTH /* type id, start_time, target_tag */ + + 3 * MAX_VAR32_LENGTH /* depth, totalHops, chain count */ + + 32 /* rootKind string */ + + (int)emitted_size * MAX_VAR32_LENGTH + + (edge_count > 0 + ? MAX_VAR32_LENGTH /* edges count */ + + (int)edge_count * + (MAX_REFERENCE_CHAIN_EDGE_LABEL + 6) + : MAX_VAR32_LENGTH /* edges count, empty array */))); + + // Multi-byte size prefix (like writeDatadogSetting() above), not + // writeEventSizePrefix()'s single byte - this event's size can exceed + // MAX_JFR_EVENT_SIZE (255) once the chain is more than a few dozen hops. + int start = buf->skip(MAX_VAR32_LENGTH); + buf->putVar64(T_REFERENCE_CHAIN); + buf->putVar64(event->_start_time); + buf->putVar64(event->_target_tag); + buf->putVar32(event->_depth); + buf->putUtf8(root_kind_name); + // Original (pre-truncation) chain length, so a consumer can tell a full + // chain (totalHops == chain.length) from a silently truncated one - + // mirrors ReferenceChainAbandonedEvent's "no silent truncation" design. + buf->putVar32(chain_size); + // T_CLASS array field (F_CPOOL|F_ARRAY, jfrMetadata.cpp) - each entry is a + // StringDictionary class id, same encoding as a scalar objectClass field + // (e.g. recordAllocation() above), just repeated `count` times. + buf->putVar32(emitted_size); + for (u32 i = 0; i < emitted_size; i++) { + buf->putVar32(event->_chain[i]); + } + // Edges array, LAST so its metadata position (after "chain", jfrMetadata.cpp) + // matches the write order - the leakTag field-order invariant + // (find-leaktag-jfr-field-misalignment) generalized. Empty count when the + // chain was truncated deeper than the collected edges (or the event predates + // label collection) rather than emitting a misaligned array. + buf->putVar32(edge_count); + for (u32 i = 0; i < edge_count; i++) { + buf->putUtf8(edge_labels[i]); + } + buf->putVar32(start, (u32)(buf->offset() - start)); + flushIfNeeded(buf); +} + +void Recording::recordReferenceChainAbandoned( + Buffer *buf, ReferenceChainAbandonedEvent *event) { + int start = buf->skip(1); + buf->putVar64(T_REFERENCE_CHAIN_ABANDONED); + buf->putVar64(event->_start_time); + // SearchAbandonReason (referenceChains.h) - kept as a small fixed table + // here rather than a T_XXX enum type, mirroring NativeSocketEvent's + // _operation -> kOpNames string mapping above. + static const char *const kReasons[] = {"none", "frontier_cap", "ttl", "canary_stuck"}; + buf->putUtf8(event->_reason < 4 ? kReasons[event->_reason] : "unknown"); + buf->putVar32(event->_passes_run); + buf->putVar32(event->_frontier_size); + buf->putVar32(event->_hop_cap); + buf->putVar32(event->_budget); + buf->putVar64(event->_ttl_ms); + buf->putVar64(event->_elapsed_ns / 1000000); + writeEventSizePrefix(buf, start); + flushIfNeeded(buf); +} + void Recording::recordMonitorBlocked(Buffer *buf, int tid, u64 call_trace_id, LockEvent *event) { int start = buf->skip(1); @@ -2520,6 +2670,32 @@ void FlightRecorder::recordHeapUsage(int lock_index, long value, bool live) { } } +void FlightRecorder::recordReferenceChainAbandoned( + int lock_index, ReferenceChainAbandonedEvent *event) { + DEBUG_ASSERT_NOT_IN_SIGNAL(); + OptionalSharedLockGuard locker(&_rec_lock); + if (locker.ownsLock()) { + Recording* rec = _rec; + if (rec != nullptr) { + Buffer *buf = rec->buffer(lock_index); + rec->recordReferenceChainAbandoned(buf, event); + } + } +} + +void FlightRecorder::recordReferenceChain(int lock_index, + ReferenceChainEvent *event) { + DEBUG_ASSERT_NOT_IN_SIGNAL(); + OptionalSharedLockGuard locker(&_rec_lock); + if (locker.ownsLock()) { + Recording* rec = _rec; + if (rec != nullptr) { + Buffer *buf = rec->buffer(lock_index); + rec->recordReferenceChain(buf, event); + } + } +} + bool FlightRecorder::recordEvent(int lock_index, int tid, u64 call_trace_id, int event_type, Event *event) { OptionalSharedLockGuard locker(&_rec_lock); diff --git a/ddprof-lib/src/main/cpp/flightRecorder.h b/ddprof-lib/src/main/cpp/flightRecorder.h index 6efbc52b45..1a4936e3cd 100644 --- a/ddprof-lib/src/main/cpp/flightRecorder.h +++ b/ddprof-lib/src/main/cpp/flightRecorder.h @@ -45,6 +45,20 @@ const int JFR_EVENT_FLUSH_THRESHOLD = RECORDING_BUFFER_LIMIT; const int MAX_VAR64_LENGTH = 10; const int MAX_VAR32_LENGTH = 5; +// Chain length Recording::recordReferenceChain() (flightRecorder.cpp) will +// actually serialize per datadog.ReferenceChain event, independent of +// ReferenceChainTracker's own _hop_cap/frontier-table cap - the frontier +// table's own defensive walk bound (FrontierTable::reconstructChain(), +// referenceChains.h) is maxCapacity(), which can run into the tens of +// thousands of entries, and neither that cap nor _hop_cap is itself +// range-validated against a buffer-safe maximum (see arguments.cpp's own +// sub-option parsing). recordReferenceChain() truncates event->_chain to +// this many entries before writing, so its own worst-case size never +// depends on trusting either of those upstream caps to stay small - a chain +// longer than this is still truncated defense-in-depth even if a caller +// changes those caps later. +const int MAX_REFERENCE_CHAIN_EVENT_HOPS = 4096; + #ifndef CONCURRENCY_LEVEL const int CONCURRENCY_LEVEL = 16; #endif @@ -387,6 +401,9 @@ class Recording { NativeSocketEvent *event); void recordHeapLiveObject(Buffer *buf, int tid, u64 call_trace_id, ObjectLivenessEvent *event); + void recordReferenceChain(Buffer *buf, ReferenceChainEvent *event); + void recordReferenceChainAbandoned(Buffer *buf, + ReferenceChainAbandonedEvent *event); void recordMonitorBlocked(Buffer *buf, int tid, u64 call_trace_id, LockEvent *event); void recordThreadPark(Buffer *buf, int tid, u64 call_trace_id, @@ -541,6 +558,24 @@ class FlightRecorder { const char *value, const char *unit); void recordHeapUsage(int lock_index, long value, bool live); + + // Mirrors recordHeapUsage()'s shape exactly - ReferenceChainAbandonedEvent + // is not stack-sample-shaped (no tid/call_trace_id), same as HeapUsage. + // Called from Profiler::writeReferenceChainAbandoned() (profiler.cpp), + // wired from Profiler::dump() the same way LivenessTracker::flush() is. + void recordReferenceChainAbandoned(int lock_index, + ReferenceChainAbandonedEvent *event); + + // Mirrors recordReferenceChainAbandoned() above exactly, for + // ReferenceChainEvent instead. Called from Profiler::writeReferenceChain() + // (profiler.cpp), itself called from + // ReferenceChainTracker::pollWatchedTargets() (referenceChains.cpp) for + // each chain event discovered this poll cycle - unlike + // recordReferenceChainAbandoned() (only reached from dump()), chain events + // are produced continuously as candidates are discovered, not only at + // dump time, so this needs its own call site rather than piggybacking on + // dump()'s flush-on-dump pattern. + void recordReferenceChain(int lock_index, ReferenceChainEvent *event); }; #endif // _FLIGHTRECORDER_H diff --git a/ddprof-lib/src/main/cpp/javaApi.cpp b/ddprof-lib/src/main/cpp/javaApi.cpp index 5b007ad5e1..31aa66f840 100644 --- a/ddprof-lib/src/main/cpp/javaApi.cpp +++ b/ddprof-lib/src/main/cpp/javaApi.cpp @@ -1104,6 +1104,223 @@ Java_com_datadoghq_profiler_JavaProfiler_dumpContext(JNIEnv* env, jclass unused) TEST_LOG("===> Context: tid:%lu, spanId=%lu, rootSpanId=%lu", OS::threadId(), spanId, rootSpanId); } +// PROF-15341: LivenessTracker/ReferenceChainTracker test seams. Unlike +// testlog()/dumpContext() above (harmless no-ops in release, via TEST_LOG's +// own release-mode expansion to nothing), these mutate real tracker state +// (tagging objects, seeding population history) - shipping them into a +// release build would let a caller corrupt the actual leak-detection state, +// not just add a silent no-op. Guarded out entirely instead, so they only +// exist in the debug build ddprof-test's `testdebug` Gradle task loads +// (`-DDEBUG`, see ConfigurationPresets.kt's configureDebug()) - never in the +// `-DNDEBUG` release build. +#ifdef DEBUG +#include "livenessTracker.h" +#include "referenceChains.h" +#include + +extern "C" DLLEXPORT jboolean JNICALL +Java_com_datadoghq_profiler_JavaProfiler_setGcGenerationsEnabled0( + JNIEnv *env, jclass unused, jboolean enabled) { + LivenessTracker::instance()->setGcGenerationsForTest(enabled); + return JNI_TRUE; +} + +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_seedKlassPopulationSample0( + JNIEnv *env, jclass unused, jint klassId, jint count, jlong epoch) { + int slot; + bool created; + LivenessTracker::instance()->klassPopulationRecordForTest( + (u32)klassId, (u16)count, (u64)epoch, &slot, &created); +} + +// Seeds one per-(klass, tid) trend sample - see tidTrendRecordForTest()'s +// own comment (livenessTracker.h) for the synthetic-flag exemption and the +// real-tid requirement scenarios must honor. +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_seedTidTrendSample0( + JNIEnv *env, jclass unused, jint klassId, jint tid, jint count, + jlong epoch) { + LivenessTracker::instance()->tidTrendRecordForTest( + (u32)klassId, (jint)tid, (u32)count, (u64)epoch); +} + +// Wires a real, caller-chosen live object in as klassId's leak-candidate +// representative, so a test-seeded slope signal (seedKlassPopulationSample0 +// above) and a directly-tagged frontier root (tagAsReferenceChainRoot0 +// below) can be joined into one deterministic end-to-end run of +// pollWatchedTargets()'s bridging step - without either LivenessTracker's +// real allocation sampler or ReferenceChainTracker's root-seeded walk ever +// running. Takes its own weak global ref (klassPopulationSetRepresentativeForTest()'s +// own contract, livenessTracker.h) rather than aliasing any handle the +// caller manages. +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_setKlassPopulationRepresentativeForTest0( + JNIEnv *env, jclass unused, jint klassId, jobject representative) { + jweak rep = env->NewWeakGlobalRef(representative); + LivenessTracker::instance()->klassPopulationSetRepresentativeForTest( + env, (u32)klassId, rep); +} + +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_resetKlassPopulationForTest0( + JNIEnv *env, jclass unused) { + LivenessTracker::instance()->klassPopulationResetForTest(); +} + +extern "C" DLLEXPORT jintArray JNICALL +Java_com_datadoghq_profiler_JavaProfiler_selectLeakCandidateKlassIds0( + JNIEnv *env, jclass unused) { + KlassCandidate candidates[5]; + int n = LivenessTracker::instance()->selectLeakCandidates(candidates, 5); + jintArray result = env->NewIntArray(n); + if (result == nullptr || n == 0) { + return result; + } + jint ids[5]; + for (int i = 0; i < n; i++) { + ids[i] = (jint)candidates[i].klass_id; + } + env->SetIntArrayRegion(result, 0, n, ids); + return result; +} + +extern "C" DLLEXPORT jlong JNICALL +Java_com_datadoghq_profiler_JavaProfiler_tagAsReferenceChainRoot0( + JNIEnv *env, jclass unused, jobject target) { + jvmtiEnv *jvmti = VM::jvmti(); + if (jvmti == nullptr) { + return 0; + } + return ReferenceChainTracker::instance()->tagAsRootForTest(jvmti, env, + target); +} + +extern "C" DLLEXPORT jboolean JNICALL +Java_com_datadoghq_profiler_JavaProfiler_runReferenceChainPass0( + JNIEnv *env, jclass unused) { + jvmtiEnv *jvmti = VM::jvmti(); + if (jvmti == nullptr) { + return JNI_FALSE; + } + return ReferenceChainTracker::instance()->runPassSerialized(jvmti, env); +} + +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_pollReferenceChainTargets0( + JNIEnv *env, jclass unused) { + jvmtiEnv *jvmti = VM::jvmti(); + if (jvmti == nullptr) { + return; + } + ReferenceChainTracker::instance()->pollWatchedTargetsSerialized(jvmti, env); +} + +extern "C" DLLEXPORT jint JNICALL +Java_com_datadoghq_profiler_JavaProfiler_drainReferenceChainEventCount0( + JNIEnv *env, jclass unused) { + std::vector events; + ReferenceChainTracker::instance()->drainPendingChainEvents(&events); + return (jint)events.size(); +} + +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_resetReferenceChainSearchForTest0( + JNIEnv *env, jclass unused) { + jvmtiEnv *jvmti = VM::jvmti(); + ReferenceChainTracker::instance()->resetSearchStateForTest(jvmti, env); +} + +// Diagnostic-only: reads target's existing JVMTI tag (does NOT tag it - +// unlike tagAsReferenceChainRoot0 above, a target the real search has not +// reached yet must be left untagged) and reports its FIFO distance from the +// front of ReferenceChainTracker's pending-expansion queue. See +// ReferenceChainTracker::pendingExpandPositionForTest()'s own comment for +// the return-value contract. +extern "C" DLLEXPORT jlong JNICALL +Java_com_datadoghq_profiler_JavaProfiler_getReferenceChainPendingPositionForTest0( + JNIEnv *env, jclass unused, jobject target) { + jvmtiEnv *jvmti = VM::jvmti(); + if (jvmti == nullptr || target == nullptr) { + return -2; + } + jlong tag = 0; + jvmtiError err = jvmti->GetTag(target, &tag); + if (err != JVMTI_ERROR_NONE) { + return -2; + } + return (jlong)ReferenceChainTracker::instance()->pendingExpandPositionForTest( + tag); +} + +extern "C" DLLEXPORT jlong JNICALL +Java_com_datadoghq_profiler_JavaProfiler_getReferenceChainPendingSizeForTest0( + JNIEnv *env, jclass unused) { + return (jlong)ReferenceChainTracker::instance()->pendingExpandSizeForTest(); +} + +// Seeds one heap-floor-ring sample directly (LivenessTracker::secondsToOOM()'s +// input), bypassing the real GarbageCollectionFinish callback - lets a test +// build an arbitrary rising/flat heap-usage-over-time history without +// waiting on real GCs. timestampNs values are only ever compared against +// each other (secondsToOOM()'s own ringThirdsStats() deltas), never against +// a real wall clock, so a test may use any self-consistent, strictly +// increasing sequence. +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_heapFloorRecordForTest0( + JNIEnv *env, jclass unused, jlong usedBytes, jlong timestampNs) { + LivenessTracker::instance()->heapFloorRecordForTest((u64)usedBytes, + (u64)timestampNs); +} + +// Bypasses initialize_table()'s JNI-dependent HeapUsage::getMaxHeap() call so +// secondsToOOM() can be exercised against a test-chosen fake max heap size, +// independent of whatever -Xmx this JVM's own shared, no-forkEvery fork +// happens to run with. +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_setMaxHeapBytesForTest0( + JNIEnv *env, jclass unused, jlong maxHeapBytes) { + LivenessTracker::instance()->setMaxHeapBytesForTest((jlong)maxHeapBytes); +} + +// Temporarily disables onGC()'s own recordHeapFloorSample() call so a test +// can seed the heap-floor ring exclusively via heapFloorRecordForTest0() +// without a real GC interleaving a sample with a real OS::nanotime() +// timestamp and real heap usage, corrupting secondsToOOM()'s projection. +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_setHeapFloorRecordingForTest0( + JNIEnv *env, jclass unused, jboolean enabled) { + LivenessTracker::instance()->setHeapFloorRecordingForTest(enabled == JNI_TRUE); +} + +// Exposes ReferenceChainTracker::shouldRunPass() directly (see that seam's +// own comment, referenceChains.h) - unlike runReferenceChainPass0() above, +// which calls runPass() unconditionally, this reports whether the +// search-restart gate itself (canAffordNewSearch() -> hasLeakSignal()) would +// currently allow a fresh/terminal search to start. +extern "C" DLLEXPORT jboolean JNICALL +Java_com_datadoghq_profiler_JavaProfiler_shouldRunPassForTest0(JNIEnv *env, + jclass unused) { + return ReferenceChainTracker::instance()->shouldRunPassForTest( + OS::nanotime()) + ? JNI_TRUE + : JNI_FALSE; +} + +// Exposes ReferenceChainTracker::passesRun() directly - not itself DEBUG-gated on the native side +// (used by production JFR event fields too), but exposed here only for test use: lets a test note +// the current pass count before creating an object, then wait for that count to advance before +// trusting any match against it - the only way to be certain a match came from a pass whose own +// expandFrontier() (and therefore collectStaleExpandedEntriesForRotation()) ran strictly after the +// object existed, rather than from the same pass racing the object's creation. +extern "C" DLLEXPORT jint JNICALL +Java_com_datadoghq_profiler_JavaProfiler_referenceChainPassesRunForTest0( + JNIEnv *env, jclass unused) { + return (jint)ReferenceChainTracker::instance()->passesRun(); +} + +#endif // DEBUG + // ---- Test-only reads of the current thread's OTEP record ----------------------------------- // Each reads the current carrier's record directly via ProfiledThread::current(), with no // detach/attach (diagnostic-only, not on any signal-handler or hot write path). diff --git a/ddprof-lib/src/main/cpp/jfrMetadata.cpp b/ddprof-lib/src/main/cpp/jfrMetadata.cpp index 4064ba241d..51dd36dd92 100644 --- a/ddprof-lib/src/main/cpp/jfrMetadata.cpp +++ b/ddprof-lib/src/main/cpp/jfrMetadata.cpp @@ -201,10 +201,38 @@ void JfrMetadata::initialize( << field("age", T_LONG, "Age", F_UNSIGNED) << field("size", T_LONG, "Original Size", F_BYTES) << field("weight", T_FLOAT, "Sample weight") + << field("leakTag", T_LONG, "Leak Tag", F_UNSIGNED) << field("spanId", T_LONG, "Span ID") << field("localRootSpanId", T_LONG, "Local Root Span ID") || contextAttributes) + << (type("datadog.ReferenceChain", T_REFERENCE_CHAIN, + "Live Object Reference Chain") + << category("Datadog", "Profiling") + << field("startTime", T_LONG, "Start Time", F_TIME_TICKS) + << field("targetTag", T_LONG, "Frontier Tag", F_UNSIGNED) + << field("depth", T_INT, "Depth") + << field("rootKind", T_STRING, "GC Root Kind") + << field("totalHops", T_INT, + "Total Chain Length Before Truncation") + << field("chain", T_CLASS, "Referrer Chain (Leaf to Root)", + F_CPOOL | F_ARRAY) + << field("edges", T_STRING, + "Retention Edge per Hop (Leaf to Root)", F_ARRAY)) + + << (type("datadog.ReferenceChainAbandoned", + T_REFERENCE_CHAIN_ABANDONED, + "Live Object Reference Chain Search Abandoned") + << category("Datadog", "Profiling") + << field("startTime", T_LONG, "Start Time", F_TIME_TICKS) + << field("reason", T_STRING, "Abandonment Reason") + << field("passesRun", T_INT, "Passes Run") + << field("frontierSize", T_INT, "Frontier Size") + << field("hopCap", T_INT, "Hop Cap") + << field("budget", T_INT, "Per-Pass Budget") + << field("ttl", T_LONG, "Search TTL", F_DURATION_MILLIS) + << field("elapsed", T_LONG, "Elapsed Time", F_DURATION_MILLIS)) + << (type("datadog.Endpoint", T_ENDPOINT, "Endpoint") << category("Datadog") << field("startTime", T_LONG, "Start Time", F_TIME_TICKS) diff --git a/ddprof-lib/src/main/cpp/jfrMetadata.h b/ddprof-lib/src/main/cpp/jfrMetadata.h index 4d3f2a86e9..7a3fb72579 100644 --- a/ddprof-lib/src/main/cpp/jfrMetadata.h +++ b/ddprof-lib/src/main/cpp/jfrMetadata.h @@ -81,6 +81,12 @@ enum JfrType { T_UNWIND_FAILURE = 126, T_MALLOC = 127, T_NATIVE_SOCKET = 128, + // PROF-15341 Phase 6: reporting surface for ReferenceChainTracker + // (referenceChains.h/.cpp) - a reconstructed referrer-type chain, and a + // distinct event for a search abandoned before reaching a target (design + // doc's "no silent truncation" requirement), see jfrMetadata.cpp. + T_REFERENCE_CHAIN = 129, + T_REFERENCE_CHAIN_ABANDONED = 130, T_ANNOTATION = 200, T_LABEL = 201, T_CATEGORY = 202, diff --git a/ddprof-lib/src/main/cpp/livenessTracker.cpp b/ddprof-lib/src/main/cpp/livenessTracker.cpp index 18fb0b4312..05dd686347 100644 --- a/ddprof-lib/src/main/cpp/livenessTracker.cpp +++ b/ddprof-lib/src/main/cpp/livenessTracker.cpp @@ -4,16 +4,24 @@ */ #include +#include +#include #include #include #include "arch.h" +#include "common.h" #include "context.h" #include "context_api.h" #include "hotspot/vmStructs.h" +#include "hotspot/vmStructs.inline.h" #include "incbin.h" #include "jniHelper.h" #include "livenessTracker.h" +#include "rcDebugLevel.h" +#include "objectSampler.h" +#include "vmEntry.h" +#include "referenceChains.h" #include "log.h" #include "nativeMem.h" #include "os.h" @@ -28,16 +36,98 @@ constexpr int LivenessTracker::MAX_TRACKING_TABLE_SIZE; constexpr int LivenessTracker::MIN_SAMPLING_INTERVAL; -void LivenessTracker::cleanup_table(bool forced) { +namespace { + +// Earliest-third/recent-third mean and minimum of a chronological ring +// window - the one computation hasQualifyingGrowth() (per-klass count_ring) +// and heapFloorRising() (the aggregate _heap_floor_ring) both need, factored +// out so the window/index derivation and the two aggregation loops exist in +// exactly one place rather than three near-identical copies. Templated on +// the reader rather than the ring's element type or storage: the per-klass +// ring is a plain array read under the caller's already-held _table_lock, +// while the heap-floor ring is lock-free and read via loadAcquire() (see +// _heap_floor_ring's own comment, livenessTracker.h) - `read(i)` lets each +// caller supply its own access discipline for physical slot `i` without +// this shared loop needing to know which one applies. +struct RingThirdsStats { + double earliest_mean; + double recent_mean; + double earliest_min; + double recent_min; +}; + +template +bool ringThirdsStats(int head, int fill, int ring_size, int min_fill, + Reader read, RingThirdsStats *out) { + if (fill < min_fill) { + return false; + } + // Chronological (oldest-first) index of the window's first sample. + int start = (head - fill + ring_size) % ring_size; + // Full-window least-squares linear regression: y = a + b*x. + // x = sample position within the window (0 = oldest, fill-1 = newest), + // y = population count. This uses all samples (not just first/last + // halves or thirds) and is far more robust for oscillating-but- + // growing trends than comparing two sub-windows. O(fill) = O(30) per + // klass per scan — negligible. + int n = fill; + if (n < 2) { + return false; + } + double sum_x = 0, sum_y = 0, sum_xx = 0, sum_xy = 0; + double earliest_min = std::numeric_limits::max(); + double recent_min = std::numeric_limits::max(); + for (int i = 0; i < n; i++) { + double v = read((start + i) % ring_size); + sum_x += i; + sum_y += v; + sum_xx += (double)i * i; + sum_xy += (double)i * v; + if (v < earliest_min) { + earliest_min = v; + } + if (i >= n / 2 && v < recent_min) { + recent_min = v; + } + } + double denom = (double)n * sum_xx - sum_x * sum_x; + if (denom == 0) { + return false; + } + double slope = ((double)n * sum_xy - sum_x * sum_y) / denom; + double intercept = (sum_y - slope * sum_x) / n; + out->earliest_mean = intercept; + out->recent_mean = intercept + slope * (n - 1); + out->earliest_min = earliest_min; + out->recent_min = recent_min; + return true; +} + +} // namespace + +void LivenessTracker::cleanup_table(bool forced, bool allow_resolve) { u64 current = load(_last_gc_epoch); u64 target_gc_epoch = load(_gc_epoch); + TEST_LOG_SUMMARY("LivenessTracker::cleanup_table forced=%d gc_generations=%d current_epoch=%llu " + "target_epoch=%llu table_size=%d", + forced, _gc_generations.load(std::memory_order_relaxed), (unsigned long long)current, + (unsigned long long)target_gc_epoch, _table_size); - if ((target_gc_epoch == _last_gc_epoch || - !__atomic_compare_exchange_n(&_last_gc_epoch, ¤t, - target_gc_epoch, false, __ATOMIC_RELAXED, __ATOMIC_RELAXED)) && - !forced) { + // is_epoch_owner is true iff this call is the one that moves _last_gc_epoch + // to target_gc_epoch - i.e. the first cleanup_table() call (forced or not) + // to observe this particular GC epoch transition. Population accounting + // below is gated on this rather than on !forced, so a forced (table- + // overflow) sweep still folds one sample per genuinely new epoch instead + // of either skipping it entirely or double-counting the same epoch across + // repeated forced sweeps. + bool is_epoch_owner = target_gc_epoch != current && + __atomic_compare_exchange_n(&_last_gc_epoch, ¤t, target_gc_epoch, + false, __ATOMIC_RELAXED, __ATOMIC_RELAXED); + + if (!is_epoch_owner && !forced) { // if the last processed GC epoch hasn't changed, or if we failed to update // it, there's nothing to do + TEST_LOG_SUMMARY("LivenessTracker::cleanup_table early-exit: epoch unchanged and not forced"); return; } @@ -46,6 +136,34 @@ void LivenessTracker::cleanup_table(bool forced) { int epoch_diff = (int)(target_gc_epoch - current); _table_lock.lock(); + + // Detect a class-map reset the same way + // ReferenceChainTracker::resolveLoadedClasses() does (referenceChains.cpp) + // - see _last_class_map_generation's own comment (livenessTracker.h). + // cached_klass_id and _klass_population's klass_id keys are + // StringDictionary ids from whatever generation was current when they were + // resolved; once Profiler::start() clears that dictionary and restarts its + // id namespace, those cached ids can silently collide with a newly + // assigned, unrelated class. Drop every such cache before this pass reads + // or writes any of them. + u64 current_class_map_generation = Profiler::instance()->classMap()->generation(); + if (current_class_map_generation != _last_class_map_generation) { + for (u32 i = 0; i < _table_size; i++) { + _table[i].cached_klass_id = 0; + } + for (int i = 0; i < _klass_population_size; i++) { + for (int r = 0; r < _klass_population[i].representative_count; r++) { + jweak rep = _klass_population[i].representatives[r]; + if (rep != nullptr) { + env->DeleteWeakGlobalRef(rep); + } + } + } + _klass_population_size = 0; + _klass_count_scratch_size = 0; + _last_class_map_generation = current_class_map_generation; + } + u32 sz = _table_size; if (sz > 0) { u64 start = OS::nanotime(), end; @@ -62,16 +180,80 @@ void LivenessTracker::cleanup_table(bool forced) { _table[i].call_trace_id = 0; } _table[target].age += epoch_diff; + + if (_gc_generations.load(std::memory_order_relaxed) && is_epoch_owner) { + // Per-klass population tracking (design doc's Open Question 3) - + // gated on _gc_generations so this new cost is paid only when the + // caller actually asked for generation/survival-shaped data + // (arguments.cpp:223-227,244), not for every liveness-tracking + // session. Gated on is_epoch_owner (not !forced) so a forced + // (table-overflow) sweep still contributes one population sample + // per genuinely new GC epoch instead of silently dropping it. + u32 klass_id = 0; + if (allow_resolve) { + // GetObjectClass + Class.getName() + StringDictionary lookup per + // surviving entry, previously paid only at JFR-flush time (see + // flush_table() below). Only affordable off the allocation-hot + // path - flush_table()/stop()'s cadence and + // LivenessTracker::maybeForceCleanup()'s background-thread tick + // both pass allow_resolve=true; track()'s hot-path forced sweep + // does not (see cleanup_table()'s own header comment). + jobject ref = env->NewLocalRef(_table[target].ref); + if (ref != nullptr) { + klass_id = resolveKlassId(env, ref); + if (klass_id != 0) { + // Cache the resolution: flush_table() runs its own + // GetObjectClass+Class.getName()+lookupClass() sequence for + // every surviving entry immediately after cleanup_table() + // returns (flush_table() always calls cleanup_table() first), + // which would otherwise repeat this exact JNI round-trip for + // the same object. An object's class is immutable, so this + // value stays valid for flush_table()'s read below, and for + // a later non-resolving sweep's read right below. + _table[target].cached_klass_id = klass_id; + } + env->DeleteLocalRef(ref); + } + } else { + // track()'s table-overflow branch calls cleanup_table(true, + // false) synchronously from the allocation-sampling call stack + // (JVMTI SampledObjectAlloc callback). resolveKlassId() calls + // Class.getName(), a genuine Java-bytecode upcall (unlike the + // plain native jvmti->GetClassSignature() call + // ObjectSampler::recordAllocation already makes on this same + // callback stack) - too costly, and too re-entrancy-prone via + // the String allocation it can trigger, to run from there. Reuse + // whatever class id an earlier resolving sweep already resolved + // for this entry instead; if it was never resolved, this entry's + // sample for this epoch is dropped rather than resolving now. + klass_id = _table[target].cached_klass_id; + } + if (klass_id != 0) { + accumulateKlassCount(klass_id, _table[target].age, _table[target].ref, + _table[target].tid); + } + } } else { jweak tmpRef = _table[i].ref; _table[i].ref = nullptr; env->DeleteWeakGlobalRef(tmpRef); _table[i].call_trace_id = 0; + if (_table[i].leak_tag != 0) { + releaseLeakTag(_table[i].leak_tag); + _table[i].leak_tag = 0; + } } } _table_size = newsz; + TEST_LOG_SUMMARY("LivenessTracker::cleanup_table survivors=%u klass_count_scratch_size=%d", + newsz, _klass_count_scratch_size); + if (_gc_generations.load(std::memory_order_relaxed) && is_epoch_owner && + _klass_count_scratch_size > 0) { + foldKlassCountsLocked(env, target_gc_epoch, allow_resolve); + } + end = OS::nanotime(); Log::debug("Liveness tracker cleanup took %.2fms (%.2fus/element)", 1.0f * (end - start) / 1000 / 1000, @@ -80,6 +262,1395 @@ void LivenessTracker::cleanup_table(bool forced) { _table_lock.unlock(); } +u32 LivenessTracker::resolveKlassId(JNIEnv *env, jobject ref) { + // Deliberately NOT flush_table()'s own Class.getName()-based resolution + // below: the ids this returns are the CANDIDATE klass ids that + // ReferenceChainTracker matches discovered instances' classes against, and + // RCT resolves those with the GetClassSignature + + // ObjectSampler::normalizeClassSignature() + lookupClass() sequence + // (resolveClassMap(), referenceChains.cpp). StringDictionary keys its + // entries by the exact string, and getName()'s "com.foo.Bar" (dot + // notation) is a DIFFERENT key from the signature's "com/foo/Bar" (slash + // notation) - so a getName()-based id can never equal the signature-based + // id the same class resolves to on the RCT side, and every candidate vs + // discovered-instance comparison failed (observed live on the pod: every + // auto-mark "resolved but no candidate match", and locally: + // LivenessTracker id 63 vs ReferenceChainTracker id 2 for the same class; + // only array classes accidentally matched, since "[B" is notation- + // identical). Same sequence as ObjectSampler::recordAllocation() + // therefore - the third user of it, after recordAllocation() and + // resolveClassMap(). Also strictly cheaper than the old getName() path: + // a plain JVMTI call instead of a Class.getName() JNI upcall that could + // allocate. + jclass clz = env->GetObjectClass(ref); + u32 id = 0; + jvmtiEnv *jvmti = VM::jvmti(); + if (clz != nullptr && jvmti != nullptr) { + char *class_name = nullptr; + if (jvmti->GetClassSignature(clz, &class_name, nullptr) == + JVMTI_ERROR_NONE && + class_name != nullptr) { + const char *name_slice = nullptr; + size_t name_len = 0; + if (ObjectSampler::normalizeClassSignature(class_name, &name_slice, + &name_len)) { + int lookup_id = Profiler::instance()->lookupClass(name_slice, name_len); + if (lookup_id > 0) { + id = (u32)lookup_id; + } + } + jvmti->Deallocate((unsigned char *)class_name); + } + } + if (clz != nullptr) { + env->DeleteLocalRef(clz); + } + return id; +} + +// Inserts (sample_source, age) into scratch.oldest[], sorted by age +// descending, capped at MAX_OLDEST_SAMPLES. Called from accumulateKlassCount() +// to bias representative selection toward long-lived instances. +void LivenessTracker::insertOldestSample(KlassCountScratch &scratch, + jweak sample_source, u32 age, + jint tid) { + int pos = scratch.oldest_count; + for (int i = 0; i < scratch.oldest_count; i++) { + if (age > scratch.oldest[i].age) { + pos = i; + break; + } + } + if (pos < KlassCountScratch::MAX_OLDEST_SAMPLES) { + if (scratch.oldest_count < KlassCountScratch::MAX_OLDEST_SAMPLES) { + scratch.oldest_count++; + } + for (int i = scratch.oldest_count - 1; i > pos; i--) { + scratch.oldest[i] = scratch.oldest[i - 1]; + } + scratch.oldest[pos].ref = sample_source; + scratch.oldest[pos].age = age; + scratch.oldest[pos].tid = tid; + } +} + +jlong LivenessTracker::acquireLeakTag(u64 call_trace_id, jint tid) { + if (_leak_tag_free_count <= 0) { + return 0; // pool exhausted + } + int idx = _leak_tag_free_list[--_leak_tag_free_count]; + _leak_tag_info[idx].call_trace_id = call_trace_id; + _leak_tag_info[idx].tid = tid; + return LEAK_TAG_BASE + idx; +} + +void LivenessTracker::releaseLeakTag(jlong tag) { + if (tag < LEAK_TAG_BASE || tag >= LEAK_TAG_BASE + LEAK_TAG_POOL_SIZE) { + return; + } + int idx = (int)(tag - LEAK_TAG_BASE); + _leak_tag_info[idx].call_trace_id = 0; + _leak_tag_info[idx].tid = 0; + _leak_tag_free_list[_leak_tag_free_count++] = idx; +} + +bool LivenessTracker::getLeakTagInfo(jlong tag, u64 *out_call_trace_id, + jint *out_tid) const { + if (tag < LEAK_TAG_BASE || tag >= LEAK_TAG_BASE + LEAK_TAG_POOL_SIZE) { + return false; + } + int idx = (int)(tag - LEAK_TAG_BASE); + if (idx >= _leak_tag_free_count && + _leak_tag_info[idx].call_trace_id == 0) { + return false; // tag is in free list + } + // Check if tag is still in use (not in free list) + // Simple check: if call_trace_id is 0 and tid is 0, it's been released + if (_leak_tag_info[idx].call_trace_id == 0 && _leak_tag_info[idx].tid == 0) { + return false; + } + *out_call_trace_id = _leak_tag_info[idx].call_trace_id; + *out_tid = _leak_tag_info[idx].tid; + return true; +} + +int LivenessTracker::tagLeakInstances(jvmtiEnv *jvmti, + const KlassCandidate *candidates, + int candidate_count) { + if (!_enabled || _table == nullptr) { + return 0; + } + JNIEnv *env = VM::jni(); + int tagged = 0; + // Tagging priority: instances from allocation sites (tids) with the + // clearest surviving-age diversity go first, then oldest within a tid. + // A continuously-leaking site keeps instances alive across many distinct + // GC generations (many distinct surviving ages), while a one-time burst + // or noise site survives at few distinct ages regardless of how old its + // oldest instance is. If the pool is contended or GC churn races the + // tagging, this ordering makes sure the strongest leak signal keeps the + // tags rather than whichever entry the table scan happens to reach first. + // The (klass, tid) MATCH below scopes this whole ranking to the leak site + // itself: each candidate only claims tracked instances its qualifying + // tids allocated (selectLeakCandidates()'s per-tid gate, livenessTracker.h), + // so machinery survivors of the same class from other threads never + // enter the ranking at all - observed on hotdog, klass-wide matching spent + // 247 pool tags on flat-retention machinery byte[]s with zero + // interceptions while the real leak-site instances churned out of the + // tracking table untagged. + struct TagCandidate { + u32 table_idx; + jint tid; + u32 age; + int distinct_ages; // age diversity of this entry's tid (computed below) + bool leak_tag_recorded; // record already holds a pool tag (reused below) + }; + // Stack scratch: matching entries are bounded by the tracking table's + // small live population (~hundreds); tag in scan order beyond capacity. + TagCandidate scratch[512]; + int n_candidates = 0; + // Per-poll (klass_id, tid) tag summary - the per-instance logging this + // replaces re-logged every stable pool tag on every poll (256 lines/poll + // once the pool saturates with real leak instances), which flooded the + // pod's container log hard enough to rotate its 10MB cap inside a + // verification window (round 5: ~600k lines in 20 min, costing log-head + // from the same window). One summary line per tagged (klass, tid) group + // carries the same diagnostics - allocation size distinguishing real + // leaked chunks from machinery survivors, the age range showing the + // site's surviving-age diversity - at constant volume per poll. + struct TagSummary { + u32 klass_id; + jint tid; + int tagged; + int need_set; + u64 min_age; + u64 max_age; + u64 max_size; + }; + TagSummary summary[16]; + int summary_count = 0; + int summary_overflow = 0; + _table_lock.lockShared(); + u32 sz = _table_size; + for (u32 i = 0; i < sz; i++) { + if (_table[i].ref == nullptr) { + continue; + } + // Check if this entry's (class, allocating thread) matches any + // candidate: class alone is not enough - only the instances a + // candidate's QUALIFYING tids allocated are in tagging scope. + u32 kid = _table[i].cached_klass_id; + if (kid == 0) { + continue; + } + bool match = false; + for (int k = 0; k < candidate_count; k++) { + if (candidates[k].klass_id != kid || + candidates[k].qualifying_tid_count <= 0) { + continue; + } + for (int q = 0; q < candidates[k].qualifying_tid_count; q++) { + if (candidates[k].qualifying_tids[q] == _table[i].tid) { + match = true; + break; + } + } + if (match) { + break; + } + } + if (!match) { + continue; + } + // Entries whose record already holds a pool tag are still collected: + // their tag survives the JVMTI tag only until the object is admitted + // or the search restarts, and the state machine below must re-act on + // the CURRENT JVMTI tag (re-establish, correlate, or leave alone). + if (n_candidates < (int)(sizeof(scratch) / sizeof(scratch[0]))) { + scratch[n_candidates].table_idx = i; + scratch[n_candidates].tid = _table[i].tid; + scratch[n_candidates].age = _table[i].age; + scratch[n_candidates].distinct_ages = 0; + scratch[n_candidates].leak_tag_recorded = _table[i].leak_tag != 0; + n_candidates++; + } + } + // Compute per-tid distinct surviving ages (matching entries only - the + // same diversity signal the epoch fold uses for clustering, computed here + // directly from the tracked entries so the ranking reflects exactly the + // population being tagged). Distinct-tid count is bounded by + // MAX_THREADS_PER_KLASS logic elsewhere but here just use a small array; + // beyond 32 tids, extras share the lowest priority tier. + struct TidAges { + jint tid; + u32 ages[32]; + int age_count; + } tid_ages[32]; + int tid_count = 0; + for (int c = 0; c < n_candidates; c++) { + TidAges *t = nullptr; + for (int ti = 0; ti < tid_count; ti++) { + if (tid_ages[ti].tid == scratch[c].tid) { + t = &tid_ages[ti]; + break; + } + } + if (t == nullptr && tid_count < (int)(sizeof(tid_ages) / sizeof(tid_ages[0]))) { + t = &tid_ages[tid_count++]; + t->tid = scratch[c].tid; + t->age_count = 0; + } + if (t != nullptr) { + bool seen = false; + for (int a = 0; a < t->age_count; a++) { + if (t->ages[a] == scratch[c].age) { + seen = true; + break; + } + } + if (!seen && t->age_count < (int)(sizeof(t->ages) / sizeof(t->ages[0]))) { + t->ages[t->age_count++] = scratch[c].age; + } + } + } + for (int c = 0; c < n_candidates; c++) { + for (int ti = 0; ti < tid_count; ti++) { + if (tid_ages[ti].tid == scratch[c].tid) { + scratch[c].distinct_ages = tid_ages[ti].age_count; + break; + } + } + } + // Sort: entries whose record already holds a pool tag first (they cost + // no pool resources - their work below is correlate-or-re-establish, not + // acquire), then highest age diversity, then oldest age. Small n, + // insertion sort is fine (n <= 512 but in practice tens). + for (int c = 1; c < n_candidates; c++) { + TagCandidate key = scratch[c]; + int j = c - 1; + while (j >= 0 && + ((!scratch[j].leak_tag_recorded && key.leak_tag_recorded) || + (scratch[j].leak_tag_recorded == key.leak_tag_recorded && + (scratch[j].distinct_ages < key.distinct_ages || + (scratch[j].distinct_ages == key.distinct_ages && + scratch[j].age < key.age))))) { + scratch[j + 1] = scratch[j]; + j--; + } + scratch[j + 1] = key; + } + // Per-candidate state machine on the object's CURRENT JVMTI tag. The + // tag on the object is shared state with ReferenceChainTracker's BFS + // (frontier tags) and must never be blindly overwritten: an object + // admitted by the BFS carries its FRONTIER tag on the object, and a + // SetTag(leak_tag) here would orphan that frontier entry (the BFS could + // never resolve it again) while making correlation depend on a parent + // re-walk ever happening. Instead: + // - leak tag on object: already waiting for interception - just make + // sure the record knows it; + // - frontier tag on object: already admitted - correlate the entry + // (ReferenceChainTracker stores the leak tag ON the entry and marks + // the instance discovered), never retag; + // - no tag: plain SetTag (first tagging, or re-establishing after a + // search restart wiped all tags via releaseSearchTags()). + for (int c = 0; c < n_candidates; c++) { + u32 i = scratch[c].table_idx; + jobject ref = env->NewLocalRef(_table[i].ref); + if (ref == nullptr) { + // Object was collected between the null check and now - its record's + // tag (if any) is released by the GC cleanup path, nothing to do. + continue; + } + jlong existing = 0; + jvmtiError tag_err = jvmti->GetTag(ref, &existing); + jlong leak_tag = 0; + bool need_set = false; + if (tag_err == JVMTI_ERROR_NONE && existing >= LEAK_TAG_BASE) { + // Already carries a leak tag (ours, or one adopted below) - waiting + // for the BFS interception. Make sure the record remembers it. + leak_tag = _table[i].leak_tag != 0 ? _table[i].leak_tag : existing; + } else if (tag_err == JVMTI_ERROR_NONE && existing > 0) { + // Frontier tag: the BFS already admitted this object. Correlate the + // existing entry rather than retagging - see the block comment above. + leak_tag = _table[i].leak_tag; + if (leak_tag == 0) { + leak_tag = acquireLeakTag(_table[i].call_trace_id, _table[i].tid); + if (leak_tag == 0) { + env->DeleteLocalRef(ref); + continue; // pool exhausted - other candidates may still correlate + } + } + if (!ReferenceChainTracker::instance()->correlateAdmittedLeakTag( + existing, leak_tag, _table[i].cached_klass_id)) { + // Not a live frontier tag after all (search just restarted) - + // fall back to plain tagging. + need_set = true; + } + } else { + // No tag: first tagging, or re-establishment after a restart wiped + // all tags (releaseSearchTags() clears every JVMTI tag while the + // record keeps its pool tag - reusing it keeps pool accounting + // stable across restarts). + leak_tag = _table[i].leak_tag; + if (leak_tag == 0) { + leak_tag = acquireLeakTag(_table[i].call_trace_id, _table[i].tid); + if (leak_tag == 0) { + env->DeleteLocalRef(ref); + break; // pool exhausted + } + } + need_set = true; + } + if (need_set) { + jvmti->SetTag(ref, leak_tag); + } + _table[i].leak_tag = leak_tag; + tagged++; + // Accumulate into the per-poll summary above instead of logging per + // instance - one stable-pool poll re-logged all 256 tags' identical + // lines every 1.4s before this. + u32 tagged_kid = _table[i].cached_klass_id; + jint tagged_tid = _table[i].tid; + u64 tagged_age = _table[i].age; + u64 tagged_size = _table[i].alloc._size; + int g = 0; + while (g < summary_count && + (summary[g].klass_id != tagged_kid || summary[g].tid != tagged_tid)) { + g++; + } + if (g == summary_count && summary_count >= + (int)(sizeof(summary) / sizeof(summary[0]))) { + // Bounded groups exceeded (candidates are <= 5, qualifying tids <= 8 + // each - 16 covers every realistic (klass, tid) pair; overflow lumps). + summary_overflow++; + } else if (g == summary_count) { + summary[g].klass_id = tagged_kid; + summary[g].tid = tagged_tid; + summary[g].tagged = 1; + summary[g].need_set = (int)need_set; + summary[g].min_age = tagged_age; + summary[g].max_age = tagged_age; + summary[g].max_size = tagged_size; + summary_count++; + } else { + summary[g].tagged++; + summary[g].need_set += (int)need_set; + if (tagged_age < summary[g].min_age) { + summary[g].min_age = tagged_age; + } + if (tagged_age > summary[g].max_age) { + summary[g].max_age = tagged_age; + } + if (tagged_size > summary[g].max_size) { + summary[g].max_size = tagged_size; + } + } + env->DeleteLocalRef(ref); + } + _table_lock.unlockShared(); + for (int g = 0; g < summary_count; g++) { + TEST_LOG("LivenessTracker::tagLeakInstances summary klass_id=%u " + "tid=%d tagged=%d need_set=%d min_age=%llu max_age=%llu " + "max_size=%llu", + summary[g].klass_id, (int)summary[g].tid, summary[g].tagged, + summary[g].need_set, (unsigned long long)summary[g].min_age, + (unsigned long long)summary[g].max_age, + (unsigned long long)summary[g].max_size); + } + if (summary_overflow > 0) { + TEST_LOG("LivenessTracker::tagLeakInstances summary overflow=%d " + "(groups beyond %d)", + summary_overflow, (int)(sizeof(summary) / sizeof(summary[0]))); + } + return tagged; +} + +void LivenessTracker::insertThreadGen(KlassCountScratch &scratch, + jint tid, u32 age) { + // Find or create the thread entry for this tid + for (int i = 0; i < scratch.thread_count; i++) { + if (scratch.threads[i].tid == tid) { + // Insert age into the thread's sorted distinct-age array + auto &t = scratch.threads[i]; + // Every surviving object counts toward the per-tid retained-count + // bar (TID_RETAINED_COUNT_BAR) BEFORE the age-dedup early return + // below: the count is per-instance, the ages are per-cohort. + t.count++; + for (u32 j = 0; j < t.age_count; j++) { + if (t.ages[j] == age) { + return; // age already counted for this thread + } + } + if (t.age_count < KlassCountScratch::MAX_AGES_PER_THREAD) { + // Insert sorted (small array, linear scan + shift) + u32 pos = t.age_count; + for (u32 j = 0; j < t.age_count; j++) { + if (age < t.ages[j]) { + pos = j; + break; + } + } + for (u32 j = t.age_count; j > pos; j--) { + t.ages[j] = t.ages[j - 1]; + } + t.ages[pos] = age; + t.age_count++; + } + return; + } + } + if (scratch.thread_count < KlassCountScratch::MAX_THREADS_PER_KLASS) { + int idx = scratch.thread_count++; + scratch.threads[idx].tid = tid; + scratch.threads[idx].ages[0] = age; + scratch.threads[idx].age_count = 1; + scratch.threads[idx].count = 1; // this object is the tid's first this epoch + } + // else: thread table full — additional threads are not tracked, but + // the oldest[] array still captures instances from all threads. +} + +void LivenessTracker::accumulateKlassCount(u32 klass_id, jlong age, + jweak sample_source, + jint tid) { + // Count distinct GC ages (generations) per klass: group surviving + // tracked objects by klass, then for each klass count the + // number of unique age values. This is the "generation + // count" — if new instances keep arriving while old ones + // survive, the number of distinct ages grows. + for (int i = 0; i < _klass_count_scratch_size; i++) { + if (_klass_count_scratch[i].klass_id == klass_id) { + auto &entry = _klass_count_scratch[i]; + // Per-class age dedup: only count each age once for the klass' + // generation count. But per-site tracking and oldest[] must see + // EVERY surviving object, not just the first per age — so those + // run unconditionally below, outside this dedup check. + bool age_seen = false; + for (u32 a : entry.ages) { + if (a == (u32)age) { + age_seen = true; + break; + } + } + if (!age_seen) { + entry.ages.push_back((u32)age); + } + // Track top-N oldest instances (Lindy bias): insert this sample + // into the oldest[] array, sorted by age descending, capped at + // MAX_OLDEST_SAMPLES. Runs for every object, not just new ages. + insertOldestSample(entry, sample_source, (u32)age, tid); + // Track per-thread distinct surviving generations (Cork/Swat + // heuristic): add this object's age to its thread's age set. + // Runs for every object — the thread's generation cardinality is + // the leak signal, and it must see all surviving objects to be + // accurate. + insertThreadGen(entry, tid, (u32)age); + return; + } + } + if (_klass_count_scratch_size < MAX_KLASS_POPULATION_ENTRIES) { + KlassCountScratch &slot = _klass_count_scratch[_klass_count_scratch_size++]; + slot.klass_id = klass_id; + slot.ages.clear(); + slot.ages.push_back((u32)age); + slot.oldest_count = 0; + slot.thread_count = 0; + insertOldestSample(slot, sample_source, (u32)age, tid); + insertThreadGen(slot, tid, (u32)age); + } + // else: this epoch's scratch snapshot already holds + // MAX_KLASS_POPULATION_ENTRIES distinct surviving klasses - klass_id's + // count for this epoch is dropped rather than growing the scratch array, + // the same best-effort tradeoff _klass_population's own fixed capacity + // already accepts. +} + +jweak LivenessTracker::recordKlassPopulationSampleLocked( + u32 klass_id, u32 count, u64 epoch, int *out_slot, bool *out_created, + jweak *out_evicted, int *out_evicted_count, int max_evicted) { + // Linear scan is fine: MAX_KLASS_POPULATION_ENTRIES is small enough that a + // full scan is cheap, the same shape NativeSocketSampler's fd LRU + // (nativeSocketSampler.h:141-142) and this class's own cleanup_table() + // pass already accept for bounded tables. + int slot = -1; + int evict_slot = -1; + for (int i = 0; i < _klass_population_size; i++) { + if (_klass_population[i].klass_id == klass_id) { + slot = i; + break; + } + if (evict_slot < 0 || + _klass_population[i].last_updated_epoch < + _klass_population[evict_slot].last_updated_epoch) { + evict_slot = i; + } + } + + jweak evicted_ref = nullptr; + bool created = false; + if (slot < 0) { + created = true; + if (_klass_population_size < MAX_KLASS_POPULATION_ENTRIES) { + slot = _klass_population_size++; + } else { + // Table full - evict the least-recently-updated entry (evict_slot is + // guaranteed set here since MAX_KLASS_POPULATION_ENTRIES > 0 implies + // at least one iteration of the loop above ran). + slot = evict_slot; + // Return evicted representatives to caller for DeleteWeakGlobalRef. + // recordKlassPopulationSampleLocked has no JNIEnv*, so it cannot + // delete them itself. The caller (foldKlassCountsLocked) has env. + // At most MAX_REPRESENTATIVES_PER_KLASS refs to return. + if (out_evicted != nullptr) { + for (int r = 0; r < _klass_population[slot].representative_count && + *out_evicted_count < max_evicted; r++) { + out_evicted[(*out_evicted_count)++] = + _klass_population[slot].representatives[r]; + } + } + } + _klass_population[slot].klass_id = klass_id; + _klass_population[slot].representative_count = 0; + memset(_klass_population[slot].representatives, 0, sizeof(_klass_population[slot].representatives)); + memset(_klass_population[slot].rep_tids, 0, sizeof(_klass_population[slot].rep_tids)); + _klass_population[slot].ring_head = 0; + _klass_population[slot].ring_fill = 0; + _klass_population[slot].consecutive_positive = 0; + _klass_population[slot].cached_slope = 0.0; + // A reused (evicted) slot's previous class's per-tid trends must not + // leak onto the new one, same as the fields above. + _klass_population[slot].tid_trend_count = 0; + // A reused (evicted) slot's PREVIOUS class's stable tag must not leak + // onto the new one - see KlassPopulationEntry::stable_class_tag's own + // comment. Minted lazily in foldKlassCountsLocked() once a live + // instance is available to resolve the class from. + _klass_population[slot].stable_class_tag = 0; + } + + KlassPopulationEntry &entry = _klass_population[slot]; + entry.count_ring[entry.ring_head] = count; + entry.ring_head = (u8)((entry.ring_head + 1) % KLASS_POPULATION_RING_SIZE); + if (entry.ring_fill < KLASS_POPULATION_RING_SIZE) { + entry.ring_fill++; + } + entry.last_updated_epoch = epoch; + + // Updated here (not in selectLeakCandidates()) so both the production path + // (foldKlassCountsLocked(), once per genuine GC epoch) and the + // klassPopulationRecordForTest() test seam - which calls this method + // directly - keep consecutive_positive in sync with the ring they just + // pushed, rather than requiring every caller to remember to do it (see + // this class's own header comment on hasQualifyingGrowth()). + if (hasQualifyingGrowth(entry)) { + if (entry.consecutive_positive < UINT8_MAX) { + entry.consecutive_positive++; + } + } else { + entry.consecutive_positive = 0; + } + + *out_slot = slot; + *out_created = created; + return evicted_ref; +} + +void LivenessTracker::mintStableClassTagIfNeeded(JNIEnv *env, int slot, + jobject instance) { + if (slot < 0 || slot >= _klass_population_size || instance == nullptr || + _klass_population[slot].stable_class_tag != 0) { + return; + } + jvmtiEnv *jvmti = VM::jvmti(); + jclass klass = env->GetObjectClass(instance); + if (jvmti != nullptr && klass != nullptr) { + jlong tag = 0; + if (jvmti->GetTag(klass, &tag) == JVMTI_ERROR_NONE) { + if (tag == 0) { + tag = ClassTagAllocator::next(); + jvmti->SetTag(klass, tag); + } + _klass_population[slot].stable_class_tag = tag; + } + } + if (klass != nullptr) { + env->DeleteLocalRef(klass); + } +} + +void LivenessTracker::foldKlassCountsLocked(JNIEnv *env, u64 epoch, + bool allow_resolve) { + TEST_LOG_SUMMARY("LivenessTracker::foldKlassCountsLocked epoch=%llu scratch_size=%d", + (unsigned long long)epoch, _klass_count_scratch_size); + for (int i = 0; i < _klass_count_scratch_size; i++) { + KlassCountScratch &s = _klass_count_scratch[i]; + TEST_LOG("LivenessTracker::foldKlassCountsLocked scratch[%d] klass_id=%u gen_count=%zu " + "thread_count=%d oldest_count=%d", + i, s.klass_id, s.ages.size(), s.thread_count, s.oldest_count); + for (int ti = 0; ti < s.thread_count; ti++) { + TEST_LOG(" thread[%d] tid=%d age_count=%u", ti, (int)s.threads[ti].tid, + s.threads[ti].age_count); + } + int slot; + bool created; + jweak evicted[KlassPopulationEntry::MAX_REPRESENTATIVES_PER_KLASS]; + int evicted_count = 0; + recordKlassPopulationSampleLocked(s.klass_id, (u32)s.ages.size(), + epoch, &slot, &created, + evicted, &evicted_count, + KlassPopulationEntry::MAX_REPRESENTATIVES_PER_KLASS); + // Per-(klass,tid) qualification fold - see recordTidTrendSamplesLocked()'s + // own comment: pure table work, no JNIEnv, so unlike the representative + // minting below it runs regardless of allow_resolve. + recordTidTrendSamplesLocked(slot, s); + for (int r = 0; r < evicted_count; r++) { + if (evicted[r] != nullptr) { + env->DeleteWeakGlobalRef(evicted[r]); + } + } + if (!allow_resolve) { + // track()'s table-overflow branch calls cleanup_table(true, false) + // synchronously from the JVMTI SampledObjectAlloc callback stack - the + // same reason resolveKlassId() is skipped there (cleanup_table()'s own + // comment above). The representative-minting NewLocalRef/ + // NewWeakGlobalRef/DeleteLocalRef churn below is no Java-bytecode + // upcall, but it is still avoidable JNI work on that hot path; leaving + // the representative unset here is safe because the retry condition + // right below picks it up again on the next allow_resolve=true sweep. + continue; + } + // Also retry minting when an existing entry's representatives are + // stale: either the count is zero, or all stored jweaks refer to + // collected objects. A jweak's pointer value never becomes nullptr + // just because its referent was collected, so we must probe each one. + // Resolving here every epoch bounds any given gap to "one epoch with + // no representative", not permanent. + // + // Mint up to MAX_REPRESENTATIVES_PER_KLASS representatives from the + // oldest surviving instances (Lindy bias: oldest = most likely to be + // leaks). Fresh independent jweaks are minted rather than reusing + // s.oldest[].ref directly — those are TrackingEntry jweaks that get + // deleted when cleanup_table() reaps the original entry. + bool need_mint = created || + _klass_population[slot].representative_count == 0; + if (!need_mint) { + // Check if all representatives are stale + bool any_live = false; + for (int r = 0; r < _klass_population[slot].representative_count; r++) { + jweak rep = _klass_population[slot].representatives[r]; + if (rep != nullptr) { + jobject probe = env->NewLocalRef(rep); + if (probe != nullptr) { + any_live = true; + env->DeleteLocalRef(probe); + break; + } + env->DeleteLocalRef(probe); + } + } + need_mint = !any_live; + } + // Compute the dominant allocating thread (highest generation + // cardinality — most distinct surviving GC ages). This reuses + // the same generation-count signal that selectLeakCandidates() + // uses per-class, applied at per-thread granularity within a + // class. A thread with 12 distinct surviving ages (continuous + // leak) outscores a thread with 1 age (one-time burst), + // regardless of raw instance count or size (Cork/Swat + // heuristic). Thread ID is used instead of call_trace_id because + // lambdas fragment call_trace_id — synthetic methods produce + // slightly different stack hashes for what is logically one + // allocation site. + jint dominant_tid = 0; + u32 dominant_gens = 0; + for (int ti = 0; ti < s.thread_count; ti++) { + if (s.threads[ti].age_count > dominant_gens) { + dominant_gens = s.threads[ti].age_count; + dominant_tid = s.threads[ti].tid; + } + } + // Re-mint if the dominant thread has >1 generation AND none of + // the current reps were minted from it. This handles the startup- + // cache problem: reps minted from noise threads during startup + // stay live even after the real leak thread becomes dominant, + // blocking re-selection because need_mint=false. By checking + // whether reps match the dominant thread, we replace stale reps + // with instances from the actual leak thread. + if (!need_mint && dominant_gens > 1) { + bool rep_matches_dominant = false; + for (int r = 0; r < _klass_population[slot].representative_count; r++) { + if (_klass_population[slot].rep_tids[r] == dominant_tid) { + rep_matches_dominant = true; + break; + } + } + if (!rep_matches_dominant) { + need_mint = true; + TEST_LOG("LivenessTracker::foldKlassCountsLocked re-minting klass_id=%u: " + "dominant_tid=%d dominant_gens=%u but no rep matches", + s.klass_id, (int)dominant_tid, dominant_gens); + } + } + if (need_mint) { + // Clean up old representatives + for (int r = 0; r < _klass_population[slot].representative_count; r++) { + if (_klass_population[slot].representatives[r] != nullptr) { + env->DeleteWeakGlobalRef(_klass_population[slot].representatives[r]); + _klass_population[slot].representatives[r] = nullptr; + } + _klass_population[slot].rep_tids[r] = 0; + } + _klass_population[slot].representative_count = 0; + // Mint fresh representatives, preferring instances from the + // dominant allocating thread. If the dominant thread has fewer + // than MAX_REPRESENTATIVES_PER_KLASS instances in oldest[], fill + // the remaining slots with other oldest instances. + bool minted_any = false; + int minted = 0; + // First pass: instances from the dominant thread (only if it has + // >1 distinct generation — otherwise all threads are equally + // uninteresting and pure oldest-first is fine) + if (dominant_gens > 1) { + for (int r = 0; r < s.oldest_count && + minted < KlassPopulationEntry::MAX_REPRESENTATIVES_PER_KLASS; r++) { + if (s.oldest[r].tid != dominant_tid) continue; + jobject strong = env->NewLocalRef(s.oldest[r].ref); + if (strong != nullptr) { + jweak rep = env->NewWeakGlobalRef(strong); + int idx = _klass_population[slot].representative_count++; + _klass_population[slot].representatives[idx] = rep; + _klass_population[slot].rep_tids[idx] = dominant_tid; + if (!minted_any) { + mintStableClassTagIfNeeded(env, slot, strong); + minted_any = true; + } + env->DeleteLocalRef(strong); + minted++; + } + } + } + // Second pass: fill remaining slots with other oldest instances + for (int r = 0; r < s.oldest_count && + minted < KlassPopulationEntry::MAX_REPRESENTATIVES_PER_KLASS; r++) { + if (dominant_gens > 1 && s.oldest[r].tid == dominant_tid) continue; + jobject strong = env->NewLocalRef(s.oldest[r].ref); + if (strong != nullptr) { + jweak rep = env->NewWeakGlobalRef(strong); + int idx = _klass_population[slot].representative_count++; + _klass_population[slot].representatives[idx] = rep; + _klass_population[slot].rep_tids[idx] = s.oldest[r].tid; + if (!minted_any) { + mintStableClassTagIfNeeded(env, slot, strong); + minted_any = true; + } + env->DeleteLocalRef(strong); + minted++; + } + } + // else: all surviving instances for this klass died before we + // could mint a representative - left with representative_count=0 + // for this epoch, retried on the next one. + TEST_LOG("LivenessTracker::foldKlassCountsLocked minted=%d for klass_id=%u " + "dominant_tid=%d dominant_gens=%u", + minted, s.klass_id, (int)dominant_tid, dominant_gens); + } + } + _klass_count_scratch_size = 0; +} + +bool LivenessTracker::hasQualifyingGrowth(const KlassPopulationEntry &entry) const { + RingThirdsStats stats; + if (!ringThirdsStats( + entry.ring_head, entry.ring_fill, KLASS_POPULATION_RING_SIZE, + KLASS_POPULATION_MIN_FILL_FOR_TREND, + [&entry](int i) { return (double)entry.count_ring[i]; }, &stats)) { + TEST_LOG("LivenessTracker::hasQualifyingGrowth klass_id=%u ring_fill=%u " + "INSUFFICIENT_FILL (need %d)", + entry.klass_id, entry.ring_fill, + KLASS_POPULATION_MIN_FILL_FOR_TREND); + return false; + } + + // Cached for selectLeakCandidates()'s ranking (KlassPopulationEntry:: + // cached_slope's own comment, livenessTracker.h) - the ring only changes + // on push, so this is the same value a later re-scan would compute. + entry.cached_slope = stats.recent_mean - stats.earliest_mean; + + double growth_bar = LEAK_GROWTH_REL_MIN * stats.earliest_mean; + if (growth_bar < LEAK_GROWTH_ABS_MIN) { + growth_bar = LEAK_GROWTH_ABS_MIN; + } + if (entry.cached_slope < growth_bar) { + TEST_LOG("LivenessTracker::hasQualifyingGrowth klass_id=%u " + "SLOPE_TOO_SMALL slope=%f growth_bar=%f", + entry.klass_id, entry.cached_slope, growth_bar); + return false; + } + + TEST_LOG("LivenessTracker::hasQualifyingGrowth klass_id=%u " + "SLOPE_OK slope=%f growth_bar=%f", + entry.klass_id, entry.cached_slope, growth_bar); + return true; +} + +bool LivenessTracker::hasQualifyingTidGrowth( + const KlassPopulationEntry::TidTrend &trend) const { + RingThirdsStats stats; + if (!ringThirdsStats( + trend.ring_head, trend.ring_fill, + KlassPopulationEntry::TID_TREND_RING_SIZE, + TID_TREND_MIN_FILL_FOR_TREND, + [&trend](int i) { return (double)trend.ring[i]; }, &stats)) { + return false; + } + // Same growth bar as the klass gate (LEAK_GROWTH_REL_MIN/ABS_MIN): + // per-tid age-cardinality is the same small-integer signal the klass gate + // already applies these thresholds to, so they transfer unchanged. + double growth_bar = LEAK_GROWTH_REL_MIN * stats.earliest_mean; + if (growth_bar < LEAK_GROWTH_ABS_MIN) { + growth_bar = LEAK_GROWTH_ABS_MIN; + } + return (stats.recent_mean - stats.earliest_mean) >= growth_bar; +} + +bool LivenessTracker::tidPushQualifies( + const KlassPopulationEntry::TidTrend &trend, u32 current_count) const { + // Second discriminator before the (usually cheaper) ring re-scan: a + // count over the bar qualifies without any history at all. + if (current_count >= TID_RETAINED_COUNT_BAR) { + return true; + } + return hasQualifyingTidGrowth(trend); +} + +void LivenessTracker::recordTidTrendSamplesLocked( + int slot, const KlassCountScratch &scratch) { + if (slot < 0 || slot >= _klass_population_size) { + return; + } + KlassPopulationEntry &entry = _klass_population[slot]; + // Present tids first: find-or-create their trend and push this epoch's + // distinct-age count. (KlassCountScratch threads always carry >=1 + // surviving instance, so every scratch thread has a real claim on a + // slot; a zero-count tid never reaches this fold.) + for (int ti = 0; ti < scratch.thread_count; ti++) { + jint tid = scratch.threads[ti].tid; + KlassPopulationEntry::TidTrend *trend = nullptr; + for (int i = 0; i < entry.tid_trend_count; i++) { + if (entry.tid_trends[i].tid == tid) { + trend = &entry.tid_trends[i]; + break; + } + } + if (trend != nullptr && trend->synthetic) { + // Seam-owned history: the test maintains this ramp itself + // (seedTidTrendSample0), and a real fold interleaving its own low + // count for the same tid would reset the trend's hysteresis every + // System.gc() the scenario runs between rounds. Production data has + // no synthetic trends, so this exemption is inert there. + continue; + } + if (trend == nullptr) { + if (entry.tid_trend_count < KlassPopulationEntry::MAX_TID_TRENDS) { + trend = &entry.tid_trends[entry.tid_trend_count++]; + } else { + // Full: evict the WEAKEST non-synthetic trend (lowest + // consecutive_positive, then lowest ring_fill) - a genuinely rising + // leak tid accumulates hysteresis fast and resists eviction, while + // machinery tids never build any. Synthetic (test-seeded) trends + // are never evicted by real data. If every slot is synthetic, the + // new tid is simply dropped - production data never marks + // anything synthetic, so this only caps a scenario's own seeding. + int victim = -1; + for (int i = 0; i < entry.tid_trend_count; i++) { + if (entry.tid_trends[i].synthetic) { + continue; + } + if (victim < 0 || + (entry.tid_trends[i].consecutive_positive < + entry.tid_trends[victim].consecutive_positive) || + (entry.tid_trends[i].consecutive_positive == + entry.tid_trends[victim].consecutive_positive && + entry.tid_trends[i].ring_fill < + entry.tid_trends[victim].ring_fill)) { + victim = i; + } + } + if (victim < 0) { + continue; // all slots synthetic - drop this tid's sample + } + trend = &entry.tid_trends[victim]; + } + trend->tid = tid; + trend->ring_head = 0; + trend->ring_fill = 0; + trend->consecutive_positive = 0; + trend->synthetic = false; + } + trend->ring[trend->ring_head] = (u8)scratch.threads[ti].age_count; + trend->count_ring[trend->ring_head] = + (u8)std::min(scratch.threads[ti].count, UINT8_MAX); + trend->ring_head = (u8)((trend->ring_head + 1) % + KlassPopulationEntry::TID_TREND_RING_SIZE); + if (trend->ring_fill < KlassPopulationEntry::TID_TREND_RING_SIZE) { + trend->ring_fill++; + } + if (tidPushQualifies(*trend, scratch.threads[ti].count)) { + if (trend->consecutive_positive < UINT8_MAX) { + trend->consecutive_positive++; + } + } else { + trend->consecutive_positive = 0; + } + } + // Tracked tids ABSENT this epoch: a thread whose instances all died + // must not keep a stale rising ring - push 0 so the next + // hasQualifyingTidGrowth() fails and the trend's hysteresis resets (the + // per-tid analogue of the population simply stopping). Synthetic + // (test-seeded) trends are exempt: scenarios interleave real + // System.gc()-driven folds with their seeded ramps, and without the + // exemption every real fold would wipe the seeded qualification before + // the test could ever use it (see TidTrend::synthetic's own comment). + for (int i = 0; i < entry.tid_trend_count; i++) { + KlassPopulationEntry::TidTrend &trend = entry.tid_trends[i]; + if (trend.synthetic) { + continue; + } + bool present = false; + for (int ti = 0; ti < scratch.thread_count; ti++) { + if (scratch.threads[ti].tid == trend.tid) { + present = true; + break; + } + } + if (present) { + continue; + } + trend.ring[trend.ring_head] = 0; + trend.count_ring[trend.ring_head] = 0; + trend.ring_head = (u8)((trend.ring_head + 1) % + KlassPopulationEntry::TID_TREND_RING_SIZE); + if (trend.ring_fill < KlassPopulationEntry::TID_TREND_RING_SIZE) { + trend.ring_fill++; + } + // A 0-count epoch cannot qualify as a rise - reset now rather than + // deferring to the next push, so a long-absent tid does not keep + // hysteresis from before its population died. + trend.consecutive_positive = 0; + } +} + +void LivenessTracker::recordHeapFloorSample(u64 used, u64 timestamp_ns, u64 container_used) { + TEST_LOG("LivenessTracker::recordHeapFloorSample called used=%llu disabled=%d", + (unsigned long long)used, + (int)_heap_floor_recording_disabled_for_test.load(std::memory_order_acquire)); +#ifdef DEBUG + if (_heap_floor_recording_disabled_for_test.load(std::memory_order_acquire)) { + TEST_LOG("LivenessTracker::recordHeapFloorSample SKIPPED (disabled for test)"); + return; + } +#endif + recordHeapFloorSampleUnchecked(used, timestamp_ns, container_used); +} + +void LivenessTracker::recordHeapFloorSampleUnchecked(u64 used, u64 timestamp_ns, u64 container_used) { + TEST_LOG("LivenessTracker::recordHeapFloorSample used=%llu timestamp_ns=%llu container_used=%llu", + (unsigned long long)used, (unsigned long long)timestamp_ns, + (unsigned long long)container_used); + // Lock-free, single-writer-at-a-time - see _heap_floor_ring's own comment + // (livenessTracker.h) for why onGC() cannot take _table_lock here. + // + // Standard SPSC publish order: the payload is a plain store, and the index + // that gates which slots are valid is what carries the release. A reader + // that loadAcquire()s the index is then guaranteed to see this payload + // write too, since it precedes the index's storeRelease() in program + // order and a release store cannot be reordered before an earlier store. + // (The previous version had this backwards - storeRelease() on the + // payload with a plain store on the index - which does not establish any + // ordering between "the index says this slot is valid" and "the payload + // for that slot is visible".) _heap_floor_time_ring's and + // _container_mem_ring's payload writes below are plain stores for the + // same reason - they precede the same storeRelease() in program order. + u8 head = load(_heap_floor_ring_head); + store(_heap_floor_ring[head], used); + store(_heap_floor_time_ring[head], timestamp_ns); + store(_container_mem_ring[head], container_used); + storeRelease(_heap_floor_ring_head, (u8)((head + 1) % KLASS_POPULATION_RING_SIZE)); + u8 fill = load(_heap_floor_ring_fill); + if (fill < KLASS_POPULATION_RING_SIZE) { + storeRelease(_heap_floor_ring_fill, (u8)(fill + 1)); + } +} + +bool LivenessTracker::heapFloorRising() const { + // Matches recordHeapFloorSample()'s storeRelease() on both index fields - + // loadAcquire() here is what makes the payload writes below visible. + u8 fill = loadAcquire(_heap_floor_ring_fill); + u8 head = loadAcquire(_heap_floor_ring_head); + RingThirdsStats stats; + if (!ringThirdsStats( + head, fill, KLASS_POPULATION_RING_SIZE, + KLASS_POPULATION_MIN_FILL_FOR_TREND, + [this](int i) { return (double)load(_heap_floor_ring[i]); }, + &stats)) { + return false; + } + + double growth_bar = HEAP_FLOOR_GROWTH_REL_MIN * stats.earliest_mean; + if (growth_bar < (double)HEAP_FLOOR_GROWTH_ABS_MIN) { + growth_bar = (double)HEAP_FLOOR_GROWTH_ABS_MIN; + } + bool mean_rising = (stats.recent_mean - stats.earliest_mean) >= growth_bar; + if (!mean_rising) { + TEST_LOG("LivenessTracker::heapFloorRising MEAN_NOT_RISING " + "recent_mean=%.0f earliest_mean=%.0f growth_bar=%.0f", + stats.recent_mean, stats.earliest_mean, growth_bar); + return false; + } + + double floor_bar = HEAP_FLOOR_FLOOR_REL_MIN * stats.earliest_min; + if (floor_bar < (double)HEAP_FLOOR_FLOOR_ABS_MIN) { + floor_bar = (double)HEAP_FLOOR_FLOOR_ABS_MIN; + } + bool floor_rising = (stats.recent_min - stats.earliest_min) >= floor_bar; + TEST_LOG("LivenessTracker::heapFloorRising %s " + "recent_mean=%.0f earliest_mean=%.0f recent_min=%.0f earliest_min=%.0f " + "floor_bar=%.0f floor_rising=%d", + floor_rising ? "FLOOR_RISING" : "FLOOR_NOT_RISING", + stats.recent_mean, stats.earliest_mean, + stats.recent_min, stats.earliest_min, + floor_bar, (int)floor_rising); + return floor_rising; +} + +double LivenessTracker::secondsToOOM() const { +#ifdef DEBUG + jlong max_heap = _max_heap_bytes_for_test.load(std::memory_order_acquire); + if (max_heap <= 0) { + max_heap = _max_heap_bytes; + } + jlong container_limit = _container_memory_limit_for_test.load(std::memory_order_acquire); + if (container_limit <= 0) { + container_limit = _container_memory_limit; + } +#else + jlong max_heap = _max_heap_bytes; + jlong container_limit = _container_memory_limit; +#endif + if (!_gc_generations.load(std::memory_order_relaxed) || max_heap <= 0) { + TEST_LOG("LivenessTracker::secondsToOOM -> -1 (gc_generations=%d max_heap=%lld)", + (int)_gc_generations.load(std::memory_order_relaxed), (long long)max_heap); + return -1; + } + + // Project against whichever of the JVM heap or the container memory + // limit is tighter, rather than projecting both and comparing results - + // an unavailable container limit (bare metal, macOS, cgroups disabled) is + // treated as unbounded so it never wins this comparison. See this + // method's own comment (livenessTracker.h) for why the two are + // independent boundaries worth checking at all. + jlong effective_container_limit = + container_limit > 0 ? container_limit : std::numeric_limits::max(); + bool use_container = effective_container_limit < max_heap; + jlong limit = use_container ? container_limit : max_heap; + + u8 fill = loadAcquire(_heap_floor_ring_fill); + u8 head = loadAcquire(_heap_floor_ring_head); + TEST_LOG("LivenessTracker::secondsToOOM ring fill=%d head=%d source=%s limit=%lld", + (int)fill, (int)head, use_container ? "container" : "heap", (long long)limit); + + RingThirdsStats byte_stats; + bool have_byte_stats = use_container + ? ringThirdsStats( + head, fill, KLASS_POPULATION_RING_SIZE, + KLASS_POPULATION_MIN_FILL_FOR_TREND, + [this](int i) { return (double)load(_container_mem_ring[i]); }, + &byte_stats) + : ringThirdsStats( + head, fill, KLASS_POPULATION_RING_SIZE, + KLASS_POPULATION_MIN_FILL_FOR_TREND, + [this](int i) { return (double)load(_heap_floor_ring[i]); }, + &byte_stats); + if (!have_byte_stats) { + TEST_LOG("LivenessTracker::secondsToOOM -> -1 (INSUFFICIENT_FILL fill=%d need=%d)", + (int)fill, KLASS_POPULATION_MIN_FILL_FOR_TREND); + return -1; + } + RingThirdsStats time_stats; + ringThirdsStats( + head, fill, KLASS_POPULATION_RING_SIZE, KLASS_POPULATION_MIN_FILL_FOR_TREND, + [this](int i) { return (double)load(_heap_floor_time_ring[i]); }, + &time_stats); + + double bytes_delta = byte_stats.recent_mean - byte_stats.earliest_mean; + double time_delta_ns = time_stats.recent_mean - time_stats.earliest_mean; + TEST_LOG("LivenessTracker::secondsToOOM bytes_delta=%.0f time_delta_ns=%.0f " + "earliest_mean=%.0f recent_mean=%.0f earliest_min=%.0f recent_min=%.0f", + bytes_delta, time_delta_ns, + byte_stats.earliest_mean, byte_stats.recent_mean, + byte_stats.earliest_min, byte_stats.recent_min); + if (bytes_delta <= 0 || time_delta_ns <= 0) { + TEST_LOG("LivenessTracker::secondsToOOM -> -1 (NOT_RISING bytes_delta=%.0f time_delta_ns=%.0f)", + bytes_delta, time_delta_ns); + return -1; + } + + // Corroborate with a fit over just the most recent half of the window + // (HEAP_FLOOR_RECENT_HALF_MIN_FILL's own comment) - a one-time step + // change that has already plateaued still passes the full-window check + // above for as long as any of its samples remain in the window, but its + // own recent half is flat. + int half_fill = fill / 2; + RingThirdsStats recent_half_stats; + bool have_recent_half = use_container + ? ringThirdsStats( + head, half_fill, KLASS_POPULATION_RING_SIZE, + HEAP_FLOOR_RECENT_HALF_MIN_FILL, + [this](int i) { return (double)load(_container_mem_ring[i]); }, + &recent_half_stats) + : ringThirdsStats( + head, half_fill, KLASS_POPULATION_RING_SIZE, + HEAP_FLOOR_RECENT_HALF_MIN_FILL, + [this](int i) { return (double)load(_heap_floor_ring[i]); }, + &recent_half_stats); + double recent_half_delta = + have_recent_half ? recent_half_stats.recent_mean - recent_half_stats.earliest_mean : 0; + if (!have_recent_half || recent_half_delta <= 0) { + TEST_LOG("LivenessTracker::secondsToOOM -> -1 (RECENT_HALF_FLAT " + "half_fill=%d have_recent_half=%d recent_half_delta=%.0f)", + half_fill, (int)have_recent_half, recent_half_delta); + return -1; + } + + double bytes_per_ns = bytes_delta / time_delta_ns; + double remaining_bytes = (double)limit - byte_stats.recent_mean; + if (remaining_bytes <= 0) { + // The chosen ring's own recent mean has already reached (or passed) its + // limit - exhaustion is not "in N seconds", it's now. + return 0; + } + return (remaining_bytes / bytes_per_ns) / 1e9; // ns -> seconds +} + +int LivenessTracker::selectLeakCandidates(KlassCandidate *out, int max) { + int cap = max < MAX_LEAK_CANDIDATES ? max : MAX_LEAK_CANDIDATES; + if (cap <= 0) { + return 0; + } + + // Kept sorted descending by slope magnitude, at most `cap` (<= + // MAX_LEAK_CANDIDATES == 5) entries - not one per klass - so an + // insertion-sort-style insert per candidate (O(cap) per insert, O(N*cap) + // overall for N <= MAX_KLASS_POPULATION_ENTRIES == 256 klasses) is cheaper + // and simpler than collecting every qualifying candidate and calling + // std::sort. + // A single call, shared by every candidate this scan considers - see + // LEAK_TREND_HYSTERESIS_BASE/CORROBORATED's own comment (livenessTracker.h) + // for why an aggregate, non-attributed signal can only raise or lower the + // bar uniformly, never reorder candidates against each other. Lock-free + // (heapFloorRising()'s own comment), so no relation to _table_lock below. + const int required_hysteresis = heapFloorRising() + ? LEAK_TREND_HYSTERESIS_CORROBORATED + : LEAK_TREND_HYSTERESIS_BASE; + + double best_slopes[MAX_LEAK_CANDIDATES]; + int count = 0; + + // Read-only pass over _klass_population - mirrors getLiveTraceIds()'s own + // shared-lock read pattern above, the same table cleanup_table() writes + // under the exclusive lock this shared lock is taken against. + _table_lock.lockShared(); + // Only log when there is actually something to scan - this runs on every + // BFS-thread wake (once per second), so logging an empty scan turns the + // steady, idle state into per-second noise. + if (_klass_population_size > 0) { + TEST_LOG("LivenessTracker::selectLeakCandidates scanning %d klass_population entries", + _klass_population_size); + } + for (int i = 0; i < _klass_population_size; i++) { + const KlassPopulationEntry &entry = _klass_population[i]; + // cached_slope was computed by hasQualifyingGrowth() the last time this + // entry was pushed (recordKlassPopulationSampleLocked()) - the ring only + // changes on push, so re-scanning it here would just recompute the same + // value a moment later. + bool has_trend = entry.ring_fill >= KLASS_POPULATION_MIN_FILL_FOR_TREND; + double slope = entry.cached_slope; + TEST_LOG("LivenessTracker::selectLeakCandidates entry[%d] klass_id=%u ring_fill=%u " + "has_trend=%d slope=%f consecutive_positive=%u required=%d rep_count=%d", + i, entry.klass_id, entry.ring_fill, has_trend, has_trend ? slope : 0.0, + entry.consecutive_positive, required_hysteresis, entry.representative_count); + if (!has_trend || slope <= 0 || entry.consecutive_positive < required_hysteresis) { + // Not enough history yet, flat/shrinking, or hasn't shown a + // qualifying rise (hasQualifyingGrowth()) for enough consecutive + // epochs yet to trust it over sampling/oscillation noise. + continue; + } + // (klass,tid) qualification: the klass-level rise above is also + // produced by churn spread across many threads each retaining a + // stable handful of instances (observed live on hotdog - see + // KlassPopulationEntry::TidTrend's own comment), so a klass only + // becomes a candidate if at least ONE allocating thread's own per-tid + // trend has held a qualifying rise for the same hysteresis. The + // qualifying tids are handed to the caller so tagLeakInstances() can + // scope its pool tags to the leak-site instances only. + jint qualifying_tids[KlassPopulationEntry::MAX_TID_TRENDS]; + int qualifying_tid_count = 0; + for (int t = 0; t < entry.tid_trend_count && + qualifying_tid_count < + (int)(sizeof(qualifying_tids) / + sizeof(qualifying_tids[0])); t++) { + if (entry.tid_trends[t].consecutive_positive >= + (u8)required_hysteresis) { + qualifying_tids[qualifying_tid_count++] = entry.tid_trends[t].tid; + } + } + if (qualifying_tid_count == 0) { + TEST_LOG("LivenessTracker::selectLeakCandidates entry[%d] klass_id=%u " + "klass trend OK but no qualifying tid - skipped", + i, entry.klass_id); + continue; + } + if (count == cap && slope <= best_slopes[cap - 1]) { + // Already holding `cap` stronger (or equal) candidates - this one + // doesn't make the cut. + continue; + } + + int pos = count < cap ? count++ : cap - 1; + best_slopes[pos] = slope; + KlassCandidate &cand = out[pos]; + cand.klass_id = entry.klass_id; + cand.representative = + entry.representative_count > 0 ? entry.representatives[0] : nullptr; + cand.qualifying_tid_count = qualifying_tid_count; + memcpy(cand.qualifying_tids, qualifying_tids, + sizeof(jint) * (size_t)qualifying_tid_count); + while (pos > 0 && best_slopes[pos - 1] < best_slopes[pos]) { + double tmp_slope = best_slopes[pos - 1]; + best_slopes[pos - 1] = best_slopes[pos]; + best_slopes[pos] = tmp_slope; + KlassCandidate tmp_cand = out[pos - 1]; + out[pos - 1] = out[pos]; + out[pos] = tmp_cand; + pos--; + } + } + _table_lock.unlockShared(); + TEST_LOG("LivenessTracker::selectLeakCandidates returning %d candidates (required_hysteresis=%d, heapFloorRising=%d)", + count, required_hysteresis, + (int)heapFloorRising()); + return count; +} + +int LivenessTracker::topKlassesByGenerationCount(u32 *out, int max) { + int cap = max < MAX_LEAK_CANDIDATES ? max : MAX_LEAK_CANDIDATES; + if (cap <= 0) { + return 0; + } + + // Same insertion-sort-style top-k selection as selectLeakCandidates() + // above (small, fixed cap - cheaper than collecting everything and + // sorting), ranked by most-recent count_ring sample instead of slope, and + // with no hysteresis/trend gate at all - see this method's own header + // comment (livenessTracker.h). + u32 best_counts[MAX_LEAK_CANDIDATES]; + int count = 0; + + _table_lock.lockShared(); + for (int i = 0; i < _klass_population_size; i++) { + const KlassPopulationEntry &entry = _klass_population[i]; + if (entry.ring_fill == 0 || entry.stable_class_tag == 0) { + // Never sampled, or a live instance has not been resolved yet to mint + // its stable_class_tag from (foldKlassCountsLocked()'s own comment) - + // nothing usable to rank or return in either case. + continue; + } + // ring_head is "next slot to write" (recordKlassPopulationSampleLocked(), + // livenessTracker.cpp) - the most recently written slot is one behind it, + // wrapping. + u32 latest = entry.count_ring[(entry.ring_head + KLASS_POPULATION_RING_SIZE - 1) % + KLASS_POPULATION_RING_SIZE]; + if (count == cap && latest <= best_counts[cap - 1]) { + continue; + } + int pos = count < cap ? count++ : cap - 1; + best_counts[pos] = latest; + out[pos] = (u32)entry.stable_class_tag; + while (pos > 0 && best_counts[pos - 1] < best_counts[pos]) { + u32 tmp_count = best_counts[pos - 1]; + best_counts[pos - 1] = best_counts[pos]; + best_counts[pos] = tmp_count; + u32 tmp_id = out[pos - 1]; + out[pos - 1] = out[pos]; + out[pos] = tmp_id; + pos--; + } + } + _table_lock.unlockShared(); + TEST_LOG("LivenessTracker::topKlassesByGenerationCount returning %d klass_ids", count); + return count; +} + +jobject LivenessTracker::resolveCandidateRepresentative(JNIEnv *env, u32 klass_id) { + // Shared lock excludes cleanup_table()'s exclusive lock (the only writer, + // and the only place that can DeleteWeakGlobalRef() an entry's + // representatives via foldKlassCountsLocked()'s eviction path above) for + // the whole lookup+resolve, so the value NewLocalRef() runs on here is + // always the table's current one for klass_id, never a snapshot that + // eviction could have invalidated in the meantime - see + // selectLeakCandidates()'s own comment for the race this closes. + // + // Returns the first live representative (oldest first, per the Lindy + // bias in KlassCountScratch::oldest[]). Callers that need all live + // representatives (e.g. pollWatchedTargets() tagging all of them) + // use resolveCandidateRepresentatives() instead. + _table_lock.lockShared(); + jobject obj = nullptr; + for (int i = 0; i < _klass_population_size; i++) { + if (_klass_population[i].klass_id == klass_id) { + for (int r = 0; r < _klass_population[i].representative_count; r++) { + jweak rep = _klass_population[i].representatives[r]; + if (rep != nullptr) { + obj = env->NewLocalRef(rep); + if (obj != nullptr) { + break; + } + } + } + break; + } + } + _table_lock.unlockShared(); + return obj; +} + +int LivenessTracker::resolveCandidateRepresentatives( + JNIEnv *env, u32 klass_id, jobject *out, int max_out) { + // Returns all live representatives for klass_id, oldest first. + // Used by pollWatchedTargets() to tag all representatives with marker + // tags so the canary mechanism has multiple chances to find a + // long-lived instance. See KlassCountScratch::oldest's comment for + // why multiple representatives matter. + _table_lock.lockShared(); + int count = 0; + for (int i = 0; i < _klass_population_size; i++) { + if (_klass_population[i].klass_id == klass_id) { + for (int r = 0; r < _klass_population[i].representative_count && + count < max_out; r++) { + jweak rep = _klass_population[i].representatives[r]; + if (rep != nullptr) { + jobject obj = env->NewLocalRef(rep); + if (obj != nullptr) { + out[count++] = obj; + } + } + } + break; + } + } + _table_lock.unlockShared(); + return count; +} + void LivenessTracker::flush(std::set &tracked_thread_ids) { if (!_enabled) { // disabled @@ -112,16 +1683,33 @@ void LivenessTracker::flush_table(std::set *tracked_thread_ids) { event._alloc = _table[i].alloc; event._skipped = _table[i].skipped; event._ctx = _table[i].ctx; + event.leak_tag = _table[i].leak_tag; - jclass clz = env->GetObjectClass(ref); - jstring name_str = (jstring)env->CallObjectMethod(clz, _Class_getName); - env->DeleteLocalRef(clz); - jniExceptionCheck(env); - const char *name = env->GetStringUTFChars(name_str, nullptr); - int class_id = name != nullptr - ? Profiler::instance()->lookupClass(name, strlen(name)) - : 0; - env->ReleaseStringUTFChars(name_str, name); + int class_id = 0; + if (_table[i].cached_klass_id != 0) { + // Already resolved by cleanup_table()'s survivor loop this epoch + // (resolveKlassId(), only when _gc_generations is enabled) - reuse + // it instead of repeating the GetObjectClass+Class.getName()+ + // lookupClass() JNI round-trip for the same object. + class_id = _table[i].cached_klass_id; + } else { + jclass clz = env->GetObjectClass(ref); + jstring name_str = (jstring)env->CallObjectMethod(clz, _Class_getName); + env->DeleteLocalRef(clz); + jniExceptionCheck(env); + // name_str can be null if the call above threw and + // jniExceptionCheck() cleared the pending exception rather than + // propagating it - GetStringUTFChars()/ReleaseStringUTFChars() + // require a non-null jstring (mirrors resolveKlassId()'s own guard). + if (name_str != nullptr) { + const char *name = env->GetStringUTFChars(name_str, nullptr); + if (name != nullptr) { + class_id = Profiler::instance()->lookupClass(name, strlen(name)); + env->ReleaseStringUTFChars(name_str, name); + } + env->DeleteLocalRef(name_str); + } + } // lookupClass() returns -1 when the class map is at capacity; do not // assign it to the u32 event id (it would wrap to 0xFFFFFFFF and @@ -139,13 +1727,8 @@ void LivenessTracker::flush_table(std::set *tracked_thread_ids) { _table_lock.unlock(); if (_record_heap_usage) { - bool isLastGc = HeapUsage::isLastGCUsageSupported(); - size_t used = isLastGc ? HeapUsage::get()._used_at_last_gc - : loadAcquire(_used_after_last_gc); - if (used == 0) { - used = HeapUsage::get()._used; - isLastGc = false; - } + bool isLastGc; + size_t used = resolvePostGcHeapUsage(&isLastGc); Profiler::instance()->writeHeapUsage(used, isLastGc); } @@ -164,6 +1747,15 @@ Error LivenessTracker::initialize_table(JNIEnv *jni, int sampling_interval) { return Error("Can not track liveness for allocation samples without heap " "size information."); } + // Cached for secondsToOOM() - see _max_heap_bytes' own comment + // (livenessTracker.h) for why this is resolved once here rather than + // re-querying HeapUsage::getMaxHeap() on every projection. + _max_heap_bytes = max_heap; + // Cached the same way and for the same reason - see _container_memory_limit's + // own comment (livenessTracker.h). -1 (unavailable) is a valid outcome + // here, unlike max_heap above: not every JVM runs under a memory-limited + // cgroup. + _container_memory_limit = OS::getContainerMemoryLimit(); int required_table_capacity = sampling_interval > 0 ? max_heap / sampling_interval : max_heap; @@ -188,6 +1780,13 @@ Error LivenessTracker::start(Arguments &args) { if (err) { return err; } + // Initialize leak tag free list + for (int i = 0; i < LEAK_TAG_POOL_SIZE; i++) { + _leak_tag_free_list[i] = i; + _leak_tag_info[i].call_trace_id = 0; + _leak_tag_info[i].tid = 0; + } + _leak_tag_free_count = LEAK_TAG_POOL_SIZE; if (!_enabled) { // disabled return Error::OK; @@ -221,6 +1820,14 @@ void LivenessTracker::stop() { Error LivenessTracker::initialize(Arguments &args) { _enabled = args._gc_generations || args._record_liveness; + // Gates per-klass population tracking (see the _gc_generations member's + // own comment in livenessTracker.h). Updated unconditionally alongside + // _record_heap_usage below, ahead of the _initialized guard, for the same + // reason: each profiler start should observe the flag it was actually + // started with, even though the tracking table itself persists across + // recordings. + _gc_generations.store(args._gc_generations, std::memory_order_relaxed); + if (!_enabled) { return Error::OK; } @@ -240,6 +1847,26 @@ Error LivenessTracker::initialize(Arguments &args) { } _initialized = true; + // Sync the class-map-generation baseline to what it already is by this + // point, rather than leaving it at the constructor's 0 sentinel (see + // _last_class_map_generation's own comment, livenessTracker.h). By the time + // this runs, ObjectSampler::start() -> LivenessTracker::start() has already + // happened strictly after Profiler::start()'s own _class_map.clearAll() + // (referenceChains.cpp's own comment on this same ordering) - so + // classMap()->generation() here already reflects this process's first + // recording, not the pre-clearAll() baseline the 0 sentinel implies. + // Without this, cleanup_table()'s class-map-reset branch always sees a + // spurious mismatch (0 vs. whatever generation() has already reached) the + // very first time it runs after ANY start() - regardless of how much + // genuinely post-reset, still-valid population/leak-tracking history has + // already accumulated in _klass_population by then - and wipes it all, + // found the hard way via StaticFieldGrowingCollectionScenario silently + // losing its seeded candidate the moment the first post-start GC finished. + // A real subsequent generation bump (a later recording's own clearAll()) + // still trips the mismatch correctly, since by then this field holds + // whatever value cleanup_table() last actually observed, not this sentinel. + _last_class_map_generation = Profiler::instance()->classMap()->generation(); + if (VM::hotspot_version() < 11) { Log::warn("Liveness tracking requires Java 11+"); // disable liveness tracking @@ -275,6 +1902,11 @@ Error LivenessTracker::initialize(Arguments &args) { // decision in track() is an integer compare rather than a double multiply. _subsample = SubsampleRate(args._live_samples_ratio); + // Fresh recording: no chase is open, so no watched tids and no urgency + // boost may leak in from a previous recording's lifecycle. + __atomic_store_n(&_watched_tid_count, 0, __ATOMIC_RELEASE); + __atomic_store_n(&_urgent_tracking, false, __ATOMIC_RELEASE); + _table_size = 0; _table_cap = std::min(2048, _table_max_cap); // with default 512k sampling interval, it's @@ -305,6 +1937,116 @@ Error LivenessTracker::initialize(Arguments &args) { static ThreadLocal rng; static ThreadLocal skipped; +// track()'s admission gate: chase-phase admission boost (see +// admitForTracking()'s declaration comment in livenessTracker.h). Both boost +// paths precede the configured-ratio draw, so a boosted allocation never +// consumes an RNG draw and the non-boosted path keeps the exact +// reject-on-draw > ratio semantics (and per-thread RNG stream position drift +// only when boosts actually happen - harmless, the streams are independent +// per thread and probabilistic by design). +bool LivenessTracker::admitForTracking(jint tid) { + if (__atomic_load_n(&_urgent_tracking, __ATOMIC_ACQUIRE)) { + return true; + } + // Count+array two-phase publish (noteSelectedCandidates() writes the + // slots before release-storing the count): the acquire load pairs with + // that release store so every slot read below is at least as fresh as the + // count observed here - see the project's atomic memory ordering rule + // for why RELAXED is not an option on arm64. + int n = __atomic_load_n(&_watched_tid_count, __ATOMIC_ACQUIRE); + for (int i = 0; i < n; i++) { + if (_watched_tids[i] == tid) { + return true; + } + } + if (_subsample.ratio >= 1.0) { + return true; + } + u64 state = rng.get(); + if (state == 0) { + // Seeded on a thread's first tracked allocation and kept until its TLS is + // released at thread end or JNI detach: the tick keeps two threads that + // share a tid across that boundary on different streams, and the tid + // separates threads alive at the same time. A thread that outlives a + // recording keeps its stream rather than restarting it -- there is no + // per-recording reseed, because stop/start does not clear the slot. + state = xorshift::seed(TSC::ticks(), (u64)tid); + } + u64 draw = xorshift::next(state); + rng.set(state); + return draw < _subsample.threshold; +} + +// Publishes the current poll's qualifying tids as track()'s watched-admission +// set. Called from ReferenceChainTracker's pollWatchedTargets() with the FULL +// candidate selection (never the hasLeakSignal() max=1 probe - that one's +// partial view could drop tids of other still-active candidates), including +// the zero-candidate case, which clears the set: a tid left watched after the +// chase ends would keep admitting that thread at 100% forever (tids get +// recycled by the OS into unrelated threads), so the set must track the +// candidate selection exactly, poll by poll. +void LivenessTracker::noteSelectedCandidates(const KlassCandidate *candidates, + int count) { + jint tids[KlassCandidate::MAX_QUALIFYING_TIDS]; + int n = 0; + if (candidates != nullptr && count > 0) { + for (int i = 0; i < count; i++) { + const KlassCandidate &kc = candidates[i]; + for (int j = 0; j < kc.qualifying_tid_count; j++) { + jint tid = kc.qualifying_tids[j]; + bool dup = false; + for (int k = 0; k < n && !dup; k++) { + dup = (tids[k] == tid); + } + if (!dup) { + tids[n++] = tid; + if (n == KlassCandidate::MAX_QUALIFYING_TIDS) { + goto full; + } + } + } + } + } +full: + // Copy into the live array before publishing the count (two-phase + // publish mirrored by admitForTracking()'s acquire load). A reader + // mid-scan may transiently mix old and new slot values below the OLD + // count - harmless: admission is advisory, and the worst case is one + // allocation admitted per the previous poll's set. + for (int i = 0; i < n; i++) { + _watched_tids[i] = tids[i]; + } + __atomic_store_n(&_watched_tid_count, n, __ATOMIC_RELEASE); + if (n > 0) { + // TEST_LOG, not Log::debug: one line per poll on the reference-chain + // thread (never the allocation hot path), and pod-side verification of + // the boost engaging needs the tid list in the extracted lib's logs. + TEST_LOG("LivenessTracker::noteSelectedCandidates watched tids[%d]:", + n); + for (int i = 0; i < n; i++) { + TEST_LOG(" watched tid=%d", _watched_tids[i]); + } + } +} + +void LivenessTracker::setUrgentTracking(bool urgent) { + // Transition-only TEST_LOG (same pattern as ReferenceChainTracker::isUrgent()'s + // latch/release logging): the setter runs every threadLoop() iteration, so + // per-call logging would repeat the steady-state value line indefinitely. + bool prev = __atomic_exchange_n(&_urgent_tracking, urgent, __ATOMIC_ACQ_REL); + if (prev != urgent) { + TEST_LOG_SUMMARY("LivenessTracker::setUrgentTracking urgent=%d", (int)urgent); + } +} + +void LivenessTracker::admissionResetForTest() { + __atomic_store_n(&_urgent_tracking, false, __ATOMIC_RELEASE); + __atomic_store_n(&_watched_tid_count, 0, __ATOMIC_RELEASE); + _subsample = SubsampleRate(0.0); + rng.clear(); + skipped.set(0); +} + void LivenessTracker::releaseThreadLocalState() { rng.clear(); skipped.clear(); @@ -321,23 +2063,9 @@ void LivenessTracker::track(JNIEnv *env, AllocEvent &event, jint tid, return; } - if (_subsample.ratio < 1.0) { - u64 state = rng.get(); - if (state == 0) { - // Seeded on a thread's first tracked allocation and kept until its TLS is - // released at thread end or JNI detach: the tick keeps two threads that - // share a tid across that boundary on different streams, and the tid - // separates threads alive at the same time. A thread that outlives a - // recording keeps its stream rather than restarting it -- there is no - // per-recording reseed, because stop/start does not clear the slot. - state = xorshift::seed(TSC::ticks(), (u64)tid); - } - u64 draw = xorshift::next(state); - rng.set(state); - if (draw >= _subsample.threshold) { - skipped.set(skipped.get() + static_cast(event._weight) * event._size); - return; - } + if (!admitForTracking(tid)) { + skipped.set(skipped.get() + static_cast(event._weight) * event._size); + return; } jweak ref = env->NewWeakGlobalRef(object); @@ -370,7 +2098,9 @@ void LivenessTracker::track(JNIEnv *env, AllocEvent &event, jint tid, skipped.set(0); _table[idx].age = 0; _table[idx].call_trace_id = call_trace_id; + _table[idx].leak_tag = 0; _table[idx].ctx = ContextApi::snapshot(); + _table[idx].cached_klass_id = 0; } _table_lock.unlockShared(); @@ -381,8 +2111,10 @@ void LivenessTracker::track(JNIEnv *env, AllocEvent &event, jint tid, retried = true; // try cleanup before resizing - there is a good chance it will free some - // space - cleanup_table(true); + // space. allow_resolve=false: this runs synchronously on the + // allocation-sampling callback stack (see cleanup_table()'s own header + // comment for why resolveKlassId() is unsafe here). + cleanup_table(true, false); if (_table_cap < _table_max_cap) { @@ -426,11 +2158,51 @@ void LivenessTracker::track(JNIEnv *env, AllocEvent &event, jint tid, } } +void LivenessTracker::maybeForceCleanup(u64 now_ns) { + if (!_enabled || !_gc_generations.load(std::memory_order_relaxed)) { + return; + } + constexpr u64 FORCE_CLEANUP_INTERVAL_NS = 30ULL * 1000 * 1000 * 1000; + u64 last_cleanup_ns = load(_last_cleanup_ns); + if (now_ns - last_cleanup_ns < FORCE_CLEANUP_INTERVAL_NS) { + return; + } + if (load(_gc_epoch) == load(_last_gc_epoch)) { + // Nothing happened since the last sweep (organic, forced, or a prior + // call to this method) - re-walking an unchanged table would just + // re-fold the same survivor counts into this epoch's scratch, skewing + // the slope computed from it. Leave _last_cleanup_ns alone so the next + // wake keeps checking at the same ~1s cadence rather than restarting a + // fresh 30s wait with nothing to show for it. + return; + } + store(_last_cleanup_ns, now_ns); + cleanup_table(true, true); +} + void JNICALL LivenessTracker::GarbageCollectionFinish(jvmtiEnv *jvmti_env) { ProfiledThread::initCurrentThreadSignalSafe(); LivenessTracker::instance()->onGC(); } +size_t LivenessTracker::resolvePostGcHeapUsage(bool *out_is_last_gc) { + bool isLastGc = HeapUsage::isLastGCUsageSupported(); + size_t used = isLastGc ? HeapUsage::get()._used_at_last_gc + : loadAcquire(_used_after_last_gc); + TEST_LOG("LivenessTracker::resolvePostGcHeapUsage isLastGc=%d used_at_last_gc=%zu", + (int)isLastGc, used); + if (used == 0) { + used = HeapUsage::get(false)._used; + isLastGc = false; + TEST_LOG("LivenessTracker::resolvePostGcHeapUsage used==0, falling back to HeapUsage::get(false)._used=%zu", + used); + } + if (out_is_last_gc != nullptr) { + *out_is_last_gc = isLastGc; + } + return used; +} + void LivenessTracker::onGC() { if (!_initialized) { return; @@ -442,6 +2214,32 @@ void LivenessTracker::onGC() { if (!HeapUsage::isLastGCUsageSupported()) { store(_used_after_last_gc, HeapUsage::get(false)._used); } + + if (_gc_generations.load(std::memory_order_relaxed)) { + // Feeds heapFloorRising()'s corroboration check (selectLeakCandidates()) + // - gated on _gc_generations, same as the per-klass population table + // itself, since this ring exists purely to support that feature. + // recordHeapFloorSample() itself checks _heap_floor_recording_disabled_for_test + // (debug-only) so a test can seed the ring exclusively. + size_t used = resolvePostGcHeapUsage(nullptr); + TEST_LOG_SUMMARY("LivenessTracker::onGC recording heap floor used=%zu gc_epoch=%llu", + used, (unsigned long long)load(_gc_epoch)); + if (used > 0) { + // A failed read (-1, e.g. transient /sys/fs/cgroup access error) is + // recorded as 0 rather than skipping the sample outright - see + // _container_mem_ring's own comment (livenessTracker.h) for why this + // ring must stay index-aligned with _heap_floor_ring/ + // _heap_floor_time_ring. Harmless when _container_memory_limit is + // itself unavailable (secondsToOOM() never selects this ring then), + // and a rare, self-correcting blip otherwise (the next successful + // read re-establishes the real growth rate). + long container_usage = OS::getContainerMemoryUsage(); + recordHeapFloorSample((u64)used, OS::nanotime(), + container_usage >= 0 ? (u64)container_usage : 0); + } else { + TEST_LOG_SUMMARY("LivenessTracker::onGC used<=0, skipping heap floor record"); + } + } } void LivenessTracker::getLiveTraceIds(CallTraceIdSet& out_buffer) { diff --git a/ddprof-lib/src/main/cpp/livenessTracker.h b/ddprof-lib/src/main/cpp/livenessTracker.h index a100126831..addabedae0 100644 --- a/ddprof-lib/src/main/cpp/livenessTracker.h +++ b/ddprof-lib/src/main/cpp/livenessTracker.h @@ -8,11 +8,13 @@ #include "arch.h" #include "callTraceHashTable.h" +#include "classTagAllocator.h" #include "context.h" #include "engine.h" #include "event.h" #include "spinLock.h" #include "xorshift.h" +#include #include #include #include @@ -28,18 +30,19 @@ typedef struct TrackingEntry { jint tid; jlong time; jlong age; + jlong leak_tag; // 0 = untagged; otherwise a tag from the leak tag pool Context ctx; + // Set by cleanup_table()'s survivor loop via resolveKlassId() when + // _gc_generations is enabled (0 otherwise, or if resolution failed - 0 is + // StringDictionary's own "no entry" sentinel, so a real id is never 0). + // flush_table() reuses this instead of re-resolving the same object's + // class via a second GetObjectClass+Class.getName()+lookupClass() JNI + // round-trip - an object's class never changes, so a value resolved here + // stays valid for flush_table()'s later read of the same entry. track() + // resets this to 0 for every newly tracked entry. + u32 cached_klass_id; } TrackingEntry; -// The liveness subsampling rate, held as the ratio and the xorshift draw -// threshold derived from it. -// -// The two have to move together: the threshold decides which allocations are -// kept, while the ratio is what Recording reports into the JFR chunk, so a -// threshold left behind by a ratio change samples at one rate and claims -// another -- silently, and in a direction no assertion would notice. The only -// constructor derives the threshold from the ratio, so a caller cannot set one -// without the other; assigning a new SubsampleRate replaces both. struct SubsampleRate { double ratio; u64 threshold; @@ -48,6 +51,142 @@ struct SubsampleRate { : ratio(subsample_ratio), threshold(xorshift::threshold(subsample_ratio)) {} }; +// Fixed-capacity, LRU-evicted per-klass population history, keyed by klass +// StringDictionary id (Profiler::classMap(), the same id TrackingEntry/ +// AllocEvent resolves lazily today only at flush time, flush_table() below). +// This is the data doc/architecture/LiveHeapReferenceChains.md's Open +// Question 3 "positive population-slope ranking" proposal needs: a rolling +// window of how many tracked instances of a klass are alive at each GC +// epoch, plus one representative instance to chase a chain for if the trend +// looks leak-shaped (see selectLeakCandidates() below for the ranking, and +// referenceChains.cpp's pollWatchedTargets() for how a ranked candidate gets +// consumed - this struct only stores the raw history). +typedef struct KlassPopulationEntry { + u32 klass_id; // StringDictionary id; 0 means "unused slot" (0 is + // also StringDictionary's own "no entry" sentinel, + // so a real id is never 0 - see resolveKlassId()). + // Up to MAX_REPRESENTATIVES_PER_KLASS representative instances of this + // klass, biased toward the oldest surviving instances (highest GC age). + // The first live one is used by resolveCandidateRepresentative(); all live + // ones are tagged by pollWatchedTargets(). See KlassCountScratch::oldest's + // comment for why multiple representatives matter. + static constexpr int MAX_REPRESENTATIVES_PER_KLASS = 3; + jweak representatives[MAX_REPRESENTATIVES_PER_KLASS]; + jint rep_tids[MAX_REPRESENTATIVES_PER_KLASS]; // tid each rep was minted from + int representative_count; + u32 count_ring[30]; // ring buffer of per-epoch generation counts: the + // number of distinct GC ages among a klass's + // surviving tracked instances that epoch (see + // accumulateKlassCount(), livenessTracker.cpp). + u8 ring_head; // next slot to write + u8 ring_fill; // samples written so far, caps at 30 + // Number of consecutive epochs (most recent first) for which + // LivenessTracker::hasQualifyingGrowth() found this entry's ring to show a + // leak-shaped rise, updated every time a new sample is pushed + // (recordKlassPopulationSampleLocked()) - reset to 0 the moment a single + // epoch fails the test. selectLeakCandidates() requires this to reach a + // hysteresis threshold before trusting the klass, rather than acting on + // one qualifying epoch alone: a population merely oscillating (no net + // growth) satisfies a single-epoch test on roughly half of all epochs, so + // without this counter it gets reported as a leak candidate almost as + // often as a real leak does. + u8 consecutive_positive; + // Slope (recent third's mean minus earliest third's mean) as of the last + // push, computed and cached by hasQualifyingGrowth() alongside + // consecutive_positive above - selectLeakCandidates() reads this directly + // for ranking instead of re-scanning the ring: the ring only changes on + // push, so a second scan at scan time would just recompute the same + // value. Meaningless (left at its previous value, or 0 for a newly + // created entry) whenever ring_fill < KLASS_POPULATION_MIN_FILL_FOR_TREND + // - callers must check ring_fill first, exactly as before this field + // existed. + mutable double cached_slope; + u64 last_updated_epoch; // _gc_epoch value as of the last write, for LRU + // eviction when the table is full + // Stable per-class identifier, from the process-wide, negative-tag + // allocator shared with ReferenceChainTracker (classTagAllocator.h) - NOT + // the same value as klass_id above. klass_id (Profiler::classMap()'s + // dictionary id) can end up different for the exact same class depending + // on when/which subsystem resolves it, because that dictionary can be + // compacted/regenerated independently of this table - found the hard way + // correlating ReferenceChainTracker::FrontierEntry::referrer_klass values + // (also a classMap id, resolved at a different time by a different + // subsystem) against a LivenessTracker-reported growing klass_id: the + // exact same class ("[B" in the reproducing case) resolved to two + // different classMap ids depending on which subsystem asked. This field + // exists specifically so cross-subsystem klass matching (referenceChains.h's + // own _watched_leak_klass_ids) has something both sides can agree on + // regardless of classMap's own housekeeping. 0 until minted (see + // foldKlassCountsLocked()'s own comment for when that happens) - always + // negative once minted (ClassTagAllocator::next()'s own convention). + jlong stable_class_tag; + // Per-(klass,tid) trend qualification (the disjoint-tagged-vs-frontier + // pod finding: a whole-klass rising generation count can come from churn + // spread across MANY allocating threads, each of which retains a STABLE + // handful of instances - observed live on the hotdog pod where the only + // qualifying [B candidate's tagged instances were 24-16KB machinery + // byte[]s whose per-site retention never rose). Each TidTrend holds one + // allocating thread's per-epoch count of distinct surviving GC ages (the + // same generation-cardinality signal the klass ring above uses, at + // per-thread granularity), and selectLeakCandidates() only reports the + // klass as a candidate if at least one tid here shows a sustained rise of + // its own - continuous retention concentrated in one thread is the leak + // shape; retention spread thinly and stably across threads is machinery. + // A real leak filled from many threads still qualifies: every thread that + // keeps adding surviving instances grows its own per-tid age span. The + // age trend alone cannot see one-cohort-per-thread accumulation (each + // one-shot thread's instances share one age, so its distinct-age count + // stays 1 forever), which is why qualification is that trend OR the + // retained-count bar (TID_RETAINED_COUNT_BAR): machinery threads retain + // neither rising age spans nor a large surviving count, and fail both. + static constexpr int MAX_TID_TRENDS = 8; + static constexpr int TID_TREND_RING_SIZE = 16; + struct TidTrend { + jint tid; // allocating thread id (TrackingEntry::tid's space) + u8 ring[TID_TREND_RING_SIZE]; // per-epoch distinct surviving age counts + u8 count_ring[TID_TREND_RING_SIZE]; // per-epoch surviving tracked- + // instance counts (same head/fill as ring - + // always pushed together; feeds the + // TID_RETAINED_COUNT_BAR discriminator) + u8 ring_head; + u8 ring_fill; + u8 consecutive_positive; // sustained-qualification hysteresis, same + // push-time update pattern as the klass + // consecutive_positive above + // Seeded by tidTrendRecordForTest() only: the real fold + // (recordTidTrendSamplesLocked()) never marks entries synthetic, and + // exempts synthetic entries from both its zero-pushes-for-absent-tids + // decay and its slot-eviction search - a scenario's seeded ramp would + // otherwise be wiped/reset by the very next real GC's fold. Production + // data has no synthetic entries, so the exemption is inert there. + bool synthetic; + }; + TidTrend tid_trends[MAX_TID_TRENDS]; + int tid_trend_count; +} KlassPopulationEntry; + +// One leak-candidate result from selectLeakCandidates() below: the klass to +// chase, a currently-live representative instance of it, and the threads +// whose per-tid trends qualified it - ready to hand to +// referenceChains.cpp's pollWatchedTargets() (the design doc's Open Question +// 3 bridging step). Deliberately excludes the slope/rank that produced the +// ranking - the caller only needs identity, matching the design doc's own +// "KlassCandidate { u32 klass_id; jweak representative; }" sketch, extended +// with the qualifying tids selectLeakCandidates() now requires. +typedef struct KlassCandidate { + u32 klass_id; + jweak representative; + // The tids whose per-tid trends cleared the same hysteresis gate that + // qualified this klass (selectLeakCandidates() fills this; empty never + // happens for a returned candidate). tagLeakInstances() only tags + // tracked instances of klass_id allocated by one of these threads - + // the per-(klass,tid) tagging scope that keeps machinery churn of the + // same class from consuming the leak-tag pool. + static constexpr int MAX_QUALIFYING_TIDS = KlassPopulationEntry::MAX_TID_TRENDS; + jint qualifying_tids[MAX_QUALIFYING_TIDS]; + int qualifying_tid_count; +} KlassCandidate; + // Aligned to satisfy SpinLock member alignment requirement (64 bytes) // Required because this class contains SpinLock _table_lock member class alignas(alignof(SpinLock)) LivenessTracker { @@ -58,6 +197,112 @@ class alignas(alignof(SpinLock)) LivenessTracker { constexpr static int MAX_TRACKING_TABLE_SIZE = 262144; constexpr static int MIN_SAMPLING_INTERVAL = 524288; // 512kiB + // _klass_population/_klass_count_scratch below are both scanned linearly + // (lookup and LRU-eviction search) - this size keeps every such scan cheap + // enough that a plain linear scan is fine, rather than requiring an index. + constexpr static int MAX_KLASS_POPULATION_ENTRIES = 256; + // Design doc's Open Question 3 proposal: "ring buffer of up to 30 recent + // population counts". + constexpr static int KLASS_POPULATION_RING_SIZE = 30; + // Design doc's Open Question 3 proposal: "Only trust the trend once the + // window has a minimum fill (e.g. ≥10 samples) to avoid noise right after + // a klass starts being tracked." + constexpr static int KLASS_POPULATION_MIN_FILL_FOR_TREND = 10; + // Per-tid trend gate's own minimum fill (see KlassPopulationEntry:: + // TidTrend): 6 samples of the 16-slot ring leaves a 2-3 sample thirds + // comparison, small enough that a per-tid qualification (6 pushes + + // the 3-5 hysteresis epochs, ~9-11) completes no later than the klass + // gate's own ~13-15-epoch latency, keeping the klass ring the sole + // latency driver for candidate emergence. + constexpr static int TID_TREND_MIN_FILL_FOR_TREND = 6; + // Per-tid qualification's second discriminator, OR-ed with the age + // trend: a thread whose surviving tracked-instance count of THIS ONE + // klass clears this bar qualifies without waiting out the trend/hysteresis + // latency. Covers the shape the age-trend gate structurally cannot see - + // one-cohort-per-thread accumulation (each one-shot worker thread's + // instances all share one age, so its distinct-age count stays 1 forever, + // but it retains hundreds of instances - LeakingCacheScenario's exact + // shape), and shortens qualification for any big per-thread retained set. + // 8 tracked survivors of a single class from one thread is roughly an + // order of magnitude above the machinery shapes this gate exists to + // exclude (buffer/thread pools retain 1-5 tracked per class/thread, + // observed on the hotdog pod's tagged machinery byte[]s); the live-heap + // subsample makes real instance counts ~10x the tracked count, so the bar + // corresponds to ~80+ retained same-class instances from one thread. + constexpr static u32 TID_RETAINED_COUNT_BAR = 8; + // secondsToOOM() below corroborates the full-window heap-floor regression + // with a second regression over just the most recent half of the same + // ring, and requires both to show a positive slope. Guards against a + // one-time step change that immediately plateaus (e.g. a cache warming up + // once at startup) - the full-window fit alone stays "rising" for as long + // as the step's samples remain in the window, but the recent half's own + // slope collapses back to ~0 as soon as growth actually stops, well + // before the full window ages the step out. fill/2 is always >= + // HEAP_FLOOR_RECENT_HALF_MIN_FILL once the full-window fit itself has + // cleared KLASS_POPULATION_MIN_FILL_FOR_TREND (10), so this never blocks + // a reading the full-window check wouldn't already have blocked. + constexpr static int HEAP_FLOOR_RECENT_HALF_MIN_FILL = 5; + // Design doc's Open Question 3 proposal: "seed only the top 3-5 by trend + // magnitude" - this is the upper end of that range. selectLeakCandidates() + // also honors the caller-supplied `max`, so the effective cutoff is + // min(max, MAX_LEAK_CANDIDATES, ); + // "no separate budget constant is needed" per the design doc, this top-N + // cutoff doubles as the per-pass seeding cap. + // (Moved to public section for ReferenceChainTracker access.) + + // --- Sustained-trend gate (hasQualifyingGrowth() below) --- + // count_ring holds, per epoch, the number of distinct GC ages + // (generations) among a klass's surviving tracked instances + // (accumulateKlassCount(), livenessTracker.cpp) rather than the raw + // surviving instance count: a klass whose survivors keep spanning more + // distinct allocation cohorts over time is one where old instances are + // not dying as new ones arrive, which is the leak shape this gate looks + // for. The gate itself is a single condition - the recent third's mean + // generation count must exceed the earliest third's by a meaningful + // margin (LEAK_GROWTH_REL_MIN/LEAK_GROWTH_ABS_MIN, whichever is larger). + // An earlier revision of this gate also required the recent third's + // *minimum* to exceed the earliest third's minimum (a floor-rise check, + // to reject oscillations whose peak alone passes the growth test) - that + // check was tuned for raw population counts (which can run into the + // thousands) and does not transfer to generation counts, which are small + // integers bounded by how many distinct cohorts a klass can realistically + // accumulate; it was dropped rather than re-tuned. + constexpr static double LEAK_GROWTH_REL_MIN = 0.15; + // Absolute floor for the growth bar: the slope must exceed + // max(LEAK_GROWTH_REL_MIN * earliest_mean, LEAK_GROWTH_ABS_MIN). + constexpr static int LEAK_GROWTH_ABS_MIN = 1; + + // Required number of consecutive qualifying epochs + // (KlassPopulationEntry::consecutive_positive) before selectLeakCandidates() + // trusts a klass as a leak candidate. Lower (CORROBORATED) when the + // aggregate post-GC live heap (heapFloorRising() below) is independently + // showing a sustained rise of its own over the same horizon - that is + // whole-heap evidence this klass's growth isn't an isolated artifact + // (redistribution/churn that nets out heap-wide, or per-klass sampling + // noise), so fewer of this klass's own epochs are needed to trust it. + // heapFloorRising() is a single call per selectLeakCandidates() scan, not + // per candidate: the aggregate heap has no per-klass attribution, so it + // cannot single out which klass (if any) is responsible for its rise - + // it can only raise or lower the bar for every candidate in that scan + // uniformly, never reorder them against each other. + constexpr static int LEAK_TREND_HYSTERESIS_BASE = 5; + constexpr static int LEAK_TREND_HYSTERESIS_CORROBORATED = 3; + + // --- Aggregate post-GC heap floor (heapFloorRising() below) --- + // Same "mean-of-thirds growth + floor rise" shape as the per-klass test + // above, applied to a single global ring of post-GC live heap size + // instead of one klass's sampled population - see this class's own ring + // (_heap_floor_ring below). Its own thresholds are deliberately looser + // (fraction-of-heap, not fraction-of-one-klass): this signal is diluted by + // every other klass's allocation activity (a leak far smaller than these + // thresholds is invisible against the rest of the heap), so it is not + // sensitive enough to gate on directly - it is used only as the + // LEAK_TREND_HYSTERESIS_BASE/CORROBORATED selector above. + constexpr static double HEAP_FLOOR_GROWTH_REL_MIN = 0.02; + constexpr static u64 HEAP_FLOOR_GROWTH_ABS_MIN = 1ULL << 20; // 1MiB + constexpr static double HEAP_FLOOR_FLOOR_REL_MIN = 0.01; + constexpr static u64 HEAP_FLOOR_FLOOR_ABS_MIN = 1ULL << 19; // 512KiB + bool _initialized; bool _enabled; Error _stored_error; @@ -70,6 +315,14 @@ class alignas(alignof(SpinLock)) LivenessTracker { SubsampleRate _subsample; + // Chase-phase admission-boost state (admitForTracking()'s declaration + // comment below). _watched_tids is two-phase-published: slots first, then + // _watched_tid_count with RELEASE (admitForTracking()'s ACQUIRE load pairs + // with it) - a reader never trusts a slot beyond the count it observed. + jint _watched_tids[KlassCandidate::MAX_QUALIFYING_TIDS]; + volatile int _watched_tid_count; + volatile bool _urgent_tracking; + bool _record_heap_usage; jclass _Class; @@ -78,12 +331,252 @@ class alignas(alignof(SpinLock)) LivenessTracker { volatile u64 _gc_epoch; volatile u64 _last_gc_epoch; + // Timestamp (OS::nanotime()) of the last cleanup_table() sweep that + // actually ran, whether organic (flush_table()'s JFR cadence) or forced + // (track()'s table-overflow branch). Read/written only by + // maybeForceCleanup() below - see that method's own comment for why a + // third, time-based trigger is needed on top of those two. + volatile u64 _last_cleanup_ns; + size_t _used_after_last_gc; + // Ring of post-GC live heap sizes, one sample per GC epoch, feeding + // heapFloorRising() below - same shape as KlassPopulationEntry::count_ring + // but a single global instance rather than one per klass, and lock-free + // rather than _table_lock-guarded: onGC() runs from the JVMTI + // GarbageCollectionFinish callback, which can fire synchronously mid-way + // through a JNI upcall this class itself is making while already holding + // _table_lock (e.g. cleanup_table()'s Class.getName() call, if that + // allocation triggers a GC) - taking the same lock here would risk a + // self-deadlock on a non-reentrant SpinLock. GC completions are never + // concurrent with each other (HotSpot never runs two GCs at once), so + // onGC() is always a single writer at a time, matching the existing + // lock-free _gc_epoch/_used_after_last_gc fields' own assumption - see + // recordHeapFloorSample()/heapFloorRising() (livenessTracker.cpp) for the + // load/store ordering this relies on. + u64 _heap_floor_ring[KLASS_POPULATION_RING_SIZE]; + // OS::nanotime() paired index-for-index with _heap_floor_ring above (same + // head/fill, always pushed together by recordHeapFloorSample() - see that + // method's comment) - heapFloorRising() itself has no use for elapsed + // time, but secondsToOOM() below needs it to turn the ring's byte growth + // into a rate rather than just a magnitude. + u64 _heap_floor_time_ring[KLASS_POPULATION_RING_SIZE]; + // Container (cgroup) memory usage, one sample per GC epoch, pushed to the + // same head/fill index as _heap_floor_ring/_heap_floor_time_ring above by + // the same recordHeapFloorSample() call - so this ring shares + // _heap_floor_time_ring's timestamps rather than needing its own. Exists + // because container memory can grow from causes _heap_floor_ring never + // sees at all (native/off-heap allocations - direct buffers, JNI/native + // libraries) - see secondsToOOM()'s own comment for how this ring and + // _heap_floor_ring are used as two independent OOM boundaries. + u64 _container_mem_ring[KLASS_POPULATION_RING_SIZE]; + volatile u8 _heap_floor_ring_head; + volatile u8 _heap_floor_ring_fill; + + // Debug-only test seam: when true, onGC() skips recordHeapFloorSample() + // so a test can seed the ring exclusively via heapFloorRecordForTest() + // without a real GC interleaving a sample with a real OS::nanotime() + // timestamp and real heap usage, corrupting secondsToOOM()'s projection. + // See AggressiveLeakReferenceChainTest's own comment for the race this + // closes (a real GC firing between resetKlassPopulationForTest0() and + // shouldRunPassForTest0()). +#ifdef DEBUG + // Atomic (not plain bool) so the GC thread sees the test thread's store + // on arm64's weak memory model. Checked inside recordHeapFloorSample() + // at the actual write point, not just in onGC(), so a GC already past the + // onGC() gate when the test sets this flag still cannot corrupt the ring. + std::atomic _heap_floor_recording_disabled_for_test{false}; +#endif + + // Runtime.maxMemory(), resolved once by initialize_table() (the same call + // that already requires it to enable liveness tracking at all - see that + // method's own Error path) and cached here so secondsToOOM() - polled once + // per ReferenceChainTracker::threadLoop wake, ~1s - does not repeat + // HeapUsage::getMaxHeap()'s handful of JNI calls on every poll. A JVM's max + // heap does not change at runtime, so a value resolved once stays valid. + // -1 if never resolved (mirrors getMaxHeap()'s own sentinel). + jlong _max_heap_bytes; + +#ifdef DEBUG + // Atomic mirror used only by the setMaxHeapBytesForTest() seam so the + // test thread's store is published to the BFS thread's secondsToOOM() + // read on arm64's weak memory model. _max_heap_bytes above stays plain + // for the production path (written once from the control thread during + // initialize_table(), before the BFS thread starts, so no cross-thread + // visibility issue there). + std::atomic _max_heap_bytes_for_test{-1}; +#endif + + // OS::getContainerMemoryLimit(), resolved once by initialize_table() next + // to _max_heap_bytes above (same call site, same "resolved once, does not + // change at runtime" reasoning) and cached here so secondsToOOM() does not + // re-walk the cgroup hierarchy on every poll. -1 if never resolved, or if + // this process is not running under a memory-limited cgroup (bare metal, + // macOS, cgroups disabled) - secondsToOOM() treats -1 as "unbounded" so + // this boundary never wins over the heap-based one. + jlong _container_memory_limit; + +#ifdef DEBUG + // Mirrors _max_heap_bytes_for_test above for the same reason: lets a test + // exercise secondsToOOM()'s container boundary without a real cgroup. + std::atomic _container_memory_limit_for_test{-1}; +#endif + + // Gates the per-klass population tracking below. Set from + // args._gc_generations in initialize() - deliberately not folded into + // _enabled (which also covers plain _record_liveness): this doesn't + // resolve the design doc's own "still undecided" bullet under Open + // Question 3 by itself, but the plan built on top of this table requires + // liveness tracking *and* _gc_generations, matching the doc's stated + // fallback of "no target-seeding" when generations tracking isn't on + // (arguments.cpp:223-227,244). std::atomic (relaxed) since initialize() + // writes it from the control thread while the BFS thread + // (maybeForceCleanup()) and the GC-callback thread (cleanup_table()) can + // still be reading it from a session that persists across a restart. + std::atomic _gc_generations; + + // Per-klass population history table (see KlassPopulationEntry above). + // Populated only from cleanup_table()'s GC-epoch-advance pass, never from + // track() (the allocation sampling hot path) - see + // accumulateKlassCount()/foldKlassCountsLocked() below. Guarded by + // _table_lock, the same lock cleanup_table() already holds for the + // duration of its epoch-advance pass, rather than adding a second lock. + KlassPopulationEntry _klass_population[MAX_KLASS_POPULATION_ENTRIES]; + int _klass_population_size; + + // Scratch space reused across cleanup_table() calls (a member field, not a + // per-call stack/heap allocation - cleanup_table() runs on a GC-signal + // cadence, not the allocation hot path, but this codebase's + // allocation-free preference still applies wherever avoiding an + // allocation is cheap) to accumulate this epoch's per-klass surviving + // counts before folding them into _klass_population's ring buffers at the + // end of the pass. + typedef struct KlassCountScratch { + u32 klass_id; + // Distinct GC ages (generations) of surviving tracked instances + // of this klass at this epoch. The size of this vector + // is the klass' generation count. + std::vector ages; + // Top-N oldest surviving instances of this klass seen this epoch, + // sorted by age descending. Used by foldKlassCountsLocked() to mint + // representatives biased toward long-lived instances (Lindy effect: + // the oldest surviving instances are the most likely to be leaks). + // Fixed-size to avoid heap allocation in the GC callback path. + static constexpr int MAX_OLDEST_SAMPLES = 3; + struct OldestSample { + jweak ref; + u32 age; + jint tid; // allocating thread of this instance + }; + OldestSample oldest[MAX_OLDEST_SAMPLES]; + int oldest_count; + // Per-thread generation tracking (Cork/Swat heuristic adapted): + // track distinct surviving GC ages per allocating thread (tid) + // within this klass. The thread with the most distinct surviving + // generations is the strongest leak signal — it reuses the same + // generation-count signal that selectLeakCandidates() uses + // per-class, applied at per-thread granularity within a class. + // A thread with 12 distinct surviving ages (continuous leak) + // outscores a thread with 1 age (one-time burst). + // + // Thread ID is used instead of call_trace_id because lambdas + // fragment call_trace_id — synthetic methods produce slightly + // different stack hashes for what is logically one allocation + // site, yielding N sites × 1 generation instead of 1 site × N + // generations. Thread ID is stable and naturally separates leak + // threads from noise threads. + static constexpr int MAX_THREADS_PER_KLASS = 16; + static constexpr int MAX_AGES_PER_THREAD = 32; + struct ThreadGens { + jint tid; + u32 ages[MAX_AGES_PER_THREAD]; // sorted distinct ages + u32 age_count; + u32 count; // surviving tracked instances of this klass this epoch + // (increments per object, before the age-dedup early + // return below) - feeds the per-tid retained-count bar + // (TID_RETAINED_COUNT_BAR) + }; + ThreadGens threads[MAX_THREADS_PER_KLASS]; + int thread_count; + } KlassCountScratch; + KlassCountScratch _klass_count_scratch[MAX_KLASS_POPULATION_ENTRIES]; + int _klass_count_scratch_size; + + // Profiler::classMap()'s generation as of the last cleanup_table() call + // that checked it, mirroring ReferenceChainTracker::_last_class_map_generation + // (referenceChains.h). Profiler::start() calls _class_map.clearAll() + // (profiler.cpp) whenever `reset || _start_time == 0`, restarting that + // StringDictionary's id namespace at 1 - but TrackingEntry::cached_klass_id + // and _klass_population's klass_id keys are ids resolved from that + // dictionary, and both survive stop()/start() cycles (this class's table is + // designed to persist across recordings). Left unguarded, an id cached + // before a reset would silently collide with whatever unrelated class the + // new generation reassigns that same id to. cleanup_table() compares this + // against Profiler::instance()->classMap()->generation() and, on a + // mismatch, drops every such cached id before resuming. Constructor- + // initialized to 0 (StringDictionary's own initial generation) but + // immediately re-synced in initialize() to whatever generation() already is + // by then - NOT left at 0, despite 0 also being correct for a + // never-yet-reset classMap in isolation: ObjectSampler::start() -> + // LivenessTracker::start() always runs strictly after Profiler::start()'s + // own _class_map.clearAll() (referenceChains.cpp's comment on this same + // ordering), so by the time initialize() runs, generation() has *already* + // bumped once for this process's first recording. Leaving the 0 sentinel + // in place here would make cleanup_table()'s very first call always see a + // spurious mismatch against that already-happened bump, discarding + // whatever genuinely post-reset population history had already + // accumulated by then - found the hard way (see initialize()'s own + // comment). + u64 _last_class_map_generation; + + // --- Leak tag pool --- + // Reusable pool of JVMTI tags for directly tagging tracked leaking + // objects. Tags are in range [LEAK_TAG_BASE, LEAK_TAG_BASE+POOL_SIZE). + // When a tracked object is GC'd, its tag is returned to the pool. + // This lets the BFS find the exact leaking objects (not just any + // instance of the same class) and correlate chains with HeapLiveObject. + static constexpr int LEAK_TAG_POOL_SIZE = 256; + static constexpr jlong LEAK_TAG_BASE = 0x40000000LL; + int _leak_tag_free_list[LEAK_TAG_POOL_SIZE]; + int _leak_tag_free_count; + // Side table: for each tag in the pool, the (call_trace_id, tid) of + // the tracked object it was assigned to. Used by ReferenceChainTracker + // for coverage tracking (adaptive CPU budget). + struct LeakTagInfo { + u64 call_trace_id; + jint tid; + }; + LeakTagInfo _leak_tag_info[LEAK_TAG_POOL_SIZE]; + + jlong acquireLeakTag(u64 call_trace_id, jint tid); + void releaseLeakTag(jlong tag); + Error initialize(Arguments &args); Error initialize_table(JNIEnv *jni, int sampling_interval); - void cleanup_table(bool force = false); + // force=true is used by track()'s table-overflow branch to run a cleanup + // synchronously from the allocation-sampling call stack, bypassing the + // GC-epoch-changed check below. The per-klass population tracking below + // (_gc_generations) runs on both paths, once per genuinely new GC epoch + // (see "is_epoch_owner" in livenessTracker.cpp). + // + // allow_resolve gates resolveKlassId() - a real Class.getName() + // Java-bytecode upcall, unlike the plain native JVMTI calls already made + // elsewhere on track()'s callback stack - independently of force: force + // only says "bypass the epoch-unchanged early-exit", it says nothing about + // which call stack this is running on. track()'s hot-path call passes + // force=true, allow_resolve=false (too costly/re-entrancy-prone to resolve + // from the SampledObjectAlloc callback stack - reuses whatever + // cached_klass_id an entry already picked up from an earlier resolving + // sweep, or skips accounting for that entry this epoch if it was never + // resolved). flush_table()/stop() pass the defaults (force=false, + // allow_resolve=true) - the original organic, GC-cadence path. LivenessTracker::maybeForceCleanup() passes force=true, + // allow_resolve=true: it runs on ReferenceChainTracker's own background + // thread (referenceChains.cpp), not the allocation hot path, so the same + // upcalls flush_table() already makes safely are just as safe there - see + // that method's own comment for why a third caller needs both bypassing + // the early-exit *and* resolution. + void cleanup_table(bool force = false, bool allow_resolve = true); void flush_table(std::set *tracked_thread_ids); @@ -92,11 +585,205 @@ class alignas(alignof(SpinLock)) LivenessTracker { jlong getMaxMemory(JNIEnv *env); + // Resolves the best available post-GC heap usage sample, mirroring + // flush_table()'s own resolution order (JDK17+ exact + // CollectedHeap::_used_at_last_gc when supported, otherwise onGC()'s own + // _used_after_last_gc snapshot, falling back to a live usage read if + // neither has produced anything yet, e.g. before the first GC). Shared by + // flush_table()'s JFR event and onGC()'s heap-floor ring sample so both + // read the same value the same way. Returns 0 only if HeapUsage itself has + // nothing to offer. *out_is_last_gc (if non-null) reports which case was + // used, for callers (flush_table()) that need to say so in the JFR event. + size_t resolvePostGcHeapUsage(bool *out_is_last_gc); + + // --- Per-klass population tracking (cleanup_table()'s epoch-advance pass only) --- + + // Resolves the StringDictionary id for `ref`'s class, mirroring + // flush_table()'s existing class-name resolution above (GetObjectClass + + // Class.getName() + Profiler::lookupClass()) - this is the "genuinely new + // cost on an existing pass" the design doc flags, previously paid only at + // JFR-flush time. Returns 0 (StringDictionary's own "no entry" sentinel) + // if the name could not be resolved or interned. + u32 resolveKlassId(JNIEnv *env, jobject ref); + + // Increments klass_id's running sample count in _klass_count_scratch for + // the epoch currently being processed, creating a new scratch slot (with + // `sample_source` remembered for a possible new KlassPopulationEntry) if + // this is the first surviving instance of this klass seen so far this + // epoch. No-op if the scratch table is already full and klass_id is not + // present - the same fixed-capacity/best-effort tradeoff + // _klass_population's own table already accepts, one level up. + void accumulateKlassCount(u32 klass_id, jlong age, jweak sample_source, + jint tid); + void insertOldestSample(KlassCountScratch &scratch, jweak sample_source, + u32 age, jint tid); + void insertThreadGen(KlassCountScratch &scratch, jint tid, u32 age); + + // Pushes `count` into klass_id's ring buffer, creating the entry (evicting + // the least-recently-updated entry first if the table is already at + // MAX_KLASS_POPULATION_ENTRIES capacity - the same evict-LRU-on-insert- + // when-full shape NativeSocketSampler's fd cache already solves, + // nativeSocketSampler.h:141-142/184's insertFdAddrLocked(), and the same + // "single agent-owned pass, lock already held by caller" shape + // cleanup_table() itself already uses) if klass_id has never been seen. + // A newly-created entry's `representative` is left null - it is the + // caller's job (foldKlassCountsLocked(), which owns the JNIEnv this + // method deliberately does not touch) to fill it in, which keeps this + // method free of any JNI call and therefore directly exercisable by gtest + // without a live JVM. On return, *out_slot is the table slot used for + // klass_id and *out_created is true iff a new entry was created (an + // evicted-and-reused slot counts as "created", since the old klass_id's + // data was fully replaced). Returns the evicted entry's representative + // jweak (nullptr if nothing was evicted, or the evicted entry had none) + // so the caller can DeleteWeakGlobalRef() it. + // Precondition: _table_lock is held (by cleanup_table(), the only + // production caller). + jweak recordKlassPopulationSampleLocked(u32 klass_id, u32 count, u64 epoch, + int *out_slot, bool *out_created, + jweak *out_evicted = nullptr, + int *out_evicted_count = nullptr, + int max_evicted = 0); + + // Drains _klass_count_scratch into _klass_population for the epoch that + // just finished, minting a fresh representative jweak (from each entry's + // KlassCountScratch::sample_source) for klasses not already present, and + // retrying the mint for existing entries whose representative is still + // null (a previous epoch's mint attempt can fail if sample_source died in + // the window between cleanup_table()'s survival check and the mint - see + // this method's own retry-condition comment, livenessTracker.cpp) - see + // recordKlassPopulationSampleLocked()'s comment for why that JNI work + // happens here rather than inside it. A fresh weak global ref + // is used instead of aliasing sample_source directly because + // sample_source is the corresponding TrackingEntry's own jweak: that + // entry's slot in _table is reused (and its jweak deleted via + // DeleteWeakGlobalRef) the moment the tracked object dies and + // cleanup_table() reaps it, which would leave _klass_population holding a + // dangling handle if it aliased the same jweak. Resets + // _klass_count_scratch_size to 0 once drained. Called with _table_lock + // held, at the end of cleanup_table()'s epoch-advance pass. + // allow_resolve mirrors resolveKlassId()'s own parameter (cleanup_table()'s + // header comment): when false, this runs synchronously on the JVMTI + // SampledObjectAlloc callback stack (track()'s table-overflow branch), so + // the representative-minting NewLocalRef/NewWeakGlobalRef/DeleteLocalRef + // churn below is skipped - the ring/count bookkeeping still happens, and a + // missing representative is retried on the next allow_resolve=true sweep + // (see the retry-condition comment in livenessTracker.cpp). + void foldKlassCountsLocked(JNIEnv *env, u64 epoch, bool allow_resolve); + + // Get-or-mints slot's stable_class_tag (see that field's own comment) from + // `instance`'s class, if not already minted for this slot's current + // occupant. Shared by foldKlassCountsLocked() (the production allocation- + // sampling path) and klassPopulationSetRepresentativeForTest() (the + // debug-only test seam StaticFieldGrowingCollectionScenario-style + // external-process tests drive instead of real sampling) - both hand this + // a live representative instance to resolve the class from, so neither + // needs its own copy of the GetObjectClass/GetTag/SetTag sequence. No-op + // if slot is out of range, instance is null, or a tag is already minted. + void mintStableClassTagIfNeeded(JNIEnv *env, int slot, jobject instance); + + // --- Slope computation and candidate ranking (selectLeakCandidates() below) --- + + // Per-tid sustained-trend gate half #1: the same mean-of-thirds growth + // test hasQualifyingGrowth() below applies to a klass's ring, at + // KlassPopulationEntry::TidTrend granularity (TID_TREND_MIN_FILL_FOR_TREND + // samples of that smaller ring, same LEAK_GROWTH_REL_MIN/ABS_MIN growth + // bar - per-tid age-cardinality is the same small-integer signal the + // klass gate already uses, so the same bar transfers). Pure read over the + // trend's ring; no cached slope (nothing ranks tids against each other by + // slope - tagLeakInstances ranks its candidates by the live tracking + // table's own age diversity, not by this history). + bool hasQualifyingTidGrowth(const KlassPopulationEntry::TidTrend &trend) const; + + // Per-tid qualification for ONE epoch push: the age-trend test above OR + // the just-pushed surviving-instance count clearing + // TID_RETAINED_COUNT_BAR (the one-cohort-per-thread accumulation shape + // the age trend structurally cannot see - see that constant's own + // comment). The caller applies the result to the trend's own + // consecutive_positive, the same push-time pattern + // recordKlassPopulationSampleLocked() uses for the klass-level counter; + // the required hysteresis then makes both discriminators SUSTAINED + // (a thread over the bar must stay over it for required_hysteresis + // consecutive epochs - a transient burst that GCs away resets). + bool tidPushQualifies(const KlassPopulationEntry::TidTrend &trend, + u32 current_count) const; + + // Folds one epoch's per-thread scratch (KlassCountScratch::threads - + // per-tid distinct surviving GC ages, already maintained by + // accumulateKlassCount()/insertThreadGen()) into slot's per-tid trend + // rings: present tids push their epoch count; tracked tids ABSENT this + // epoch push 0 (a thread whose instances all died must not keep a stale + // rising ring - the 0 push fails hasQualifyingTidGrowth() and resets that + // trend's consecutive_positive, the per-tid analogue of the population + // simply stopping); synthetic (test-seeded) trends are exempt from that + // decay and from eviction. New tids beyond MAX_TID_TRENDS evict the + // non-synthetic trend with the lowest consecutive_positive (then the + // lowest ring_fill) - a genuinely rising leak tid accumulates hysteresis + // fast and resists eviction, machinery tids never do. Best-effort caveat: + // KlassCountScratch caps per-klass threads at 16 > MAX_TID_TRENDS, so a + // klass with more than 16 allocating threads in one epoch can miss a + // tracked tid from the scratch and push it a spurious 0 - the same + // fixed-capacity best-effort the scratch itself already accepts. + // _table_lock held exclusively by the caller (foldKlassCountsLocked()). + void recordTidTrendSamplesLocked(int slot, const KlassCountScratch &scratch); + + // The sustained-trend gate (this class's own header comment above, + // "Sustained-trend gate") - both-required growth-magnitude and floor-rise + // tests, design doc's explicit "mean of thirds" choice over full + // least-squares regression (cheap, allocation-free, one pass over the + // ring, no sorting or extra storage). A single scan + // (ringThirdsStats(), livenessTracker.cpp) both derives the pass/fail + // result below AND updates entry.cached_slope (recent third's mean minus + // earliest third's mean) for selectLeakCandidates()'s ranking, rather than + // that method re-scanning the same unchanged ring a moment later. Returns + // false (leaving entry.cached_slope untouched) if entry.ring_fill is below + // KLASS_POPULATION_MIN_FILL_FOR_TREND - not enough history yet to trust a + // trend; callers must check ring_fill themselves before trusting + // cached_slope, exactly as they checked this method's own return value + // before cached_slope existed. + // + // Called from recordKlassPopulationSampleLocked() every time a new sample + // is pushed (both the production path, + // foldKlassCountsLocked()->recordKlassPopulationSampleLocked(), and the + // klassPopulationRecordForTest() test seam that calls the same method + // directly), so KlassPopulationEntry::consecutive_positive/cached_slope + // are always kept in sync with the ring they summarize, regardless of + // caller. + bool hasQualifyingGrowth(const KlassPopulationEntry &entry) const; + + // Pushes `used`/`timestamp_ns`/`container_used` into _heap_floor_ring/ + // _heap_floor_time_ring/_container_mem_ring - see those members' own + // comments for why this is lock-free rather than _table_lock-guarded. + // Called only from onGC() (single-writer-at-a-time, same comment), which + // supplies OS::nanotime() and OS::getContainerMemoryUsage() explicitly + // rather than this method calling them internally - keeps this method + // itself deterministic for the heapFloorRecordForTest() test seam below. + void recordHeapFloorSample(u64 used, u64 timestamp_ns, u64 container_used); + + // Internal: pushes to the rings without checking + // _heap_floor_recording_disabled_for_test. Called by + // recordHeapFloorSample() (after the debug-only flag check) and by + // heapFloorRecordForTest() (which bypasses the check so a test can + // seed the rings even while real GC recording is disabled). + void recordHeapFloorSampleUnchecked(u64 used, u64 timestamp_ns, u64 container_used); + + // Reads whether the aggregate post-GC live heap has itself shown a + // sustained rise over _heap_floor_ring's horizon - see + // LEAK_TREND_HYSTERESIS_BASE/CORROBORATED's own comment above for how + // selectLeakCandidates() uses this (a uniform hysteresis-threshold + // selector for the whole scan, never a per-candidate veto or boost). + bool heapFloorRising() const; + public: static LivenessTracker *instance() { static LivenessTracker instance; return &instance; } + + // Public read accessor for the auto-tuner (ReferenceChainTracker::autoTuneDefaults) + // and secondsToOOM(). See _max_heap_bytes's own comment above. + jlong maxHeapBytes() const { return _max_heap_bytes; } + + constexpr static int MAX_LEAK_CANDIDATES = 5; // Delete copy constructor and assignment operator to prevent copies LivenessTracker(const LivenessTracker&) = delete; LivenessTracker& operator=(const LivenessTracker&) = delete; @@ -104,24 +791,683 @@ class alignas(alignof(SpinLock)) LivenessTracker { LivenessTracker() : _initialized(false), _enabled(false), _stored_error(Error::OK), _table_size(0), _table_cap(0), _table_max_cap(0), _table(NULL), - _subsample(0.1), _record_heap_usage(false), _Class(NULL), + _subsample(0.1), _watched_tid_count(0), _urgent_tracking(false), + _record_heap_usage(false), _Class(NULL), _Class_getName(0), _gc_epoch(0), _last_gc_epoch(0), - _used_after_last_gc(0) {} + _last_cleanup_ns(0), _used_after_last_gc(0), + _heap_floor_ring_head(0), _heap_floor_ring_fill(0), + _max_heap_bytes(-1), _container_memory_limit(-1), + _gc_generations(false), + _klass_population_size(0), _klass_count_scratch_size(0), + _last_class_map_generation(0), + _leak_tag_free_count(LEAK_TAG_POOL_SIZE) {} Error start(Arguments &args); void stop(); void track(JNIEnv *env, AllocEvent &event, jint tid, jobject object, u64 call_trace_id); + + // track()'s admission gate: a chase-phase raise of the live-samples + // tracking probability over the configured _subsample ratio (default 10% - + // a 90% probabilistic drop whose thinning of small per-(klass, tid) + // populations was observed live as the intermittent zero-tag runs in + // LeakTagCorrelationReferenceChainTest - see that test's own comment for + // the full lottery analysis). Two independent raises, both advisory-only + // (a missed raise just falls back to the configured ratio's behavior): + // - watched tids: noteSelectedCandidates() publishes + // selectLeakCandidates()'s qualifying tids - exactly the (klass, tid) + // scope tagLeakInstances() tags and the reference-chain chase intercepts + // - and admitForTracking() admits them at 100%. Bounded by the candidate + // threads' own allocation rate (at most MAX_QUALIFYING_TIDS threads), so + // the tracking table's volume/cleanup cost scales with the leak's own + // threads, not the process's whole allocation rate. + // - urgency: setUrgentTracking() from ReferenceChainTracker::threadLoop()'s + // seconds_to_oom ramp admits everything. Under the OOM ramp the process + // is expected to die soon; maximizing what the last chapter captures + // outweighs the tracking table's transient volume. + // Fail-open by construction: the boost only ever adds admissions on top of + // the configured ratio - a stale or missed boost cannot drop an allocation + // that the ratio would have admitted. + bool admitForTracking(jint tid); + + // Publishes the current candidate poll's qualifying tids as the watched + // set above. Called from ReferenceChainTracker's full poll + // (pollWatchedTargets()) every wake - including the zero-candidate case, + // which clears the set (a tid left watched after the chase ends would keep + // admitting that thread at 100% across OS tid reuse). NOT called from + // hasLeakSignal()'s max=1 probe: that partial view could silently drop + // other still-active candidates' tids. + void noteSelectedCandidates(const KlassCandidate *candidates, int count); + + // Urgency raise toggle - see admitForTracking()'s comment above. Set every + // threadLoop() iteration with that iteration's urgency computation, so the + // boost tracks the ramp exactly (both engaging and releasing). + void setUrgentTracking(bool urgent); + void flush(std::set &tracked_thread_ids); - // Frees this thread's subsampling RNG state (track()'s gen/dis/skipped + // Frees this thread's subsampling RNG state (track()'s rng/skipped // ThreadLocals, livenessTracker.cpp). Must be called from a thread that is // about to detach/terminate - see those ThreadLocal's own comment for why // their pthread-key destructors alone cannot be relied on for JNI-attached // threads. Safe to call even if this thread never called track(). static void releaseThreadLocalState(); + // Reads the per-klass population histories (_klass_population) and + // writes up to `max` leak candidates into `out`: klasses whose recent + // population trend is positive (growing), ranked by trend magnitude + // descending, capped at MAX_LEAK_CANDIDATES regardless of `max` (design + // doc's Open Question 3 "top 3-5" cutoff). Returns the number of + // candidates written (0 if _gc_generations was never enabled - + // _klass_population stays empty in that case, since population tracking + // is gated on it, so no separate guard is needed here). Called on demand + // by the BFS-pass poll, not on any timer of its own; does no JNI work, so it is safe + // to call from any thread that can take _table_lock (mirrors + // getLiveTraceIds()'s own shared-lock read pattern, livenessTracker.cpp). + // + // The `representative` jweak copied into KlassCandidate here is a snapshot + // only - callers MUST NOT resolve it directly (e.g. via NewLocalRef()) + // after this method has returned and _table_lock released. This table's + // LRU eviction (recordKlassPopulationSampleLocked(), livenessTracker.cpp) + // can DeleteWeakGlobalRef() that exact handle at any point afterwards + // (from cleanup_table()'s epoch-advance pass, running on a different + // thread), which invalidates the handle - a later NewLocalRef() on it is + // undefined behavior per the JNI spec, not merely "returns null". Use + // resolveCandidateRepresentative() below instead, which re-reads the + // table's current value for klass_id atomically with the resolve. + int selectLeakCandidates(KlassCandidate *out, int max); + + // Tag the tracked instances of the given leak candidates - only the + // instances allocated by a candidate's QUALIFYING tids + // (KlassCandidate::qualifying_tids, filled by selectLeakCandidates() + // from the per-tid trends that cleared the same hysteresis gate) - with + // JVMTI tags from the leak tag pool. Called by pollWatchedTargets() + // with this poll's candidate list. The tid scope is the fix for the + // disjoint-tagged-vs-frontier pod finding's pool-economy half: a leak + // klass's tracked population also contains machinery instances of the + // same class from other threads, and klass-wide tagging spent pool tags + // on those (observed on hotdog: 247 tagged, all machinery byte[]s with + // flat per-site retention, zero ever intercepted) while the real + // leak-site instances churned out of the pool. The BFS recognizes these + // tags by range check (isLeakTag) and admits the specific objects into + // the frontier, storing the leak tag for correlation with HeapLiveObject + // events. Returns the number of tags assigned. + int tagLeakInstances(jvmtiEnv *jvmti, const KlassCandidate *candidates, + int candidate_count); + + // Look up the (call_trace_id, tid) recorded for a leak tag. Returns + // false if the tag is not a valid leak tag or has been returned to the + // pool. Used by ReferenceChainTracker for coverage tracking. + bool getLeakTagInfo(jlong tag, u64 *out_call_trace_id, + jint *out_tid) const; + + // Reads _klass_population and writes up to `max` STABLE CLASS TAGS + // (KlassPopulationEntry::stable_class_tag - NOT the classMap dictionary + // klass_id selectLeakCandidates() above deals in; see that field's own + // comment for why the distinction matters) into `out`, ranked by MOST + // RECENT count_ring sample (the "generation count" - accumulateKlassCount()'s + // own comment: the number of distinct GC ages among a klass's surviving + // tracked instances - livenessTracker.cpp) descending, capped at + // MAX_LEAK_CANDIDATES. Unlike selectLeakCandidates() above, this applies + // NO trend/hysteresis gate at all (no ring_fill minimum beyond "at least + // one sample", no consecutive_positive requirement, no positive-slope + // requirement) - it is meant to be called only once + // ReferenceChainTracker::hasLeakSignal() has ALREADY fired via the + // slower, hysteresis-gated selectLeakCandidates() path, as a faster, + // broader follow-up ranking that does not itself need to wait out that + // same hysteresis a second time for ReferenceChainTracker's own rotation- + // priority use (see referenceChains.h's own comment on + // _watched_leak_klass_ids for why). Same shared-lock read pattern as + // selectLeakCandidates(); a klass whose ring is entirely empty (never + // sampled) or whose stable_class_tag has not been minted yet (no live + // instance resolved so far - foldKlassCountsLocked()'s own comment) is + // skipped, since neither has anything usable to rank or return. Returns + // the number of tags written. + int topKlassesByGenerationCount(u32 *out, int max); + + // Re-reads klass_id's current representative from _klass_population and + // resolves it to a fresh JNI local ref, both under the same _table_lock + // critical section - closes the race selectLeakCandidates()'s own comment + // above describes: a KlassCandidate snapshot returned by that method can + // go stale (LRU-evicted and DeleteWeakGlobalRef()'d) at any point before a + // caller gets around to resolving it. Looking the entry up again by + // klass_id here, under lock, guarantees NewLocalRef() only ever runs on a + // representative jweak this table still actually owns at the moment of the + // call: if klass_id has since been evicted (or was never assigned a + // representative), the lookup simply fails to find it and this returns + // nullptr without ever touching the stale handle. Returns nullptr if + // klass_id is no longer present, has no representative yet, or the + // representative's referent has since been collected (NewLocalRef() on a + // jweak returns null in that case, JNI spec). Mirrors the shared-lock read + // pattern selectLeakCandidates()/getLiveTraceIds() already use. + jobject resolveCandidateRepresentative(JNIEnv *env, u32 klass_id); + int resolveCandidateRepresentatives(JNIEnv *env, u32 klass_id, + jobject *out, int max_out); + + // Exposes the _gc_generations gate (see that member's own comment) so a + // caller outside this class - ReferenceChainTracker::pollWatchedTargets() + // (referenceChains.cpp), PROF-15341's LivenessTracker-to-ReferenceChainTracker + // bridging step - can skip + // calling selectLeakCandidates() entirely when the feature isn't in use, + // rather than relying on that method's own "returns 0" fallback to make + // the no-op cheap. Read-only; this accessor never toggles the flag. + bool gcGenerationsEnabled() const { + return _gc_generations.load(std::memory_order_relaxed); + } + + // Time-to-OOM projection against whichever of two independent boundaries + // is tighter: the JVM heap (_heap_floor_ring vs _max_heap_bytes) or the + // container/cgroup memory limit (_container_mem_ring vs + // _container_memory_limit). The two rings are pushed together (same + // head/fill index, shared _heap_floor_time_ring timestamps - see + // _container_mem_ring's own comment) so this compares _max_heap_bytes + // against _container_memory_limit up front (the latter treated as + // unbounded when unavailable) and runs the "mean of thirds" rate + // extrapolation (allocation-free, one ring scan) only once, against + // whichever limit is smaller - not once per boundary. This matters + // because container memory can grow from causes the heap-floor ring never + // sees at all (native/off-heap allocations), so a heap-only projection can + // under-warn right up until an OOM-kill from container memory pressure + // that had nothing to do with heap occupancy. + // + // Exists because selectLeakCandidates()'s per-klass gate + // (KLASS_POPULATION_MIN_FILL_FOR_TREND ring samples plus + // LEAK_TREND_HYSTERESIS_BASE/CORROBORATED consecutive qualifying epochs) + // can take longer to trust a candidate than a fast leak has left before + // OOM - this gives ReferenceChainTracker::hasLeakSignal() an independent, + // rate-based signal to start a search immediately instead of waiting on + // that gate. Not a claim that growth stays linear, only that a + // short-horizon linear extrapolation is a reasonable urgency signal. + // Returns a negative value if the chosen ring is not filled enough yet + // (gcGenerationsEnabled() is off, or too few GC epochs have happened), not + // rising, or neither _max_heap_bytes nor _container_memory_limit was ever + // resolved - callers must treat any non-positive return as "no projection + // available", not "zero seconds". Returns 0 if the chosen ring's recent + // mean has already reached its limit. + // + // Known limitation, inherited from heapFloorRising() rather than + // introduced here: cleanup_table()'s class-map-reset branch clears + // _klass_population but never _heap_floor_ring/_heap_floor_time_ring/ + // _container_mem_ring, so a stop()/start() gap with a real wall-clock + // pause in between can still mix pre-gap and post-gap samples into the + // same window. heapFloorRising() only risked a magnitude error from this; + // this method additionally divides by elapsed time, so the same gap + // understates the growth rate (overstates the projected time-to-OOM) + // rather than the reverse - not solved here. + double secondsToOOM() const; + + // Third trigger for cleanup_table(), alongside track()'s table-overflow + // branch (forced) and flush_table()'s JFR-flush cadence (organic): those + // two both depend on ObjectSampler's allocation-sampling callback firing + // often enough. ObjectSampler::updateConfiguration()'s PID controller + // throttles the JVMTI heap sampling interval toward a fixed target *event + // rate*, not a fixed *byte* rate - under sustained, fast heap growth this + // can push the interval high enough that SampledObjectAlloc (and therefore + // track()) stops firing in practice, starving cleanup_table() of both its + // forced trigger and the per-klass population samples + // selectLeakCandidates()'s slope computation needs. If that happens, the + // history cleanup_table() would otherwise have advanced goes stale and + // ReferenceChainTracker::hasLeakSignal() can never see a positive trend + // again, no matter how much the leaking population actually grows. + // + // Called once per ReferenceChainTracker::threadLoop wake (~1s cadence, see + // referenceChains.cpp) with a live JNIEnv already in hand - a convenient, + // already-existing periodic tick, not a new thread. No-ops unless both: + // (a) at least 30s have passed since the last cleanup_table() sweep + // (organic, forced, or one run by this method), and (b) at least one GC + // has happened since then (gcEpoch() != _last_gc_epoch) - so this never + // does a pointless sweep of an unchanged table. + void maybeForceCleanup(u64 now_ns); + static void JNICALL GarbageCollectionFinish(jvmtiEnv *jvmti_env); + // Test seams - not part of the production API. Mirrors + // NativeSocketSampler's own "for testing only" accessors + // (nativeSocketSampler.h's fdAddrCacheSizeForTest()/ + // fdAddrCacheInsertForTest()) rather than befriending the test binary. + // These only exercise the JNI-free ring/eviction mechanics + // (recordKlassPopulationSampleLocked() takes no JNIEnv), never + // foldKlassCountsLocked()'s representative-minting step, which needs a + // live JVM and is therefore out of gtest's reach. + int klassPopulationSizeForTest() const { return _klass_population_size; } + + // Leak-tag pool test seams - same "for testing only" rationale as the + // klass-population seams above: the pool acquire/release/info mechanics + // are JNI-free pure logic, so they are directly testable; only + // tagLeakInstances() itself needs a live JVM (SetTag/NewLocalRef) and stays + // out of gtest's reach. + void leakTagPoolResetForTest() { + for (int i = 0; i < LEAK_TAG_POOL_SIZE; i++) { + _leak_tag_free_list[i] = i; + _leak_tag_info[i].call_trace_id = 0; + _leak_tag_info[i].tid = 0; + } + _leak_tag_free_count = LEAK_TAG_POOL_SIZE; + } + + jlong acquireLeakTagForTest(u64 call_trace_id, jint tid) { + return acquireLeakTag(call_trace_id, tid); + } + + void releaseLeakTagForTest(jlong tag) { releaseLeakTag(tag); } + + int leakTagFreeCountForTest() const { return _leak_tag_free_count; } + + // Admission-boost test seams - same "for testing only" rationale as the + // leak-tag pool seams above: admitForTracking()/noteSelectedCandidates() + // are JNI-free pure logic (atomic reads + the per-thread RNG draw), so they + // are directly testable; only track() beyond the gate needs a live JNIEnv + // (NewWeakGlobalRef) and stays out of gtest's reach. + bool admitForTrackingForTest(jint tid) { return admitForTracking(tid); } + + void setSubsampleRatioForTest(double ratio) { _subsample = SubsampleRate(ratio); } + + // Reset for tests: clears both boost paths, forces ratio=0 (deterministic + // reject for unboosted tids - xorshift::threshold(0) is 0, which no draw + // compares <, so a ratio of 0 can never admit), and returns this thread's + // rng ThreadLocal to the unseeded sentinel so the next admitForTracking() + // draw comes from a freshly-seeded stream. + void admissionResetForTest(); + + int watchedTidCountForTest() const { + return __atomic_load_n(&_watched_tid_count, __ATOMIC_ACQUIRE); + } + + jint watchedTidForTest(int i) const { return _watched_tids[i]; } + + static jlong leakTagBaseForTest() { return LEAK_TAG_BASE; } + + static int leakTagPoolSizeForTest() { return LEAK_TAG_POOL_SIZE; } + bool klassPopulationLookupForTest(u32 klass_id, KlassPopulationEntry *out) const { + for (int i = 0; i < _klass_population_size; i++) { + if (_klass_population[i].klass_id == klass_id) { + *out = _klass_population[i]; + return true; + } + } + return false; + } + // recordKlassPopulationSampleLocked()'s own precondition is "_table_lock is + // held (by cleanup_table(), the only production caller)" - this seam is + // called from a Java/test thread while the BFS thread + // (ReferenceChainTracker::threadLoop()) may concurrently be inside + // cleanup_table()'s epoch-advance pass, which holds _table_lock while + // mutating the very same _klass_population/_klass_population_size fields + // (including its class-map-generation-reset branch, which can wipe the + // whole table). Without taking the lock here too, a seeded sample could + // race that wipe and silently vanish moments after being recorded. Mirrors + // klassPopulationSetRepresentativeForTest() below, which already does this + // correctly. + // TEST-ONLY synthetic-to-real klass-id aliasing. Since the leak-tag pool + // redesign, every consumer of a candidate klass id keys on the REAL id + // space of Profiler::lookupClass()/ReferenceChainTracker's class tags: + // tagLeakInstances() scans the live-heap tracking table's cached_klass_id, + // and discovered-instance recording resolves an admitted object's class + // to the same space. A purely synthetic seeded id matches none of those + // (observed live: tagged=0 for every poll, "resolved but no candidate + // match" for every auto-mark, no chains ever built). The scenarios' debug + // seams therefore alias each synthetic id to the representative's real + // class id, established by klassPopulationSetRepresentativeForTest() + // below (the only seam holding an actual instance) and applied by + // klassPopulationRecordForTest() above. + struct TestKlassAlias { + u32 synthetic; + u32 real; + }; + static constexpr int MAX_TEST_KLASS_ALIASES = 8; + TestKlassAlias _test_klass_aliases[MAX_TEST_KLASS_ALIASES]; + int _test_klass_alias_count = 0; + + // All three below require _table_lock held (same as every other + // _klass_population mutator). + u32 resolveTestKlassAliasLocked(u32 klass_id) const { + for (int i = 0; i < _test_klass_alias_count; i++) { + if (_test_klass_aliases[i].synthetic == klass_id) { + return _test_klass_aliases[i].real; + } + } + return klass_id; + } + + void registerTestKlassAliasLocked(u32 synthetic, u32 real) { + for (int i = 0; i < _test_klass_alias_count; i++) { + if (_test_klass_aliases[i].synthetic == synthetic) { + _test_klass_aliases[i].real = real; + return; + } + } + if (_test_klass_alias_count < MAX_TEST_KLASS_ALIASES) { + _test_klass_aliases[_test_klass_alias_count++] = {synthetic, real}; + } + } + + void removeKlassPopulationEntryLocked(int slot) { + if (slot < 0 || slot >= _klass_population_size) { + return; + } + memmove(&_klass_population[slot], &_klass_population[slot + 1], + sizeof(KlassPopulationEntry) * + (size_t)(_klass_population_size - slot - 1)); + _klass_population_size--; + } + + jweak klassPopulationRecordForTest(u32 klass_id, u32 count, u64 epoch, + int *out_slot, bool *out_created, + jweak *out_evicted = nullptr, + int *out_evicted_count = nullptr, + int max_evicted = 0) { + _table_lock.lock(); + klass_id = resolveTestKlassAliasLocked(klass_id); + jweak evicted = recordKlassPopulationSampleLocked(klass_id, count, epoch, + out_slot, out_created, + out_evicted, + out_evicted_count, + max_evicted); + _table_lock.unlock(); + return evicted; + } + + // Seeds one per-tid trend sample for tests/scenarios: pushes `count` as + // tid's epoch sample on klass_id's KlassPopulationEntry, marking the + // trend SYNTHETIC (exempt from the real fold's absent-tid decay and slot + // eviction - see TidTrend::synthetic's own comment) so a scenario's ramp + // survives the interleaved real GC folds that the same test's + // System.gc() churn triggers. The tid MUST be the allocating thread's real + // profiler tid (ProfiledThread::currentTid()'s space - + // JavaProfiler.getTid() on the leaking thread) whenever the test also + // relies on tagLeakInstances() tagging real tracked instances: the + // production tagging scope is exactly the qualifying-tid set, and a fake + // tid matches no tracked instance. Creates the klass entry if absent + // (count-0/epoch-0 first sample, same creation branch + // klassPopulationSetRepresentativeForTest() relies on) and resolves the + // synthetic-id aliasing exactly like klassPopulationRecordForTest() + // above. Out of gtest's reach in one respect: gtest call sites have no + // live tracked instances to tag anyway (they exercise the qualification + // gate only), so a distinct gtest-chosen tid is fine there. + void tidTrendRecordForTest(u32 klass_id, jint tid, u32 count, u64 epoch) { + _table_lock.lock(); + klass_id = resolveTestKlassAliasLocked(klass_id); + int slot = -1; + for (int i = 0; i < _klass_population_size; i++) { + if (_klass_population[i].klass_id == klass_id) { + slot = i; + break; + } + } + if (slot < 0) { + int out_slot; + bool created; + recordKlassPopulationSampleLocked(klass_id, 0, 0, &out_slot, &created); + slot = out_slot; + } + KlassPopulationEntry &entry = _klass_population[slot]; + KlassPopulationEntry::TidTrend *trend = nullptr; + for (int i = 0; i < entry.tid_trend_count; i++) { + if (entry.tid_trends[i].tid == tid) { + trend = &entry.tid_trends[i]; + break; + } + } + if (trend == nullptr && entry.tid_trend_count < + KlassPopulationEntry::MAX_TID_TRENDS) { + trend = &entry.tid_trends[entry.tid_trend_count++]; + trend->tid = tid; + trend->ring_head = 0; + trend->ring_fill = 0; + trend->consecutive_positive = 0; + } + if (trend != nullptr) { + trend->synthetic = true; + // The one seeded value lands in BOTH the age ring and the retained- + // count ring, so a test can qualify a tid either way the production + // gate does: a rising small-value ramp exercises the age-trend + // discriminator, while one flat value over TID_RETAINED_COUNT_BAR + // exercises the retained-count bar (the one-cohort-per-thread + // accumulation shape - see that constant's own comment). + trend->ring[trend->ring_head] = (u8)count; + trend->count_ring[trend->ring_head] = (u8)count; + trend->ring_head = (u8)((trend->ring_head + 1) % + KlassPopulationEntry::TID_TREND_RING_SIZE); + if (trend->ring_fill < KlassPopulationEntry::TID_TREND_RING_SIZE) { + trend->ring_fill++; + } + if (tidPushQualifies(*trend, count)) { + if (trend->consecutive_positive < UINT8_MAX) { + trend->consecutive_positive++; + } + } else { + trend->consecutive_positive = 0; + } + (void)epoch; // the per-tid ring is push-ordered like the klass ring; + // recordKlassPopulationSampleLocked()'s own callers + // likewise only use epoch for last_updated_epoch LRU + } + _table_lock.unlock(); + } + // Sets an entry's representative directly - production code only ever + // does this via foldKlassCountsLocked()'s JNI-dependent minting step + // (out of gtest's reach, see the class comment above), so tests use this + // seam instead to set up a fake representative and assert it comes back + // out of recordKlassPopulationSampleLocked() as the evicted jweak when + // that entry is later LRU-evicted. No-op if klass_id is not present. + // Also called from a live-JVM test (not just gtest) while the BFS thread + // (ReferenceChainTracker::threadLoop()) may concurrently be inside + // cleanup_table()'s epoch-advance pass, which holds _table_lock while + // mutating _klass_population/_klass_population_size - so this seam takes + // the same lock rather than writing the field unguarded (mirrors + // klassPopulationResetForTest() immediately below). Deletes any previous + // representative via DeleteWeakGlobalRef() before overwriting, the same + // way foldKlassCountsLocked() handles a stale representative on eviction - + // otherwise repeated calls for the same klass_id leak a JNI weak global + // ref per call. + void klassPopulationSetRepresentativeForTest(JNIEnv *env, u32 klass_id, jweak rep) { + // ALIAS RESOLUTION (see _test_klass_aliases' own comment above): resolve + // the representative's real klass id BEFORE taking _table_lock, so the + // Class.getName() JNI upcall never runs under the lock. Needs a live JVM + // (a valid Class.getName methodID and a resolvable representative); in + // gtest (env == nullptr or _Class_getName == 0) the alias is skipped and + // the old synthetic-only behavior applies. + u32 real_id = klass_id; + jobject strong = nullptr; + if (env != nullptr && rep != nullptr && _Class_getName != nullptr) { + strong = env->NewLocalRef(rep); + if (strong != nullptr) { + u32 resolved = resolveKlassId(env, strong); + if (resolved != 0 && resolved != klass_id) { + real_id = resolved; + } + } + } + _table_lock.lock(); + if (real_id != klass_id) { + registerTestKlassAliasLocked(klass_id, real_id); + // Re-key any already-seeded synthetic entry to the real id so its + // ring history (hysteresis ramp included) survives the alias. If a + // real-keyed entry already exists too (real allocation sampling + // already folded genuine samples for this same class), the real + // entry wins and the synthetic one is dropped: genuine history for + // the same class outranks seeded history. + int syn_slot = -1; + int real_slot = -1; + for (int i = 0; i < _klass_population_size; i++) { + if (_klass_population[i].klass_id == klass_id) { + syn_slot = i; + } else if (_klass_population[i].klass_id == real_id) { + real_slot = i; + } + } + if (syn_slot >= 0 && real_slot < 0) { + _klass_population[syn_slot].klass_id = real_id; + } else if (syn_slot >= 0 && real_slot >= 0) { + removeKlassPopulationEntryLocked(syn_slot); + } + } + // Find-or-create the entry under real_id (previously a silent no-op + // when absent - but with aliasing, set-representative-first is the + // load-bearing order: scenarios must establish the alias BEFORE any + // seeding, and with no entry yet there is nothing to store the + // representative into. Creating via a count-0, epoch-0 ring sample + // reuses recordKlassPopulationSampleLocked()'s own creation branch + // exactly; the seeded rising ramp that follows still clears + // hasQualifyingGrowth() (a single 0 at the ring's start only lowers the + // earliest-third mean, which RAISES the slope). + int slot = -1; + for (int i = 0; i < _klass_population_size; i++) { + if (_klass_population[i].klass_id == real_id) { + slot = i; + break; + } + } + if (slot < 0) { + jweak evicted[KlassPopulationEntry::MAX_REPRESENTATIVES_PER_KLASS]; + int evicted_count = 0; + bool created = false; + recordKlassPopulationSampleLocked(real_id, 0, 0, &slot, &created, + evicted, &evicted_count, + KlassPopulationEntry::MAX_REPRESENTATIVES_PER_KLASS); + if (env != nullptr) { + for (int r = 0; r < evicted_count; r++) { + if (evicted[r] != nullptr) { + env->DeleteWeakGlobalRef(evicted[r]); + } + } + } + } + if (slot >= 0) { + // Clean up old representatives + for (int r = 0; r < _klass_population[slot].representative_count; r++) { + jweak prev = _klass_population[slot].representatives[r]; + if (prev != nullptr && env != nullptr) { + env->DeleteWeakGlobalRef(prev); + } + } + _klass_population[slot].representative_count = 0; + if (rep != nullptr) { + _klass_population[slot].representatives[0] = rep; + _klass_population[slot].representative_count = 1; + } + // Mint stable_class_tag from this same representative if this slot + // has not gotten one yet - this seam is how live-JVM, + // StaticFieldGrowingCollectionScenario-style tests seed a candidate + // instead of real allocation sampling (foldKlassCountsLocked()), + // which would otherwise never run for them, leaving + // stable_class_tag permanently unminted and + // topKlassesByGenerationCount() unable to report this klass at all - + // found the hard way, exactly the failure this comment is warning + // about. + // env is nullptr in some gtest call sites that only exercise the + // representative-swap bookkeeping above and don't care about + // stable_class_tag - guard against it rather than crash. + if (strong != nullptr) { + mintStableClassTagIfNeeded(env, slot, strong); + } + } + _table_lock.unlock(); + if (strong != nullptr) { + env->DeleteLocalRef(strong); + } + } + // Sets an entry's stable_class_tag directly - production code only ever + // mints this via foldKlassCountsLocked()'s JVMTI-dependent get-or-assign + // step (out of gtest's reach, same rationale as + // klassPopulationSetRepresentativeForTest() above), so tests that exercise + // topKlassesByGenerationCount() via klassPopulationRecordForTest() (which + // bypasses foldKlassCountsLocked() entirely) need this seam - without it, + // stable_class_tag would stay 0 (never minted) and + // topKlassesByGenerationCount() would skip every entry. No-op if klass_id + // is not present. + void klassPopulationSetStableClassTagForTest(u32 klass_id, jlong tag) { + _table_lock.lock(); + for (int i = 0; i < _klass_population_size; i++) { + if (_klass_population[i].klass_id == klass_id) { + _klass_population[i].stable_class_tag = tag; + break; + } + } + _table_lock.unlock(); + } + // Unlike the other klassPopulation*ForTest() seams above, this one is + // also called from a live-JVM test (not just gtest) while the BFS thread + // (ReferenceChainTracker::threadLoop()) may concurrently be inside + // cleanup_table()'s epoch-advance pass, which holds _table_lock while + // mutating _klass_population_size/_klass_population - so this seam must + // take the same lock rather than writing the field unguarded. + void klassPopulationResetForTest() { + _table_lock.lock(); + _klass_population_size = 0; + _test_klass_alias_count = 0; + _table_lock.unlock(); + // Also reset the heap-floor ring: it is a sibling piece of the same + // _gc_generations-gated feature, read by every selectLeakCandidates() + // scan (heapFloorRising()), so leaving it populated across tests in the + // same gtest binary would leak one test's heap-usage history into the + // next test's hysteresis threshold. storeRelease() (not plain store()) + // so the reset is published to the BFS/GC thread on arm64's weak memory + // model - a relaxed store may never be visible to a reader doing + // loadAcquire() on these same indices. + storeRelease(_heap_floor_ring_head, (u8)0); + storeRelease(_heap_floor_ring_fill, (u8)0); + } + + // Test seams for the heap-floor ring (mirrors klassPopulation*ForTest()'s + // own seams immediately above) - lock-free, see _heap_floor_ring's own + // comment, so no locking wrapper is needed here either. timestamp_ns + // defaults to 0 for existing callers that only exercise + // heapFloorRising()/heapFloorRisingForTest() (which never reads the time + // ring) - a test exercising secondsToOOM() must pass real, increasing + // values explicitly. + // container_used defaults to 0 for existing callers that only exercise + // heapFloorRising()/heapFloorRisingForTest() (which never reads + // _container_mem_ring) - a test exercising secondsToOOM()'s container + // boundary must pass real, increasing values explicitly. + void heapFloorRecordForTest(u64 used, u64 timestamp_ns = 0, u64 container_used = 0) { + recordHeapFloorSampleUnchecked(used, timestamp_ns, container_used); + } + bool heapFloorRisingForTest() const { return heapFloorRising(); } + // Bypasses initialize_table()'s JNI-dependent HeapUsage::getMaxHeap() call + // (out of gtest's reach, same reason setGcGenerationsForTest() exists) so + // secondsToOOM() can be exercised directly against a fake max heap size. + void setMaxHeapBytesForTest(jlong v) { +#ifdef DEBUG + _max_heap_bytes_for_test.store(v, std::memory_order_release); +#else + _max_heap_bytes = v; +#endif + } + + // Mirrors setMaxHeapBytesForTest() above, for secondsToOOM()'s container + // boundary - bypasses initialize_table()'s real OS::getContainerMemoryLimit() + // call so a test can exercise the container-vs-heap comparison with a + // fake, deterministic limit. + void setContainerMemoryLimitForTest(jlong v) { +#ifdef DEBUG + _container_memory_limit_for_test.store(v, std::memory_order_release); +#else + _container_memory_limit = v; +#endif + } + +#ifdef DEBUG + // Temporarily disables onGC()'s own recordHeapFloorSample() call so a + // test can seed the ring exclusively via heapFloorRecordForTest() without + // a real GC interleaving a sample. See _heap_floor_recording_disabled_for_test's + // own comment above. + void setHeapFloorRecordingForTest(bool enabled) { + _heap_floor_recording_disabled_for_test.store(!enabled, std::memory_order_release); + } +#endif + + // Sets _gc_generations directly, bypassing initialize() (which requires a + // live JVM - VM::hotspot_version()/VM::jni(), see that method's own code - + // out of gtest's reach the same way foldKlassCountsLocked()'s + // representative-minting step is, per this seam block's own comment + // above). Callers outside this class that only need to exercise + // gcGenerationsEnabled()'s gate (e.g. referenceChains_ut.cpp's + // pollWatchedTargets() tests) use this instead of standing up a full + // initialize()/start() call. + void setGcGenerationsForTest(bool v) { + _gc_generations.store(v, std::memory_order_relaxed); + } + private: void getLiveTraceIds(CallTraceIdSet& out_buffer); }; diff --git a/ddprof-lib/src/main/cpp/objectSampler.cpp b/ddprof-lib/src/main/cpp/objectSampler.cpp index 1cc4cabe38..bdae094380 100644 --- a/ddprof-lib/src/main/cpp/objectSampler.cpp +++ b/ddprof-lib/src/main/cpp/objectSampler.cpp @@ -175,11 +175,17 @@ Error ObjectSampler::start(Arguments &args) { return error; } if (_interval > 0) { - if (_record_liveness || _gc_generations) { - error = LivenessTracker::instance()->start(args); - if (error) { - return error; - } + // Always call through, even when this start's own args request neither + // liveness recording nor gc generations: LivenessTracker::start() -> + // initialize() refreshes its own _gc_generations/_enabled from args + // unconditionally (see that method's own comment) and is a no-op beyond + // that when disabled. Gating this call on ObjectSampler's own + // (freshly-set, correct) flags left LivenessTracker's flags stuck at + // whatever the previous recording in this process last set them to, + // since it never got a chance to observe this recording's request at all. + error = LivenessTracker::instance()->start(args); + if (error) { + return error; } jvmtiEnv *jvmti = VM::jvmti(); @@ -205,9 +211,9 @@ void ObjectSampler::stop() { jvmti->SetEventNotificationMode(JVMTI_DISABLE, JVMTI_EVENT_SAMPLED_OBJECT_ALLOC, NULL); - if (_record_liveness || _gc_generations) { - LivenessTracker::instance()->stop(); - } + // See start()'s own comment on why this call is unconditional - + // LivenessTracker::stop() already self-guards on its own _enabled. + LivenessTracker::instance()->stop(); } Error ObjectSampler::updateConfiguration(u64 events, double time_coefficient) { diff --git a/ddprof-lib/src/main/cpp/os.h b/ddprof-lib/src/main/cpp/os.h index a3d0283819..ee3a74b866 100644 --- a/ddprof-lib/src/main/cpp/os.h +++ b/ddprof-lib/src/main/cpp/os.h @@ -212,6 +212,7 @@ class OS { static int getCpuCount(); static int getCgroupCpuMillicores(); static long getContainerMemoryLimit(); + static long getContainerMemoryUsage(); static u64 getProcessCpuTime(u64* utime, u64* stime); static u64 getTotalCpuTime(u64* utime, u64* stime); diff --git a/ddprof-lib/src/main/cpp/os_linux.cpp b/ddprof-lib/src/main/cpp/os_linux.cpp index 361f4284d9..1bd49e88d3 100644 --- a/ddprof-lib/src/main/cpp/os_linux.cpp +++ b/ddprof-lib/src/main/cpp/os_linux.cpp @@ -1015,6 +1015,71 @@ long OS::getContainerMemoryLimit() { return -1; } +// Reads the current usage from this process's own cgroup leaf only - unlike +// getContainerMemoryLimit()'s ancestor walk (the most restrictive limit can +// live at any level), a leaf's memory.current/memory.usage_in_bytes already +// counts everything charged to it (including descendants), so there is +// nothing further to gain by also reading ancestors' usage here. +long OS::getContainerMemoryUsage() { + char subpath[PATH_MAX]; + char path[PATH_MAX]; + + // Try cgroup v2 first, resolved from this process's own cgroup path. + if (getOwnCgroupPath("", subpath, sizeof(subpath))) { + size_t base_len = strlen("/sys/fs/cgroup"); + size_t sub_len = strlen(subpath); + if (base_len + sub_len < sizeof(path)) { + memcpy(path, "/sys/fs/cgroup", base_len); + memcpy(path + base_len, subpath, sub_len + 1); + + char leaf[PATH_MAX]; + if ((size_t)snprintf(leaf, sizeof(leaf), "%s/memory.current", path) < sizeof(leaf)) { + int fd = open(leaf, O_RDONLY); + if (fd != -1) { + char buf[32] = {0}; + ssize_t r = read(fd, buf, sizeof(buf) - 1); + close(fd); + if (r > 0) { + long usage = atol(buf); + if (usage >= 0) { + return usage; + } + } + } + } + } + } + + // Fall back to cgroup v1, likewise resolved from the process's own path. + if (getOwnCgroupPath("memory", subpath, sizeof(subpath))) { + const char* base = "/sys/fs/cgroup/memory"; + size_t base_len = strlen(base); + size_t sub_len = strlen(subpath); + if (base_len + sub_len < sizeof(path)) { + memcpy(path, base, base_len); + memcpy(path + base_len, subpath, sub_len + 1); + + char leaf[PATH_MAX]; + if ((size_t)snprintf(leaf, sizeof(leaf), "%s/memory.usage_in_bytes", path) < sizeof(leaf)) { + int fd = open(leaf, O_RDONLY); + if (fd != -1) { + char buf[32] = {0}; + ssize_t r = read(fd, buf, sizeof(buf) - 1); + close(fd); + if (r > 0) { + long usage = atol(buf); + if (usage >= 0) { + return usage; + } + } + } + } + } + } + + return -1; +} + u64 OS::getProcessCpuTime(u64* utime, u64* stime) { struct tms buf; clock_t real = times(&buf); @@ -1050,9 +1115,12 @@ int OS::createMemoryFile(const char* name) { void OS::copyFile(int src_fd, int dst_fd, off_t offset, size_t size) { // copy_file_range() is probably better, but not supported on all kernels + size_t requested = size; while (size > 0) { ssize_t bytes = sendfile(dst_fd, src_fd, &offset, size); if (bytes <= 0) { + TEST_LOG("OS::copyFile sendfile returned %zd, errno=%d, remaining=%zu of requested=%zu", + bytes, errno, size, requested); break; } size -= (size_t)bytes; diff --git a/ddprof-lib/src/main/cpp/os_macos.cpp b/ddprof-lib/src/main/cpp/os_macos.cpp index 805a11b84f..3d3aa8ef6b 100644 --- a/ddprof-lib/src/main/cpp/os_macos.cpp +++ b/ddprof-lib/src/main/cpp/os_macos.cpp @@ -384,6 +384,10 @@ long OS::getContainerMemoryLimit() { return -1; // macOS has no cgroup support. } +long OS::getContainerMemoryUsage() { + return -1; // macOS has no cgroup support. +} + u64 OS::getProcessCpuTime(u64* utime, u64* stime) { struct tms buf; clock_t real = times(&buf); diff --git a/ddprof-lib/src/main/cpp/painBudget.h b/ddprof-lib/src/main/cpp/painBudget.h new file mode 100644 index 0000000000..3e1b4a39d4 --- /dev/null +++ b/ddprof-lib/src/main/cpp/painBudget.h @@ -0,0 +1,92 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +#ifndef _PAINBUDGET_H +#define _PAINBUDGET_H + +#include "arch.h" + +/* + * A leaky bucket over *cost* (milliseconds of expensive work already spent), + * not over an event *rate* - unlike PidController/RateLimiter (which target + * a steady events-per-second throughput), this answers "have we spent too + * much recently to justify doing more expensive work right now?" + * + * Typical use: a subsystem that occasionally does one genuinely expensive, + * bounded operation (here: a full-heap BFS pass) wants to avoid doing that + * operation back-to-back if it keeps being expensive, while still allowing + * it immediately again if the last one was cheap. spend() records how much + * an operation cost; canStartNow() drains the balance by however much + * wall-clock time has passed (at _refill_rate) and reports whether the + * debt has cleared. + * + * _refill_rate is the one tunable: the fraction of wall-clock time this + * budget is willing to let its owner spend on the expensive operation, on + * average (e.g. 0.01 = "at most ~1% of wall-clock time, averaged over + * time"). Unlike PidController's gain triples (P/I/D), this single ratio + * has a direct, human-interpretable meaning and needs no derivation beyond + * picking that target fraction. + */ +class PainBudget { +private: + double _balance_ms; // accumulated debt in ms; 0 means "clear to spend" + double _refill_rate; // fraction of wall-clock time allowed, e.g. 0.01 + u64 _last_update_ns; // OS::nanotime() as of the last drain(); 0 = never drained yet + + void drain(u64 now_ns) { + if (_last_update_ns == 0) { + // First call ever - nothing to drain yet, just establish the baseline. + _last_update_ns = now_ns; + return; + } + u64 elapsed_ns = now_ns - _last_update_ns; + double elapsed_ms = (double)elapsed_ns / 1000000.0; + // _refill_rate == 0.0 (the default constructor argument) makes this a + // no-op forever: the balance never drains, so once spend() has pushed it + // above 0 canStartNow() stays false permanently. Callers that want the + // budget to actually refill must pass a positive _refill_rate. + _balance_ms -= elapsed_ms * _refill_rate; + if (_balance_ms < 0) { + _balance_ms = 0; + } + _last_update_ns = now_ns; + } + +public: + explicit PainBudget(double refill_rate = 0.0) + : _balance_ms(0), _refill_rate(refill_rate), _last_update_ns(0) {} + + // Records that an operation just cost `pain_ms` milliseconds of + // wall-clock time. Does not drain first - the cost is added on top of + // whatever debt (already correctly drained as of the last canStartNow() + // call) currently exists. + void spend(u64 pain_ms) { _balance_ms += (double)pain_ms; } + + // True once the debt has drained back to zero at _refill_rate - i.e. it + // is now affordable, on average, to spend more pain. Drains the balance + // as a side effect, so repeated calls correctly reflect elapsed time + // even if spend() is never called again. + bool canStartNow(u64 now_ns) { + drain(now_ns); + return _balance_ms <= 0; + } + + // Test/introspection only - current debt after draining as of now_ns. + double balanceMs(u64 now_ns) { + drain(now_ns); + return _balance_ms; + } + + // Changes the refill rate without resetting accumulated debt - unlike + // assigning a freshly-constructed PainBudget(rate), which would zero + // _balance_ms. Drains at the *old* rate up to now_ns first, so the rate + // change only affects time elapsed after this call. + void setRefillRate(double refill_rate, u64 now_ns) { + drain(now_ns); + _refill_rate = refill_rate; + } +}; + +#endif // _PAINBUDGET_H diff --git a/ddprof-lib/src/main/cpp/profiler.cpp b/ddprof-lib/src/main/cpp/profiler.cpp index 576c2555f8..58dcc20c6d 100644 --- a/ddprof-lib/src/main/cpp/profiler.cpp +++ b/ddprof-lib/src/main/cpp/profiler.cpp @@ -31,6 +31,7 @@ #include "objectSampler.h" #include "os.h" #include "perfEvents.h" +#include "referenceChains.h" #include "safeAccess.h" #include "stackFrame.h" #include "stackWalker.h" @@ -110,6 +111,11 @@ void Profiler::onThreadEnd(jvmtiEnv *jvmti, JNIEnv *jni, jthread thread) { // ProfiledThread is alive - do full cleanup and use efficient tid access int slot_id = current->filterSlotId(); tid = current->tid(); + // NOT gated on reference-chains enabled: a thread registered while a + // recording ran must release its global ref when it ends, even if the + // recording has since stopped (see unregisterThreadObject()'s comment, + // referenceChains.h). + ReferenceChainTracker::instance()->unregisterThreadObject(jni, tid); if (_thread_filter.enabled()) { _thread_filter.unregisterThread(slot_id); @@ -859,6 +865,89 @@ void Profiler::writeHeapUsage(long value, bool live) { _locks[lock_index].unlock(); } +void Profiler::writeReferenceChainAbandoned(ReferenceChainAbandonedEvent *event) { + int tid = ProfiledThread::currentTid(); + if (tid < 0) { + return; + } + u32 lock_index = getLockIndex(tid); + if (!_locks[lock_index].tryLock() && + !_locks[lock_index = (lock_index + 1) % CONCURRENCY_LEVEL].tryLock() && + !_locks[lock_index = (lock_index + 2) % CONCURRENCY_LEVEL].tryLock()) { + return; + } + _jfr.recordReferenceChainAbandoned(lock_index, event); + _locks[lock_index].unlock(); +} + +// Unlike writeReferenceChainAbandoned() above (mirroring CPU/wall's signal-handler-safe +// non-blocking pattern out of caution, even though its own call site - Profiler::dump(), +// profiler.cpp - isn't a signal handler either), this call site genuinely cannot be one: +// this is called from Profiler::dump()'s drain loop, on dump()'s own calling thread, once +// per event snapshotted from ReferenceChainTracker::_resolved_chains (up to +// MAX_RESOLVED_CHAINS per dump) - never from pollWatchedTargets() or any other call on +// ReferenceChainTracker's own BFS agent thread, and never from a signal handler. A single +// bare 3-slot tryLock() sweep with no wait - correct for a signal handler, which must never +// block - was found, by running PROF-15341's end-to-end integration test +// (ddprof-test's ReferenceChainTrackingTest.shouldReconstructReferrerChainToGcRoot) for real, +// to drop this event under perfectly ordinary contention: the same _locks[] pool is shared +// with every other sample type (recordJVMTISample() et al.), and any nontrivial allocation +// throughput keeps enough of CONCURRENCY_LEVEL's slots busy that 3 immediate, back-to-back +// attempts routinely all miss. A bounded retry with a short sleep between sweeps costs +// nothing the dump()-thread cannot afford, but the retry budget below is a single deadline +// shared across the *entire* drain batch (see the caller in dump()) rather than per event: +// with up to MAX_RESOLVED_CHAINS events snapshotted, a fresh per-event budget could stall +// the dump/JFR-flush thread for seconds under contention. Once the shared deadline has +// passed this degrades to the same single non-blocking 3-slot sweep as +// writeReferenceChainAbandoned() above for the remainder of the batch. +void Profiler::writeReferenceChain(ReferenceChainEvent *event, u64 deadline_ns) { + int tid = ProfiledThread::currentTid(); + if (tid < 0) { + TEST_LOG("Profiler::writeReferenceChain drop: currentTid() < 0"); + return; + } + u32 lock_index; + bool locked = false; + int sweeps = 0; + u64 start_ns = OS::nanotime(); + for (;;) { + sweeps++; + lock_index = getLockIndex(tid); + if (_locks[lock_index].tryLock() || + _locks[lock_index = (lock_index + 1) % CONCURRENCY_LEVEL].tryLock() || + _locks[lock_index = (lock_index + 2) % CONCURRENCY_LEVEL].tryLock()) { + locked = true; + break; + } + if (OS::nanotime() >= deadline_ns) { + // Shared batch budget exhausted - the sweep just above was already a + // single non-blocking attempt, so stop retrying rather than sleeping + // again. + break; + } + usleep(1000); + } + if (!locked) { + // Unlike the drain-once era, this drop is NOT permanent: the event was + // only copied out of ReferenceChainTracker::_resolved_chains + // (drainPendingChainEvents() snapshots without clearing), so as long as + // the sample stays live the next dump re-emits it and gets another chance + // at the lock. Still counted like every other counted-drop path + // (REFERENCE_CHAIN_WRITE_DROPPED's own comment) rather than dropping it + // silently. + Counters::increment(REFERENCE_CHAIN_WRITE_DROPPED); + TEST_LOG("Profiler::writeReferenceChain drop: lock contention exhausted shared " + "deadline after sweeps=%d waited_us=%llu", + sweeps, (unsigned long long)((OS::nanotime() - start_ns) / 1000)); + return; + } + TEST_LOG("Profiler::writeReferenceChain locked lock_index=%u after sweeps=%d " + "waited_us=%llu", + lock_index, sweeps, (unsigned long long)((OS::nanotime() - start_ns) / 1000)); + _jfr.recordReferenceChain(lock_index, event); + _locks[lock_index].unlock(); +} + bool Profiler::prewarmUnwinder() { #ifdef __linux__ // Force libgcc_s.so.1 to load now and report whether that succeeded. This @@ -1732,6 +1821,42 @@ Error Profiler::start(Arguments &args, bool reset) { // Paired with drainInflight() on the stop side. _cpu_engine->enableEvents(true); + // Independent of the CPU/wall/alloc engine mask above (GC-triggered, not + // sample-triggered) - same pattern as malloc_tracer/NativeSocketSampler + // being gated on their own flags rather than folded into `activated`. + // Placed after the engines are confirmed running (inside this + // `if (activated)` block) so there is nothing to unwind here if it + // fails - see this method's failure path below, which never reaches + // this point. + // Called unconditionally, not gated on args._reference_chains: start() + // is the only place that refreshes ReferenceChainTracker::_enabled + // (stop() deliberately leaves it unchanged - see that method's own + // comment), so a previous recording's `referencechains=true` session + // must still reach start() when this one opts out, or the tracker keeps + // reporting enabled()==true - and Profiler::dump()'s reference-chains + // gate keeps emitting that stale session's cached chains/abandonment + // state - for the entire duration of this new, opted-out recording. + // start() itself sets `_enabled = args._reference_chains` up front and + // returns early when that is false, so this call is a cheap no-op for + // an opted-out recording. + error = ReferenceChainTracker::instance()->start(args); + if (error) { + Log::warn("%s", error.message()); + error = Error::OK; // recoverable + } else if (args._reference_chains) { + // Only safe once the JVM/JVMTI environment is fully up, which is + // guaranteed at this point in Profiler::start() - see + // ReferenceChainTracker::start()'s own comment (referenceChains.cpp) + // for why this is not called from inside start() itself. + ReferenceChainTracker::instance()->startThread(); + // Pre-existing threads (alive since before this recording began) + // never fired onThreadStart() - same lifecycle rationale as + // startThread() above for why this runs here rather than inside + // ReferenceChainTracker::start(). + ReferenceChainTracker::instance()->registerExistingThreads( + VM::jvmti(), VM::jni()); + } + _state.store(RUNNING, std::memory_order_release); _start_time = time(NULL); __atomic_add_fetch(&_epoch, 1, __ATOMIC_RELAXED); @@ -1776,6 +1901,13 @@ Error Profiler::stop() { _alloc_engine->stop(); if (_event_mask & EM_NATIVEMEM) malloc_tracer.stop(); + // Not part of _event_mask (see the matching start() block above) - gated + // on enabled() instead, which start() set from args._reference_chains for + // this session. + if (ReferenceChainTracker::instance()->enabled()) { + ReferenceChainTracker::instance()->stopThread(); + ReferenceChainTracker::instance()->stop(); + } // Stop the refresher BEFORE socket unpatch: the refresher calls // install_socket_hooks() which re-reads _socket_active before acquiring the // patch lock. If the refresher runs concurrently with unpatch_socket_functions() @@ -1930,6 +2062,60 @@ Error Profiler::dump(const char *path, const int length) { // by the live objects LivenessTracker::instance()->flush(thread_ids); + // ReferenceChainTracker::_resolved_chains (and the search-state fields + // read below) are intentionally left populated across a stop()/start() + // cycle - see _resolved_chains' own comment (referenceChains.h) - but + // that means they can still hold state from a *previous* recording that + // had referencechains enabled, even once the current recording started + // with referencechains=false (in which case ReferenceChainTracker:: + // start() sets _enabled=false and no BFS thread is polling to ever + // refresh or prune them). Gate both emissions on the current session's + // flag so an opted-out recording does not keep re-reporting a dead + // session's abandoned search or stale resolved chains. + if (ReferenceChainTracker::instance()->enabled()) { + // ReferenceChainTracker's BFS thread restarts an ABANDONED search on + // its own ~1s cadence (referenceChains.cpp shouldRunPass() -> + // restartSearch()), which clears the very state + // buildAbandonedEvent() needs. A live re-read of searchState() here + // would almost always miss that ~1s window against dump()'s much + // slower JFR-chunk-rotation cadence. Instead each abandon is + // snapshotted into a queue at the moment it happens + // (enqueuePendingAbandonedEvent(), called from runPass()) and drained + // here - a true drain, unlike drainPendingChainEvents() below, since + // an abandon is a one-off past occurrence rather than an ongoing live + // sample. + std::vector pending_abandoned_events; + ReferenceChainTracker::instance()->drainPendingAbandonedEvents( + &pending_abandoned_events); + for (auto &rc_event : pending_abandoned_events) { + rc_event._start_time = TSC::ticks(); + writeReferenceChainAbandoned(&rc_event); + } + + // Re-emit every currently-cached datadog.ReferenceChain pollWatchedTargets() + // (referenceChains.cpp) has resolved - snapshotted here, on this call's + // own thread, rather than written eagerly from the BFS scheduling thread + // that discovered them (see ReferenceChainTracker::_resolved_chains' own + // comment for why the cache re-emits on every dump rather than draining). + std::vector pending_chain_events; + ReferenceChainTracker::instance()->drainPendingChainEvents( + &pending_chain_events); + // One ~50ms retry budget for the *whole* batch, not per event - + // writeReferenceChain()'s own comment for why: up to + // MAX_RESOLVED_CHAINS events can be snapshotted, and a fresh per-event + // budget would let this dump()-thread stall for seconds under ordinary + // _locks[] contention. + const u64 kChainDrainBudgetNs = 50 * 1000000ULL; + u64 chain_drain_deadline_ns = OS::nanotime() + kChainDrainBudgetNs; + long long write_dropped_before = Counters::getCounter(REFERENCE_CHAIN_WRITE_DROPPED); + for (auto &rc_event : pending_chain_events) { + writeReferenceChain(&rc_event, chain_drain_deadline_ns); + } + TEST_LOG("Profiler::dump reference-chain batch=%d write_dropped=%lld", + (int)pending_chain_events.size(), + Counters::getCounter(REFERENCE_CHAIN_WRITE_DROPPED) - write_dropped_before); + } + Libraries::instance()->refresh(); updateJavaThreadNames(); updateNativeThreadNames(); @@ -1945,6 +2131,9 @@ Error Profiler::dump(const char *path, const int length) { err = _jfr.dump(path, length); __atomic_add_fetch(&_epoch, 1, __ATOMIC_SEQ_CST); }); + if (err) { + TEST_LOG("Profiler::dump _jfr.dump failed: %s", err.message()); + } _thread_info.clearAll(thread_ids); _thread_info.reportCounters(); diff --git a/ddprof-lib/src/main/cpp/profiler.h b/ddprof-lib/src/main/cpp/profiler.h index 7227960359..397e0ea513 100644 --- a/ddprof-lib/src/main/cpp/profiler.h +++ b/ddprof-lib/src/main/cpp/profiler.h @@ -189,7 +189,9 @@ class alignas(alignof(SpinLock)) Profiler { // // rotate() is self-contained: it uses _accepting + RefCountGuard to drain // concurrent JNI readers, and SignalBlocker prevents profiling signals on - // this thread from inserting into old_active between Phase 1 and Phase 2. + // this thread from inserting into old_active between the pre-populate copy + // step and the catch-up copy step of the dictionary's two-step rotation + // (see StringDictionary::rotate(), stringDictionary.h). // No external lock is required for rotation. // // lockAll() wraps jfr_op only — to gate call-trace writers (signal handlers @@ -464,6 +466,20 @@ class alignas(alignof(SpinLock)) Profiler { void writeDatadogProfilerSetting(int tid, int length, const char *name, const char *value, const char *unit); void writeHeapUsage(long value, bool live); + // Mirrors writeHeapUsage()'s shape exactly. Called from dump() whenever + // ReferenceChainTracker's search has ended in SearchState::ABANDONED, + // the same way LivenessTracker::flush() is called from dump(). + void writeReferenceChainAbandoned(ReferenceChainAbandonedEvent *event); + // Unlike writeReferenceChainAbandoned() above, this is NOT a bare 3-slot + // tryLock() sweep - it retries with a bounded, sleeping loop because its + // call site is dump()'s drain loop (profiler.cpp), on dump()'s own calling + // thread, which can tolerate blocking, unlike a signal handler; see this + // method's own comment in profiler.cpp for why that retry exists. + // `deadline_ns` is a single retry budget shared across dump()'s *entire* + // drain batch (not reset per event) - see the caller in dump() and this + // method's own comment in profiler.cpp for why a per-event budget would be + // unbounded across a large batch. + void writeReferenceChain(ReferenceChainEvent *event, u64 deadline_ns); int eventMask() const { return _event_mask; } bool isRemoteSymbolication() const { return _remote_symbolication; } bool sanityCheckFailed() const { return _sanity_check_failed; } diff --git a/ddprof-lib/src/main/cpp/rcDebugLevel.h b/ddprof-lib/src/main/cpp/rcDebugLevel.h new file mode 100644 index 0000000000..2042590621 --- /dev/null +++ b/ddprof-lib/src/main/cpp/rcDebugLevel.h @@ -0,0 +1,65 @@ +#ifndef _RC_DEBUG_LEVEL_H +#define _RC_DEBUG_LEVEL_H + +// Runtime level gate for the reference-chains subsystem's TEST_LOG +// diagnostics (referenceChains.cpp + livenessTracker.cpp). Included +// AFTER common.h, this re-points THIS translation unit's TEST_LOG at a +// level check and adds TEST_LOG_SUMMARY: +// +// level 0 silent (the default - keeps DEBUG builds pod-safe) +// level 1 lifecycle/summary: state-machine transitions, per-pass and +// per-poll outcomes (candidates, canary, rotation counters, +// drain/re-emit, leak-tag correlation) +// level 2 full diagnostics: per-object/per-klass/per-entry lines +// (heap admits, auto-marks, sweep/fold internals) +// +// Sources, in order: the env var below at first use, overridden at +// runtime by the file below (re-checked about once per second from the +// reference-chains thread loop - refresh does open/read, so it never +// runs in heap callbacks; heap callbacks only read the cached atomic). +// +// env: DD_PROFILING_REFERENCE_CHAINS_DEBUG=0|1|2 +// file: /tmp/ddprof_root/refchains_debug_level (single digit 0/1/2; +// remove the file to fall back to the env value) +// +// All machinery is DEBUG-build-only: in non-debug builds TEST_LOG is +// already a no-op (common.h) and TEST_LOG_SUMMARY matches it, so this +// header costs nothing. + +#include "common.h" + +// The level machinery is compiled in ALL builds (the gtest binary is a +// non-DEBUG build and tests it directly; in non-DEBUG builds nothing calls +// it because the TEST_LOG macros are no-ops), while the macros below stay +// DEBUG-only like TEST_LOG itself. +int rcDebugLevel(); // cached; lazy env init on first use +void rcDebugLevelRefresh(bool force = false); // file override check, ~1s TTL +int parseRcDebugLevel(const char *value); // pure: NULL/invalid -> -1, else 0/1/2 +int readRcDebugLevelFile(const char *path); // pure: -1 missing/invalid + +#ifdef DEBUG + +#undef TEST_LOG +#define TEST_LOG(fmt, ...) \ + do { \ + if (rcDebugLevel() >= 2) { \ + fprintf(stdout, "[TEST::INFO] " fmt "\n", ##__VA_ARGS__); \ + fflush(stdout); \ + } \ + } while (0) + +#define TEST_LOG_SUMMARY(fmt, ...) \ + do { \ + if (rcDebugLevel() >= 1) { \ + fprintf(stdout, "[TEST::INFO] " fmt "\n", ##__VA_ARGS__); \ + fflush(stdout); \ + } \ + } while (0) + +#else // DEBUG + +#define TEST_LOG_SUMMARY(fmt, ...) // No-op in non-debug mode + +#endif // DEBUG + +#endif // _RC_DEBUG_LEVEL_H diff --git a/ddprof-lib/src/main/cpp/referenceChains.cpp b/ddprof-lib/src/main/cpp/referenceChains.cpp new file mode 100644 index 0000000000..7e57c121ec --- /dev/null +++ b/ddprof-lib/src/main/cpp/referenceChains.cpp @@ -0,0 +1,7042 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +#include "referenceChains.h" +#include "common.h" +#include "counters.h" +#include "jniHelper.h" +#include "jvmThread.h" +#include "livenessTracker.h" +#include "log.h" +#include "objectSampler.h" +#include "os.h" +#include "profiler.h" +#include "rcDebugLevel.h" +#include "tsc.h" +#include "vmEntry.h" +#include +#include +#include +#include +#include +#include +#include +#include + +// --------------------------------------------------------------------------- +// Reference-chains debug-log level (see rcDebugLevel.h). Level 0 silent +// (default), 1 lifecycle/summary, 2 full per-object diagnostics. Sources: +// env DD_PROFILING_REFERENCE_CHAINS_DEBUG, overridden at runtime by +// /tmp/ddprof_root/refchains_debug_level, re-checked ~1/s from threadLoop +// (rcDebugLevelRefresh never runs in heap callbacks - they only read the +// cached atomic below). Compiled in all builds (harmless in non-DEBUG: +// nothing calls it, the macros are no-ops) so the gtest binary can test it. +// --------------------------------------------------------------------------- +namespace { +constexpr const char *kRcDebugLevelEnv = "DD_PROFILING_REFERENCE_CHAINS_DEBUG"; +constexpr const char *kRcDebugLevelFile = "/tmp/ddprof_root/refchains_debug_level"; +constexpr u64 kRcDebugLevelRefreshTtlNs = 1000000000ULL; // 1s +std::atomic g_rc_debug_level{-1}; // -1 = not yet resolved from env +std::atomic g_rc_debug_level_last_refresh_ns{0}; + +int envRcDebugLevel() { + int lvl = parseRcDebugLevel(getenv(kRcDebugLevelEnv)); + return lvl < 0 ? 0 : lvl; // invalid/unset env means silent +} +} // namespace + +int rcDebugLevel() { + int lvl = g_rc_debug_level.load(std::memory_order_relaxed); + if (lvl >= 0) { + return lvl; + } + // Lazy one-time env resolve; may fire from a heap callback on the very + // first log line, which is still strictly cheaper than the fprintf the + // same line performs in a DEBUG build. + lvl = envRcDebugLevel(); + g_rc_debug_level.store(lvl, std::memory_order_relaxed); + return lvl; +} + +int parseRcDebugLevel(const char *value) { + if (value == nullptr || *value == '\0') { + return -1; + } + // Trim surrounding whitespace (files written via `echo N >` end with \n). + while (*value == ' ' || *value == '\t' || *value == '\n' || *value == '\r') { + ++value; + } + const char *end = value + strlen(value); + while (end > value && (end[-1] == ' ' || end[-1] == '\t' || + end[-1] == '\n' || end[-1] == '\r')) { + --end; + } + if (end == value || end - value != 1) { + return -1; // exactly one digit + } + if (*value < '0' || *value > '2') { + return -1; + } + return *value - '0'; +} + +int readRcDebugLevelFile(const char *path) { + if (path == nullptr) { + return -1; + } + FILE *f = fopen(path, "r"); + if (f == nullptr) { + return -1; + } + char buf[16]; + size_t n = fread(buf, 1, sizeof(buf) - 1, f); + fclose(f); + buf[n] = '\0'; + return parseRcDebugLevel(buf); +} + +void rcDebugLevelRefresh(bool force) { + u64 now = OS::nanotime(); + u64 last = g_rc_debug_level_last_refresh_ns.load(std::memory_order_relaxed); + if (!force && last != 0 && now >= last && + now - last < kRcDebugLevelRefreshTtlNs) { + return; + } + g_rc_debug_level_last_refresh_ns.store(now, std::memory_order_relaxed); + int lvl = readRcDebugLevelFile(kRcDebugLevelFile); + if (lvl < 0) { + lvl = envRcDebugLevel(); // file absent/invalid -> fall back to env + } + g_rc_debug_level.store(lvl, std::memory_order_relaxed); +} + +// --------------------------------------------------------------------------- +// FrontierTable (tag-indexed frontier metadata table) +// --------------------------------------------------------------------------- + +FrontierTable::FrontierTable(int max_cap) + : _table_size(0), _table_cap(0), _table_max_cap(std::max(max_cap, 0)), + _table(nullptr) { + _table_cap = std::min(INITIAL_TABLE_CAPACITY, _table_max_cap); + if (_table_cap > 0) { + _table = (FrontierEntry *)calloc(_table_cap, sizeof(FrontierEntry)); + if (_table == nullptr) { + _table_cap = 0; + } + } + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, + (jlong)_table_cap * sizeof(FrontierEntry)); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, _table_cap); +} + +FrontierTable::~FrontierTable() { + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, + -(jlong)_table_cap * sizeof(FrontierEntry)); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, -_table_cap); + free(_table); +} + +void FrontierTable::resetCapacityForTest(int max_cap) { + _table_lock.lock(); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, + -(jlong)_table_cap * sizeof(FrontierEntry)); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, -_table_cap); + free(_table); + _table = nullptr; + _table_max_cap = std::max(max_cap, 0); + _table_cap = std::min(INITIAL_TABLE_CAPACITY, _table_max_cap); + if (_table_cap > 0) { + _table = (FrontierEntry *)calloc(_table_cap, sizeof(FrontierEntry)); + if (_table == nullptr) { + _table_cap = 0; + } + } + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, + (jlong)_table_cap * sizeof(FrontierEntry)); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, _table_cap); + _table_size.store(0, std::memory_order_relaxed); + _table_lock.unlock(); +} + +bool FrontierTable::growLocked(int required_cap) { + if (required_cap <= _table_cap) { + return true; + } + if (_table_cap >= _table_max_cap) { + return false; + } + + int newcap = _table_cap; + while (newcap < required_cap && newcap < _table_max_cap) { + newcap = newcap == 0 ? std::min(INITIAL_TABLE_CAPACITY, _table_max_cap) + : std::min(newcap * 2, _table_max_cap); + } + if (newcap <= _table_cap) { + return false; + } + + FrontierEntry *tmp = + (FrontierEntry *)realloc(_table, sizeof(FrontierEntry) * newcap); + if (tmp == nullptr) { + Log::debug( + "ReferenceChains: frontier table resize to %d entries failed", newcap); + return false; + } + // realloc() does not zero the newly grown region - clear it so lookup() + // never returns garbage state for a slot that hasn't been inserted yet. + memset(tmp + _table_cap, 0, sizeof(FrontierEntry) * (newcap - _table_cap)); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, + (jlong)(newcap - _table_cap) * sizeof(FrontierEntry)); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, + newcap - _table_cap); + _table = tmp; + _table_cap = newcap; + return _table_cap >= required_cap; +} + +bool FrontierTable::insert(jlong tag, jlong parent_tag, u32 referrer_klass, + u32 depth, u8 state, u8 root_kind, + jlong class_tag, jint referrer_field_index, + u8 edge_kind, jlong referrer_class_tag) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return false; + } + int idx = (int)(tag - 1); + + // Exclusive lock for the whole write (growLocked() already requires it) - + // a shared lock here would not exclude lookup()'s own shared-mode read of + // the same slot, letting a concurrent reader observe a torn entry. + _table_lock.lock(); + if (idx >= _table_cap && !growLocked(idx + 1)) { + _table_lock.unlock(); + Log::debug("ReferenceChains: frontier table capacity exhausted " + "(cap=%d, max=%d, tag=%lld)", + _table_cap, _table_max_cap, (long long)tag); + return false; + } + _table[idx].parent_tag = parent_tag; + _table[idx].referrer_klass = referrer_klass; + _table[idx].depth = depth; + _table[idx].state = state; + _table[idx].root_kind = root_kind; + _table[idx].class_tag = class_tag; + _table[idx].leak_tag = 0; + _table[idx].referrer_field_index = referrer_field_index; + _table[idx].edge_kind = edge_kind; + _table[idx].referrer_class_tag = referrer_class_tag; + _table_lock.unlock(); + + int sz = _table_size.load(std::memory_order_relaxed); + while (sz < idx + 1 && + !_table_size.compare_exchange_weak(sz, idx + 1, + std::memory_order_relaxed)) { + // sz reloaded with the current value by compare_exchange_weak on + // failure; retry until either this thread wins or another thread + // already advanced _table_size past idx + 1. + } + return true; +} + +bool FrontierTable::lookup(jlong tag, FrontierEntry *out) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return false; + } + int idx = (int)(tag - 1); + + bool found = false; + _table_lock.lockShared(); + if (idx < _table_size) { + *out = _table[idx]; + found = true; + } + _table_lock.unlockShared(); + return found; +} + +bool FrontierTable::lookupLocked(jlong tag, FrontierEntry *out) const { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return false; + } + int idx = (int)(tag - 1); + if (idx < _table_size) { + *out = _table[idx]; + return true; + } + return false; +} + +void FrontierTable::clear(jlong tag) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return; + } + int idx = (int)(tag - 1); + + // Exclusive lock: this mutates a slot lookup() may be reading concurrently + // under its own shared lock (see insert()'s own comment above). + _table_lock.lock(); + if (idx < _table_size) { + _table[idx].state = FrontierEntryState::ABANDONED; + } + _table_lock.unlock(); +} + +void FrontierTable::markEdge(jlong tag) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return; + } + int idx = (int)(tag - 1); + + _table_lock.lock(); + if (idx < _table_size) { + _table[idx].state = FrontierEntryState::EDGE; + } + _table_lock.unlock(); +} + +void FrontierTable::markExpanded(jlong tag) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return; + } + int idx = (int)(tag - 1); + + _table_lock.lock(); + if (idx < _table_size) { + _table[idx].state = FrontierEntryState::EXPANDED; + } + _table_lock.unlock(); +} + +void FrontierTable::updateRootKind(jlong tag, u8 root_kind) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return; + } + int idx = (int)(tag - 1); + + _table_lock.lock(); + if (idx < _table_size) { + _table[idx].root_kind = root_kind; + } + _table_lock.unlock(); +} + +bool FrontierTable::improveChain(jlong tag, jlong parent_tag, + u32 referrer_klass, u32 depth, + u8 root_kind, jint referrer_field_index, + u8 edge_kind, jlong referrer_class_tag) { + // Replace a shallow root-attached entry (parent_tag == 0, depth == 0) + // with a deeper chain-attached entry when the object is reached via a + // longer path. This fixes the "depth=1 chain with no holder" problem: + // an object first admitted as a JNI-local root (parent_tag == 0) gets + // its frontier entry overwritten when the static-field → ... → object + // path reaches it later with a non-zero parent_tag. + // Returns true if the entry was actually improved (new depth > old). + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return false; + } + // Round 16 (pod round-15 measurement, ev-leaktag-onpod-round15-results): + // a "chain" whose parent is the entry itself is never an improvement - + // it is the self-edge a this-field produces, and it is REAL in the heap: + // every java.util.Collections$Synchronized* holder carries mutex == this, + // so walking such a holder's own subtree (the rotation anchor walk or a + // BFS descent) re-reports the holder as its own child through that field. + // On the pod this exact edge demoted the LEAK_BUFFER wrapper: admitted + // root-attached (parent=0, root_kind=STATIC_FIELD) in search #2 and + // walked once, then every later search's entry read parent==its own tag + // root_kind=0 referrer_klass= - collector-invisible + // (the parent_tag==0 eligibility filter) and never re-walked, because + // improveChain(depth=holder.depth+1) "improved" the entry with itself. + // A self-parent would also dead-loop reconstructChain(). Refuse, and + // count the refusal (selfEdgeGuardSkips()): the counter climbing on the + // pod is the verification that the guard fires on the real wrapper. + if (parent_tag == tag) { + _self_edge_guard_skips.fetch_add(1, std::memory_order_relaxed); + return false; + } + int idx = (int)(tag - 1); + + _table_lock.lock(); + bool improved = false; + if (idx < _table_size && depth > _table[idx].depth) { + _table[idx].parent_tag = parent_tag; + _table[idx].referrer_klass = referrer_klass; + _table[idx].depth = depth; + _table[idx].root_kind = root_kind; + _table[idx].referrer_field_index = referrer_field_index; + _table[idx].edge_kind = edge_kind; + _table[idx].referrer_class_tag = referrer_class_tag; + improved = true; + } + _table_lock.unlock(); + return improved; +} + +bool FrontierTable::reparentToDurableRoot(jlong tag, jlong new_parent_tag, + u32 referrer_klass, + jint referrer_field_index, + u8 edge_kind) { + // See the declaration's own comment (referenceChains.h) for why this + // exists as a sibling of improveChain(): equal-depth depth-1 noise->real + // re-parenting. All lookups happen under one lock - three index reads, + // no allocation, O(1). + // The parent==tag self-edge guard mirrors improveChain's (round 16); + // today this cannot self-reparent - the swap below requires the entry's + // own parent_tag > 0 while the new parent slot (the same slot) must hold + // parent_tag == 0, a contradiction - but the two siblings' contracts + // stay identical so a future caller change cannot reintroduce the + // self-parent a this-field (mutex == this) would deliver. + if (tag <= 0 || tag - 1 > (jlong)INT_MAX || new_parent_tag <= 0 || + new_parent_tag - 1 > (jlong)INT_MAX || new_parent_tag == tag) { + if (new_parent_tag == tag) { + _self_edge_guard_skips.fetch_add(1, std::memory_order_relaxed); + } + return false; + } + int idx = (int)(tag - 1); + int new_par_idx = (int)(new_parent_tag - 1); + + _table_lock.lock(); + bool swapped = false; + if (idx < _table_size && _table[idx].depth == 1 && + _table[idx].parent_tag > 0 && _table[idx].parent_tag != new_parent_tag) { + int old_par_idx = (int)(_table[idx].parent_tag - 1); + if (old_par_idx >= 0 && old_par_idx < _table_size && + new_par_idx < _table_size && + _table[new_par_idx].parent_tag == 0 && + _table[new_par_idx].root_kind != 0 && + !isTransientRootKind(_table[new_par_idx].root_kind) && + _table[old_par_idx].parent_tag == 0 && + isTransientRootKind(_table[old_par_idx].root_kind)) { + // New parent is a root-attached DURABLE root (static field, JNI + // global, thread) and the current parent is a root-attached TRANSIENT + // one - same depth, strictly better retention explanation. + _table[idx].parent_tag = new_parent_tag; + _table[idx].referrer_klass = referrer_klass; + _table[idx].referrer_field_index = referrer_field_index; + _table[idx].edge_kind = edge_kind; + swapped = true; + } + } + _table_lock.unlock(); + return swapped; +} + +bool FrontierTable::reconstructChain(jlong target_tag, + std::vector *out_chain, + u8 *out_root_kind, + std::vector *out_edges) { + FrontierEntry entry{}; + if (!lookup(target_tag, &entry)) { + return false; + } + + std::vector chain; + std::vector edges; + jlong tag = target_tag; + u8 root_kind = 0; + // Bounded by maxCapacity(): every tag maps to a distinct slot (this table's + // "tags/slots are never reused" invariant, see the class comment above), + // so a well-formed parent_tag chain can visit at most maxCapacity() slots + // before either reaching parent_tag == 0 or repeating a slot. + for (int hops = 0; hops <= maxCapacity() && tag != 0; hops++) { + if (!lookup(tag, &entry)) { + // parent_tag pointed at a tag that was never inserted - should not + // happen for a chain built entirely within one BFS pass, but do not + // fabricate a partial chain silently. + return false; + } + chain.push_back(entry.referrer_klass); + if (out_edges != nullptr) { + // edges[i] describes the edge INTO chain[i]: the entry's own recorded + // edge identity, plus the referrer's class tag - the parent entry's + // own class for interior hops, the declaring class for root-attached + // static edges (FrontierEntry::referrer_class_tag, filled only there, + // since a class-object referrer has no parent entry to read from). + ChainHopEdge hop{}; + hop.field_index = entry.referrer_field_index; + if (entry.parent_tag == 0) { + hop.edge_kind = entry.root_kind; + hop.referrer_class_tag = entry.referrer_class_tag; + } else { + hop.edge_kind = entry.edge_kind; + FrontierEntry parent_entry{}; + hop.referrer_class_tag = + lookup(entry.parent_tag, &parent_entry) ? parent_entry.class_tag : 0; + } + edges.push_back(hop); + } + markEdge(tag); + root_kind = entry.root_kind; + tag = entry.parent_tag; + } + if (tag != 0) { + // Ran past the defensive hop bound without reaching a root-attached + // entry (parent_tag == 0) - a corrupted/cyclic chain. Report failure + // rather than returning a truncated, possibly-misleading chain. + return false; + } + + *out_chain = std::move(chain); + if (out_edges != nullptr) { + *out_edges = std::move(edges); + } + if (out_root_kind != nullptr) { + // The loop's last iteration is always the root-attached entry (the one + // whose parent_tag == 0 that just ended the loop), so root_kind here is + // that entry's own FrontierEntry::root_kind. + *out_root_kind = root_kind; + } + return true; +} + +// --------------------------------------------------------------------------- +// ReferenceChainTracker +// --------------------------------------------------------------------------- + +// Marks the calling thread as executing inside the GarbageCollectionStart/ +// Finish JVMTI callback for the duration of the guard's lifetime. Used by the +// tag helpers below as a debug-only self-consistency check that this class +// never issues a Heap-category JVMTI call (SetTag/GetTag/...) from a context +// where the JVMTI spec forbids it (see referenceChains.h). Thread-local +// because the JVMTI spec only guarantees the callback runs on the VM thread +// delivering the event, and this must not leak across threads. +static thread_local bool t_inGCCallback = false; + +namespace { +class GCCallbackGuard { +public: + GCCallbackGuard() { t_inGCCallback = true; } + ~GCCallbackGuard() { t_inGCCallback = false; } +}; +} // namespace + +void ReferenceChainTracker::autoTuneDefaults(Arguments &args) { + // Only tune defaults the operator did not set explicitly. + const u8 tuned = args._reference_chains_tuned_mask; + + // Max heap is resolved by LivenessTracker::initialize_table() at this + // point (ObjectSampler::start() -> LivenessTracker::start() runs + // before ReferenceChainTracker::start() in Profiler::start()). + jlong max_heap = LivenessTracker::instance()->maxHeapBytes(); + if (max_heap <= 0) { + return; // can't tune without heap size + } + + // Available processors from JVMTI (cached by FlightRecorder, but we + // can query JVMTI directly here). + jint nprocs = 1; + jvmtiEnv *jvmti = VM::jvmti(); + if (jvmti != nullptr) { + jvmti->GetAvailableProcessors(&nprocs); + } + if (nprocs < 1) nprocs = 1; + + // Heap size in MiB. + double heap_mib = (double)max_heap / (1024.0 * 1024.0); + + // --- Budget (edges per BFS pass) --- + // Scale with sqrt(heap_mib): a 4 GiB heap gets 2x, a 16 GiB heap + // gets 4x, a 64 GiB heap gets 8x the default 1000. This keeps the + // per-pass safepoint pause proportional to sqrt(heap) — the + // number of edges explored per pass grows, but not quadratically. + if (!(tuned & REF_CHAINS_TUNED_BUDGET)) { + int scaled = (int)(DEFAULT_REFERENCE_CHAINS_BUDGET * std::sqrt(heap_mib / 512.0)); + args._reference_chains_budget = std::max(DEFAULT_REFERENCE_CHAINS_BUDGET, + std::min(scaled, MAX_REFERENCE_CHAINS_BUDGET)); + } + + // --- First-pass budget --- + // The root enumeration pass is one-shot per search and can + // afford a much larger budget. Scale it 10x the per-pass budget + // so the first pass covers more roots. + if (!(tuned & REF_CHAINS_TUNED_FIRST_PASS_BUDGET)) { + int fpb = args._reference_chains_budget * 10; + args._reference_chains_first_pass_budget = std::min(fpb, + MAX_REFERENCE_CHAINS_FIRST_PASS_BUDGET); + } + + // --- TTL (per-search wall-clock lifetime) --- + // The search needs enough time to cover the heap at the tuned + // budget. At ~1 pass/sec, TTL_seconds >= heap_edges / budget. + // We don't know heap_edges, but it scales with heap size. Use + // heap_mib as a proxy: TTL = base_ttl * (heap_mib / 512), + // clamped to [60s, 30min]. + if (!(tuned & REF_CHAINS_TUNED_TTL)) { + long scaled_ttl = (long)(DEFAULT_REFERENCE_CHAINS_TTL_MS * (heap_mib / 512.0)); + scaled_ttl = std::max(DEFAULT_REFERENCE_CHAINS_TTL_MS, std::min(scaled_ttl, + (long)(30 * 60 * 1000))); // 30 min max + args._reference_chains_ttl_ms = scaled_ttl; + } + + // --- Frontier cap --- + // The frontier grows with the number of edges admitted per + // pass. Scale with budget so a larger budget doesn't + // immediately hit the cap. Use a floating-point ratio - integer + // division here would truncate the scale factor (e.g. a budget + // of 3741 against a default of 1000 would floor to a 3x + // multiplier instead of ~3.74x, undershooting the cap by ~20%). + if (!(tuned & REF_CHAINS_TUNED_FRONTIER_CAP)) { + int scaled_cap = (int)(DEFAULT_REFERENCE_CHAINS_FRONTIER_CAP * + ((double)args._reference_chains_budget / DEFAULT_REFERENCE_CHAINS_BUDGET)); + args._reference_chains_frontier_cap = std::max( + DEFAULT_REFERENCE_CHAINS_FRONTIER_CAP, + std::min(scaled_cap, MAX_REFERENCE_CHAINS_FRONTIER_CAP)); + } + + // --- Pause target --- + // More available processors = the JVM can afford a slightly + // longer per-pass safepoint without impacting application + // throughput. Scale linearly: 1 core = 50ms, 4 cores = 100ms, + // 8 cores = 150ms, capped at 50ms (the per-call STW cap from the + // safepoint budget model). + if (!(tuned & REF_CHAINS_TUNED_PAUSE_TARGET)) { + long scaled_pause = DEFAULT_REFERENCE_CHAINS_PAUSE_TARGET_MS * + (1 + (nprocs - 1) / 3); + args._reference_chains_pause_target_ms = std::min(scaled_pause, (long)50); + } + + // --- Pain budget percent --- + // More cores = more spare capacity for background work. + // Scale: 1 core = 1%, 4 cores = 2%, 8 cores = 3%, capped at 5%. + if (!(tuned & REF_CHAINS_TUNED_PAIN_BUDGET)) { + int scaled_pain = DEFAULT_REFERENCE_CHAINS_PAIN_BUDGET_PERCENT * + (1 + (nprocs - 1) / 4); + args._reference_chains_pain_budget_percent = std::min(scaled_pain, 5); + } + + Log::info("Reference chain auto-tuner: heap=%.0f MiB nprocs=%d -> " + "budget=%d ttl=%ldms framecap=%d pausetarget=%ldms painbudget=%d%% firstpassbudget=%d", + heap_mib, (int)nprocs, + args._reference_chains_budget, args._reference_chains_ttl_ms, + args._reference_chains_frontier_cap, + args._reference_chains_pause_target_ms, + args._reference_chains_pain_budget_percent, + args._reference_chains_first_pass_budget); +} + +Error ReferenceChainTracker::start(Arguments &args) { + _enabled = args._reference_chains; + + if (!_enabled) { + Log::info("Reference chain tracking is disabled"); + return Error::OK; + } + + // Auto-tune defaults that the operator did not set explicitly, + // based on max heap size and available processors. Must run before + // _configured_frontier_cap is read below. + autoTuneDefaults(args); + + Log::info("Reference chain tracking is enabled (hops=%d, budget=%d, " + "ttl=%ldms, framecap=%d, pausetarget=%ldms, painbudget=%d%%)", + args._reference_chains_hop_cap, args._reference_chains_budget, + args._reference_chains_ttl_ms, args._reference_chains_frontier_cap, + args._reference_chains_pause_target_ms, + args._reference_chains_pain_budget_percent); + + // Like LivenessTracker's table (livenessTracker.cpp:225-232), construct the + // frontier table once and keep it across repeated start()/stop() cycles - + // do not reallocate on a second start() with a possibly different cap, for + // the same reason LivenessTracker keeps its first-initialize() result. + // Recorded unconditionally, even on a start() call that finds _frontier + // already constructed (see _configured_frontier_cap's own comment) - this + // is what resetSearchStateForTest() rebuilds the table at, undoing + // whatever cap an earlier test in this same JVM happened to construct it + // with. + _configured_frontier_cap = args._reference_chains_frontier_cap; + if (_frontier == nullptr) { + _frontier = new FrontierTable(_configured_frontier_cap); + } + + _hop_cap = args._reference_chains_hop_cap; + _budget = args._reference_chains_budget; + // 0 (unset) auto-scales from _budget instead of falling back to it plainly + // - see this field's own comment (referenceChains.h) for why a + // steady-state per-pass budget is the wrong size for the first pass. + _first_pass_budget = args._reference_chains_first_pass_budget > 0 + ? args._reference_chains_first_pass_budget + : std::min(_budget * AUTO_FIRST_PASS_BUDGET_MULTIPLIER, + AUTO_FIRST_PASS_BUDGET_CAP); + _ttl_ms = args._reference_chains_ttl_ms; + + // Pause-time pacing controller: (re)seed the controller's ceiling and the + // adaptive values it drives. _effective_budget/_effective_cadence_ns start + // exactly at their pre-pacing-controller fixed-constant equivalents + // (_budget/PASS_CADENCE_NS) so a tracker that has not yet measured a pass + // behaves identically to before the controller was added - updatePacing() + // only moves them once a real pass duration is + // available. _pause_pid is reconstructed (not just reset()) because its + // target is only known now, from args - same reason RateLimiter::start() + // reconstructs its own _pid rather than mutating it in place. + _pause_target_ms = args._reference_chains_pause_target_ms; + _effective_pause_target_ms = _pause_target_ms; + _effective_budget = _budget; + _effective_cadence_ns = PASS_CADENCE_NS; + _candidate_count = 0; + _candidate_found_bits = 0; + memset(_candidate_discovered_count, 0, sizeof(_candidate_discovered_count)); + memset(_candidate_qualifying_tid_count, 0, + sizeof(_candidate_qualifying_tid_count)); + _passes_since_last_candidate_progress = 0; + _last_candidate_progress_mark = 0; + // Fresh chase gets a fresh back-to-back spacing allowance (see + // _canary_backoff_mult's own comment). + _canary_backoff_mult = 1; + _canary_pass_ema_ms = 0; + _last_canary_pass_ns = 0; + _canary_stuck_restart_count = 0; + // Budget-borrowing (referenceChains.h's _borrowed_budget comment): reset + // alongside the rest of the pacing controller's state, so a restarted + // search never inherits headroom earned by a previous one. + _borrowed_budget = 0; + _consecutive_under_target_passes = 0; + _pause_pid = PidController((u64)std::max(_pause_target_ms, 0L), + 10, // proportional gain: reacts to a single + // pass's over/under-ceiling error without + // needing many passes to notice - a + // duration-ms error is typically single/ + // low-double-digit in magnitude (unlike + // the shared triple's event-count scale), + // so a smaller P keeps a one-pass + // overshoot from swinging the budget by + // more than a modest fraction of itself + 1, // integral gain: small and round - + // pidController.cpp's `_integral_value` + // has no built-in clamp, and this + // controller is invoked once per BFS pass + // rather than on the other three usages' + // roughly-periodic one-call-per-second + // cadence, so windup accumulates faster + // per wall-clock second than it does there + 2, // derivative gain: small, matching the + // shared triple's own "the derivational + // gain is rather small" rationale + // (objectSampler.cpp) - a single slow/ + // fast pass should not itself trigger a + // large swing + 1, // sampling_window=1: one compute() call + // *is* one pass, not a fixed real-time + // window like the other three usages + // assume (see _pause_pid's own comment) + 5.0 // cutoff_secs: a round value, halved from + // the shared triple's own "15" since a + // pass-scoped signal is naturally + // noisier per-call than a roughly-1s- + // cadence one + ); + + // Search restart (this class's own header comment): (re)seed _safepoint_pain_budget + // from the configured refill rate, mirroring _pause_pid's own + // reconstruct-in-start() pattern above. A search's already-accumulated + // _search_pain_ms is deliberately left untouched here - only restartSearch() + // spends it, so a start()/stop() cycle mid-search (if that ever happens) + // does not erase cost the current search has already incurred. + _safepoint_pain_budget = PainBudget( + std::max(args._reference_chains_pain_budget_percent, 0) / 100.0); + _pain_budget_refill_rate = std::max(args._reference_chains_pain_budget_percent, 0) / 100.0; + // Same refill rate as _safepoint_pain_budget above - one operator-facing + // "how much background cost is acceptable" percentage covers both + // leaky buckets (see _cpu_pain_budget's own comment, referenceChains.h). + _cpu_pain_budget = PainBudget(_pain_budget_refill_rate); + + // Lazy-enable, matching LivenessTracker::start() (livenessTracker.cpp:194-196): + // the GC callbacks are wired unconditionally in vmEntry.cpp, but the events + // themselves are only turned on for this JVMTI env when the flag is on. + jvmtiEnv *jvmti = VM::jvmti(); + jvmti->SetEventNotificationMode( + JVMTI_ENABLE, JVMTI_EVENT_GARBAGE_COLLECTION_START, nullptr); + jvmti->SetEventNotificationMode( + JVMTI_ENABLE, JVMTI_EVENT_GARBAGE_COLLECTION_FINISH, nullptr); + + // Deliberately does NOT create the BFS thread (threadEntry()/threadLoop() + // below) here - threadLoop()'s VM::attachThread() call dereferences + // VM::_vm unconditionally (vmEntry.h:191-195) and crashes if the VM is not + // yet attached, which is exactly the case in this file's own gtest binary + // (referenceChains_ut.cpp calls start() directly with no live JVM). + // startThread() (referenceChains.h) owns spawning the thread instead, and + // is called from Profiler::start() (profiler.cpp) immediately after this + // method returns Error::OK - by that point in the real profiler lifecycle + // the JVM/JVMTI environment is already fully up, so VM::attachThread() is + // safe there. runPass() - the actual BFS engine - does not depend on the + // thread either way and is called directly by this file's own tests. + + return Error::OK; +} + +void ReferenceChainTracker::stop() { + if (!_enabled) { + return; + } + Log::info("Reference chain tracking stopped"); + + // Do not disable GC notifications here - LivenessTracker follows the same + // rule (livenessTracker.cpp:209-210) since the JVMTI env and its tracker + // singletons are expected to survive across multiple start/stop recording + // cycles. The BFS thread itself is stopped separately, by + // Profiler::stop() calling stopThread() (profiler.cpp) - mirroring + // start()'s split between this method and startThread(). +} + +void ReferenceChainTracker::startThread() { + if (!_enabled || _running.load(std::memory_order_acquire)) { + return; + } + // Reset from any previous stopThread() call - a dynamic-attach profiler + // can go through multiple start()/stop() cycles in one JVM lifetime (this + // class's own start()/stop() header comments), and a stale abort request + // left set from the prior cycle would make heapReferenceCallback() abort + // this new cycle's very first pass instantly. + _abort_pass_requested.store(false, std::memory_order_relaxed); + + // Publish _running=true *before* creating the thread, not after. If the + // OS schedules the new thread ahead of the parent, threadLoop()'s startup + // check (`while (_running.load(...))`) would otherwise be racing against + // this store: the child could see the still-`false` initial value, fall + // straight through the loop, detach and exit - and the parent would then + // publish `true` regardless, leaving startThread() reporting the tracker + // as running while no BFS thread is actually alive for the rest of the + // recording. pthread_create() itself is the fix's synchronization point: + // POSIX guarantees everything the calling thread writes before this call + // is visible to the new thread once it starts running, so ordering the + // store first removes the race outright rather than narrowing it. Roll + // back on a failed create so a later startThread() call is not blocked by + // a stale `_running=true` with no thread behind it. stopThread() is only + // ever called after this method has returned (Profiler::start()/stop() + // pair the two sequentially - see this class's own start()/stop() header + // comments), so its use of _thread below is unaffected by this reordering. + _running.store(true, std::memory_order_release); + pthread_t thread; + if (pthread_create(&thread, NULL, threadEntry, this) != 0) { + Log::warn("Unable to create ReferenceChains BFS thread"); + _running.store(false, std::memory_order_release); + return; + } + _thread = thread; +} + +void ReferenceChainTracker::stopThread() { + if (!_running.load(std::memory_order_acquire)) { + return; + } + _running.store(false, std::memory_order_release); + // Ask any in-flight JVMTI FollowReferences walk (heapReferenceCallback()) + // to abort at its next callback invocation - set before pthread_kill() + // below, since that signal alone cannot interrupt a call already inside + // the JVM/JVMTI implementation. + _abort_pass_requested.store(true, std::memory_order_relaxed); + // Same wake-then-join shape as BaseWallClock::stop() (wallClock.cpp:324-333): + // pthread_kill(WAKEUP_SIGNAL) interrupts threadLoop()'s OS::sleep() early + // (WAKEUP_SIGNAL/SIGIO is installed with a no-op handler unconditionally + // in vmEntry.cpp, so this signal never terminates the thread) so it + // re-checks _running and exits promptly rather than waiting out the rest + // of the current sleep interval. + pthread_kill(_thread, WAKEUP_SIGNAL); + int res = pthread_join(_thread, NULL); + if (res != 0) { + Log::warn("Unable to join ReferenceChains BFS thread on stop %d", res); + } +} + +// Not yet started by anything (see start()'s comment above for why) - but +// now implements the real scheduling loop the design doc asks for, matching +// J9WallClock's attach/park/detach lifecycle (j9WallClock.cpp:28-57): each +// wake (adaptive cadence, or earlier via onGCFinish()'s pthread_kill below) +// checks shouldRunPass() and calls runPass() if it says so. The pause-time +// pacing controller sleeps for _effective_cadence_ns rather than the fixed +// PASS_CADENCE_NS, so a +// controller-driven relaxed cadence (updatePacing()) actually shortens how +// long an idle, no-GC-event search waits between passes, not just +// shouldRunPass()'s own comparison. +void ReferenceChainTracker::threadLoop() { + struct Cleanup { + ReferenceChainTracker *tracker; + ~Cleanup() { + // No cached-class cleanup needed before detaching: + // _cached_object_class is a global ref, deliberately valid across + // attach/detach cycles (see its own comment in referenceChains.h) - + // unlike the per-attach local ref it replaced, which this destructor + // used to have to clear here. + VM::detachThread(); + } + } cleanup{this}; + JNIEnv *jni = VM::attachThread("java-profiler ReferenceChains"); + jvmtiEnv *jvmti = VM::jvmti(); + if (jni == nullptr) { + // AttachCurrentThreadAsDaemon() failed - mirror pollWatchedTargets()'s + // own jni==nullptr early return rather than letting a null JNIEnv flow + // into runPass()/resolveLoadedClasses()/expandFrontier()/ + // releaseSearchTags() below: those only guard their DeleteLocalRef() + // calls on `jni != nullptr`, so without this check every + // GetLoadedClasses()/GetObjectsWithTags() local ref returned on this + // (permanently un-attached) thread would leak for the rest of the + // process's lifetime. Nothing this thread does is safe without a live + // JNIEnv, so give up on the whole loop rather than retrying per + // iteration - detachThread() in Cleanup is a safe no-op if attach never + // actually succeeded. + Log::warn("ReferenceChains: VM::attachThread failed; BFS thread exiting"); + return; + } + DEBUG_ONLY(rcDebugLevelRefresh(true)); // apply the override file before the first log line + TEST_LOG_SUMMARY("ReferenceChainTracker::threadLoop started, cadence=%lluns rc_debug_level=%d", (unsigned long long)_effective_cadence_ns, rcDebugLevel()); + + int iteration = 0; + while (_running.load(std::memory_order_acquire)) { + // Fixed ~1s cadence, no early wake on GC (see onGCFinish()'s own + // comment) - stopThread() still interrupts this via its own + // pthread_kill so shutdown stays prompt. + // Urgency-driven dynamic tuning: as secondsToOOM() falls within + // OOM_RAMP_START_S of projected exhaustion, ramp the per-pass pause + // target and cadence exponentially toward their ceilings (see + // OOM_RAMP_START_S/URGENT_PAUSE_TARGET_MS/URGENT_CADENCE_NS's own + // comments) - slow at the 30-minute mark, aggressive right before OOM. + // secondsToOOM() itself already gates on a confirmed rising trend (its + // NOT_RISING check), so a non-negative value here is real growth, not + // noise. The PID controller is reconstructed whenever the (rounded) + // target changes so its ceiling tracks the new value. + double seconds_to_oom = LivenessTracker::instance()->secondsToOOM(); + bool urgent = seconds_to_oom >= 0 && seconds_to_oom < OOM_RAMP_START_S; + long target_ms = _pause_target_ms; + u64 cadence_ns = _effective_cadence_ns; + if (urgent) { + double x = 1.0 - seconds_to_oom / OOM_RAMP_START_S; // 0 at 30min out, 1 at OOM + target_ms = std::lround(_pause_target_ms * + std::pow((double)URGENT_PAUSE_TARGET_MS / std::max(_pause_target_ms, 1L), x)); + // Ramp from the fixed configured cadence, not the currently-adaptive + // _effective_cadence_ns - using the live value as the ramp's own + // moving anchor would compound the exponent across iterations instead + // of tracking urgency directly from a stable baseline. + cadence_ns = (u64)std::llround((double)PASS_CADENCE_NS * + std::pow((double)URGENT_CADENCE_NS / (double)PASS_CADENCE_NS, x)); + // While urgent, the ramp owns _effective_cadence_ns outright so + // shouldRunPass()'s cadence gate and the per-pass log actually + // reflect it. updatePacing()'s own overflow-driven widen/narrow + // adjustment (see _effective_cadence_ns's header comment) resumes + // sole ownership the instant urgency clears - this block simply stops + // touching the field then, so there is nothing to snap back from. + _effective_cadence_ns = cadence_ns; + } + if (target_ms != _effective_pause_target_ms) { + _effective_pause_target_ms = target_ms; + _pause_pid = PidController((u64)std::max(_effective_pause_target_ms, 0L), + 10, 1, 2, 1, 5.0); + // Once in the ramp window, hold the budget ceiling raised for its + // entire duration rather than only right before OOM: the process is + // likely to die anyway, so it's worth spending whatever budget it + // takes to collect good diagnostic data for as long as we have. + if (urgent) { + _budget = std::min(_budget * 4, MAX_REFERENCE_CHAINS_BUDGET); + } + TEST_LOG_SUMMARY("ReferenceChainTracker::threadLoop urgency=%d pauseTarget=%ldms " + "cadence=%lluns budget=%d", + (int)urgent, _effective_pause_target_ms, + (unsigned long long)cadence_ns, _budget); + } + // Third trigger for LivenessTracker::cleanup_table() (see + // LivenessTracker::maybeForceCleanup()'s own comment): track()'s + // table-overflow branch and flush_table()'s JFR cadence can both starve + // under ObjectSampler's PID-controlled sampling interval, leaving + // hasLeakSignal() below stuck on a stale population history no matter + // how long a real leak keeps growing. This thread already wakes every + // ~1s with a live JNIEnv, so it doubles as that fallback tick - cheap, + // and a no-op unless 30s have actually elapsed with a GC in between (see + // that method for the exact gate). + u64 wake_now_ns = OS::nanotime(); + LivenessTracker::instance()->maybeForceCleanup(wake_now_ns); + + // No fast-path skip here: shouldRunPass() below already returns false + // cheaply (a couple of atomic loads/comparisons, no JVMTI call) for a + // RUNNING search with no new GC and cadence not yet elapsed. An earlier + // revision additionally gated this on hasLeakSignal() (LivenessTracker's + // population-trend signal, also used by canAffordNewSearch() below to gate + // the first-ever search and every restart), but that signal answers "is + // there a leak candidate right now", which is unrelated to whether an + // already-RUNNING search's own frontier still has pending work - gating a + // RUNNING search's every pass on it would stall that search's own + // convergence for as long as no leak candidate happens to be visible, + // even with GC epochs advancing or cadence elapsed. hasLeakSignal() + // remains the right gate for starting a *new* search, whether that is the + // first one ever or a restart of a *terminal* one (shouldRunPass()'s own + // canAffordNewSearch() call). + u64 now_ns = OS::nanotime(); + + // Re-check the runtime debug-level override file (~1s TTL; see + // rcDebugLevel.h) - never in heap callbacks, which only read the + // cached atomic. + DEBUG_ONLY(rcDebugLevelRefresh()); + + // Hand this iteration's ramp state to shouldRunPass() before it decides + // - the canary-backoff gate is bypassed while the OOM urgency ramp is + // active (see _oom_ramp_active's own comment) - and raise + // LivenessTracker's tracking admission to 100% for the same ramp + // (see setUrgentTracking()'s own comment, livenessTracker.h): same state, + // same iteration, so the boost tracks the ramp exactly, engaging and + // releasing together. + _oom_ramp_active = urgent; + LivenessTracker::instance()->setUrgentTracking(urgent); + + bool should_run = shouldRunPass(now_ns); + // Only sleep when idle (no pass will run). When a canary + // search is active or a pass is about to run, skip the + // sleep to run passes back-to-back. + if (!should_run && cadence_ns > 0) { + OS::sleep(cadence_ns); + if (!_running.load(std::memory_order_acquire)) { + break; + } + now_ns = OS::nanotime(); + } + // Log the loop state only when a pass is actually going to run - the idle + // wakes (should_run == false) are the common steady state and logging them + // every second is pure noise. + if (should_run) { + TEST_LOG_SUMMARY("ReferenceChainTracker::threadLoop iteration=%d shouldRunPass=%d searchState=%d " + "passesRun=%d effectiveCadenceNs=%llu effectiveBudget=%d gcFinishEpoch=%llu " + "lastPassGcFinishEpoch=%llu nowMinusLastPassNs=%llu", + ++iteration, should_run, (int)_search_state, _passes_run, + (unsigned long long)_effective_cadence_ns, _effective_budget, + (unsigned long long)gcFinishEpoch(), (unsigned long long)_last_pass_gc_finish_epoch, + (unsigned long long)(now_ns - _last_pass_ns)); + runPassSerialized(jvmti, jni); + } + // Target-selection bridging step: poll once per scheduling cycle, after + // runPass() - so this poll always sees the most recent pass's tagging (see + // pollWatchedTargets()'s own comment). Unconditional, not gated on + // shouldRunPass()'s decision above: a candidate discovered by an + // earlier pass may still be waiting for its first poll even on a cycle + // where this cycle's own pass was skipped. + pollWatchedTargetsSerialized(jvmti, jni); + } +} + +void JNICALL ReferenceChainTracker::GarbageCollectionStart(jvmtiEnv *jvmti_env) { + ReferenceChainTracker::instance()->onGCStart(); +} + +void JNICALL ReferenceChainTracker::GarbageCollectionFinish(jvmtiEnv *jvmti_env) { + ReferenceChainTracker::instance()->onGCFinish(); +} + +void ReferenceChainTracker::onGCStart() { + if (!_enabled) { + return; + } + // JVMTI spec: only Memory Management category calls (Allocate/Deallocate) + // are allowed from inside this callback - nothing else may run here. + GCCallbackGuard guard; + atomicIncRelaxed(_gc_start_epoch, (u64)1); +} + +void ReferenceChainTracker::onGCFinish() { + if (!_enabled) { + return; + } + GCCallbackGuard guard; + // Design doc's Triggering section: GC callbacks are only a scheduling + // *signal*, never a pass's execution vehicle (Heap-category JVMTI calls + // are forbidden here - see this file's header comment). Deliberately just + // bookkeeping - no pthread_kill/early wake here. threadLoop() below wakes + // on its own fixed ~1s cadence and reads this epoch then; waking it early + // on every GC gains at most ~1s of latency but, under any GC-heavy + // workload, collapses the loop's cadence to GC frequency instead (each + // early wake is itself a full iteration's worth of shouldRunPass()/ + // pollWatchedTargets() work), which is not worth the latency win. + atomicIncRelaxed(_gc_finish_epoch, (u64)1); +} + +bool ReferenceChainTracker::shouldRunPass(u64 now_ns) { + if (!_search_started) { + // Same gate as a restart (canAffordNewSearch() below) - a brand-new + // tracker must not pay for the first whole-heap walk/tagging pass either + // when there is no leak candidate to justify it. The pain-budget half is + // always a no-op here (nothing has ever been spent yet), so this reduces + // to hasLeakSignal() in practice, but sharing the one gate keeps both + // call sites from drifting apart. + bool afford = canAffordNewSearch(now_ns); + TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass search_not_started " + "canAffordNewSearch=%d", (int)afford); + if (!afford) { + return false; + } + // This episode's one urgency-authorized search (_urgent_search_spent's + // own comment, referenceChains.h) is the one about to start. + _urgent_search_spent = _urgent_latched; + TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass -> true (search not started yet)"); + return true; // nothing has run yet - always worth taking the first pass + } + if (_search_state != SearchState::RUNNING) { + // Terminal outcome already reached (runPass()'s Termination section). + if (!_tags_released) { + // releaseSearchTags() failed to confirm every live tag this search + // owned was actually cleared - restartSearch() must never run until + // that is confirmed (see _tags_released's own comment), so return + // true unconditionally here: that drives threadLoop() to call + // runPass() again, whose terminal-state branch retries the release, + // rather than letting canAffordNewSearch()/restartSearch() below run + // ahead of it. + TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass -> true (retrying tag " + "release before restart is allowed)"); + return true; + } + // Restart (this class's own header comment) if the pain budget has + // drained and there is still (or again) a leak indication to chase - + // canAffordNewSearch() is always true when LivenessTracker's population + // trends are not in use at all, so this only ever changes behavior for a + // search that already ran once. + if (canAffordNewSearch(now_ns)) { + // Same entitlement bookkeeping as the first-search branch above. + _urgent_search_spent = _urgent_latched; + restartSearch(); + TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass -> true (restarting search)"); + return true; + } + // No log here: a terminal search waiting for a restart to become + // warranted is the common idle state, re-evaluated every second, so + // logging it is pure per-second noise (see threadLoop()). + TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass terminal_blocked " + "tags_released=%d safepoint_pain=%d search_state=%d", + (int)_tags_released, + (int)_safepoint_pain_budget.canStartNow(now_ns), + (int)_search_state); + return false; + } + // Canary search active with candidates still to find - computed ahead of + // the pain-budget check below so the refill-rate raise and the backoff + // gate further down agree on the same snapshot of _candidate_found_bits. + bool canary_active = _candidate_count > 0 && + __builtin_popcountll(_candidate_found_bits) < (u64)_candidate_count; + // Adaptive CPU budget: 100x refill while a canary chase is open - NOT a + // rate control (the canary lane's rate is bounded by _canary_backoff_ns's + // progress-driven exponential backoff, see its own comment) but a + // double-throttle guard: the base refill rate is tuned for the ordinary + // ~1 pass/s whole-graph cadence and would otherwise starve a chase the + // backoff has already paced. 1x in every other mode. + double multiplier = + canary_active ? CANARY_PAIN_BUDGET_REFILL_MULTIPLIER : 1.0; + _cpu_pain_budget.setRefillRate( + std::min(_pain_budget_refill_rate * multiplier, 1.0), + now_ns); + if (!_cpu_pain_budget.canStartNow(now_ns)) { + TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass blocked by cpu_pain_budget " + "balance=%.1fms refill_rate=%.4f canary_active=%d " + "multiplier=%.1f", + _cpu_pain_budget.balanceMs(now_ns), + std::min(_pain_budget_refill_rate * multiplier, 1.0), + (int)canary_active, multiplier); + return false; + } + if (canary_active) { + // Canary-lane pacing: the chase's rate bound - work-scaled spacing + // (_canary_backoff_mult's own comment for the law and the live burn it + // bounds). Deliberately placed ABOVE the gc-finish-epoch trigger below: + // a GC-heavy workload bumps the epoch on virtually every wake (minor + // young GCs included), so letting the epoch trigger bypass the backoff + // would make the backoff unreachable exactly on the GC-churning + // deployments that burn the most. The pass that eventually runs sees + // whatever the graph looks like then - freshness is not lost, only + // re-checked at the paced rate. The OOM urgency ramp is the one + // override (see _oom_ramp_active's own comment). + u64 spacing_ns = + (u64)_canary_backoff_mult * _canary_pass_ema_ms * 1000000ULL; + // mult == 1 (fresh chase, or last pass made progress) means the gate is + // OFF - the chase runs at its natural pass rate, one pass starting as + // soon as the last ended. + if (!_oom_ramp_active && _canary_backoff_mult > 1 && + now_ns - _last_canary_pass_ns < spacing_ns) { + TEST_LOG("ReferenceChainTracker::shouldRunPass held off by canary " + "backoff mult=%d ema_ms=%llu since_last_pass=%llums", + _canary_backoff_mult, + (unsigned long long)_canary_pass_ema_ms, + (unsigned long long)((now_ns - _last_canary_pass_ns) / + 1000000ULL)); + return false; + } + TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass -> true (canary search, " + "%d/%d candidates found, backoff_mult=%d ema_ms=%llu)", + (int)__builtin_popcountll(_candidate_found_bits), + (int)_candidate_count, _canary_backoff_mult, + (unsigned long long)_canary_pass_ema_ms); + return true; + } + u64 gc_finish_epoch = gcFinishEpoch(); + if (gc_finish_epoch != _last_pass_gc_finish_epoch) { + // Triggering section: "a GC just happened, a pass may be worth running + // soon". + TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass -> true (gcFinishEpoch=%llu != " + "lastPassGcFinishEpoch=%llu)", + (unsigned long long)gc_finish_epoch, + (unsigned long long)_last_pass_gc_finish_epoch); + return true; + } + // Pause-time pacing controller: compares against _effective_cadence_ns, not + // the fixed PASS_CADENCE_NS - see that + // field's own comment (referenceChains.h) for how updatePacing() widens or + // relaxes it from the measured pause-time signal. + bool cadence_elapsed = now_ns - _last_pass_ns >= _effective_cadence_ns; + // Only log when the cadence actually elapsed (a pass will run). The + // not-yet-elapsed case is the common idle wake and logging it every second + // is noise. + if (cadence_elapsed) { + TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass -> true (now_ns=%llu last_pass_ns=%llu " + "delta=%llu effectiveCadenceNs=%llu)", + (unsigned long long)now_ns, (unsigned long long)_last_pass_ns, + (unsigned long long)(now_ns - _last_pass_ns), + (unsigned long long)_effective_cadence_ns); + } + return cadence_elapsed; +} + +// Search restart gate (this class's own header comment). Deliberately a +// probe (max=1) rather than reusing pollWatchedTargets()'s own +// selectLeakCandidates() call - that one runs after runPass() in +// threadLoop()'s own iteration and needs the *list* to poll each candidate's +// tag; this only needs to know whether at least one exists. +// Latching, hysteretic read of LivenessTracker::secondsToOOM() - see +// _urgent_latched's own comment (referenceChains.h) for why a bare threshold +// comparison here flaps, and OOM_URGENT_RELEASE_S for the release bar. +bool ReferenceChainTracker::isUrgent() const { + double seconds_to_oom = LivenessTracker::instance()->secondsToOOM(); + if (seconds_to_oom >= 0 && seconds_to_oom < OOM_URGENT_THRESHOLD_S) { + _urgent_release_ticks = 0; + if (!_urgent_latched) { + _urgent_latched = true; + // A fresh episode gets a fresh entitlement to one search. + _urgent_search_spent = false; + TEST_LOG_SUMMARY("ReferenceChainTracker::isUrgent latching urgency " + "(secondsToOOM=%.1f < OOM_URGENT_THRESHOLD_S=%.1f)", + seconds_to_oom, OOM_URGENT_THRESHOLD_S); + } + return true; + } + if (_urgent_latched) { + // Negative means "no rising trend to project from" (secondsToOOM()'s own + // unknown/NOT_RISING encoding), which counts toward release just like a + // comfortably distant projection does. + if (seconds_to_oom < 0 || seconds_to_oom >= OOM_URGENT_RELEASE_S) { + if (++_urgent_release_ticks >= URGENT_RELEASE_CONSECUTIVE) { + _urgent_latched = false; + _urgent_release_ticks = 0; + _urgent_search_spent = false; + TEST_LOG_SUMMARY("ReferenceChainTracker::isUrgent releasing urgency " + "(secondsToOOM=%.1f clear of OOM_URGENT_RELEASE_S=%.1f for " + "%d consecutive observations)", + seconds_to_oom, OOM_URGENT_RELEASE_S, + URGENT_RELEASE_CONSECUTIVE); + return false; + } + } else { + // Between the two bars, or a single noisy reading past the release bar + // followed by one that is not - neither releases the latch. + _urgent_release_ticks = 0; + } + return true; + } + return false; +} + +bool ReferenceChainTracker::hasLeakSignal() { + if (!LivenessTracker::instance()->gcGenerationsEnabled()) { + // No population-trend signal to gate on at all - see this method's own + // header comment for why that means "always true" here. + return true; + } + double seconds_to_oom = LivenessTracker::instance()->secondsToOOM(); + // isUrgent() is called unconditionally, not short-circuited behind + // _urgent_search_spent: it is what maintains the latch/release counter, so + // skipping it would freeze the episode state (see _urgent_latched). + bool urgent = isUrgent(); + if (urgent && !_urgent_search_spent) { + // Heap-wide floor is rising fast enough that OOM is imminent - don't + // wait for a specific klass to clear selectLeakCandidates()'s own + // per-klass ring-fill/hysteresis gate; see OOM_URGENT_THRESHOLD_S's own + // comment (referenceChains.h) for why that gate alone is too slow here. + // canAffordNewSearch() can still defer this via the pain-budget check it + // runs before calling this method (this method's own header comment). + // + // Only until this episode's one search has been started + // (_urgent_search_spent): past that point the per-klass probe below is + // the sole remaining trigger, so a completed urgent search is not torn + // down and restarted from scratch on the very next tick. + TEST_LOG_SUMMARY("ReferenceChainTracker::hasLeakSignal -> true (urgent, " + "secondsToOOM=%.1f)", + seconds_to_oom); + return true; + } + KlassCandidate probe[1]; + int n = LivenessTracker::instance()->selectLeakCandidates(probe, 1); + TEST_LOG_SUMMARY("ReferenceChainTracker::hasLeakSignal -> %s (secondsToOOM=%.1f, " + "candidates=%d, urgent=%d, urgentSearchSpent=%d)", + n > 0 ? "true" : "false", seconds_to_oom, n, urgent, + _urgent_search_spent); + return n > 0; +} + +bool ReferenceChainTracker::canAffordNewSearch(u64 now_ns) { + if (!_safepoint_pain_budget.canStartNow(now_ns)) { + TEST_LOG_SUMMARY("ReferenceChainTracker::canAffordNewSearch blocked by " + "safepoint_pain_budget balance=%.1fms refill_rate=%.4f", + _safepoint_pain_budget.balanceMs(now_ns), + _pain_budget_refill_rate); + return false; // still cooling down from the last search's own cost + } + return hasLeakSignal(); +} + +// Search restart (this class's own header comment). Called only from +// shouldRunPass() once canAffordNewSearch() has approved it, immediately +// before returning true for this same iteration - runPass() then sees +// _search_started == false and takes the first-pass branch, exactly like a +// brand-new tracker. +void ReferenceChainTracker::restartSearch() { + // Only called once shouldRunPass() has confirmed _tags_released - never + // while a prior search's release might still be pending (see + // _tags_released's own comment): resetting _next_tag to 1 / the frontier + // table below while some object could still hold this search's now- + // ambiguous tag would let the restarted search's fresh tags collide with + // it. + assert(_tags_released && + "restartSearch() must not run before releaseSearchTags() has " + "confirmed every live tag was cleared"); + + // Spend the finishing search's own cost before clearing the accumulator - + // canAffordNewSearch()'s *next* call must see this search's cost, not a + // reset-to-zero balance. + _safepoint_pain_budget.spend(_search_pain_ms); + _search_pain_ms = 0; + + if (_frontier != nullptr) { + _frontier->resetForRestart(); + } + _next_tag = 1; + // Hop-edge label cache: keyed by raw class tags, which survive a restart + // (the shared class-tag allocator is deliberately not reset - see this + // method's own declaration comment) - but the frontier entries referencing + // them do not, and a restart is the natural bounded clear point for a + // cache capped by HOP_LABEL_CLASS_CACHE_CAP wholesale. + _hop_label_cache.clear(); + // The shared class-tag counter (classTagAllocator.h)/_class_tags + // intentionally untouched - see this method's own declaration comment + // (referenceChains.h). + + _search_started = false; + store(_search_state, (u8)SearchState::RUNNING); + store(_abandon_reason, (u8)SearchAbandonReason::NONE); + store(_search_start_ns, (u64)0); + _pending_expand.clear(); + _priority_expand.clear(); + _priority_expand_set.clear(); + _static_anchor_fifo.clear(); + _static_anchor_fifo_set.clear(); + _static_anchor_fifo_klass_counts.clear(); + _static_anchor_index.clear(); + _static_anchor_own_class_tags.clear(); + _static_anchor_index_tags.clear(); + _anchor_container_cursor = 0; + _anchor_other_cursor = 0; + // Fresh lane: nothing admits before the search does, so nothing can + // have a pending first look either. + _static_anchor_fresh_queue.clear(); + // Discovered-instance tags are FRONTIER tags - the reset above just + // invalidated every one of them (fresh tags restart from 1). Leaving + // the slots populated lets pollWatchedTargets() resolve stale tags into + // whatever unrelated object the new search assigns them to: a dead + // slot fails reconstructChain() (observed on-pod round 13: 'buildChainEvent + // failed ... reconstructChain failed for target_tag=8851'), and a live + // one emits a chain event for the WRONG OBJECT (the likely origin of + // the earlier session's noise-[B event). _class_shape_cache is + // deliberately NOT cleared here: class tags are stable for the JVM's + // lifetime (the class-tag allocator is not reset), so a classification + // remains valid across searches. + memset(_candidate_discovered_tags, 0, sizeof(_candidate_discovered_tags)); + memset(_candidate_discovered_count, 0, sizeof(_candidate_discovered_count)); + // Both keyed by frontier tags this restart is about to invalidate (fresh + // tags start again from 1) - a stale entry surviving past a restart would + // be compared against whatever unrelated object the new search has since + // reassigned that tag to. _watched_leak_klass_ids itself is left alone: + // it reflects LivenessTracker's own growth signal, unrelated to this + // search's lifecycle, and simply gets refreshed again on the next + // pollWatchedTargets() tick regardless. + _leak_signature_totals.clear(); + _leak_signature_prev_totals.clear(); + _leak_parent_fanout.clear(); + _leak_tags_assigned = 0; + _leak_tags_resolved = 0; + _last_pass_gc_finish_epoch = 0; + store(_last_pass_ns, (u64)0); + store(_passes_run, 0); + // Reset back to their just-constructed values (0 / -1) like every other + // per-search field this method touches: resolveLoadedClasses() and + // admitStaticFieldRoots() must both run unconditionally on the restarted + // search's first pass, exactly as they do for a brand-new tracker. + _last_resolved_class_count = 0; + _last_static_field_class_count = -1; + // _resolved_chains is intentionally left intact: a chain resolved by the + // finishing search stays cached (and keeps being re-emitted on every dump) + // across the restart, since it describes a sample that is still live. The + // restarted search re-tags that sample under a fresh _search_start_ns, and + // pollWatchedTargets() refreshes the cached entry then (its own comment); + // it prunes the entry if the sample has since been collected. +} + +void ReferenceChainTracker::resetSearchStateForTest(jvmtiEnv *jvmti, + JNIEnv *jni) { + // Every field touched below is otherwise only ever mutated by the BFS + // thread itself (threadLoop()/runPass()/pollWatchedTargets()) - without + // stopping it first, a pass already in flight on that thread can observe + // this reset only partially, or overwrite it right back (e.g. finish a + // pass that was already headed for SearchState::ABANDONED after this + // method has just forced SearchState::RUNNING below), a race found in + // practice, not just in theory. stopThread() (now that it can abort an + // in-flight JVMTI walk promptly - see its own comment) makes this a cheap, + // clean stop/reset/restart rather than an indefinite wait. + stopThread(); + + // Clear every live tag this search still holds before resetting - the + // same ordering restartSearch() itself requires (its own assert), so a + // stale tag from whatever search a previous test left running cannot + // collide with the fresh search's own tags once _next_tag is rewound + // below. + if (jvmti != nullptr && jni != nullptr) { + releaseSearchTags(jvmti, jni); + } + _tags_released = true; + + _safepoint_pain_budget.spend(_search_pain_ms); + _search_pain_ms = 0; + // Reset the pain budget entirely so a fresh test starts from zero debt, + // independent of how much wall-clock time has elapsed since the last + // test's spend(). Without this, a fast CI runner (musl, small heap, + // no GC pauses) may not have drained enough debt between tests. + _safepoint_pain_budget = PainBudget(_pain_budget_refill_rate); + // Mirror the reset for the non-safepoint budget - same test-isolation + // rationale as _safepoint_pain_budget above. + _cpu_pain_budget = PainBudget(_pain_budget_refill_rate); + // Same test-isolation rationale: a latched urgency episode left behind by + // an earlier test would otherwise deny this one its own + // urgency-authorized search (see _urgent_search_spent). + _urgent_latched = false; + _urgent_release_ticks = 0; + _urgent_search_spent = false; + + if (_frontier != nullptr) { + // Rebuilds the table at this test's own _configured_frontier_cap, + // undoing any smaller framecap= an earlier test left it permanently + // sized at (this class's own header comment on @TestMethodOrder) - + // restartSearch()'s production path only calls the cheaper + // resetForRestart() since it never needs to change the cap mid-JVM. + _frontier->resetCapacityForTest(_configured_frontier_cap); + } + _next_tag = 1; + + _search_started = false; + store(_search_state, (u8)SearchState::RUNNING); + store(_abandon_reason, (u8)SearchAbandonReason::NONE); + store(_search_start_ns, (u64)0); + _pending_expand.clear(); + _priority_expand.clear(); + _priority_expand_set.clear(); + _static_anchor_fifo.clear(); + _static_anchor_fifo_set.clear(); + _static_anchor_fifo_klass_counts.clear(); + _static_anchor_fifo_quota_drops = 0; + _static_anchor_fifo_pushed = 0; + _static_anchor_index.clear(); + _static_anchor_own_class_tags.clear(); + _static_anchor_index_tags.clear(); + _anchor_container_cursor = 0; + _anchor_other_cursor = 0; + _static_anchor_fresh_queue.clear(); + // Same stale-frontier-tag hygiene as restartSearch() (see its comment): + // discovered tags are frontier tags, invalid across the test reset just + // as across a restart. + memset(_candidate_discovered_tags, 0, sizeof(_candidate_discovered_tags)); + memset(_candidate_discovered_count, 0, sizeof(_candidate_discovered_count)); + // Test-only extra: production restartSearch() keeps _class_shape_cache + // (class tags are JVM-lifetime-stable there), but test scenarios script + // class-tag values directly and a later test can reuse an earlier one + // for a different mock class - clear the cache between tests. + _class_shape_cache.clear(); + // Same reset rationale as restartSearch()'s own comment. + _leak_signature_totals.clear(); + _leak_signature_prev_totals.clear(); + _leak_parent_fanout.clear(); + _leak_tags_assigned = 0; + _leak_tags_resolved = 0; + _last_pass_gc_finish_epoch = 0; + store(_last_pass_ns, (u64)0); + store(_passes_run, 0); + _passes_since_last_progress = 0; + _passes_since_last_candidate_progress = 0; + _last_candidate_progress_mark = 0; + _canary_backoff_mult = 1; + _canary_pass_ema_ms = 0; + _last_canary_pass_ns = 0; + _canary_stuck_restart_count = 0; + // Same "just-constructed values" contract resetForRestart() already + // documents for these two fields - without it, a prior test's fully-swept + // (or partially-swept) state survives in this process-wide singleton + // (ReferenceChainTracker::instance()) and can wrongly skip + // admitStaticFieldRoots() entirely on this test's first pass if its + // resolved class count happens to match whatever an earlier test last + // left behind (found via a real gtest-suite-order failure, not + // hypothetical). + _last_resolved_class_count = 0; + _last_static_field_class_count = -1; + _static_field_sweep_cursor = 0; + _static_field_sweep_cycle_truncated = false; + _candidate_count = 0; + _candidate_found_bits = 0; + memset(_candidate_discovered_count, 0, sizeof(_candidate_discovered_count)); + memset(_candidate_qualifying_tid_count, 0, + sizeof(_candidate_qualifying_tid_count)); + // _candidate_parent_tags/_candidate_referrer_klasses/_candidate_depths + // will be filled at pruning time. + // across a production restart, a test reset starts from a blank cache so + // one test's resolved chains cannot leak into the next. + _resolved_chains_lock.lock(); + _resolved_chains.clear(); + _resolved_chains_lock.unlock(); + _pending_abandoned_events_lock.lock(); + _pending_abandoned_events.clear(); + _pending_abandoned_events_lock.unlock(); + + // Restart the BFS thread against this freshly reset state - startThread() + // itself clears _abort_pass_requested, so the new thread's very first + // pass is not instantly aborted by the flag stopThread() just set above. + startThread(); +} + +long ReferenceChainTracker::pendingExpandPositionForTest(jlong tag) const { + if (tag == 0) { + return -2; + } + // _priority_expand drains first (expandFrontier()'s own comment), so its + // entries are reported as coming before _pending_expand's. + long pos = 0; + for (jlong queued : _priority_expand) { + if (queued == tag) { + return pos; + } + pos++; + } + for (jlong queued : _pending_expand) { + if (queued == tag) { + return pos; + } + pos++; + } + return -1; +} + +size_t ReferenceChainTracker::pendingExpandSizeForTest() const { + return _pending_expand.size() + _priority_expand.size(); +} + +jlong ReferenceChainTracker::tagObject(jvmtiEnv *jvmti, jobject obj) { + assert(!t_inGCCallback && + "SetTag is a JVMTI Heap-category call and must not be made from " + "GarbageCollectionStart/Finish"); + jlong tag = nextTag(); + jvmtiError err = jvmti->SetTag(obj, tag); + if (err != JVMTI_ERROR_NONE) { + return 0; + } + return tag; +} + +jlong ReferenceChainTracker::getTag(jvmtiEnv *jvmti, jobject obj) { + assert(!t_inGCCallback && + "GetTag is a JVMTI Heap-category call and must not be made from " + "GarbageCollectionStart/Finish"); + jlong tag = 0; + jvmtiError err = jvmti->GetTag(obj, &tag); + if (err != JVMTI_ERROR_NONE) { + return 0; + } + return tag; +} + +void ReferenceChainTracker::clearTag(jvmtiEnv *jvmti, jobject obj) { + assert(!t_inGCCallback && + "SetTag is a JVMTI Heap-category call and must not be made from " + "GarbageCollectionStart/Finish"); + jvmti->SetTag(obj, 0); +} + +jlong ReferenceChainTracker::tagAsRootForTest(jvmtiEnv *jvmti, JNIEnv *jni, + jobject obj) { + if (_frontier == nullptr || jvmti == nullptr || jni == nullptr || + obj == nullptr) { + return 0; + } + // Resolves the klass_id via the same GetClassSignature + + // normalizeClassSignature + Profiler::lookupClass sequence every + // consumer in this subsystem uses (ObjectSampler::recordAllocation(), + // LivenessTracker::resolveKlassId(), resolveClassMap() above) - the + // id space is load-bearing here: pollWatchedTargets() matches frontier + // entries against leak candidates by klass_id, and the candidate ids + // come from that signature-notation space (Class.getName()'s dot form + // is a DIFFERENT StringDictionary key - see + // find-klass-id-notation-mismatch). Test-only, off-hot-path. + u32 klass_id = 0; + jclass klass = jni->GetObjectClass(obj); + char *class_name = nullptr; + if (jvmti->GetClassSignature(klass, &class_name, nullptr) == + JVMTI_ERROR_NONE && + class_name != nullptr) { + const char *name_slice = nullptr; + size_t name_len = 0; + if (ObjectSampler::normalizeClassSignature(class_name, &name_slice, + &name_len)) { + int id = Profiler::instance()->lookupClass(name_slice, name_len); + if (id != -1) { + klass_id = (u32)id; + } + } + jvmti->Deallocate((unsigned char *)class_name); + } + jni->DeleteLocalRef(klass); + + // Tags `obj` and inserts it as a frontier root (parent_tag=0, depth=0), + // exactly the convention runPass()'s heap-root callback path already uses + // (referenceChains.cpp's heapReferenceCallback(), referrer_tag_ptr == + // nullptr branch) - this lets a test drive the real BFS/chain- + // reconstruction logic (runPass()/pollWatchedTargets()/buildChainEvent()) + // against a known, caller-chosen live object, decoupled from whether the + // real root-seeded walk or LivenessTracker's probabilistic sampler happens + // to reach/select it on its own. + jlong tag = tagObject(jvmti, obj); + if (tag == 0) { + return 0; + } + if (!_frontier->insert(tag, 0, klass_id, 0)) { + clearTag(jvmti, obj); + return 0; + } + return tag; +} + +// --------------------------------------------------------------------------- +// Heap-walk engine +// --------------------------------------------------------------------------- + +void ReferenceChainTracker::resolveLoadedClasses(jvmtiEnv *jvmti, + JNIEnv *jni) { + // Profiler::start() resets the class-name StringDictionary + // (_class_map.clearAll(), profiler.cpp) whenever `reset || _start_time == + // 0` - which restarts its id namespace at 1, but does NOT touch any + // class's JVMTI-level class-object tag (JVM-level state, unrelated to our + // dictionary). Detect that reset via the dictionary's own generation + // counter and drop every id this table cached from the now-gone + // generation before the scan below - see _last_class_map_generation's own + // comment (referenceChains.h) for why leaving them in place would keep + // resolving heap references to the wrong (or nonexistent) class name. + u64 current_generation = Profiler::instance()->classMap()->generation(); + bool class_map_reset = current_generation != _last_class_map_generation; + if (class_map_reset) { + TEST_LOG_SUMMARY("ReferenceChainTracker::resolveLoadedClasses class_map generation " + "changed: old=%llu new=%llu - clearing _class_tags and " + "candidate klass_ids may be stale", + (unsigned long long)_last_class_map_generation, + (unsigned long long)current_generation); + _class_tags.clear(); + // Force the scan below to run even if GetLoadedClasses()'s count happens + // to match the last-seen count - -1 can never equal `class_count` + // (always >= 0), unlike 0 which is a legitimate "no classes loaded yet" + // starting value. + _last_resolved_class_count = -1; + _last_class_map_generation = current_generation; + } + + jclass *classes = nullptr; + jint class_count = 0; + if (jvmti->GetLoadedClasses(&class_count, &classes) != JVMTI_ERROR_NONE || + classes == nullptr) { + return; + } + + // Skip the per-class GetTag()/GetClassSignature() scan entirely once the + // loaded-class count has not CHANGED since the last time this ran it: + // every already-tagged class stays tagged forever (tags are never + // cleared once assigned - see _class_tags' own comment), so a resumed + // pass with no newly-loaded classes has nothing left to resolve. Without + // this, every single pass pays a full GetTag() call per loaded class + // (potentially thousands) even though almost all of them are already + // resolved, and that cost is invisible to the pause-time-SLO pacing + // controller (runPass()'s pass_wall_ticks measurement deliberately scopes + // out this call - see that field's own comment). + // + // Deliberately `!=`, not `>`: GetLoadedClasses()'s count is NOT monotonic + // - class unloading (a GC'd custom classloader, JSP/bytecode-macro + // recompilation, etc.) can shrink it. A `>` check would then stay + // permanently skipped once new classes are loaded back up to, but not + // past, a prior historical peak - e.g. 1000 classes loaded then unloaded + // down to 400, then 50 different new classes loaded (total 450, still + // below the 1000 peak) - silently leaving those 50 new classes' tag == 0 + // forever, so any object of theirs discovered by the BFS walk never + // resolves a referrer_klass. `!=` catches both directions; the only + // residual gap is the count-preserving unload-then-reload-same-count case, + // far narrower than the permanent gap `>` left open. + if (class_count != _last_resolved_class_count) { + for (jint i = 0; i < class_count; i++) { + jclass klass = classes[i]; + jlong tag = 0; + // Resolve if not yet tagged (ordinary case: a newly-loaded class), or + // unconditionally on a class-map reset (class_map_reset above) - a + // class already tagged from a prior generation still carries that same + // JVMTI tag (untouched by clearAll()), but the dictionary id it used to + // map to is gone, so its name must be re-resolved into the new + // generation too. + if (jvmti->GetTag(klass, &tag) == JVMTI_ERROR_NONE && + (tag == 0 || class_map_reset)) { + // Resolve its name now, via the same GetClassSignature + + // normalizeClassSignature + Profiler::lookupClass sequence + // ObjectSampler::recordAllocation() already uses + // (objectSampler.cpp:76-90), reused rather than re-derived. + char *class_name = nullptr; + if (jvmti->GetClassSignature(klass, &class_name, nullptr) == + JVMTI_ERROR_NONE && + class_name != nullptr) { + const char *name_slice = nullptr; + size_t name_len = 0; + if (ObjectSampler::normalizeClassSignature(class_name, &name_slice, + &name_len)) { + int id = Profiler::instance()->lookupClass(name_slice, name_len); + if (id != -1) { + TEST_LOG("ReferenceChainTracker::resolveClassMap id=%d name=%.*s", + id, (int)name_len, name_slice); + // Reuse the existing tag if this class was already tagged by a + // prior generation - only the resolved id needs refreshing, + // not the tag identity heapReferenceCallback() keys off of. + jlong class_tag = tag != 0 ? tag : nextClassTag(); + if (tag != 0 || + jvmti->SetTag(klass, class_tag) == JVMTI_ERROR_NONE) { + _class_tags.insert(class_tag, (u32)id); + } + } + } + jvmti->Deallocate((unsigned char *)class_name); + } + } + // GetLoadedClasses() hands back class_count fresh JNI local refs - + // delete each immediately rather than holding all of them alive at + // once, since class_count can run into the thousands. + if (jni != nullptr) { + jni->DeleteLocalRef(klass); + } + } + _last_resolved_class_count = class_count; + } else if (jni != nullptr) { + // Still owe DeleteLocalRef for every fresh local ref GetLoadedClasses() + // just handed back, even though the scan above was skipped. + for (jint i = 0; i < class_count; i++) { + jni->DeleteLocalRef(classes[i]); + } + } + jvmti->Deallocate((unsigned char *)classes); +} + +namespace { +// Per-runPass() state threaded through heapReferenceCallback() via +// FollowReferences' user_data parameter. Private to this .cpp - the type +// never needs to be visible in referenceChains.h since only runPass() +// constructs one and only heapReferenceCallback() reads it. +struct PassContext { + ReferenceChainTracker *tracker; + FrontierTable *frontier; + int hop_cap; + int budget; + int edges_admitted; + bool truncated; + + // Set only when `truncated` became true because frontier->insert() itself + // reported capacity exhaustion, as opposed to edges_admitted reaching + // budget. runPass() uses this to distinguish "this pass ran out + // of budget, more work remains for a later pass" (search stays RUNNING) + // from "the frontier table itself is full" (design doc's Termination + // section: grounds to ABANDON the whole search, not just this pass). + bool frontier_cap_hit; + + // ARRAY-HOLDER BATCHING: when non-null, expandFrontier() is driving a + // one-hop expansion of a batch of boundary objects passed to a single + // FollowReferences(initial_object=holder_array) call. heapReferenceCallback() + // then descends ONLY into objects whose tag is in this set (the boundary + // objects we deliberately put in the array), and returns "do not descend" + // (0) for everything else - so a freshly-admitted child is tagged but its + // own subtree is left for a later pass, and an already-expanded object from + // a prior pass is never re-traversed. Null on the whole-heap first pass + // (runPass()'s !_search_started branch) and IterateOverReachableObjects + // root enumeration, which keep the unconditional-descend behavior. + std::unordered_set *batch_tags = nullptr; + + // Rolling resume cursor for expandFrontier(): tracks the tag of the last + // batch entry that FollowReferences visited (the callback at the + // batch_tags descent-gate updates this). After FollowReferences returns + // truncated, expandFrontier() uses this to pop fully-processed entries + // from the source queue (mark EXPANDED) and leave only the + // partially-processed and unvisited entries for the next pass — same + // resumable-cursor pattern as admitStaticFieldRoots()'s sweep cursor. + // Without this, a truncated batch is retried in its entirety next pass: + // GetObjectsWithTags + FollowReferences re-walks already-expanded + // entries (their children are ALREADY_ADMITTED, so idempotent but + // wasteful — re-paying the full O(tag_map × batch) GOTW cost and the + // FollowReferences STW for entries that need no work). 0 = no batch + // entry visited yet this FollowReferences call. + jlong _last_visited_batch_tag = 0; + + // Set only by admitStaticFieldRoots(): the seed holder array for that + // sweep holds loaded-class objects (negative-tagged by + // resolveLoadedClasses(), see the *tag_ptr < 0 branch below), and the + // whole point of the sweep is to walk past that holder->class edge into + // each class's own outgoing references - chiefly STATIC_FIELD - which the + // *tag_ptr < 0 check would otherwise stop cold before FollowReferences + // ever gets to report them. Left false everywhere else (expandFrontier()'s + // batching, root enumeration, the whole-heap first pass), where a + // negative-tagged referee must never be descended into. + bool static_field_seed = false; + + // PER-CLASS NON-STATIC QUOTA (admitStaticFieldRoots() only). JVMTI + // reports a class's entire metadata graph through the same + // static_field_seed opening - CONSTANT_POOL (resolved String/Class/ + // MethodHandle/MethodType/CallSite constants), INTERFACE, SUPERCLASS, + // CLASS_LOADER, ... - not just its STATIC_FIELD edges. CONSTANT_POOL + // alone is 5-15x STATIC_FIELD volume per class, systemically. Admitting + // all of them would burn the per-chunk callback/deadline budget on + // non-static-field edges and starve static-field discovery; dropping + // them entirely would exclude a real (if rarer) leak category. Instead, + // STATIC_FIELD edges are always admitted and non-STATIC_FIELD edges from + // a class are admitted up to _class_other_cap per class, then dropped + // for the rest of that class this lap. The cap resets on class + // boundary (detected by referrer tag change), so one fat class cannot + // exhaust the quota for any other. admitStaticFieldRoots() sets + // _class_other_cap from STATIC_FIELD_SWEEP_NON_STATIC_CAP_PER_CLASS; + // everywhere else these are zero/unused. + jlong _seed_class_tag = 0; // negative tag of the class currently + // being descended (0 before the first + // class edge is seen) + int _class_other_admitted = 0; // non-STATIC_FIELD edges admitted for + // the current class this lap + int _class_other_cap = 0; // per-class cap; 0 disables the quota + // (admit all) when not in seed sweep + // Number of distinct classes entered so far in this chunk's descent + // (incremented on each class-boundary tag change). Used by + // admitStaticFieldRoots() to compute the resumable cursor on truncation: + // resume at chunk_start + count - 1 (redo the partial class) rather + // than skipping to chunk_end and losing the rest of the chunk. + int _classes_in_chunk_visited = 0; + + // Amortizes tracker->_pass_deadline_ns's OS::nanotime() check (heapReference + // Callback()/heapRootCallback() run once per visited edge/root - checking + // wall-clock on literally every call would add real overhead on a large + // heap) - checked only every 4096th call, local to this ctx so each of + // runPassManualWalk()'s several sub-calls (root enum, static-field sweep, + // expandFrontier(), rotation) starts its own count. + int deadline_check_counter = 0; + + // True while expandFrontier() is walking a batch drawn from + // _priority_expand (a rotation-selected, already-EXPANDED parent) rather + // than the ordinary _pending_expand backlog - see _priority_expand's own + // comment. Newly admitted children inherit the fast lane so the whole + // re-discovered subtree, not just the immediate child, skips the backlog. + bool admit_priority = false; + + // TEMP DIAGNOSTIC (see doc/temp/ investigation notes): per-jvmtiHeapReference + // Kind callback tally for admitStaticFieldRoots()'s current chunk, to find + // which reference kind actually drives per-chunk callback volume (e.g. + // static fields vs. constant-pool entries vs. interfaces). Null everywhere + // else - only admitStaticFieldRoots() sets this to a non-null, zeroed, + // stack-local array sized for the full jvmtiHeapReferenceKind range. + int *kind_counts = nullptr; + + // DESCEND-WALK controls (descendFromAnchor()'s calls only; null/0 + // everywhere else, so every gate below is a no-op for the ordinary + // walk phases): + // + // _no_descend_class_tags: exact class tags to neither admit nor descend + // into for the duration of this walk. The fat-metadata classes whose + // graphs would otherwise turn a bounded anchor walk into sweep-scale + // cost (java.lang.ClassLoader from Thread.contextClassLoader reaching + // every loaded class, java.lang.ThreadGroup reaching every thread, + // java.security.ProtectionDomain). Exact tag match only - a SUBCLASS of + // one of these is a distinct class tag and is descended into normally + // (documented limitation: subclass instances are ordinary app objects, + // and an is-a hierarchy walk is not affordable per callback). + // + // _descent_anchor_tag/_anchor_descend_class_tag: when non-zero, the + // ANCHOR object's own outgoing edges are gated to descend only into + // referees of that exact class (walkCandidateThreadLocals() passes + // java.lang.ThreadLocal$ThreadLocalMap - the value type of BOTH of + // Thread's threadLocals and inheritableThreadLocals fields - so the walk + // never enumerates the Thread's other instance fields; the class gate + // is used instead of jvmtiHeapReferenceInfoField.index because the + // index-vs-GetClassFields-order correspondence is a spec subtlety this + // design has no need to depend on). Below the anchor, descent is gated + // by _no_descend_class_tags as above. + static constexpr int NO_DESCEND_CLASS_CAP = 8; + jlong _no_descend_class_tags[NO_DESCEND_CLASS_CAP] = {0}; + int _no_descend_class_tag_count = 0; + jlong _descent_anchor_tag = 0; + jlong _anchor_descend_class_tag = 0; + + // TEMP DIAGNOSTIC (pod round 12 - wrapper walked but never intercepted): + // when _diag_trace is set (only descendFromAnchor() sets it, and only for + // an anchor whose class is Collections$UnmodifiableRandomAccessList - + // the LEAK_BUFFER wrapper shape), heapReferenceCallback()'s admission and + // already-tagged-encounter sites record (klass_id, how-seen) pairs for the + // first DIAG_MAX_ENTRIES entries, so one walk's actual traversal is + // observable: does the wrapper -> list -> elementData -> [B chunk chain + // get enumerated AT ALL, and did the enumerated chunks carry leak tags at + // that moment. _diag_leak_flags: 0 = fresh admission, 1 = admission via + // leak-tag conversion (an interception - never observed for the wrapper + // so far), 2 = already-frontier-tagged re-encounter. TEMP: overfit to the + // round-12 wrapper question - remove once the pod answers it. + static constexpr int DIAG_MAX_ENTRIES = 48; + bool _diag_trace = false; + int _diag_count = 0; + u32 _diag_klass_ids[DIAG_MAX_ENTRIES]; + u8 _diag_leak_flags[DIAG_MAX_ENTRIES]; +}; +} // namespace + +jint JNICALL ReferenceChainTracker::heapReferenceCallback( + jvmtiHeapReferenceKind reference_kind, + const jvmtiHeapReferenceInfo *reference_info, jlong class_tag, + jlong referrer_class_tag, jlong size, jlong *tag_ptr, + jlong *referrer_tag_ptr, jint length, void *user_data) { + PassContext *ctx = (PassContext *)user_data; + + // TEMP DIAGNOSTIC (see doc/temp/ investigation notes): tally every callback + // by kind before any early-return below, so an aborted/truncated chunk + // still reports what it actually saw. kind_counts is only non-null for + // admitStaticFieldRoots()'s call - zero overhead elsewhere. + if (ctx->kind_counts != nullptr && (int)reference_kind >= 0 && + (int)reference_kind < 32) { + ctx->kind_counts[(int)reference_kind]++; + } + + if (ctx->tracker->_abort_pass_requested.load(std::memory_order_relaxed)) { + // stopThread() has set this right before pthread_kill()/pthread_join() - + // see that method's own comment. pthread_kill(WAKEUP_SIGNAL) only + // interrupts threadLoop()'s OS::sleep(); it cannot interrupt an + // in-flight JVMTI FollowReferences call, so without this check + // pthread_join() would block until this pass's walk finishes on its own + // - potentially the whole reachable graph, well past any caller's + // shutdown timeout. Treat it exactly like an ordinary budget exhaustion + // (ctx->truncated = true): this pass ends early and the search stays + // non-terminal - fine, since the tracker is shutting down and simply + // never resumes it. + ctx->truncated = true; + return JVMTI_VISIT_ABORT; + } + + if (ctx->tracker->_pass_deadline_ns != 0 && + (++ctx->deadline_check_counter & 0xFFF) == 0 && + OS::nanotime() >= ctx->tracker->_pass_deadline_ns) { + // This pass has run past its wall-clock share (see _pass_deadline_ns's + // own comment) - treat it exactly like ordinary budget exhaustion so it + // ends early without abandoning the search; a later pass re-enumerates + // whatever roots/edges this one didn't get to. + ctx->truncated = true; + return JVMTI_VISIT_ABORT; + } + + // Retention-edge identity for every admission site below: the JVMTI + // heap callback's field ordinal (the JVMTI-SPECIFICATION numbering over + // the referrer's flattened field space - see FrontierEntry:: + // referrer_field_index's own comment) for FIELD/STATIC_FIELD edges, -1 + // otherwise; and the referrer's class tag when the referrer is a + // CLASS OBJECT (root-attached static edges - interior hops get their + // referrer class from the parent entry at chain-reconstruction time, + // so only the parent_tag==0 case needs it recorded here). Captured + // here, before the early-return branches, because a live heap callback + // cannot be replayed after the fact - the same reason FrontierEntry:: + // class_tag is stored at admission. + jint edge_field_index = -1; + if (reference_info != nullptr && + (reference_kind == JVMTI_HEAP_REFERENCE_FIELD || + reference_kind == JVMTI_HEAP_REFERENCE_STATIC_FIELD)) { + edge_field_index = reference_info->field.index; + } + jlong edge_referrer_class_tag = 0; + if (referrer_tag_ptr != nullptr && *referrer_tag_ptr < 0) { + edge_referrer_class_tag = *referrer_tag_ptr; + } + + // Canary pruning: if this object is the pre-tagged representative of a + // leaked candidate (marker tag MARKER_TAG_BASE - i, negative), record its + // chain link but do NOT enqueue its children (treat as a leaf). Do not + // return JVMTI_VISIT_ABORT -- that aborts the entire FollowReferences walk + // (JVMTI spec). Just skip admitObject() for this object. + // + // MUST run before the *tag_ptr < 0 class-tag check below, since marker + // tags are negative and would be caught by that check first (returning 0 + // without recording the chain link). + // + // Matches by object identity (the marker tag SetTag() put on this exact + // representative object in pollWatchedTargets()), not by class: matching + // by class alone would record a chain for whichever instance of that class + // the walk happens to visit first, which for a common class (e.g. byte[]) + // is almost certainly an unrelated, possibly short-lived object - not the + // specific instance LivenessTracker flagged as growing. + if (ctx->tracker->_candidate_count > 0 && + *tag_ptr <= ReferenceChainTracker::MARKER_TAG_BASE) { + int candidate_idx = (int)(ReferenceChainTracker::MARKER_TAG_BASE - *tag_ptr); + if (candidate_idx >= 0 && candidate_idx < ctx->tracker->_candidate_count) { + jlong rtag = (referrer_tag_ptr != nullptr) ? *referrer_tag_ptr : 0; + u32 candidate_klass = ctx->tracker->classTags()->resolve(class_tag); + if (rtag > 0) { + FrontierEntry parent{}; + if (ctx->frontier->lookup(rtag, &parent)) { + // Use the marker tag itself as the frontier table key - it is + // already a unique per-candidate value, so no nextTag() is needed. + jlong frontier_tag = *tag_ptr; + ctx->frontier->insert(frontier_tag, rtag, + parent.referrer_klass, + parent.depth + 1, + FrontierEntryState::FRONTIER, + parent.root_kind, + /*class_tag=*/0, edge_field_index, + (u8)reference_kind); + ctx->tracker->_candidate_parent_tags[candidate_idx] = rtag; + ctx->tracker->_candidate_frontier_tags[candidate_idx] = frontier_tag; + ctx->tracker->_candidate_referrer_klasses[candidate_idx] = candidate_klass; + ctx->tracker->_candidate_depths[candidate_idx] = parent.depth + 1; + ctx->tracker->_candidate_found_bits |= (1ULL << candidate_idx); + TEST_LOG("ReferenceChainTracker::heapReferenceCallback canary " + "pruned candidate %d (klass_id=%u frontier_tag=%lld)", + candidate_idx, candidate_klass, (long long)frontier_tag); + } + } else { + // Root-referenced candidate. + jlong frontier_tag = *tag_ptr; + ctx->frontier->insert(frontier_tag, 0, + candidate_klass, 1, + FrontierEntryState::FRONTIER, + (u8)reference_kind, + /*class_tag=*/0, edge_field_index, + /*edge_kind=*/0, edge_referrer_class_tag); + ctx->tracker->_candidate_parent_tags[candidate_idx] = 0; + ctx->tracker->_candidate_frontier_tags[candidate_idx] = frontier_tag; + ctx->tracker->_candidate_referrer_klasses[candidate_idx] = candidate_klass; + ctx->tracker->_candidate_depths[candidate_idx] = 1; + ctx->tracker->_candidate_found_bits |= (1ULL << candidate_idx); + TEST_LOG("ReferenceChainTracker::heapReferenceCallback canary " + "pruned root-referenced candidate %d (klass_id=%u)", + candidate_idx, candidate_klass); + } + // Do NOT enqueue children for this object. + return 0; + } + } + + if (*tag_ptr < 0) { + if (ctx->static_field_seed && + reference_kind == JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT && + referrer_tag_ptr != nullptr && *referrer_tag_ptr == 0) { + // admitStaticFieldRoots()'s own holder[i] -> class edge: referrer_tag_ptr + // points at the transient, never-tagged seed array itself (tag 0), not + // at a frontier-admitted parent. Continue the walk into this class's + // own outgoing references - static fields chief among them - instead + // of stopping here; that is the entire purpose of the sweep. The class + // object itself is still never admitted into the frontier (tag_ptr is + // left untouched, so it stays negative). + return JVMTI_VISIT_OBJECTS; + } + // Referee is a class object already tagged negative by + // resolveLoadedClasses() (that pre-pass runs before FollowReferences in + // runPass(), so every loaded class already carries a negative tag by + // this point). Never admit a class object into the frontier as if it + // were an ordinary retained instance, and - outside the + // admitStaticFieldRoots() seed edge handled above - never expand from a + // class's own metadata graph (static fields, superclass, interfaces, + // constant pool, class loader, ...). Out of scope per the design doc's + // non-goals (no field-level/exhaustive paths) and keeps the walk bounded + // to the instance-reachability graph that actually explains "why is + // this object alive". + return 0; + } + if (reference_kind == JVMTI_HEAP_REFERENCE_CLASS || + reference_kind == JVMTI_HEAP_REFERENCE_SYSTEM_CLASS) { + // Definitionally a class by reference_kind (CLASS: "reference from an + // object to its class"; SYSTEM_CLASS: a root reference to a class) even + // if resolveLoadedClasses() failed to resolve/tag this particular one + // (e.g. a transient StringDictionary contention failure) and its tag is + // therefore not yet negative. Same non-goal as above: never expand from + // or admit a class object. + return 0; + } + + if (ctx->truncated) { + // Defensive: FollowReferences should already have stopped delivering + // callbacks after a JVMTI_VISIT_ABORT return below; this just avoids + // doing further work if one more callback arrives anyway. + return JVMTI_VISIT_ABORT; + } + + jlong parent_tag = 0; + u32 depth = 0; + if (referrer_tag_ptr != nullptr) { + jlong rtag = *referrer_tag_ptr; + if (rtag > 0) { + FrontierEntry parent{}; + if (ctx->frontier->lookup(rtag, &parent)) { + parent_tag = rtag; + depth = parent.depth + 1; + } + // lookup() failing for a positive rtag should not happen - a referrer + // must already be one of our tagged frontier objects for its own + // outgoing edges to be traversed at all (FollowReferences only + // explores past an object this callback returned JVMTI_VISIT_OBJECTS + // for) - but fall back to root-like (parent_tag=0/depth=0) rather + // than corrupt the chain if it ever does. + } + // rtag < 0: referrer is a pre-tagged class object (e.g. a static field + // holding this reference) - treated as root-like rather than attributed + // to a parent hop, since class objects are never admitted as frontier + // entries and so have no depth/parent_tag of their own (see the + // *tag_ptr < 0 check above). rtag == 0: referrer not yet tagged, should + // not happen for the same reason noted above. + } + // referrer_tag_ptr == nullptr: a heap-root reference (JNI global, thread + // stack local/JNI local, monitor, thread, system class, ...) - parent_tag + // and depth stay 0. + + if (depth >= (u32)ctx->hop_cap) { + // Hop cap: do not admit this object into the frontier, and do not + // expand further from it - enforced here rather than + // discovering-then-discarding, per the plan. + return 0; + } + + if (ctx->static_field_seed && referrer_tag_ptr != nullptr && + *referrer_tag_ptr < 0) { + // Referrer is the class object opened by the static_field_seed branch + // above. JVMTI reports that class's entire metadata reference graph + // through this same opening, not just its static fields - CONSTANT_POOL + // (resolved String/Class/MethodHandle/MethodType/CallSite constants), + // INTERFACE, SUPERCLASS, CLASS_LOADER, ... Those are real reachability + // edges, just lower-priority for this static-field-root sweep than + // STATIC_FIELD. Admit STATIC_FIELD unconditionally; admit non-STATIC_FIELD + // up to the per-class cap (PassContext::_class_other_cap) so one fat + // class cannot starve the rest, then drop further non-static edges for + // this class this lap. Track the current class tag for + // admitStaticFieldRoots()'s resumable cursor (see that method's own + // comment) and reset the cap counter on class boundary. + if (*referrer_tag_ptr != ctx->_seed_class_tag) { + ctx->_seed_class_tag = *referrer_tag_ptr; + ctx->_class_other_admitted = 0; + ctx->_classes_in_chunk_visited++; + } + if (reference_kind != JVMTI_HEAP_REFERENCE_STATIC_FIELD) { + if (ctx->_class_other_cap > 0 && + ctx->_class_other_admitted >= ctx->_class_other_cap) { + // Quota exhausted for this class - drop the edge. Count every + // drop, and count the first drop for this class separately so + // the two counters together distinguish "a few fat outlier + // classes dropping many edges" from "systematic drops across + // almost all classes" (cap too low). + Counters::increment(REFERENCE_CHAIN_STATIC_SWEEP_NON_STATIC_DROPPED); + if (ctx->_class_other_admitted == ctx->_class_other_cap) { + Counters::increment(REFERENCE_CHAIN_STATIC_SWEEP_CLASSES_CAPPED); + } + return 0; + } + ctx->_class_other_admitted++; + } + } + + // Leak tag: this object was directly tagged by LivenessTracker's + // tagLeakInstances() because it's a tracked leaking object. Convert + // the leak tag to a frontier tag so the BFS can build its chain, and + // store the leak tag in the frontier entry for correlation with + // HeapLiveObject events. + if (isLeakTag(*tag_ptr)) { + jlong leak_tag = *tag_ptr; + // Allocate a frontier tag for this object + jlong frontier_tag = ctx->tracker->nextTag(); + u32 referrer_klass = ctx->tracker->classTags()->resolve(class_tag); + u8 root_kind = parent_tag == 0 ? (u8)reference_kind : 0; + if (ctx->frontier->insert(frontier_tag, parent_tag, referrer_klass, + depth, FrontierEntryState::FRONTIER, + root_kind, class_tag, edge_field_index, + (u8)reference_kind, + parent_tag == 0 ? edge_referrer_class_tag : 0)) { + // Store the leak tag in the frontier entry + ctx->frontier->setLeakTag(frontier_tag, leak_tag); + *tag_ptr = frontier_tag; + ctx->edges_admitted++; + TEST_LOG("ReferenceChainTracker::heapReferenceCallback leak-tag " + "intercepted: leak_tag=%lld -> frontier_tag=%lld depth=%u " + "parent_tag=%lld", + (long long)leak_tag, (long long)frontier_tag, depth, + (long long)parent_tag); + if (ctx->_diag_trace && + ctx->_diag_count < PassContext::DIAG_MAX_ENTRIES) { + ctx->_diag_klass_ids[ctx->_diag_count] = + ctx->tracker->classTags()->resolve(class_tag); + ctx->_diag_leak_flags[ctx->_diag_count] = 1; + ctx->_diag_count++; + } + ctx->tracker->trackLeakAccumulation(ctx->frontier, class_tag, + parent_tag, frontier_tag); + // Index maintenance: a leak-tagged object admitted root-attached by + // a durable root edge (e.g. a static field directly holding a tagged + // chunk) is the highest-priority anchor tier (leak_tag != 0). + if (parent_tag == 0 && + (root_kind == (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD || + root_kind == (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL)) { + ctx->tracker->addToStaticAnchorIndex(frontier_tag, class_tag, + root_kind); + } + // Auto-mark: record this as a discovered instance, with eviction + // rights over uncorrelated noise slots (see recordDiscoveredInstance). + if (ctx->tracker->_candidate_count > 0) { + u32 klass_id = ctx->tracker->classTags()->resolve(class_tag); + ctx->tracker->recordDiscoveredInstance(klass_id, frontier_tag, true); + } + } else { + // Frontier cap hit + ctx->truncated = true; + ctx->frontier_cap_hit = true; + return JVMTI_VISIT_ABORT; + } + return JVMTI_VISIT_OBJECTS; + } + + // DESCEND-WALK GATES (no-ops on every ordinary walk - the PassContext + // fields below are zero-initialized and only descendFromAnchor() sets + // them). Deliberately placed AFTER the leak-tag interception branch + // above: a leak-tagged instance of a no-descend class (e.g. a leaking + // ClassLoader - a classic leak category) must still be intercepted and + // correlated; only its own subtree is not descended into. Returning 0 + // skips admission AND descent for this edge, which also skips the + // improveChain/root-upgrade branches below - correct for this walk: a + // descend walk's purpose is reaching tagged instances below the anchor, + // not re-attributing metadata objects the ordinary BFS already owns. + if (ctx->_no_descend_class_tag_count > 0) { + for (int i = 0; i < ctx->_no_descend_class_tag_count; i++) { + if (ctx->_no_descend_class_tags[i] == class_tag) { + return 0; + } + } + } + if (ctx->_descent_anchor_tag != 0 && ctx->_anchor_descend_class_tag != 0 && + referrer_tag_ptr != nullptr && + *referrer_tag_ptr == ctx->_descent_anchor_tag && + class_tag != ctx->_anchor_descend_class_tag) { + // This descend walk's ANCHOR object's own edge, and the referee is not + // the gate class (see PassContext::_anchor_descend_class_tag's own + // comment - e.g. walkCandidateThreadLocals() walks ONLY the Thread's + // ThreadLocalMap edges, never enumerating the Thread's other fields). + return 0; + } + + if (*tag_ptr == 0) { + // First time this object is visited in this pass. + u32 referrer_klass = ctx->tracker->classTags()->resolve(class_tag); + // reference_kind describes this admitting edge; only meaningful for a + // root-attached entry (parent_tag == 0) - see FrontierEntry::root_kind's + // own comment for why a non-root entry's edge kind is not recorded. + u8 root_kind = parent_tag == 0 ? (u8)reference_kind : 0; + ReferenceChainTracker::AdmitResult result = ctx->tracker->admitObject( + ctx->frontier, ctx->hop_cap, ctx->budget, &ctx->edges_admitted, + tag_ptr, parent_tag, referrer_klass, depth, root_kind, class_tag, + ctx->admit_priority, edge_field_index, (u8)reference_kind, + parent_tag == 0 ? edge_referrer_class_tag : 0); + switch (result) { + case ReferenceChainTracker::AdmitResult::BUDGET_EXHAUSTED: + ctx->truncated = true; + return JVMTI_VISIT_ABORT; + case ReferenceChainTracker::AdmitResult::FRONTIER_CAP_HIT: + // Frontier-size cap hit (FrontierTable::insert() returned false + // without partially writing) - stop admitting new entries and report + // the truncation (design doc: "stop admitting new entries ... report + // it"), rather than silently dropping this object and continuing. + // Distinct from ordinary budget exhaustion above - runPass() keeps + // the search RUNNING on this, deferring to the no-progress detector + // to abandon only if the frontier then stops growing (see runPass()'s + // frontier_cap_hit handling). + ctx->truncated = true; + ctx->frontier_cap_hit = true; + return JVMTI_VISIT_ABORT; + default: + // ADMITTED, or HOP_CAP/ALREADY_ADMITTED (neither reachable here: the + // hop-cap check above already returned before this branch, and + // *tag_ptr == 0 rules out ALREADY_ADMITTED) - nothing further to do. + break; + } + if (ctx->_diag_trace && + ctx->_diag_count < PassContext::DIAG_MAX_ENTRIES && + result == ReferenceChainTracker::AdmitResult::ADMITTED) { + ctx->_diag_klass_ids[ctx->_diag_count] = referrer_klass; + ctx->_diag_leak_flags[ctx->_diag_count] = 0; + ctx->_diag_count++; + } + // TEMP DIAGNOSTIC (pod round 12): log every root-attached + // STATIC_FIELD admission so we can see which classes are admitted + // as static-field holders by the sweep. The LEAK_BUFFER wrapper + // (SynchronizedRandomAccessList) must appear here if the sweep + // processes ProfileAnalyzer's class. Remove once the wrapper + // question is answered. + if (result == ReferenceChainTracker::AdmitResult::ADMITTED && + parent_tag == 0 && + root_kind == (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD) { + TEST_LOG("ReferenceChainTracker::heapReferenceCallback " + "root-attached STATIC_FIELD admit klass_id=%u " + "frontier_tag=%lld depth=%u", + referrer_klass, (long long)*tag_ptr, depth); + } + // Index maintenance: track root-attached durable anchors for O(anchors) + // collector iteration instead of O(frontier_size) table scan. class_tag + // is the anchor object's OWN class tag (the callback's class_tag param + // describes the referee, i.e. the object being admitted here) - kept so + // the collector can tier by class shape without JNI. + if (result == ReferenceChainTracker::AdmitResult::ADMITTED && + parent_tag == 0) { + ctx->tracker->addToStaticAnchorIndex(*tag_ptr, class_tag, root_kind); + } + // Auto-mark: if this object's class matches a watched leak class, + // record its frontier tag so pollWatchedTargets() can build a chain + // event for it. A leaking class typically has many live instances, + // and each one's reference chain is independently useful — the + // pre-tagged representative is just one sample, and its representative + // may change (LRU-evicted) between polls. Recording all discovered + // instances ensures we emit chain events for all of them, not just + // whichever single object happened to be the representative when the + // canary slot was first filled. See _candidate_discovered_tags's own + // comment. + if (result == ReferenceChainTracker::AdmitResult::ADMITTED && + ctx->tracker->_candidate_count > 0) { + u32 klass_id = ctx->tracker->classTags()->resolve(class_tag); + if (klass_id == 0) { + // class_tag not in _class_tags - either class map rotated + // (resolveLoadedClasses hasn't re-resolved yet) or this class + // was never tagged. Log once per pass to diagnose class-map + // rotation issues. + TEST_LOG("ReferenceChainTracker::auto-mark class_tag=%lld " + "unresolved (not in _class_tags)", + (long long)class_tag); + } else { + bool matched = false; + for (int s = 0; s < ctx->tracker->_candidate_count; s++) { + if (ctx->tracker->_candidate_klass_ids[s] == klass_id) { + matched = true; + ctx->tracker->recordDiscoveredInstance(klass_id, *tag_ptr, + false); + break; + } + } + if (!matched && klass_id != 0) { + // klass_id resolved but doesn't match any candidate - likely + // class map rotation made candidate klass_ids stale + TEST_LOG("ReferenceChainTracker::auto-mark klass_id=%u " + "resolved but no candidate match (candidates=[%u,%u,%u,%u,%u])", + klass_id, + ctx->tracker->_candidate_count > 0 ? ctx->tracker->_candidate_klass_ids[0] : 0, + ctx->tracker->_candidate_count > 1 ? ctx->tracker->_candidate_klass_ids[1] : 0, + ctx->tracker->_candidate_count > 2 ? ctx->tracker->_candidate_klass_ids[2] : 0, + ctx->tracker->_candidate_count > 3 ? ctx->tracker->_candidate_klass_ids[3] : 0, + ctx->tracker->_candidate_count > 4 ? ctx->tracker->_candidate_klass_ids[4] : 0); + } + } + } + } else if (*tag_ptr > 0) { + if (ctx->_diag_trace && + ctx->_diag_count < PassContext::DIAG_MAX_ENTRIES) { + ctx->_diag_klass_ids[ctx->_diag_count] = + ctx->tracker->classTags()->resolve(class_tag); + ctx->_diag_leak_flags[ctx->_diag_count] = 2; + ctx->_diag_count++; + } + // TEMP DIAGNOSTIC (pod round 12): log every STATIC_FIELD edge + // from the sweep that hits an already-admitted entry, with the + // entry's current shape (parent_tag, root_kind, state). This shows + // whether the LEAK_BUFFER wrapper (SynchronizedRandomAccessList) + // is ever reached by the sweep and what its frontier entry looks + // like. Remove once the wrapper question is answered. + if (ctx->static_field_seed && + reference_kind == JVMTI_HEAP_REFERENCE_STATIC_FIELD) { + FrontierEntry e{}; + bool found = ctx->frontier->lookup(*tag_ptr, &e); + TEST_LOG("ReferenceChainTracker::sweep STATIC_FIELD " + "already-admitted klass_id=%u tag=%lld " + "parent=%lld root_kind=%u state=%u found=%d", + ctx->tracker->classTags()->resolve(class_tag), + (long long)*tag_ptr, (long long)(found ? e.parent_tag : 0), + (unsigned)(found ? e.root_kind : 0), + (unsigned)(found ? e.state : 0), (int)found); + } + // Already-tagged object reached via a new edge. This arm - NOT the + // first-admission block above - is where an already-admitted entry's + // shape can be corrected: improveChain/reparentToDurableRoot for a + // deeper/equal-durable path, maybeUpgradeRootAttachedRootKind for a + // new root-like edge. These branches were originally nested INSIDE + // the *tag_ptr == 0 block (misplaced by 57aec4895, whose own message + // says "improveChain needs to run when *tag_ptr != 0"), where they + // were dead code for their stated purpose: a freshly-admitted entry + // carries exactly this edge's (parent_tag, depth), so both improve- + // Chain's depth> check and the durability upgrade's strict-> check + // are guaranteed no-ops there. On the pod this silently disabled + // every already-admitted re-attribution: the static sweep's edge onto + // a holder born chain-attached could never re-root it + // (find-anchor-holder-eviction). + if (parent_tag != 0) { + // This new path is deeper - replace the shallow root-attached entry + // with the deeper chain-attached entry. This fixes the "depth=1 chain + // with no holder" problem: an object first admitted as a JNI-local + // root (parent_tag == 0) gets its frontier entry improved when the + // static-field → ... → object path reaches it. + u32 referrer_klass = ctx->tracker->classTags()->resolve(class_tag); + // Pre-read the CURRENT shape: if improveChain() below succeeds, this + // root-attached durable entry is about to be replaced with a deeper + // chain-attached one - i.e. it is leaving the population + // collectStaticFieldAnchorsForRotation() can select, at exactly this + // moment. That eviction (find-anchor-holder-eviction) needs a B' + // at-risk push fired HERE, not only from the sweep's static-edge + // site: the sweep gate re-laps only while the class count is in flux + // (see admitStaticFieldRoots' gate in runPassManualWalk), so a + // post-lap demotion in a stable-class JVM would otherwise never see + // another static edge onto this entry. + FrontierEntry pre_improve_entry{}; + bool was_root_attached_durable = + ctx->frontier->lookup(*tag_ptr, &pre_improve_entry) && + pre_improve_entry.parent_tag == 0 && + rootKindDurability(pre_improve_entry.root_kind) >= 2; + if (ctx->frontier->improveChain(*tag_ptr, parent_tag, referrer_klass, + depth, 0, edge_field_index, + (u8)reference_kind)) { + // Chain was improved — invalidate any cached chain for this tag + // so pollWatchedTargets rebuilds it with the deeper path. + ctx->tracker->invalidateResolvedChain(*tag_ptr); + if (was_root_attached_durable) { + // Demotion push (B'): the replaced entry's static/JNI-global + // attribution was its only anchor-tier eligibility, and it is + // gone now. rootKindDurability() >= 2 is exactly the durable set + // the collector selects (STATIC_FIELD, JNI_GLOBAL); SYSTEM_CLASS + // scores 3 too but a class object is never admitted as a frontier + // entry, so it cannot appear here. + ctx->tracker->pushAtRiskStaticAnchor( + *tag_ptr, ctx->tracker->classTags()->resolve(class_tag)); + } + } else if (ctx->frontier->reparentToDurableRoot( + *tag_ptr, parent_tag, referrer_klass, edge_field_index, + (u8)reference_kind)) { + // Equal-depth re-parent from a transient root to a durable one + // (improveChain() cannot express it - see its declaration) - same + // cache invalidation so the rebuilt chain uses the durable root. + ctx->tracker->invalidateResolvedChain(*tag_ptr); + } + } else { + // Already-admitted entry reached via a NEW root-like edge + // (parent_tag == 0): the static-field sweep's class -> field edge + // reports the class as the referrer with a negative tag, which the + // rtag < 0 branch above treats as root-like (class objects are never + // frontier entries), and heap-root references arrive here with + // referrer_tag_ptr == nullptr. Without this, an entry first admitted + // through a stack local keeps its transient classification forever + // even after a later static-field sweep proves the same object is + // the direct value of a static field - exactly the durable-root + // discovery maybeUpgradeRootAttachedRootKind() exists for (same + // tie-break heapRootCallback() applies on its own ALREADY_ADMITTED + // case), so reuse it: upgrade only when this edge's kind is strictly + // more durable, and drop any cached chain so it is rebuilt with the + // upgraded root kind. + if (ctx->tracker->maybeUpgradeRootAttachedRootKind(ctx->frontier, + *tag_ptr, + (u8)reference_kind)) { + ctx->tracker->invalidateResolvedChain(*tag_ptr); + } else if (reference_kind == JVMTI_HEAP_REFERENCE_STATIC_FIELD) { + // The upgrade refused (maybeUpgradeRootAttachedRootKind returns + // false for parent_tag != 0 by design), so this STATIC_FIELD edge + // just proved an at-risk static attachment the anchor tier's + // parent_tag == 0 filter can never see: a holder already admitted + // as a non-root child (find-anchor-holder-eviction). Feed it to + // _static_anchor_fifo (B', see its declaration comment) so the + // static-anchor walk drains it ahead of the root-attached cohort. + // Only the sweep emits STATIC_FIELD edges onto already-tagged + // entries: the edge's referrer is the class object, and the BFS/ + // descend walks never expand classes (the tag < 0 and CLASS-kind + // early returns above), so no other walk can flood the FIFO. + FrontierEntry entry{}; + if (ctx->frontier->lookup(*tag_ptr, &entry) && + entry.parent_tag != 0) { + ctx->tracker->pushAtRiskStaticAnchor( + *tag_ptr, ctx->tracker->classTags()->resolve(class_tag)); + } + } + } + } + + if (ctx->batch_tags != nullptr) { + // ARRAY-HOLDER BATCHING one-hop descent control (see PassContext:: + // batch_tags). Descend only into this pass's boundary objects so the + // single FollowReferences(holder_array) call expands exactly one hop: + // a boundary object yields its direct children (which get tagged above), + // but those children are not themselves descended into, and any + // already-expanded object from a prior pass is skipped rather than + // re-traversed. + jlong my_tag = *tag_ptr; + if (my_tag > 0 && ctx->batch_tags->count(my_tag) != 0) { + // Track this batch entry as visited for the rolling resume cursor + // (see _last_visited_batch_tag's own comment). + ctx->_last_visited_batch_tag = my_tag; + return JVMTI_VISIT_OBJECTS; + } + return 0; + } + + return JVMTI_VISIT_OBJECTS; +} + +ReferenceChainTracker::AdmitResult ReferenceChainTracker::admitObject( + FrontierTable *frontier, int hop_cap, int budget, int *edges_admitted, + jlong *tag_ptr, jlong parent_tag, u32 referrer_klass, u32 depth, + u8 root_kind, jlong class_tag, bool priority, + jint edge_field_index, u8 edge_kind, jlong edge_referrer_class_tag) { + // edge_* default-declared in the header; heapRootCallback() passes the + // defaults (a root reference is not a field edge) unchanged. + if (*tag_ptr != 0) { + return AdmitResult::ALREADY_ADMITTED; + } + if (depth >= (u32)hop_cap) { + return AdmitResult::HOP_CAP; + } + if (*edges_admitted >= budget) { + return AdmitResult::BUDGET_EXHAUSTED; + } + jlong tag = nextTag(); + if (!frontier->insert(tag, parent_tag, referrer_klass, depth, + FrontierEntryState::FRONTIER, root_kind, class_tag, + edge_field_index, edge_kind, edge_referrer_class_tag)) { + return AdmitResult::FRONTIER_CAP_HIT; + } + *tag_ptr = tag; + (*edges_admitted)++; + // Queue for expandFrontier()/markAllFrontierExpanded() - see + // _pending_expand's/_priority_expand's own declaration comments for why + // this replaces a scan over the admitted range, and for why a + // rotation-discovered child (priority=true) skips the ordinary backlog. + if (priority && _priority_expand.size() < PRIORITY_EXPAND_CAP) { + _priority_expand.push_back(tag); + _priority_expand_set.insert(tag); + } else { + // Priority lane full: the rotation backpressure falls back to the + // ordinary backlog rather than silently dropping the re-discovered + // subtree (see PRIORITY_EXPAND_CAP's own comment). + _pending_expand.push_back(tag); + } + trackLeakAccumulation(frontier, class_tag, parent_tag, tag); + return AdmitResult::ADMITTED; +} + +void ReferenceChainTracker::trackLeakAccumulation(FrontierTable *frontier, + jlong class_tag, + jlong parent_tag, + jlong tag) { + // Cheapest checks first: no klass_id is currently watched (the common + // case before hasLeakSignal() has ever fired - see + // _watched_leak_klass_ids' own comment), or this admission has no real + // parent to attribute to (a root-attached entry - nothing to aggregate + // by, since the "container" concept this tracks is specifically about a + // PARENT object's field holding the leaf, not the leaf itself being + // root-attached). + if (_watched_leak_klass_count <= 0 || parent_tag == 0 || class_tag == 0) { + return; + } + // (u32) truncation matches _watched_leak_klass_ids' own storage (see that + // field's comment) - class tags are small, negative, sequentially-minted + // values in practice (ClassTagAllocator::next()), so this never actually + // loses distinguishing information; it just keeps the comparison and the + // signature-key packing below in the same 32-bit space both already used + // for the (superseded) classMap-id scheme. + u32 truncated_class_tag = (u32)class_tag; + bool watched = false; + for (int i = 0; i < _watched_leak_klass_count; i++) { + if (_watched_leak_klass_ids[i] == truncated_class_tag) { + watched = true; + break; + } + } + if (!watched) { + return; + } + FrontierEntry parent_entry{}; + if (!frontier->lookup(parent_tag, &parent_entry) || + parent_entry.class_tag == 0) { + // Parent since pruned/dead between its own admission and this child's, + // or admitted before this field existed on it (should not happen in + // practice - class_tag is set at every admission - but a stale/unknown + // parent identity is not something to attribute this observation to + // either way. + return; + } + u64 key = leakSignatureKey(truncated_class_tag, (u32)parent_entry.class_tag); + _leak_signature_totals[key]++; + auto it = _leak_parent_fanout.find(parent_tag); + if (it == _leak_parent_fanout.end()) { + TEST_LOG("ReferenceChainTracker::trackLeakAccumulation fanout-insert " + "parent_tag=%lld parent_class_tag=%lld child_class_tag=%lld", + (long long)parent_tag, (long long)parent_entry.class_tag, + (long long)class_tag); + _leak_parent_fanout.emplace(parent_tag, LeakParentFanoutEntry{key, 1}); + } else { + // The signature key for a given parent_tag is fixed once recorded + // (parent_entry.class_tag never changes once admitted; the LEAF side of + // the key is fixed by which klass_id is currently watched at the time + // of THIS call, which could in principle differ between two children of + // the same parent if _watched_leak_klass_ids itself changed between + // them - overwrite rather than accumulate under a stale key in that + // case, since the stored signature_key should always reflect the most + // recently observed watched klass_id for this parent). + it->second.signature_key = key; + it->second.fanout++; + } + // ANCESTOR FANOUT: the direct parent is not necessarily the part of the + // holder chain that STAYS LIVE. A container that replaces its internals + // (the canonical unmaintained-singleton leak: a growing ArrayList swaps + // elementData on growth, a HashMap resizes its table) leaves the watched + // instances' direct parents dead - observed live: the fanout filled with + // old backing arrays while the live holder was never re-walked, its new + // internals never admitted, and zero tagged chunks ever intercepted. + // The ancestors up to the root ARE the durable holders, so record every + // hop of the holder chain, not just the last one. Bounded: this only runs + // for watched-klass admissions (rare - leak-candidate classes only), and + // the walk stops at the root-attached entry (parent_tag == 0), typically + // a handful of lookups. + jlong ancestor = parent_entry.parent_tag; + int hops = 0; + while (ancestor != 0 && hops++ < _hop_cap) { + FrontierEntry ancestor_entry{}; + if (!_frontier->lookup(ancestor, &ancestor_entry)) { + break; + } + if (_leak_parent_fanout.find(ancestor) == _leak_parent_fanout.end()) { + _leak_parent_fanout.emplace(ancestor, LeakParentFanoutEntry{key, 1}); + } + if (ancestor_entry.parent_tag == 0) { + break; // root-attached: the holder chain ends here + } + ancestor = ancestor_entry.parent_tag; + } +} + +void ReferenceChainTracker::seedLeakAccumulationForNewlyWatchedKlass( + u32 klass_id) { + if (_frontier == nullptr) { + // pollWatchedTargets() can run before the first pass has ever created + // the frontier table - nothing to seed from yet. + return; + } + int table_size = _frontier->size(); + if (table_size <= 0) { + return; + } + // Inlines trackLeakAccumulation()'s own signature/fanout update logic + // (rather than calling it per matching entry) deliberately: this whole + // scan already holds _frontier's shared lock for its duration (matching + // collectStaleExpandedEntriesForRotation()'s own lockShared() rationale - + // a per-tag SpinLock acquisition would double the cost of this O(table_size) + // sweep), and trackLeakAccumulation() takes that same lock itself via + // frontier->lookup() - calling it from inside an already-held shared + // section would risk a reentrant-lock deadlock if a writer is ever + // concurrently pending, so this uses lookupLocked() throughout instead. + // Compares against (u32) FrontierEntry::class_tag - see that field's own + // comment for why this, and not referrer_klass, is the stable identifier + // klass_id (itself a truncated class_tag - _watched_leak_klass_ids' own + // comment) can actually be matched against. + _frontier->withSharedLock([&](const FrontierTable *frontier) { + for (jlong tag = 1; tag <= table_size; tag++) { + FrontierEntry entry{}; + if (!frontier->lookupLocked(tag, &entry) || + entry.state != FrontierEntryState::EXPANDED || + entry.parent_tag == 0 || (u32)entry.class_tag != klass_id) { + continue; + } + FrontierEntry parent_entry{}; + if (!frontier->lookupLocked(entry.parent_tag, &parent_entry) || + parent_entry.class_tag == 0) { + continue; + } + u64 key = leakSignatureKey(klass_id, (u32)parent_entry.class_tag); + _leak_signature_totals[key]++; + auto it = _leak_parent_fanout.find(entry.parent_tag); + if (it == _leak_parent_fanout.end()) { + _leak_parent_fanout.emplace(entry.parent_tag, + LeakParentFanoutEntry{key, 1}); + } else { + it->second.signature_key = key; + it->second.fanout++; + } + } + }); +} + +bool ReferenceChainTracker::maybeUpgradeRootAttachedRootKind( + FrontierTable *frontier, jlong tag, u8 new_root_kind) { + FrontierEntry entry{}; + if (!frontier->lookup(tag, &entry)) { + return false; + } + if (entry.parent_tag != 0) { + // Not root-attached - per this phase's option (a) resolution of the + // parent_tag==0/root_kind invariant conflict (referenceChains.h's + // FrontierEntry::root_kind comment), only a root-context update may ever + // write a non-zero root_kind, and only onto an entry that is already + // root-attached. An object that happens to also be a genuine GC root but + // was first discovered as a non-root child (e.g. via frontier + // expansion) keeps its original, non-root attribution - a known, + // documented limitation rather than an attempt to retroactively flip + // parent_tag to 0, which reconstructChain()'s parent-link walk does not + // support. + return false; + } + if (rootKindDurability(new_root_kind) <= rootKindDurability(entry.root_kind)) { + return false; + } + frontier->updateRootKind(tag, new_root_kind); + // TEMP DIAGNOSTIC (pod round 12): include klass_id so the LEAK_BUFFER + // wrapper (SynchronizedRandomAccessList) can be identified among the + // upgraded entries. Remove once the wrapper question is answered. + TEST_LOG("ReferenceChainTracker::maybeUpgradeRootAttachedRootKind tag=%lld " + "old_root_kind=%d -> new_root_kind=%d klass_id=%u", + (long long)tag, (int)entry.root_kind, (int)new_root_kind, + entry.referrer_klass); + addToStaticAnchorIndex(tag, entry.class_tag, new_root_kind); + return true; +} + +std::vector +ReferenceChainTracker::collectStaleRootKindEntriesForRotation( + int max_count) { + std::vector selected; + int table_size = _frontier->size(); + if (max_count <= 0 || table_size <= 0) { + return selected; + } + if (_root_kind_rotation_cursor <= 0 || + _root_kind_rotation_cursor > table_size) { + _root_kind_rotation_cursor = 1; + } + + // Held for the whole sweep below (potentially wrapping all the way around + // table_size) rather than once per tag via lookup() - the same rationale + // as collectStaleExpandedEntriesForRotation()'s own lockShared() use: a + // per-tag SpinLock acquisition would double this scan's cost under a large + // frontier table. + jlong start_tag = _root_kind_rotation_cursor; + jlong tag = start_tag; + _frontier->withSharedLock([&](const FrontierTable *frontier) { + do { + FrontierEntry entry{}; + if (frontier->lookupLocked(tag, &entry) && + entry.state == FrontierEntryState::EXPANDED && + entry.parent_tag == 0 && isTransientRootKind(entry.root_kind) && + !isQueuedForRotation(tag) && + _priority_expand.size() < PRIORITY_EXPAND_CAP) { + selected.push_back(tag); + _priority_expand.push_back(tag); + _priority_expand_set.insert(tag); + if ((int)selected.size() >= max_count) { + tag = tag % table_size + 1; + break; + } + } + tag = tag % table_size + 1; + } while (tag != start_tag); + }); + + _root_kind_rotation_cursor = tag; + return selected; +} + +std::vector +ReferenceChainTracker::collectStaleExpandedEntriesForRotation( + int max_count) { + std::vector selected; + int table_size = _frontier->size(); + if (max_count <= 0 || table_size <= 0) { + return selected; + } + // LEAK-PARENT PRIORITY, FAIR-SHARED WITH THE BLIND LAP: _leak_parent_fanout + // knows the EXPANDED parents that actually lead to watched leak-klass + // children - re-walking one of those re-sees its current children + // (improveChain() upgrades children first admitted via a shallower path, + // leak-tag interception for the tagged ones) and catches elements added + // since its expansion, which is exactly the mutation this rotation exists + // to observe. The fanout is orders of magnitude smaller than the table; + // select from it first (rotating via _leak_parent_rotation_cursor for + // coverage), up to HALF the budget (ceil) - then the blind table lap below + // fills the remainder. + // + // Why capped at half rather than fanout-first-until-exhausted (the + // original design, observed broken live): the fanout only ever contains + // parents of watched instances ALREADY ADMITTED as their direct children - + // and for a container that REPLACES its internals (the canonical + // unmaintained-singleton case: a growing ArrayList swaps elementData on + // growth), the watched instances' direct parents are the OLD, now-dead + // backing arrays, while the live holder's new internals are never in the + // fanout at all (the holder's own direct children are non-watched + // container internals). Re-walking the LIVE holder is what admits each + // new backing array; only the blind lap selects an arbitrary EXPANDED + // holder. With an unbounded fanout-first policy and a fanout grown to + // ~11k entries, the fanout filled ALL 256 selections every pass + // (observed live: rotation edges admitted in only 4 of 206 passes, the + // sink's resized backing arrays never admitted, zero interceptions) and + // the lap never ran - the exact starvation this rotation was built to + // prevent, reproduced one level down. A half/half split guarantees both + // tiers make progress every pass. + // + // FANOUT HYGIENE: entries whose parent no longer resolves in the + // frontier (pruned: dead object, search-restart wipe) can never be + // re-walked again, yet accumulate forever without this erase - observed + // live as an 11k-entry fanout of overwhelmingly-dead old backing arrays, + // which both bloats this scan and makes _leak_parent_rotation_cursor's + // lap arithmetic cover mostly corpses. Entries that exist but are not + // EXPANDED yet (still pending expansion) are kept - their children have + // not even been seen once. + if (!_leak_parent_fanout.empty() && + _priority_expand.size() < PRIORITY_EXPAND_CAP) { + int fanout_budget = (max_count + 1) / 2; + size_t fanout_size = _leak_parent_fanout.size(); + u64 skip = _leak_parent_rotation_cursor % fanout_size; + auto it = _leak_parent_fanout.begin(); + while (it != _leak_parent_fanout.end()) { + if ((int)selected.size() >= fanout_budget || + _priority_expand.size() >= PRIORITY_EXPAND_CAP) { + break; + } + if (skip > 0) { + skip--; + ++it; + continue; + } + jlong parent_tag = it->first; + if (isQueuedForRotation(parent_tag)) { + ++it; + continue; + } + FrontierEntry entry{}; + // Dead parent: either the frontier slot is gone entirely, or it was + // clear()'d (dead object / restart wipe) - clear() marks the slot + // ABANDONED rather than removing it, so both conditions must erase + // (tags are never reused within a search and the fanout is wiped on + // restart, so an ABANDONED parent can never come back to life). + if (!_frontier->lookup(parent_tag, &entry) || + entry.state == FrontierEntryState::ABANDONED) { + it = _leak_parent_fanout.erase(it); + continue; + } + if (entry.state != FrontierEntryState::EXPANDED) { + ++it; + continue; + } + selected.push_back(parent_tag); + _priority_expand.push_back(parent_tag); + _priority_expand_set.insert(parent_tag); + ++it; + } + _leak_parent_rotation_cursor += selected.size() + 1; + if ((int)selected.size() >= max_count) { + // Budget exhausted by the fanout alone (only possible for + // max_count == 1, where the fanout's ceil-half share is the whole + // budget) - fanout-priority preserved, and the lap below has nothing + // left to do this pass. + return selected; + } + } + if (_stale_expanded_rotation_cursor <= 0 || + _stale_expanded_rotation_cursor > table_size) { + _stale_expanded_rotation_cursor = 1; + } + // Resume scanning from _stale_expanded_rotation_cursor rather than always + // restarting at tag 1: a frontier table can accumulate far more than + // max_count entries that are EXPANDED and stay that way forever + // (long-lived infrastructure objects - caches, maps, bootstrap classes). + // An always-from-1 scan would let that low-tag population fill this + // sweep's entire cap on every single call, permanently starving any + // EXPANDED entry with a higher tag (e.g. a static field's collection, + // admitted only once its class loads well after startup) of ever being + // re-queued. A wrapping cursor, like collectStaleRootKindEntriesForRotation() + // above already uses, guarantees every entry gets a turn within + // ceil(table_size / max_count) calls instead of never. + // + // This scan's own EXPANDED criterion is a strict superset of + // collectStaleRootKindEntriesForRotation()'s (which additionally requires + // parent_tag == 0 and a transient root_kind), and that function always + // runs first within the same pass and pushes its picks onto + // _priority_expand before this one runs - so without a check here, a tag + // it already selected would be pushed a second time, and + // expandFrontier() re-expands each deque entry as its own independent + // unit of work. isQueuedForRotation() also covers any entries still + // sitting there from a prior pass's truncated batch (expandFrontier() + // leaves those at the front of the queue for a later retry rather than + // popping them). + // Held for the whole scan below instead of once per tag via lookup() - a + // per-tag SpinLock acquisition/release would double the cost of this + // O(table_size) sweep under a large frontier table (the exact scenario - + // tens of thousands of entries - this rotation mechanism targets). + // + // A sparse stretch of non-EXPANDED/already-queued tags could make this + // scan run long chasing max_count with no wall-clock bound of its own - + // unlike heapReferenceCallback()'s per-edge check, this scan isn't itself + // a JVMTI/STW call, but it still steals from the same _pass_deadline_ns + // window the actual walk needs (see that field's own comment). Amortized + // the same way heapReferenceCallback() amortizes its own check: an + // OS::nanotime() call every iteration would + // itself be a meaningful fraction of this loop's per-tag cost. + int deadline_check_counter = 0; + jlong start_tag = _stale_expanded_rotation_cursor; + jlong tag = start_tag; + _frontier->withSharedLock([&](const FrontierTable *frontier) { + do { + if (_pass_deadline_ns != 0 && + (++deadline_check_counter & 0xFFF) == 0 && + OS::nanotime() >= _pass_deadline_ns) { + // Ran past this pass's wall-clock share - stop scanning with + // whatever was already selected (possibly none) and resume from + // here next call. The wrapping cursor already tolerates a call that + // selects fewer than max_count, so this composes without any + // special-casing. + break; + } + FrontierEntry entry{}; + if (frontier->lookupLocked(tag, &entry) && + entry.state == FrontierEntryState::EXPANDED && + !isQueuedForRotation(tag) && + _priority_expand.size() < PRIORITY_EXPAND_CAP) { + selected.push_back(tag); + _priority_expand.push_back(tag); + _priority_expand_set.insert(tag); + if ((int)selected.size() >= max_count) { + tag = tag % table_size + 1; + break; + } + } + tag = tag % table_size + 1; + } while (tag != start_tag); + }); + _stale_expanded_rotation_cursor = tag; + return selected; +} + +// Bounded rotating re-expansion targeting the accumulation point of a +// klass LivenessTracker has flagged as growing - the design's actual +// targeted tier, replacing an earlier structural (depth + root-durability + +// class-shape) heuristic that measurement against a real classpath showed +// selects far too much of the reachable graph to fit in a small budget (see +// git history and doc/temp/ investigation notes). This design instead uses +// the one signal that CAN distinguish "the specific container that is +// leaking" from "the many unrelated objects that happen to hold instances +// of a common leaf class" without needing a full dominator-tree/retained- +// size computation (a wider web search into how heap analysis tools solve +// this - Eclipse MAT's accumulation-point/big-drop-in-dominator-tree +// heuristic, Cork's class-level points-from summary diffed across GCs, +// LeakBot's rank-cheaply-then-track-only-the-winners discipline - converged +// on the same two-tier shape implemented here): +// +// Tier 1 (class-level, cheap, aggregated at admission time by +// trackLeakAccumulation() into _leak_signature_totals/_leak_parent_fanout - +// no full-table scan): rank (leaf_klass_id, parent_class_id) signatures by +// growth since the previous pass (current total minus the snapshot rolled +// forward at the end of that pass), Cork-style. This collapses "thousands +// of objects holding a common leaf class" into a handful of signatures - a +// legitimate cache class that merely holds MANY instances, but not a +// GROWING number of them pass over pass, never wins here, regardless of its +// absolute size. +// +// Tier 2 (spent only within the winning signature): rank the concrete +// parent objects contributing to it by their own fanout of the flagged +// leaf klass_id - the highest-fanout parent is the one whose already- +// EXPANDED state is most likely stale (i.e. its own children were admitted +// once and it has since accumulated more that were never observed), so it +// is the one worth spending this pass's rotation budget re-expanding. +// +// Unlike the other two rotation collectors, this one does not use a +// wrapping cursor: it always selects the current best candidates rather +// than guaranteeing fair coverage of a population, since re-selecting the +// same still-growing parent every pass is exactly the desired behavior, +// not something a fairness guarantee needs to correct for. +std::vector +ReferenceChainTracker::collectLeakAccumulationCandidatesForRotation( + int max_count) { + std::vector selected; + if (max_count <= 0 || _leak_signature_totals.empty()) { + return selected; + } + + // Tier 1: rank signatures by growth since the last pass's snapshot. A + // signature with no prior snapshot (brand new this pass) compares against + // an implicit prev_total of 0 - see _leak_signature_prev_totals' own + // comment for why that is the correct behavior, not a special case. + u64 winning_key = 0; + bool have_winner = false; + u32 best_delta = 0; + for (const auto &kv : _leak_signature_totals) { + u32 prev = 0; + auto prev_it = _leak_signature_prev_totals.find(kv.first); + if (prev_it != _leak_signature_prev_totals.end()) { + prev = prev_it->second; + } + u32 delta = kv.second > prev ? kv.second - prev : 0; + if (delta > 0 && (!have_winner || delta > best_delta)) { + have_winner = true; + best_delta = delta; + winning_key = kv.first; + } + } + // Roll the snapshot forward for the NEXT pass's comparison regardless of + // whether this pass found a winner - a signature that didn't grow this + // pass still needs its current total remembered so a future pass's delta + // is computed against the right baseline, not against however many + // passes ago it was last checked. + _leak_signature_prev_totals = _leak_signature_totals; + if (!have_winner) { + // Nothing grew since last pass - nothing to prioritize this tier this + // time (collectStaleExpandedEntriesForRotation()'s unprioritized + // fallback still covers this population eventually). + return selected; + } + + // Tier 2: within the winning signature only, rank concrete parent objects + // by their own fanout - collected first, then partially sorted, since + // _leak_parent_fanout's total size is what bounds this method's cost (not + // table_size), and is expected to be small (see that map's own comment). + // + // Two parent states qualify: + // - EXPANDED: the ranking's original case - the parent's children were + // admitted once and its EXPANDED state is stale (it has since + // accumulated more children that were never observed), so re-expanding + // it admits the new ones. + // - FRONTIER: the parent is a known holder of watched-klass children + // that has never been expanded at all. Measured live on hotdog (round + // 4, ev-leaktag-onpod-round4): with a 126,895-entry _pending_expand + // backlog draining at ~120-200 objects/min, the growing holders (an + // elementData-sized Object[] of the leaking collection) stay FRONTIER + // for hours, and an EXPANDED-only filter made this tier select ZERO + // every pass while holding 51k known parent candidates + // (leak_accumulation_tags=0 on all 183 passes) - the targeted tier + // going dead in exactly the regime it exists for. Allowing FRONTIER + // parents means a first expansion, and it must JUMP the backlog + // rather than join it: selections are pushed to the FRONT of + // _priority_expand so the next expandFrontier() batch reaches them + // (that deque already held ~1016 stale re-walk entries on the same + // pod; push_back would queue the targeted holder behind all of them). + // + // A FRONTIER-state selection may still have a stale copy sitting in + // _pending_expand (queued there at admission time); the pending-lane + // drain eventually pops that copy and re-expands an already-EXPANDED + // object - one wasted expansion, deduped to edges=0 by the + // ALREADY_ADMITTED check. Bounded (once per selection) and harmless next + // to the value of reaching the holder at all. + std::vector> candidates; // (parent_tag, fanout) + for (const auto &kv : _leak_parent_fanout) { + if (kv.second.signature_key == winning_key && !isQueuedForRotation(kv.first)) { + FrontierEntry entry{}; + if (_frontier->lookup(kv.first, &entry) && + (entry.state == FrontierEntryState::EXPANDED || + entry.state == FrontierEntryState::FRONTIER)) { + candidates.emplace_back(kv.first, kv.second.fanout); + } + } + } + std::sort(candidates.begin(), candidates.end(), + [](const std::pair &a, const std::pair &b) { + return a.second > b.second; + }); + for (const auto &c : candidates) { + if ((int)selected.size() >= max_count || + _priority_expand.size() >= PRIORITY_EXPAND_CAP) { + break; + } + selected.push_back(c.first); + _priority_expand_set.insert(c.first); + FrontierEntry state_entry{}; + bool is_expanded = _frontier->lookup(c.first, &state_entry) && + state_entry.state == FrontierEntryState::EXPANDED; + TEST_LOG("ReferenceChainTracker::" + "collectLeakAccumulationCandidatesForRotation selected " + "parent_tag=%lld state=%s fanout=%u", + (long long)c.first, is_expanded ? "EXPANDED" : "FRONTIER", + c.second); + } + // Place the whole selection at the head of the priority lane, keeping + // the fanout ranking order (see the FRONTIER-state case in the Tier 2 + // comment above for why the head and not the tail): push_front reverses, + // so insert back-to-front. + for (auto it = selected.rbegin(); it != selected.rend(); ++it) { + _priority_expand.push_front(*it); + } + return selected; +} + +// --------------------------------------------------------------------------- +// Retention-edge labels: naming the field each chain hop is retained +// through, from the JVMTI-specification field ordinal captured at +// admission (see FrontierEntry::referrer_field_index's own comment). +// --------------------------------------------------------------------------- +namespace { + +// Fallback label for a hop whose edge is not a field reference (or whose +// field ordinal could not be decoded) - the edge KIND, never a fabricated +// name. Numbering per jvmti.h's jvmtiHeapReferenceKind. +const char *hopEdgeKindLabel(u8 kind) { + switch (kind) { + case JVMTI_HEAP_REFERENCE_CLASS: + return "class"; + case JVMTI_HEAP_REFERENCE_FIELD: + return "field"; + case JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT: + return "element"; + case JVMTI_HEAP_REFERENCE_CLASS_LOADER: + return "class_loader"; + case JVMTI_HEAP_REFERENCE_SIGNERS: + return "signers"; + case JVMTI_HEAP_REFERENCE_PROTECTION_DOMAIN: + return "protection_domain"; + case JVMTI_HEAP_REFERENCE_INTERFACE: + return "interface"; + case JVMTI_HEAP_REFERENCE_STATIC_FIELD: + return "static_field"; + case JVMTI_HEAP_REFERENCE_CONSTANT_POOL: + return "constant_pool"; + case JVMTI_HEAP_REFERENCE_SUPERCLASS: + return "superclass"; + case JVMTI_HEAP_REFERENCE_JNI_GLOBAL: + return "jni_global"; + case JVMTI_HEAP_REFERENCE_SYSTEM_CLASS: + return "system_class"; + case JVMTI_HEAP_REFERENCE_MONITOR: + return "monitor"; + case JVMTI_HEAP_REFERENCE_STACK_LOCAL: + return "stack_local"; + case JVMTI_HEAP_REFERENCE_JNI_LOCAL: + return "jni_local"; + case JVMTI_HEAP_REFERENCE_THREAD: + return "thread"; + case JVMTI_HEAP_REFERENCE_OTHER: + return "other"; + default: + return "unknown"; + } +} + +// Appends the own-declared field names of `cls` to *out, in GetClassFields() +// order. Does NOT delete `cls`'s local ref - the caller owns every jclass +// local ref it passes in and deletes them on all paths exactly once. +// Returns false on any JVMTI failure (caller treats the whole class +// as undecodable - partial lists would silently MISNAME ordinals). +bool appendClassFieldNames(jvmtiEnv *jvmti, JNIEnv *jni, jclass cls, + std::vector *out) { + jint count = 0; + jfieldID *fields = nullptr; + if (jvmti->GetClassFields(cls, &count, &fields) != JVMTI_ERROR_NONE) { + return false; + } + bool ok = true; + for (jint i = 0; i < count; i++) { + char *name = nullptr; + if (jvmti->GetFieldName(cls, fields[i], &name, nullptr, nullptr) != + JVMTI_ERROR_NONE || + name == nullptr) { + ok = false; + break; + } + out->emplace_back(name); + jvmti->Deallocate((unsigned char *)name); + } + jvmti->Deallocate((unsigned char *)fields); + return ok; +} + +// Sums the own-declared field counts of every interface transitively +// implemented/extended by `cls`, each counted once (a diamond interface graph +// shared by two branches must not double-count - the spec's ordinal space +// counts interface fields once each, and HotSpot's transitive_interfaces +// array contains each interface exactly once). `seen` dedupes by raw class +// tag. Returns -1 on JVMTI failure. +jlong interfaceFieldCount(jvmtiEnv *jvmti, JNIEnv *jni, jclass cls, + std::unordered_set *seen) { + // GetTag is safe on any jclass; the shared class-tag allocator + // (classTagAllocator.h) tags every class resolveLoadedClasses() sees, and + // classes here always come from that set. + jlong tag = 0; + if (jvmti->GetTag(cls, &tag) != JVMTI_ERROR_NONE || tag == 0) { + // Untagged interface: cannot dedupe reliably - fail the whole decode + // rather than risk double counting. + return -1; + } + if (!seen->insert(tag).second) { + return 0; // already counted this interface (a shared subinterface) + } + jint iface_count = 0; + jclass *ifaces = nullptr; + if (jvmti->GetImplementedInterfaces(cls, &iface_count, &ifaces) != + JVMTI_ERROR_NONE) { + return -1; + } + jlong total = 0; + for (jint i = 0; i < iface_count; i++) { + if (ifaces[i] == nullptr) { + continue; + } + // An interface's own fields count toward any implementor's ordinal + // base - matching the spec's "count of the fields in all the interfaces + // implemented by C" (jvmtiHeapReferenceInfoField). + jint field_count = 0; + jfieldID *fields = nullptr; + if (jvmti->GetClassFields(ifaces[i], &field_count, &fields) == + JVMTI_ERROR_NONE) { + total += field_count; + jvmti->Deallocate((unsigned char *)fields); + } + total += interfaceFieldCount(jvmti, jni, ifaces[i], seen); + if (total < 0) { + return -1; + } + jni->DeleteLocalRef(ifaces[i]); + } + jvmti->Deallocate((unsigned char *)ifaces); + return total; +} + +} // namespace + +const ReferenceChainTracker::HopLabelClass * +ReferenceChainTracker::hopLabelClassFor(jvmtiEnv *jvmti, JNIEnv *jni, + jlong class_tag) { + auto it = _hop_label_cache.find(class_tag); + if (it != _hop_label_cache.end()) { + return &it->second; + } + // Bounded: chains reference few distinct referrer classes; a wholesale + // clear at the cap (rather than LRU eviction) keeps this O(1) and is + // correct because the cache is purely derived state - any cleared entry + // is transparently rebuilt on its next hop. + if (_hop_label_cache.size() >= HOP_LABEL_CLASS_CACHE_CAP) { + _hop_label_cache.clear(); + } + HopLabelClass entry{}; + entry.class_tag = class_tag; + entry.decode_failed = true; // until proven otherwise + do { + // GetSuperclass is a JNI (not JVMTI) function - modern JVMTI dropped it + // (the spec delivers superclass references via heap callbacks, + // jvmti.xml's JVMTI_HEAP_REFERENCE_SUPERCLASS note); the rest are JVMTI + // slots. Partial function tables (gtest mock environments stub only + // the slots their tests drive) degrade to kind labels rather than + // calling a null slot. + if (jvmti->functions->GetObjectsWithTags == nullptr || + jvmti->functions->IsInterface == nullptr || + jvmti->functions->GetImplementedInterfaces == nullptr || + jvmti->functions->GetClassFields == nullptr || + jvmti->functions->GetFieldName == nullptr || + jni->functions->GetSuperclass == nullptr) { + break; + } + // Resolve the class object from its raw tag (negative - the shared + // allocator's class tags; GetObjectsWithTags accepts any tag value). + jint count = 0; + jobject *objects = nullptr; + jlong *tags = nullptr; + if (jvmti->GetObjectsWithTags(1, &class_tag, &count, &objects, &tags) != + JVMTI_ERROR_NONE || + count != 1 || objects == nullptr || objects[0] == nullptr) { + break; + } + jclass cls = (jclass)objects[0]; + jboolean is_interface = JNI_FALSE; + std::vector names; + bool ok = false; + if (jvmti->IsInterface(cls, &is_interface) == JVMTI_ERROR_NONE) { + if (is_interface) { + // The spec's INTERFACE branch: base = fields of all superinterfaces + // of I, then I's own fields (jvmtiHeapReferenceInfoField). + std::unordered_set seen; + jlong base = interfaceFieldCount(jvmti, jni, cls, &seen); + if (base >= 0) { + names.resize((size_t)base); // positioned but unnamed: ordinal [0, + // base) is interface fields, only + // reachable through an interface + // branch decode of a superinterface + ok = appendClassFieldNames(jvmti, jni, cls, &names); + } + } else { + // The spec's CLASS branch: base = fields of all interfaces + // implemented by C, then the superclass chain root-first + // (java.lang.Object's fields first, C's own last), each class's + // fields in GetClassFields() order. + std::unordered_set seen; + jlong base = interfaceFieldCount(jvmti, jni, cls, &seen); + if (base >= 0) { + names.resize((size_t)base); + // GetSuperclass walks UP, so gather then append in reverse + // (root first). supers[] holds local refs of every class along + // the way, cls included - deleted together below, exactly once. + jclass supers[128]; + int depth = 0; + jclass k = cls; + while (k != nullptr && + depth < (int)(sizeof(supers) / sizeof(supers[0]))) { + supers[depth++] = k; + k = jni->GetSuperclass(k); + } + ok = (k == nullptr); // deeper than 128 classes: fail rather than + // misname + for (int i = depth - 1; ok && i >= 0; i--) { + ok = appendClassFieldNames(jvmti, jni, supers[i], &names); + } + for (int i = 0; i < depth; i++) { + jni->DeleteLocalRef(supers[i]); + } + } + } + } + jni->DeleteLocalRef(cls); + if (!ok) { + break; + } + entry.field_names = std::move(names); + entry.decode_failed = false; + } while (false); + auto inserted = _hop_label_cache.emplace(class_tag, std::move(entry)); + return &inserted.first->second; +} + +void ReferenceChainTracker::resolveHopEdgeLabel(jvmtiEnv *jvmti, JNIEnv *jni, + ChainHopEdge edge, char *out, + size_t out_cap) { + if (jvmti != nullptr && jni != nullptr && + (edge.edge_kind == JVMTI_HEAP_REFERENCE_FIELD || + edge.edge_kind == JVMTI_HEAP_REFERENCE_STATIC_FIELD) && + edge.field_index >= 0 && edge.referrer_class_tag != 0 && out_cap > 0) { + const HopLabelClass *labels = + hopLabelClassFor(jvmti, jni, edge.referrer_class_tag); + if (labels != nullptr && !labels->decode_failed && + (size_t)edge.field_index < labels->field_names.size()) { + const std::string &name = + labels->field_names[(size_t)edge.field_index]; + if (!name.empty()) { + size_t n = name.size() < out_cap - 1 ? name.size() : out_cap - 1; + memcpy(out, name.data(), n); + out[n] = '\0'; + return; + } + // Empty positioned slot (an interface-field ordinal below the + // class's own base) - fall through to the kind label. + } + } + if (out_cap > 0) { + snprintf(out, out_cap, "%s", hopEdgeKindLabel(edge.edge_kind)); + } +} + +void ReferenceChainTracker::fillHopEdgeLabels( + jvmtiEnv *jvmti, JNIEnv *jni, const std::vector &edges, + std::vector *out) { + out->clear(); + out->reserve(edges.size()); + char label[MAX_HOP_EDGE_LABEL + 1]; + for (ChainHopEdge edge : edges) { + resolveHopEdgeLabel(jvmti, jni, edge, label, sizeof(label)); + out->emplace_back(label); + } +} + +// --------------------------------------------------------------------------- +// Candidate-scoped reach: bounded descend walks from anchor objects (see +// descendFromAnchor()'s declaration comment, referenceChains.h). Reaching +// the tagged leak instances is a correctness requirement, not a +// throughput optimization: breadth-first FIFO expansion over a rising +// heap can never drain (pod rounds 5-6), so the tagged instances sit +// under holders the ordinary crawl reaches only after hours. +// --------------------------------------------------------------------------- +namespace { + +// Class tags to neither admit nor descend into during a descend walk (see +// PassContext::_no_descend_class_tags' own comment). All three are +// bootstrap-resolvable, so FindClass works from any JNI context; a class +// not yet tagged by resolveLoadedClasses() (tag 0) is simply not added - +// the gate then stays off for that class until a later pass re-resolves. +const char *const kNoDescendClassNames[] = { + "java/lang/ClassLoader", "java/lang/ThreadGroup", + "java/security/ProtectionDomain", +}; + +int resolveNoDescendClassTags(jvmtiEnv *jvmti, JNIEnv *jni, + jlong *out, int cap) { + int count = 0; + for (const char *name : kNoDescendClassNames) { + if (count >= cap) { + break; + } + jclass cls = jni->FindClass(name); + if (cls == nullptr) { + // Not loadable in this JVM (e.g. java.security classes stripped by a + // minimal runtime) - skip; the gate simply does not cover it. + jni->ExceptionClear(); + continue; + } + jlong tag = 0; + if (jvmti->GetTag(cls, &tag) == JVMTI_ERROR_NONE && tag != 0) { + out[count++] = tag; + } + jni->DeleteLocalRef(cls); + } + return count; +} + +// java.lang.ThreadLocal$ThreadLocalMap's class tag for +// walkCandidateThreadLocals()'s anchor gate (see PassContext:: +// _anchor_descend_class_tag's own comment): the value type of BOTH of +// Thread's threadLocals and inheritableThreadLocals fields, and its exact +// class tag is what the anchor gate compares against. The class is +// package-private but already loaded in any JVM that has ever touched a +// ThreadLocal (FindClass resolves by name regardless of access), and the +// class name is stable across JDK 8-26. Returns 0 if not resolvable (not +// yet loaded / FindClass refused) - the caller then walks the Thread's +// edges generically, gated only by the no-descend class set + hop cap, +// rather than skipping the walk entirely. +jlong resolveThreadLocalMapClassTag(jvmtiEnv *jvmti, JNIEnv *jni) { + jlong tag = 0; + jclass cls = jni->FindClass("java/lang/ThreadLocal$ThreadLocalMap"); + if (cls == nullptr) { + jni->ExceptionClear(); + return 0; + } + jvmti->GetTag(cls, &tag); + jni->DeleteLocalRef(cls); + return tag; +} + +} // namespace + +void ReferenceChainTracker::descendFromAnchor( + jvmtiEnv *jvmti, JNIEnv *jni, jobject anchor, jlong anchor_tag, + u32 anchor_depth, jlong anchor_descend_class_tag, int budget, + int *edges_admitted, bool *truncated, bool *frontier_cap_hit, + u64 *safepoint_ticks, bool diag_trace) { + PassContext ctx; + ctx.tracker = this; + ctx.frontier = _frontier; + ctx._diag_trace = diag_trace; + // Bound admission to DESCENT_HOPS below the anchor, still subject to the + // global hop cap. Caveat (accepted, bounded): a pre-existing frontier + // entry reachable inside the subgraph carries its GLOBAL depth (from + // whatever path first admitted it), so the walk can descend more than + // DESCENT_HOPS below the anchor through such an entry - never past + // _hop_cap + the pass deadline though, the same bounds every other walk + // phase lives under. + int descent_cap = (int)anchor_depth + DESCENT_HOPS; + ctx.hop_cap = descent_cap < _hop_cap ? descent_cap : _hop_cap; + ctx.budget = budget; + ctx.edges_admitted = 0; + ctx.truncated = false; + ctx.frontier_cap_hit = false; + + ctx._no_descend_class_tag_count = + resolveNoDescendClassTags(jvmti, jni, ctx._no_descend_class_tags, + PassContext::NO_DESCEND_CLASS_CAP); + if (anchor_descend_class_tag != 0) { + ctx._descent_anchor_tag = anchor_tag; + ctx._anchor_descend_class_tag = anchor_descend_class_tag; + } + + jvmtiHeapCallbacks callbacks; + memset(&callbacks, 0, sizeof(callbacks)); + callbacks.heap_reference_callback = heapReferenceCallback; + u64 follow_start_ticks = TSC::ticks(); + jvmti->FollowReferences(0, nullptr, anchor, &callbacks, &ctx); + *safepoint_ticks += TSC::ticks() - follow_start_ticks; + if (diag_trace) { + // TEMP DIAGNOSTIC (pod round 12, PassContext::_diag_trace's own + // comment): dump this walk's admission/re-encounter sequence - the + // wrapper-walk question is exactly "was list/elementData enumerated, + // and did any leak-class ([B) entry carry a leak tag". + for (int i = 0; i < ctx._diag_count; i++) { + TEST_LOG("ReferenceChainTracker::descendFromAnchor diag anchor=%lld " + "entry=%d klass_id=%u seen_as=%u", + (long long)anchor_tag, i, ctx._diag_klass_ids[i], + (unsigned)ctx._diag_leak_flags[i]); + } + TEST_LOG("ReferenceChainTracker::descendFromAnchor diag anchor=%lld " + "recorded=%d edges=%d truncated=%d", + (long long)anchor_tag, ctx._diag_count, ctx.edges_admitted, + (int)ctx.truncated); + } + *edges_admitted += ctx.edges_admitted; + *truncated = *truncated || ctx.truncated; + *frontier_cap_hit = *frontier_cap_hit || ctx.frontier_cap_hit; +} + +std::vector +ReferenceChainTracker::collectStaticFieldAnchorsForRotation(int max_count) { + std::vector selected; + if (max_count <= 0 || _static_anchor_index.empty()) { + return selected; + } + // Tiered selection over _static_anchor_index (O(anchors) per pass, + // under ONE shared lock - the lookups below are lookupLocked()). The + // tiers, in walk order: + // 0. leak-tagged anchors (entry.leak_tag != 0) - always selected + // first (rare; no cursor needed). These are the only anchors that + // already lead to known-leaking objects. + // 1. FRESH anchors (the _static_anchor_fresh_queue drain) - anchors + // admitted since their last first-look attempt, container-shaped + // OR not-yet-classified. Admission order = sweep order = loaded- + // class order, so a leak holder held by a late-loaded class (the + // hotdog LEAK_BUFFER wrapper: holder class at sweep index 24627 of + // 33270, round-15 measurement) is admitted at the index TAIL - + // the very END of the fair container tier's lap. The measured + // hotdog search lifetime (44-75 passes) is shorter than + // ceil(container_cohort/budget) (1633/16 = 102), so fair-only + // coverage deterministically never reaches it; the fresh lane + // walks it within a pass or two of admission instead. Each anchor + // gets exactly ONE first look: a drain that outranks it (budget + // exhausted by earlier fresh picks) or classifies it as a + // non-container drops it from the queue, and it falls back to the + // fair tiers at its index position - covered eventually, just not + // urgently. Not-yet-classified anchors ride the lane because the + // wrapper admits one pass before reconcileAnchorClassShapes() can + // classify its class; "not-yet-classified" is bounded in practice + // (the shape cache is JVM-lifetime, so only genuinely new classes + // arrive unclassified, and churn classes are lambdas with no + // static fields). + // 2. container-shaped anchors, cursor-fair. A leak holder is + // typically a container, and the container cohort is far smaller + // than the full anchor population (round-14 hotdog measurement: + // 1633-1680 containers of 27739-30711 anchors). Fair so a cohort + // larger than the budget still rotates to coverage instead of + // hammering the same prefix. + // 3. everything else (String/Class/boxed/enum statics), cursor-fair, + // eventually covered within ceil(tier/budget) passes - explicitly + // NOT guaranteed within one search lifetime; that is the accepted + // cost of prioritizing containers (the round-13 starvation + // analysis). + // Eligibility filters (unchanged from the single-cursor version): + // liveness (entry cleared/ABANDONED since indexing), root-attached + // durable root kinds, FRONTIER/EXPANDED states, !isQueuedForRotation. + size_t idx_size = _static_anchor_index.size(); + if (_anchor_container_cursor >= idx_size) { + _anchor_container_cursor = 0; + } + if (_anchor_other_cursor >= idx_size) { + _anchor_other_cursor = 0; + } + struct TierPick { + size_t pos; + jlong tag; + }; + std::vector leak_picks; + std::vector fresh_picks; + std::vector container_picks; + std::vector other_picks; + leak_picks.reserve(16); + // Fresh picks kept by the queue drain (bounded by max_count) - used to + // keep the fair-tier consumption below from double-selecting them. + std::unordered_set fresh_kept_tags; + const size_t fresh_queue_len = _static_anchor_fresh_queue.size(); + _frontier->withSharedLock([&](const FrontierTable *frontier) { + // Index scan: partition every eligible anchor into the leak tier or + // one of the two fair tiers (the fresh lane is decided by the queue + // drain below - a fresh-kept anchor also lands in a fair pick vector + // here and is skipped at consumption time via fresh_kept_tags). + for (size_t i = 0; i < idx_size; i++) { + jlong tag = _static_anchor_index[i]; + FrontierEntry entry{}; + if (!frontier->lookupLocked(tag, &entry) || + entry.parent_tag != 0 || + (entry.root_kind != (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD && + entry.root_kind != (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL) || + (entry.state != FrontierEntryState::FRONTIER && + entry.state != FrontierEntryState::EXPANDED) || + isQueuedForRotation(tag)) { + continue; + } + if (entry.leak_tag != 0) { + leak_picks.push_back(TierPick{i, tag}); + } else if (i < _static_anchor_own_class_tags.size()) { + auto shape_it = + _class_shape_cache.find(_static_anchor_own_class_tags[i]); + if (shape_it != _class_shape_cache.end() && + shape_it->second == (u8)AnchorClassShape::CONTAINER) { + container_picks.push_back(TierPick{i, tag}); + } else { + other_picks.push_back(TierPick{i, tag}); + } + } else { + other_picks.push_back(TierPick{i, tag}); + } + } + // Fresh-lane drain. Every queue entry is popped (its ONE first look + // is spent either way): kept if eligible AND (container-shaped OR + // not-yet-classified) AND room remains in the budget; dropped + // otherwise. Leak-tagged anchors are dropped here - the leak tier + // above already owns them and its picks lead the selection anyway. + // The queue is a contiguous, ordered slice of the index (the two + // append together in addToStaticAnchorIndex; every removal - a cap + // drop, a spent first look - pops from the front), so entries drain + // in lockstep with index positions counting up from + // idx_size - fresh_queue_len: O(1) per entry, no tag search needed. + int fresh_room = max_count - (int)leak_picks.size(); + size_t drain_pos = fresh_queue_len <= idx_size ? idx_size - fresh_queue_len : 0; + while (!_static_anchor_fresh_queue.empty()) { + if (fresh_room <= 0) { + // Budget exhausted before the queue drained: everything + // remaining spends its first look now and falls back to the + // fair tiers at its index position (covered, not urgent). + _static_anchor_fresh_queue.clear(); + break; + } + jlong tag = _static_anchor_fresh_queue.front(); + _static_anchor_fresh_queue.pop_front(); + size_t pos = drain_pos; + drain_pos++; + if (pos >= idx_size || _static_anchor_index[pos] != tag) { + // The suffix-window invariant broke (cannot happen today; + // defensive): fall back to a search rather than mis-shape the + // entry - the queue is small, this is not a hot path once + // healthy. + auto it = + std::find(_static_anchor_index.begin(), + _static_anchor_index.end(), tag); + if (it == _static_anchor_index.end()) { + continue; + } + pos = (size_t)(it - _static_anchor_index.begin()); + } + FrontierEntry entry{}; + if (!frontier->lookupLocked(tag, &entry) || + entry.parent_tag != 0 || + (entry.root_kind != (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD && + entry.root_kind != (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL) || + (entry.state != FrontierEntryState::FRONTIER && + entry.state != FrontierEntryState::EXPANDED) || + isQueuedForRotation(tag) || entry.leak_tag != 0) { + continue; // dead/demoted/queued/leak-tier: first look spent, not + // fresh-kept (the leak tier selects it via the scan if + // it is leak-tagged) + } + bool keep = false; // container or not-yet-classified rides the + // lane; the wrapper admits one pass before + // reconcile can classify its class + if (pos < _static_anchor_own_class_tags.size()) { + auto shape_it = + _class_shape_cache.find(_static_anchor_own_class_tags[pos]); + keep = shape_it == _class_shape_cache.end() || + shape_it->second == (u8)AnchorClassShape::CONTAINER; + } else { + keep = true; // no own-class tag recorded - treat as unknown + } + if (!keep) { + continue; // classified non-container: the other tier owns it + } + fresh_picks.push_back(TierPick{pos, tag}); + fresh_kept_tags.insert(tag); + fresh_room--; + } + }); + // TEMP DIAGNOSTIC (pod round 14/15): the cohort histogram - the + // measured sizes of the tiers against the per-pass budget. This is + // the arithmetic gate for the container tier (a container cohort much + // larger than ~4k cannot be covered within a ~44-75-pass search + // lifetime and the frontier-cap conversation becomes the next lever) + // and for the fresh lane (fresh > budget means the lane runs as a + // backlog - the leak holder then waits fresh/budget passes, still + // bounded, but the number belongs in the next analysis). Remove once + // the pod verifies the arithmetic. + if (!leak_picks.empty() || !fresh_picks.empty() || + !container_picks.empty() || !other_picks.empty()) { + TEST_LOG_SUMMARY("ReferenceChainTracker::anchorTierHistogram " + "index=%zu leak_tier=%zu fresh_tier=%zu fresh_queue=%zu " + "container_tier=%zu other_tier=%zu budget=%d", + idx_size, leak_picks.size(), fresh_picks.size(), + fresh_queue_len, container_picks.size(), other_picks.size(), + max_count); + } + // Cursor-fair consumption of one tier: scan picks (sorted by pos by + // construction) starting at entries with pos >= cursor, stop at `want` + // OR at the lap end (NO within-call wrap: re-walking anchors this same + // call already covered would waste walk budget - the leftover budget + // flows to the next tier instead, and the cursor resets to 0 so the + // NEXT call starts a fresh lap). Fresh-kept tags are skipped - the + // fresh lane already selected them this call - but the cursor still + // passes their positions (they were covered this call, in effect). + auto consume_tier_fair = [&](const std::vector &picks, + size_t &cursor, int want) { + int took = 0; + if (want <= 0 || picks.empty()) { + return took; + } + size_t consumed_pos = 0; + for (size_t k = 0; k < picks.size() && took < want; k++) { + const TierPick &p = picks[k]; + if (p.pos < cursor) { + continue; + } + if (fresh_kept_tags.count(p.tag) > 0) { + continue; + } + selected.push_back(p.tag); + consumed_pos = p.pos; + took++; + } + if (took > 0) { + cursor = consumed_pos + 1 >= idx_size ? 0 : consumed_pos + 1; + } + return took; + }; + int budget_left = max_count; + for (const TierPick &p : leak_picks) { + if (budget_left <= 0) { + break; + } + selected.push_back(p.tag); + budget_left--; + } + // Fresh lane: queue order (admission order) so a burst larger than the + // budget spends the oldest first looks first and nothing jumps the + // queue; outranked fresh anchors fall back to the fair tiers at their + // positions (the drain already dropped them from the queue). + for (const TierPick &p : fresh_picks) { + if (budget_left <= 0) { + break; + } + selected.push_back(p.tag); + budget_left--; + } + budget_left -= consume_tier_fair(container_picks, _anchor_container_cursor, + budget_left); + budget_left -= consume_tier_fair(other_picks, _anchor_other_cursor, + budget_left); + return selected; +} + +void ReferenceChainTracker::pushAtRiskStaticAnchor(jlong tag, u32 klass_id) { + if (_static_anchor_fifo_set.contains(tag)) { + return; + } + if (_static_anchor_fifo.size() >= STATIC_ANCHOR_FIFO_CAP) { + // Cap-full drop. With the per-class quota below this is legitimate + // saturation: a full 1024-entry FIFO necessarily holds >= 16 distinct + // under-quota classes (measured pod contrast, round 15: the cap was + // pinned by three classes' floods - klass 1 at 1396 pushes, klass 215 + // at 1063, klass 1733 at 988+ - so the wrapper's own pushes were + // dropped here and B' was dead for exactly the holder it exists for). + return; + } + auto count_it = _static_anchor_fifo_klass_counts.find(klass_id); + if (count_it != _static_anchor_fifo_klass_counts.end() && + count_it->second >= STATIC_ANCHOR_ATRISK_PER_KLASS_CAP) { + // Per-class quota drop: this class already holds its share of the + // lane, and its oldest entry drains within a few passes + // (STATIC_ANCHOR_FIFO_DRAIN=16/pass). A dropped push is retried by + // the feed's next event (the next static edge / next demotion), so + // nothing is lost - the entry just cannot crowd out every other + // class's repair. + _static_anchor_fifo_quota_drops++; + return; + } + if (count_it == _static_anchor_fifo_klass_counts.end()) { + count_it = _static_anchor_fifo_klass_counts.emplace(klass_id, 0U).first; + } + count_it->second++; + _static_anchor_fifo.push_back(AtRiskAnchor{tag, klass_id}); + _static_anchor_fifo_set.insert(tag); + _static_anchor_fifo_pushed++; + // TEMP DIAGNOSTIC (pod round 12/16): track which classes enter the + // at-risk FIFO and the cumulative quota drops (round 16: the drops + // should be the flood classes' pushes, while fifo_size stays well + // under the 1024 cap so the LEAK_BUFFER wrapper's pushes land). Remove + // once the wrapper question is answered. + TEST_LOG("ReferenceChainTracker::pushAtRiskStaticAnchor tag=%lld " + "klass_id=%u fifo_size=%zu quota_drops=%llu", + (long long)tag, klass_id, _static_anchor_fifo.size(), + (unsigned long long)_static_anchor_fifo_quota_drops); +} + +void ReferenceChainTracker::addToStaticAnchorIndex(jlong tag, + jlong own_class_tag, + u8 root_kind) { + if (root_kind != (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD && + root_kind != (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL) { + return; + } + // Dedup: a push at first admission and another at upgrade would double-add. + // The round-13 pod measurement sized the real anchor population at ~28k + // per search - a linear scan per add is O(n^2) over a search (observed + // cost ~2ms/pass amortized, wasted inside the sweep callback), so dedupe + // via a companion hash set (the PriorityExpandSet pattern, but unbounded: + // the set is cleared with the index on restart). + if (!_static_anchor_index_tags.insert(tag).second) { + return; + } + _static_anchor_index.push_back(tag); + _static_anchor_own_class_tags.push_back(own_class_tag); + // Fresh lane (round 15, _static_anchor_fresh_queue's own comment): a + // newly admitted anchor gets ONE first-look walk priority ahead of the + // fair cursors. Cap-drop from the front: the oldest fresh chances fall + // back to the fair tiers (their index positions are unchanged) - never + // lost, only deprioritized, same as a drain that outranks them. + _static_anchor_fresh_queue.push_back(tag); + if (_static_anchor_fresh_queue.size() > STATIC_ANCHOR_FRESH_CAP) { + _static_anchor_fresh_queue.pop_front(); + } +} + +bool ReferenceChainTracker::resolveContainerInterfaceTags( + jvmtiEnv *jvmti, JNIEnv *jni) { + if (_collection_iface_class_tag != 0 && _map_iface_class_tag != 0) { + return true; + } + // resolveLoadedClasses() tags every loaded class (including these + // bootstrap interfaces) with its class-tag-allocator tag (a NEGATIVE + // value - see nextClassTag()'s own comment) before any + // anchor can be admitted, but classify defensively: if an interface + // object somehow carries no tag yet, mint one via the shared allocator + // (same sequence resolveLoadedClasses() itself uses) so the comparison + // below is well-defined. + struct Iface { + const char *name; + jlong *tag_out; + }; + Iface ifaces[2] = {{"java/util/Collection", &_collection_iface_class_tag}, + {"java/util/Map", &_map_iface_class_tag}}; + for (const Iface &iface : ifaces) { + if (*iface.tag_out != 0) { + continue; + } + jclass local = jni->FindClass(iface.name); + if (jniExceptionCheck(jni) || local == nullptr) { + jni->ExceptionClear(); + return false; + } + jlong tag = 0; + bool ok = jvmti->GetTag(local, &tag) == JVMTI_ERROR_NONE; + if (ok && tag == 0) { + tag = nextClassTag(); + ok = jvmti->SetTag(local, tag) == JVMTI_ERROR_NONE; + } + if (ok && tag != 0) { + *iface.tag_out = tag; + } + jni->DeleteLocalRef(local); + if (!ok) { + return false; + } + } + return _collection_iface_class_tag != 0 && _map_iface_class_tag != 0; +} + +bool ReferenceChainTracker::classImplementsContainerOrMap(jvmtiEnv *jvmti, + JNIEnv *jni, + jclass klass) { + // BFS over the superclass chain + every visited class's interfaces, + // comparing GetTag() against the two cached interface class tags. + // Interface diamonds exist (e.g. both List and Set through Collection), + // so a visited set (by class tag) is required for termination; the + // visited set doubles as the memo the caller caches per class tag. + std::vector work; + std::unordered_set visited; + work.push_back(klass); + bool found = false; + int hops = 0; + while (!found && !work.empty() && hops++ < 64) { + jclass cur = work.back(); + work.pop_back(); + jlong cur_tag = 0; + if (jvmti->GetTag(cur, &cur_tag) != JVMTI_ERROR_NONE || cur_tag == 0) { + continue; + } + if (visited.count(cur_tag) > 0) { + continue; + } + visited.insert(cur_tag); + if (cur_tag == _collection_iface_class_tag || + cur_tag == _map_iface_class_tag) { + found = true; + // The popped `cur` ref never reaches the loop's bottom delete. + if (cur != klass) { + jni->DeleteLocalRef(cur); + } + break; + } + jclass super = jni->GetSuperclass(cur); + if (!jniExceptionCheck(jni) && super != nullptr) { + work.push_back(super); + } else { + jni->ExceptionClear(); + } + jint iface_count = 0; + jclass *ifaces = nullptr; + if (jvmti->GetImplementedInterfaces(cur, &iface_count, &ifaces) == + JVMTI_ERROR_NONE && + ifaces != nullptr) { + for (jint i = 0; i < iface_count; i++) { + if (ifaces[i] != nullptr) { + work.push_back(ifaces[i]); + } + } + jvmti->Deallocate((unsigned char *)ifaces); + } + // `cur` is either the caller-provided klass (caller-managed ref - NOT + // deleted here) or a ref this walk minted (GetSuperclass/ + // GetImplementedInterfaces locals, deleted immediately after use). + if (cur != klass) { + jni->DeleteLocalRef(cur); + } + } + // Single exit: every remaining ref minted into `work` (early hop-bound + // exit or the found-break) is deleted here rather than leaking locals + // for the process lifetime (the engine thread never detaches). + for (jclass r : work) { + if (r != nullptr && r != klass) { + jni->DeleteLocalRef(r); + } + } + return found; +} + +void ReferenceChainTracker::reconcileAnchorClassShapes(jvmtiEnv *jvmti, + JNIEnv *jni) { + if (jni == nullptr) { + return; + } + if (_static_anchor_own_class_tags.empty()) { + return; + } + // Collect up to ANCHOR_SHAPE_RECONCILE_BUDGET distinct class tags that + // appear in the anchor index but are not yet classified. + std::vector unknown; + unknown.reserve(8); + std::unordered_set seen; + for (jlong class_tag : _static_anchor_own_class_tags) { + if (class_tag == 0 || seen.count(class_tag) > 0 || + _class_shape_cache.count(class_tag) > 0) { + continue; + } + seen.insert(class_tag); + unknown.push_back(class_tag); + if ((int)unknown.size() >= ANCHOR_SHAPE_RECONCILE_BUDGET) { + break; + } + } + if (unknown.empty()) { + return; + } + if (!resolveContainerInterfaceTags(jvmti, jni)) { + return; + } + // The per-class interface walk below mints local refs (GetSuperclass, + // GetImplementedInterfaces) that are only deleted as the BFS pops + // them; bound the outstanding count explicitly rather than relying on + // the JVM to grow the local-ref table. + if (jni->EnsureLocalCapacity(512) < 0 || jniExceptionCheck(jni)) { + jni->ExceptionClear(); + return; + } + // One GetObjectsWithTags call resolves the class objects for the whole + // batch (class objects are tagged with their class tags). + jint obj_count = 0; + jobject *objs = nullptr; + jlong *obj_tags = nullptr; + if (jvmti->GetObjectsWithTags((jint)unknown.size(), unknown.data(), + &obj_count, &objs, &obj_tags) != + JVMTI_ERROR_NONE || + obj_count <= 0) { + if (objs != nullptr) { + jvmti->Deallocate((unsigned char *)objs); + } + if (obj_tags != nullptr) { + jvmti->Deallocate((unsigned char *)obj_tags); + } + return; + } + for (jint i = 0; i < obj_count; i++) { + jclass klass = (jclass)objs[i]; + jlong class_tag = obj_tags[i]; + // class tags are NEGATIVE (a namespace disjoint from positive + // frontier tags); 0 means the object was never tagged - skip only that. + if (class_tag == 0 || klass == nullptr) { + continue; + } + AnchorClassShape shape = classImplementsContainerOrMap(jvmti, jni, klass) + ? AnchorClassShape::CONTAINER + : AnchorClassShape::NON_CONTAINER; + _class_shape_cache[class_tag] = (u8)shape; + // TEMP DIAGNOSTIC (pod round 14): name newly classified container + // classes so the cohort is identifiable in logs. Remove once the + // pod verifies the arithmetic. + if (shape == AnchorClassShape::CONTAINER) { + char *sig = nullptr; + if (jvmti->GetClassSignature(klass, &sig, nullptr) == + JVMTI_ERROR_NONE && + sig != nullptr) { + TEST_LOG("ReferenceChainTracker::reconcileAnchorClassShapes " + "container class %s (class_tag=%lld)", + sig, (long long)class_tag); + jvmti->Deallocate((unsigned char *)sig); + } + } + } + jvmti->Deallocate((unsigned char *)objs); + jvmti->Deallocate((unsigned char *)obj_tags); +} + +int ReferenceChainTracker::drainStaticAnchorFifo(int max_count, + std::vector &out) { + if (max_count <= 0 || _static_anchor_fifo.empty()) { + return 0; + } + int drained = 0; + while (drained < max_count && !_static_anchor_fifo.empty()) { + AtRiskAnchor entry = _static_anchor_fifo.front(); + _static_anchor_fifo.pop_front(); + auto count_it = _static_anchor_fifo_klass_counts.find(entry.klass_id); + if (count_it != _static_anchor_fifo_klass_counts.end() && + --count_it->second == 0) { + // Erased at zero so the map is bounded by the FIFO's live contents + // (<= 1024 distinct classes), not by the search lifetime. + _static_anchor_fifo_klass_counts.erase(count_it); + } + out.push_back(entry); + drained++; + } + _static_anchor_fifo_set.rebuildFrom(_static_anchor_fifo); + return drained; +} + +void ReferenceChainTracker::requeueStaticAnchorFifoFront( + const std::vector &entries) { + if (entries.empty()) { + return; + } + // Reverse order onto the front preserves the tags' relative FIFO order + // (push_front of the LAST entry first leaves the FIRST entry at the + // deque's front). The tags were popped by this pass's + // drainStaticAnchorFifo() and nothing runs a sweep between that drain and + // here, so no tag can already be in the deque - rebuildFrom() would + // silently keep the FIRST slot for a duplicate, but there are none by + // construction. Re-increment each entry's class occupancy: drain + // decremented it, and the requeued entry occupies a FIFO slot again + // exactly as before the drain. + for (size_t i = entries.size(); i-- > 0;) { + _static_anchor_fifo_klass_counts[entries[i].klass_id]++; + _static_anchor_fifo.push_front(entries[i]); + } + _static_anchor_fifo_set.rebuildFrom(_static_anchor_fifo); +} + +void ReferenceChainTracker::walkStaticFieldAnchors( + jvmtiEnv *jvmti, JNIEnv *jni, const std::vector &anchor_tags, + int budget, int *edges_admitted, bool *truncated, bool *frontier_cap_hit, + u64 *safepoint_ticks, std::vector *unwalked) { + if (anchor_tags.empty()) { + return; + } + // Resolve all anchors with ONE GetObjectsWithTags call - the call's + // O(tag_map) cost is the dominant term on a large tag map (pod round 4: + // 14-41ms floor), exactly why expandFrontier() batches its own resolves. + jint resolved_count = 0; + jobject *objects = nullptr; + jlong *resolved_tags = nullptr; + if (jvmti->GetObjectsWithTags((jint)anchor_tags.size(), anchor_tags.data(), + &resolved_count, &objects, + &resolved_tags) != JVMTI_ERROR_NONE) { + return; + } + int walked = 0; + // First index the walk did NOT consume (breaks before an anchor's walk + // report i, breaks after report i+1; a completed loop keeps the + // resolved_count sentinel). The un-walked set is only meaningful at a + // break - the caller requeues FIFO-sourced anchors, collector-sourced + // ones keep their own cursor retention, and dead tags never resolved + // are intentionally absent (they must not be requeued anywhere). + jint first_unwalked = resolved_count; + for (jint i = 0; i < resolved_count; i++) { + FrontierEntry entry{}; + if (!_frontier->lookup(resolved_tags[i], &entry)) { + // Dead-or-stale between selection and here - skip; release machinery + // owns dead-entry cleanup, never here. + jni->DeleteLocalRef(objects[i]); + continue; + } + // TEMP DIAGNOSTIC (pod round 8: interception still zero over 16-hop + // walks of every selected anchor): name WHICH + // anchor objects are actually being walked, with their recorded chain + // shape, so a holder that never makes it into this tier (wrong + // root_kind / chain-attached / never-admitted) is distinguishable from + // one that gets walked without reaching the tagged chunks. + bool wrapper_trace = false; + { + char *sig = nullptr; + if (jni->GetObjectClass(objects[i]) != nullptr) { + jvmti->GetClassSignature(jni->GetObjectClass(objects[i]), &sig, + nullptr); + } + TEST_LOG("ReferenceChainTracker::walkStaticFieldAnchors anchor " + "tag=%lld class=%s klass_id=%u parent=%lld root_kind=%u state=%u " + "field_index=%d", + (long long)resolved_tags[i], sig != nullptr ? sig : "", + entry.referrer_klass, + (long long)entry.parent_tag, (unsigned)entry.root_kind, + (unsigned)entry.state, (int)entry.referrer_field_index); + // TEMP DIAGNOSTIC (pod round 12): the LEAK_BUFFER wrapper shape is a + // Collections.synchronizedList value held by a static field. The + // wrapper gets walked (observed rounds 10-11) yet never intercepts, + // so for wrapper-class anchors additionally name the HOLDER class + // (the class owning the admitting static field - resolves the + // "which synchronized list is this" question) and trace the walk's + // admission sequence (descendFromAnchor's diag_trace). TEMP: remove + // once the pod answers the wrapper question. + wrapper_trace = + sig != nullptr && + (strstr(sig, "UnmodifiableRandomAccessList") != nullptr || + strstr(sig, "SynchronizedRandomAccessList") != nullptr || + strstr(sig, "SynchronizedList") != nullptr); + if (wrapper_trace && entry.referrer_class_tag < 0) { + jlong holder_tag = entry.referrer_class_tag; + jint holder_count = 0; + jobject *holder_objs = nullptr; + jlong *holder_tags = nullptr; + if (jvmti->GetObjectsWithTags(1, &holder_tag, &holder_count, + &holder_objs, &holder_tags) == + JVMTI_ERROR_NONE && + holder_count > 0) { + char *holder_sig = nullptr; + jvmti->GetClassSignature((jclass)holder_objs[0], &holder_sig, + nullptr); + TEST_LOG("ReferenceChainTracker::walkStaticFieldAnchors wrapper " + "anchor tag=%lld holder_class=%s field_index=%d", + (long long)resolved_tags[i], + holder_sig != nullptr ? holder_sig : "", + (int)entry.referrer_field_index); + if (holder_sig != nullptr) { + jvmti->Deallocate((unsigned char *)holder_sig); + } + } + if (holder_objs != nullptr) { + jvmti->Deallocate((unsigned char *)holder_objs); + } + if (holder_tags != nullptr) { + jvmti->Deallocate((unsigned char *)holder_tags); + } + } + if (sig != nullptr) { + jvmti->Deallocate((unsigned char *)sig); + } + } + int remaining = budget - *edges_admitted; + if (remaining <= 0) { + jni->DeleteLocalRef(objects[i]); + first_unwalked = i; + break; + } + int edges_before = *edges_admitted; + descendFromAnchor(jvmti, jni, objects[i], resolved_tags[i], entry.depth, + /*anchor_descend_class_tag=*/0, remaining, edges_admitted, + truncated, frontier_cap_hit, safepoint_ticks, + wrapper_trace); + TEST_LOG("ReferenceChainTracker::walkStaticFieldAnchors anchor walk " + "outcome tag=%lld edges=%d truncated=%d cap_hit=%d", + (long long)resolved_tags[i], *edges_admitted - edges_before, + (int)*truncated, (int)*frontier_cap_hit); + walked++; + jni->DeleteLocalRef(objects[i]); + if (*truncated && !*frontier_cap_hit) { + // Budget/deadline exhausted mid-set - remaining anchors keep their + // rotation turn via the cursor next pass (the wrapping cursor already + // tolerates a short selection). + first_unwalked = i + 1; + break; + } + if (*frontier_cap_hit) { + first_unwalked = i + 1; + break; + } + } + if (unwalked != nullptr && first_unwalked < resolved_count) { + unwalked->insert(unwalked->end(), resolved_tags + first_unwalked, + resolved_tags + resolved_count); + } + jvmti->Deallocate((unsigned char *)objects); + jvmti->Deallocate((unsigned char *)resolved_tags); + TEST_LOG_SUMMARY("ReferenceChainTracker::walkStaticFieldAnchors selected=%zu " + "walked=%d edges_admitted=%d truncated=%d frontier_cap_hit=%d", + anchor_tags.size(), walked, *edges_admitted, (int)*truncated, + (int)*frontier_cap_hit); +} + +void ReferenceChainTracker::walkCandidateThreadLocals( + jvmtiEnv *jvmti, JNIEnv *jni, int budget, int *edges_admitted, + bool *truncated, bool *frontier_cap_hit, u64 *safepoint_ticks) { + if (_candidate_count <= 0) { + return; + } + // Flatten the per-slot qualifying-tid snapshot into (slot, tid) pairs, + // then walk up to THREAD_WALK_MAX_ANCHORS of them per pass, rotating via + // _thread_walk_anchor_cursor so every qualifying tid gets a turn within + // ceil(total / THREAD_WALK_MAX_ANCHORS) passes instead of always walking + // the first candidates' tids. + int slot[MAX_CANDIDATE_QUALIFYING_TIDS * MAX_LEAK_CANDIDATES_FROM_LT]; + jint tid[sizeof(slot) / sizeof(slot[0])]; + int total = 0; + for (int s = 0; s < _candidate_count; s++) { + for (int q = 0; q < _candidate_qualifying_tid_count[s]; q++) { + if (total >= (int)(sizeof(slot) / sizeof(slot[0]))) { + break; + } + slot[total] = s; + tid[total] = _candidate_qualifying_tids[s][q]; + total++; + } + } + if (total == 0) { + return; + } + jlong descend_class_tag = resolveThreadLocalMapClassTag(jvmti, jni); + if (_thread_walk_anchor_cursor < 0 || + _thread_walk_anchor_cursor >= total) { + _thread_walk_anchor_cursor = 0; + } + int walked = 0; + const int start = _thread_walk_anchor_cursor; + int i = start; + do { + jobject thread_obj; + { + MutexLocker ml(_thread_objects_lock); + auto it = _thread_objects.find(tid[i]); + if (it == _thread_objects.end()) { + thread_obj = nullptr; // Thread died/never registered - skip + } else { + thread_obj = it->second; + } + } + if (thread_obj != nullptr) { + // Anchor admission, idempotent across passes: a tag that still maps + // to a live entry is reused as-is (the Thread object is commonly + // root-attached by root enumeration already); a stale positive tag + // (search restart reissued tags from 1, releaseSearchTags() did not + // clear this object because release only touches FrontierTable + // entries) must be re-minted, otherwise the walk would parent new + // children onto a dead table slot or, worse, onto the entry a + // reissued tag now belongs to. + jlong anchor_tag = getTag(jvmti, thread_obj); + u32 anchor_depth = 0; + FrontierEntry anchor_entry{}; + if (anchor_tag > 0 && _frontier->lookup(anchor_tag, &anchor_entry)) { + anchor_depth = anchor_entry.depth; + } else { + jclass thread_class = jni->GetObjectClass(thread_obj); + jlong class_tag = 0; + jvmti->GetTag(thread_class, &class_tag); + u32 referrer_klass = classTags()->resolve(class_tag); + jlong fresh_tag = tagObject(jvmti, thread_obj); + if (fresh_tag != 0 && + _frontier->insert(fresh_tag, 0, referrer_klass, 0, + FrontierEntryState::FRONTIER, + (u8)JVMTI_HEAP_REFERENCE_THREAD, class_tag)) { + anchor_tag = fresh_tag; + } else { + anchor_tag = 0; + } + jni->DeleteLocalRef(thread_class); + } + if (anchor_tag != 0) { + int remaining = budget - *edges_admitted; + if (remaining > 0) { + descendFromAnchor(jvmti, jni, thread_obj, anchor_tag, anchor_depth, + descend_class_tag, remaining, edges_admitted, + truncated, frontier_cap_hit, safepoint_ticks); + walked++; + } + } + } + i = (i + 1) % total; + if (*frontier_cap_hit || walked >= THREAD_WALK_MAX_ANCHORS || + budget - *edges_admitted <= 0) { + break; + } + } while (i != start); + _thread_walk_anchor_cursor = i; + TEST_LOG_SUMMARY("ReferenceChainTracker::walkCandidateThreadLocals candidates=%d " + "tids=%d walked=%d edges_admitted=%d truncated=%d " + "frontier_cap_hit=%d", + _candidate_count, total, walked, *edges_admitted, (int)*truncated, + (int)*frontier_cap_hit); +} + +void ReferenceChainTracker::registerExistingThreads(jvmtiEnv *jvmti, + JNIEnv *jni) { + if (!_enabled || jvmti == nullptr || jni == nullptr) { + return; + } + // Register PRE-EXISTING threads into the tid -> Thread-object registry + // (see _thread_objects' own comment): Profiler::onThreadStart() only sees + // threads started after the recording began, and a leaking thread is + // typically alive since well before the profiler attached (observed live: + // ThreadLocalLeakScenario's leak thread - started before the profiler to + // seed the fixture - stayed unregistered, walkCandidateThreadLocals() + // reporting walked=0 while the only thing standing between the walk and + // the tagged chunks was the registry lookup). Same native-tid mapping the + // profiler's own thread-name refresh uses for these same pre-existing + // threads (Profiler::updateThreadName -> JVMThread::nativeThreadId, + // profiler.cpp). Best-effort per thread: a thread whose native tid cannot + // be resolved is simply skipped - a later ThreadEnd unregister for it is + // a harmless no-op lookup miss. Runs from Profiler::start() rather than + // this class's own start() so gtest binaries that call start() directly + // with partial mock JVMTI tables (no GetAllThreads slot) never reach the + // real-call path - only the real profiler lifecycle guarantees a fully + // populated JVMTI table here. + jint thread_count = 0; + jthread *thread_objects = nullptr; + if (jvmti->GetAllThreads(&thread_count, &thread_objects) != JVMTI_ERROR_NONE) { + return; + } + for (jint i = 0; i < thread_count; i++) { + jthread thread = thread_objects[i]; + if (thread == nullptr) { + continue; + } + int tid = JVMThread::nativeThreadId(jni, thread); + if (jni->ExceptionCheck()) { + jni->ExceptionClear(); + continue; + } + if (tid >= 0) { + registerThreadObject(jni, tid, thread); + } + jni->DeleteLocalRef(thread); + } + jvmti->Deallocate((unsigned char *)thread_objects); +} + +void ReferenceChainTracker::registerThreadObject(JNIEnv *jni, int tid, + jthread thread) { + if (!_enabled || jni == nullptr || thread == nullptr) { + return; + } + jobject ref = jni->NewGlobalRef(thread); + if (ref == nullptr) { + return; + } + MutexLocker ml(_thread_objects_lock); + auto it = _thread_objects.find(tid); + if (it != _thread_objects.end()) { + jni->DeleteGlobalRef(it->second); + } + _thread_objects[tid] = ref; +} + +void ReferenceChainTracker::unregisterThreadObject(JNIEnv *jni, int tid) { + if (jni == nullptr) { + return; + } + MutexLocker ml(_thread_objects_lock); + auto it = _thread_objects.find(tid); + if (it != _thread_objects.end()) { + jni->DeleteGlobalRef(it->second); + _thread_objects.erase(it); + } +} + +// --------------------------------------------------------------------------- +// Manual walk driver - IterateOverReachableObjects root/stack-ref enumeration +// plus expandFrontier()'s batched array-holder FollowReferences hop expansion. +// The only path driven by runPass() below. +// --------------------------------------------------------------------------- + +namespace { +// jvmtiHeapRootKind (IterateOverReachableObjects's root/stack-ref callbacks, +// ordinals 1-7) and jvmtiHeapReferenceKind (FrontierEntry::root_kind's own +// type, FollowReferences' callback, ordinals 8/21-27) are different, disjoint +// enums per the real jvmti.h - storing a raw jvmtiHeapRootKind value into +// root_kind unmodified would make flightRecorder.cpp's rootKindName() report +// "unknown" for every root-callback-attributed chain. Every jvmtiHeapRootKind +// value maps onto its jvmtiHeapReferenceKind namesake; there is no root-kind +// equivalent of STATIC_FIELD (that value only ever arises from +// heapReferenceCallback()'s own referrer-is-a-tagged-class case), so it is +// never produced here. +u8 translateHeapRootKind(jvmtiHeapRootKind root_kind) { + switch (root_kind) { + case JVMTI_HEAP_ROOT_JNI_GLOBAL: + return (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL; + case JVMTI_HEAP_ROOT_SYSTEM_CLASS: + return (u8)JVMTI_HEAP_REFERENCE_SYSTEM_CLASS; + case JVMTI_HEAP_ROOT_MONITOR: + return (u8)JVMTI_HEAP_REFERENCE_MONITOR; + case JVMTI_HEAP_ROOT_STACK_LOCAL: + return (u8)JVMTI_HEAP_REFERENCE_STACK_LOCAL; + case JVMTI_HEAP_ROOT_JNI_LOCAL: + return (u8)JVMTI_HEAP_REFERENCE_JNI_LOCAL; + case JVMTI_HEAP_ROOT_THREAD: + return (u8)JVMTI_HEAP_REFERENCE_THREAD; + case JVMTI_HEAP_ROOT_OTHER: + default: + return (u8)JVMTI_HEAP_REFERENCE_OTHER; + } +} + +} // namespace + +jvmtiIterationControl JNICALL ReferenceChainTracker::heapRootCallback( + jvmtiHeapRootKind root_kind, jlong class_tag, jlong size, jlong *tag_ptr, + void *user_data) { + PassContext *ctx = (PassContext *)user_data; + if (ctx->tracker->_abort_pass_requested.load(std::memory_order_relaxed)) { + ctx->truncated = true; + return JVMTI_ITERATION_ABORT; + } + if (ctx->truncated) { + return JVMTI_ITERATION_ABORT; + } + + u32 referrer_klass = ctx->tracker->classTags()->resolve(class_tag); + u8 translated_root_kind = translateHeapRootKind(root_kind); + AdmitResult result = ctx->tracker->admitObject( + ctx->frontier, ctx->hop_cap, ctx->budget, &ctx->edges_admitted, tag_ptr, + /*parent_tag=*/0, referrer_klass, /*depth=*/0, translated_root_kind, + class_tag); + switch (result) { + case AdmitResult::BUDGET_EXHAUSTED: + ctx->truncated = true; + return JVMTI_ITERATION_ABORT; + case AdmitResult::FRONTIER_CAP_HIT: + ctx->truncated = true; + ctx->frontier_cap_hit = true; + return JVMTI_ITERATION_ABORT; + case AdmitResult::ALREADY_ADMITTED: + // Rediscovery via a second heap root - either later in this same pass's + // root enumeration, or in a later pass re-enumerating roots entirely + // (design doc's durability tie-break / "opportunistic upgrade", "Fix for + // root-attribution staleness" point 1 and Phase 5 item 1): apply the + // same durability ranking admitObject() would have used on first + // discovery, upgrading root_kind if this root is more durable than + // whatever is currently recorded. Restricted to root-attached entries + // only (parent_tag == 0) - see maybeUpgradeRootAttachedRootKind()'s own + // comment for why. + ctx->tracker->maybeUpgradeRootAttachedRootKind(ctx->frontier, *tag_ptr, + translated_root_kind); + break; + default: + break; + } + return JVMTI_ITERATION_CONTINUE; +} + +jvmtiIterationControl JNICALL ReferenceChainTracker::stackRefCallback( + jvmtiHeapRootKind root_kind, jlong class_tag, jlong size, jlong *tag_ptr, + jlong thread_tag, jint depth, jmethodID method, jint slot, + void *user_data) { + // Stack-local/JNI-local roots carry thread/frame/slot detail JVMTI reports + // via this callback's richer shape, but FrontierEntry has nowhere to + // record it (depth/method/slot are not part of the record) - admission is + // otherwise identical to heapRootCallback() above, so this just forwards. + return heapRootCallback(root_kind, class_tag, size, tag_ptr, user_data); +} + +void ReferenceChainTracker::runPassManualWalk(jvmtiEnv *jvmti, JNIEnv *jni, + bool run_root_enum, + int root_enum_budget, + int expand_budget, + int *edges_admitted, + bool *truncated, + bool *frontier_cap_hit, + u64 *safepoint_ticks) { + assert(!t_inGCCallback && + "IterateOverReachableObjects/FollowReferences are JVMTI " + "Heap-category calls and must not be made from " + "GarbageCollectionStart/Finish"); + + *safepoint_ticks = 0; + + // Shared wall-clock ceiling for this whole call's static-field sweep, + // expandFrontier(), and rotation sub-calls below (see _pass_deadline_ns's + // own comment) - deliberately NOT applied to root/stack-ref enumeration + // itself, which is instead cadence-gated by run_root_enum/ + // ROOT_ENUM_MIN_INTERVAL_NS. + _pass_deadline_ns = _effective_pause_target_ms > 0 + ? OS::nanotime() + (u64)_effective_pause_target_ms * 1000000ULL + : 0; + + *edges_admitted = 0; + *truncated = false; + *frontier_cap_hit = false; + + // Reserve a slice for rotation up front, across all three tiers (see + // ROOT_KIND_ROTATION_BUDGET/LEAK_ACCUMULATION_ROTATION_BUDGET/ + // STALE_EXPANDED_ROTATION_BUDGET's own comments) so rotation still gets to + // run this pass even when ordinary work below spends everything else and + // truncates. Also capped at half of expand_budget: without that cap, a + // pacing-throttled pass (expand_budget down near MIN_EFFECTIVE_BUDGET) + // would hand rotation its full reservation and leave ordinary expansion + // with 0 - exactly the priority inversion this reservation exists to + // avoid, just for the other side. Capping at half means each side + // degrades proportionally as pacing throttles down, instead of either one + // hitting a hard 0. + int rotation_reserved_budget = std::min( + expand_budget / 2, ROOT_KIND_ROTATION_BUDGET + + LEAK_ACCUMULATION_ROTATION_BUDGET + + STALE_EXPANDED_ROTATION_BUDGET); + int budget = expand_budget - rotation_reserved_budget; + + // Root/stack-ref enumeration alone (unlike a root-seeded FollowReferences + // call on the fallback path) never discovers a root's own transitive + // children - IterateOverReachableObjects's root/stack-ref callbacks are + // given no oop, only a tag_ptr (see heapRootCallback()'s own comment) - so + // even when it runs this pass, the expandFrontier() call below is still + // needed to make any further progress. Gated behind run_root_enum (see + // ROOT_ENUM_MIN_INTERVAL_NS's own comment) since the call's fixed + // root-walk-and-dispatch cost is paid in full every time it runs, + // regardless of budget. + if (run_root_enum) { + PassContext ctx; + ctx.tracker = this; + ctx.frontier = _frontier; + ctx.hop_cap = _hop_cap; + ctx.budget = root_enum_budget; + ctx.edges_admitted = 0; + ctx.truncated = false; + ctx.frontier_cap_hit = false; + + u64 root_enum_start_ticks = TSC::ticks(); + jvmtiError root_err = jvmti->IterateOverReachableObjects( + heapRootCallback, stackRefCallback, /*object_ref_callback=*/nullptr, + &ctx); + *safepoint_ticks += TSC::ticks() - root_enum_start_ticks; + + // expand_budget is spent independently of root_enum_budget below (see + // ROOT_ENUM_MIN_INTERVAL_NS's own comment) - ctx.edges_admitted is + // written straight into *edges_admitted so the static-field/expand/ + // rotation budget math below is never shrunk by whatever root + // enumeration admitted. + *edges_admitted = ctx.edges_admitted; + _last_root_enum_ns = OS::nanotime(); + + if (root_err != JVMTI_ERROR_NONE) { + *truncated = true; + *frontier_cap_hit = false; + _root_enum_truncated_last_time = false; + return; + } + if (ctx.truncated) { + *truncated = true; + *frontier_cap_hit = ctx.frontier_cap_hit; + // Only a budget-exhausted truncation (not a frontier-cap-hit, which + // abandons the search outright) is grounds to retry root enumeration + // on the very next pass - see _root_enum_truncated_last_time's own + // comment. + _root_enum_truncated_last_time = !ctx.frontier_cap_hit; + return; + } + _root_enum_truncated_last_time = false; + } + + int expand_phase_edges_admitted = 0; + + // Candidate-scoped reach, prong 1: descend-walk the current candidates' + // qualifying threads' ThreadLocalMap subgraphs BEFORE any breadth-first + // work this pass - reaching the tagged instances under a thread-retained + // holder must not queue behind the ordinary backlog (see + // walkCandidateThreadLocals()'s own comment). Own deadline slice + // (per-sub-op reset, same as expand/rotation below) and its own budget + // draw from the ordinary expand slice; a truncation here behaves exactly + // like a truncated static sweep below (search stays RUNNING, remaining + // work resumes next pass). + if (_candidate_count > 0) { + int thread_walk_edges_admitted = 0; + bool thread_walk_truncated = false; + bool thread_walk_frontier_cap_hit = false; + // Give the thread walk its own fresh deadline so the root-enum walk + // above never eats its slice (per-sub-op reset rationale, see expand + // below). + _pass_deadline_ns = _effective_pause_target_ms > 0 + ? OS::nanotime() + + (u64)_effective_pause_target_ms * 1000000ULL + : 0; + walkCandidateThreadLocals(jvmti, jni, budget, &thread_walk_edges_admitted, + &thread_walk_truncated, + &thread_walk_frontier_cap_hit, safepoint_ticks); + expand_phase_edges_admitted += thread_walk_edges_admitted; + *edges_admitted += thread_walk_edges_admitted; + if (thread_walk_frontier_cap_hit) { + // Frontier-cap mid-thread-walk is the same search-abandonment grounds + // as anywhere else - do not spend more of this pass's budget. + *truncated = true; + *frontier_cap_hit = true; + return; + } + if (thread_walk_truncated) { + *truncated = true; + } + } + + // Static-field roots (SomeClass.staticField -> obj) are not reachable via + // IterateOverReachableObjects' root/stack-ref callbacks above - see + // admitStaticFieldRoots()'s own comment - so this pass would otherwise + // never discover an object retained only that way. Best-effort: failures + // here do not truncate the pass, they just mean this sweep found nothing + // new this time around. + // + // Only run the sweep when the loaded-class set has actually changed since + // the last time it completed (same guard shape resolveLoadedClasses() uses + // for its own GetLoadedClasses()-driven scan, and reusing the count that + // call already refreshed via resolveLoadedClasses() earlier this same + // runPass() - see _last_static_field_class_count's own comment). Without + // this, admitStaticFieldRoots() would re-run its own GetLoadedClasses() + // call and a FollowReferences over every loaded class - a stop-the-world + // HeapWalkOperation - on every pass, forever, at the per-second pass + // cadence, even once every loaded class's static fields have already been + // swept and no new class has appeared to introduce new ones. + TEST_LOG_SUMMARY("ReferenceChainTracker::runPassManualWalk static_sweep_gate " + "resolved=%d swept=%d cursor=%d", + _last_resolved_class_count, _last_static_field_class_count, + _static_field_sweep_cursor); + if (_last_resolved_class_count != _last_static_field_class_count) { + int static_field_edges_admitted = 0; + bool static_field_truncated = false; + bool static_field_frontier_cap_hit = false; + bool static_field_cycle_complete = false; + int static_field_budget = std::max(budget - expand_phase_edges_admitted, 0); + admitStaticFieldRoots(jvmti, jni, _hop_cap, static_field_budget, + &static_field_edges_admitted, &static_field_truncated, + &static_field_frontier_cap_hit, + &static_field_cycle_complete, safepoint_ticks); + expand_phase_edges_admitted += static_field_edges_admitted; + *edges_admitted += static_field_edges_admitted; + // TEMP DIAGNOSTIC (see doc/temp/ investigation notes): split out how much + // of this pass's budget/deadline the static-field sweep's current chunk + // alone consumed, and whether that chunk completed / the lap wrapped - + // to distinguish "a chunk never finishes within the per-pass deadline" + // from "chunks finish but rotation/expansion still can't find the + // target". + TEST_LOG_SUMMARY("ReferenceChainTracker::runPassManualWalk static_field_phase " + "edges_admitted=%d truncated=%d frontier_cap_hit=%d " + "cycle_complete=%d sweep_cursor=%d " + "last_resolved_class_count=%d last_static_field_class_count=%d", + static_field_edges_admitted, (int)static_field_truncated, + (int)static_field_frontier_cap_hit, (int)static_field_cycle_complete, + _static_field_sweep_cursor, _last_resolved_class_count, + _last_static_field_class_count); + if (static_field_truncated) { + *truncated = true; + *frontier_cap_hit = static_field_frontier_cap_hit; + if (static_field_frontier_cap_hit) { + // Frontier-size cap hit while admitting static-field roots is the + // same "grounds to ABANDON the whole search" outcome + // BUDGET_EXHAUSTED/FRONTIER_CAP_HIT handling above gives root + // enumeration - do not spend any more of this pass's budget on the + // ordinary expansion below. + return; + } + } + if (static_field_cycle_complete) { + // The chunk cursor completed a full lap over the loaded-class list + // with no chunk truncating along the way (possibly discovering + // nothing, if every static field seen was already ALREADY_ADMITTED) - + // remember the class count it covered so a later pass with no new + // classes can skip re-running the sweep entirely. Left unset if any + // chunk in the lap truncated (admitStaticFieldRoots() already started + // the next lap immediately in that case) so passes keep retrying + // instead of wrongly treating a still-incomplete sweep as done. + _last_static_field_class_count = _last_resolved_class_count; + } + } + + int expand_edges_admitted = 0; + bool expand_truncated = false; + bool expand_frontier_cap_hit = false; + int remaining_budget = std::max(budget - expand_phase_edges_admitted, 0); + // Give expand its own fresh deadline so the static-field sweep's + // FollowReferences calls don't eat expand's time. Each sub-operation + // (sweep, expand, rotation) gets its own _effective_pause_target_ms + // wall-clock budget — the cumulative rate is still capped by the pass + // cadence (effectiveCadenceNs). See q-safepoint-budget-model. + _pass_deadline_ns = _effective_pause_target_ms > 0 + ? OS::nanotime() + (u64)_effective_pause_target_ms * 1000000ULL + : 0; + expandFrontier(jvmti, jni, _hop_cap, remaining_budget, + &expand_edges_admitted, &expand_truncated, + &expand_frontier_cap_hit, safepoint_ticks); + expand_phase_edges_admitted += expand_edges_admitted; + *edges_admitted += expand_edges_admitted; + *truncated = *truncated || expand_truncated; + *frontier_cap_hit = expand_frontier_cap_hit; + TEST_LOG_SUMMARY("ReferenceChainTracker::runPassManualWalk expand_phase " + "edges_admitted=%d truncated=%d frontier_cap_hit=%d " + "remaining_budget=%d", + expand_edges_admitted, (int)expand_truncated, + (int)expand_frontier_cap_hit, remaining_budget); + + // Note: unlike a hard truncation during root/stack-ref enumeration or the + // static-field sweep above (which return early - the pass never even + // reached ordinary expansion), a truncated ordinary expansion does NOT + // skip rotation below: rotation runs on its own reserved slice of budget + // (see rotation_reserved_budget's own comment above) precisely because + // ordinary expansion truncates on nearly every pass under a sustained + // fast-growing backlog, and that is exactly the situation - a mutable + // field reassigned out from under an already-EXPANDED entry - rotation + // exists to correct. + + // Three-tier bounded rotating re-expansion (design doc's closing section, + // Phase 5 item 3, extended - see each collector's own comment for why it + // exists as its own tier): re-walk a bounded, rotating subset of + // already-EXPANDED entries so mutations to an already-expanded object's + // fields - a durable root discovered elsewhere for a stale attribution, or + // a mutable collection field reassigned out from under a container object + // - get a chance to be observed on a later pass. Runs after the ordinary + // expansion above so it only ever spends whatever budget that left + // unused, plus its own reserved slice. Ordered highest-value/cheapest + // first: root-attribution re-verification (small, bounded population), + // then the leak-accumulation growth-catching tier (also small, bounded, + // and the one that actually targets this leak shape), then the + // unprioritized whole-table fallback last. + std::vector rotation_tags = + collectStaleRootKindEntriesForRotation(ROOT_KIND_ROTATION_BUDGET); + std::vector leak_accumulation_tags = + collectLeakAccumulationCandidatesForRotation( + LEAK_ACCUMULATION_ROTATION_BUDGET); + // Also re-walk a bounded, rotating subset of EXPANDED entries regardless + // of root attribution: a mutable field reassigned since an + // object's one-time expansion - e.g. HashMap.table on resize - is + // otherwise never observed again, silently orphaning everything only + // reachable through the field's current value. See + // collectStaleExpandedEntriesForRotation()'s own comment. + std::vector stale_expanded_tags = + collectStaleExpandedEntriesForRotation(STALE_EXPANDED_ROTATION_BUDGET); + // Candidate-scoped reach, prong 2: root-attached static holders are + // descend-walked directly (see collectStaticFieldAnchorsForRotation()/ + // walkStaticFieldAnchors()'s own comments) - not pushed onto the priority + // lane, so they are independent of the queue tiers above. The at-risk + // FIFO (B', _static_anchor_fifo's declaration comment) is drained BEHIND + // the collector's selection: the collector's small root-attached cohort + // walks first (a single-referrer static holder like the LEAK_BUFFER + // wrapper lives there - it never demotes, so it can never be at-risk), + // and a truncated pass falls on the FIFO's at-risk suffix, which the + // requeue path below protects - observed on the pod (round 10) that the + // reverse order starved the collector's picks in ~60% of passes + // (walked=6-16 of selected=20) against a cap-pinned at-risk flood. + // Classify any not-yet-shaped anchor classes (up to + // ANCHOR_SHAPE_RECONCILE_BUDGET per pass, one GOTW call) BEFORE the + // collector runs, so this pass's tiering sees as much of the container + // cohort as possible. Runs on the engine thread with JNI available, + // outside heap callbacks. + reconcileAnchorClassShapes(jvmti, jni); + std::vector static_anchor_tags = + collectStaticFieldAnchorsForRotation(STATIC_ANCHOR_ROTATION_BUDGET); + std::vector static_anchor_fifo_drained; + int static_anchor_fifo_drained_count = + drainStaticAnchorFifo(STATIC_ANCHOR_FIFO_DRAIN, static_anchor_fifo_drained); + for (const AtRiskAnchor &at_risk : static_anchor_fifo_drained) { + static_anchor_tags.push_back(at_risk.tag); + } + // TEMP DIAGNOSTIC (see static_field_phase log above). The fifo fields are + // the round-10 verification channel for B': fifo_pushed_total sizes the + // at-risk population (the design's drain-rate argument was inferred, not + // measured). Round 16: fifo_quota_drops_total is the flood-classes' dropped + // pushes (should climb steadily on the pod while fifo_size stays far + // below the 1024 cap, so the wrapper's pushes land), and self_edge_skips + // is FrontierTable's self-edge guard count (should climb every pass that + // walks a Synchronized* holder - the LEAK_BUFFER wrapper's mutex==this + // edge proving the guard fires on the real object). + TEST_LOG_SUMMARY("ReferenceChainTracker::runPassManualWalk rotation_candidates " + "root_kind_tags=%zu leak_accumulation_tags=%zu stale_expanded_tags=%zu " + "static_anchor_tags=%zu static_anchor_fifo_size=%zu " + "static_anchor_fifo_drained=%d static_anchor_fifo_pushed_total=%llu " + "static_anchor_fifo_quota_drops_total=%llu " + "self_edge_skips=%llu watched_leak_klass_count=%d " + "leak_signatures=%zu leak_parents=%zu", + rotation_tags.size(), leak_accumulation_tags.size(), + stale_expanded_tags.size(), static_anchor_tags.size(), + _static_anchor_fifo.size(), static_anchor_fifo_drained_count, + (unsigned long long)_static_anchor_fifo_pushed, + (unsigned long long)_static_anchor_fifo_quota_drops, + (unsigned long long)_frontier->selfEdgeGuardSkips(), + _watched_leak_klass_count, + _leak_signature_totals.size(), _leak_parent_fanout.size()); + if (rotation_tags.empty() && leak_accumulation_tags.empty() && + stale_expanded_tags.empty() && static_anchor_tags.empty()) { + return; + } + // rotation_reserved_budget + max(budget - expand_phase_edges_admitted, 0) is + // exactly expand_budget - expand_phase_edges_admitted: budget already IS + // expand_budget - rotation_reserved_budget (above), and expand_phase_edges_ + // admitted can never exceed budget (the static-field sweep and ordinary + // expandFrontier() calls above are both capped to budget-derived slices), + // so the max() is never actually needed to avoid going negative. Folding + // rotation_reserved_budget back into expand_budget here - rather than + // subtracting it out and then adding it back - says directly what this + // value is: whatever of the whole pass's budget the phases above didn't + // spend. + int rotation_budget = expand_budget - expand_phase_edges_admitted; + int rotation_edges_admitted = 0; + bool rotation_truncated = false; + // Give rotation its own fresh deadline, same as expand above. + _pass_deadline_ns = _effective_pause_target_ms > 0 + ? OS::nanotime() + (u64)_effective_pause_target_ms * 1000000ULL + : 0; + bool rotation_frontier_cap_hit = false; + // Prong 2 static-anchor descend walks run FIRST inside rotation's slice: + // they are the highest-value rotation work (bounded, targeted, and the + // only rotation tier that can reach a collection-shaped static holder's + // internals in one pass), and their edges draw down the same rotation + // budget the queue-tier batch below uses - a pass whose anchor walks admit + // the holder's whole internal structure needs less one-hop rotation work, + // not more. + if (!static_anchor_tags.empty() && rotation_budget > 0) { + int static_anchor_edges_admitted = 0; + bool static_anchor_truncated = false; + bool static_anchor_frontier_cap_hit = false; + std::vector static_anchor_unwalked; + walkStaticFieldAnchors(jvmti, jni, static_anchor_tags, rotation_budget, + &static_anchor_edges_admitted, + &static_anchor_truncated, + &static_anchor_frontier_cap_hit, safepoint_ticks, + &static_anchor_unwalked); + rotation_edges_admitted += static_anchor_edges_admitted; + rotation_budget -= static_anchor_edges_admitted; + *truncated = *truncated || static_anchor_truncated; + // B' requeue: resolved-but-unwalked anchors that came from this pass's + // FIFO drain go back to the FIFO front, order-preserving, so a pass + // whose budget died mid-batch walks them first next pass instead of + // waiting for the next sweep lap's re-push. Collector-sourced un-walked + // anchors are deliberately dropped from this - their retention is the + // collector cursor's own. Frontier lookups filter entries that died + // between selection and the walk (requeueing a dead tag would only + // re-drop it). The scan is drained x unwalked (<= 16 x <= 20), well + // under the small-set linear-scan cutoff. Entries keep their klass so + // the requeue re-increments the per-class occupancy exactly. + if (!static_anchor_unwalked.empty() && + !static_anchor_fifo_drained.empty()) { + std::vector static_anchor_requeue; + for (jlong tag : static_anchor_unwalked) { + FrontierEntry entry{}; + if (!_frontier->lookup(tag, &entry)) { + continue; + } + for (const AtRiskAnchor &at_risk : static_anchor_fifo_drained) { + if (tag == at_risk.tag) { + static_anchor_requeue.push_back(at_risk); + break; + } + } + } + if (!static_anchor_requeue.empty()) { + requeueStaticAnchorFifoFront(static_anchor_requeue); + } + } + if (static_anchor_frontier_cap_hit) { + *frontier_cap_hit = true; + return; + } + } + // expandFrontier() SETS (does not add into) its edges output - see its + // entry - so the anchor walks' edges are kept in a separate counter and + // summed here. + int queue_tier_edges_admitted = 0; + expandFrontier(jvmti, jni, _hop_cap, rotation_budget, + &queue_tier_edges_admitted, &rotation_truncated, + &rotation_frontier_cap_hit, safepoint_ticks); + rotation_edges_admitted += queue_tier_edges_admitted; + *edges_admitted += rotation_edges_admitted; + // TEMP DIAGNOSTIC (see static_field_phase log above). + TEST_LOG_SUMMARY("ReferenceChainTracker::runPassManualWalk rotation_phase " + "edges_admitted=%d truncated=%d frontier_cap_hit=%d " + "rotation_budget=%d", + rotation_edges_admitted, (int)rotation_truncated, + (int)rotation_frontier_cap_hit, rotation_budget); + // OR, not overwrite: the ordinary expand phase above may have already set + // these to true (real truncation/cap-hit left in _pending_expand), and a + // rotation batch that happens to finish cleanly must not erase that - + // has_pending_frontier (runPass()) and the FRONTIER_CAP abandon check both + // read these as "did any of this pass's sub-phases truncate/cap-hit", not + // just the last one that ran. + *truncated = *truncated || rotation_truncated; + *frontier_cap_hit = *frontier_cap_hit || rotation_frontier_cap_hit; +} + +// --------------------------------------------------------------------------- +// Incremental resumption across passes. +// --------------------------------------------------------------------------- + +void ReferenceChainTracker::markAllFrontierExpanded() { + while (!_priority_expand.empty()) { + _frontier->markExpanded(_priority_expand.front()); + _priority_expand.pop_front(); + } + _priority_expand_set.clear(); + while (!_pending_expand.empty()) { + _frontier->markExpanded(_pending_expand.front()); + _pending_expand.pop_front(); + } +} + +void ReferenceChainTracker::expandFrontier(jvmtiEnv *jvmti, JNIEnv *jni, + int hop_cap, int budget, + int *edges_admitted, + bool *truncated, + bool *frontier_cap_hit, + u64 *safepoint_ticks) { + assert(!t_inGCCallback && + "GetObjectsWithTags/FollowReferences are JVMTI Heap-category calls " + "and must not be made from GarbageCollectionStart/Finish"); + + PassContext ctx; + ctx.tracker = this; + ctx.frontier = _frontier; + ctx.hop_cap = hop_cap; + ctx.budget = budget; + ctx.edges_admitted = 0; + ctx.truncated = false; + ctx.frontier_cap_hit = false; + + // ARRAY-HOLDER BATCHING: expand a whole batch of boundary objects with ONE + // FollowReferences(initial_object=holder_array) call per BFS level, instead + // of one FollowReferences PER frontier entry. batch_tags gates + // heapReferenceCallback() to a single hop (see its own comment). This is + // the unconditional default expansion path (runPass()'s only non-fallback + // walk), not a prototype relative to anything else still in the codebase. + std::unordered_set batch_tags; + ctx.batch_tags = &batch_tags; + + jvmtiHeapCallbacks callbacks; + memset(&callbacks, 0, sizeof(callbacks)); + callbacks.heap_reference_callback = heapReferenceCallback; + + // java/lang/Object element type for the transient frontier-holder array. + // Cached across calls on this same (attached) JNIEnv rather than re-resolved + // via a fresh FindClass() every call - expandFrontier() runs roughly once + // per BFS-thread wake for the tracker's lifetime, and the class never + // changes, so a per-pass class-loader lookup is unnecessary churn. Without + // a JNIEnv (some test seams) the array-holder path cannot run; a JNIEnv + // change (fresh attach) invalidates the cache since the previous call's + // local ref is only guaranteed valid for that attach's lifetime. + if (jni != nullptr && _cached_object_class == nullptr) { + jclass local = jni->FindClass("java/lang/Object"); + if (!jniExceptionCheck(jni) && local != nullptr) { + _cached_object_class = (jclass)jni->NewGlobalRef(local); + } + if (local != nullptr) { + jni->DeleteLocalRef(local); + } + } + jclass object_class = _cached_object_class; + + bool progress = true; + // FAIR-SHARE DRAIN: alternate batches between _priority_expand and + // _pending_expand whenever both are non-empty (priority still takes the + // first batch of each call). The original strict priority-first drain + // starved the ordinary backlog whenever rotation's inflow + // (~STALE_EXPANDED_ROTATION_BUDGET+ROOT_KIND_ROTATION_BUDGET per pass) + // exceeded the deadline-bounded drain (~2-3 GetObjectsWithTags calls per + // phase) - observed live on hotdog: _priority_expand grew 39k->103k in 20 + // minutes while the BFS's own _pending_expand (66k entries) was never + // drained by a single batch, freezing all new-territory crawl. + // Alternation guarantees the ordinary frontier at least every other + // batch regardless of queue depths; an empty lane falls back to the + // other one. The toggle is the _expand_lane_prefer_priority MEMBER + // (not a local of this invocation): the phase deadlines bound a typical + // invocation to a single batch, so a per-invocation reset made priority + // win every invocation - observed live on hotdog with round 3's build, + // where _pending_expand GREW 109k->113k across 260 passes while every + // gotw call drained the priority lane's stale re-walks (edges=0). + while (!ctx.truncated && progress && object_class != nullptr) { + // Wall-clock deadline check per iteration: GetObjectsWithTags runs OUTSIDE + // any FollowReferences callback, so heapReferenceCallback()'s amortized + // deadline check never sees its cost. Measured live on hotdog: a + // collapsed batch (batch=2) let ~1400 unchecked GetObjectsWithTags calls + // (~20ms each) run in one expand phase, spending 10.4s of CPU and ~30s of + // wall time in a single pass. Checking here bounds each phase to + // _effective_pause_target_ms regardless of batch health. + if (_pass_deadline_ns != 0 && OS::nanotime() >= _pass_deadline_ns) { + ctx.truncated = true; + break; + } + progress = false; + + // Alternate lanes (see FAIR-SHARE DRAIN above); priority still goes + // first so a rotation-selected parent's re-discovery keeps its + // head-of-queue property, but no lane can monopolize the drain. + bool from_priority; + if (_priority_expand.empty()) { + from_priority = false; + } else if (_pending_expand.empty()) { + from_priority = true; + } else { + from_priority = _expand_lane_prefer_priority; + _expand_lane_prefer_priority = !_expand_lane_prefer_priority; + } + std::deque &source = + from_priority ? _priority_expand : _pending_expand; + ctx.admit_priority = from_priority; + if (source.empty()) { + break; // nothing pending in either lane + } + + // SELF-CALIBRATING ADAPTIVE BATCH SIZE for GetObjectsWithTags. + // GetObjectsWithTags iterates the whole JVMTI tag map per call, so its + // cost has a batch-independent floor that grows with the frontier + // (measured live: ~20ms at a 225k-entry map regardless of batch_size). + // Calibrating batch_size from a per-tag EMA collapses in that regime + // (small batch inflates per-tag cost, which shrinks the batch further — + // observed live driving batch from ~400 to 2). Instead, AIMD directly on + // batch size against the measured per-CALL time vs GOTW_CPU_BUDGET_NS — + // see _gotw_batch_size's own comment. + // + // Still capped at `budget` and `_budget` for the original reasons + // (first-pass budget can be far larger than the backlog; a single + // huge batch risks JNI local-capacity/OOM with zero progress). + size_t gotw_batch_size = + _gotw_batch_size != 0 ? _gotw_batch_size : GOTW_INITIAL_BATCH_SIZE; + size_t batch_size = std::min( + source.size(), + std::min((size_t)std::max(std::min(budget, _budget), 1), + gotw_batch_size)); + std::vector candidate_tags(source.begin(), + source.begin() + batch_size); + + // Resolve this batch's live boundary objects. GetObjectsWithTags iterates + // the whole tag map, but does so under a no-safepoint mutex on this + // (Java) thread - it is NOT a stop-the-world VM operation, unlike the + // FollowReferences below (jvmtiTagMap.cpp: get_objects_with_tags takes + // Mutex::_no_safepoint_check_flag and calls entry_iterate directly, + // whereas follow_references does VMThread::execute()). + jint resolved_count = 0; + jobject *resolved_objects = nullptr; + jlong *resolved_tags = nullptr; + u64 gotw_start_ns = OS::nanotime(); + jvmtiError resolve_err = jvmti->GetObjectsWithTags( + (jint)candidate_tags.size(), candidate_tags.data(), &resolved_count, + &resolved_objects, &resolved_tags); + u64 gotw_elapsed_ns = OS::nanotime() - gotw_start_ns; + // Self-calibrate (PROPORTIONAL batch control): update the EMA of + // PER-CALL elapsed time, then scale the batch so ONE call fills the + // remaining wall-clock window. This replaces the earlier per-call AIMD + // (fixed budget, halve/add-64): measured live on hotdog, the tag map + // grew until the per-call floor alone (~27ms at a 243k-entry map) + // exceeded the fixed 25ms budget, so AIMD ratcheted to GOTW_MIN_BATCH + // and stayed there (batch=8 forever) even though batch=72 cost only + // +36% for 9x the objects - the floor-dominated regime in which a + // BIGGER batch is the right move, and only a proportion against the + // remaining deadline can see that. See _gotw_batch_size's own + // comment for the full history (incl. the earlier per-tag collapse). + if (batch_size > 0 && gotw_elapsed_ns > 0) { + if (_gotw_ema_call_ns == 0) { + _gotw_ema_call_ns = gotw_elapsed_ns; + } else { + _gotw_ema_call_ns = _gotw_ema_call_ns * 4 / 5 + gotw_elapsed_ns / 5; + } + u64 now_ns = OS::nanotime(); + u64 window_ns = + gotwWindowNs( + _pass_deadline_ns != 0 && _pass_deadline_ns > now_ns + ? _pass_deadline_ns - now_ns + : 0, + source.size()); + // window_ns / ema_call_ns == how many such calls fit the window; + // scaling the CURRENT calibration batch by that ratio sizes the next + // call to consume the whole window in one go. Extrapolate from the + // stored _gotw_batch_size (the intended size), not from batch_size: + // batch_size is capped by the lane depth (min(source.size(), ...)), + // and a shallow lane would calibrate the stored size toward its own + // depth even though the stored size is what the next deep-lane call + // will use. Integer division biases the next batch slightly small - + // safe (an under-filled window just runs a second call; an + // over-filled one overruns the deadline). + size_t calib_batch = + _gotw_batch_size != 0 ? _gotw_batch_size : GOTW_INITIAL_BATCH_SIZE; + size_t next_batch = (size_t)((u64)calib_batch * window_ns / + std::max(_gotw_ema_call_ns, 1ULL)); + _gotw_batch_size = std::min(std::max(next_batch, GOTW_MIN_BATCH), + GOTW_MAX_BATCH); + } + if (resolve_err != JVMTI_ERROR_NONE) { + ctx.truncated = true; + break; + } + + // TEMP DIAGNOSTIC: verify adaptive batch_size is working + TEST_LOG("ReferenceChainTracker::expandFrontier gotw " + "batch_size=%zu resolved=%d edges=%d gotw_ms=%llu ema_call_ms=%llu " + "next_batch=%llu", + batch_size, resolved_count, ctx.edges_admitted, + (unsigned long long)(gotw_elapsed_ns / 1000000ULL), + (unsigned long long)(_gotw_ema_call_ns / 1000000ULL), + (unsigned long long)(_gotw_batch_size != 0 ? _gotw_batch_size + : GOTW_INITIAL_BATCH_SIZE)); + + std::unordered_map live; + for (jint i = 0; i < resolved_count; i++) { + live[resolved_tags[i]] = resolved_objects[i]; + } + + // Build the frontier-holder array from the live boundary objects and + // record their tags so heapReferenceCallback() descends into exactly + // these (one hop). + batch_tags.clear(); + jobjectArray holder = nullptr; + if (resolved_count > 0) { + jint capacity_err = jni->EnsureLocalCapacity(resolved_count + 16); + if (capacity_err < 0 || jniExceptionCheck(jni)) { + // Could not guarantee local-ref headroom for this batch - treat like + // any other batch-level failure below (JVMTI error / OOM building the + // holder array): retry this batch on a later pass rather than + // proceeding into NewObjectArray with no capacity guarantee. + ctx.truncated = true; + } else { + holder = jni->NewObjectArray(resolved_count, object_class, nullptr); + if (jniExceptionCheck(jni)) { + // OutOfMemoryError building the holder array (or any other + // exception NewObjectArray raised) left `holder` null; make sure + // the pending exception does not survive into the next JNI call + // below or the next expandFrontier() invocation on this same + // long-lived BFS-thread JNIEnv (JNI spec: undefined behavior with + // a pending exception across ordinary JNI calls). + holder = nullptr; + } + if (holder != nullptr) { + for (jint i = 0; i < resolved_count; i++) { + jni->SetObjectArrayElement(holder, i, resolved_objects[i]); + if (jniExceptionCheck(jni)) { + // e.g. an array-store-class failure. Abort building this + // batch's holder rather than handing a partially-populated + // array (with a just-cleared pending exception) to + // FollowReferences. + ctx.truncated = true; + break; + } + batch_tags.insert(resolved_tags[i]); + } + } + if (holder == nullptr) { + // NewObjectArray failed (OOM/local-ref exhaustion) - the + // FollowReferences call below (which would have discovered this + // batch's children) never runs. Falling through to the + // mark-EXPANDED-and-dequeue path further down would silently and + // permanently drop these still-undiscovered children, so this must + // be treated exactly like a failed FollowReferences/JVMTI call: + // retry the batch on a later pass instead. + ctx.truncated = true; + } else if (!ctx.truncated) { + // A single FollowReferences over the holder array expands this whole + // BFS level in one stop-the-world HeapWalkOperation (instead of one + // per frontier entry). initial_object=holder means the traversal + // starts from the array only (never enumerates roots / the whole + // heap); heapReferenceCallback() returns "descend" for the array's + // elements (the boundary objects, in batch_tags) and "no descend" for + // their children, so exactly one hop past the boundary is explored. + ctx._last_visited_batch_tag = 0; // reset rolling cursor + u64 follow_start_ticks = TSC::ticks(); + jvmtiError follow_err = + jvmti->FollowReferences(0, nullptr, holder, &callbacks, &ctx); + *safepoint_ticks += TSC::ticks() - follow_start_ticks; + if (follow_err != JVMTI_ERROR_NONE) { + ctx.truncated = true; + } + } + } + } + + if (!ctx.truncated) { + // The whole batch had all its direct children admitted this level: + // dead entries are pruned, live ones are marked EXPANDED, and all are + // popped off the front. New children were appended to the back by + // admitObject() and become the next level's batch. + for (jlong tag : candidate_tags) { + if (live.find(tag) == live.end()) { + _frontier->clear(tag); + } else { + _frontier->markExpanded(tag); + } + source.pop_front(); + } + progress = true; + } else if (ctx._last_visited_batch_tag != 0) { + // ROLLING RESUME: FollowReferences truncated mid-batch, but we know + // which batch entry was being visited when it stopped (tracked by + // the callback's batch_tags descent-gate). Pop entries that were + // fully processed BEFORE that entry (mark live ones EXPANDED, clear + // dead ones), and leave the partially-processed entry and everything + // after it at the front of the source queue for the next pass to + // retry. Same resumable-cursor pattern as admitStaticFieldRoots()'s + // sweep cursor — avoids re-walking already-expanded entries (and + // re-paying GetObjectsWithTags's O(tag_map × batch) cost for them) + // on every retry. + // + // The partially-visited entry (at _last_visited_batch_tag) stays: + // some of its children may have been admitted before the truncation, + // and the rest are discovered on retry (admitObject is idempotent — + // already-admitted children return ALREADY_ADMITTED). + for (size_t i = 0; i < candidate_tags.size(); i++) { + if (candidate_tags[i] == ctx._last_visited_batch_tag) { + break; // stop at the partially-visited entry + } + jlong tag = candidate_tags[i]; + if (live.find(tag) == live.end()) { + _frontier->clear(tag); + } else { + _frontier->markExpanded(tag); + } + source.pop_front(); + } + } + // else truncated with no batch entry visited (e.g. GetObjectsWithTags + // error, holder allocation failure, or truncation before the first + // batch entry was reached): leave the entire batch at the front of the + // source queue for a later pass to retry, same as before. + + if (from_priority) { + // This batch popped entries off _priority_expand's front (or, on + // truncation, was left untouched) - re-derive the membership index + // from the deque's current contents either way so + // isQueuedForRotation() stays exact for the rotation collectors that + // run later in this same pass. A full rebuild is <= + // PRIORITY_EXPAND_CAP inserts, a few microseconds against the ~20ms + // GetObjectsWithTags call this batch already paid + // (PriorityExpandSet's own comment, referenceChains.h). + _priority_expand_set.rebuildFrom(_priority_expand); + } + + if (holder != nullptr) { + jni->DeleteLocalRef(holder); + } + if (jni != nullptr) { + for (jint i = 0; i < resolved_count; i++) { + jni->DeleteLocalRef(resolved_objects[i]); + } + } + if (resolved_objects != nullptr) { + jvmti->Deallocate((unsigned char *)resolved_objects); + } + if (resolved_tags != nullptr) { + jvmti->Deallocate((unsigned char *)resolved_tags); + } + } + + // object_class is NOT deleted here - it is now cached in + // _cached_object_class and reused across calls on this same JNIEnv (see + // above), not a per-call local ref. + + if (!ctx.truncated && jni != nullptr && object_class == nullptr && + (!_pending_expand.empty() || !_priority_expand.empty())) { + // FindClass("java/lang/Object") failed for this (attached) JNIEnv, so + // the batching loop above never ran even though pending frontier work + // remains. Report truncated rather than leaving *truncated false: the + // caller (runPassManualWalk()/runPass()) treats false as "no pending + // frontier work", which would falsely mark the search + // SearchState::COMPLETED instead of retrying - directly contradicting + // this subsystem's documented "no silent truncation" requirement (see + // SearchAbandonReason's header comment). + ctx.truncated = true; + } + + *edges_admitted = ctx.edges_admitted; + *truncated = ctx.truncated; + *frontier_cap_hit = ctx.frontier_cap_hit; +} + +void ReferenceChainTracker::admitStaticFieldRoots(jvmtiEnv *jvmti, JNIEnv *jni, + int hop_cap, int budget, + int *edges_admitted, + bool *truncated, + bool *frontier_cap_hit, + bool *cycle_complete, + u64 *safepoint_ticks) { + assert(!t_inGCCallback && + "GetLoadedClasses/FollowReferences are JVMTI Heap-category calls " + "and must not be made from GarbageCollectionStart/Finish"); + *edges_admitted = 0; + *truncated = false; + *frontier_cap_hit = false; + *cycle_complete = false; + + if (jni == nullptr) { + // No JNIEnv to build the holder array on (some test seams) - see + // expandFrontier()'s own identical guard. Best-effort sweep: nothing + // discovered this call, not this pass's own truncation. + return; + } + + jint class_count = 0; + jclass *classes = nullptr; + jvmtiError classes_err = jvmti->GetLoadedClasses(&class_count, &classes); + if (classes_err != JVMTI_ERROR_NONE) { + return; + } + if (class_count <= 0) { + if (classes != nullptr) { + jvmti->Deallocate((unsigned char *)classes); + } + return; + } + + // GetLoadedClasses() gives no ordering guarantee across separate calls, so + // the cursor below is only meaningful as an index into THIS call's array - + // reprioritize it every call rather than trying to cache an ordering. + // Application/library classes (any non-bootstrap classloader) are moved to + // the front so a chunked sweep (below) reaches a likely leak source within + // its first several chunks instead of only after every JDK/platform class + // (typically the majority of a real JVM's loaded-class count) has been + // swept first. In-place two-way partition, no extra allocation. + jint app_boundary = 0; + for (jint i = 0; i < class_count; i++) { + jobject loader = nullptr; + jvmtiError loader_err = jvmti->GetClassLoader(classes[i], &loader); + bool is_app_class = (loader_err == JVMTI_ERROR_NONE) && (loader != nullptr); + if (loader != nullptr) { + jni->DeleteLocalRef(loader); + } + if (is_app_class) { + if (i != app_boundary) { + std::swap(classes[i], classes[app_boundary]); + } + app_boundary++; + } + } + + if (_static_field_sweep_cursor >= class_count) { + // Loaded-class count shrank since the last chunk (classes unloaded) - + // restart the lap rather than reading out of range. + _static_field_sweep_cursor = 0; + _static_field_sweep_cycle_truncated = false; + } + jint chunk_start = _static_field_sweep_cursor; + jint chunk_end = + std::min(chunk_start + STATIC_FIELD_SWEEP_CHUNK_CLASSES, class_count); + jint chunk_count = chunk_end - chunk_start; + + // Same java/lang/Object element-type cache expandFrontier() uses for its + // own frontier-holder array - shared across both call sites on this same + // attached JNIEnv rather than a second FindClass() per pass. + if (_cached_object_class == nullptr) { + jclass local = jni->FindClass("java/lang/Object"); + if (!jniExceptionCheck(jni) && local != nullptr) { + _cached_object_class = (jclass)jni->NewGlobalRef(local); + } + if (local != nullptr) { + jni->DeleteLocalRef(local); + } + } + jclass object_class = _cached_object_class; + + if (object_class == nullptr || + jni->EnsureLocalCapacity(class_count + 16) < 0 || + jniExceptionCheck(jni)) { + for (jint i = 0; i < class_count; i++) { + jni->DeleteLocalRef(classes[i]); + } + jvmti->Deallocate((unsigned char *)classes); + return; + } + + jobjectArray holder = jni->NewObjectArray(chunk_count, object_class, nullptr); + if (jniExceptionCheck(jni)) { + // OutOfMemoryError (or any other exception) building the holder - + // clear it rather than let it survive into the DeleteLocalRef() calls + // below (JNI spec: undefined behavior with a pending exception across + // ordinary JNI calls), same as expandFrontier()'s identical case. + holder = nullptr; + } + if (holder != nullptr) { + // Fill in REVERSE chunk order: holder[0] = classes[chunk_end-1], ..., + // holder[chunk_count-1] = classes[chunk_start]. HotSpot's + // FollowReferences visits the initial_object (the holder array) by + // pushing it on a LIFO visit_stack and popping (jvmtiTagMap.cpp: + // iterate_over_array pushes elements 0..n-1 in order, the while-loop + // pops LIFO), so classes are descended in REVERSE holder order. + // Reversing the fill makes the descent visit classes in ASCENDING + // original index order (chunk_start first), which is what + // admitStaticFieldRoots()'s resumable cursor below assumes: an abort + // at class p means classes chunk_start..p-1 are done and p+1..chunk_end-1 + // are pending, so the cursor resumes at p (redoing the partial class) + // without re-walking completed classes. + for (jint i = 0; i < chunk_count; i++) { + jni->SetObjectArrayElement(holder, i, classes[chunk_end - 1 - i]); + if (jniExceptionCheck(jni)) { + holder = nullptr; + break; + } + } + } + + // TEMP DIAGNOSTIC (pod round 13): ground-truth probe for the + // LEAK_BUFFER wrapper. Every deduction path (root-attached admit -> + // index -> collector walk; chain-attached -> at-risk FIFO walk; + // already-admitted re-hit -> upgrade or push) should end in a walk + // log with class=...SynchronizedRandomAccessList, yet none is ever + // observed. This probe bypasses the sweep callback entirely: it reads + // ProfileAnalyzer.LEAK_BUFFER directly via JNI and reports the + // wrapper's CURRENT tag and frontier entry shape each time the class + // passes through a chunk. Runs BEFORE the sweep's FollowReferences for + // this chunk, so the first lap shows tag=0 (pre-admission) and later + // laps show the steady-state entry. Remove once the wrapper question + // is answered. + for (jint i = chunk_start; i < chunk_end; i++) { + char *probe_sig = nullptr; + if (jvmti->GetClassSignature(classes[i], &probe_sig, nullptr) != + JVMTI_ERROR_NONE) { + continue; + } + if (probe_sig == nullptr || + strstr(probe_sig, "ProfileAnalyzer") == nullptr) { + if (probe_sig != nullptr) { + jvmti->Deallocate((unsigned char *)probe_sig); + } + continue; + } + TEST_LOG("ReferenceChainTracker::admitStaticFieldRoots LEAK_BUFFER " + "probe: sweeping holder class %s at index %d", + probe_sig, (int)i); + jvmti->Deallocate((unsigned char *)probe_sig); + jfieldID leak_fid = + jni->GetStaticFieldID(classes[i], "LEAK_BUFFER", "Ljava/util/List;"); + if (jniExceptionCheck(jni) || leak_fid == nullptr) { + jni->ExceptionClear(); + TEST_LOG("ReferenceChainTracker::admitStaticFieldRoots LEAK_BUFFER " + "probe: GetStaticFieldID failed"); + continue; + } + jobject probe_wrapper = jni->GetStaticObjectField(classes[i], leak_fid); + if (jniExceptionCheck(jni) || probe_wrapper == nullptr) { + jni->ExceptionClear(); + TEST_LOG("ReferenceChainTracker::admitStaticFieldRoots LEAK_BUFFER " + "probe: static value null/exception"); + continue; + } + jlong probe_tag = 0; + jvmti->GetTag(probe_wrapper, &probe_tag); + FrontierEntry probe_entry{}; + bool probe_found = + probe_tag > 0 ? _frontier->lookup(probe_tag, &probe_entry) : false; + TEST_LOG("ReferenceChainTracker::admitStaticFieldRoots LEAK_BUFFER " + "probe wrapper_tag=%lld frontier_found=%d parent=%lld " + "root_kind=%u state=%u leak_tag=%lld depth=%u " + "referrer_klass=%u", + (long long)probe_tag, (int)probe_found, + (long long)(probe_found ? probe_entry.parent_tag : 0), + (unsigned)(probe_found ? probe_entry.root_kind : 0), + (unsigned)(probe_found ? probe_entry.state : 0), + (long long)(probe_found ? probe_entry.leak_tag : 0), + (unsigned)(probe_found ? probe_entry.depth : 0), + probe_found ? probe_entry.referrer_klass : 0); + jni->DeleteLocalRef(probe_wrapper); + break; // one holder class per chunk is enough + } + + // GetLoadedClasses() returned a local ref for every class regardless of + // chunk selection - free all of them here, not just the chunk. + for (jint i = 0; i < class_count; i++) { + jni->DeleteLocalRef(classes[i]); + } + jvmti->Deallocate((unsigned char *)classes); + + if (holder == nullptr) { + // OOM/local-ref exhaustion/array-store failure - skip this pass's sweep + // rather than treating it like the manual walk's own truncation (see + // this method's own header comment). + return; + } + + PassContext ctx; + ctx.tracker = this; + ctx.frontier = _frontier; + ctx.hop_cap = hop_cap; + ctx.budget = budget; + ctx.edges_admitted = 0; + ctx.truncated = false; + ctx.frontier_cap_hit = false; + // Empty (not null) batch_tags forces heapReferenceCallback() to stop at + // exactly one hop past each class - see this method's own header comment + // for why a deeper descent here would reintroduce the whole-graph + // FollowReferences cost the array-holder batching design otherwise avoids. + std::unordered_set empty_batch_tags; + ctx.batch_tags = &empty_batch_tags; + // Lets heapReferenceCallback() walk past the holder->class seed edge (see + // PassContext::static_field_seed's own comment) so this sweep actually + // reaches each class's static fields instead of stopping at the + // negative-tagged class object itself. + ctx.static_field_seed = true; + // Per-class non-STATIC_FIELD admission cap (see PassContext::_class_other_cap's + // own comment). STATIC_FIELD edges are always admitted; non-static edges + // (CONSTANT_POOL, INTERFACE, SUPERCLASS, CLASS_LOADER, ...) are admitted + // up to this many per class per lap, then dropped for the rest of that + // class. 32 covers a typical class's full constant-pool/interface set; + // outlier classes are bounded so they cannot blow the chunk's deadline. + ctx._class_other_cap = STATIC_FIELD_SWEEP_NON_STATIC_CAP_PER_CLASS; + // TEMP DIAGNOSTIC (see doc/temp/ investigation notes): see PassContext:: + // kind_counts's own comment. + int kind_counts[32]; + memset(kind_counts, 0, sizeof(kind_counts)); + ctx.kind_counts = kind_counts; + + jvmtiHeapCallbacks callbacks; + memset(&callbacks, 0, sizeof(callbacks)); + callbacks.heap_reference_callback = heapReferenceCallback; + u64 follow_start_ticks = TSC::ticks(); + jvmtiError follow_err = + jvmti->FollowReferences(0, nullptr, holder, &callbacks, &ctx); + *safepoint_ticks += TSC::ticks() - follow_start_ticks; + jni->DeleteLocalRef(holder); + // TEMP DIAGNOSTIC (see doc/temp/ investigation notes): kind indices per + // jvmti.h's jvmtiHeapReferenceKind - 1=CLASS 2=FIELD 3=ARRAY_ELEMENT + // 4=CLASS_LOADER 5=SIGNERS 6=PROTECTION_DOMAIN 7=INTERFACE 8=STATIC_FIELD + // 9=CONSTANT_POOL 10=SUPERCLASS (21-27 are root kinds, not expected here). + TEST_LOG("ReferenceChainTracker::admitStaticFieldRoots kind_counts " + "k1=%d k2=%d k3=%d k4=%d k5=%d k6=%d k7=%d k8=%d k9=%d k10=%d", + kind_counts[1], kind_counts[2], kind_counts[3], kind_counts[4], + kind_counts[5], kind_counts[6], kind_counts[7], kind_counts[8], + kind_counts[9], kind_counts[10]); + if (follow_err != JVMTI_ERROR_NONE) { + return; + } + + *edges_admitted = ctx.edges_admitted; + *truncated = ctx.truncated; + *frontier_cap_hit = ctx.frontier_cap_hit; + + if (ctx.truncated) { + _static_field_sweep_cycle_truncated = true; + // Resumable cursor: instead of skipping to chunk_end (losing every + // class after the interruption point for the rest of this lap), resume + // at the class we were inside when the walk aborted. The holder was + // filled in reversed order so descent visits classes in ascending + // original index order; _classes_in_chunk_visited counts how many + // classes were entered before the abort. Resume at chunk_start + count + // - 1 to redo the partial class (its already-admitted edges hit + // ALREADY_ADMITTED cheaply; with the per-class quota its non-static + // edges complete within the cap). Classes before it are done; classes + // after it are pending and will be reached on the next pass. + if (ctx._classes_in_chunk_visited > 0) { + _static_field_sweep_cursor = + chunk_start + ctx._classes_in_chunk_visited - 1; + } else { + // Aborted before any class's own edges were seen (e.g. during the + // holder->class seed edges) - redo the whole chunk. + _static_field_sweep_cursor = chunk_start; + } + } else { + // Full advance: every class in the chunk was processed. + _static_field_sweep_cursor = chunk_end; + } + if (_static_field_sweep_cursor >= class_count) { + *cycle_complete = !_static_field_sweep_cycle_truncated; + _static_field_sweep_cursor = 0; + _static_field_sweep_cycle_truncated = false; + } +} + +bool ReferenceChainTracker::releaseSearchTags(jvmtiEnv *jvmti, JNIEnv *jni) { + assert(!t_inGCCallback && + "GetObjectsWithTags is a JVMTI Heap-category call and must not be " + "made from GarbageCollectionStart/Finish"); + if (jvmti == nullptr || _frontier == nullptr) { + return true; // nothing to release + } + + jlong scan_limit = _frontier->size(); + std::vector live_tags; + for (jlong tag = 1; tag <= scan_limit; tag++) { + FrontierEntry entry{}; + if (_frontier->lookup(tag, &entry) && + entry.state != FrontierEntryState::ABANDONED) { + live_tags.push_back(tag); + } + } + if (live_tags.empty()) { + return true; + } + + jint resolved_count = 0; + jobject *resolved_objects = nullptr; + jlong *resolved_tags = nullptr; + if (jvmti->GetObjectsWithTags((jint)live_tags.size(), live_tags.data(), + &resolved_count, &resolved_objects, + &resolved_tags) != JVMTI_ERROR_NONE) { + // GetObjectsWithTags() itself failed (e.g. JVMTI_ERROR_OUT_OF_MEMORY): + // we do NOT know which, if any, of live_tags are still live objects, so + // do not mark any of them ABANDONED here - doing so while their JVMTI + // tag might still be set would let a restarted search's nextTag() + // sequence eventually reissue the same numeric tag to a brand-new + // object, corrupting FrontierTable's tag-uniqueness invariant (see this + // method's own header comment). Report failure so the caller retries + // this same batch later instead of proceeding to restart. + Counters::increment(REFERENCE_CHAIN_TAG_RELEASE_FAILED); + Log::warn("ReferenceChains: GetObjectsWithTags failed while releasing " + "%zu search tag(s); will retry before allowing a search " + "restart", + live_tags.size()); + return false; + } + + for (jint i = 0; i < resolved_count; i++) { + // clearTag() rather than a raw SetTag() call - reuses the + // same helper (and its GC-callback self-consistency assert) tagObject/ + // getTag already go through. + clearTag(jvmti, resolved_objects[i]); + if (jni != nullptr) { + jni->DeleteLocalRef(resolved_objects[i]); + } + } + if (resolved_objects != nullptr) { + jvmti->Deallocate((unsigned char *)resolved_objects); + } + if (resolved_tags != nullptr) { + jvmti->Deallocate((unsigned char *)resolved_tags); + } + // Tags that failed to resolve above are already dead (JVMTI forgot them + // with their object) - nothing to release, just mark the record ABANDONED + // below like every other entry this search owned. Only reached once + // GetObjectsWithTags() itself succeeded, so every live_tags entry has now + // either been resolved-and-cleared or confirmed dead. + for (jlong tag : live_tags) { + _frontier->clear(tag); + } + return true; +} + +bool ReferenceChainTracker::runPass(jvmtiEnv *jvmti, JNIEnv *jni, + bool *out_truncated) { + if (!_enabled || jvmti == nullptr || _frontier == nullptr) { + TEST_LOG_SUMMARY("ReferenceChainTracker::runPass early-exit: enabled=%d jvmti=%p frontier=%p", + _enabled, (void *)jvmti, (void *)_frontier); + return false; + } + + if (_search_state != SearchState::RUNNING) { + // The search already reached a terminal outcome - nothing left for + // another pass to do until shouldRunPass() decides to restartSearch() + // (this class's header comment), which flips _search_started back to + // false before this method is called again. If a prior terminal-state + // transition's releaseSearchTags() call failed, retry it here rather + // than leaving _tags_released false forever - shouldRunPass() refuses + // to restart the search until this succeeds (see _tags_released's own + // comment), so this is the only remaining call site that can make + // progress on the retry. + if (!_tags_released) { + _tags_released = releaseSearchTags(jvmti, jni); + } + TEST_LOG_SUMMARY("ReferenceChainTracker::runPass no-op: searchState=%d already terminal " + "tagsReleased=%d", + (int)_search_state, _tags_released); + if (out_truncated != nullptr) { + *out_truncated = false; + } + return true; + } + + resolveLoadedClasses(jvmti, jni); + + TEST_LOG_SUMMARY("ReferenceChainTracker::runPass starting JVMTI walk: " + "search_started=%d frontierSize=%zu", + _search_started, _frontier != nullptr ? _frontier->size() : (size_t)0); + + int edges_admitted = 0; + bool truncated = false; + bool frontier_cap_hit = false; + jvmtiError err; + // Whole-call wall-clock duration of runPassManualWalk() below - includes + // root/stack-ref enumeration dispatch, frontier-table bookkeeping, and + // rotation-candidate collection, in addition to the actual in-safepoint + // JVMTI calls. Used only to derive non_safepoint_ticks below for + // _cpu_pain_budget; NOT fed to updatePacing()/_pause_pid directly (see + // safepoint_ticks below for that). Measured via TSC::ticks() rather than + // OS::nanotime(), matching this codebase's other interval-timing call + // sites (LivenessTracker::track(), pollWatchedTargets() below); + // TSC::ticks() itself falls back to OS::nanotime() when the TSC is + // unavailable/disabled, so this is a strict upgrade with no behavior + // change on hosts without a usable timestamp counter. + u64 pass_wall_ticks = 0; + // Genuine in-safepoint cost of this pass, accumulated by + // runPassManualWalk() across every IterateOverReachableObjects/ + // FollowReferences call it makes (root enum, static-field sweep, ordinary + // expansion, rotation re-expansion) - explicitly excluding + // GetObjectsWithTags (not a safepoint call) and every bookkeeping line in + // between. This, not pass_wall_ticks, is what updatePacing()/ + // maybeRevokeBorrowForRootEnumPass() below actually regulate: JFR + // (jdk.ExecuteVMOperation[operation=HeapWalkOperation]) confirmed the two + // can differ substantially - a pass's non-safepoint bookkeeping must not + // be mistaken for pause-time-SLO pressure and throttle the PID controller + // on its behalf. + u64 safepoint_ticks = 0; + + // Every pass is driven by the manual walk (runPassManualWalk() - + // IterateOverReachableObjects for roots, then a batched array-holder + // FollowReferences per BFS level in expandFrontier()), on every collector + // including ZGC. The walk issues only JVMTI heap calls, which run inside + // the VM_HeapWalkOperation safepoint and honor ZGC's load barriers, so + // concurrent relocation cannot corrupt it - it reads no raw oop. Batching + // one hop per level keeps each FollowReferences bounded, avoiding the + // multi-hundred-ms-to-second STW pauses a whole-graph FollowReferences + // would impose. + bool manual_first_pass = !_search_started; + if (manual_first_pass) { + _search_started = true; + store(_search_start_ns, OS::nanotime()); + } + + // Root/stack-ref enumeration alone never discovers a root's transitive + // children (runPassManualWalk()'s own comment) - there is no "first pass + // walks the whole graph inline" shortcut here, so every pass (first or + // resumed) takes the same expand-frontier shape. Root/stack-ref + // enumeration itself, though, does NOT run on every pass: its fixed + // native dispatch cost is paid in full regardless of budget (see + // ROOT_ENUM_MIN_INTERVAL_NS's own comment), so it is cadence-gated to the + // first pass, a still-truncated retry from last time, or once + // ROOT_ENUM_MIN_INTERVAL_NS has elapsed since it last ran - not every + // pass, unlike expandFrontier()'s cheap incremental work below. + u64 now_ns = OS::nanotime(); + bool run_root_enum = manual_first_pass || _root_enum_truncated_last_time || + (now_ns - _last_root_enum_ns >= ROOT_ENUM_MIN_INTERVAL_NS); + + int frontier_size_before_pass = _frontier != nullptr ? _frontier->size() : 0; + + u64 call_start_ticks = TSC::ticks(); + runPassManualWalk(jvmti, jni, run_root_enum, _first_pass_budget, + _effective_budget, &edges_admitted, &truncated, + &frontier_cap_hit, &safepoint_ticks); + pass_wall_ticks = TSC::ticks() - call_start_ticks; + // TSC::ticks() is monotonic but not necessarily free of measurement noise + // between the outer call_start_ticks snapshot and the several inner + // TSC::ticks() snapshots safepoint_ticks is built from - clamp rather than + // underflow if the accumulated safepoint portion ever reads back larger + // than the whole-call wall time it's a subset of. + u64 non_safepoint_ticks = + pass_wall_ticks > safepoint_ticks ? pass_wall_ticks - safepoint_ticks : 0; + err = JVMTI_ERROR_NONE; + + store(_passes_run, load(_passes_run) + 1); + _last_pass_gc_finish_epoch = gcFinishEpoch(); + store(_last_pass_ns, OS::nanotime()); + if (!run_root_enum) { + // A pass that ran root/stack-ref enumeration spends _first_pass_budget, + // not _effective_budget - its duration is not a signal about the + // per-pass cost updatePacing() is trying to regulate (expandFrontier()'s + // cheap, per-node expansion calls), so feeding it in here would + // throttle _effective_budget down for every one of those unrelated + // later passes based on a single, deliberately oversized outlier. + updatePacing(safepoint_ticks); + } else { + // Excluded from the budget/cadence controller above, but not from the + // borrow ceiling's revocation check (see maybeRevokeBorrowForRootEnumPass()'s + // own comment) - a root-enum pass's in-safepoint cost is real pause time + // and must still be able to revoke a borrowed-budget grant the pacing + // controller would otherwise keep believing is safe. + maybeRevokeBorrowForRootEnumPass(safepoint_ticks); + } + // Search restart (this class's own header comment): accumulate this + // pass's own in-safepoint cost toward the running total restartSearch() + // will spend into _safepoint_pain_budget once the search reaches a terminal state - + // same TSC::ticks_to_millis() conversion updatePacing() already uses for + // its own pass-duration signal. + _search_pain_ms += TSC::ticks_to_millis(safepoint_ticks); + // Independent leaky bucket for the non-safepoint remainder of this pass + // (root/stack-ref enumeration dispatch, frontier-table admission, + // rotation-candidate collection) - see _cpu_pain_budget's own comment + // (referenceChains.h) for why this needs to be tracked separately from + // both _safepoint_pain_budget above and _pause_pid's safepoint_ticks signal. + // Spent every pass, root-enum or not: none of this cost is + // cadence-gated the way root enum's in-safepoint dispatch is. + _cpu_pain_budget.spend(TSC::ticks_to_millis(non_safepoint_ticks)); + + // Design doc's Termination section, decided in priority order: + // 1. Frontier-size cap hit -> abandon immediately, regardless of TTL. + // 2. No pending frontier entries left (this pass wasn't truncated) AND no + // active leak-accumulation watch -> the reachable graph was fully + // explored within the hop cap with nothing left to keep re-checking; + // natural completion (the hop cap alone is a normal boundary, not + // truncation - see heapReferenceCallback()'s own comment). See the + // _watched_leak_klass_count clause's own comment below for why an + // active watch withholds completion here even with an empty pending + // queue. + // 3. TTL exceeded while work is still pending -> abandon. + // 4. Otherwise stay RUNNING - more pending work, no cap hit yet. + // Write the abandon reason (and every other detail field + // buildAbandonedEvent() reads: _passes_run/_last_pass_ns/_search_start_ns + // above, _frontier's size, ...) BEFORE the _search_state transition below, + // and publish that transition with a release store - dump()'s reader side + // (buildAbandonedEvent()/searchState()) pairs it with an acquire load, so + // observing the new _search_state also guarantees every detail field + // written before this release store is visible too, even on a weakly + // ordered CPU (e.g. arm64) where relaxed stores to two different atomics + // carry no such guarantee. + bool has_pending_frontier = truncated; + int frontier_size_after = _frontier->size(); + if (frontier_cap_hit) { + // Frontier table is full -- no new entries can ever be admitted, so + // frontier_size_after can never exceed frontier_size_before_pass again. + // Deferring to the no-progress detector below (as a prior version of + // this branch did) would never actually reach it: this same `if` would + // keep matching every subsequent pass, permanently short-circuiting the + // else-if chain before _passes_since_last_progress is ever read. Abandon + // immediately instead, matching this function's own design-doc priority + // list above (frontier-size cap hit abandons regardless of TTL). + store(_abandon_reason, (u8)SearchAbandonReason::FRONTIER_CAP); + storeRelease(_search_state, (u8)SearchState::ABANDONED); + enqueuePendingAbandonedEvent(); + TEST_LOG_SUMMARY("ReferenceChainTracker::runPass frontier cap hit -- " + "abandoning search (size=%d)", + frontier_size_after); + } else if (!has_pending_frontier && _watched_leak_klass_count == 0) { + storeRelease(_search_state, (u8)SearchState::COMPLETED); + } else if (!has_pending_frontier) { + // Reachable graph fully explored, but LivenessTracker still has at least + // one klass under active leak watch (_watched_leak_klass_count's own + // comment) - do NOT complete. Rotation (collectLeakAccumulationCandidates + // ForRotation() et al., runPassManualWalk()'s own comment) exists + // precisely to re-observe already-EXPANDED entries whose fields mutate + // after their one-time expansion - e.g. an element appended to a + // static-field-rooted collection well after the walk first visited it. + // Once every reachable object has been visited once, has_pending_frontier + // goes permanently false and runPass()'s terminal-state branch would + // otherwise make every future call to this method a no-op forever + // (search_state != RUNNING short-circuits before rotation ever runs + // again) - silently disabling the one mechanism built to catch that + // exact mutation. Falling through here leaves _search_state at RUNNING, + // so the next pass (still gated by shouldRunPass()'s normal cadence/pain + // budget) runs the rotation collectors again with a fresh view of + // whatever object identities LivenessTracker is currently watching. + // _passes_since_last_progress below still counts this pass as "no + // progress" (frontier size is genuinely unchanged), so a watch that + // never resolves anything still bounds out via the TTL/no-progress + // branch below once isUrgent() clears - this only keeps a search alive + // while there is an active signal to keep probing for, not forever + // unconditionally. + } else if (_passes_since_last_progress >= NO_PROGRESS_PASS_LIMIT && + !isUrgent()) { + // The frontier hasn't grown for NO_PROGRESS_PASS_LIMIT consecutive + // passes — the search is genuinely stuck (not just slow), so + // abandon. A large heap takes more passes simply because + // there are more objects to explore; that is not "stuck". + // Only abandon when the frontier stops growing entirely. + // Suppressed when urgent (secondsToOOM() < OOM_URGENT_THRESHOLD_S): + // the search must complete to find the leak before the app OOMs. + store(_abandon_reason, (u8)SearchAbandonReason::TTL); + storeRelease(_search_state, (u8)SearchState::ABANDONED); + enqueuePendingAbandonedEvent(); + } else if (_candidate_count > 0 && + __builtin_popcountll(_candidate_found_bits) == + (u64)_candidate_count) { + // Canary early termination: all leaked candidates have been + // found -- the search is complete. + storeRelease(_search_state, (u8)SearchState::COMPLETED); + Counters::increment(REFERENCE_CHAIN_CANDIDATES_FOUND, + __builtin_popcountll(_candidate_found_bits)); + } else if (_candidate_count > 0 && + _passes_since_last_progress >= NO_PROGRESS_PASS_LIMIT && + _passes_since_last_candidate_progress >= + canaryStuckPassLimit()) { + // Canary-specific stuck detector - deliberately NOT suppressed by + // isUrgent() (contrast the ordinary TTL check above). The ordinary + // check's !isUrgent() guard protects a search that's still making real + // (whole-graph) progress from being killed just because the process is + // close to OOM; but _passes_since_last_candidate_progress only advances + // when NO candidate has been newly found and NO new candidate has been + // admitted, which frontier growth elsewhere in the graph does not + // affect. A canary that has made zero discovery progress for this many + // passes is provably not converging regardless of urgency, so letting + // isUrgent() keep it RUNNING would only burn urgency-boosted STW pause + // budget during the same OOM approach this search exists to diagnose. + // + // Also requires the whole-graph frontier to have stalled + // (_passes_since_last_progress >= NO_PROGRESS_PASS_LIMIT): a search + // whose frontier is still growing is making real progress toward + // eventually reaching the candidate even if it hasn't yet, so it is + // not "stuck" in the sense this detector exists to catch - see + // canaryStuckPassLimit()'s own comment for why the pass limit itself + // also escalates across consecutive restarts of the same chase. + store(_abandon_reason, (u8)SearchAbandonReason::CANARY_STUCK); + storeRelease(_search_state, (u8)SearchState::ABANDONED); + enqueuePendingAbandonedEvent(); + if (_canary_stuck_restart_count < MAX_CANARY_STUCK_BACKOFF_SHIFT) { + _canary_stuck_restart_count++; + } + } + + // Track progress: if the frontier grew this pass, reset the no-progress + // counter. Otherwise increment it. + if (frontier_size_after > frontier_size_before_pass) { + _passes_since_last_progress = 0; + } else { + _passes_since_last_progress++; + } + + // Track canary-specific progress separately - see + // _passes_since_last_candidate_progress's own comment for why frontier + // growth above does not substitute for this. + // Pass-cost EMA for the canary lane's work-scaled backoff (see + // _canary_pass_ema_ms's own comment) - updated from every pass's whole- + // call wall duration so it is warm before the first held-off decision. + // 0.8/0.2, same smoothing as _gotw_ema_call_ns. + { + u64 pass_wall_ms = (u64)TSC::ticks_to_millis(pass_wall_ticks); + _canary_pass_ema_ms = _canary_pass_ema_ms == 0 + ? pass_wall_ms + : _canary_pass_ema_ms * 4 / 5 + pass_wall_ms / 5; + } + int candidate_progress_mark = + _candidate_count + (int)__builtin_popcountll(_candidate_found_bits); + if (candidate_progress_mark > _last_candidate_progress_mark) { + _last_candidate_progress_mark = candidate_progress_mark; + _passes_since_last_candidate_progress = 0; + // Real chase progress (a candidate found or a new one admitted) - the + // canary lane gets its back-to-back spacing back (multiplier 1, see + // _canary_backoff_mult's own comment). + _canary_backoff_mult = 1; + } else { + _passes_since_last_candidate_progress++; + // No chase progress: double the canary lane's work-scaled spacing + // multiplier, capped. Only while a chase is actually open - an + // all-found or candidate-free pass should not accumulate backoff for + // a chase that no longer exists. + if (_candidate_count > 0 && + __builtin_popcountll(_candidate_found_bits) < (u64)_candidate_count) { + _canary_backoff_mult = + std::min(_canary_backoff_mult * 2, CANARY_BACKOFF_MULT_MAX); + _last_canary_pass_ns = OS::nanotime(); + } + } + + if (load(_search_state) != SearchState::RUNNING) { + _tags_released = releaseSearchTags(jvmti, jni); + // Release canary marker tags: use GetObjectsWithTags + // to find all live marker-tagged objects and clear + // them. DeleteLocalRef each object before + // Deallocate-ing the array (matches the + // existing pattern at referenceChains.cpp:2125-2133). + if (_candidate_count > 0) { + for (int i = 0; i < _candidate_count; i++) { + jlong tag = _candidate_tags[i]; + jint count = 0; + jobject *objects = nullptr; + jlong *result_tags = nullptr; + jvmtiError cerr = jvmti->GetObjectsWithTags( + 1, &tag, &count, &objects, &result_tags); + if (cerr == JVMTI_ERROR_NONE && count > 0) { + for (jint j = 0; j < count; j++) { + jvmti->SetTag(objects[j], 0); + jni->DeleteLocalRef(objects[j]); + } + jvmti->Deallocate((unsigned char *)objects); + jvmti->Deallocate((unsigned char *)result_tags); + } + } + _candidate_count = 0; + _candidate_found_bits = 0; + memset(_candidate_discovered_count, 0, sizeof(_candidate_discovered_count)); + memset(_candidate_qualifying_tid_count, 0, + sizeof(_candidate_qualifying_tid_count)); + _passes_since_last_candidate_progress = 0; + } + // Only CANARY_STUCK should keep escalating canaryStuckPassLimit() - + // any other terminal reason (natural completion, all candidates found, + // frontier cap, TTL) is an unrelated outcome for this chase sequence, + // so a fresh restart afterward should start back at the base limit. + if (load(_abandon_reason) != SearchAbandonReason::CANARY_STUCK) { + _canary_stuck_restart_count = 0; + } + } + + if (out_truncated != nullptr) { + *out_truncated = truncated; + } + + TEST_LOG_SUMMARY("ReferenceChainTracker::runPass done: err=%d edges_admitted=%d truncated=%d " + "frontier_cap_hit=%d searchState=%d abandonReason=%d frontierSize=%d " + "effectiveBudget=%d effectiveCadenceNs=%llu pendingExpand=%zu priorityExpand=%zu " + "candidateFound=%d/%d discoveredCounts=[%d,%d,%d,%d,%d]", + (int)err, edges_admitted, truncated, frontier_cap_hit, (int)load(_search_state), + (int)_abandon_reason, _frontier->size(), _effective_budget, + (unsigned long long)_effective_cadence_ns, + _pending_expand.size(), _priority_expand.size(), + (int)__builtin_popcountll(_candidate_found_bits), _candidate_count, + _candidate_count > 0 ? _candidate_discovered_count[0] : 0, + _candidate_count > 1 ? _candidate_discovered_count[1] : 0, + _candidate_count > 2 ? _candidate_discovered_count[2] : 0, + _candidate_count > 3 ? _candidate_discovered_count[3] : 0, + _candidate_count > 4 ? _candidate_discovered_count[4] : 0); + + return err == JVMTI_ERROR_NONE; +} + +// --------------------------------------------------------------------------- +// Pause-time-SLO feedback loop (see this method's declaration in +// referenceChains.h for the full mechanism). +// --------------------------------------------------------------------------- + +void ReferenceChainTracker::updatePacing(u64 pass_wall_ticks) { + // Truncating to whole milliseconds matches every other PidController usage + // in this codebase (ObjectSampler/MallocTracer/NativeSocketSampler all feed + // it integer counts, pidController.h's `compute(u64 input, ...)`) - sub-ms + // precision is not meaningful against a millisecond-scale target anyway. + // TSC::ticks_to_millis() already falls back to a nanotime-based conversion + // when the TSC is unavailable/disabled (tsc.h), matching runPass()'s own + // TSC::ticks() fallback for pass_wall_ticks itself. + u64 pass_ms = TSC::ticks_to_millis(pass_wall_ticks); + // time_delta_coefficient is deliberately 1.0, not a real-elapsed-time + // ratio - unlike ObjectSampler's usage (objectSampler.cpp), which + // rescales an event count accumulated over a variable-length real-time + // window against a fixed-real-time target, _pause_pid was constructed + // with sampling_window=1 (its own constructor comment above, in start()): + // one compute() call *is* one pass, and pass_ms already IS the per-call + // quantity being compared against the per-call ceiling _target encodes. + // Rescaling pass_ms by how much real wall-clock time elapsed since the + // previous call would compare it against a target calibrated for a + // different unit (per-second, not per-pass), double-counting the same + // irregular-cadence effect this coefficient exists to correct for in the + // per-second case. (Re-litigated after review: an earlier pass flagged + // this as a bug and a fix using TSC-measured elapsed time was drafted, + // but re-checking against this constructor's own documented design + // confirmed 1.0 is correct here - see this comment instead of changing + // it again.) + double signal = _pause_pid.compute(pass_ms, 1.0); + + // Budget-borrowing (referenceChains.h's _borrowed_budget comment): only a + // sustained run of comfortably-under-target passes earns extra headroom + // above _budget, and any pass that is not comfortably under target revokes + // it immediately - _budget itself must stay the ceiling the instant this + // search stops proving it has pause-time room to spare. + bool comfortably_under_target = + _effective_pause_target_ms > 0 && + (double)pass_ms <= (double)_effective_pause_target_ms * BORROW_UNDER_TARGET_FRACTION; + if (comfortably_under_target) { + if (_consecutive_under_target_passes < BORROW_WARMUP_PASSES) { + _consecutive_under_target_passes++; + } + if (_consecutive_under_target_passes >= BORROW_WARMUP_PASSES) { + int64_t max_borrow = (int64_t)_budget * (BORROW_CEILING_MULTIPLIER - 1); + int64_t grown = _borrowed_budget + + (int64_t)std::llround((double)_budget * BORROW_GROWTH_FRACTION); + _borrowed_budget = std::min(grown, max_borrow); + } + } else { + _consecutive_under_target_passes = 0; + _borrowed_budget = 0; + } + + int64_t ceiling = (int64_t)_budget + _borrowed_budget; + int64_t floor = ceiling > 0 ? std::min((int64_t)MIN_EFFECTIVE_BUDGET, ceiling) + : 0; + int64_t desired = (int64_t)_effective_budget + (int64_t)std::lround(signal); + int64_t clamped = std::max(floor, std::min(ceiling, desired)); + // Whatever part of `desired` the clamp above could not absorb - positive + // when there was more headroom than the ceiling allows, negative when the + // pass is still over the pause-time target even at the floor. Drives + // _effective_cadence_ns below, per this method's own comment on folding + // Open Question 5 into the same controller output. + int64_t overflow = desired - clamped; + _effective_budget = (int)clamped; + + if (overflow < 0) { + // Still over the pause-time ceiling even at the minimum budget - widen + // the fallback interval instead of shrinking the budget further. + u64 step = (u64)(-overflow) * CADENCE_NS_PER_EDGE_OVERFLOW; + _effective_cadence_ns = + std::min(_effective_cadence_ns + step, MAX_EFFECTIVE_CADENCE_NS); + } else if (overflow > 0) { + // Comfortably under the ceiling even at the maximum (config) budget - + // relax the fallback interval. The GC-finish-epoch trigger already fires + // independently of cadence (shouldRunPass() above), so this only + // shortens how long an idle, no-GC-event search waits between passes. + u64 step = (u64)overflow * CADENCE_NS_PER_EDGE_OVERFLOW; + _effective_cadence_ns = + step >= _effective_cadence_ns + ? MIN_EFFECTIVE_CADENCE_NS + : std::max(_effective_cadence_ns - step, MIN_EFFECTIVE_CADENCE_NS); + } + // overflow == 0: the budget clamp alone fully absorbed this pass's + // correction - leave the cadence at its current value. +} + +// A root/stack-ref enumeration pass never reaches updatePacing() above (see +// runPass()'s own comment on why its wall-clock cost is excluded from the +// per-pass PID/effective-budget signal), but it still spends real +// pause-time-SLO time. _borrowed_budget's own comment requires the grant be +// revoked the instant ANY pass is not comfortably under target, so this +// mirrors updatePacing()'s comfortably_under_target check for that one +// purpose only - it never grows _consecutive_under_target_passes/ +// _borrowed_budget, since the warmup streak is calibrated against +// expandFrontier()'s per-node cost, not this call's unrelated fixed +// dispatch cost. +void ReferenceChainTracker::maybeRevokeBorrowForRootEnumPass( + u64 pass_wall_ticks) { + if (_effective_pause_target_ms <= 0) { + return; + } + u64 pass_ms = TSC::ticks_to_millis(pass_wall_ticks); + bool comfortably_under_target = + (double)pass_ms <= (double)_effective_pause_target_ms * BORROW_UNDER_TARGET_FRACTION; + if (!comfortably_under_target) { + _consecutive_under_target_passes = 0; + _borrowed_budget = 0; + // The ceiling updatePacing() would compute right now collapses to + // _budget alone (no _borrowed_budget term above) - re-clamp + // _effective_budget immediately instead of leaving the borrow-inflated + // value in place until the next ordinary pass's updatePacing() call. + _effective_budget = std::min(_effective_budget, (int)_budget); + } +} + +// --------------------------------------------------------------------------- +// Target-selection bridging step - LivenessTracker's leak-candidate ranking feeds +// this tracker's already-running BFS search (design doc's Open Question 3, +// corrected mechanism - see this method's own comment below and the plan +// doc's "Correction to the design doc's Open Question 3 mechanism"). +// --------------------------------------------------------------------------- + +void ReferenceChainTracker::requeueChainRootForRotation(jlong tag) { + if (_frontier == nullptr || tag <= 0) { + return; + } + // Walk the parent chain up to the root-attached entry - the same links + // reconstructChain() walks, but we only need the tag, not the class ids. + // Bounded by _hop_cap (the frontier's own invariant: depth <= hop_cap), + // so a corrupt cycle cannot spin here. + jlong root_tag = tag; + FrontierEntry entry{}; + int hops = 0; + while (hops++ < _hop_cap) { + if (!_frontier->lookup(root_tag, &entry) || entry.parent_tag == 0) { + break; + } + root_tag = entry.parent_tag; + } + if (root_tag == tag) { + return; // tag IS the root - nothing above it to requeue + } + if (!_frontier->lookup(root_tag, &entry) || + entry.state != FrontierEntryState::EXPANDED) { + return; // root pruned or still pending expansion - nothing to re-walk + } + if (isQueuedForRotation(root_tag) || + _priority_expand.size() >= PRIORITY_EXPAND_CAP) { + return; + } + TEST_LOG("ReferenceChainTracker::requeueChainRootForRotation root_tag=%lld " + "target_tag=%lld", + (long long)root_tag, (long long)tag); + _priority_expand.push_back(root_tag); + _priority_expand_set.insert(root_tag); +} + +namespace { + +// The discovered-chain gate's suppression predicate, shared by EVERY site +// that caches a resolved chain - the poll's discovered-instances loop AND +// both representative build paths (the canary/marker path and the +// normal-tag path). Chains shallower than the first real holder hop +// (depth < 2) rooted at a TRANSIENT root (stack local / JNI local) are the +// observed noise shape - a momentarily-live frame's variable holding the +// instance - whose retention explanation evaporates when the frame dies. +// Everything else is real: a depth==1 chain rooted at a durable root is the +// singleton-collection leak shape (a depth-0 static root's elements are +// depth 1), and a depth==0 chain rooted at a durable root is the +// direct-retention shape (the root-retained object itself - e.g. a static +// field's value, or a Thread object for thread-local leaks) - suppressing +// those unconditionally would drop exactly the retention categories the +// search exists to report. Anything deeper passes regardless of root kind +// (at depth >= 2 the chain has at least one real holder hop). +// Representative-driven builds used to bypass this check entirely (found +// live: the canary path cached a stack-local-rooted depth-1 chain for a +// seeded noise-class representative and snapshot-and-keep re-emitted it +// forever) - every cacheResolvedChain() call site in pollWatchedTargets() +// must pass this gate. +bool suppressChainEvent(const ReferenceChainEvent &event) { + return event._depth < 2 && isTransientRootKind(event._root_kind); +} + +} // namespace + +// Chain-event reconstruction for a discovered/correlated instance (out of +// line from referenceChains.h so these sites share the TU's level-gated +// TEST_LOG; per-instance outcomes are level-2 diagnostics). +bool ReferenceChainTracker::buildChainEvent(jvmtiEnv *jvmti, JNIEnv *jni, + jlong target_tag, + ReferenceChainEvent *out) { + if (_frontier == nullptr || out == nullptr) { + TEST_LOG("ReferenceChainTracker::buildChainEvent false: " + "frontier=%p out=%p", (void *)_frontier, (void *)out); + return false; + } + FrontierEntry entry{}; + if (!_frontier->lookup(target_tag, &entry)) { + TEST_LOG("ReferenceChainTracker::buildChainEvent false: " + "target_tag=%lld not in frontier", (long long)target_tag); + return false; + } + std::vector chain; + std::vector edges; + u8 root_kind = 0; + if (!_frontier->reconstructChain(target_tag, &chain, &root_kind, &edges)) { + TEST_LOG("ReferenceChainTracker::buildChainEvent false: " + "reconstructChain failed for target_tag=%lld", + (long long)target_tag); + return false; + } + TEST_LOG("ReferenceChainTracker::buildChainEvent target_tag=%lld chain_size=%zu " + "chain[0]=%u depth=%u root_kind=%u leak_tag=%lld", + (long long)target_tag, chain.size(), chain.empty() ? 0u : chain[0], + entry.depth, (unsigned)root_kind, (long long)entry.leak_tag); + out->_target_tag = entry.leak_tag != 0 ? (u64)entry.leak_tag : (u64)target_tag; + out->_depth = entry.depth; + out->_root_kind = root_kind; + out->_chain = std::move(chain); + // Retention-edge labels, aligned with _chain (see fillHopEdgeLabels()). + fillHopEdgeLabels(jvmti, jni, edges, &out->_edges); + return true; +} + +// Canary chain reconstruction (out of line for the same reason). The +// canary's build outcomes are level-1: they are the chase's lifecycle +// story, rare (bounded by the candidate count) and chase-relevant. +bool ReferenceChainTracker::buildCanaryChainEvent(int candidate_idx, + ReferenceChainEvent *out) { + if (_frontier == nullptr || out == nullptr || candidate_idx < 0 || + candidate_idx >= _candidate_count) { + TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " + "frontier=%p out=%p idx=%d count=%d", + (void *)_frontier, (void *)out, candidate_idx, + _candidate_count); + return false; + } + jlong parent_tag = _candidate_parent_tags[candidate_idx]; + u32 candidate_klass = _candidate_referrer_klasses[candidate_idx]; + jlong frontier_tag = _candidate_frontier_tags[candidate_idx]; + std::vector chain; + u8 root_kind = 0; + if (parent_tag > 0) { + // Walk parent_tag back to root through the frontier table. + FrontierEntry entry{}; + if (!_frontier->lookup(parent_tag, &entry)) { + TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " + "parent_tag=%lld not in frontier (candidate=%d)", + (long long)parent_tag, candidate_idx); + return false; + } + root_kind = entry.root_kind; + for (jlong tag = parent_tag; tag > 0;) { + if (!_frontier->lookup(tag, &entry)) { + TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " + "chain walk: tag=%lld not in frontier (candidate=%d)", + (long long)tag, candidate_idx); + return false; + } + chain.push_back(entry.referrer_klass); + tag = entry.parent_tag; + } + } else if (parent_tag == 0 && frontier_tag > 0) { + // Root-referenced candidate: chain is just [candidate_klass]. + // root_kind was stored in the frontier entry at pruning time; + // re-read it from the frontier table (the candidate's own entry). + FrontierEntry entry{}; + if (!_frontier->lookup(frontier_tag, &entry)) { + TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " + "frontier_tag=%lld not in frontier (candidate=%d)", + (long long)frontier_tag, candidate_idx); + return false; + } + root_kind = entry.root_kind; + } else { + TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " + "never pruned (candidate=%d parent_tag=%lld frontier_tag=%lld)", + candidate_idx, (long long)parent_tag, + (long long)frontier_tag); + return false; // never pruned (candidate not reached) + } + // Prepend the candidate's own referrer_klass. + chain.push_back(candidate_klass); + // The chain was built root-to-parent; reverse to get candidate-to-root. + std::reverse(chain.begin(), chain.end()); + out->_target_tag = (u64)frontier_tag; + out->_depth = _candidate_depths[candidate_idx]; + out->_root_kind = root_kind; + out->_chain = std::move(chain); + TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent candidate=%d " + "parent_tag=%lld chain_size=%zu", + candidate_idx, (long long)parent_tag, chain.size()); + return true; +} + +void ReferenceChainTracker::pollWatchedTargets(jvmtiEnv *jvmti, JNIEnv *jni) { + if (!_enabled || jvmti == nullptr || jni == nullptr || + !LivenessTracker::instance()->gcGenerationsEnabled()) { + // Explicit guard, even though selectLeakCandidates() below already + // returns 0 candidates whenever its own _gc_generations gate + // (livenessTracker.h) is off - keeps this method's cost at the four + // checks above, not even a shared-lock-guarded table scan, when the + // feature isn't in use (design doc's Open Question 3 "still undecided" + // fallback: referencechains=... alone gets no target-seeding). + return; + } + + // Stamp every entry this poll refreshes with the current search + // generation. _search_start_ns changes each time restartSearch() begins a + // new search (runPass() sets it on the restarted search's first pass); a + // cached chain whose source_search_ns predates the current one was + // reconstructed from a FrontierTable the restart has since reset, so it is + // refreshed below the moment the restarted search re-tags its sample - + // trusting the stale source_tag would risk matching a tag the reset has + // reassigned to an unrelated object. + const u64 current_search_ns = load(_search_start_ns); + + // klass_ids resolved (and therefore already pruned-if-dead) by the + // candidate loop below, so the prune pass afterwards skips re-resolving + // them - it only needs to cover cached klasses that are no longer flagged. + // Per-instance caching: no per-klass prune needed (see comment below + // where the prune logic used to be). + + // Sized generously above LivenessTracker::selectLeakCandidates()'s own + // private MAX_LEAK_CANDIDATES cap (design doc: top 3-5) - that method + // clamps internally to whichever of `max`/its own cap/the qualifying- + // candidate count is smallest, so this local bound only needs to be + // "large enough", not exactly synchronized to a constant this class has + // no visibility into (MAX_LEAK_CANDIDATES is private to LivenessTracker). + constexpr int kMaxWatchedCandidates = 8; + KlassCandidate candidates[kMaxWatchedCandidates]; + int candidate_count = LivenessTracker::instance()->selectLeakCandidates( + candidates, kMaxWatchedCandidates); + + // Publish this poll's qualifying tids as LivenessTracker's watched-admission + // set (see noteSelectedCandidates()'s own comment, livenessTracker.h): + // exactly the (klass, tid) scope tagLeakInstances() tags and this chase + // intercepts gets its allocations admitted at 100% instead of the default + // 10% ratio lottery. Includes the candidate_count == 0 case, which clears + // the set - the boost must track the candidate selection poll by poll. + LivenessTracker::instance()->noteSelectedCandidates(candidates, + candidate_count); + + // Refresh the faster, un-hysteresis-gated klass_id ranking rotation + // priority uses (see _watched_leak_klass_ids' own comment) - but only + // once selectLeakCandidates() above has ALREADY found at least one + // qualifying candidate via its own slower hysteresis gate: this mechanism + // is meant to crank once the trend detector has triggered, not to run the + // ranking independently before that gate has ever fired. Refreshed even + // when candidate_count's specific candidates are unrelated to whichever + // klass ends up ranked highest by generation count - the two lists serve + // different purposes (canary marker output vs. rotation priority) and are + // deliberately not required to agree. + if (candidate_count > 0) { + // Snapshot the OLD watched set before overwriting it, so any klass_id + // that's newly appearing this refresh can get its one-time retroactive + // catch-up (seedLeakAccumulationForNewlyWatchedKlass() - see + // _watched_leak_klass_ids' own comment for why admission-time tracking + // alone cannot see objects admitted before watching started). + u32 previously_watched[MAX_WATCHED_LEAK_KLASSES]; + int previously_watched_count = _watched_leak_klass_count; + for (int i = 0; i < previously_watched_count; i++) { + previously_watched[i] = _watched_leak_klass_ids[i]; + } + _watched_leak_klass_count = LivenessTracker::instance()->topKlassesByGenerationCount( + _watched_leak_klass_ids, MAX_WATCHED_LEAK_KLASSES); + for (int i = 0; i < _watched_leak_klass_count; i++) { + u32 klass_id = _watched_leak_klass_ids[i]; + bool already_watched = false; + for (int j = 0; j < previously_watched_count; j++) { + if (previously_watched[j] == klass_id) { + already_watched = true; + break; + } + } + if (!already_watched) { + seedLeakAccumulationForNewlyWatchedKlass(klass_id); + } + } + } + // Only log when there are candidates to act on - this poll runs on every + // BFS-thread wake (once per second), so logging a zero count is per-second + // noise for the common idle case. + if (candidate_count > 0) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate_count=%d", candidate_count); + // Admit any candidate selectLeakCandidates() returns this poll that + // doesn't already occupy a slot, into the next free slot. This runs on + // every poll (not gated to "only the first time") because + // selectLeakCandidates()'s result set can change across polls - a new + // klass_id can start qualifying well after the search began. Slots are + // never retired or reassigned once occupied: a klass_id that stops + // qualifying just keeps whatever slot it has (and can still be found + // there), it is never freed for reuse by a different klass_id. That + // keeps the marker tag (MARKER_TAG_BASE - slot) a stable, search-lifetime + // identity for heapReferenceCallback()'s decode (referenceChains.cpp, + // near the *tag_ptr <= MARKER_TAG_BASE check) - reusing a slot mid-search + // would let a live marker tag on one object suddenly decode to a + // different klass_id's bookkeeping. + // Use resolveCandidateRepresentative() (re-reads under lock) + // instead of candidates[i].representative (stale jweak). + for (int i = 0; i < candidate_count; i++) { + u32 klass_id = candidates[i].klass_id; + bool already_tracked = false; + for (int s = 0; s < _candidate_count; s++) { + if (_candidate_klass_ids[s] == klass_id) { + already_tracked = true; + break; + } + } + if (already_tracked) { + continue; + } + if (_candidate_count >= MAX_LEAK_CANDIDATES_FROM_LT) { + TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary: klass_id=%u " + "qualifies but all %d slots are occupied - not tracked this search", + klass_id, MAX_LEAK_CANDIDATES_FROM_LT); + continue; + } + // Tag this candidate's specific representative object with a distinct + // marker tag (MARKER_TAG_BASE - slot) so heapReferenceCallback() can + // identify that exact object by identity when the walk reaches it - + // matching by class alone would record a chain for whichever instance + // of that class the walk happens to visit, not necessarily the one + // LivenessTracker flagged as growing. + int slot = _candidate_count; + _candidate_klass_ids[slot] = klass_id; + _candidate_tags[slot] = 0; // no marker tags — using leak tags now + _candidate_count = slot + 1; + TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary: admitted klass_id=%u " + "into slot=%d (candidate_count now %d)", + klass_id, slot, _candidate_count); + Counters::increment(REFERENCE_CHAIN_CANDIDATE_COUNT, 1); + } + // Refresh the per-slot qualifying-tid snapshot the walk phases read + // (walkCandidateThreadLocals()): zero every slot first, then fill from + // THIS poll's candidates - a klass whose per-tid trend stopped + // qualifying must stop having its tids walked, exactly like it stops + // consuming pool tags (tagLeakInstances() below keeps the same + // per-poll-candidates scope for the same reason). + memset(_candidate_qualifying_tid_count, 0, + sizeof(_candidate_qualifying_tid_count)); + for (int i = 0; i < candidate_count; i++) { + for (int s = 0; s < _candidate_count; s++) { + if (_candidate_klass_ids[s] != candidates[i].klass_id) { + continue; + } + int n = candidates[i].qualifying_tid_count; + if (n > MAX_CANDIDATE_QUALIFYING_TIDS) { + n = MAX_CANDIDATE_QUALIFYING_TIDS; + } + for (int q = 0; q < n; q++) { + _candidate_qualifying_tids[s][q] = candidates[i].qualifying_tids[q]; + } + _candidate_qualifying_tid_count[s] = n; + break; + } + } + // Tag the tracked instances of THIS poll's candidates with leak tags. + // This replaces the old single-representative marker-tag approach — + // the BFS will find these specific leaking objects by tag, not by + // class match, eliminating noise from unrelated instances of the same + // class. Passing the per-poll candidates (not the ever-occupied + // _candidate_klass_ids slots) is deliberate: the candidates carry the + // qualifying tids selectLeakCandidates() just computed, and a klass + // whose per-tid trend stopped qualifying should stop consuming pool + // tags even though its slot persists (tagLeakInstances()'s own + // comment, livenessTracker.h). + int tagged = LivenessTracker::instance()->tagLeakInstances( + jvmti, candidates, candidate_count); + _leak_tags_assigned = tagged; + _leak_tags_resolved = 0; // reset on each tagging round + TEST_LOG("ReferenceChainTracker::pollWatchedTargets tagLeakInstances tagged=%d", + tagged); + } + + for (int i = 0; i < candidate_count; i++) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate[%d] klass_id=%u", i, + candidates[i].klass_id); + // Deliberately does NOT resolve candidates[i].representative directly: + // that field is a snapshot taken under selectLeakCandidates()'s own + // shared-lock scan, which can go stale (LRU-evicted and + // DeleteWeakGlobalRef()'d by LivenessTracker's cleanup_table(), running + // concurrently on a different thread) at any point between that call and + // this one - see selectLeakCandidates()'s comment (livenessTracker.h) for + // why resolving it here would be undefined behavior, not just a null + // result. resolveCandidateRepresentative() re-reads the table's current + // value for this klass_id and resolves it atomically under the same + // lock, so it is always safe to call from here. + // Per-instance caching: no per-klass prune needed. + const u32 klass_id = candidates[i].klass_id; + jobject obj = LivenessTracker::instance()->resolveCandidateRepresentative( + jni, klass_id); + if (obj == nullptr) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate[%d] klass_id=%u " + "representative could not be resolved (died/evicted)", + i, klass_id); + // The representative died, but the canary chain (if the candidate + // was pruned by BFS before the representative died) only needs the + // frontier table — not the live representative. Try to build it + // before erasing the cached chain and skipping this candidate. + bool built_from_canary = false; + for (int s = 0; s < _candidate_count; s++) { + if (_candidate_klass_ids[s] != klass_id) continue; + if ((_candidate_found_bits & (1ULL << s)) && + _candidate_frontier_tags[s] != 0) { + jlong canary_ftag = _candidate_frontier_tags[s]; + _resolved_chains_lock.lock(); + bool need = (_resolved_chains.find(canary_ftag) == _resolved_chains.end()); + _resolved_chains_lock.unlock(); + if (need) { + ReferenceChainEvent event; + built_from_canary = buildCanaryChainEvent(s, &event); + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "buildCanaryChainEvent(dead rep, slot=%d) -> %d", + s, (int)built_from_canary); + if (built_from_canary && suppressChainEvent(event)) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "filtered depth=%u root_kind=%d canary_ftag=%lld " + "klass_id=%u (dead-rep path)", + event._depth, (int)event._root_kind, + (long long)canary_ftag, klass_id); + built_from_canary = false; + invalidateResolvedChain(canary_ftag); + } else if (built_from_canary) { + event._start_time = TSC::ticks(); + cacheResolvedChain(canary_ftag, std::move(event), + canary_ftag, current_search_ns); + } + } + } + break; + } + if (!built_from_canary) { + // The representative died. Per-instance caching means we don't erase + // by klass_id — chains for other instances of this class may still + // be valid. The dead representative's chain (if any) will expire + // when the search restarts and the frontier is wiped. + } + continue; // candidate died, or was evicted, since LivenessTracker flagged it + } + + { + jclass obj_klass = jni->GetObjectClass(obj); + char *obj_class_name = nullptr; + if (obj_klass != nullptr && + jvmti->GetClassSignature(obj_klass, &obj_class_name, nullptr) == + JVMTI_ERROR_NONE && + obj_class_name != nullptr) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate[%d] " + "klass_id=%u class_name=%s", + i, klass_id, obj_class_name); + jvmti->Deallocate((unsigned char *)obj_class_name); + } + if (obj_klass != nullptr) { + jni->DeleteLocalRef(obj_klass); + } + } + + // Corrected mechanism (the plan doc's own correction to the design doc's + // original proposal): a READ, never a SetTag + // seed. runPass()'s whole-graph walk is the only thing that ever + // assigns a tag; if it already has (tag > 0), heapReferenceCallback() + // already recorded a correct parent_tag/depth chain for this object the + // moment it was first visited - pre-tagging it here instead would make + // that callback's `*tag_ptr == 0` branch (the only branch that records + // parent_tag/depth, referenceChains.h) skip it entirely the next time a + // pass reached it. + jlong tag = getTag(jvmti, obj); + + // Canary search: if this candidate was pre-tagged with a marker tag + // (negative, set above), use the canary chain reconstruction. The + // marker tag itself stays on the representative for the whole search + // (heapReferenceCallback() never overwrites it), so this stays true + // regardless of whether the walk has actually reached it yet this pass - + // buildCanaryChainEvent() below is what distinguishes "found" (parent_tag + // or frontier_tag populated) from "not yet pruned". + if (tag <= MARKER_TAG_BASE) { + // The marker tag encodes the slot this object was pre-tagged at + // (MARKER_TAG_BASE - slot, mirroring heapReferenceCallback()'s own + // decode at referenceChains.cpp:1510). Decode it from the tag itself + // rather than reusing the loop index `i`: selectLeakCandidates() is + // not guaranteed to return candidates in the same order across polls, + // so `i` can drift from the slot this object was actually tagged at. + int candidate_slot = (int)(MARKER_TAG_BASE - tag); + if (candidate_slot < 0 || candidate_slot >= _candidate_count) { + TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary candidate[%d] " + "klass_id=%u marker_tag=%lld decodes to out-of-range slot=%d " + "(candidate_count=%d) - skipping", + i, klass_id, (long long)tag, candidate_slot, _candidate_count); + jni->DeleteLocalRef(obj); + continue; + } + bool need_refresh = false; + jlong canary_ftag = _candidate_frontier_tags[candidate_slot]; + _resolved_chains_lock.lock(); + auto it = _resolved_chains.find(canary_ftag); + need_refresh = (it == _resolved_chains.end() || + it->second.source_search_ns != current_search_ns); + _resolved_chains_lock.unlock(); + TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary candidate[%d] " + "klass_id=%u marker_tag=%lld slot=%d needRefresh=%d", + i, klass_id, (long long)tag, candidate_slot, need_refresh); + if (need_refresh) { + ReferenceChainEvent event; + bool built = buildCanaryChainEvent(candidate_slot, &event); + TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary " + "buildCanaryChainEvent(slot=%d) -> %d", + candidate_slot, built); + if (built && suppressChainEvent(event)) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "filtered depth=%u root_kind=%d canary_ftag=%lld " + "klass_id=%u (canary path)", + event._depth, (int)event._root_kind, + (long long)canary_ftag, klass_id); + built = false; + invalidateResolvedChain(canary_ftag); + } + if (built) { + event._start_time = TSC::ticks(); + cacheResolvedChain(canary_ftag, std::move(event), + canary_ftag, current_search_ns); + } + } + // Fall through to discovered-instances check below — the canary + // representative may not have been reached by BFS yet, but other + // instances of the same class may have been admitted and their + // chains can be built now. + } + + // Normal (non-canary) path: tag > 0 means the walk visited this + // object and assigned it a frontier tag. + + // Keep the holder chain's root warm in the rotation queue: a growing + // container's current internals are only reachable via the holder's + // re-walk (requeueChainRootForRotation()'s own comment). Every poll, + // not just on cache refresh - the holder must be re-walked CONTINUOUSLY + // to observe each resize as it happens. + if (tag > 0) { + requeueChainRootForRotation(tag); + } + + // Reconstruct only when this klass has no current chain cached: either + // nothing cached yet, or what is cached was built from a different tag or + // an earlier search generation (see current_search_ns above). A klass + // that keeps getting flagged, unchanged, across many polls is left alone + // - its cached chain is already being re-emitted on every dump. + // + // Round-19 (pod 289f8): resolve by the SLOT's chain key, not the + // representative's leak tag. The resolved-chains cache is keyed by + // frontier tags (disc_tag); under the leak-tag design the rep's leak + // tag is never a frontier key (interceptions insert with a fresh + // frontier tag, entry.leak_tag rides inside), so find(leak_tag) can + // never hit — the rep stayed needRefresh=1 forever, retrying a + // buildChainEvent(leak_tag) that always fails "not in frontier". + // buildDiscoveredInstanceChains() records the slot's frontier tag + // when a leak-tag chain first lands (the found criterion); until + // then there is nothing to refresh — the discovered path below both + // creates the chain and marks the slot found. + bool need_refresh = false; + jlong rep_chain_key = 0; + for (int s = 0; s < _candidate_count; s++) { + if (_candidate_klass_ids[s] == klass_id) { + rep_chain_key = _candidate_frontier_tags[s]; + break; + } + } + if (rep_chain_key != 0) { + _resolved_chains_lock.lock(); + auto it = _resolved_chains.find(rep_chain_key); + need_refresh = (it == _resolved_chains.end() || + it->second.source_search_ns != current_search_ns); + _resolved_chains_lock.unlock(); + } + TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate[%d] klass_id=%u tag=%lld " + "needRefresh=%d", + i, klass_id, (long long)tag, need_refresh); + if (need_refresh) { + ReferenceChainEvent event; + bool built = buildChainEvent(jvmti, jni, tag, &event); + TEST_LOG("ReferenceChainTracker::pollWatchedTargets buildChainEvent(tag=%lld) -> %d", + (long long)tag, built); + if (built && suppressChainEvent(event)) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "filtered depth=%u root_kind=%d rep_tag=%lld klass_id=%u " + "(representative path)", + event._depth, (int)event._root_kind, (long long)tag, + klass_id); + built = false; + invalidateResolvedChain(tag); + } + if (built) { + // Provisional stamp; drainPendingChainEvents() re-stamps each copy at + // dump time so the event lands in that chunk's window. + event._start_time = TSC::ticks(); + cacheResolvedChain(tag, std::move(event), tag, current_search_ns); + } + } + // tag == 0: The representative object has no tag — the BFS walk + // hasn't reached it yet AND it is not yet leak-tagged. No action here: + // tagLeakInstances() (earlier in this same poll) tags the tracked + // instances of candidate classes' QUALIFYING tids from the reusable + // pool, so the next tagLeakInstances round or the next walk pass will + // pick it up (a representative of a klass whose candidates all carry + // the rep's tid - foldKlassCountsLocked()'s dominant-tid rep bias - is + // in scope by construction). The + // old marker-tag re-tag path is gone — marker tags are no longer the + // candidate discovery mechanism (leak tags are), and re-tagging with + // a marker tag here would resurrect the dead mechanism on objects the + // leak-tag pool has not yet reached. + + // Build chain events for auto-marked discovered instances of this class. + // Runs via buildDiscoveredInstanceChains() - see that method's own + // comment for why it is slot-driven (the orphan fix), and the + // per-poll-candidate call plus the slot sweep below for the two + // call sites. + buildDiscoveredInstanceChains(jvmti, jni, klass_id, current_search_ns); + + jni->DeleteLocalRef(obj); + } + + // Orphan fix: slots whose klass is NOT among this poll's candidates. A + // candidate that qualified long enough for the walk to record discovered + // instances, then stops qualifying (per-tid trend decay, thread switch - + // the exact shape observed live: the walk recorded 8 discovered + // instances the pass AFTER the candidate's seeded trend aged out, and + // nothing ever built their chains for the remaining 116 passes, because + // the per-candidate loop above only iterates the CURRENT poll's + // candidates), must not strand those instances: the slot persists by + // design precisely so the klass "can still be found there" + // (_candidate_klass_ids' own comment above) - this sweep is what finds + // them. Slots whose klass IS a poll candidate are skipped here to avoid + // double-processing (the loop above already handled them) - the build + // itself is idempotent either way (already-cached chains are skipped + // under the same source_search_ns check), this is purely to keep the + // per-poll log volume unchanged for still-qualifying candidates. + for (int s = 0; s < _candidate_count; s++) { + u32 slot_klass = _candidate_klass_ids[s]; + bool in_poll = false; + for (int i = 0; i < candidate_count; i++) { + if (candidates[i].klass_id == slot_klass) { + in_poll = true; + break; + } + } + if (!in_poll && slot_klass != 0) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets orphan slot " + "sweep: slot=%d klass_id=%u not in poll candidates - building " + "its discovered chains", + s, slot_klass); + buildDiscoveredInstanceChains(jvmti, jni, slot_klass, current_search_ns); + } + } + + // Per-instance caching: chains are keyed by frontier tag, not klass_id. + // There is no per-klass prune — chains for dead instances are harmless + // (they describe a reference path that was valid at resolution time) and + // expire naturally when the search restarts (frontier is wiped, all tags + // become invalid, _resolved_chains is cleared in restartSearch()). + // The backend can filter stale chains by cross-referencing with + // HeapLiveObject events from the same chunk. +} + +// Inserts or refreshes klass_id's resolved chain - see _resolved_chains' +// comment (referenceChains.h) for why a resolved chain is cached and +// re-emitted rather than emitted once. A refresh (klass_id already present) +// always succeeds; only a brand-new klass_id arriving with the cache already +// full is dropped (counted, not silent), rather than evicting some other +// still-live sample's chain. Split out of pollWatchedTargets() so +// ResolvedChainCacheTest (referenceChains_ut.cpp) can drive the overflow path +// directly, without standing up hundreds of real LivenessTracker candidates. +void ReferenceChainTracker::cacheResolvedChain(jlong source_tag, + ReferenceChainEvent &&event, + jlong source_tag_val, + u64 source_search_ns) { + _resolved_chains_lock.lock(); + auto it = _resolved_chains.find(source_tag); + if (it == _resolved_chains.end() && + (int)_resolved_chains.size() >= MAX_RESOLVED_CHAINS) { + _resolved_chains_lock.unlock(); + Counters::increment(REFERENCE_CHAIN_EVENTS_DROPPED); + TEST_LOG("ReferenceChainTracker::cacheResolvedChain dropped new source_tag=%lld, " + "cache full (at MAX_RESOLVED_CHAINS=%d)", + (long long)source_tag, MAX_RESOLVED_CHAINS); + return; + } + CachedChain &slot = _resolved_chains[source_tag]; + slot.event = std::move(event); + slot.source_tag = source_tag_val; + slot.source_search_ns = source_search_ns; + TEST_LOG("ReferenceChainTracker::cacheResolvedChain source_tag=%lld cache_size=%d", + (long long)source_tag, (int)_resolved_chains.size()); + _resolved_chains_lock.unlock(); +} + +void ReferenceChainTracker::invalidateResolvedChain(jlong source_tag) { + _resolved_chains_lock.lock(); + auto it = _resolved_chains.find(source_tag); + if (it != _resolved_chains.end()) { + _resolved_chains.erase(it); + TEST_LOG("ReferenceChainTracker::invalidateResolvedChain source_tag=%lld", + (long long)source_tag); + } + _resolved_chains_lock.unlock(); +} + +// Builds and caches chain events for every auto-marked discovered +// instance recorded against a slot holding klass_id (see the auto-mark +// block in heapReferenceCallback() for how instances get recorded). +// +// Slot-driven, deliberately NOT poll-candidate-driven: the slots persist +// for the whole search ("a klass_id that stops qualifying just keeps +// whatever slot it has ... it is never freed for reuse", +// pollWatchedTargets()'s own slot-admission comment) specifically so the +// klass can still be found after it stops qualifying - a candidate that +// qualified long enough for the walk to record discovered instances and +// then stopped qualifying (per-tid trend decay, thread switch) must not +// strand them. Two call sites make that true: the per-poll-candidate loop +// (klass still qualifying - chains build as fresh as possible, before the +// trend can age out) and the orphan slot sweep (klass no longer in this +// poll's candidates - the recorded instances still get their chains). +// Idempotent per instance: buildDiscoveredInstanceChains skips any tag +// already cached for the current search generation, so calling both +// paths for the same slot in one poll is safe - the sweep simply never +// overlaps the per-candidate call for the same klass. +void ReferenceChainTracker::buildDiscoveredInstanceChains(jvmtiEnv *jvmti, + JNIEnv *jni, + u32 klass_id, + u64 current_search_ns) { +for (int s = 0; s < _candidate_count; s++) { + if (_candidate_klass_ids[s] != klass_id) continue; + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "discovered loop: klass_id=%u slot=%d discovered_count=%d", + klass_id, s, _candidate_discovered_count[s]); + if (_candidate_discovered_count[s] == 0) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "no discovered instances for klass_id=%u slot=%d", + klass_id, s); + } + for (int d = 0; d < _candidate_discovered_count[s]; d++) { + jlong disc_tag = _candidate_discovered_tags[s][d]; + if (disc_tag == 0) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "disc_tag=0 at idx=%d for klass_id=%u slot=%d", + d, klass_id, s); + continue; + } + // Skip if already cached for this instance + _resolved_chains_lock.lock(); + bool already_cached = (_resolved_chains.find(disc_tag) != _resolved_chains.end()); + _resolved_chains_lock.unlock(); + if (already_cached) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "already_cached disc_tag=%lld klass_id=%u slot=%d idx=%d", + (long long)disc_tag, klass_id, s, d); + continue; + } + ReferenceChainEvent event; + bool built = buildChainEvent(jvmti, jni, disc_tag, &event); + // Retention-explanation filter. Only applies to discovered + // instances, not canary. Suppress: + // - depth==0: the instance IS the root (chain is just [object], + // no holder to explain anything); + // - depth==1 rooted at a TRANSIENT root (stack local / JNI + // local): the observed noise shape - a momentarily-live + // frame's variable holding the instance. The chain explains a + // retention that evaporates when the frame dies. + // Both durable-rooted shapes are REAL direct-retention chains and + // must NOT be caught by a blanket depth filter: depth==1 rooted at + // a static field is the singleton-collection leak shape (a depth-0 + // static root's elements are depth 1), and depth==0 rooted at a + // durable root is the root-retained object itself (a static field's + // value, a Thread object for thread-local leaks) - the actual + // retention categories the search exists to report. Anything + // deeper passes regardless of root kind (at depth >= 2 the chain + // has at least one real holder hop). + if (built && suppressChainEvent(event)) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "filtered depth=%u root_kind=%d disc_tag=%lld klass_id=%u", + event._depth, (int)event._root_kind, + (long long)disc_tag, klass_id); + built = false; + // Also drop any chain cached for this tag before the filter + // existed (or before an improveChain/reparent upgraded it) - + // drainPendingChainEvents() re-emits cached chains + // unconditionally, so suppressing only the build would leave the + // noise chains re-emitting forever. + invalidateResolvedChain(disc_tag); + } + if (built) { + event._start_time = TSC::ticks(); + cacheResolvedChain(disc_tag, std::move(event), disc_tag, + current_search_ns); + // Track coverage for adaptive CPU budget + if (event._target_tag >= (u64)LEAK_TAG_BASE) { + _leak_tags_resolved++; + // Round-19 (pod 289f8): the canary chase's found criterion. The + // marker-tag design this code replaced ("no marker tags — using + // leak tags now", the slot registration in pollWatchedTargets()) + // never migrated heapReferenceCallback()'s marker-keyed + // found-bit setting — under leak tags nothing ever set + // _candidate_found_bits, so the chase could never exit (0/1 + // across every search, exits only via frontier-cap/no-progress). + // A leak-tag-target chain for the slot IS the leak-tag-world + // "canary found": a walk reached the leaked population and the + // correlation carried (target_tag = the leak tag). Record the + // link so buildCanaryChainEvent()'s root-referenced branch works + // too (parent_tag 0, frontier_tag = disc_tag). + if (!(_candidate_found_bits & (1ULL << s))) { + _candidate_found_bits |= (1ULL << s); + _candidate_frontier_tags[s] = disc_tag; + _candidate_parent_tags[s] = 0; + _candidate_depths[s] = event._depth; + _candidate_referrer_klasses[s] = klass_id; + TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary " + "found: klass_id=%u slot=%d leak chain target_tag=%llu " + "via disc_tag=%lld (%d/%d candidates found)", + klass_id, s, (unsigned long long)event._target_tag, + (long long)disc_tag, + __builtin_popcountll(_candidate_found_bits), + _candidate_count); + } + } + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "auto-marked chain for klass_id=%u tag=%lld target_tag=%llu", + klass_id, (long long)disc_tag, + (unsigned long long)event._target_tag); + } else { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "buildChainEvent failed for discovered tag=%lld " + "klass_id=%u slot=%d disc_idx=%d", + (long long)disc_tag, klass_id, s, d); + } + } + break; +} +} + +void ReferenceChainTracker::recordDiscoveredInstance(u32 klass_id, + jlong frontier_tag, + bool leak_correlated) { + // See the declaration's own comment (referenceChains.h) for the + // noise-eviction rationale. Bounded: at most MAX_DISCOVERED_INSTANCES- + // PER_CLASS frontier lookups when evicting, zero allocation (slots are + // fixed arrays). + for (int s = 0; s < _candidate_count; s++) { + if (_candidate_klass_ids[s] != klass_id) { + continue; + } + if (_candidate_discovered_count[s] < MAX_DISCOVERED_INSTANCES_PER_CLASS) { + _candidate_discovered_tags[s][_candidate_discovered_count[s]++] = + frontier_tag; + TEST_LOG("ReferenceChainTracker::recordDiscoveredInstance slot=%d " + "klass_id=%u tag=%lld leak_correlated=%d count=%d", + s, klass_id, (long long)frontier_tag, (int)leak_correlated, + _candidate_discovered_count[s]); + return; + } + if (!leak_correlated) { + return; // full - noise never displaces anything + } + // All slots full and this instance is leak-correlated: evict the first + // slot held by an entry with no leak tag (a noise instance). Also drop + // the evicted instance's cached chain so it stops re-emitting - the + // discovered-loop gate below suppresses new noise builds, but a chain + // cached before that gate keeps draining forever. + for (int d = 0; d < _candidate_discovered_count[s]; d++) { + jlong victim = _candidate_discovered_tags[s][d]; + FrontierEntry victim_entry{}; + if (_frontier == nullptr || + !_frontier->lookup(victim, &victim_entry) || + victim_entry.leak_tag == 0) { + _candidate_discovered_tags[s][d] = frontier_tag; + invalidateResolvedChain(victim); + TEST_LOG("ReferenceChainTracker::recordDiscoveredInstance evicted " + "noise slot=%d idx=%d victim_tag=%lld for leak tag=%lld", + s, d, (long long)victim, (long long)frontier_tag); + return; + } + } + TEST_LOG("ReferenceChainTracker::recordDiscoveredInstance all slots " + "leak-correlated, dropping tag=%lld klass_id=%u", + (long long)frontier_tag, klass_id); + return; + } +} + +bool ReferenceChainTracker::correlateAdmittedLeakTag(jlong frontier_tag, + jlong leak_tag, + u32 klass_id) { + // See the declaration's own comment (referenceChains.h). Called from + // LivenessTracker::tagLeakInstances() on this same thread + // (pollWatchedTargets -> tagLeakInstances), so _candidate_* slot access + // here never races heapReferenceCallback's auto-mark path. + if (_frontier == nullptr) { + return false; + } + FrontierEntry entry{}; + if (!_frontier->lookup(frontier_tag, &entry)) { + return false; // not a live frontier tag (or the search restarted) + } + if (entry.leak_tag != 0) { + // Already correlated (idempotent) - e.g. a second tagLeakInstances + // round after a post-restart re-admission. + return true; + } + _frontier->setLeakTag(frontier_tag, leak_tag); + TEST_LOG_SUMMARY("ReferenceChainTracker::correlateAdmittedLeakTag " + "frontier_tag=%lld leak_tag=%lld depth=%u parent_tag=%lld", + (long long)frontier_tag, (long long)leak_tag, entry.depth, + (long long)entry.parent_tag); + recordDiscoveredInstance(klass_id, frontier_tag, true); + return true; +} + +void ReferenceChainTracker::drainPendingChainEvents( + std::vector *out) { + if (out == nullptr) { + return; + } + // Snapshot-and-keep, not a drain: every cached chain is copied out (and + // re-stamped so it lands in the dumping chunk's window) while the cache + // itself is left intact, so the same live sample's chain re-emits into + // every chunk it survives into (see _resolved_chains' comment). `now` is + // read once, before the lock, so every event in one dump shares a stamp. + u64 now = TSC::ticks(); + _resolved_chains_lock.lock(); + for (const auto &kv : _resolved_chains) { + out->push_back(kv.second.event); + out->back()._start_time = now; + } + _resolved_chains_lock.unlock(); + TEST_LOG_SUMMARY("ReferenceChainTracker::drainPendingChainEvents re-emitted=%d", + (int)out->size()); +} + +void ReferenceChainTracker::enqueuePendingAbandonedEvent() { + // Called right after runPass() (referenceChains.cpp) writes + // SearchState::ABANDONED, on the same thread, before shouldRunPass() gets + // a chance to call restartSearch() - so buildAbandonedEvent()'s live read + // of _search_state/_abandon_reason/etc. is guaranteed to still succeed + // here even though it cannot be trusted to succeed later, from dump()'s + // independent clock (see _pending_abandoned_events' own comment). + ReferenceChainAbandonedEvent event; + if (!buildAbandonedEvent(&event)) { + return; + } + _pending_abandoned_events_lock.lock(); + if ((int)_pending_abandoned_events.size() >= MAX_PENDING_ABANDONED_EVENTS) { + _pending_abandoned_events_lock.unlock(); + Counters::increment(REFERENCE_CHAIN_EVENTS_DROPPED); + TEST_LOG_SUMMARY("ReferenceChainTracker::enqueuePendingAbandonedEvent dropped, " + "queue full (at MAX_PENDING_ABANDONED_EVENTS=%d)", + MAX_PENDING_ABANDONED_EVENTS); + return; + } + _pending_abandoned_events.push_back(event); + TEST_LOG_SUMMARY("ReferenceChainTracker::enqueuePendingAbandonedEvent reason=%d " + "queue_size=%d", + (int)event._reason, (int)_pending_abandoned_events.size()); + _pending_abandoned_events_lock.unlock(); +} + +void ReferenceChainTracker::drainPendingAbandonedEvents( + std::vector *out) { + if (out == nullptr) { + return; + } + // True drain, unlike drainPendingChainEvents() above: each queued event + // describes a discrete past occurrence, not an ongoing live sample, so + // once Profiler::dump() (profiler.cpp) has emitted it there is nothing + // left to re-report on the next dump. + _pending_abandoned_events_lock.lock(); + out->insert(out->end(), _pending_abandoned_events.begin(), + _pending_abandoned_events.end()); + _pending_abandoned_events.clear(); + _pending_abandoned_events_lock.unlock(); + TEST_LOG_SUMMARY("ReferenceChainTracker::drainPendingAbandonedEvents drained=%d", + (int)out->size()); +} diff --git a/ddprof-lib/src/main/cpp/referenceChains.h b/ddprof-lib/src/main/cpp/referenceChains.h new file mode 100644 index 0000000000..8a452d1729 --- /dev/null +++ b/ddprof-lib/src/main/cpp/referenceChains.h @@ -0,0 +1,3482 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +#ifndef _REFERENCECHAINS_H +#define _REFERENCECHAINS_H + +#include "arch.h" +#include "arguments.h" +#include "classTagAllocator.h" +#include "common.h" +#include "event.h" +#include "painBudget.h" +#include "pidController.h" +#include "spinLock.h" +#include "mutex.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +// PROF-15341: incremental resumption across passes (see +// ReferenceChainTracker::runPass() below), building on an earlier +// proof-of-concept that established two things end-to-end: +// 1. A cheap "a GC just happened" signal reaches this subsystem via the +// GarbageCollectionStart/Finish JVMTI callbacks (vmEntry.cpp), mirroring +// LivenessTracker::onGC() (livenessTracker.cpp:415-426) - just bumping an +// atomic epoch counter, nothing else. +// 2. JVMTI object tags round-trip a live object across a GC (SetTag/GetTag), +// via the minimal tagObject()/getTag()/clearTag() helpers below. +// The tag-indexed FrontierTable was added next, followed by the actual heap +// walk (runPass() calling jvmtiEnv::IterateOverReachableObjects to enumerate +// heap roots via heapRootCallback()/stackRefCallback(), populating +// FrontierTable subject to the hop cap/budget/frontier cap) - but that walk +// originally ran as a single, +// non-resumable pass with no cross-pass persistence, no GC-epoch-driven +// scheduling, and no tag release. This revision makes the search resumable +// and terminating: +// - runPass() now distinguishes a search's first pass (IterateOverReachableObjects +// to enumerate heap roots, exactly as the original single-pass walk did) +// from a resumed pass (expandFrontier() below: resolve each +// not-yet-expanded frontier entry via GetObjectsWithTags - dead ones are +// pruned for free - then call FollowReferences with that object as +// initial_object to discover its own outgoing edges, continuing until +// the per-pass budget or the frontier cap is hit). +// - The Termination section's cutoffs are enforced across passes: the hop +// cap already carried over via FrontierEntry::depth; this adds a +// wall-clock TTL cutoff (_ttl_ms, from first pass) and treats the +// frontier-size cap as immediate search abandonment rather than a +// per-pass truncation. +// - releaseSearchTags() clears (SetTag(obj, 0)) every live tag this search +// still owns once it completes or is abandoned, without discarding the +// FrontierTable's own records - reconstructChain() keeps working from +// memory after the search ends, only the underlying JVMTI tag map entry +// is released (design doc's Open Question 4 concern about leftover-tag +// overhead). +// - shouldRunPass()/threadLoop() implement the Triggering section's pass- +// scheduling signal (GC-finish epoch advanced, or a fixed cadence +// elapsed) - see threadLoop()'s own comment for why the thread this runs +// on is still not spawned by start(). +// +// PROF-15341 (doc/architecture/LiveHeapReferenceChains-RemainingWorkPlan.md): +// pollWatchedTargets() below is the LivenessTracker-to-ReferenceChainTracker +// target-selection bridge, closing the gap left by buildChainEvent() having +// no caller. It polls LivenessTracker::selectLeakCandidates() +// (livenessTracker.h's Open Question 3 population-slope ranking) and, for +// each candidate already tagged by an ordinary runPass() walk, reconstructs +// and emits its chain via Profiler::writeReferenceChain(). This is a READ of +// getTag(), never a SetTag seed - see pollWatchedTargets()'s own comment for +// the plan doc's "Correction to the design doc's Open Question 3 mechanism" +// this implements instead of the design doc's original seeding proposal. +// +// `can_tag_objects` and `can_generate_garbage_collection_events` are already +// requested unconditionally in vmEntry.cpp, so this bridging step only adds +// callback wiring and lazy event enablement, not capability requests. +// +// PROF-15341 (doc/architecture/LiveHeapReferenceChains-RemainingWorkPlan.md): +// the pause-time pacing controller replaces the fixed _budget/PASS_CADENCE_NS +// constants' role as the literal per-pass values with a measured +// pause-time-SLO feedback loop (design doc's Open Questions 2/5, "Proposed +// mechanism" paragraphs). runPass() now times its own FollowReferences/ +// GetObjectsWithTags call (already the thread blocked inside the safepoint +// those trigger, see the Triggering section) and feeds that duration to +// updatePacing() below, which scales _effective_budget/_effective_cadence_ns +// - the values runPass()/shouldRunPass()/threadLoop() now actually use - +// via this tracker's own PidController instance (_pause_pid). _budget/ +// PASS_CADENCE_NS survive as this controller's ceiling/baseline +// respectively, not as the literal per-pass values anymore. See +// updatePacing()'s own comment for the full mechanism, including why its +// gains are not copied from ObjectSampler/MallocTracer/NativeSocketSampler's +// shared triple. +// +// PROF-15341 (doc/architecture/LiveHeapReferenceChains-RemainingWorkPlan.md): +// search restart. Earlier revisions of this class only ever ran a single +// search for the tracker's entire lifetime (runPass()'s own comment used to +// read "starting a *new* search once one ends is not implemented"). That is +// a real gap: LivenessTracker::selectLeakCandidates() only trusts a klass's +// population trend once it has accumulated +// LivenessTracker::KLASS_POPULATION_MIN_FILL_FOR_TREND GC epochs of history +// (livenessTracker.cpp), which takes real wall-clock time - but a +// large-enough-budget search can finish walking the whole reachable graph, +// and permanently stop, before that time has passed. Any object allocated +// after the search already completed is then structurally undiscoverable +// forever, not just unlucky. +// +// The fix: a (re)started search's first pass is now gated on +// LivenessTracker already reporting at least one leak candidate, rather than +// starting unconditionally - by the time a candidate is flagged, the +// underlying object has necessarily survived several epochs already, so a +// fresh root-seeded walk started right then is very likely to still find it +// reachable. restartSearch() (referenceChains.cpp) resets the per-search +// state (frontier table, tag counter, emitted-target set) once a prior +// search reaches COMPLETED/ABANDONED, so shouldRunPass() can treat the next +// candidate-driven trigger exactly like a first-ever search. +// +// Restarting is still an expensive full-heap walk, so canAffordNewSearch() +// also gates it on _safepoint_pain_budget (PainBudget, painBudget.h) - a leaky bucket +// over the wall-clock cost of past searches, not a fixed cooldown, so a +// search that finished cheaply can restart again soon while an expensive one +// has to wait proportionally longer. shouldRunPass() reuses this same +// canAffordNewSearch() check for the very first search too, though the pain +// budget half of it is always a no-op there (nothing has been spent yet). +// With LivenessTracker::gcGenerationsEnabled() off there is no candidate +// signal to gate on at all, so both the first search and every restart start +// unconditionally, exactly as before this revision - gating without a +// leak-detection mechanism running would have no signal to justify one. +// +// JVMTI spec restriction: GarbageCollectionStart/Finish run while the VM is +// at a safepoint, and only the Memory Management category (Allocate/ +// Deallocate) is allowed from inside them - Heap category calls (SetTag, +// GetTag, GetObjectsWithTags, FollowReferences, IterateThroughHeap) are not. +// onGCStart()/onGCFinish() below must therefore never call anything but the +// atomic counter bump. GCCallbackGuard (referenceChains.cpp) marks this +// thread as "inside the GC callback" for the duration of that bump; the tag +// helpers assert() (debug builds only) that they are never entered while the +// guard is active, as a self-consistency check - it does not catch every way +// this restriction could be violated, only calls routed through this class. +// +// Per-tag frontier metadata state (design doc: Frontier/EdgeStore records). +// FRONTIER->EXPANDED is driven by ReferenceChainTracker::expandFrontier()/ +// markAllFrontierExpanded() once an entry's own outgoing edges have +// been visited; FRONTIER/EXPANDED->ABANDONED is driven by expandFrontier()'s +// resolve-or-drop path (dead objects) and releaseSearchTags() (search +// completion/abandonment). +namespace FrontierEntryState { +constexpr u8 FRONTIER = 0; // discovered, not yet expanded by FollowReferences +constexpr u8 EXPANDED = 1; // expanded; children (if any) are in the table +constexpr u8 EDGE = 2; // on a path toward a target sample (EdgeStore) +constexpr u8 ABANDONED = 3; // tag released; entry kept only to avoid reuse +} // namespace FrontierEntryState + +// Search-level outcome (design doc's Termination section), distinct from a +// single pass's per-call truncation (ReferenceChainTracker::runPass()'s +// `out_truncated`, unchanged from the original single-pass heap-walk engine): +// a pass can be truncated - budget +// or frontier cap exhausted for *that call* - without the search itself +// being ABANDONED, because there may be nothing left to do (RUNNING is still +// correct) or plenty left for the next pass to pick up. See runPass()'s own +// comment for exactly which conditions move _search_state out of RUNNING. +namespace SearchState { +constexpr u8 RUNNING = 0; // at least one more pass may still make progress +constexpr u8 COMPLETED = 1; // reachable graph fully explored within caps +constexpr u8 ABANDONED = 2; // TTL or frontier-size cap forced an incomplete stop +} // namespace SearchState + +// Records which cutoff actually moved a search from RUNNING to ABANDONED +// (runPass()'s Termination-section decision, referenceChains.cpp) - recorded +// so abandonReason() (and the T_REFERENCE_CHAIN_ABANDONED JFR event built +// from it, see buildAbandonedEvent()) can report *why*, per the design doc's +// "no silent truncation" requirement, rather than just *that* it happened. +// Values match Recording::recordReferenceChainAbandoned()'s kReasons table +// (flightRecorder.cpp) index-for-index. +namespace SearchAbandonReason { +constexpr u8 NONE = 0; // not (yet) abandoned +constexpr u8 FRONTIER_CAP = 1; // frontier-size cap hit +constexpr u8 TTL = 2; // wall-clock TTL exceeded with work still pending +// Canary candidate-discovery has made no progress for +// NO_PROGRESS_PASS_LIMIT consecutive passes. Unlike TTL above, this fires +// even while isUrgent() holds - see CANARY_NO_PROGRESS_PASS_LIMIT's own +// comment for why the ordinary TTL's !isUrgent() guard must not apply here. +constexpr u8 CANARY_STUCK = 3; +} // namespace SearchAbandonReason + +// Frontier/EdgeStore record (design doc: "Data structures" / +// "Frontier metadata storage"). Deliberately does not hold a live +// jclass/jobject: retaining either would defeat the point of using +// non-retaining JVMTI tags for frontier identity. `referrer_klass` is a +// StringDictionary id (Profiler::classMap(), profiler.h:260 - the same +// interning table LivenessTracker uses via Profiler::lookupClass(), +// livenessTracker.cpp:120-122) resolved from a class name string; the +// heap-walk engine populates it from GetClassSignature, and FrontierEntry +// only needs the field. +typedef struct FrontierEntry { + jlong parent_tag; // links back to the record that discovered this one + u32 referrer_klass; // StringDictionary id, 0 = unresolved/none + u32 depth; // hop count from the frontier's seed, for the hop cap + u8 state; // one of FrontierEntryState's constants + // The leak tag assigned by LivenessTracker to this specific tracked + // object, copied from the JVMTI tag at admission time. 0 = not a + // leak-tagged object (ordinary BFS admission). When non-zero, this is + // the stable correlation ID written into both ReferenceChain.targetTag + // and HeapLiveObject.leakTag — the backend joins on this field to + // match a reference chain to the specific leaking heap object it + // describes. + jlong leak_tag; + // jvmtiHeapReferenceKind of the edge that admitted this entry, but only + // meaningful when parent_tag == 0 (this entry is root-attached) - 0 (no + // JVMTI_HEAP_REFERENCE_* value is 0) for every other entry, since a + // non-root entry's own referrer edge kind is not what + // reconstructChain()'s callers want to report (they want to label the + // chain's root, not every hop). Set by heapReferenceCallback() + // (referenceChains.cpp) at insert() time. + u8 root_kind; + // Raw JVMTI class tag of THIS entry's own object, from the shared, + // process-wide allocator (classTagAllocator.h) - NOT referrer_klass above + // (a classMap dictionary id, which can differ for the same class at + // different times if that dictionary gets compacted/regenerated - see + // LivenessTracker::KlassPopulationEntry::stable_class_tag's own comment + // for the bug this was found fixing). Populated at admission time + // (admitObject()) directly from the class_tag value heapReferenceCallback()/ + // heapRootCallback() already receive as a JVMTI callback parameter - no + // extra JVMTI call needed. Stored (rather than only used transiently at + // admission time) specifically so ReferenceChainTracker:: + // seedLeakAccumulationForNewlyWatchedKlass()'s retroactive scan can read + // it back later for entries admitted long before that scan runs - a live + // JVMTI callback cannot be replayed after the fact. 0 = unresolved/none, + // same convention as referrer_klass. + jlong class_tag; + + // Retention-edge identity of the edge that admitted THIS entry, captured + // at admission time for the same cannot-replay-the-callback reason as + // class_tag above: it lets the emitted datadog.ReferenceChain name the + // field each hop is retained through, turning the bare class list into a + // readable path ("LeakHolder.SINK -> HashMap.table -> Entry.value") - + // which is the diagnostic point of the whole feature. + // + // referrer_field_index: for a FIELD/STATIC_FIELD admitting edge, the + // JVMTI-SPECIFICATION field ordinal of that field in the REFERRER's + // flattened field space - the ordinal space the heap callbacks' reference + // reference_info->field.index is defined over (JVMTI spec, + // jvmtiHeapReferenceInfoField: interface-implemented fields, then the + // superclass chain root-first, then the class's own fields, each in + // GetClassFields order - verified against HotSpot's implementation, + // jvmtiTagMap.cpp ClassFieldMap::create_map_of_*). NOT a position in any + // one GetClassFields result alone - the referrer class's FULL hierarchy is + // needed to decode it (resolveFieldEdgeName(), below). -1 = the admitting + // edge was not a field/static-field reference (or reference_info was + // unavailable). + jint referrer_field_index; + + // jvmtiHeapReferenceKind of the admitting edge for an INTERIOR hop (an + // entry with parent_tag != 0 - "this object was reached from its parent + // via this kind of edge"). Root-attached entries do not use this field + // (their edge kind IS root_kind by definition). 0 = not recorded. + u8 edge_kind; + + // Referrer's class tag when the referrer is a CLASS OBJECT rather than a + // frontier entry (a root-attached static-field admission - the referrer + // is the declaring class, parent_tag == 0 so there is no parent entry to + // read a class from). Interior hops leave 0 (their referrer class is the + // parent entry's own class_tag). Needed at emission to know WHICH + // hierarchy the referrer_field_index ordinal is defined over. + jlong referrer_class_tag; +} FrontierEntry; + +// Per-hop retention-edge identity collected by reconstructChain() alongside +// the class chain: everything needed at emission to label HOW chain[i] is +// retained (the field of chain[i+1] pointing at it, or the root edge for the +// last hop). Aligned with the leaf-to-root chain order - edges[i] describes +// the edge INTO chain[i]. +typedef struct ChainHopEdge { + // FrontierEntry::referrer_field_index/edge_kind of the entry for chain[i] + jint field_index; // -1 = not a field/static-field edge + u8 edge_kind; // admitting edge kind (root hops: root_kind) + // The referrer's raw class tag: the PARENT entry's class_tag for interior + // hops, FrontierEntry::referrer_class_tag for root-attached hops. 0 = + // unknown (label degrades to the edge kind, never a fabricated name). + jlong referrer_class_tag; +} ChainHopEdge; + +// Durability ranking for FrontierEntry::root_kind (design doc's "Fix for +// root-attribution staleness" point 1 / this plan's Phase 5 item 1): higher +// is more durable. Used to decide whether a newly-observed root reference to +// an already-admitted, root-attached entry should replace its recorded +// root_kind rather than keeping whichever root happened to be enumerated +// first. Only the three tiers the design doc actually names are ranked with +// confidence ("static/class/CLD > JNI global > JNI local/stack local/ +// monitor"); JVMTI_HEAP_REFERENCE_THREAD and _OTHER have no documented tier +// and are conservatively bucketed with the least-durable tier rather than +// assumed durable. +inline int rootKindDurability(u8 root_kind) { + switch (root_kind) { + case JVMTI_HEAP_REFERENCE_STATIC_FIELD: + case JVMTI_HEAP_REFERENCE_SYSTEM_CLASS: + return 3; + case JVMTI_HEAP_REFERENCE_JNI_GLOBAL: + return 2; + case JVMTI_HEAP_REFERENCE_MONITOR: + case JVMTI_HEAP_REFERENCE_STACK_LOCAL: + case JVMTI_HEAP_REFERENCE_JNI_LOCAL: + case JVMTI_HEAP_REFERENCE_THREAD: + case JVMTI_HEAP_REFERENCE_OTHER: + return 1; + default: + return 0; // root_kind's own "not set"/non-root-attached value + } +} + +// True for the two root kinds the design doc calls "first observed via" +// rather than "rooted by" evidence (design doc point 2): a stack-local or +// JNI-local reference is only alive for as long as its owning frame/handle +// scope is on some thread's stack, so an entry admitted through one is +// always a candidate both for a durability upgrade (rootKindDurability() +// above) and for the softer output label (flightRecorder.cpp's +// rootKindName()) and for Phase 5's bounded rotating re-expansion +// (ReferenceChainTracker::collectStaleRootKindEntriesForRotation()). +inline bool isTransientRootKind(u8 root_kind) { + return root_kind == JVMTI_HEAP_REFERENCE_STACK_LOCAL || + root_kind == JVMTI_HEAP_REFERENCE_JNI_LOCAL; +} + +// Tag-indexed slot table storing FrontierEntry metadata, modeled on +// LivenessTracker's TrackingEntry table (livenessTracker.h:21-30): CAS-safe +// doubling resize under a signal-safe SpinLock (spinLock.h), reusing its +// shared/exclusive split so reads (lookup) never race a resize. +// +// Structural difference from LivenessTracker's table: the slot index is the +// JVMTI tag value itself (tag - 1), not an externally-assigned array +// position. This works because ReferenceChainTracker::nextTag() hands out +// tags sequentially starting at 1 and never reuses one, so each +// tag maps to exactly one slot for the table's lifetime. +// +// Capacity is an explicit constructor parameter (wired from +// Arguments::_reference_chains_frontier_cap), not derived from heap +// size the way LivenessTracker sizes its table (livenessTracker.cpp:152-176) +// - the design doc explicitly flags that sizing formula as non-transferable +// to a BFS frontier (Open Question 2: frontier width is driven by per-hop +// fan-out, not an allocation sampling rate). Only the doubling-resize +// *mechanics* are reused from LivenessTracker, not its sizing heuristic. +// +// Concurrency: unlike LivenessTracker::track() (called from the allocation +// sampling hot path, which must never block), FrontierTable::insert()/ +// clear()/markEdge()/markExpanded() are only ever called from the single +// agent-owned BFS thread (design doc's Algorithm; the heap-walk engine), so +// they use the blocking exclusive lock() rather than LivenessTracker's +// non-blocking tryLockShared() bailout - exclusive, not shared, so a writer +// actually excludes a concurrent lookup() reader instead of merely +// serializing against other writers. lookup() may still be called +// concurrently from a reader walking parent_tag links (e.g. chain +// reconstruction), hence the shared lock there: shared mode only ever +// contends with other shared-mode readers, never with a writer's exclusive +// lock. +class alignas(alignof(SpinLock)) FrontierTable { +private: + // Provisional default pending empirical tuning (see + // doc/architecture/LiveHeapReferenceChains-ImplementationPlan.md) - not + // benchmark-derived. Reuses LivenessTracker's doubling-resize *mechanics* + // (growLocked() below), but this starting size is a conservative guess, + // not scaled from LivenessTracker's own initial size (which that class + // derives from max_heap/sampling_interval, a formula the design doc + // explicitly flags as non-transferable to a BFS frontier - see + // arguments.h's DEFAULT_REFERENCE_CHAINS_FRONTIER_CAP comment). Small + // enough to avoid over-allocating for a search that never grows a wide + // frontier, large enough to avoid the first several growLocked() calls + // for an ordinary one; a future frontier-table peak-occupancy + // measurement pass is the intended way to replace this guess. + static constexpr int INITIAL_TABLE_CAPACITY = 1024; + + // mutable: capacity()/maxCapacity() below are const accessors that still + // need to take this lock to read _table_cap/_table_max_cap safely. + mutable SpinLock _table_lock; + // 1 + highest index ever inserted (informational upper bound for + // lookup(); never shrinks, since tags/slots are never reused). atomic + // (not volatile) because insert() updates it via a CAS loop concurrently + // with plain reads from size()/resetForRestart() - a volatile int mixed + // with __sync_bool_compare_and_swap has no synchronizes-with edge under + // the C++ memory model, so those plain reads and the CAS are a genuine + // data race (caught by TSAN), even though relaxed/informational + // semantics are all that's needed here. + std::atomic _table_size; + int _table_cap; + int _table_max_cap; + FrontierEntry *_table; + + // Cumulative improveChain()/reparentToDurableRoot() refusals of a + // parent==tag SELF-EDGE (see improveChain's own guard). A single + // callback-delivered self-edge trips BOTH sibling guards (improveChain + // refuses, then the already-admitted block's else-if offers the same + // self-parent to reparentToDurableRoot), so the count climbs 2 per + // delivered edge - on the hotdog pod, the LEAK_BUFFER wrapper's own + // rotation walk delivers its mutex == this edge every pass, so the + // counter climbing is the verification that the guard fires on the real + // wrapper. Relaxed-atomic: the heap-callback writers and the engine + // thread's per-pass log reader only need a monotonic tally, not + // synchronization. + std::atomic _self_edge_guard_skips{0}; + + // Grows _table (doubling) until it holds at least `required_cap` slots or + // _table_max_cap is reached. Must be called with _table_lock held + // exclusively. Returns false (capacity exhausted) without partially + // resizing if `required_cap` exceeds _table_max_cap. + bool growLocked(int required_cap); + +public: + // `max_cap` <= 0 disables the table (capacity() stays 0, every insert() + // reports exhaustion) - callers are expected to guard on the config flag + // before constructing one, but this makes a misconfigured cap fail safe + // rather than crash. + explicit FrontierTable(int max_cap); + ~FrontierTable(); + + FrontierTable(const FrontierTable &) = delete; + FrontierTable &operator=(const FrontierTable &) = delete; + + // Cumulative self-edge guard refusals (see _self_edge_guard_skips). + u64 selfEdgeGuardSkips() const { + return _self_edge_guard_skips.load(std::memory_order_relaxed); + } + + // Writes (parent_tag, referrer_klass, depth, state) into the slot for + // `tag` (index = tag - 1), growing the table if needed. Returns false + // without writing anything if `tag` is not positive, or the table is + // already at max_cap and still too small for this tag - the design doc's + // frontier-size-cap requirement is "stop admitting new entries and report + // it", so this reports failure to the caller rather than crashing or + // silently dropping the write. + // `root_kind` is the jvmtiHeapReferenceKind of the admitting edge - only + // meaningful when `parent_tag == 0` (see FrontierEntry::root_kind's own + // comment); callers that are not admitting a root-attached entry can + // leave it at the default 0. `class_tag` is FrontierEntry::class_tag - see + // its own comment; defaults to 0 (unresolved) so call sites that do not + // participate in leak-accumulation matching (canary pruning, test seams) + // need no change. + bool insert(jlong tag, jlong parent_tag, u32 referrer_klass, u32 depth, + u8 state = FrontierEntryState::FRONTIER, u8 root_kind = 0, + jlong class_tag = 0, + jint referrer_field_index = -1, u8 edge_kind = 0, + jlong referrer_class_tag = 0); + + // Reads the slot for `tag` into *out. Returns false (leaving *out + // untouched) if `tag` is not positive or has never been inserted. + bool lookup(jlong tag, FrontierEntry *out); + + // Runs `fn(this)` with the shared lock held for the whole call, for a + // caller that needs to look up many tags back to back (e.g. the rotation + // collectors' O(size()) sweeps in referenceChains.cpp) under ONE lock + // acquisition, instead of paying SpinLock's lock/unlock cost on every + // single lookup() call. RAII (SharedLockGuard, spinLock.h) releases the + // lock on every exit path from `fn`, including an early return - unlike a + // manual lockShared()/unlockShared() pair, a `fn` that returns early can't + // leak the lock. `fn` should only call lookupLocked() on this table, never + // another FrontierTable method that tries to take the lock again. + template void withSharedLock(Fn &&fn) const { + SharedLockGuard guard(&_table_lock); + fn(this); + } + + // Same as lookup() above, but assumes the caller already holds the shared + // lock via withSharedLock() below. + bool lookupLocked(jlong tag, FrontierEntry *out) const; + + // Marks the slot for `tag` as ABANDONED in place. This is only the + // metadata-table side of tag release (design doc's Termination section); + // the caller is still responsible for SetTag(obj, 0) via + // ReferenceChainTracker::clearTag() - clear() here does not touch JVMTI. + // No-op if `tag` was never inserted. + void clear(jlong tag); + + // Marks the slot for `tag` as EDGE in place (design doc: "on a path + // toward a target sample (EdgeStore)"). No-op if `tag` was never + // inserted. Used by reconstructChain() below to mark every hop it walks. + void markEdge(jlong tag); + + // Marks the slot for `tag` as EXPANDED in place (design doc: "expanded; + // children (if any) are in the table") - the resumed-pass counterpart to + // markEdge(): ReferenceChainTracker::expandFrontier() calls this + // once an entry's own outgoing edges have been fully visited by a + // FollowReferences(initial_object=) call, so a later + // pass's scan for pending work (which only considers FRONTIER-state + // entries) skips it. No-op if `tag` was never inserted. + void markExpanded(jlong tag); + + // Overwrites the slot for `tag`'s root_kind in place, touching no other + // field - the durability-upgrade counterpart to insert()'s one-time + // root_kind write (design doc's "opportunistic upgrade during root + // re-enumeration", Phase 5 item 1). No-op if `tag` was never inserted. + // + // Callers MUST only invoke this when the update itself originates from a + // root discovery (a root callback rediscovering an already-tagged object + // as a heap root), never from an ordinary edge admission/re-expansion - + // and only on an entry that is already root-attached (parent_tag == 0). + // FrontierEntry::root_kind is documented as meaningful only when + // parent_tag == 0; this mutator does not itself touch parent_tag, so + // calling it from a non-root discovery context (e.g. an + // edge-driven re-expansion rediscovering an edge to an already-tracked, + // non-root-attached object) would silently leave a non-zero root_kind on + // an entry nothing else treats as root-attached. See + // ReferenceChainTracker::maybeUpgradeRootAttachedRootKind() (the sole + // caller) for how this is enforced. + void updateRootKind(jlong tag, u8 root_kind); + + // Set the leak tag on a frontier entry (the JVMTI tag assigned by + // LivenessTracker to this specific tracked leaking object). + void setLeakTag(jlong tag, jlong leak_tag) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return; + } + int idx = (int)(tag - 1); + _table_lock.lock(); + if (idx < _table_size) { + _table[idx].leak_tag = leak_tag; + } + _table_lock.unlock(); + } + + // Replace a shallow root-attached entry (parent_tag == 0, depth == 0) + // with a deeper chain-attached entry when the object is reached via a + // longer path. This fixes the "depth=1 chain with no holder" problem: + // an object first admitted as a JNI-local root (parent_tag == 0) gets + // its frontier entry overwritten when the static-field → ... → object + // path reaches it later with a non-zero parent_tag. + // Returns true if the entry was actually improved (new depth > old). + bool improveChain(jlong tag, jlong parent_tag, u32 referrer_klass, + u32 depth, u8 root_kind, jint referrer_field_index = -1, + u8 edge_kind = 0, jlong referrer_class_tag = 0); + + // Equal-depth re-parenting, the one case improveChain() above cannot + // express: a depth-1 entry whose current parent is a TRANSIENT root + // (stack local / JNI local - a momentarily-live frame) is re-parented to + // a DURABLE root-attached parent (static field, JNI global, thread) when + // one is seen admitting the same object at the same depth. The retention + // explanation of a depth-1 chain is entirely its root hop, so a transient + // root makes the chain noise even though the depth is legitimate for the + // real static path (the actual hotdog shape: a singleton collection is a + // depth-0 static root, its elements depth 1 - equal depth means + // improveChain() sees no improvement and the noise path would stick + // forever). Only depth-1 targets qualify: at deeper depths judging root + // durability would require walking both chains, not just two lookups. + // Returns true if the parent was swapped. + bool reparentToDurableRoot(jlong tag, jlong new_parent_tag, + u32 referrer_klass, + jint referrer_field_index = -1, u8 edge_kind = 0); + + // Walks parent_tag links starting at `target_tag` back to a root-attached + // entry (parent_tag == 0), appending each visited entry's referrer_klass + // to *out_chain in leaf-to-root order, and marking each visited entry + // EDGE via markEdge() - this table's degenerate EdgeStore (design doc: + // "a chain can be walked back from a target sample to a root by + // following parent_tag across EdgeStore records"). Returns false (leaving + // *out_chain untouched) if target_tag was never inserted. Bounds the walk + // at maxCapacity() hops as a defensive guard against a corrupted/cyclic + // parent_tag chain - nextTag() only ever hands out a strictly larger value + // than any tag already assigned (true across resumed passes too, not just + // within one), so a child's parent_tag always points at an + // already-existing, strictly smaller tag and a cycle should be + // unreachable in practice; this is not a correctness dependency. + // + // `out_root_kind` (if non-null) receives the root-attached entry's own + // FrontierEntry::root_kind - the jvmtiHeapReferenceKind of whichever edge + // first admitted this chain into the frontier, letting a caller label the + // chain with why it is reachable at all (JNI global, thread stack, static + // field, ...) instead of just how (the referrer_klass hops in *out_chain). + bool reconstructChain(jlong target_tag, std::vector *out_chain, + u8 *out_root_kind = nullptr, + std::vector *out_edges = nullptr); + + // Search restart (ReferenceChainTracker::restartSearch(), this class's own + // header comment): marks every slot unoccupied again without releasing + // _table's allocation - a new search's nextTag() sequence restarts at 1, + // reusing these same slot indices, so lookup()/insert() must not read back + // the previous search's now-irrelevant entries for them. Safe to call + // only once releaseSearchTags() has already cleared every live JVMTI tag + // this search owned (restartSearch()'s own caller ordering) - this method + // has no way to release tags itself, it only forgets the metadata table's + // record of them. + void resetForRestart() { + _table_lock.lock(); + _table_size.store(0, std::memory_order_relaxed); + _table_lock.unlock(); + } + + // Debug-only test seam (ReferenceChainTracker::resetSearchStateForTest()). + // Unlike resetForRestart(), which only forgets this table's occupancy, + // this discards the table's whole allocation and rebuilds it at + // `max_cap` - the only way to undo the "sized once, on the first start() + // in this JVM" capacity choice (this class's own constructor comment) + // that a differently-configured test running earlier in the same, + // no-forkEvery JVM (ProfilerTestPlugin.kt) would otherwise leave every + // later test permanently stuck with. Defined in referenceChains.cpp + // alongside the constructor it mirrors. + void resetCapacityForTest(int max_cap); + + // _table_cap/_table_max_cap are plain ints, not atomics like _table_size, + // and resetCapacityForTest() (debug-only test seam, see its own comment) + // rewrites both under _table_lock after freeing/reallocating _table. Every + // other reader of these fields (growLocked() and its callers) already + // holds _table_lock; these two accessors take it too so a concurrent + // resetCapacityForTest() during shared-JVM test overlap can't race an + // unsynchronized read here. + int capacity() const { + _table_lock.lock(); + int cap = _table_cap; + _table_lock.unlock(); + return cap; + } + int maxCapacity() const { + _table_lock.lock(); + int max_cap = _table_max_cap; + _table_lock.unlock(); + return max_cap; + } + + // Current upper bound on assigned slots (mirrors _table_size's own + // comment: "1 + highest index ever inserted"). expandFrontier() + // uses this to know how far a resumed pass's scan for FRONTIER-state + // entries needs to go. Relaxed/informational like _table_size itself: a + // concurrent insert() racing this read only makes the caller's scan + // window one tag short for this call, which self-corrects on the next + // call once _table_size has caught up. + int size() const { return _table_size.load(std::memory_order_relaxed); } +}; + +// Tag-indexed table mapping a *class* tag (see +// ReferenceChainTracker::nextClassTag() - always negative, a namespace +// disjoint from the positive FrontierTable object tags above so a raw tag +// value alone always tells the heap-walk callback which table it belongs +// to) to the StringDictionary id of that class's resolved name +// (Profiler::classMap(), the same interning table LivenessTracker uses via +// Profiler::lookupClass(), livenessTracker.cpp:120-122 - see Open Item 2 in +// the implementation plan). +// +// Populated once per loaded class by +// ReferenceChainTracker::resolveLoadedClasses() - a GetLoadedClasses() + +// GetClassSignature() pass run *before* FollowReferences starts, specifically +// so heapReferenceCallback() (referenceChains.cpp) never needs a class-name +// lookup of its own: GetClassSignature is a JNI/Class-category call, and the +// JVMTI spec forbids Heap-callback functions like heapReferenceCallback from +// calling anything but "callback safe" functions (see the header comment +// above) - resolving names inline inside the callback is not an option. +// +// Concurrency: like FrontierTable, only ever touched by the single +// agent-owned BFS thread (design doc's Algorithm "Thread" bullet), so no locking is +// needed - unlike FrontierTable there is also no cross-thread reader to +// guard against (chain reconstruction only needs FrontierTable). +class ClassTagTable { +private: + std::unordered_map _table; + +public: + void insert(jlong class_tag, u32 dict_id) { _table[class_tag] = dict_id; } + + // Returns the StringDictionary id for `class_tag`, or 0 if it was never + // inserted (0 is StringDictionary's own "no entry" sentinel too, so this + // composes with FrontierEntry::referrer_klass's documented 0 = + // unresolved/none convention without a separate "found" out-parameter). + u32 resolve(jlong class_tag) const { + auto it = _table.find(class_tag); + return it != _table.end() ? it->second : 0; + } + + size_t size() const { return _table.size(); } + + // Drops every cached class_tag -> dict_id mapping - used when the + // underlying StringDictionary itself was reset (see + // ReferenceChainTracker::_last_class_map_generation's comment) and every + // id here now points at a namespace that no longer exists. + void clear() { _table.clear(); } +}; + +// Singleton shape mirrors LivenessTracker (livenessTracker.h). +class ReferenceChainTracker { + // Test-only accessor (referenceChains_ut.cpp), mirroring vmEntry.h's + // VMTestAccessor pattern: since instance() is a process-wide singleton, + // the search-lifecycle fields (_search_state/_search_started/etc.) + // would otherwise leak across separate TEST_F cases in the same gtest + // binary. The accessor resets them back to their just-constructed values + // between tests; it does not change any production behavior. + friend class ReferenceChainsTestAccessor; + +private: + bool _enabled; + + // Frontier metadata table. Constructed lazily on the first + // start() with the flag enabled, sized from + // args._reference_chains_frontier_cap; like LivenessTracker's table + // (livenessTracker.cpp:209-210) it survives stop() so it persists across + // multiple start/stop recording cycles. + FrontierTable *_frontier; + + // args._reference_chains_frontier_cap as of the most recent start() call - + // recorded unconditionally (even once _frontier already exists and start() + // itself skips reconstructing it), so resetSearchStateForTest() has + // something to rebuild the table at other than whatever cap the first + // start() in this JVM happened to use (see _frontier's own comment). + int _configured_frontier_cap; + + // Class-tag -> StringDictionary id table. Populated by + // resolveLoadedClasses(), read by heapReferenceCallback(). Survives + // stop()/start() cycles for the same reason _frontier does - a class, + // once resolved, does not need re-resolving just because the profiler + // recording was restarted - UNLESS the underlying dictionary itself was + // reset (see _last_class_map_generation below), in which case every id + // cached here is for an id namespace that no longer exists. + ClassTagTable _class_tags; + + // Profiler::classMap()'s generation as of the last resolveLoadedClasses() + // call. Profiler::start() calls _class_map.clearAll() (profiler.cpp) + // whenever `reset || _start_time == 0`, which restarts that + // StringDictionary's id namespace at 1 - but a class's JVMTI-level + // class-object tag (GetTag(klass, ...)) is JVM-level state, untouched by + // that reset, so resolveLoadedClasses()'s "already tagged -> already + // resolved, skip it" check (tag == 0) would otherwise keep _class_tags + // pointing at ids from a dictionary generation that clearAll() already + // wiped. resolveLoadedClasses() compares this against + // Profiler::instance()->classMap()->generation() and, on a mismatch, + // re-resolves every loaded class's name (reusing its existing tag rather + // than assigning a new one) instead of only the untagged ones - see that + // method's own comment. Initialized to 0 (StringDictionary's own initial + // generation), not a sentinel, since a resolveLoadedClasses() call before + // any clearAll() has ever run must NOT treat that as a mismatch. + u64 _last_class_map_generation; + + // GetLoadedClasses() count as of the last resolveLoadedClasses() call that + // actually ran its per-class GetTag()/GetClassSignature() scan - lets that + // method skip the scan entirely on a resumed pass where the loaded-class + // count has not CHANGED (see resolveLoadedClasses()'s own comment for why + // this must be an equality check, not just a "grew" check: the count is + // not monotonic once class unloading is in play). Survives stop()/start() + // cycles for the same reason _class_tags does. Written and read only from + // the single BFS thread, like _last_pass_gc_finish_epoch. Forced to -1 + // (a value class_count, always >= 0, can never equal) by a + // _last_class_map_generation mismatch, so the scan is never skipped on the + // very call that must re-resolve every already-tagged class. + int _last_resolved_class_count; + + // GetLoadedClasses() count as of the last runPassManualWalk() call whose + // admitStaticFieldRoots() sweep actually ran (i.e. was not skipped by the + // guard below) AND completed without being truncated. Distinct from + // _last_resolved_class_count even though both are populated from the same + // GetLoadedClasses() count: resolveLoadedClasses() runs once per runPass() + // unconditionally (it is cheap to skip its own per-class scan once + // unchanged), whereas admitStaticFieldRoots() re-walks EVERY loaded class + // via FollowReferences - a stop-the-world HeapWalkOperation - so + // runPassManualWalk() only calls it at all when this differs from + // resolveLoadedClasses()'s freshly-observed _last_resolved_class_count, + // i.e. only when the loaded-class set has actually changed since the last + // completed sweep. Left unset (mismatched) on a truncated sweep so the + // next pass retries rather than silently treating a still-incomplete sweep + // as done. Initialized to -1 (a value class_count, always >= 0, can never + // equal) so the very first pass always runs the sweep once. + int _last_static_field_class_count; + + // Index into the (per-call, app-classes-first-partitioned) loaded-class + // list that admitStaticFieldRoots() resumes from on its next call - see + // that method's own comment for why a single FollowReferences over every + // loaded class at once (no cursor) could never finish within one pass's + // safepoint deadline on a JVM with tens of thousands of loaded classes. + // Wrapped back to 0 once a chunk reaches the end of the current + // GetLoadedClasses() count. Clamped to 0 if the loaded-class count shrinks + // below the cursor (classes unloaded) rather than reading out of range. + int _static_field_sweep_cursor; + + // Set when any chunk within the current lap (the cursor's walk from 0 + // back to 0) truncates. Read when the cursor wraps: a lap that truncated + // even once must not mark _last_static_field_class_count as done - the + // next lap starts immediately (cursor is already back at 0) to keep + // retrying, same "no silent truncation" contract the untruncated case + // documents. Cleared at the start of each new lap. + bool _static_field_sweep_cycle_truncated; + + // Per-call cap on how many classes admitStaticFieldRoots() includes in one + // FollowReferences call - see that method's own comment. Provisional and + // unbenchmarked like this subsystem's other per-pass caps (e.g. + // ROOT_KIND_ROTATION_BUDGET): small enough that building the holder array + // and walking one chunk's static fields fits comfortably inside the 5-50ms + // per-pass safepoint deadline even when a class in the chunk has an + // unusually large static-field graph, large enough that a JVM with a + // realistic loaded-class count (tens of thousands) completes a full lap in + // well under a minute of wall-clock passes. + static constexpr int STATIC_FIELD_SWEEP_CHUNK_CLASSES = 512; + + // Per-class cap on non-STATIC_FIELD edges admitted during one + // admitStaticFieldRoots() lap. STATIC_FIELD edges are always admitted + // (high-priority leak root); non-static edges (CONSTANT_POOL, INTERFACE, + // SUPERCLASS, CLASS_LOADER, ...) are admitted up to this many per class, + // then dropped for the rest of that class this lap. 32 covers a typical + // class's full constant-pool/interface set; outlier classes are bounded + // so they cannot blow the chunk's safepoint deadline. See + // PassContext::_class_other_cap's own comment for the admission logic. + static constexpr int STATIC_FIELD_SWEEP_NON_STATIC_CAP_PER_CLASS = 32; + + // "GC just happened" signals. Bumped only from onGCStart()/onGCFinish(); + // gcFinishEpoch() is now read by shouldRunPass() as one of the + // two pass-scheduling triggers (design doc's Triggering section). + volatile u64 _gc_start_epoch; + volatile u64 _gc_finish_epoch; + + // Monotonically increasing tag source for frontier objects. 0 is reserved + // (JVMTI convention: an untagged object reads back tag 0, and + // SetTag(obj, 0) clears a tag), so this starts at 1. Always hands out + // positive values - see nextClassTag() below for why classes use a + // disjoint (negative) range instead of sharing this counter. + volatile jlong _next_tag; + + // Per-pass tunables, copied from Arguments in start() (design doc: Open + // Question 2 defaults, from the config-flag scaffolding). A future + // measurement pass will decide whether/how these can change between passes + // of the same search; for now this only needs one fixed value per + // start()/stop() cycle, exactly like LivenessTracker's _subsample + // (livenessTracker.h). + int _hop_cap; + int _budget; + + // Edge budget for just the search's one-shot, root-seeded first pass + // (runPass()'s !_search_started branch) - copied from + // Arguments::_reference_chains_first_pass_budget in start(), auto-scaled + // from _budget (AUTO_FIRST_PASS_BUDGET_MULTIPLIER, capped at + // AUTO_FIRST_PASS_BUDGET_CAP) when unset (0), rather than falling back to + // plain _budget: a steady-state per-pass budget sized for cheap incremental + // expansion truncates a cold root-seeded walk of a real JVM's object graph + // long before it reaches anything interesting (see + // ddprof-stresstest's ReferenceChainLeakDemo, whose whole class comment is + // about exactly this trap). Only the first pass's own edge budget is this + // large - runPassManualWalk()'s IterateOverReachableObjects root/stack-ref + // enumeration itself reruns on every pass, first or resumed (see its own + // comment); already-admitted roots short-circuit cheaply via + // admitObject()'s ALREADY_ADMITTED check, so a root this pass doesn't + // reach before the budget runs out is still picked up by a later pass, not + // permanently lost. Unlike _budget/_effective_budget, this is spent at + // most once per search, not once per pass, so a much larger ceiling is + // affordable without the per-pass pacing controller (updatePacing()) ever + // seeing it - runPass() deliberately excludes the first pass's own + // duration from that signal (see runPass()'s own comment) so a large + // first-pass cost cannot throttle down every cheap expansion pass that + // follows. + int _first_pass_budget; + + // Wall-clock TTL, copied from Arguments in start() (design doc's + // Termination section: "a hard cap on passes-per-search or wall-clock TTL + // from first observation"). This implements the TTL half of that + // "or" - the config-flag scaffolding only added a TTL sub-option (no + // separate pass-count cap), and Open Question 2 leaves the choice between + // the two open pending a future measurement pass. <= 0 disables the TTL + // cutoff (a search can only still end via the frontier-size cap or natural + // completion). + long _ttl_ms; + + // Pause-time pacing controller: pause-time-SLO ceiling copied from + // Arguments in start() (Arguments::_reference_chains_pause_target_ms) - + // the "single target ceiling" the plan asks for in place of guessing + // _budget/PASS_CADENCE_NS directly. Used only to (re)construct _pause_pid + // in start(); updatePacing() itself never reads it again, since it lives + // inside _pause_pid's own _target once constructed. + long _pause_target_ms; + // Runtime-adjusted pause target: when isUrgent(), this is bumped + // to URGENT_PAUSE_TARGET_MS so each pass can explore more edges. + // Restored to _pause_target_ms (the configured default) once + // urgency clears. The PID controller (_pause_pid) is + // reconstructed whenever this changes so its ceiling tracks + // the new target. + long _effective_pause_target_ms; + // Passes since the frontier last grew. Reset to 0 whenever + // _frontier->size() increases (new entries admitted). + // When this exceeds NO_PROGRESS_PASS_LIMIT without + // the frontier growing, the search is genuinely + // stuck (not just slow) and is abandoned. + int _passes_since_last_progress; + // Passes since _candidate_found_bits last changed (a candidate was newly + // found, or a new candidate was admitted into a slot). Distinct from + // _passes_since_last_progress above: that one tracks whole-graph frontier + // growth, which keeps resetting to 0 for as long as there is any unvisited + // reachable object left, regardless of whether the canary's specific + // candidates are ever pruned - so it never reflects "the canary search + // itself is stuck", only "the graph walk is stuck". Read by runPass()'s + // canary-stuck check, which (unlike the ordinary TTL check just above it) + // is deliberately NOT suppressed by isUrgent(): the isUrgent() TTL + // suppression exists so a search that's still making real progress isn't + // killed just because the process is close to OOM, but a canary search + // that has made zero candidate-discovery progress for + // CANARY_NO_PROGRESS_PASS_LIMIT consecutive passes is provably not + // converging - continuing to run it at urgency-boosted budget/cadence only + // burns STW pause budget the rest of the process needs during the same + // OOM approach this search was launched to diagnose. + int _passes_since_last_candidate_progress; + // candidate_count + popcount(found_bits) as of the last pass this was + // updated. A monotonic non-decreasing marker: _candidate_count only grows + // (pollWatchedTargets()'s admission loop never retires a slot) and + // found_bits only gains bits (heapReferenceCallback() only ever ORs a bit + // in) for as long as a search is RUNNING, so a rising sum means real + // canary progress happened since the last check; an unchanged sum means + // none did. + int _last_candidate_progress_mark; + + // Canary-lane pass pacing, work-scaled: the chase's inter-pass spacing is + // _canary_backoff_mult x _canary_pass_ema_ms - a multiple of what a pass + // actually COSTS, not a fixed wall-clock constant. The multiplier starts + // at 1 (back-to-back: the next pass starts as soon as the last one ended, + // which a cheap pass makes harmless by construction), doubles on each + // pass with NO candidate progress up to CANARY_BACKOFF_MULT_MAX, and + // resets to 1 on any candidate progress (a candidate found or a new + // candidate admitted). shouldRunPass() holds a canary pass off while + // now - _last_canary_pass_ns is below the spacing. + // + // Why work-scaled rather than a fixed cap: a fixed cap only binds when + // it exceeds the pass's own duration - measured live on hotdog (rounds + // 3-4), an un-findable candidate held the chase open for 32 minutes at + // ~88 passes/min (a full core; round 4's passes ran 0.7-4s each, so a 1s + // cap would have changed nothing at all - the loop is work-bound, never + // sleep-bound, when the pass itself exceeds the cap). Scaled against the + // pass's measured cost, the burn bound is structural: at the multiplier + // cap the chase spends <= 1/CANARY_BACKOFF_MULT_MAX of a core on pass + // work, whatever that work is - ~6% on the pod, and proportionally less + // blocking for a deep-but-cheap chase whose passes cost milliseconds + // (those keep a dense, fast-resolving chase at the same multiplier). + int _canary_backoff_mult; + // 0.8/0.2 EMA of each pass's whole-call wall duration in ms + // (TSC::ticks_to_millis(pass_wall_ticks), runPass()), updated every pass + // - kept warm regardless of canary state so a chase that opens on a + // known-cost crawl sizes correctly from its first held-off decision. + u64 _canary_pass_ema_ms; + // End-of-pass timestamp of the last pass that ran with a canary chase + // still open (the reference point for the spacing above). 0 = no canary + // pass has run since the search (re)started. + u64 _last_canary_pass_ns; + // Whether threadLoop()'s OOM urgency ramp is currently active, set each + // loop iteration BEFORE shouldRunPass() (same thread, no atomics + // needed). While set, shouldRunPass()'s canary-backoff gate is bypassed: + // imminent OOM is the one regime where the chase is deliberately allowed + // to burn budget back-to-back, exactly as before the backoff existed. + bool _oom_ramp_active; + + // How many consecutive times in a row runPass()'s canary-stuck check + // (CANARY_NO_PROGRESS_PASS_LIMIT below) has abandoned this same + // candidate-chase sequence. Never reset by restartSearch() - it must + // survive across restarts to make the escalation in + // canaryStuckPassLimit() actually widen with repeated failures. Reset to + // 0 whenever a search leaves RUNNING for a reason other than + // CANARY_STUCK (natural completion, candidate-complete, frontier cap, + // TTL) - see the terminal-state block in runPass(). + int _canary_stuck_restart_count; + + // Canary-search candidate set: pre-tagged with distinct + // marker tags (MARKER_TAG_BASE - i) before the walk, applied to each + // candidate's specific representative object (identity match) - matching + // by class alone would let the walk record a chain for an unrelated, + // possibly short-lived, instance of the same class instead of the one + // LivenessTracker actually flagged as growing. + // _candidate_found_bits is a packed bitmap (bit i = candidate i found). + // _candidate_tags[i] holds the marker tag (MARKER_TAG_BASE - i) for candidate i. + // _candidate_frontier_tags[i] holds the frontier tag assigned by + // heapReferenceCallback() when it pruned the candidate (equal to the + // marker tag itself, since the marker is already a unique table key). + // Reset by resetSearchStateForTest(). + // + // MAX_LEAK_CANDIDATES_FROM_LT must match + // LivenessTracker::MAX_LEAK_CANDIDATES (livenessTracker.h:133). + // Duplicated here to avoid a heavy include chain. + static constexpr int MAX_LEAK_CANDIDATES_FROM_LT = 5; + + // How many klass_ids _watched_leak_klass_ids tracks at once - matches + // LivenessTracker::MAX_LEAK_CANDIDATES (livenessTracker.h), the cap + // LivenessTracker::topKlassesByGenerationCount() itself already enforces; + // duplicated here for the same reason MAX_LEAK_CANDIDATES_FROM_LT above + // already duplicates it, rather than depending on a private LivenessTracker + // constant. + static constexpr int MAX_WATCHED_LEAK_KLASSES = 5; + int _candidate_count; + u64 _candidate_found_bits; + // klass_id occupying each slot, so pollWatchedTargets() can tell whether a + // klass_id selectLeakCandidates() returns this poll already has a slot + // (and must not be re-tagged/re-admitted) or is new (and should be + // admitted into the next free slot). Slots are never retired or reused + // once assigned for the lifetime of a search - see pollWatchedTargets()'s + // admission loop for why. + u32 _candidate_klass_ids[MAX_LEAK_CANDIDATES_FROM_LT]; + jlong _candidate_tags[MAX_LEAK_CANDIDATES_FROM_LT]; + jlong _candidate_frontier_tags[MAX_LEAK_CANDIDATES_FROM_LT]; + // Per-candidate chain link recorded at pruning time: + // parent_tag (referrer's frontier tag, positive) and referrer_klass. + // Used by buildCanaryChainEvent() to reconstruct the + // chain without a frontier table lookup on the + // negative marker tag (which lookup() rejects). + jlong _candidate_parent_tags[MAX_LEAK_CANDIDATES_FROM_LT]; + u32 _candidate_referrer_klasses[MAX_LEAK_CANDIDATES_FROM_LT]; + u32 _candidate_depths[MAX_LEAK_CANDIDATES_FROM_LT]; + + // Per-SLOT snapshot of the qualifying allocating-thread tids + // LivenessTracker::selectLeakCandidates() reported for that slot's klass + // this poll, refreshed by pollWatchedTargets() (zeroed first, then filled + // from the poll's candidates so a klass that stops qualifying stops + // walking its tids too). The BFS-thread walk phases run between polls + // under the same engine serialization and read this snapshot to reach + // the exact (klass, tid) scope tagLeakInstances() tags - see + // walkCandidateThreadLocals()'s own comment for why that reach matters. + // Sized generously above LivenessTracker's + // KlassCandidate::MAX_QUALIFYING_TIDS (= 8, itself + // KlassPopulationEntry::MAX_TID_TRENDS): the snapshot loop clamps, so + // this bound only needs to be "large enough" (same pattern as + // kMaxWatchedCandidates in pollWatchedTargets()). + static constexpr int MAX_CANDIDATE_QUALIFYING_TIDS = 16; + jint _candidate_qualifying_tids[MAX_LEAK_CANDIDATES_FROM_LT] + [MAX_CANDIDATE_QUALIFYING_TIDS]; + int _candidate_qualifying_tid_count[MAX_LEAK_CANDIDATES_FROM_LT]; + + // tid -> JNI global reference to the live java.lang.Thread object, fed + // from Profiler::onThreadStart/onThreadEnd + // (registerThreadObject()/unregisterThreadObject()). A strong global + // reference to a live thread's Thread object does not distort reachability + // (a running thread's Thread object is reachable via the VM anyway) and is + // released at ThreadEnd, so it cannot outlive the thread. Walked by + // walkCandidateThreadLocals() as the FollowReferences initial object of a + // bounded descend walk - the candidate-scoped reach mechanism (see + // descendFromAnchor()'s own comment). Mutex-guarded rather than relying + // on the engine lock: ThreadStart/End fire on arbitrary JVM threads, + // outside the engine serialization the walk phases run under. + Mutex _thread_objects_lock; + std::unordered_map _thread_objects; + + // Auto-marked instances: when the BFS walk discovers ANY object whose + // class matches a watched leak class (not just the pre-tagged + // representative), its frontier tag is recorded here so pollWatchedTargets() + // can build chain events for all of them. A leaking class typically has + // many live instances, and each one's reference chain is independently + // useful for diagnosis — the pre-tagged representative is just one sample, + // and its chain may differ from other instances' chains (different parents, + // different retention paths). Fixed-size per slot to avoid heap allocation + // in the callback (safepoint context). When the per-slot array fills, + // further instances are silently dropped (the representative + up to + // MAX_DISCOVERED_INSTANCES_PER_CLASS others is still far more coverage + // than the single-representative design it replaces). + static constexpr int MAX_DISCOVERED_INSTANCES_PER_CLASS = 8; + jlong _candidate_discovered_tags[MAX_LEAK_CANDIDATES_FROM_LT] + [MAX_DISCOVERED_INSTANCES_PER_CLASS]; + int _candidate_discovered_count[MAX_LEAK_CANDIDATES_FROM_LT]; + + // klass_ids from LivenessTracker::topKlassesByGenerationCount() (a faster, + // un-hysteresis-gated ranking than the canary candidate set above - see + // that method's own comment), refreshed once per BFS-thread tick but only + // once hasLeakSignal() has already fired via the slower, hysteresis-gated + // selectLeakCandidates() path (per design discussion: this whole mechanism + // only cranks once the trend detector has already triggered, so it never + // needs to wait out that same hysteresis a second time on its own). + // Consulted at admission time (heapReferenceCallback()) to decide whether + // a newly-admitted object's class is worth tracking for rotation priority + // - see _leak_signature_totals/_leak_parent_fanout's own comments for what + // that tracking actually does. Whenever a klass_id newly enters this + // array (was not present in the previous refresh), pollWatchedTargets() + // also calls seedLeakAccumulationForNewlyWatchedKlass() for it once - see + // that method's own comment for why admission-time tracking alone cannot + // see objects admitted before watching started. _watched_leak_klass_count + // entries are valid; the rest of the array is unspecified. + u32 _watched_leak_klass_ids[MAX_WATCHED_LEAK_KLASSES]; + int _watched_leak_klass_count = 0; + + // Packs a (leaf_klass_id, parent_class_id) pair into one map key for + // _leak_signature_totals/_leak_signature_prev_totals below - both are + // StringDictionary ids (u32), so this never loses information and avoids + // defining a custom hash/equality functor for a 2-field struct key. + static u64 leakSignatureKey(u32 leaf_klass_id, u32 parent_class_id) { + return ((u64)leaf_klass_id << 32) | (u64)parent_class_id; + } + + // Tier 1 of the leak-accumulation rotation design (see + // collectLeakAccumulationCandidatesForRotation()'s own comment for the + // full design): aggregate, per (leaf_klass_id, parent_class_id) signature + // - not per object - how many admitted children of that leaf klass_id + // have been observed under a parent of that class. Incremented at + // admission time (heapReferenceCallback(), O(1) per matching new + // admission - see _watched_leak_klass_ids' own comment), so this reflects + // the CURRENT cumulative total, never decreasing within a search. Small: + // bounded by (distinct leaf klass_ids ever watched) x (distinct parent + // classes ever seen holding one of them) - nowhere near the whole + // table's size, even for a common leaf class held by many unrelated + // parents, because it aggregates BY CLASS, not by individual parent + // object (that finer granularity is _leak_parent_fanout below). + std::unordered_map _leak_signature_totals; + + // Snapshot of _leak_signature_totals as of the END of the previous pass - + // runPassManualWalk() computes each signature's delta (totals - this) to + // rank signatures by growth before rolling this forward to the current + // totals for the next pass's comparison. A signature with no prior + // snapshot (brand new this pass) is treated as prev_total == 0, so its + // delta is its whole total - correctly ranks a suddenly-appearing + // signature as growing, without a special first-seen case. + std::unordered_map _leak_signature_prev_totals; + + // Tier 2 of the leak-accumulation rotation design: per PARENT TAG (not + // per class), how many admitted children of a watched leaf klass_id this + // specific parent object holds, plus which signature it belongs to (so + // rotation-selection can filter to the pass's winning signature without a + // second lookup). Incremented at the same admission-time hook as + // _leak_signature_totals above. Bounded by however many distinct parent + // objects have ever been observed holding a watched leaf klass_id - + // small in practice even for a common leaf type, since it is scoped to at + // most MAX_WATCHED_LEAK_KLASSES specific klass_ids, not every collection- + // shaped class in the JVM (the structural heuristic this design replaced - + // see git history for why that was measured and found not to work). + struct LeakParentFanoutEntry { + u64 signature_key; + u32 fanout; + }; + std::unordered_map _leak_parent_fanout; + + // Rotating skip-count over _leak_parent_fanout's iteration order for + // collectStaleExpandedEntriesForRotation()'s leak-parent-priority tier - + // advances by the number of parents selected each pass so, across passes, + // every fanout parent gets re-walked within ceil(fanout_size/budget) + // passes instead of only whichever entries the hash iteration happens to + // yield first. Same single-BFS-thread access as the fanout map itself. + u64 _leak_parent_rotation_cursor = 0; + + // Pause-time pacing controller: the actual per-pass budget runPass() passes + // to FollowReferences/expandFrontier(), replacing _budget's old role as a + // literal per-pass value - _budget above becomes this controller's ceiling + // instead (never exceeded, see updatePacing()), while this field is what + // updatePacing() actually raises/lowers pass to pass. Starts at _budget in + // start(), so a tracker that has not measured a pass yet behaves exactly as + // before the pacing controller was added. + int _effective_budget; + + // Pause-time pacing controller: the actual fallback cadence + // shouldRunPass()/threadLoop() compare against, replacing the fixed + // PASS_CADENCE_NS constant below in that role once updatePacing() starts + // adjusting it - see PASS_CADENCE_NS's own comment for why that constant + // survives as this field's starting value rather than being deleted + // outright. + u64 _effective_cadence_ns; + + // Budget-borrowing: extra headroom updatePacing() has temporarily granted + // above _budget's own ceiling, earned by a sustained run of comfortably- + // under-target passes (see BORROW_WARMUP_PASSES's own comment). This is + // the one exception to "_budget is never exceeded" (_effective_budget's + // own comment) - it exists because a fast-growing frontier (a real + // leaking-cache workload, not just a synthetic one) can otherwise starve + // under a steady-state budget sized for ordinary incremental expansion, + // never converging within this search's TTL even though pause time has + // visible headroom to spare. Revoked immediately (reset to 0, see + // updatePacing()) the moment a pass is no longer comfortably under target, + // so a search that starts abusing its pause-time budget loses the + // borrowed headroom before the very next pass - _budget itself remains the + // hard ceiling in that case, same as before this field existed. + int64_t _borrowed_budget; + + // Budget-borrowing: number of consecutive passes (since the last reset) + // that came in comfortably under _pause_target_ms (see + // BORROW_UNDER_TARGET_FRACTION). Reset to 0 the moment a pass does not + // qualify - see _borrowed_budget's own comment on why this must be a + // consecutive-run counter, not a cumulative one: a single expensive pass + // means the frontier is not, in fact, converging with room to spare, and + // borrowing more budget for the next pass on the strength of an unrelated + // earlier streak would defeat the point of gating growth on *sustained* + // headroom at all. + int _consecutive_under_target_passes; + + // Pause-time pacing controller: this tracker's own PidController instance - + // see updatePacing() + // below for the full mechanism, and PASS_CADENCE_NS's neighboring + // constants for why its gains are not copied from ObjectSampler/ + // MallocTracer/NativeSocketSampler's shared triple. Placeholder-constructed + // here (target=1, unit gains); start() reconstructs it once + // _pause_target_ms is known, mirroring RateLimiter's own + // placeholder-then-reconstruct pattern (rateLimiter.h's + // `_pid{1, 1.0, 1.0, 1.0, 1, 1.0}` member default, replaced in + // RateLimiter::start()). + PidController _pause_pid; + + // Self-calibrating adaptive batch sizing for GetObjectsWithTags (see + // expandFrontier()'s own comment). GetObjectsWithTags iterates the whole + // JVMTI tag map per call, so its cost has a batch-independent floor that + // grows with the frontier (~20-25ms at a 225-245k-entry map, measured live + // on hotdog) plus a small per-searched-tag component (measured live: + // batch 8 -> 27.4ms, batch 72 -> 36.1ms, i.e. ~0.12ms per extra tag on a + // ~25ms floor). Two earlier designs both collapsed: + // - per-TAG EMA (ema = elapsed / batch_size): the floor dominates at + // small batch sizes, so a smaller batch INFLATES the per-tag cost, + // shrinking the batch further (observed live: ~400 -> 2). + // - per-CALL AIMD against a FIXED budget: once the map grows enough that + // the floor alone exceeds the budget (25ms budget vs ~27ms floor at a + // 243k-entry map), every call "overran" regardless of batch size, so + // AIMD ratcheted to GOTW_MIN_BATCH and stayed there - measured live + // batch=8 on every call while batch=72 cost only +36% for 9x the + // objects. + // In the floor-dominated regime the right move is the OPPOSITE of + // shrinking: a bigger batch amortizes the floor. The control law is a + // direct proportion - scale the batch so ONE call fills the remaining + // wall-clock window: + // ema_call_ns = ema × 0.8 + measured × 0.2 (per-CALL, not per-tag) + // window_ns = remaining _pass_deadline_ns (nominal GOTW_CPU_BUDGET_NS + // when no deadline is set) + // batch = clamp(batch × window_ns / ema_call_ns, MIN, MAX) + // Converges upward while calls come in under the window (a near-free + // call scales the batch to GOTW_MAX_BATCH), shrinks as the window + // drains so tail calls still fit, and tracks the floor automatically + // as the tag map grows or shrinks. + // + // The window itself is computed by gotwWindowNs() below, which widens it + // under backlog pressure when the measured per-call floor already + // exceeds the remaining pass window - see that method's own comment. + size_t _gotw_batch_size = 0; // 0 = unset, use GOTW_INITIAL_BATCH_SIZE + u64 _gotw_ema_call_ns = 0; // EMA of per-call elapsed, 0 = unset + + // Nominal per-call window for the proportional batch control above when + // no phase deadline is set (expandFrontier()'s window is the REMAINING + // deadline, which the phases refresh per invocation). Also doubles as a + // CPU-overhead sanity target: ~25ms per call at ~2-4 calls/pass is + // ~50-100ms/sec of CPU overhead on one core, acceptable for a background + // search thread on a multi-core machine. + static constexpr u64 GOTW_CPU_BUDGET_NS = 25000000; // 25ms + + // Effective window for the proportional batch control (expandFrontier() + // calls this with the remaining pass deadline and the depth of the lane + // the next GetObjectsWithTags call will drain). The remaining-deadline + // window can be SMALLER than the measured per-call floor - measured live + // on hotdog (round 4, ev-leaktag-onpod-round4): a ~10ms expand window vs + // a 14-41ms floor at a 242k-entry tag map collapsed the proportional law + // to GOTW_MIN_BATCH on every call (batch=8 forever) while the lane it + // was draining held 127k entries - a self-inflicted ~120-200 + // objects/min drain. In that regime the floor is paid by EVERY call + // regardless of batch size, so the right move is to AMORTIZE it: when + // the lane is deeper than GOTW_BACKLOG_MIN_DEPTH and the floor exceeds + // the remaining window, widen the window to EMA x + // GOTW_BACKLOG_WINDOW_MULT so the proportional law sizes the batch UP + // (batch = calib x window/ema = calib x MULT - the floor's per-tag + // surcharge, ~0.12ms/tag measured, stays the real limit). The widening + // only sizes the NEXT batch; _pass_deadline_ns itself is unchanged, so + // the per-pass overrun is bounded by one widened call, and it never + // applies to shallow lanes (rotation fast-lane batches stay + // deadline-sized, keeping targeted re-walks cheap and frequent). + u64 gotwWindowNs(u64 remaining_ns, size_t lane_depth) const { + u64 window_ns = remaining_ns != 0 ? remaining_ns : GOTW_CPU_BUDGET_NS; + if (lane_depth >= GOTW_BACKLOG_MIN_DEPTH && + _gotw_ema_call_ns > window_ns) { + window_ns = std::max(window_ns, + _gotw_ema_call_ns * GOTW_BACKLOG_WINDOW_MULT); + } + return window_ns; + } + + // Lane depth beyond which gotwWindowNs()'s backlog widening applies. + // Above this, the lane itself proves that draining it matters more than + // keeping each pass strictly inside its remaining window; below it the + // ordinary proportional law applies unchanged. Chosen an order of + // magnitude above the rotation inflow per pass (~25-64 entries) so + // ordinary rotation cycling never widens the window. + static constexpr size_t GOTW_BACKLOG_MIN_DEPTH = 4096; + + // How many measured per-call floors one widened window may cost - see + // gotwWindowNs() above. 3x converges the batch upward by 3x per call in + // the floor regime (8 -> 24 -> 72 -> 216 -> GOTW_MAX_BATCH) while keeping + // a single call's overrun bounded to a small multiple of what the floor + // already forced. + static constexpr u64 GOTW_BACKLOG_WINDOW_MULT = 3; + + // Conservative initial batch_size before the first GetObjectsWithTags + // measurement. Small enough to be safe on any machine regardless of + // tag-map size, large enough to make meaningful progress per + // FollowReferences call. + static constexpr int GOTW_INITIAL_BATCH_SIZE = 64; + // AIMD bounds for the adaptive batch. The cap bounds JNI local refs + // (resolved objects + holder array) per call; the floor keeps a + // degenerate tiny batch from making each FollowReferences call + // resolve a single object (observed live: batch=2 collapsed BFS + // throughput ~20x while per-call cost stayed ~20ms). + static constexpr size_t GOTW_MAX_BATCH = 512; + static constexpr size_t GOTW_MIN_BATCH = 8; + + // Search lifecycle state. _search_started distinguishes a search's first + // pass (seed FollowReferences from the heap roots) from a resumed pass + // (expand the persisted frontier, see expandFrontier()) - runPass() below. + // _search_state starts RUNNING and only ever moves forward (RUNNING -> + // COMPLETED or RUNNING -> ABANDONED, never back) - see runPass()'s comment + // for the exact conditions. Both fields are written only by runPass(), + // called from the single agent-owned BFS thread, but are read cross-thread + // by searchState()/buildAbandonedEvent() (called from Profiler::dump(), + // e.g. profiler.cpp's JFR-flush path) - so, like _gc_start_epoch/ + // _gc_finish_epoch above, they are volatile and accessed via load()/ + // store() rather than a plain load/store the compiler could reorder or + // cache across threads. + bool _search_started; + volatile u8 _search_state; + + // True once releaseSearchTags() has confirmed every live tag this search + // owned was actually cleared (or there were none) - see that method's own + // comment for why a GetObjectsWithTags() failure must NOT be treated as + // "released". Starts true (nothing to release for a not-yet-run search); + // set false the moment a search reaches a terminal state and is only ever + // reset back to true once releaseSearchTags() itself confirms success - + // possibly across several retried runPass() calls first, see runPass()'s + // terminal-state branch. shouldRunPass() refuses to restartSearch() while + // this is false, so _next_tag/the frontier table are never reset out from + // under a search whose tags might still be live. Written and read only + // from the single BFS thread (runPass()/shouldRunPass()), like + // _search_started above, so no volatile/load()/store() is needed. + bool _tags_released; + + // Hysteresis state behind isUrgent(), which used to be a bare + // `secondsToOOM() < OOM_URGENT_THRESHOLD_S` comparison. That estimate is + // computed from a short ring of heap deltas, so it swings by orders of + // magnitude between consecutive observations of the very same steadily + // growing heap (observed in one run: 128s, then 52769s, then back). + // _urgent_latched is set the first time the projection drops below + // OOM_URGENT_THRESHOLD_S and only cleared once it has stayed at or above + // OOM_URGENT_RELEASE_S (or gone unknown, i.e. negative) for + // URGENT_RELEASE_CONSECUTIVE consecutive observations, counted by + // _urgent_release_ticks. + // + // _urgent_search_spent makes each urgency episode authorize exactly one + // search. hasLeakSignal()'s urgency shortcut bypasses the per-klass + // hysteresis gate, so without this every terminal search reaching + // shouldRunPass()'s restart branch while still urgent immediately called + // restartSearch() again - which discards the frontier table and the + // leak-signature/parent-fanout accumulators (see restartSearch()), so the + // rotation heuristics that need several passes to converge were wiped + // before they ever could. Set when a search is started under a latched + // urgency, cleared together with the latch. The per-klass leak-candidate + // half of hasLeakSignal() is untouched and can still authorize restarts + // during an episode. + // + // All three are written and read only from the single BFS thread, but + // mutable because the latch is maintained inside isUrgent() const. + mutable bool _urgent_latched; + mutable int _urgent_release_ticks; + mutable bool _urgent_search_spent; + + // Set (once) at the same point runPass() moves _search_state to ABANDONED - + // see SearchAbandonReason's own comment for why this exists and + // buildAbandonedEvent()/abandonReason() below for how it is read. Same + // cross-thread read pattern as _search_state above. + volatile u8 _abandon_reason; + + // Wall-clock timestamp (OS::nanotime()) of the search's first pass - + // baseline for the TTL cutoff above. Set once, in runPass(), the first + // time _search_started flips true; read cross-thread by + // buildAbandonedEvent() (elapsed-time calculation), so volatile/load()- + // accessed like _search_state above. + volatile u64 _search_start_ns; + + // Tags currently in FrontierEntryState::FRONTIER (admitted but not yet + // expanded), in admission order. Pushed by heapReferenceCallback() at the + // moment it admits a tag (both for the first pass's root-seeded walk and + // for expandFrontier()'s own per-node FollowReferences calls, since both + // share that one callback), popped by expandFrontier()/ + // markAllFrontierExpanded() as entries are expanded. This replaces a + // former O(range) scan over every tag between a cursor and the frontier's + // current size just to filter down to the FRONTIER-state subset - a scan + // whose cost was proportional to everything admitted since the cursor + // last advanced, not to what was actually pending, so a pass immediately + // following a large one-shot admission (e.g. a restart's first pass) paid + // for the whole batch just to discover a handful of genuinely pending + // entries. Only ever touched from the single BFS thread that runs + // heapReferenceCallback()/expandFrontier(), so no locking is needed. + std::deque _pending_expand; + + // Fast-lane counterpart to _pending_expand above: entries admitted while + // re-walking a rotation-selected (already-EXPANDED) parent go here + // instead, and expandFrontier() drains this queue ahead of the ordinary + // one. Without this, a mutable field re-observed via rotation (e.g. + // HashMap.table after a resize) admits a fresh child that then has to + // travel through however much of the ordinary backlog is still ahead of + // it - under a fast-growing leak that backlog can be tens of thousands of + // entries deep, so the re-admitted chain would never visibly progress + // within any reasonable search window. Same single-BFS-thread-only + // access as _pending_expand, no locking needed. + // + // HARD CAP (PRIORITY_EXPAND_CAP): the rotation collectors enqueue up to + // STALE_EXPANDED_ROTATION_BUDGET+ROOT_KIND_ROTATION_BUDGET entries per + // pass, but each phase's wall-clock deadline admits only ~2-3 + // GetObjectsWithTags calls (~50-200 entries) of drain per pass, so an + // uncapped queue grows without bound - observed live: 39k->103k in 20 + // minutes while _pending_expand (the BFS frontier itself) was never + // drained once, because expandFrontier() drains this queue first. The + // cap bounds both the memory and isQueuedForRotation()'s linear scan; + // collectors and admitObject() skip pushing when full, which throttles + // rotation to whatever the drain can actually consume. + // + // expandFrontier() alternates batches between this queue and + // _pending_expand when both are non-empty (see its own comment) - the + // original priority-first drain is what let this queue starve the + // ordinary backlog above. + std::deque _priority_expand; + + // Which lane the NEXT expandFrontier() batch comes from when both lanes + // are non-empty (the alternation toggle). Deliberately a MEMBER, not a + // local: the phase deadlines bound a typical expandFrontier() invocation + // to ONE batch (a single GetObjectsWithTags costs ~25-30ms of a 50ms + // window at a ~240k-entry tag map), and a per-invocation local reset to + // "priority first" made priority win EVERY invocation - observed live on + // hotdog, the ordinary _pending_expand lane (109k entries) was never + // drained by a single batch while the priority lane livelocked on stale + // re-walks. Persisting the toggle across invocations makes the two + // phases of each pass (expand + rotation) drain alternating lanes. + bool _expand_lane_prefer_priority = true; + + // Upper bound on _priority_expand above. 1024 holds a few passes' worth of + // rotation selection (budgets sum to ~272/pass) so a truncating rotation + // phase still has work waiting next pass. + static constexpr size_t PRIORITY_EXPAND_CAP = 1024; + + // O(1) membership index over _priority_expand, backing + // isQueuedForRotation(): the rotation collectors run that check for + // EVERY FrontierTable slot they visit (~199k EXPANDED entries on a + // large heap), and the original linear scan over the deque cost up to + // ~200M comparisons per rotation pass at the cap - observed prominently + // in profiles. Fixed-capacity open addressing with no allocation after + // construction: PRIORITY_EXPAND_CAP (1024) live entries in a + // 2*PRIORITY_EXPAND_SLOT_SHIFT-power-of-two slot table at <=0.5 load + // factor, linear probing over Fibonacci-hashed tags (frontier tags are + // near-sequential, so a plain (tag % slots) index would cluster). + // Deletion needs NO tombstones: expandFrontier() pops a batch off the + // deque's front and rebuildFrom() re-derives the index from the deque's + // remaining contents - a full rebuild is <=1024 inserts, a few + // microseconds, against the ~20ms GetObjectsWithTags call the same + // batch already paid. The deque remains the drain-order source of + // truth; this index only answers membership. All mutation happens on + // the engine thread under the _engine_lock mutex (pushes in the + // collectors/admitObject()/requeueChainRootForRotation(), pops in + // expandFrontier()/markAllFrontierExpanded(), clears in + // startSearch()/restartSearch()), so plain non-atomic access is safe. + class PriorityExpandSet { + private: + // 2^11 == 2 * PRIORITY_EXPAND_CAP == 2048 slots. The shift below + // derives from it; keep both in sync. + static constexpr u64 SLOT_SHIFT = 11; + static constexpr u64 SLOT_MASK = (1ULL << SLOT_SHIFT) - 1; + jlong _keys[1ULL << SLOT_SHIFT]; + u8 _used[1ULL << SLOT_SHIFT]; // 0 = empty, 1 = occupied + + static u64 mix(jlong tag) { + // Fibonacci hashing: spreads near-sequential integer tags evenly + // across the table's power-of-two slot space. + return (u64)tag * 0x9E3779B97F4A7C15ULL; + } + + public: + bool contains(jlong tag) const { + u64 i = mix(tag) >> (64 - SLOT_SHIFT); + while (_used[i]) { + if (_keys[i] == tag) { + return true; + } + i = (i + 1) & SLOT_MASK; + } + return false; + } + + // Idempotent: returns false if `tag` is already indexed. + bool insert(jlong tag) { + u64 i = mix(tag) >> (64 - SLOT_SHIFT); + while (_used[i]) { + if (_keys[i] == tag) { + return false; + } + i = (i + 1) & SLOT_MASK; + } + _used[i] = 1; + _keys[i] = tag; + return true; + } + + void clear() { + memset(_used, 0, sizeof(_used)); + } + + // Re-derives the index from the deque's CURRENT contents. Call after + // any pops so membership matches the queue exactly again; the + // deque's own size is the only bound needed here (the engine thread + // guarantees it stays <= PRIORITY_EXPAND_CAP by construction). + template void rebuildFrom(const Deque &queue) { + clear(); + for (jlong tag : queue) { + insert(tag); + } + } + } _priority_expand_set; + + // B' (find-anchor-holder-eviction / find-anchor-live-feed-design): a + // static holder richly referenced from the running graph is EXCLUDED + // from the static-anchor tier forever once its frontier entry is + // chain-attached (first admitted via a non-root path, or demoted by + // improveChain) - maybeUpgradeRootAttachedRootKind() refuses entries + // with parent_tag != 0 by design, so collectStaticFieldAnchorsForRotation() + // (parent_tag == 0 filter) can never select it. This FIFO is the live + // feed that repairs the hole: the static sweep's class->field edge onto + // such an entry (heapReferenceCallback's root-like-onto-already-admitted + // block, on the failed maybeUpgradeRootAttachedRootKind()) pushes the + // tag here, and runPassManualWalk() drains it into the same + // walkStaticFieldAnchors() batch BEHIND the collector's small + // root-attached cohort (pod round 10: a cap-pinned at-risk flood starving + // that cohort in ~60% of passes is what the reverse order caused) - + // anchor selection no longer depends on the entry's attribution shape + // for this population, so the eviction is structurally impossible. + // Feed semantics: pushes happen every sweep lap (the sweep re-proves + // the static edge each lap), drained entries are NOT re-pushed by the + // walk - an un-intercepted holder comes back via the next lap's push, + // which bounds steady-state occupancy to one lap's worth of at-risk + // holders instead of a permanent rotation cohort. Same engine-thread + // under-_engine_lock discipline as _priority_expand above (the push + // site runs inside the sweep's FollowReferences in runPassSerialized(), + // the drain in the same pass's rotation phase), so no locking. Deque + // node allocations are bounded by the cap; pushes inside the JVMTI + // callback allocate only when a new deque node is needed - bounded, + // amortized, and on the engine thread, never a signal context. + // Round 16: entries carry their klass_id so occupancy is countable per + // class (see _static_anchor_fifo_klass_counts below) - drain/requeue + // maintain the counts exactly without a frontier lookup, so a holder + // whose entry died between push and drain cannot leak a stale count. + struct AtRiskAnchor { + jlong tag; + u32 klass_id; + // PriorityExpandSet::rebuildFrom() iterates `jlong tag : queue` - the + // implicit conversion keeps that template generic over both the + // plain-jlong _priority_expand deque and this pair deque. + operator jlong() const { return tag; } + }; + std::deque _static_anchor_fifo; + + // Membership index over _static_anchor_fifo (push-side dedupe, so one + // lap's repeated static edges onto the same chain-attached holder push + // it once) - a second instance of PriorityExpandSet, whose fixed 2048 + // slot table keeps the cap at PRIORITY_EXPAND_CAP (1024). Full-FIFO + // pushes are dropped (the natural throttle: at-risk holders far beyond + // one lap's drain rate wait for the next lap's re-push, never lost). + PriorityExpandSet _static_anchor_fifo_set; + static constexpr size_t STATIC_ANCHOR_FIFO_CAP = PRIORITY_EXPAND_CAP; + + // Cumulative at-risk pushes, for the per-pass TEST_LOG line (round-10 + // verification: sizes the at-risk population the design's drain-rate + // argument was inferred, not measured, from). + u64 _static_anchor_fifo_pushed = 0; + + // Round 16 (pod round-15 measurement, ev-leaktag-onpod-round15-results): + // the B' repair was DEAD on the pod - the FIFO sat cap-pinned at 1024 + // because three classes flooded it (klass 1: 1396 pushes, klass 215: + // 1063, klass 1733: 988+), so the LEAK_BUFFER wrapper's own pushes + // (klass 28366) were dropped at the cap check and the wrapper never + // rode B' - the exact at-risk holder the lane exists for. Per-class + // occupancy, maintained exactly by push/drain/requeue (the klass rides + // in each AtRiskAnchor), so one class cannot dominate the lane: at + // STATIC_ANCHOR_ATRISK_PER_KLASS_CAP entries per class the 1024 cap + // necessarily holds >= 16 distinct classes, and a class at its quota + // dropping its 65th push is correct twice over - it is already + // represented (its oldest entry drains within a few passes at + // STATIC_ANCHOR_FIFO_DRAIN=16/pass), and the alternative measured on + // the pod was every other class's repair being locked out. Counters + // are erased at zero on drain so the map is bounded by the FIFO's own + // contents (<= 1024 distinct classes), not by the search lifetime. + std::unordered_map _static_anchor_fifo_klass_counts; + static constexpr u32 STATIC_ANCHOR_ATRISK_PER_KLASS_CAP = 64; + // Cumulative quota drops (class at cap), for the per-pass TEST_LOG + // line: on the pod this should climb steadily with the flood classes' + // pushes while the wrapper's pushes stop dropping (the round-16 + // verification channel, alongside fifo_size dropping below 1024). + u64 _static_anchor_fifo_quota_drops = 0; + + // Index of root-attached STATIC_FIELD/JNI_GLOBAL frontier entries, + // so collectStaticFieldAnchorsForRotation() iterates O(anchors) instead + // of scanning the full frontier table O(frontier_size). An entry is + // added when it is first admitted root-attached with a durable root_kind + // (STATIC_FIELD or JNI_GLOBAL), or when maybeUpgradeRootAttachedRootKind() + // upgrades it to one of those. Cleared on restartSearch(). Engine thread + // only — all mutation sites run under _engine_lock or inside the BFS + // thread's own pass. + std::vector _static_anchor_index; + + // Parallel to _static_anchor_index: the OWN class tag of each anchor + // object (the class of the static field's VALUE, not the holder class). + // The selection tiers anchors by that class's shape (container vs other, + // see AnchorClassShape below) via a lookup into _class_shape_cache, so a + // later classification automatically upgrades an entry's tier without + // any index mutation. Cleared with the index on restartSearch(). + std::vector _static_anchor_own_class_tags; + + // Dedup companion for _static_anchor_index (O(1) membership; the + // population is ~28k on a real JVM - see addToStaticAnchorIndex()'s own + // comment). Cleared with the index. + std::unordered_set _static_anchor_index_tags; + + // Shape of a class as an anchor candidate: does the class implement + // java/util/Collection or java/util/Map (directly or via superclasses/ + // interfaces)? A leak holder is typically a container (the hotdog leak: + // Collections$SynchronizedRandomAccessList, a List) while the anchor + // population is dominated by non-containers (String/Class/boxed/ + // primitive-array/enum statics - round 13 measured ~28k anchors on the + // hotdog JVM vs ~4k walkable per search). Selection walks leak-tagged + // anchors first, then containers, then everything else cursor-fairly — + // a container anchor is thus covered within ceil(container_cohort / + // budget) passes of admission instead of within ceil(28k / budget) + // passes (= never, at ~190-pass search lifetimes). + enum class AnchorClassShape : u8 { UNKNOWN = 0, CONTAINER = 1, NON_CONTAINER = 2 }; + + // class tag -> AnchorClassShape, process-lifetime (class tags are never + // reused - the shared class-tag allocator is deliberately not reset by + // restartSearch()). Filled lazily by reconcileAnchorClassShapes(): an + // anchor's own class is classified on first need, never re-classified. + // Engine thread only. Entries are never evicted: bounded by loaded-class + // count (~34k on hotdog), u8 values. + std::unordered_map _class_shape_cache; + + // java/util/Collection and java/util/Map class tags, resolved once + // lazily by resolveContainerInterfaceTags() (0 = not yet resolved; a + // resolved value is NEGATIVE - class tags are a negative namespace, + // see nextClassTag()'s own comment). Class tags are stable for the + // JVM's lifetime, so these are safe to cache across searches. Engine + // thread only. + jlong _collection_iface_class_tag = 0; + jlong _map_iface_class_tag = 0; + + // Fair-rotation cursors (index POSITIONS, not tags) for the two + // cursor-fair tiers of collectStaticFieldAnchorsForRotation(): + // _anchor_container_cursor for container-shaped anchors, + // _anchor_other_cursor for everything else. Leak-tagged anchors are + // always selected (they are rare) and need no cursor. Wrapping + // semantics: the next selection for a tier scans for entries at + // positions >= the cursor, stops at the lap end (no within-call wrap), + // and leaves the cursor just past the last consumed position. + size_t _anchor_container_cursor = 0; + size_t _anchor_other_cursor = 0; + + // Fresh-admission lane (round 15): FIFO of anchor tags that have not + // yet had their ONE first-look walk priority. addToStaticAnchorIndex() + // appends here alongside the index (the two append together, so the + // queue is always an ordered suffix window of the index); the collector + // drains it each call - keeping container-shaped or not-yet-classified + // entries up to the walk budget, and DROPPING everything else out of + // the queue (a dropped anchor keeps its index position and is owned by + // the fair tiers from then on - it got its one first look and was + // outranked, not lost). Why a queue and not a "positions since mark" + // watermark: the mark form reorders selection when state leaks between + // units/tests (a stale mark demotes an early anchor behind a later one + // for no reason - caught by StaticAnchorRotation* in test-order runs); + // the queue is per-anchor, so each anchor's first look is exactly once, + // in admission order, regardless of what any earlier search/test left + // behind. Rationale for the lane itself: admission order = sweep order + // = loaded-class order, so a leak holder held by a late-loaded class + // (the hotdog wrapper: holder class at sweep index 24627 of 33270) + // admits at the index TAIL - the very END of the fair container tier's + // lap - and the measured search lifetime (44-75 passes, round 15) is + // shorter than ceil(container_cohort/budget) (1633/16 = 102), so + // fair-only coverage never reaches it; the fresh lane walks it within a + // pass or two of admission instead. Bounded by STATIC_ANCHOR_FRESH_CAP + // (overflow drops the OLDEST fresh chances - they fall back to the fair + // tiers, never lost). Cleared with the index on restartSearch(). + std::deque _static_anchor_fresh_queue; + + // Cap for _static_anchor_fresh_queue. The queue normally holds only + // ~one pass of sweep admits (admits happen in passes, the collector + // drains every pass, and the drain DROPS everything it does not keep, + // so the queue empties each call); the cap only guards a pathological + // burst (e.g. a pass admitting thousands) from growing it unbounded in + // native memory. + static constexpr size_t STATIC_ANCHOR_FRESH_CAP = 1024; + + // java/lang/Object jclass cache for expandFrontier()'s and + // admitStaticFieldRoots()'s holder-array element type (referenceChains.cpp) + // - resolved once via FindClass()+NewGlobalRef() and reused for the + // tracker's lifetime. MUST be a GLOBAL ref, not a local one: callers + // include JNI-entered test seams (runReferenceChainPass0), and a local + // ref is freed the moment its creating JNI invocation returns to Java - + // caching one across invocations crashed in NewObjectArray() on the + // second pass (observed). A global ref is also valid across the BFS + // thread's detach/attach cycles, so no JNIEnv* keying or detach-time + // invalidation is needed. Never freed: java/lang/Object is a bootstrap + // class (never unloaded) and this tracker is a process-lifetime + // singleton, so the single ref is reclaimed with the JVM. + jclass _cached_object_class = nullptr; + + // Rotation cursor for collectStaleRootKindEntriesForRotation() (Phase 5 + // item 3): 1-based tag to resume scanning from on the next call, so + // consecutive calls sweep forward through the table instead of always + // re-examining the same low-tag entries first. Wraps back to 1 once it + // reaches _frontier->size(). Persisted across passes (not per-search-reset + // by ReferenceChainsTestAccessor::reset(), same as _next_tag is not reset + // by restartSearch() logic elsewhere) since a stale cursor value only ever + // costs one wasted scan step before self-correcting, never a correctness + // problem. + jlong _root_kind_rotation_cursor; + + // Same role as _root_kind_rotation_cursor above, but for + // collectStaleExpandedEntriesForRotation(): without its own persistent + // cursor, that sweep always restarted from tag 1 on every call, so a + // frontier table holding >= STALE_EXPANDED_ROTATION_BUDGET low-tag entries + // that stay EXPANDED forever (long-lived infrastructure objects) filled + // its entire per-pass cap from that population alone, every pass, + // permanently starving any EXPANDED entry with a higher tag (e.g. a + // static field's collection, admitted only once its class loads well + // after startup) of ever being re-queued. + jlong _stale_expanded_rotation_cursor; + + // Per-pass cap on how many transient-root_kind entries + // collectStaleRootKindEntriesForRotation() selects - round, provisional + // like this subsystem's other unbenchmarked constants (see e.g. + // MIN_EFFECTIVE_BUDGET's own comment): small enough that a pass dominated + // by rotation work never meaningfully competes with genuinely new + // discoveries for the same pass's budget, large enough that a search with + // a modest number of transient roots converges to durable attribution + // within a handful of passes rather than needing hundreds. + static constexpr int ROOT_KIND_ROTATION_BUDGET = 16; + + // Per-pass cap on how many EXPANDED entries + // collectStaleExpandedEntriesForRotation() re-queues for expansion, + // uniformly across the WHOLE frontier table regardless of lineage. This is + // the low-priority fallback tier of the rotation design (see + // collectLeakAccumulationCandidatesForRotation()'s own comment for the + // targeted tier): coverage of the frontier table is only guaranteed within + // ceil(table_size / this) passes, which can be far longer than any one + // search realistically survives before completing/restarting on a large + // table - two earlier versions of this code tried to compensate by scaling + // this cap with table size, then by adding a structural (depth + root- + // durability + class-shape) priority tier, but both were solving the wrong + // problem (see git history): no purely structural property can distinguish + // the one specific container that is actually leaking from the thousands + // of ordinary ones a real classpath contains. collectLeakAccumulationCandidatesForRotation() + // instead targets that population directly, using LivenessTracker's own + // growth signal plus fanout-of-a-flagged-klass, at a small fixed budget + // regardless of table size. This constant stays flat because the rest of + // the table (everything NOT tied to a currently-flagged klass) genuinely + // doesn't need a faster guarantee - eventual coverage is enough. + static constexpr int STALE_EXPANDED_ROTATION_BUDGET = 256; + + // Candidate-scoped reach (see descendFromAnchor()/walkCandidateThreadLocals()/ + // walkStaticFieldAnchors()'s own comments): how many hops BELOW a descend + // walk's anchor the walk may admit. A static Map -> table[] -> Entry -> + // leaked chunk is 3-4 hops below its root-attached holder; a Thread -> + // ThreadLocalMap -> table[] -> Entry -> value -> holder -> chunk is 5-6 + // below the Thread object. Raised 6 -> 16 after pod round 7: with both + // prongs live, interception stayed zero while the static-anchor walks + // admitted whole executor-task subgraphs (3000+ edges in single walks) + // - the holder interior is reachable but the tagged chunks sit one or + // more hops PAST the 6-hop cap (the static-ExecutorService -> queue -> + // task -> accumulator -> list -> chunk shape is ~7 deep). The walk is + // deadline-bounded per slice either way, so the cap's cost model does + // not change - a deeper cap just lets each bounded walk cover the whole + // holder interior instead of stopping mid-way, and already-admitted + // entries are skipped so repeated passes march deeper each time. + static constexpr int DESCENT_HOPS = 16; + + // Per-pass cap on how many candidate threads walkCandidateThreadLocals() + // descend-walks. Each anchor walk is a whole bounded FollowReferences + // call, so unlike queue-push rotation tiers, this cap is the real cost + // control next to the deadline; a per-pass rotation of 4 gives every + // qualifying tid a turn within a few passes (qualifying tid sets are + // small - bounded by MAX_QUALIFYING_TIDS per candidate, and typically 1-2). + static constexpr int THREAD_WALK_MAX_ANCHORS = 4; + + // Per-pass cap on how many root-attached static holders + // walkStaticFieldAnchors() resolve + descend-walk. Same "the cap IS the + // cost control" reasoning as THREAD_WALK_MAX_ANCHORS (each anchor is a + // bounded FollowReferences call); the tiered cursors in + // collectStaticFieldAnchorsForRotation() guarantee coverage of each + // tier within ceil(tier / this) passes - in particular the container + // tier (the likely leak-holder cohort) within ceil(container_cohort / + // this) passes of admission, regardless of how large the full anchor + // population is. Round 15 raised 16 -> 32: the measured fresh-container + // admit rate on hotdog (~10-37/pass depending on the sweep's class + // band) exceeded the old cap, which would have made the fresh-priority + // lane a growing backlog - same tail-starvation shape one level down. + // STW safety does NOT depend on this number: walkStaticFieldAnchors() + // stops at the per-pass deadline and requeues whatever it could not + // walk (TruncatedAnchorWalkRequeuesUnwalkedFifoTags covers the requeue + // path), so a larger selection can only spend selection-scan time + // (O(anchors), no safepoint), never extend the pause. Observed on-pod: + // the deadline already truncates 16-selection passes ("walked 6-16 of + // selected 20", round 10) - selection size and walked size are + // decoupled. + static constexpr int STATIC_ANCHOR_ROTATION_BUDGET = 32; + + // Per-pass cap on how many DISTINCT anchor classes + // reconcileAnchorClassShapes() classifies (one GetObjectsWithTags call + // for the batch + a depth-bounded interface walk per class). Bounds the + // first-lap classification of a ~34k-class JVM to ~270 passes worst case + // (in practice far fewer: only classes that actually own admitted + // anchors need classifying), after which the cache is warm for the JVM's + // lifetime. Same "the cap IS the cost control" reasoning as the budgets + // above. + static constexpr int ANCHOR_SHAPE_RECONCILE_BUDGET = 128; + + // Per-pass cap on how many AT-RISK static holders (frontier entries + // with parent_tag != 0 - the find-anchor-holder-eviction population) + // drainStaticAnchorFifo() pops for the same walkStaticFieldAnchors() + // batch. Deliberately larger than STATIC_ANCHOR_ROTATION_BUDGET: the + // whole batch resolves in ONE GetObjectsWithTags call whose cost is + // dominated by the O(tag_map) scan floor, so a larger batch is nearly + // free per anchor; each anchor's own walk still draws down the same + // rotation budget, and truncation re-queues un-walked entries + // (requeueStaticAnchorFifoFront()), so a big batch costs only the + // per-entry GOTW bookkeeping, not extra STW. + static constexpr int STATIC_ANCHOR_FIFO_DRAIN = 16; + + // (Removed: the old single wrapping-index-cursor selection was replaced + // by tiered selection with per-tier cursors — see + // collectStaticFieldAnchorsForRotation()'s own comment and + // _anchor_container_cursor/_anchor_other_cursor.) + + // Cursor over the flattened (slot, tid) enumeration of + // _candidate_qualifying_tids above, so walkCandidateThreadLocals()'s + // THREAD_WALK_MAX_ANCHORS-per-pass cap rotates fairly instead of always + // walking the same first candidates' tids. + int _thread_walk_anchor_cursor; + + // Per-pass cap on how many EXPANDED entries + // collectLeakAccumulationCandidatesForRotation() re-queues - see that + // method's own comment. Small and fixed like the other two tiers' + // budgets: the population it selects from (_leak_parent_fanout) is itself + // already small, bounded by how many distinct parent objects have ever + // been observed holding an instance of one of the currently-watched leak + // klass_ids (see _watched_leak_klass_ids), not by total table size. + static constexpr int LEAK_ACCUMULATION_ROTATION_BUDGET = 16; + + // Snapshot of gcFinishEpoch() as of the end of the last pass. Written only + // by runPass(), read only by shouldRunPass() - both always called from the + // same thread (the BFS thread once wired up, or directly by a caller/test + // standing in for it), so no locking is needed. + u64 _last_pass_gc_finish_epoch; + + // OS::nanotime() as of the end of the last pass. Written by runPass() on + // the BFS thread; also read cross-thread by buildAbandonedEvent()'s + // elapsed-time calculation, so volatile/load()-accessed like + // _search_state above. + volatile u64 _last_pass_ns; + + // Total passes run this search. Written by runPass() on the BFS thread; + // read cross-thread by passesRun()/buildAbandonedEvent(), so volatile/ + // load()-accessed like _search_state above. + volatile int _passes_run; + + // Resolved reference chains, keyed by the leak-candidate klass_id + // pollWatchedTargets() reconstructed each one for. An entry lives here for + // as long as LivenessTracker can still resolve a representative for its + // klass_id: pollWatchedTargets() prunes it the moment the representative + // stops resolving (collected, or LRU-evicted from the population table). + // Every Profiler::dump() re-emits the whole cache (drainPendingChainEvents() + // snapshots without clearing), so a datadog.ReferenceChain lands in every + // JFR chunk the sample survives into - mirroring how LivenessTracker + // re-emits its live-object samples on each flush, rather than the emit-once + // model where a chain reached only the single transient dump that drained + // it. klass_id is the identity that survives a search restart (which resets + // FrontierTable tags); CachedChain remembers the tag and search generation + // the chain was reconstructed from, so a poll after a restart re-tags the + // object rebuilds the chain instead of trusting a tag value the reset has + // since reassigned to a different object. + struct CachedChain { + ReferenceChainEvent event; + jlong source_tag; + u64 source_search_ns; + }; + // Bounded by the number of discovered instances across all candidate + // slots: MAX_LEAK_CANDIDATES_FROM_LT * MAX_DISCOVERED_INSTANCES_PER_CLASS + // = 5 * 8 = 40, plus up to 5 canary chains. 128 gives headroom for + // chains from the normal (non-canary) path too. + static constexpr int MAX_RESOLVED_CHAINS = 128; + // Keyed by frontier tag (individual instance identity), NOT by klass_id. + // For common classes like [B or [Ljava/lang/String; there may be many + // live instances with different reference chains — the first chain found + // may be noise (a shallow JNI-local instance), while the actual leak is + // a deep static-field-held instance. Caching per-instance ensures all + // chains are emitted to JFR and the profiling backend can aggregate them + // by class. Chains expire when the search restarts (frontier is wiped, + // all tags become invalid). + std::unordered_map _resolved_chains; + SpinLock _resolved_chains_lock; + + // Abandoned-search events awaiting Profiler::dump() (profiler.cpp). + // Populated synchronously by runPass() (referenceChains.cpp), at the exact + // point it writes SearchState::ABANDONED, by building a + // ReferenceChainAbandonedEvent from the fields that are only valid right + // then. Reading those fields lazily later (buildAbandonedEvent(), a live + // re-read of _search_state/_abandon_reason/etc.) does not work in + // production: shouldRunPass() (this class's own header comment) can call + // restartSearch() as soon as the very next BFS-thread loop iteration + // (~1s later), which both flips _search_state back to RUNNING and clears + // _abandon_reason/_search_start_ns/_passes_run - while Profiler::dump() + // only samples this tracker's state on JFR chunk rotation, an independent + // and much slower clock (tens of seconds). A dump landing outside that + // ~1s window previously saw nothing to report at all, silently dropping + // every abandoned-search event. Queueing the fully-built event at the + // moment of abandon removes the race entirely: dump() drains whatever + // accumulated since its last call, however long that took. + // + // Bounded like _resolved_chains above: an abandon queued between two + // dumps should be rare (dump cadence is normally much shorter than how + // long a search takes to get stuck), so hitting the cap and dropping + // (counted via REFERENCE_CHAIN_EVENTS_DROPPED, "no silent truncation") + // signals abandons happening far faster than normal, worth surfacing + // rather than growing without bound. + static constexpr int MAX_PENDING_ABANDONED_EVENTS = 16; + std::vector _pending_abandoned_events; + SpinLock _pending_abandoned_events_lock; + + // Search restart (this class's own header comment): leaky bucket over the + // wall-clock cost of past searches, gating how soon a *restarted* search + // may take its first pass - see PainBudget's own comment (painBudget.h) + // and canAffordNewSearch() below. _search_pain_ms accumulates the current + // search's own cost (each pass's pass_wall_ticks, converted to ms) as it + // runs; restartSearch() spends the total into _safepoint_pain_budget and zeroes this + // back out for the next search. Constructed with the configured refill + // rate in start(), mirroring _pause_pid's own placeholder-then-reconstruct + // pattern. + PainBudget _safepoint_pain_budget; + // Cached refill rate from start(), reused by resetSearchStateForTest() + // so a test reset rebuilds the budget with the same rate. + double _pain_budget_refill_rate = 0.0; + u64 _search_pain_ms; + + // Non-safepoint CPU-time pain budget: gates shouldRunPass() independently + // of both _safepoint_pain_budget above (which only cools down *restarts*, spent once + // per finished search) and _pause_pid's per-pass signal (updatePacing(), + // now fed only the genuine in-safepoint portion of each pass - see + // runPass()'s own comment). Spent every pass, root-enum or not, with the + // wall-clock cost of runPassManualWalk() *outside* its actual JVMTI calls + // - root/stack-ref enumeration dispatch, frontier-table admission, + // rotation-candidate collection - none of which the pause-time PID has any + // visibility into. Seeded from the same configured refill rate as + // _safepoint_pain_budget (_reference_chains_pain_budget_percent) rather than a + // separate knob - one operator-facing "how much background cost is + // acceptable" percentage covers both leaky buckets. + PainBudget _cpu_pain_budget; + + // The cache above is mutated on this tracker's own BFS scheduling thread + // (pollWatchedTargets()) and read on whatever thread calls Profiler::dump() + // (drainPendingChainEvents()); _resolved_chains_lock (declared with the + // cache) is the only synchronization between them. The write itself is + // still deferred to the dump() thread - Profiler::writeReferenceChain() + // (profiler.cpp) can block up to ~50ms per event under _locks[] contention, + // which must never delay the next scheduled BFS pass - exactly mirroring + // how buildAbandonedEvent()'s output is deferred to that same call site + // rather than written eagerly. + + // Fallback cadence for shouldRunPass()'s cadence trigger (design doc's + // Triggering section / Open Question 5). Provisional default pending + // empirical tuning (see + // doc/architecture/LiveHeapReferenceChains-ImplementationPlan.md) - not + // benchmark-derived: a round one-second value chosen only so an idle + // search still makes some progress between GC-triggered wakeups without + // polling so tightly that an idle tracker burns CPU. The pause-time pacing + // controller folds Open Question 5's cadence decision into updatePacing() + // below rather than solving it separately (design doc's explicit "one + // shared mechanism" framing): this constant now only serves as + // _effective_cadence_ns's starting value (start()) and as the unit + // MAX_EFFECTIVE_CADENCE_NS below scales from - shouldRunPass()/threadLoop() + // themselves compare against _effective_cadence_ns, not this constant + // directly, once a pass has run. + static constexpr u64 PASS_CADENCE_NS = 1000000000ULL; // 1s + + // Heap-wide time-to-OOM urgency threshold (LivenessTracker::secondsToOOM()) + // - hasLeakSignal() below forces a search to start immediately once the + // projection drops under this, rather than waiting for a klass to clear + // selectLeakCandidates()'s own per-klass ring-fill/hysteresis gate + // (KLASS_POPULATION_MIN_FILL_FOR_TREND plus LEAK_TREND_HYSTERESIS_BASE/ + // CORROBORATED epochs, livenessTracker.h). Without this, an aggressive, + // heap-wide leak can OOM the process before any single klass ever clears + // that gate. Provisional, same status as every other un-benchmarked + // *_NS/*_S constant in this file: 5 minutes is chosen only to leave some + // margin for a BFS pass plus a JFR flush to actually run before the + // projected exhaustion, not measured against a real OOM race. + static constexpr double OOM_URGENT_THRESHOLD_S = 300.0; // 5 minutes + + // Release side of OOM_URGENT_THRESHOLD_S's hysteresis (see + // _urgent_latched). secondsToOOM() is derived from a short ring of heap + // deltas, so consecutive readings on the same monotonically growing heap + // routinely swing across OOM_URGENT_THRESHOLD_S in both directions - a + // single bare threshold comparison therefore flaps, and each flap back to + // "urgent" used to authorize a brand-new whole-heap search via + // hasLeakSignal(). Urgency is only released once the projection has stayed + // clear of this (deliberately higher) bar for URGENT_RELEASE_CONSECUTIVE + // consecutive observations. + static constexpr double OOM_URGENT_RELEASE_S = 2 * OOM_URGENT_THRESHOLD_S; + static constexpr int URGENT_RELEASE_CONSECUTIVE = 5; + + // Horizon over which threadLoop() ramps the pause target and cadence + // toward their urgent ceilings as secondsToOOM() falls, once it reports a + // confirmed rising trend (see secondsToOOM()'s own NOT_RISING gate — a + // non-negative value already means real growth, not noise). Profiles are + // flushed once per minute by default, so a search that only ramps up in + // the last OOM_URGENT_THRESHOLD_S (5 minutes) may not get enough elevated + // passes recorded in a JFR before the process dies. 30 minutes gives it + // ~30 recording rotations to actually land a chain. Separate from + // OOM_URGENT_THRESHOLD_S, which still gates hasLeakSignal()'s forced + // search start and runPass()'s TTL-abandonment suppression. + static constexpr double OOM_RAMP_START_S = 1800.0; // 30 minutes + + // Ceilings the pause target and cadence ramp toward as secondsToOOM() + // approaches zero within OOM_RAMP_START_S: the ramp is exponential (slow + // near OOM_RAMP_START_S out, aggressive near OOM) since the process is + // likely to die anyway and diagnostic data collected right before that is + // worth spending STW time and CPU on. TTL abandonment is suppressed + // whenever isUrgent() (see OOM_URGENT_THRESHOLD_S). + static constexpr long URGENT_PAUSE_TARGET_MS = 100; // ceiling STW ms per pass + static constexpr u64 URGENT_CADENCE_NS = 10000000ULL; // 10ms floor between passes + + // Abandon the search after this many consecutive passes with + // zero new frontier entries admitted (genuinely stuck, not just slow). + // A large heap takes more passes simply because there are more + // objects to explore — that is not "stuck". Only abandon when the + // frontier stops growing entirely. + + // True when LivenessTracker::secondsToOOM() projects exhaustion sooner + + // Auto-scaled default for _first_pass_budget when + // Arguments::_reference_chains_first_pass_budget is unset (0) - see + // _first_pass_budget's own comment for why plain _budget is the wrong + // fallback. 50x is a round, provisional guess (same status as every other + // _reference_chains* constant here), picked to comfortably clear a cold + // JVM's root set without needing firstpassbudget spelled out explicitly for + // every reasonably-sized heap; the cap keeps a pathologically large + // _budget (e.g. an operator-supplied 100000) from ballooning the first + // pass's own one-shot cost unbounded. + static constexpr int AUTO_FIRST_PASS_BUDGET_MULTIPLIER = 50; + static constexpr int AUTO_FIRST_PASS_BUDGET_CAP = 200000; + + // Minimum wall-clock gap between root/stack-ref enumeration attempts + // (runPassManualWalk()'s IterateOverReachableObjects call) after the + // search's own first pass. That call re-walks every live GC root and + // stack/JNI-local on every attempt regardless of budget (see its own + // comment) - a fixed tax independent of how much of it is new. Retrying it + // on every cheap steady-state tick (PASS_CADENCE_NS once relaxed) pays that + // tax far more often than it buys new admissions, which two live + // experiments confirmed nets LESS total progress than one large, + // infrequent attempt (each retry given _first_pass_budget-sized headroom - + // see _last_root_enum_ns's own comment). Seconds, not milliseconds: large + // enough that most ticks take the cheap expandFrontier()-only path, small + // enough that a ~20s search window still gets several independent attempts + // at whatever root JVMTI's enumeration order didn't reach the first time. + // Round, provisional guess, like this subsystem's other unbenchmarked + // constants. + static constexpr u64 ROOT_ENUM_MIN_INTERVAL_NS = 2000000000ULL; // 2s + + // Pause-time pacing controller: bounds and conversion constants for + // updatePacing()'s budget/cadence adjustment - see that method's own + // comment for the full mechanism. Every value here is a round, provisional + // guess like every other _reference_chains* constant in this codebase + // (arguments.h's own DEFAULT_REFERENCE_CHAINS_* header comment sets the + // pattern) - a future benchmark plan is the intended path to replacing + // all of them with measured values, not a design decision made here. + // + // Floor updatePacing() will never shrink _effective_budget below (clamped + // further down to _budget itself when the configured budget is smaller + // than this floor - see updatePacing()). Not 0: a floor of 0 would let a + // single pathological pass shrink the search to "admit nothing, ever", + // stalling all progress instead of just slowing it. + static constexpr int MIN_EFFECTIVE_BUDGET = 2000; + + // Bounds for _effective_cadence_ns. The lower bound is not 0: threadLoop() + // sleeps for exactly this many nanoseconds each loop iteration (below), so + // a true 0 would busy-loop the BFS thread. The upper bound reuses + // PASS_CADENCE_NS (this field's own pre-pacing-controller baseline) as the unit for + // a round, provisional multiplier, so a search that is persistently over + // the pause-time ceiling still makes some progress rather than backing + // off indefinitely. + static constexpr u64 MIN_EFFECTIVE_CADENCE_NS = 10000000ULL; // 10ms + static constexpr u64 MAX_EFFECTIVE_CADENCE_NS = PASS_CADENCE_NS * 4; // 4s + + // Conversion factor from "edges of budget signal updatePacing()'s clamp + // could not absorb" to a cadence adjustment in nanoseconds - the two are + // different units (edge count vs. wall-clock time) with no natural + // exchange rate, so this is a round, provisional choice: large enough + // that a sustained, deeply-saturated overflow visibly moves the cadence + // within a handful of passes, small enough that a single borderline pass + // does not swing the whole cadence range at once. + static constexpr u64 CADENCE_NS_PER_EDGE_OVERFLOW = 1000000ULL; // 1ms/edge + + // Budget-borrowing (see _borrowed_budget's own comment): how many + // consecutive comfortably-under-target passes (BORROW_UNDER_TARGET_FRACTION) + // must be observed before updatePacing() starts growing _borrowed_budget at + // all. Round, provisional like this subsystem's other unbenchmarked + // constants - large enough that a brief lull (e.g. one quiet pass right + // after a GC) cannot itself unlock extra headroom, small enough that a + // workload with a genuinely fast-growing frontier converges within a few + // seconds of passes rather than needing to wait out most of the search's + // own TTL just to start borrowing. + static constexpr int BORROW_WARMUP_PASSES = 5; + + // Budget-borrowing: a pass counts toward BORROW_WARMUP_PASSES/keeps + // _borrowed_budget only when pass_ms is at most this fraction of + // _pause_target_ms - deliberately stricter than merely "under the + // ceiling" (which the ordinary _effective_budget clamp already + // guarantees), so growth is gated on *comfortable* headroom, not on + // shaving the pass in just under the wire. + static constexpr double BORROW_UNDER_TARGET_FRACTION = 0.5; + + // Budget-borrowing: hard cap on how far updatePacing() may grow + // (_budget + _borrowed_budget) above _budget alone - _borrowed_budget + // itself is clamped so the resulting ceiling never exceeds + // _budget * BORROW_CEILING_MULTIPLIER. _budget remains a real ceiling in + // the sense that it still bounds how much headroom borrowing can ever + // reach; this only relaxes "never exceeded" into "never exceeded by more + // than a bounded, revocable multiple", which is the whole point of the + // extension (see _borrowed_budget's own comment). + static constexpr int BORROW_CEILING_MULTIPLIER = 4; + + // Budget-borrowing: fraction of _budget by which _borrowed_budget grows on + // each pass once BORROW_WARMUP_PASSES has been reached - a fraction of the + // configured budget rather than of the current borrowed amount, so growth + // stays linear (predictable, boundable within a known number of passes) + // rather than compounding. + static constexpr double BORROW_GROWTH_FRACTION = 0.25; + + // Agent-owned BFS thread (design doc's Triggering section: "an agent-owned, + // already-attached thread ... calling FollowReferences/IterateThroughHeap + // directly; the safepoint is a side effect of that call, not something the + // profiler builds or schedules"). threadLoop() mirrors J9WallClock's + // pthread lifecycle (j9WallClock.cpp:28-57) rather than BaseWallClock's, + // since J9WallClock's is the simpler of the two shapes actually used for a + // single dedicated thread in this codebase. threadLoop() implements the + // actual scheduling loop (shouldRunPass() below). + // + // start()/stop() themselves still do NOT create/join this thread - + // threadLoop()'s VM::attachThread() call crashes on a null VM::_vm if the + // VM is not yet attached, and referenceChains_ut.cpp calls start() + // directly with no live JVM, so spawning unconditionally from start() + // would crash that gtest binary. startThread()/stopThread() (public API + // above) own the thread's lifecycle instead, and are called from + // Profiler::start()/stop() (profiler.cpp) - the only place in this + // codebase that also calls ReferenceChainTracker::start()/stop() itself, + // and only once the JVM/JVMTI environment is already up. onGCFinish() + // below wakes this thread via pthread_kill(WAKEUP_SIGNAL) whenever it is + // running (i.e. once startThread() has been called) - inert otherwise. + pthread_t _thread; + // std::atomic rather than plain volatile bool - volatile alone gives + // no C++ memory-model acquire/release guarantees (it only prevents the + // compiler from eliding/reordering that one variable's own accesses), so a + // weakly-ordered CPU (e.g. arm64) could let the BFS thread's stopThread()- + // side write (see stopThread()'s own comment) become visible to + // threadLoop() later than intended, missing the shutdown request on one + // wakeup and sleeping/looping an extra cycle before pthread_join() unblocks + // it. Written with memory_order_release from startThread()/stopThread(), + // read with memory_order_acquire from threadLoop() - the same cross-thread + // shape _abort_pass_requested above already uses atomic for. + std::atomic _running; + + // Cooperative-cancellation flag for an in-flight JVMTI FollowReferences + // walk: stopThread() sets this before pthread_kill()/pthread_join() + // (that signal alone cannot interrupt a call already inside the JVM/JVMTI + // implementation), and heapReferenceCallback() checks it on every + // invocation, aborting the walk within one callback rather than letting + // pthread_join() block until the walk finishes on its own - see both + // methods' own comments. startThread() resets it back to false, since a + // dynamic-attach profiler can cycle through multiple start()/stop() calls + // in one JVM lifetime and a stale abort request would instantly kill the + // next cycle's very first pass. std::atomic: written from + // stopThread()/startThread() on the calling (shutdown) thread and read + // from heapReferenceCallback() on the BFS thread - the same cross-thread + // shape _running above already has, just made explicit via atomic rather + // than a plain volatile bool. + std::atomic _abort_pass_requested; + + // Wall-clock deadline for the pass currently in flight (OS::nanotime() + // ticks; 0 = no deadline). Set once at the top of runPassManualWalk() from + // _pause_target_ms and shared across that same call's static-field sweep + // and expandFrontier() calls (both read it via heapReferenceCallback()'s + // own periodic check), and collectStaleExpandedEntriesForRotation()'s + // candidate scan (its own periodic check, same amortization pattern - that + // scan is plain C++ under _frontier's shared lock, not a JVMTI/STW call + // itself, but an unbounded scan there would still steal from this same + // pass's wall-clock share before the actual walk even starts). Deliberately + // NOT applied to root/stack-ref + // enumeration (heapRootCallback()) - a live experiment truncating that + // call early on a wall-clock basis measurably reduced total edges admitted + // over a fixed test window versus letting it run to its own (much larger) + // edge budget, because the call's fixed root-walk-and-dispatch cost is paid + // in full regardless of how early it's cut off - see + // ROOT_ENUM_MIN_INTERVAL_NS's own comment for the mechanism that now + // controls how often that call runs instead. Single-threaded: only + // threadLoop() ever calls runPassManualWalk(), so this needs no atomicity. + u64 _pass_deadline_ns = 0; + + // Last time root/stack-ref enumeration actually ran (OS::nanotime() ticks; + // 0 before the search's first pass). runPass() compares this against + // ROOT_ENUM_MIN_INTERVAL_NS to decide whether the current pass re-runs + // IterateOverReachableObjects or takes the cheap expandFrontier()-only + // path over the already-persisted frontier. Single-threaded, same as + // _pass_deadline_ns above. + u64 _last_root_enum_ns = 0; + + // Set true when the most recent root/stack-ref enumeration attempt ended + // via BUDGET_EXHAUSTED (not FRONTIER_CAP_HIT, which stops admitting new + // frontier entries but leaves the search RUNNING - see runPass()'s + // frontier_cap_hit handling) - runPass() treats this as grounds to retry + // root enumeration + // on the very next pass regardless of ROOT_ENUM_MIN_INTERVAL_NS, so a + // still-incomplete attempt is not left waiting out the full interval + // before continuing. Cleared as soon as an attempt completes without + // truncating. + bool _root_enum_truncated_last_time = false; + + ReferenceChainTracker() + : _enabled(false), + _frontier(nullptr), _configured_frontier_cap(0), + _last_class_map_generation(0), + _last_resolved_class_count(0), + _last_static_field_class_count(-1), + _static_field_sweep_cursor(0), + _static_field_sweep_cycle_truncated(false), + _gc_start_epoch(0), + _gc_finish_epoch(0), _next_tag(1), + _hop_cap(0), _budget(0), _first_pass_budget(0), _ttl_ms(0), _pause_target_ms(0), + _effective_pause_target_ms(0), _passes_since_last_progress(0), + _passes_since_last_candidate_progress(0), _last_candidate_progress_mark(0), + _canary_backoff_mult(1), _canary_pass_ema_ms(0), + _last_canary_pass_ns(0), _oom_ramp_active(false), + _canary_stuck_restart_count(0), + _effective_budget(0), _effective_cadence_ns(PASS_CADENCE_NS), + _pause_pid(1, 1.0, 1.0, 1.0, 1, 1.0), _search_started(false), + _tags_released(true), _urgent_latched(false), + _urgent_release_ticks(0), _urgent_search_spent(false), + _search_state(SearchState::RUNNING), + _abandon_reason(SearchAbandonReason::NONE), _search_start_ns(0), + _last_pass_gc_finish_epoch(0), _last_pass_ns(0), + _passes_run(0), + _root_kind_rotation_cursor(1), + _stale_expanded_rotation_cursor(1), + _thread_walk_anchor_cursor(0), + _safepoint_pain_budget(0.0), _search_pain_ms(0), _cpu_pain_budget(0.0), + _thread(), _running(false), _abort_pass_requested(false) {} + + void onGCStart(); + void onGCFinish(); + + static void *threadEntry(void *self) { + ((ReferenceChainTracker *)self)->threadLoop(); + return nullptr; + } + void threadLoop(); + + // Combines Open Question 5's two candidate pass-scheduling triggers + // (design doc's Triggering section) rather than picking one: true if the + // GC-finish epoch has advanced since the last pass ("a GC just happened, a + // pass may be worth running soon") or PASS_CADENCE_NS has elapsed since + // the last pass, whichever comes first. Also true before the first pass + // has ever run. A future measurement pass decides whether one of these + // two triggers should be dropped as unnecessary once real cost data + // exists - for now both are implemented, combined, rather than adding an + // unmeasured config knob to switch between them. + bool shouldRunPass(u64 now_ns); + + // Cheap probe (max=1, not the real poll pollWatchedTargets() makes) into + // LivenessTracker's population-trend table: true if at least one klass + // shows a positive population slope worth chasing. Always true when + // LivenessTracker::gcGenerationsEnabled() is off, since there is no + // candidate signal to gate on in that mode - callers fall back to their + // pre-existing behavior in that case. Shared by canAffordNewSearch() + // (restart gate) and threadLoop()'s own steady-state gate below, so a GC + // with no accompanying population growth doesn't trigger either a restart + // or a fresh pass. + // + // Also true, independent of the per-klass check above, while isUrgent() is + // latched and this urgency episode has not yet authorized a search + // (_urgent_search_spent) - see OOM_URGENT_THRESHOLD_S's own comment for why + // the per-klass gate alone is too slow for an aggressive, heap-wide leak, + // and _urgent_search_spent's for why the shortcut is limited to one search + // per episode. This only removes *this* gate: canAffordNewSearch() (the + // actual restart/first-search decision) still checks the pain budget before + // ever calling this method, so a search already cooling down from a recent + // one's own cost can still be deferred even while this returns true. + bool hasLeakSignal(); + + // Latched, hysteretic view of LivenessTracker::secondsToOOM() crossing + // OOM_URGENT_THRESHOLD_S - see _urgent_latched for the latch/release rules + // and why the raw comparison flaps. When true, runPass() suppresses TTL + // abandonment and threadLoop() tightens cadence + raises the pause + // target so the search completes before the app OOMs. The only SLO is + // the STW pause time, bounded by URGENT_PAUSE_TARGET_MS. const, but + // maintains the latch state (declared mutable) as a side effect, so it + // must be called on every scheduling tick to advance the release counter. + bool isUrgent() const; + + // Abandon the search after this many consecutive passes with + // zero new frontier entries admitted (genuinely stuck, not just slow). + // A large heap takes more passes simply because there are more + // objects to explore — that is not "stuck". Only abandon when the + // frontier stops growing entirely. + // (Moved to public section for test access.) + + // Abandon the search after this many consecutive passes with + // zero new frontier entries admitted (genuinely stuck, not just slow). + // A large heap takes more passes simply because there are more + // objects to explore — that is not "stuck". Only abandon when the + // frontier stops growing entirely. + // (Public for test access.) + + // Search restart gate (this class's own header comment): true once + // _safepoint_pain_budget has drained back to zero (canStartNow()) *and* + // hasLeakSignal() above reports at least one leak candidate. Also reused by + // shouldRunPass() to gate the very first search, not just restarts - the + // pain-budget half is always a no-op there (nothing has been spent yet). + // Always true when LivenessTracker::gcGenerationsEnabled() is off, since + // there is no candidate signal to gate on in that mode (see the header + // comment's last paragraph), so a reference-chains-without-generations + // setup is unaffected either way. + bool canAffordNewSearch(u64 now_ns); + + // Resets every per-search field back to its just-constructed value so the + // next runPass() call takes the "first pass of a search" branch again, + // exactly like a fresh ReferenceChainTracker would. Called by + // shouldRunPass() once a terminal search's tags have already been released + // (runPass() calls releaseSearchTags() itself before returning, so that + // has always already happened by the time this runs) and + // canAffordNewSearch() has approved a restart. Spends the finishing + // search's accumulated cost into _safepoint_pain_budget first, so the *next* + // restart's gate reflects what this one actually cost. Does not touch + // _class_tags/the shared class-tag counter (classTagAllocator.h) - + // classes do not change identity + // across searches, so their resolved names stay valid and do not need + // re-resolving (mirrors _frontier's own stop()/start()-survival + // rationale). frontierTable()'s own resetForRestart() keeps the + // table's allocation but marks every slot unoccupied again, so tags + // restarting from 1 (nextTag()'s only outstanding scheme) do not read back + // stale metadata from the previous search. + void restartSearch(); + + // Marks every entry still queued in _pending_expand EXPANDED and drains the + // queue. Called after a *first-pass*, root-seeded FollowReferences call + // that completed without truncation: an uninterrupted walk from the heap + // roots already visits every admitted object's own outgoing edges inline + // (as part of the same call, not a separate one per object - see + // heapReferenceCallback()'s own comment), so nothing is left FRONTIER by + // accident; this just makes that explicit so a later resumed pass has + // nothing pending to expand. + void markAllFrontierExpanded(); + + // Resumed-pass counterpart to the first pass's root-seeded FollowReferences + // call in runPass(): resolves every not-yet-expanded entry queued in + // _pending_expand via GetObjectsWithTags - design doc Algorithm step 2's + // "resolve currently- + // live tagged frontier objects; objects that fail to resolve are dropped + // (dead - free pruning)" - then calls FollowReferences with the resolved + // object as initial_object to discover its own outgoing edges, exactly as + // the root walk does inline for a first pass. Repeats over newly- + // discovered entries within the same call, stopping the moment + // *edges_admitted reaches `budget` or the frontier table reports capacity + // exhaustion (the same "abort expansion past the cap rather than + // discovering-then-discarding" rule the root-seeded path already follows) + // - leaving the remaining range untouched for a later call to retry. + // + // *frontier_cap_hit distinguishes "budget for this call ran out" (normal; + // the search stays RUNNING, more work remains for the next pass) from + // "the frontier table itself is full" (design doc: "stop admitting new + // entries ... report it" - runPass() treats this as grounds to ABANDON the + // whole search, not just truncate this pass). If GetObjectsWithTags itself + // fails, *truncated is set (there is pending work, just not resolvable + // this call) so runPass() does not mistake that for the search having + // reached natural completion. + // safepoint_ticks: added to (not overwritten - callers may invoke this + // more than once per pass, e.g. ordinary expansion then rotation) with the + // TSC::ticks() duration of just this call's FollowReferences invocation(s) + // - the genuine in-safepoint VM_HeapWalkOperation cost, excluding + // GetObjectsWithTags (not a safepoint call - its own comment above) and + // every other bookkeeping line in this function. See runPass()'s own + // comment for why this needed splitting out from the whole call's + // wall-clock time. + void expandFrontier(jvmtiEnv *jvmti, JNIEnv *jni, int hop_cap, int budget, + int *edges_admitted, bool *truncated, + bool *frontier_cap_hit, u64 *safepoint_ticks); + + // Static-field counterpart to heapRootCallback()'s GC-root enumeration: + // IterateOverReachableObjects's root/stack-ref callbacks never report a + // class's static fields (there is no jvmtiHeapRootKind for STATIC_FIELD - + // translateHeapRootKind()'s own comment), so without this call an object + // retained only via `SomeClass.staticField` is never discovered by either + // root enumeration or expandFrontier() (which only descends from + // already-admitted, non-class frontier entries - class objects are never + // admitted, see heapReferenceCallback()'s own comment). This drives one + // batched FollowReferences(initial_object=) call - + // mirroring expandFrontier()'s array-holder batching, one FollowReferences + // for every loaded class rather than one per class - so + // heapReferenceCallback()'s existing referrer-is-a-pre-tagged-class + // ("rtag < 0") root-like handling actually gets invoked. An empty + // batch_tags set forces exactly one hop past each class, exactly like + // expandFrontier()'s per-level batching: each admitted static-field + // referent becomes an ordinary frontier entry that a later + // expandFrontier() call expands on its own turn. Best-effort: on any + // failure (no JNIEnv, OOM/local-ref exhaustion building the holder array, + // JVMTI error) this simply skips the sweep for this pass rather than + // treating it as this pass's own truncation - it is discovery on top of + // the manual walk, not part of its budget/frontier-cap accounting. + // safepoint_ticks: same accumulate-not-overwrite contract as + // expandFrontier()'s own parameter above - added to with just this call's + // FollowReferences duration. + // + // Chunked and resumable via _static_field_sweep_cursor + // (STATIC_FIELD_SWEEP_CHUNK_CLASSES classes per call, not every loaded + // class at once): a JVM with tens of thousands of loaded classes cannot + // have its entire static-field graph walked by one FollowReferences call + // within a single pass's 5-50ms safepoint deadline (confirmed on a live + // pod - see doc/temp/ investigation notes - truncated=1 on 275/275 + // observed passes, 0-1 edges admitted out of ~34k classes' worth of + // static fields). Restarting from class 0 every truncated attempt, as a + // single-call sweep must, means whichever classes come after wherever the + // deadline hits are structurally unreachable no matter how many times it + // retries. Chunking instead makes guaranteed forward progress through the + // loaded-class list across passes regardless of any one chunk truncating. + // The holder array is filled in reversed order so HotSpot's LIFO + // FollowReferences descent visits classes in ascending original index + // order; on truncation the cursor resumes at the class that was being + // processed (not chunk_end), so classes after the interruption point are + // reached on the next pass rather than skipped for the rest of the lap. + // Each call also reprioritizes the loaded-class list app-classes-first + // before selecting its chunk (see the .cpp body) so a likely leak source + // is reached within the first several chunks instead of only after every + // JDK/platform class has been swept. *cycle_complete is set when this + // call's chunk reaches + // the end of the loaded-class list (a full lap), which the caller uses + // in place of the old single-call "not truncated" check to decide whether + // to update _last_static_field_class_count. + void admitStaticFieldRoots(jvmtiEnv *jvmti, JNIEnv *jni, int hop_cap, + int budget, int *edges_admitted, + bool *truncated, bool *frontier_cap_hit, + bool *cycle_complete, u64 *safepoint_ticks); + + // Clears the live JVMTI tag (via clearTag(), i.e. SetTag(obj, 0)) for + // every FrontierTable entry this search has not already marked ABANDONED - + // design doc's Termination section: "on abandonment or completion, every + // JVMTI tag this search assigned ... must be cleared before the search's + // state is discarded." Does not discard the FrontierTable's own records + // (referrer_klass/parent_tag/depth survive, so reconstructChain() keeps + // working from memory after the search ends) - only the underlying + // object's live JVMTI tag is released, via the same batch + // resolve-then-clear sequence GetObjectsWithTags makes possible for + // expandFrontier()'s resolve-or-drop path. + // + // Returns true if every live tag scanned this call was successfully + // resolved-and-cleared (or there were none to begin with), false if + // GetObjectsWithTags() itself failed - in which case NO entry is marked + // ABANDONED (unlike a resolve failure for an individual tag, which means + // the object is already dead and safe to treat as released): a batch + // GetObjectsWithTags() failure tells us nothing about which, if any, + // objects in the batch are still live, so marking them ABANDONED here + // would let restartSearch() reset _next_tag/the frontier table while a + // still-live object could still be holding this search's JVMTI tag, + // corrupting the next search's tag-uniqueness invariant. Callers must not + // allow a restart until this returns true; calling it again later safely + // retries only the tags still not marked ABANDONED from a prior failed + // call. + bool releaseSearchTags(jvmtiEnv *jvmti, JNIEnv *jni); + + // Pause-time pacing controller (doc/architecture/LiveHeapReferenceChains- + // RemainingWorkPlan.md): feeds `pass_wall_ns` - the wall-clock duration of the FollowReferences/ + // GetObjectsWithTags call runPass() just made (the safepoint-triggering + // call itself, per the design doc's Triggering section; no new + // instrumentation needed, since this class is already the thread blocked + // inside it) - into `_pause_pid`, and scales `_effective_budget`/ + // `_effective_cadence_ns` from its output. Folds Open Questions 2 and 5 + // into the one controller call the plan asks for, rather than two + // separately tuned mechanisms. + // + // `_pause_pid.compute()`'s sign convention (pidController.cpp): a positive + // signal means the measured value came in *under* the controller's + // target - here, the last pass finished comfortably inside + // `_pause_target_ms`, so there is headroom to admit a larger budget next + // time. This is the opposite of ObjectSampler/MallocTracer/ + // NativeSocketSampler's own usage (objectSampler.cpp:220-224, + // mallocTracer.cpp:317-318, rateLimiter.h), which *subtract* the signal + // from their interval because their controlled variable (a sampling + // interval) is inversely related to their target rate; `_effective_budget` + // is directly related to pass duration (more budget -> longer pass), so + // the signal is *added* here instead. + // + // The result is clamped to [floor, _budget] - `_budget` (the config value, + // Arguments::_reference_chains_budget) becomes this controller's ceiling + // rather than a fixed per-pass value, per the plan's "clamped, never above + // the frontier/hop caps": the hop cap and frontier cap stay untouched, + // fixed correctness bounds exactly as before (design doc: "not + // controller-tuned"). Whatever part of the signal the clamp could not + // absorb (`overflow` below) drives `_effective_cadence_ns` instead - the + // plan's "fold the cadence decision into the same controller output rather + // than a second mechanism": a search still running long even at the + // minimum budget backs off the fallback cadence instead of trying to + // shrink the budget further (avoiding a degenerate near-zero budget just + // to hit an aggressive cadence, per the plan's own wording); a search with + // spare headroom even at the maximum (config) budget relaxes the cadence + // toward MIN_EFFECTIVE_CADENCE_NS instead, letting the GC-finish-epoch + // trigger (shouldRunPass(), already unconditional on cadence) make + // progress as often as it fires. + void updatePacing(u64 pass_wall_ticks); + + // Root/stack-ref enumeration passes never reach updatePacing() (runPass()'s + // own comment: their fixed dispatch cost would wrongly throttle + // _effective_budget for every unrelated later pass), but a slow one still + // spends real pause-time-SLO budget the borrow ceiling promised was safe to + // hand out. _borrowed_budget's own comment requires it be revoked the + // instant ANY pass is not comfortably under target, so this call - made in + // updatePacing()'s place for a root-enum pass - only ever revokes, never + // grows the streak/borrow: the warmup counter is calibrated against + // expandFrontier()'s per-node cost, not this call's unrelated fixed cost. + void maybeRevokeBorrowForRootEnumPass(u64 pass_wall_ticks); + + // Tags every not-yet-tagged loaded class (GetLoadedClasses()) with a + // fresh nextClassTag() and resolves its name into _class_tags, via the + // same GetClassSignature + normalizeClassSignature + Profiler::lookupClass + // sequence ObjectSampler::recordAllocation() already uses + // (objectSampler.cpp:76-90) - reusing that normalization helper rather + // than re-deriving it. Run once at the start of every runPass() (before + // FollowReferences) rather than lazily during the walk, because + // GetClassSignature/JNI calls are not allowed from inside + // heapReferenceCallback() (see the file header comment) - by pre-tagging, + // every class_tag the callback sees is already resolvable with no further + // JVMTI/JNI calls of its own. Already-tagged classes (from a previous + // pass) are skipped, not re-resolved. + void resolveLoadedClasses(jvmtiEnv *jvmti, JNIEnv *jni); + + // jvmtiHeapReferenceCallback for runPass()'s FollowReferences call (see + // runPass() below for the full walk). `user_data` is a PassContext* + // (referenceChains.cpp, private to the .cpp - the type never needs to be + // visible here since only runPass() constructs one). + static jint JNICALL heapReferenceCallback( + jvmtiHeapReferenceKind reference_kind, + const jvmtiHeapReferenceInfo *reference_info, jlong class_tag, + jlong referrer_class_tag, jlong size, jlong *tag_ptr, + jlong *referrer_tag_ptr, jint length, void *user_data); + + // Outcome of admitObject() below - lets each of its two call sites + // (heapReferenceCallback() above and the IterateOverReachableObjects root/ + // stack-ref callbacks, both in referenceChains.cpp) translate the same + // admission decision into its own callback-shape-appropriate return + // value/truncation flag, instead of duplicating the decision twice. + enum class AdmitResult { + ALREADY_ADMITTED, // *tag_ptr != 0: nothing to do, not a truncation + HOP_CAP, // depth >= hop_cap: not admitted, not a truncation + BUDGET_EXHAUSTED, // edges_admitted >= budget: this pass's cap + FRONTIER_CAP_HIT, // FrontierTable::insert() itself is full: stops + // admitting new entries but does not itself abandon + // the search (see runPass()'s frontier_cap_hit + // handling - the no-progress detector abandons only + // if the frontier then stops growing) + ADMITTED, + }; + + // First-discovery admission core (implementation plan's Phase 4 item 2): + // factored out of heapReferenceCallback()'s inline admission branch so the + // manual-walk driver's root/stack-ref callbacks stay in sync with + // FollowReferences' own admission by construction, not by copy-paste. + // `*tag_ptr` is the in/out tag slot each call site already has as a real + // JVMTI out-parameter, and `*edges_admitted` + // is likewise each call site's own running counter for this call. On + // AdmitResult::ADMITTED, `*tag_ptr` is filled with the freshly assigned tag + // and the tag is queued onto _pending_expand (or, when `priority` is true, + // onto _priority_expand instead - see that field's own comment) exactly as + // heapReferenceCallback() already did inline. + // `class_tag` is the raw JVMTI class tag of the object being admitted - + // both real call sites (heapReferenceCallback()/heapRootCallback()) have + // this in hand already as their own JVMTI callback parameter, so no new + // JVMTI call is needed to supply it. Stored into the new entry's + // FrontierEntry::class_tag (see that field's own comment) and forwarded + // to trackLeakAccumulation() below. + AdmitResult admitObject(FrontierTable *frontier, int hop_cap, int budget, + int *edges_admitted, jlong *tag_ptr, + jlong parent_tag, u32 referrer_klass, u32 depth, + u8 root_kind, jlong class_tag, + bool priority = false, + jint edge_field_index = -1, u8 edge_kind = 0, + jlong edge_referrer_class_tag = 0); + + // Called by admitObject() on every successful ADMITTED result (root or + // non-root, ordinary or priority) - the single shared admission path, so + // this needs no duplicate call site at heapReferenceCallback()/ + // heapRootCallback()/stackRefCallback(). O(1) in the common case + // (_watched_leak_klass_count == 0, before any leak signal has fired - + // just one integer compare) and O(MAX_WATCHED_LEAK_KLASSES) plus one + // frontier lookup when something is being watched - see + // _leak_signature_totals/_leak_parent_fanout's own comments for what this + // actually records and collectLeakAccumulationCandidatesForRotation()'s + // own comment for the full design this feeds. `class_tag` is the newly- + // admitted object's own raw JVMTI class tag (FrontierEntry::class_tag), + // NOT referrer_klass (a classMap dictionary id) - see class_tag's own + // comment for why matching against _watched_leak_klass_ids requires the + // stable tag, not the compactable dictionary id. + void trackLeakAccumulation(FrontierTable *frontier, jlong class_tag, + jlong parent_tag, jlong tag); + + // Record a discovered instance for a watched candidate class: store its + // frontier tag in the class's discovery slots so pollWatchedTargets() can + // build its chain event. Shared by the ordinary auto-mark path + // (heapReferenceCallback on admission), the leak-tag interception path, + // and correlateAdmittedLeakTag() below - the latter two pass + // leak_correlated=true, which lets them EVICT a slot held by an + // uncorrelated (noise) instance when all slots are full. Slots are bounded + // by MAX_DISCOVERED_INSTANCES_PER_CLASS and noise instances can fill them + // before the leak-tagged ones are ever reached - without preferential + // eviction the tracked leak instances would be silently discarded + // (observed live: 8 depth-1 jni_local/stack_local noise instances + // permanently occupying all slots of the watched [B class). + // Single-thread: only called from the pass thread (heapReferenceCallback) + // and the tracker's own thread (correlateAdmittedLeakTag via + // tagLeakInstances from pollWatchedTargets) - never concurrently. + // Builds and caches chain events for every discovered instance recorded + // against a slot holding klass_id - slot-driven so instances recorded while + // a candidate still qualified are not stranded when it stops qualifying + // (see the definition's own comment in referenceChains.cpp for the observed + // live failure this closes and the two call sites). + void buildDiscoveredInstanceChains(jvmtiEnv *jvmti, JNIEnv *jni, + u32 klass_id, u64 current_search_ns); + + void recordDiscoveredInstance(u32 klass_id, jlong frontier_tag, + bool leak_correlated); + + // Correlate a leak tag with an instance the BFS admitted BEFORE + // tagLeakInstances() tagged it (its JVMTI tag is a frontier tag, its + // frontier entry has leak_tag == 0). Sets the entry's leak_tag so chain + // events emit targetTag = the leak tag (the HeapLiveObject correlation + // key), and records the instance as discovered. Returns false if the tag + // resolves to no live frontier entry (caller should treat the object as + // un-tagged). Idempotent: an entry that already carries a leak tag just + // returns true. Also handles post-restart re-admission: the new entry for + // a re-admitted instance gets the SAME leak tag the pool already holds + // for it (LivenessTracker's record survives the search restart). + // Public: LivenessTracker::tagLeakInstances() (livenessTracker.cpp) + // calls it - see the public section below for the declaration. + + // One-time retroactive catch-up for a klass_id the moment it FIRST enters + // _watched_leak_klass_ids (pollWatchedTargets() calls this only for the + // newly-added ids in each refresh, never for ones already being watched). + // trackLeakAccumulation() above only fires on NEW admissions + // (admitObject()'s ADMITTED result) - it cannot see objects that were + // already admitted before this klass_id started being watched, which for + // a klass that has been growing for a while (the exact case this + // mechanism targets) can be nearly all of them, found the hard way: the + // container that actually needs re-expansion typically already got fully + // admitted in an early pass, long before selectLeakCandidates()'s own + // hysteresis gate ever authorized watching it, leaving + // _leak_signature_totals/_leak_parent_fanout permanently empty with no + // way to ever get their first data point. This scans the WHOLE frontier + // table once (bounded by table size, same cost class as + // collectStaleExpandedEntriesForRotation()'s existing per-pass scan, but + // this one runs only on the rare newly-watched-klass event - at most + // MAX_WATCHED_LEAK_KLASSES times per search, not every pass) for + // already-EXPANDED entries whose (u32) class_tag matches klass_id (both + // sides of the comparison are truncated the same way - see + // _watched_leak_klass_ids' own comment for why a full jlong is not + // needed), and feeds each one through the same aggregation logic + // trackLeakAccumulation() uses for new admissions, applied retroactively - + // so ongoing incremental updates compose cleanly on top of this baseline + // without double-counting. + void seedLeakAccumulationForNewlyWatchedKlass(u32 klass_id); + + // Durability tie-break (design doc's "Fix for root-attribution staleness" + // point 1 / Phase 5 item 1) for an object rediscovered as a heap root by + // heapRootCallback()/stackRefCallback() while already admitted (same pass + // or a previous one) - factored out of those callbacks, rather than + // inlined, so it is unit-testable without a PassContext/JVMTI mock (both + // callbacks' user_data type is private to referenceChains.cpp). Only ever + // overwrites root_kind - never parent_tag - and only for an entry that is + // already root-attached (entry.parent_tag == 0); this is the option (a) + // resolution of the parent_tag==0/root_kind invariant conflict (Phase 5's + // own callout): a re-expansion-driven, non-root rediscovery of an edge to + // some already-tracked, non-root-attached object must never reach this + // method at all (re-expansion child admission always passes + // root_kind=0, which loses every tie-break, so it structurally cannot + // trigger an upgrade even if it were mistakenly routed here). Returns true + // if an upgrade was applied, false otherwise (already at least as durable, + // not root-attached, or not found) - purely informational for callers/ + // tests, not required for correctness. + bool maybeUpgradeRootAttachedRootKind(FrontierTable *frontier, jlong tag, + u8 new_root_kind); + + // True if `tag` is already sitting in _priority_expand - either queued + // earlier this same pass by the other rotation collector, or left over + // from a prior pass's truncated batch (expandFrontier() leaves those at + // the front of the queue for a later retry rather than popping them). + // Shared by both rotation collectors below so neither can push a tag + // that's already pending re-expansion. O(1) via _priority_expand_set - + // the rotation collectors run this check for EVERY FrontierTable slot + // they visit (~199k EXPANDED entries on a large heap), so the original + // linear scan over the deque was ~200M comparisons per rotation pass at + // the PRIORITY_EXPAND_CAP (observed prominently in profiles); the + // "sub-millisecond" claim its original comment made only held for the + // per-SELECTION calls it was written for, not the per-slot visits the + // collectors actually make. + bool isQueuedForRotation(jlong tag) const { + return _priority_expand_set.contains(tag); + } + + // Bounded rotating re-expansion (design doc's closing section / Phase 5 + // item 3): each manual-walk pass, feed up to `max_count` already-EXPANDED, + // root-attached entries whose root_kind is still transient + // (isTransientRootKind()) back into _priority_expand so + // expandFrontier() re-walks their fields - giving a stale root_kind + // another chance to be superseded by a durable root discovered elsewhere + // in the interim, via the same admitObject()/tie-break machinery every + // other admission uses. Scans FrontierTable slots in tag order starting + // from _root_kind_rotation_cursor, wrapping at size(), so repeated calls + // sweep the whole table over time instead of only ever revisiting the + // first `max_count` transient entries. Pure table scan/queue push - no + // JVMTI call of its own - so it is unit-testable directly. + // Returns the tags selected (also already pushed onto _priority_expand). + std::vector collectStaleRootKindEntriesForRotation(int max_count); + + // Bounded rotating re-expansion for stale mutable fields: + // expandFrontier() observes an object's outgoing references exactly once + // (on the FollowReferences call that marks it EXPANDED) and never + // revisits it, so a field that is later reassigned to point at a + // different object - e.g. HashMap.table on resize - has its new value + // permanently unobserved once the map itself is EXPANDED; the old table + // array's own frontier entry eventually resolves to a dead object via + // GetObjectsWithTags and gets silently cleared with zero children, + // orphaning everything only reachable through the *current* table. + // Feeds up to `max_count` already-EXPANDED entries (any parent_tag/ + // root_kind - unlike collectStaleRootKindEntriesForRotation() above, which + // is scoped to root-attached transient entries for a different reason) + // back into _priority_expand so expandFrontier() re-runs FollowReferences + // on them and observes their current field values. Already-admitted + // children are ALREADY_ADMITTED no-ops (admitObject()'s own idempotency); + // only a genuinely new edge (i.e. a mutated field) is admitted. Scans + // FrontierTable slots in tag order starting from + // _stale_expanded_rotation_cursor, wrapping at size() - same rationale as + // collectStaleRootKindEntriesForRotation() above: a fixed always-from-1 + // scan lets a large, permanently-EXPANDED low-tag population (long-lived + // infrastructure objects) monopolize every pass's cap forever, starving + // any higher-tag entry (e.g. a static field's collection, admitted only + // once its class loads) of ever being re-queued. + std::vector collectStaleExpandedEntriesForRotation(int max_count); + + // Bounded rotating re-expansion targeting the accumulation point of a + // klass LivenessTracker has flagged as growing (LivenessTracker:: + // topKlassesByGenerationCount(), _watched_leak_klass_ids) - the design's + // actual targeted tier. See its own definition comment (referenceChains.cpp) + // for the full two-tier design (class-level growth ranking, then per- + // parent fanout ranking within the winner) and why depth/root-durability/ + // class-shape heuristics alone were measured and found insufficient. + // Unlike the other two rotation collectors, has no wrapping cursor - it + // always selects the current best candidate(s), which is the desired + // behavior here (re-selecting a still-growing parent every pass), not + // something a fairness-across-passes guarantee needs to correct for. + std::vector collectLeakAccumulationCandidatesForRotation( + int max_count); + + // CANDIDATE-SCOPED REACH: bounded descend walk from an anchor object. + // FollowReferences(initial_object=anchor) with batch_tags == nullptr so + // descent is gated only by hop_cap + the pass deadline, reusing + // heapReferenceCallback() unchanged - leak-tag interception, canary + // pruning, auto-mark and improveChain all work as-is, and every admitted + // child chains back to the anchor's own frontier entry, so an interception + // during the walk yields the complete root->...->chunk chain in ONE + // bounded STW call. The walk's ctx.hop_cap is lowered to + // min(_hop_cap, anchor_depth + DESCENT_HOPS), bounding admission to a few + // hops below the anchor. Motivation (pod rounds 5-6, + // ev-leaktag-onpod-round5/6): breadth-first FIFO expansion over a rising + // heap can never drain (_pending_expand net-growing, per-lap sweep + // re-admissions replenishing it), so the tagged leak instances sit under + // holders the crawl reaches only after hours - candidate-scoped walks + // make the reach independent of the backlog. The walk prunes descent + // (and admission) into a small fixed set of fat-metadata classes + // (java.lang.ClassLoader, java.lang.ThreadGroup, + // java.security.ProtectionDomain - exact class-tag match) resolved fresh + // per call: without this, a Thread anchor's contextClassLoader edge would + // descend into every loaded class and its statics, sweep-scale cost per + // thread per pass. When `anchor_descend_class_tag` is non-zero, the + // ANCHOR's own outgoing edges are additionally gated to descend only + // into referees of that exact class (used by walkCandidateThreadLocals() + // to descend only into ThreadLocal$ThreadLocalMap - the value type of + // BOTH of Thread's threadLocals and inheritableThreadLocals fields), + // letting the walk skip the Thread's other (transient, metadata-heavy) + // instance fields entirely. Deliberately a class-tag comparison, NOT + // jvmtiHeapReferenceInfoField.index matching: that index is a jint field + // ordinal whose correspondence to GetClassFields() order the design has + // no need to depend on. + void descendFromAnchor(jvmtiEnv *jvmti, JNIEnv *jni, jobject anchor, + jlong anchor_tag, u32 anchor_depth, + jlong anchor_descend_class_tag, int budget, + int *edges_admitted, bool *truncated, + bool *frontier_cap_hit, u64 *safepoint_ticks, + bool diag_trace = false); + + // Prong 1 of the candidate-scoped reach design (thread-retained taxonomy: + // ThreadLocal-held caches and thread-owned collections): per pass, walk + // up to THREAD_WALK_MAX_ANCHORS of the current candidates' qualifying + // tids' live Thread objects (registerThreadObject()'s map above) with + // descendFromAnchor() (anchor-gated to ThreadLocalMap, see above). A + // thread-local accumulation's holder chain + // is entirely inside the Thread object's own ThreadLocalMap subgraph, so + // the walk reaches the tagged instances regardless of the ordinary + // BFS backlog - which the whole-heap crawl demonstrably never does in a + // rising heap. Anchor admission is idempotent across passes (GetTag + + // FrontierTable lookup, reuse-or-retag), so re-walking a thread per pass + // is cheap and ALREADY_ADMITTED-safe. + void walkCandidateThreadLocals(jvmtiEnv *jvmti, JNIEnv *jni, int budget, + int *edges_admitted, bool *truncated, + bool *frontier_cap_hit, u64 *safepoint_ticks); + + // Prong 2 (durable-root-retained taxonomy): select up to `max_count` root- + // attached entries held by a DURABLE root kind (parent_tag == 0, + // root_kind STATIC_FIELD or JNI_GLOBAL, FRONTIER or EXPANDED) with a + // TIERED cursor (leak-tagged first, then FRESH anchors - not yet given + // their one first-look walk priority, the round-15 queue lane - then + // container-shaped anchors, then everything else cursor-fairly — see + // collectStaticFieldAnchorsForRotation()'s own comment) - and + // descend-walk each via walkStaticFieldAnchors(). A static Map/List's + // leaked chunks sit 3-4 hops below its root-attached holder, deeper + // than the one-hop Tier-2 rotation can reach from an un-expanded + // FRONTIER holder; a descend walk covers the holder's whole internal + // structure in one bounded call. JNI_GLOBAL joined the filter after pod + // round 7 (interception zero with statics fully walked): a leaking + // holder retained through a JNI global is the same taxonomy shape and + // the same walk covers it - covering it in the SAME rebuild avoids a + // second deploy cycle if the pod's holder turns out to be one. + std::vector collectStaticFieldAnchorsForRotation(int max_count); + // The B' at-risk push itself (both call sites below): dedupe via the + // set, cap-drop when the FIFO is full, per-class quota drop at + // STATIC_ANCHOR_ATRISK_PER_KLASS_CAP (round 16 - see + // _static_anchor_fifo_klass_counts), count admitted pushes. No return + // value - a dropped push is silently retried by the feed's next event + // (the next static edge onto the entry, or the next demotion). Engine + // thread only. + void pushAtRiskStaticAnchor(jlong tag, u32 klass_id); + // Add `tag` to _static_anchor_index if its root_kind is a durable + // anchor-tier kind (STATIC_FIELD or JNI_GLOBAL). Called at first + // admission and at root-kind upgrade. `own_class_tag` is the class tag + // of the anchor OBJECT itself (heapReferenceCallback's class_tag param / + // FrontierEntry::class_tag) - kept in the parallel array so selection can + // tier by class shape without JNI. Idempotent (dedup via a linear + // scan of the small vector — the anchor population is O(hundreds), + // well under the 256-element linear-scan cutoff). Engine thread only. + void addToStaticAnchorIndex(jlong tag, jlong own_class_tag, u8 root_kind); + + // True iff `klass` implements java/util/Collection or java/util/Map, + // directly or transitively (superclass chain + interfaces of every + // visited class, depth-bounded, visited set to survive interface + // diamonds). The two interface class tags are resolved once and cached + // (_collection_iface_class_tag/_map_iface_class_tag). Caller owns local- + // ref hygiene for the jclasses this walks. Engine thread only. + bool classImplementsContainerOrMap(jvmtiEnv *jvmti, JNIEnv *jni, + jclass klass); + + // Resolve _collection_iface_class_tag/_map_iface_class_tag once; returns + // false if the interfaces cannot be resolved yet (leaves them at -1 so + // the next call retries). Engine thread only. + bool resolveContainerInterfaceTags(jvmtiEnv *jvmti, JNIEnv *jni); + + // Lazy shape reconciliation for the anchor index: scans + // _static_anchor_own_class_tags for class tags not yet in + // _class_shape_cache, resolves up to ANCHOR_SHAPE_RECONCILE_BUDGET of + // them per pass via one GetObjectsWithTags call (class objects are + // tagged with their class tags) and classifies each. Selection treats + // unclassified anchors as the lowest tier, so classification lag only + // delays a container's promotion - it never drops coverage. Runs on the + // engine thread with JNI available, outside any frontier lock and + // outside heap callbacks. TEMP: also emits the per-pass cohort + // histogram (round-14 measurement: container cohort size vs the ~4k + // per-search walk coverage) - remove once the arithmetic is verified + // on-pod. + void reconcileAnchorClassShapes(jvmtiEnv *jvmti, JNIEnv *jni); + // Pops up to max_count AtRiskAnchor entries off _static_anchor_fifo's + // front into `out` (appending), decrementing each popped entry's class + // occupancy in _static_anchor_fifo_klass_counts (erased at zero, so the + // map tracks the FIFO's live contents), and re-derives the set from the + // deque's remaining contents (PriorityExpandSet's tombstone-free + // rebuildFrom contract). Returns the drained count. Engine thread only. + int drainStaticAnchorFifo(int max_count, std::vector &out); + // Pushes `entries` back to _static_anchor_fifo's FRONT in reverse order + // (preserving FIFO order), re-incrementing each entry's class occupancy, + // and rebuilds the set - the truncated-walk requeue path. Caller passes + // ONLY entries it drained from the FIFO this pass (never collector- + // sourced ones: those keep their own cursor retention). Engine thread + // only. + void requeueStaticAnchorFifoFront(const std::vector &entries); + // When non-null, receives the tags of RESOLVED-but-unwalked anchors at + // the truncation break point - GetObjectsWithTags may return fewer + // anchors than requested (dead tags drop out) in its own order, so the + // caller cannot recover the un-walked set from a consumed index; the + // walk hands the exact tags back instead. Dead (unresolved) anchors are + // omitted: they must not be requeued anywhere. + void walkStaticFieldAnchors(jvmtiEnv *jvmti, JNIEnv *jni, + const std::vector &anchor_tags, + int budget, int *edges_admitted, bool *truncated, + bool *frontier_cap_hit, u64 *safepoint_ticks, + std::vector *unwalked = nullptr); + + // jvmtiHeapRootCallback/jvmtiStackReferenceCallback for runPassManualWalk()'s + // IterateOverReachableObjects call (referenceChains.cpp). `user_data` is a + // PassContext* (the same private-to-the-.cpp type heapReferenceCallback() + // already uses above) - both callbacks only ever admit a root-attached + // entry (parent_tag=0, depth=0), translating the JVMTI-owned + // jvmtiHeapRootKind into FrontierEntry::root_kind's jvmtiHeapReferenceKind + // numbering first (see referenceChains.cpp's translateHeapRootKind() for + // why this translation is required, not optional). + static jvmtiIterationControl JNICALL + heapRootCallback(jvmtiHeapRootKind root_kind, jlong class_tag, jlong size, + jlong *tag_ptr, void *user_data); + static jvmtiIterationControl JNICALL stackRefCallback( + jvmtiHeapRootKind root_kind, jlong class_tag, jlong size, + jlong *tag_ptr, jlong thread_tag, jint depth, jmethodID method, + jint slot, void *user_data); + + // Manual-walk pass driver (implementation plan Phase 4): when + // `run_root_enum` is true, seeds/refreshes root-attached frontier entries + // via IterateOverReachableObjects (heapRootCallback()/stackRefCallback() + // above) using `root_enum_budget`; then, regardless of `run_root_enum`, + // drains _pending_expand via admitStaticFieldRoots()/expandFrontier() up to + // `expand_budget`. The two budgets are independent, not shared - see + // ROOT_ENUM_MIN_INTERVAL_NS's own comment for why root enumeration gets its + // own, much larger, infrequent allowance instead of competing with the + // small steady-state budget every tick's expansion uses. runPass() decides + // `run_root_enum`/`root_enum_budget`; when false, this call takes the + // cheap expandFrontier()-only path over whatever the last enumeration + // already admitted into the frontier. + // *safepoint_ticks is zeroed here, then accumulated (via + // IterateOverReachableObjects's own timing below plus + // admitStaticFieldRoots()/expandFrontier()'s additive parameters) with just + // the genuine in-safepoint JVMTI call cost this pass incurred - runPass() + // uses it (not this whole call's wall-clock time) as the pacing signal. + void runPassManualWalk(jvmtiEnv *jvmti, JNIEnv *jni, bool run_root_enum, + int root_enum_budget, int expand_budget, + int *edges_admitted, bool *truncated, + bool *frontier_cap_hit, u64 *safepoint_ticks); + + // Inserts (or refreshes) klass_id's resolved chain in _resolved_chains, + // recording the source_tag/source_search_ns it was reconstructed from so a + // later poll can tell a stale entry from a current one. Drops (counting via + // REFERENCE_CHAIN_EVENTS_DROPPED) rather than evicting when a brand-new + // klass_id arrives with the cache already at MAX_RESOLVED_CHAINS - see that + // constant's own comment and this method's definition (referenceChains.cpp). + void cacheResolvedChain(jlong source_tag, ReferenceChainEvent &&event, + jlong source_tag_val, u64 source_search_ns); + + // Remove a cached chain so pollWatchedTargets rebuilds it on the next + // poll. Called when improveChain updates a frontier entry with a + // deeper path — the cached chain (built from the old shallow entry) + // must be discarded so the deeper chain is emitted instead. + void invalidateResolvedChain(jlong source_tag); + + // Snapshots the just-abandoned search into _pending_abandoned_events - + // called from runPass() (referenceChains.cpp) immediately after it writes + // SearchState::ABANDONED, while buildAbandonedEvent()'s source fields are + // still valid (see _pending_abandoned_events' own comment for why this + // cannot be deferred to dump()-time). A no-op (dropped, counted via + // REFERENCE_CHAIN_EVENTS_DROPPED) if the queue is already at + // MAX_PENDING_ABANDONED_EVENTS. + void enqueuePendingAbandonedEvent(); + +public: + static ReferenceChainTracker *instance() { + static ReferenceChainTracker instance; + return &instance; + } + + // tid -> java.lang.Thread global-ref registry (see _thread_objects). + // registerThreadObject() no-ops while !_enabled (Profiler::onThreadStart + // calls it unconditionally); unregisterThreadObject() is deliberately NOT + // gated on _enabled - a thread that started while enabled must release + // its global ref even after the recording stopped, otherwise the ref + // leaks for the JVM's lifetime. + void registerThreadObject(JNIEnv *jni, int tid, jthread thread); + void unregisterThreadObject(JNIEnv *jni, int tid); + + // One-time sweep over the JVM's CURRENTLY LIVE threads at recording start, + // registering each into the same tid -> Thread-object registry via + // JVMThread::nativeThreadId(). Profiler::onThreadStart() cannot cover this + // population: those threads started before the recording began, and a + // leaking thread is typically among them (alive since process start). + // Called from Profiler::start(), not this class's own start(): gtest + // binaries call start() directly against partial mock JVMTI tables without + // a GetAllThreads slot, and only the real profiler lifecycle guarantees a + // fully populated JVMTI/JNI environment. + void registerExistingThreads(jvmtiEnv *jvmti, JNIEnv *jni); + + // Correlate a leak tag with an instance the BFS admitted BEFORE + // tagLeakInstances() tagged it (its JVMTI tag is a frontier tag, its + // frontier entry has leak_tag == 0). Sets the entry's leak_tag so chain + // events emit targetTag = the leak tag (the HeapLiveObject correlation + // key), and records the instance as discovered. Returns false if the tag + // resolves to no live frontier entry (caller should treat the object as + // un-tagged). Idempotent: an entry that already carries a leak tag just + // returns true. Also handles post-restart re-admission: the new entry for + // a re-admitted instance gets the SAME leak tag the pool already holds + // for it (LivenessTracker's record survives the search restart). + bool correlateAdmittedLeakTag(jlong frontier_tag, jlong leak_tag, + u32 klass_id); + + // Abandon the search after this many consecutive passes with + // zero new frontier entries admitted (genuinely stuck, not just slow). + // A large heap takes more passes simply because there are more + // objects to explore — that is not "stuck". Only abandon when the + // frontier stops growing entirely. + static constexpr int NO_PROGRESS_PASS_LIMIT = 30; + + // Base limit for the canary-specific stuck detector: candidate-discovery + // must show no progress (no change to _candidate_found_bits, no new + // candidate admitted into a slot) for this many consecutive passes AND + // the whole-graph frontier must also have stalled for NO_PROGRESS_PASS_LIMIT + // passes (see runPass()'s CANARY_STUCK branch) before a canary search is + // abandoned. The whole-graph requirement was added after live evidence + // showed the frontier still growing tens of thousands of entries deep + // while chasing a specific, confirmed-reachable candidate - a canary + // search is not "stuck" just because it hasn't found its candidate yet + // if the graph walk itself is still making real progress toward it. + // canaryStuckPassLimit() escalates this base value across consecutive + // CANARY_STUCK restarts of the same candidate-chase sequence (see + // _canary_stuck_restart_count), since a fixed cutoff cannot distinguish + // "genuinely unreachable within any reasonable budget" from "reachable, + // but deeper than one restart cycle can cover" - a large/deep heap + // legitimately needs more passes, not a smaller one. + static constexpr int CANARY_NO_PROGRESS_PASS_LIMIT = 30; + + // Upper bound on how many times canaryStuckPassLimit() doubles the base + // limit (2^8 = 256x -> 7680 passes at the default base of 30) - bounds + // the escalation so a search that is ACTUALLY stuck forever (as opposed + // to merely deep) still gets abandoned in finite time rather than + // growing its patience without limit. + static constexpr int MAX_CANARY_STUCK_BACKOFF_SHIFT = 8; + + // The canary-stuck pass limit for the *current* restart attempt: + // CANARY_NO_PROGRESS_PASS_LIMIT doubled once per consecutive CANARY_STUCK + // restart of this candidate-chase sequence, capped at + // MAX_CANARY_STUCK_BACKOFF_SHIFT doublings. + int canaryStuckPassLimit() const { + return CANARY_NO_PROGRESS_PASS_LIMIT + << std::min(_canary_stuck_restart_count, + MAX_CANARY_STUCK_BACKOFF_SHIFT); + } + + // Multiplier cap for the canary lane's work-scaled backoff (see + // _canary_backoff_mult's own comment). 16 bounds a stuck chase's steady + // burn to ~1/16 of a core on pass work while keeping a deep-but-cheap + // (ms-scale passes) chase dense enough to resolve within a scenario's + // round window - measured against ReferenceChainTrackingTest's ~200-pass + // deep chase, which a fixed 1s cap starved outright (held-off wakes + // outpaced the test window). Abandonment is not this knob's job + // (CANARY_STUCK's frontier-aware detector owns that); this only paces. + static constexpr int CANARY_BACKOFF_MULT_MAX = 16; + + // While a canary search has candidates still unresolved, shouldRunPass() + // raises _cpu_pain_budget's refill rate by this factor (capped at + // 100%/wall-clock). With the canary lane's rate now bounded by + // _canary_backoff_mult (above), this is no longer a rate control at all - + // it exists only so the conservative base refill tuned for the ordinary + // ~1 pass/s whole-graph cadence does not double-throttle a chase the + // backoff has already paced. The old covering (15x) vs emergency (100x) + // distinction is gone: both existed to feed the back-to-back mode the + // backoff replaces. + static constexpr double CANARY_PAIN_BUDGET_REFILL_MULTIPLIER = 100.0; + + // Coverage tracking: how many leak tags have been assigned vs resolved. + // When all assigned tags are resolved (have chains), drop to 1x. + int _leak_tags_assigned = 0; + int _leak_tags_resolved = 0; + + // Base marker tag for canary-search candidates. Each candidate i + // gets MARKER_TAG_BASE - i (distinct negative values) so + // heapReferenceCallback() can tell which candidate was found. + // Negative to avoid collision with frontier tags (positive + // jlong from _next_tag, referenceChains.h:623). Class tags are + // always negative (nextClassTag() at referenceChains.h:1671), + // so a negative marker is disjoint from the frontier + // tag space. + static constexpr jlong MARKER_TAG_BASE = -(1LL << 62); + + // Leak tags are positive JVMTI tags in a dedicated range, assigned by + // LivenessTracker's tag pool to specific tracked leaking objects. The + // BFS recognizes them by range check and admits the object into the + // frontier, storing the leak tag in FrontierEntry::leak_tag for + // correlation with HeapLiveObject events. Unlike marker tags (one per + // candidate class), leak tags are per-instance — each tracked leaking + // object gets its own tag from a reusable pool. + static constexpr jlong LEAK_TAG_BASE = 0x40000000LL; + static constexpr int LEAK_TAG_POOL_SIZE = 256; + + // Check whether a JVMTI tag is a leak tag (from LivenessTracker's pool). + static bool isLeakTag(jlong tag) { + return tag >= LEAK_TAG_BASE && tag < LEAK_TAG_BASE + LEAK_TAG_POOL_SIZE; + } + + // Max candidates LivenessTracker::selectLeakCandidates() can return. Must match + // LivenessTracker::MAX_LEAK_CANDIDATES (livenessTracker.h:133). Duplicated here + // to avoid a heavy include chain (livenessTracker.h pulls jvmti.h). + // (Also declared in the private section above for field sizing.) + + // Test accessor for _passes_since_last_progress. + int passesSinceLastProgressForTest() const { return _passes_since_last_progress; } + // Canary-lane backoff state - see _canary_backoff_mult's own comment. + int canaryBackoffMultForTest() const { return _canary_backoff_mult; } + u64 canaryPassEmaMsForTest() const { return _canary_pass_ema_ms; } + u64 lastCanaryPassNsForTest() const { return _last_canary_pass_ns; } + void setCanaryBackoffForTest(int mult, u64 ema_ms, u64 last_pass_ns) { + _canary_backoff_mult = mult; + _canary_pass_ema_ms = ema_ms; + _last_canary_pass_ns = last_pass_ns; + } + void setOomRampActiveForTest(bool active) { _oom_ramp_active = active; } + int candidateCountForTest() const { return _candidate_count; } + void setCandidateCountForTest(int n) { _candidate_count = n; } + void setCandidateKlassIdForTest(int idx, u32 klass_id) { + _candidate_klass_ids[idx] = klass_id; + } + jlong candidateDiscoveredTagForTest(int slot, int idx) const { + return _candidate_discovered_tags[slot][idx]; + } + int candidateDiscoveredCountForTest(int slot) const { + return _candidate_discovered_count[slot]; + } + u64 candidateFoundBitsForTest() const { return _candidate_found_bits; } + void setCandidateFrontierTagForTest(int idx, jlong tag) { _candidate_frontier_tags[idx] = tag; } + int passesSinceLastCandidateProgressForTest() const { return _passes_since_last_candidate_progress; } + int canaryStuckRestartCountForTest() const { return _canary_stuck_restart_count; } + + ReferenceChainTracker(const ReferenceChainTracker &) = delete; + ReferenceChainTracker &operator=(const ReferenceChainTracker &) = delete; + + Error start(Arguments &args); + + // Scales unset referencechains defaults (budget, ttl, framecap, + // pausetarget, painbudget, firstpassbudget) from the process's max heap + // size and available processor count, so a large heap doesn't starve + // the BFS (the defaults are tuned for a small heap and abandon via + // TTL before making meaningful progress). Only overrides defaults + // that the operator did not set explicitly (tracked by + // args._reference_chains_tuned_mask). hop_cap is left alone: it bounds + // chain depth, not search breadth, and 200 is already generous. + void autoTuneDefaults(Arguments &args); + void stop(); + + // Spawns the BFS thread (threadEntry()/threadLoop()) if reference chain + // tracking is enabled and no thread is already running. Deliberately kept + // separate from start() itself: start() must stay safely callable with no + // live JVM attached (referenceChains_ut.cpp calls it directly against a + // mocked jvmtiEnv, with VM::_vm never set), while startThread()'s + // threadLoop() calls VM::attachThread() unconditionally - safe only once + // the JVM is actually up. Wired from Profiler::start() (profiler.cpp), + // which only calls this after the JVM/JVMTI environment is fully + // initialized, resolving the ordering concern start()'s own comment used + // to raise. No-op if disabled or already running. + void startThread(); + + // Stops and joins the BFS thread started by startThread(), mirroring + // BaseWallClock::stop()'s pthread_kill(WAKEUP_SIGNAL) + pthread_join() + // shape (wallClock.cpp) - WAKEUP_SIGNAL is already installed + // unconditionally in vmEntry.cpp, so no extra signal setup is needed here. + // No-op if the thread was never started. + void stopThread(); + + bool enabled() const { return _enabled; } + + u64 gcStartEpoch() { return load(_gc_start_epoch); } + u64 gcFinishEpoch() { return load(_gc_finish_epoch); } + + // Tag round-trip helpers, reused by resolveLoadedClasses()/ + // heapReferenceCallback() (the heap-walk engine) to drive FrontierTable's tag-indexed + // slots. + jlong nextTag() { return atomicIncRelaxed(_next_tag, (jlong)1); } + + // Serializes runPass()+pollWatchedTargets() between threadLoop() and the + // test seams - see runPassForTest()'s comment. A full pthread mutex, not + // a spin lock: the critical section is a whole BFS pass (tens of ms), far + // too long to spin, and neither holder is ever a signal context. + Mutex _engine_lock; + jlong tagObject(jvmtiEnv *jvmti, jobject obj); + jlong getTag(jvmtiEnv *jvmti, jobject obj); + void clearTag(jvmtiEnv *jvmti, jobject obj); + + // Hands out a fresh negative class tag, from the shared, process-wide + // counter both this class and LivenessTracker mint from - see + // classTagAllocator.h's own header comment for why this must be shared + // rather than a private counter here. Exposed (not just used internally + // by resolveLoadedClasses()) so tests can drive class tagging directly + // against a mocked jvmtiEnv without going through GetLoadedClasses. + jlong nextClassTag() { return ClassTagAllocator::next(); } + + // Returns the frontier metadata table, or nullptr if the subsystem was + // never started with the flag enabled. + FrontierTable *frontierTable() { return _frontier; } + + // Returns the class-tag resolution table. Exposed for testing in + // isolation, matching frontierTable()'s existing rationale. + ClassTagTable *classTags() { return &_class_tags; } + + // Runs exactly one bounded BFS pass and returns. The first call for a + // search seeds FollowReferences from the heap roots (heap_filter=0, + // klass=NULL, initial_object=NULL - see this method's own comment in + // referenceChains.cpp for why FollowReferences rather than + // IterateThroughHeap); every later call resumes from the persisted + // frontier via expandFrontier() instead of re-walking from the roots (see + // expandFrontier()'s comment for why - re-walking from the roots each call + // would re-traverse the entire already-discovered subgraph every pass, + // defeating the point of a per-pass budget). Newly discovered objects are + // admitted into frontierTable() up to _hop_cap/_budget/the frontier + // table's own capacity cap. + // + // Returns false if reference chain tracking is disabled, jvmti is null, or + // the frontier table was never constructed (start() never ran with the + // flag enabled). A pass that hits its budget/hop/frontier cap is still a + // *successful* call (returns true) - *out_truncated (if non-null) reports + // whether *this pass* ran to full exhaustion of the currently-known + // reachable graph or was cut short, per the design doc's "no silent + // truncation" requirement; this is call-scoped, unlike searchState() + // below which reports the whole search's outcome. + // + // Once searchState() is no longer RUNNING (the reachable graph was fully + // explored within caps, or the search was abandoned - see the Termination + // section implemented below), further calls are no-ops that return true + // immediately, *unless* shouldRunPass() has already called restartSearch() + // to begin a fresh search (this class's own header comment) - in that case + // _search_started is false again and this method takes the first-pass + // branch exactly as it would for a brand-new tracker. + bool runPass(jvmtiEnv *jvmti, JNIEnv *jni, bool *out_truncated = nullptr); + + // Serialized entry points for the two engine drivers: the real BFS thread + // (threadLoop(), below) and the debug seams (javaApi.cpp's + // runReferenceChainPass0()/pollReferenceChainTargets0()). The engine's + // non-frontier maps (_class_tags, _candidate_*, _leak_parent_fanout, ...) + // are plain containers with no cross-thread locking, so a seam-driven + // pass on a test thread while threadLoop() is mid-pass is a genuine data + // race - observed: SIGSEGV in ClassTagTable::insert's unordered_map + // rehash from a test thread inside resolveLoadedClasses() while the BFS + // thread was mid-pass of its own. Taking _engine_lock at both entry + // points makes the two drivers mutually exclusive while either can run. + bool runPassSerialized(jvmtiEnv *jvmti, JNIEnv *jni) { + MutexLocker engine_guard(_engine_lock); + return runPass(jvmti, jni); + } + + void pollWatchedTargetsSerialized(jvmtiEnv *jvmti, JNIEnv *jni) { + MutexLocker engine_guard(_engine_lock); + pollWatchedTargets(jvmti, jni); + } + + // Search-level outcome (SearchState's constants) - see runPass()'s comment + // for exactly when this leaves RUNNING. Acquire-loaded, pairing with + // runPass()'s release store of this same field (referenceChains.cpp), so a + // caller that observes a non-RUNNING value here also sees every detail + // field (_abandon_reason, _passes_run, ...) runPass() wrote before that + // release store. + u8 searchState() { return loadAcquire(_search_state); } + + // Total passes run for the current/most recent search. Exposed for tests + // to confirm multi-pass resumption actually happened. + int passesRun() { return load(_passes_run); } + + // Which SearchAbandonReason cutoff moved the search out of RUNNING, or + // SearchAbandonReason::NONE if it never left RUNNING or left via + // SearchState::COMPLETED instead. + u8 abandonReason() { return load(_abandon_reason); } + + // Reference-chain JFR event surface: fills *out from frontierTable()-> + // reconstructChain(target_tag, ...) (see that method's own comment for + // the leaf-to-root ordering and the parent_tag walk it performs). Returns + // false (leaving *out untouched) if target_tag was never inserted into + // the frontier table - the same failure case reconstructChain() itself + // reports, just wrapped into the JFR-event shape + // Recording::recordReferenceChain() (flightRecorder.cpp) expects. + // + // Deliberately does not decide *when* to call this or *which* target_tag + // to use - this codebase has no target-sample feed into + // ReferenceChainTracker yet (see runPass()'s own comment), so wiring an + // automatic call site here would have to invent + // that feed rather than reuse one. A future consumer that knows which + // tag it is chasing (e.g. an ObjectSampler-driven target) calls this + // directly once that feed exists. + // Resolves one chain hop's retention-edge label into `out` (NUL- + // terminated, at most out_cap bytes incl. the NUL). For a FIELD/STATIC_FIELD + // edge with a resolvable referrer class, the label is the field's NAME, + // decoded per the JVMTI specification's field-ordinal scheme (see + // FrontierEntry::referrer_field_index's own comment - the full scheme, with + // the interface-branch and class-branch rules of jvmtiHeapReferenceInfoField). + // Any other edge kind, or any decode failure (class gone, ordinal out of + // the computed range, JVMTI call failure), degrades to the edge KIND label + // ("element", "constant_pool", ...) - never a fabricated name: a wrong + // numbering on an unverified JVM would silently degrade to kind labels, + // not lie. HotSpot's numbering is source-verified (jvmtiTagMap.cpp + // ClassFieldMap::create_map_of_static_fields/instance_fields build exactly + // the spec's ordinal space); J9's compliance is INFERRED from the spec + // definition only (its walker arithmetic not yet source-verified - see + // find-field-name-decoding node). + // Legal only OUTSIDE heap callbacks (FindClass/GetClassFields/GetFieldName + // are not callable during heap iteration) - buildChainEvent() calls it on + // the BFS thread between walks. + void resolveHopEdgeLabel(jvmtiEnv *jvmti, JNIEnv *jni, ChainHopEdge edge, + char *out, size_t out_cap); + + // Fills *out with one label per chain hop, aligned with the chain's + // leaf-to-root order (edges[i] = the retention edge INTO chain[i]), via + // resolveHopEdgeLabel() above. Truncates labels to + // MAX_REFERENCE_CHAIN_EDGE_LABEL (event.h) bytes. Safe with null jvmti/jni + // (every label degrades to the edge kind) and against partial JVMTI + // function tables (gtest mock environments - the required slots are + // null-checked before use, same defensive rule the JFR-roundtrip crash + // taught for RCT::start()). + static constexpr size_t MAX_HOP_EDGE_LABEL = + MAX_REFERENCE_CHAIN_EDGE_LABEL; + // Per-referrer-class ordinal->name list cache behind + // resolveHopEdgeLabel(): decoding the spec ordinal requires walking the + // class's whole interface closure + superclass chain (GetClassFields + + // GetFieldName per field), and chains re-emit on every dump, so the decoded + // ordinal space of each chain-relevant class is built once here. Keyed by + // the referrer's raw class tag (stable per loaded class). Bounded by + // HOP_LABEL_CLASS_CACHE_CAP (cleared whole on search restart - the frontier + // and its class tags do not survive a restart, so nothing here may). + static constexpr size_t HOP_LABEL_CLASS_CACHE_CAP = 1024; + struct HopLabelClass { + jlong class_tag; + // One entry per ordinal in the class's flattened field space - the + // i-th element is the name of ordinal i. Empty when decoding failed + // (all hops through this class degrade to kind labels). + std::vector field_names; + bool decode_failed; + }; + std::unordered_map _hop_label_cache; + + // Cache lookup/decode behind resolveHopEdgeLabel() - see HopLabelClass's + // own comment. Never returns null (a failed decode is cached as + // decode_failed and re-reported as kind labels). + const HopLabelClass *hopLabelClassFor(jvmtiEnv *jvmti, JNIEnv *jni, + jlong class_tag); + + void fillHopEdgeLabels(jvmtiEnv *jvmti, JNIEnv *jni, + const std::vector &edges, + std::vector *out); + + bool buildChainEvent(jvmtiEnv *jvmti, JNIEnv *jni, jlong target_tag, + ReferenceChainEvent *out); + + // Canary-search chain reconstruction: builds the chain for a canary + // candidate from the per-candidate chain link recorded at + // pruning time (_candidate_parent_tags[] etc.), walking + // parent_tag through the frontier table (positive tags, + // so lookup() works). The candidate's own + // referrer_klass is prepended to the chain. + bool buildCanaryChainEvent(int candidate_idx, ReferenceChainEvent *out); + + // Abandoned-search JFR event surface for the design doc's "explicit reporting of + // abandoned searches" requirement - unlike buildChainEvent() above, this + // needs no target_tag: it reports the search's own termination state, + // which runPass() (referenceChains.cpp) already tracks unconditionally. + // Returns false (leaving *out untouched) if the search was never + // abandoned (searchState() != SearchState::ABANDONED). Called from + // Profiler::dump() (profiler.cpp), mirroring LivenessTracker::flush()'s + // own call site there, whenever a dump is requested while the search is + // ABANDONED - unlike LivenessTracker's table this does not clear any + // state, so a dump taken after the search already abandoned reports the + // same event again; this is a read of current state, not a queue drain. + bool buildAbandonedEvent(ReferenceChainAbandonedEvent *out) { + // Acquire-load, not a plain relaxed load - see searchState()'s own + // comment for why: this is the same guard-then-read-details pattern. + if (out == nullptr || loadAcquire(_search_state) != SearchState::ABANDONED) { + return false; + } + out->_reason = load(_abandon_reason); + out->_passes_run = (u32)load(_passes_run); + out->_frontier_size = _frontier != nullptr ? (u32)_frontier->size() : 0; + out->_hop_cap = _hop_cap; + out->_budget = _budget; + out->_ttl_ms = _ttl_ms; + out->_elapsed_ns = load(_last_pass_ns) - load(_search_start_ns); + return true; + } + + // Target-selection bridging step (design doc's Open Question 3, corrected + // mechanism - see this class's own header comment's bridging-step note and + // doc/architecture/LiveHeapReferenceChains-RemainingWorkPlan.md's + // "Correction to the design doc's Open Question 3 mechanism"): polls + // LivenessTracker::selectLeakCandidates() and, for each candidate whose + // representative instance has already been discovered by an ordinary + // runPass() walk (getTag() > 0 - a read, never a SetTag seed), reconstructs + // its datadog.ReferenceChain and caches it in _resolved_chains keyed by + // klass_id - see that field's own comment for why a resolved chain is + // cached (and re-emitted on every dump) rather than emitted once. The write + // itself is still deferred to drainPendingChainEvents() on the dump() + // thread, since Profiler::writeReferenceChain() can block this method's + // caller (the BFS scheduling thread) for up to ~50ms per event. A candidate + // still at tag 0 (not yet discovered) is left for a later poll to retry, + // since runPass()'s whole-graph walk eventually visits every root-reachable + // object, barring the hop/budget/frontier caps. A klass already cached from + // the current search generation is not reconstructed again; a restart + // (new _search_start_ns) or a re-tag makes the next poll refresh it. Every + // cached entry whose representative no longer resolves (collected/evicted) + // is pruned here, so the cache tracks the set of still-live flagged samples. + // Called from threadLoop() once per scheduling cycle, after runPass(), so + // this poll always sees the most recent pass's tagging. No-op if + // disabled, or if jvmti/jni is null (mirrors runPass()'s own null-safety, + // so a test can call this directly without a live JVM attached, the same + // way referenceChains_ut.cpp already does for runPass()). + void pollWatchedTargets(jvmtiEnv *jvmti, JNIEnv *jni); + + // Targeted holder re-walk: enqueues `tag`'s chain-root entry (the + // root-attached ancestor of its frontier chain) onto _priority_expand so + // the next rotation/expand pass re-walks the holder that retains + // everything below `tag`. Rationale (observed live in the correlation + // scenario): a container that replaces its internals (growing ArrayList, + // resized HashMap) never appears in the fanout as the direct parent of + // anything watched - the watched instances' direct parents are the DEAD + // old internals - and the blind lap over a large frontier is far too slow + // to reach the holder in any realistic window, so new internals are never + // admitted and tagged leak instances below them are never intercepted. + // The holder chain's root, however, is exactly what a candidate's chain + // reconstruction already walks; requeueing it per poll (bounded by + // MAX_LEAK_CANDIDATES pushes, de-duplicated by isQueuedForRotation) makes + // the holder's CURRENT children - including each new backing array - + // admitted promptly. See _leak_parent_fanout's own comment for the + // complementary (probabilistic) ancestor coverage. + void requeueChainRootForRotation(jlong tag); + + // Appends a copy of every currently-cached resolved chain to *out, + // re-stamped with a fresh _start_time so it lands in the dumping chunk's + // time window, WITHOUT clearing the cache - a repeatable snapshot, not a + // drain, so the same live sample's chain is re-emitted into every JFR chunk + // it survives into (see _resolved_chains' own comment). Called from + // Profiler::dump() (profiler.cpp), which then calls + // Profiler::writeReferenceChain() for each event on its own thread - never + // the BFS scheduling thread. A no-op (leaves *out untouched) if the cache + // is currently empty. The name is retained from the drain-once era for its + // stable call site; the semantics are now snapshot-and-keep. + void drainPendingChainEvents(std::vector *out); + + // Appends every abandoned-search event queued since the last call and + // clears the queue - a true drain, unlike drainPendingChainEvents() above: + // an abandoned search is a discrete past occurrence, not an ongoing live + // sample, so there is nothing left to re-report once Profiler::dump() + // (profiler.cpp) has emitted it. Exists because searchState()/ + // buildAbandonedEvent() alone cannot be read reliably from dump()'s thread + // (see _pending_abandoned_events' own comment): each event here was + // snapshotted synchronously, on the BFS thread, at the exact moment the + // search abandoned - before shouldRunPass() gets a chance to call + // restartSearch() and clear the live fields buildAbandonedEvent() would + // otherwise have read. + void drainPendingAbandonedEvents(std::vector *out); + + static void JNICALL GarbageCollectionStart(jvmtiEnv *jvmti_env); + static void JNICALL GarbageCollectionFinish(jvmtiEnv *jvmti_env); + + // Test seam - not part of the production API. Mirrors LivenessTracker's + // own "Test seams" block (livenessTracker.h). Production code only ever + // discovers frontier roots via runPass()'s root-seeded FollowReferences + // walk; this lets a test tag and insert one specific, caller-chosen live + // object as a frontier root directly, so runPass()/pollWatchedTargets()/ + // buildChainEvent() can be exercised end-to-end against a known target + // without depending on LivenessTracker's probabilistic allocation sampler + // to organically select and surface the same object. Returns the assigned + // tag (matching the value buildChainEvent()'s target_tag expects), or 0 on + // failure (obj/jvmti/jni null, SetTag failed, or the frontier table is at + // capacity). + jlong tagAsRootForTest(jvmtiEnv *jvmti, JNIEnv *jni, jobject obj); + + // Test seam - not part of the production API. Since ReferenceChainTracker + // is a process-wide singleton (ExternalProcessReferenceChainTest's own + // class javadoc explains why that matters: only the *first* test to ever + // call runPass() in a shared JVM gets a real root-seeded walk, since + // runPass() only re-walks from the roots once per search's whole + // lifetime), an in-process test that needs its own genuine first-ever + // root walk calls this at the start of its test body to force exactly + // that - releasing any tags a previous test's search still held, then + // resetting search/frontier state to the same "brand-new tracker" state + // restartSearch() (referenceChains.cpp) produces, plus the target- + // dedup/pending-event state restartSearch() itself intentionally leaves + // for pollWatchedTargets()/drainPendingChainEvents() to self-clear (this + // is an immediate, out-of-band reset - there is no next real pass here to + // observe the change and clear them the ordinary way). + void resetSearchStateForTest(jvmtiEnv *jvmti, JNIEnv *jni); + + // Test seam - not part of the production API. Diagnostic-only: reports how + // far a given (already-tagged) object sits from the front of + // _pending_expand's FIFO queue, to distinguish "not yet expanded because + // its own FIFO position hasn't come up yet" from "already expanded" or + // "never admitted at all" without needing a debugger. Returns >=0 (the + // 0-based distance from the front - 0 means it expands next) if tag is + // still queued, -1 if tag is nonzero but not currently queued (already + // expanded, or never admitted), or -2 if tag itself is 0. + long pendingExpandPositionForTest(jlong tag) const; + + // Test seam - not part of the production API. Companion to + // pendingExpandPositionForTest() above, for computing a position's + // fraction of the current backlog. + size_t pendingExpandSizeForTest() const; + + // Test seam - not part of the production API. Exposes the private + // shouldRunPass() gate directly, so a test can assert whether a + // fresh/terminal search would be allowed to start right now - in + // particular, whether LivenessTracker::secondsToOOM()'s urgent-OOM bypass + // (hasLeakSignal(), see OOM_URGENT_THRESHOLD_S's own comment above) opens + // this gate even with zero per-klass leak candidate (confirmable in the + // same test via LivenessTracker::selectLeakCandidates()/JavaProfiler's + // selectLeakCandidateKlassIds0() seam) - something runReferenceChainPass0() + // (javaApi.cpp) cannot show, since it calls runPass() directly and never + // consults this gate at all. + bool shouldRunPassForTest(u64 now_ns) { return shouldRunPass(now_ns); } +}; + +#endif // _REFERENCECHAINS_H diff --git a/ddprof-lib/src/main/cpp/safeAccess.h b/ddprof-lib/src/main/cpp/safeAccess.h index 564743b153..0ab3ed380c 100644 --- a/ddprof-lib/src/main/cpp/safeAccess.h +++ b/ddprof-lib/src/main/cpp/safeAccess.h @@ -1,6 +1,6 @@ /* * Copyright 2021 Andrei Pangin -* Copyright 2026 Datadog, Inc + * Copyright 2026 Datadog, Inc * * Licensed under the Apache License, Version 2.0 (the "License"); * you may not use this file except in compliance with the License. diff --git a/ddprof-lib/src/main/cpp/stringDictionary.h b/ddprof-lib/src/main/cpp/stringDictionary.h index b5572b6237..f16f802a72 100644 --- a/ddprof-lib/src/main/cpp/stringDictionary.h +++ b/ddprof-lib/src/main/cpp/stringDictionary.h @@ -459,6 +459,11 @@ class StringDictionaryBuffer { class StringDictionary { std::atomic _next_id{1}; // starts at 1; id=0 reserved as "no entry" std::atomic _accepting{true}; // false while clearAll() is resetting buffers + // Bumped by clearAll() only. Lets a cache keyed by ids from this + // dictionary (e.g. ReferenceChainTracker::_class_tags, referenceChains.h) + // detect "the id namespace was wiped out from under me" and invalidate + // itself, rather than assuming ids stay valid across a clearAll(). + std::atomic _generation{0}; StringDictionaryBuffer _a, _b, _c; TripleBufferRotator _rot; int _counter_offset; // offset into DICTIONARY_KEYS / DICTIONARY_KEYS_BYTES counter rows @@ -489,6 +494,9 @@ class StringDictionary { } } + // Current id-namespace generation; see _generation's own comment. + u64 generation() const { return _generation.load(std::memory_order_acquire); } + // Insert into active buffer; returns globally stable id. NOT signal-safe. u32 lookup(const char* key, size_t len) { if (!_accepting.load(std::memory_order_acquire)) return 0; @@ -635,6 +643,7 @@ class StringDictionary { _next_id.store(1, std::memory_order_relaxed); Counters::set(DICTIONARY_KEYS, 0, _counter_offset); Counters::set(DICTIONARY_KEYS_BYTES, 0, _counter_offset); + _generation.fetch_add(1, std::memory_order_release); _accepting.store(true, std::memory_order_release); } }; diff --git a/ddprof-lib/src/main/cpp/symbols_linux.cpp b/ddprof-lib/src/main/cpp/symbols_linux.cpp index b328fcfd56..0bb8d0c11e 100644 --- a/ddprof-lib/src/main/cpp/symbols_linux.cpp +++ b/ddprof-lib/src/main/cpp/symbols_linux.cpp @@ -575,7 +575,15 @@ void ElfParser::calcVirtualLoadAddress() { for (int i = 0; i < _header->e_phnum; i++) { ElfProgramHeader* pheader = phdrAt(i); if (pheader != NULL && pheader->p_type == PT_LOAD) { - _vaddr_diff = _base - pheader->p_vaddr; + // p_vaddr is an unrelated virtual address, not an offset within the + // _base allocation - subtracting it via pointer arithmetic can wrap + // to (or through) a null representation, which UBSan flags even + // though the resulting bit pattern is only ever used as an offset + // to add back later (at()/base()/dyn_ptr() above). Do the + // subtraction in integer space and reinterpret, matching this + // file's existing "validate in integer space before forming a + // pointer" pattern (see phdrAt() above). + _vaddr_diff = (const char*)((uintptr_t)_base - (uintptr_t)pheader->p_vaddr); return; } } @@ -598,7 +606,9 @@ void ElfParser::parseDynamicSection() { uint32_t nsyms = 0; const char* dyn_start = at(dynamic); - const char* dyn_end = dyn_start + dynamic->p_memsz; + // at(dynamic) is NULL when dynamic->p_vaddr == 0 - same null-base + // pointer-arithmetic UB as the other fixes in this file. + const char* dyn_end = (const char*)((uintptr_t)dyn_start + dynamic->p_memsz); for (ElfDyn* dyn = (ElfDyn*)dyn_start; dyn < (ElfDyn*)dyn_end; dyn++) { switch (dyn->d_tag) { case DT_SYMTAB: @@ -665,7 +675,11 @@ void ElfParser::parseDynamicSection() { loadSymbolTable(symtab, syment * nsyms, syment, strtab, strsz); } - const char* base = this->base(); + // base() is NULL for ET_EXEC (non-PIE) images - adding r->r_offset to it + // via pointer arithmetic is UB (base + r->r_offset on a null base), even + // though the intent is just "sym addresses are already absolute". Do the + // addition in integer space, same fix as the .plt case above. + uintptr_t base_addr = (uintptr_t)this->base(); if (jmprel != NULL && pltrelsz != 0) { // Parse .rela.plt table for (size_t offs = 0; offs < pltrelsz; offs += relent) { @@ -674,7 +688,7 @@ void ElfParser::parseDynamicSection() { if (sym->st_name != 0) { const char* sym_name = strAt(strtab, strsz, sym->st_name); if (sym_name != NULL) { - _cc->addImport((void**)(base + r->r_offset), sym_name); + _cc->addImport((void**)(base_addr + r->r_offset), sym_name); } } } @@ -691,7 +705,7 @@ void ElfParser::parseDynamicSection() { if (sym->st_name != 0) { const char* sym_name = strAt(strtab, strsz, sym->st_name); if (sym_name != NULL) { - _cc->addImport((void**)(base + r->r_offset), sym_name); + _cc->addImport((void**)(base_addr + r->r_offset), sym_name); } } } @@ -736,7 +750,10 @@ void ElfParser::parseDwarfInfo() { for (int i = 0; i < _header->e_phnum; i++) { ElfProgramHeader* ph = phdrAt(i); if (ph != NULL && ph->p_type == PT_LOAD) { - const char* seg_end = at(ph) + ph->p_memsz; + // at(ph) is NULL when ph->p_vaddr == 0 (a real, if rare, case for + // the first LOAD segment of some binaries) - same null-base + // pointer-arithmetic UB as the other fixes in this file. + const char* seg_end = (const char*)((uintptr_t)at(ph) + ph->p_memsz); if (seg_end > image_end) image_end = seg_end; } } @@ -792,7 +809,12 @@ void ElfParser::loadSymbols(bool use_debug) { _cc->setPlt(plt->sh_addr, plt->sh_size); ElfSection* reltab = findSection(SHT_RELA, ".rela.plt"); if (reltab != NULL || (reltab = findSection(SHT_REL, ".rel.plt")) != NULL) { - addRelocationSymbols(reltab, base() + plt->sh_addr + PLT_HEADER_SIZE); + // base() is NULL for ET_EXEC (non-PIE) images - adding a non-zero + // offset to it via pointer arithmetic is UB even though the intent + // is just "no adjustment needed, sh_addr is already absolute". + // Compute in integer space and cast once, same fix as + // calcVirtualLoadAddress()'s _vaddr_diff computation above. + addRelocationSymbols(reltab, (const char*)((uintptr_t)base() + (uintptr_t)plt->sh_addr + PLT_HEADER_SIZE)); } } } diff --git a/ddprof-lib/src/main/cpp/vmEntry.cpp b/ddprof-lib/src/main/cpp/vmEntry.cpp index ef1561a8d3..87979bb975 100644 --- a/ddprof-lib/src/main/cpp/vmEntry.cpp +++ b/ddprof-lib/src/main/cpp/vmEntry.cpp @@ -17,6 +17,7 @@ #include "log.h" #include "os.h" #include "profiler.h" +#include "referenceChains.h" #include "safeAccess.h" #include "threadLocalData.h" // Pulls in vmStructs.h plus the definitions of crashProtectionActive()/cast_to() that its inline @@ -393,6 +394,15 @@ bool VM::initLibrary(JavaVM *vm) { return true; } +// jvmtiEventCallbacks has a single function-pointer slot per event; both +// LivenessTracker and ReferenceChainTracker need GarbageCollectionFinish +// (PROF-15341), so this trampoline dispatches to both instead of one +// subsystem's registration clobbering the other's. +static void JNICALL onGarbageCollectionFinish(jvmtiEnv *jvmti_env) { + LivenessTracker::GarbageCollectionFinish(jvmti_env); + ReferenceChainTracker::GarbageCollectionFinish(jvmti_env); +} + void VM::probeJFRRequestStackTrace() { jint ext_count = 0; jvmtiExtensionFunctionInfo *ext_functions = nullptr; @@ -517,7 +527,8 @@ bool VM::initProfilerBridge(JavaVM *vm, bool attach) { callbacks.ThreadStart = Profiler::ThreadStart; callbacks.ThreadEnd = Profiler::ThreadEnd; callbacks.SampledObjectAlloc = ObjectSampler::SampledObjectAlloc; - callbacks.GarbageCollectionFinish = LivenessTracker::GarbageCollectionFinish; + callbacks.GarbageCollectionStart = ReferenceChainTracker::GarbageCollectionStart; + callbacks.GarbageCollectionFinish = onGarbageCollectionFinish; callbacks.NativeMethodBind = VMStructs::NativeMethodBind; _jvmti->SetEventCallbacks(&callbacks, sizeof(callbacks)); diff --git a/ddprof-lib/src/main/java/com/datadoghq/profiler/JavaProfiler.java b/ddprof-lib/src/main/java/com/datadoghq/profiler/JavaProfiler.java index c65d80b00d..6f5bd6374b 100644 --- a/ddprof-lib/src/main/java/com/datadoghq/profiler/JavaProfiler.java +++ b/ddprof-lib/src/main/java/com/datadoghq/profiler/JavaProfiler.java @@ -462,6 +462,19 @@ public Map getDebugCounters() { private static native int getTid0(); + /** + * Test seam (debug native builds only): returns the calling thread's profiler tid + * (ProfiledThread::currentTid()'s value on the native side - the same tid space + * {@code seedTidTrendSample0} matches tracked instances' allocating threads + * against). Scenarios seeding per-(klass, tid) trend ramps must call this ON + * the leaking thread and pass the result to {@code seedTidTrendSample0}, or + * the qualifying tid will match no tracked instance and no leak tag will + * ever be assigned to the scenario's real instances. + */ + public static int getTid() { + return getTid0(); + } + private static native boolean recordTrace0(long rootSpanId, String endpoint, String operation, int sizeLimit); private static native void dump0(String recordingFilePath); @@ -530,6 +543,181 @@ private static native void setTraceContext0(long localRootSpanId, long spanId, l */ public static native boolean testTlsPrimingAvailable(); + /** + * Test seam (debug native builds only - a no-op returning {@code false}/{@code 0}/an + * empty array in release builds): decouples LivenessTracker's leak-candidate + * detection from ReferenceChainTracker's chain reconstruction, each independently + * verifiable end-to-end without depending on both the probabilistic JVMTI heap + * sampler and the reference-chain BFS search organically producing the right + * conditions in the same test run. + *

+ * Enables/disables LivenessTracker's per-klass population tracking directly, + * bypassing {@code initialize()}'s live-JVM requirement. Returns {@code true} on + * debug builds. + */ + public static native boolean setGcGenerationsEnabled0(boolean enabled); + + /** + * Test seam (debug native builds only): seeds one epoch's worth of population + * history for {@code klassId} directly into LivenessTracker's ring buffer, + * bypassing real allocation sampling. Repeated calls (with distinct + * {@code epoch} values) build up a trend {@link #selectLeakCandidateKlassIds0()} + * can then rank, letting a test assert a slope signal would be generated for a + * chosen klass id without waiting on real GC epochs. + */ + public static native void seedKlassPopulationSample0(int klassId, int count, long epoch); + + /** + * Test seam (debug native builds only): seeds one epoch's worth of per-tid trend + * history for {@code klassId}, the (klass, tid) qualification + * selectLeakCandidates() now requires on top of the klass-level ramp seeded by + * {@link #seedKlassPopulationSample0}. Repeated calls with the same rising shape + * (and the same {@code tid}) build the sustained rise that qualifies {@code tid} + * as a leak-site thread. {@code tid} must be the leaking thread's real profiler + * tid from {@link #getTid()} whenever the scenario also relies on leak tagging + * of real tracked instances; see that method's own comment. + */ + public static native void seedTidTrendSample0(int klassId, int tid, int count, long epoch); + + /** + * Test seam (debug native builds only): wires {@code representative} in as {@code klassId}'s + * leak-candidate representative directly (a fresh weak global ref owned by LivenessTracker), + * bypassing the real allocation-sampling path that would otherwise populate this. Combined + * with {@link #seedKlassPopulationSample0} and {@link #tagAsReferenceChainRoot0}, lets a test + * join a synthetic slope signal to a real, directly-tagged object so + * {@link #pollReferenceChainTargets0()}'s bridging step can be exercised end-to-end with + * neither the real sampler nor the real root-seeded walk involved. + */ + public static native void setKlassPopulationRepresentativeForTest0(int klassId, Object representative); + + /** + * Test seam (debug native builds only): clears LivenessTracker's per-klass population table, + * so a later test in the same JVM does not observe leak candidates seeded by an earlier one. + */ + public static native void resetKlassPopulationForTest0(); + + /** + * Test seam (debug native builds only): returns the klass ids LivenessTracker's + * real leak-candidate ranking (positive population slope, top 5) currently + * selects - the same call ReferenceChainTracker's restart gate and target-polling + * bridge use in production, exposed here so a test can assert a slope signal was + * generated (real or seeded via {@link #seedKlassPopulationSample0}) without + * needing a reference-chain search to also be running. + */ + public static native int[] selectLeakCandidateKlassIds0(); + + /** + * Test seam (debug native builds only): tags {@code target} and inserts it + * directly as a reference-chain frontier root, bypassing ReferenceChainTracker's + * normal discovery path (a root-seeded FollowReferences walk) and + * LivenessTracker's leak-candidate selection entirely. Lets a test drive + * {@link #runReferenceChainPass0()}/{@link #pollReferenceChainTargets0()} against + * a known, caller-chosen live object. Returns the assigned frontier tag (matching + * the {@code target_tag} a resulting {@code datadog.ReferenceChain} event + * reports), or {@code 0} on failure (reference chains disabled, or the frontier + * table is at capacity). + */ + public static native long tagAsReferenceChainRoot0(Object target); + + /** + * Test seam (debug native builds only): runs exactly one bounded BFS pass of the + * reference-chain search synchronously, rather than waiting on the tracker's own + * background thread/cadence. Returns {@code false} if reference chains are + * disabled or the tracker was never started. + */ + public static native boolean runReferenceChainPass0(); + + /** + * Test seam (debug native builds only): runs one poll of + * ReferenceChainTracker's LivenessTracker-to-chain-reconstruction bridging step + * synchronously - for each current leak candidate already discovered by a prior + * {@link #runReferenceChainPass0()} walk, reconstructs and queues its chain + * event, rather than waiting on the background thread's own scheduling cycle. + */ + public static native void pollReferenceChainTargets0(); + + /** + * Test seam (debug native builds only): drains and returns the number of + * reference-chain events queued by {@link #pollReferenceChainTargets0()} so far + * (the same queue {@code Profiler.dump()} drains in production to write + * {@code datadog.ReferenceChain} JFR events) - lets a test assert a chain was + * actually reconstructed without needing a real JFR dump. + */ + public static native int drainReferenceChainEventCount0(); + + /** + * Test seam (debug native builds only): resets ReferenceChainTracker's search/frontier state + * back to a brand-new tracker's, releasing any tags a previous search still held. Since the + * tracker is a process-wide singleton, an in-process test that needs its own genuine first + * root-seeded walk (runPass() only re-walks from the roots once per search's whole lifetime) + * calls this at the start of its test body to force one, rather than depending on being the + * first reference-chain test to run in a shared test JVM. + */ + public static native void resetReferenceChainSearchForTest0(); + + /** + * Test seam (debug native builds only): diagnostic-only, does not tag {@code target}. Reads + * target's existing JVMTI tag (0 if the real search has never admitted it) and reports its + * FIFO distance from the front of ReferenceChainTracker's pending-expansion queue: {@code >=0} + * (0 = expands next) if still queued, {@code -1} if tagged but no longer queued (already + * expanded), or {@code -2} if never admitted at all. + */ + public static native long getReferenceChainPendingPositionForTest0(Object target); + + /** + * Test seam (debug native builds only): the current size of ReferenceChainTracker's + * pending-expansion queue, for computing {@link #getReferenceChainPendingPositionForTest0}'s + * position as a fraction of the current backlog. + */ + public static native long getReferenceChainPendingSizeForTest0(); + + /** + * Test seam (debug native builds only): seeds one heap-floor-ring sample - the input to + * {@code LivenessTracker::secondsToOOM()}'s time-to-OOM projection - directly, bypassing the + * real {@code GarbageCollectionFinish} callback. {@code timestampNs} values are only ever + * compared against each other, never against a real wall clock, so a test may use any + * self-consistent, strictly increasing sequence to build an arbitrary rising or flat + * heap-usage-over-time history without waiting on real GCs. + */ + public static native void heapFloorRecordForTest0(long usedBytes, long timestampNs); + + /** + * Test seam (debug native builds only): overrides the max-heap-size {@code secondsToOOM()} + * projects against, bypassing the real {@code Runtime.maxMemory()} resolution - so a test can + * exercise the projection deterministically, independent of whatever {@code -Xmx} this JVM's + * own shared, no-{@code forkEvery} fork happens to run with. + */ + public static native void setMaxHeapBytesForTest0(long maxHeapBytes); + + /** + * Test seam (debug native builds only): temporarily disables {@code onGC()}'s own + * {@code recordHeapFloorSample()} call so a test can seed the heap-floor ring exclusively + * via {@link #heapFloorRecordForTest0(long, long)} without a real GC interleaving a sample + * with a real {@code OS::nanotime()} timestamp and real heap usage, corrupting + * {@code secondsToOOM()}'s projection. Pass {@code false} to disable, {@code true} to restore. + */ + public static native void setHeapFloorRecordingForTest0(boolean enabled); + + /** + * Test seam (debug native builds only): reports whether ReferenceChainTracker's search-restart + * gate ({@code canAffordNewSearch()} -> {@code hasLeakSignal()}) would currently allow a + * fresh/terminal search to start - in particular, whether {@code secondsToOOM()}'s urgent-OOM + * bypass opens this gate even with zero per-klass leak candidate (confirmable in the same test + * via {@link #selectLeakCandidateKlassIds0()}). Unlike {@link #runReferenceChainPass0()}, which + * calls {@code runPass()} unconditionally, this reads the gate itself without running a pass. + */ + public static native boolean shouldRunPassForTest0(); + + /** + * Test seam (debug native builds only): the number of BFS passes run for the current/most + * recent reference-chain search. Lets a test note this count before creating an object, then + * wait for it to advance before trusting a match against that object - the only way to be + * certain the match came from a pass whose own {@code expandFrontier()} (and therefore {@code + * collectStaleExpandedEntriesForRotation()}) ran strictly after the object existed, rather + * than from the same pass racing the object's creation. + */ + public static native int referenceChainPassesRunForTest0(); + // ---- Test-only reads of the current thread's OTEP record ---------------------------------- // Each resolves the current carrier's record directly (like the write primitives above) with // no cached buffer and no per-thread Java object; introspection/test use only. From 4f8c7e5b46b744f10726362c6271c95e99fe9a80 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Wed, 16 Sep 2026 17:01:43 +0200 Subject: [PATCH 02/18] Add C++ unit tests for reference-chain tracking Unit tests for the reference-chain engine: anchor selection and admission, walk tiers and quotas, self-edge guards, leak-tag correlation, JFR roundtrip of ReferenceChain events, and debug seams for the OOM-bypass gate (arguments, jvmHeap, stale-leaf coverage). --- ddprof-lib/src/test/cpp/arguments_ut.cpp | 51 + .../src/test/cpp/lineNumberTableCopy_ut.cpp | 33 + .../src/test/cpp/livenessTracker_ut.cpp | 993 +++ .../cpp/referenceChainJfrRoundtrip_ut.cpp | 419 + .../src/test/cpp/referenceChains_ut.cpp | 7255 +++++++++++++++++ ddprof-lib/src/test/cpp/staleLeaf_ut.cpp | 129 + 6 files changed, 8880 insertions(+) create mode 100644 ddprof-lib/src/test/cpp/arguments_ut.cpp create mode 100644 ddprof-lib/src/test/cpp/referenceChainJfrRoundtrip_ut.cpp create mode 100644 ddprof-lib/src/test/cpp/referenceChains_ut.cpp create mode 100644 ddprof-lib/src/test/cpp/staleLeaf_ut.cpp diff --git a/ddprof-lib/src/test/cpp/arguments_ut.cpp b/ddprof-lib/src/test/cpp/arguments_ut.cpp new file mode 100644 index 0000000000..74bca2ce1b --- /dev/null +++ b/ddprof-lib/src/test/cpp/arguments_ut.cpp @@ -0,0 +1,51 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +#include +#include +#include "arguments.h" +#include "../../main/cpp/gtest_crash_handler.h" + +static constexpr char ARGUMENTS_TEST_NAME[] = "ArgumentsTest"; + +class ArgumentsGlobalSetup { +public: + ArgumentsGlobalSetup() { + installGtestCrashHandler(); + } + ~ArgumentsGlobalSetup() { + restoreDefaultSignalHandlers(); + } +}; + +static ArgumentsGlobalSetup global_setup; + +class ArgumentsTest : public ::testing::Test { +protected: + void SetUp() override {} + void TearDown() override {} +}; + +// hops/budget/framecap are ceiling-clamped (MAX_REFERENCE_CHAINS_HOP_CAP/ +// _BUDGET/_FRONTIER_CAP, arguments.h) as well as floored at 1 - an operator +// typo (an extra digit) must not flow straight into a loop bound or +// FrontierTable's allocation unchecked. +TEST_F(ArgumentsTest, HopsBudgetFrameCapAreCeilingClamped) { + Arguments args; + Error error = args.parse("referencechains=true:hops=2000000000:budget=2000000000:framecap=2000000000"); + EXPECT_FALSE(error); + EXPECT_EQ(args._reference_chains_hop_cap, MAX_REFERENCE_CHAINS_HOP_CAP); + EXPECT_EQ(args._reference_chains_budget, MAX_REFERENCE_CHAINS_BUDGET); + EXPECT_EQ(args._reference_chains_frontier_cap, MAX_REFERENCE_CHAINS_FRONTIER_CAP); +} + +TEST_F(ArgumentsTest, HopsBudgetFrameCapStillFlooredAtOne) { + Arguments args; + Error error = args.parse("referencechains=true:hops=-5:budget=-5:framecap=-5"); + EXPECT_FALSE(error); + EXPECT_EQ(args._reference_chains_hop_cap, 1); + EXPECT_EQ(args._reference_chains_budget, 1); + EXPECT_EQ(args._reference_chains_frontier_cap, 1); +} diff --git a/ddprof-lib/src/test/cpp/lineNumberTableCopy_ut.cpp b/ddprof-lib/src/test/cpp/lineNumberTableCopy_ut.cpp index 6c82a8ee96..902ca9a01a 100644 --- a/ddprof-lib/src/test/cpp/lineNumberTableCopy_ut.cpp +++ b/ddprof-lib/src/test/cpp/lineNumberTableCopy_ut.cpp @@ -255,3 +255,36 @@ TEST_F(LineNumberTableCopyTest, RejectsOversizedTable) { EXPECT_FALSE(isLineNumberTableReadable(line_number_table, kMaxLineNumberTableEntries + 1)); } + +// fillJavaMethodInfo() (flightRecorder.cpp) gates the copy above on +// `line_number_table_size > 0 && line_number_table_size <= +// MAX_LINE_NUMBER_TABLE_ENTRIES` (flightRecorder.cpp:374), where +// MAX_LINE_NUMBER_TABLE_ENTRIES is 65535 (flightRecorder.cpp:58, the u2 +// code_length cap). MAX_LINE_NUMBER_TABLE_ENTRIES is file-static, so these +// tests mirror the boundary condition with the literal value rather than +// calling fillJavaMethodInfo() directly (which requires a live JVMTI/JNI +// environment, impractical to fake in a plain gtest -- same rationale as the +// tests above). They exist to catch a "<=" -> "<" mutation at that line, +// which would incorrectly reject a spec-valid, exactly-max-sized table. +namespace { + +bool passesLineNumberTableSizeGuard(jint size) { + return size > 0 && size <= kMaxLineNumberTableEntries; +} +} // namespace + +TEST(LineNumberTableBoundsTest, AcceptsExactlyMaxEntries) { + EXPECT_TRUE(passesLineNumberTableSizeGuard(kMaxLineNumberTableEntries)); +} + +TEST(LineNumberTableBoundsTest, RejectsOneMoreThanMaxEntries) { + EXPECT_FALSE(passesLineNumberTableSizeGuard(kMaxLineNumberTableEntries + 1)); +} + +TEST(LineNumberTableBoundsTest, RejectsNegativeSize) { + EXPECT_FALSE(passesLineNumberTableSizeGuard(-1)); +} + +TEST(LineNumberTableBoundsTest, RejectsZeroSize) { + EXPECT_FALSE(passesLineNumberTableSizeGuard(0)); +} diff --git a/ddprof-lib/src/test/cpp/livenessTracker_ut.cpp b/ddprof-lib/src/test/cpp/livenessTracker_ut.cpp index e0117af6a2..174a871afc 100644 --- a/ddprof-lib/src/test/cpp/livenessTracker_ut.cpp +++ b/ddprof-lib/src/test/cpp/livenessTracker_ut.cpp @@ -15,6 +15,7 @@ */ #include +#include "livenessTracker.h" #include "../../main/cpp/gtest_crash_handler.h" #include #include @@ -289,3 +290,995 @@ TEST_F(LivenessTrackerTest, CapacityDoesNotExceedMaxCap) { // In the actual code, this would trigger: if (_table_cap != newcap) { ... } // which would be false, so no resize would be attempted } + +// --------------------------------------------------------------------------- +// Per-klass population tracking (LiveHeapReferenceChains-RemainingWorkPlan.md). +// These exercise LivenessTracker::instance() directly rather than +// a mock: recordKlassPopulationSampleLocked() deliberately makes no JNI call +// (see its header comment), so it is safe to call on the real singleton +// without a live JVM attached, unlike start()/track()/flush() elsewhere in +// this class. Fake jweak values below are opaque pointers the method under +// test never dereferences - only stored and handed back to the caller. +class KlassPopulationTest : public ::testing::Test { +protected: + void SetUp() override { + installGtestCrashHandler(); + // The table persists across recordings by design (see + // LivenessTracker::initialize()'s own comment on why _initialized + // survives multiple start() calls) - reset it explicitly here so + // tests don't observe leftover state from a previous test case + // sharing the same process-wide singleton. + LivenessTracker::instance()->klassPopulationResetForTest(); + } + + void TearDown() override { + LivenessTracker::instance()->klassPopulationResetForTest(); + restoreDefaultSignalHandlers(); + } + + static jweak fakeRef(uintptr_t tag) { + return reinterpret_cast(tag); + } +}; + +// A brand new klass_id creates a new entry: out_created is true, the table +// grows by one, and the single pushed sample is the ring's only member. +TEST_F(KlassPopulationTest, InsertCreatesNewEntry) { + LivenessTracker *tracker = LivenessTracker::instance(); + + int slot = -1; + bool created = false; + jweak evicted = tracker->klassPopulationRecordForTest(/*klass_id=*/1, + /*count=*/5, + /*epoch=*/1, + &slot, &created); + + EXPECT_TRUE(created); + EXPECT_EQ(evicted, nullptr); + EXPECT_EQ(tracker->klassPopulationSizeForTest(), 1); + + KlassPopulationEntry entry; + ASSERT_TRUE(tracker->klassPopulationLookupForTest(1, &entry)); + EXPECT_EQ(entry.klass_id, 1u); + EXPECT_EQ(entry.ring_fill, 1); + EXPECT_EQ(entry.ring_head, 1); + EXPECT_EQ(entry.count_ring[0], 5); + EXPECT_EQ(entry.last_updated_epoch, 1u); + EXPECT_EQ(entry.representative_count, 0); +} + +// A second sample for an already-known klass_id updates the same slot in +// place (out_created is false, table size unchanged) rather than creating a +// second entry. +TEST_F(KlassPopulationTest, InsertExistingUpdatesSameSlotInPlace) { + LivenessTracker *tracker = LivenessTracker::instance(); + + int slot1 = -1, slot2 = -1; + bool created1 = false, created2 = false; + tracker->klassPopulationRecordForTest(7, 3, 1, &slot1, &created1); + jweak evicted = tracker->klassPopulationRecordForTest(7, 4, 2, &slot2, + &created2); + + EXPECT_TRUE(created1); + EXPECT_FALSE(created2); + EXPECT_EQ(slot1, slot2); + EXPECT_EQ(evicted, nullptr); + EXPECT_EQ(tracker->klassPopulationSizeForTest(), 1); + + KlassPopulationEntry entry; + ASSERT_TRUE(tracker->klassPopulationLookupForTest(7, &entry)); + EXPECT_EQ(entry.ring_fill, 2); + EXPECT_EQ(entry.count_ring[0], 3); + EXPECT_EQ(entry.count_ring[1], 4); + EXPECT_EQ(entry.last_updated_epoch, 2u); +} + +// Ring buffer wraparound: pushing more than KLASS_POPULATION_RING_SIZE (30) +// samples must not grow ring_fill past 30, and the ring must overwrite the +// oldest slots in order rather than corrupting adjacent entries. +TEST_F(KlassPopulationTest, RingBufferWrapsAroundAtThirtySamples) { + LivenessTracker *tracker = LivenessTracker::instance(); + + const int RING_SIZE = 30; + for (int i = 0; i < RING_SIZE + 5; i++) { + int slot; + bool created; + tracker->klassPopulationRecordForTest(42, (u16)(i + 1), i + 1, &slot, + &created); + EXPECT_EQ(created, i == 0); + } + + KlassPopulationEntry entry; + ASSERT_TRUE(tracker->klassPopulationLookupForTest(42, &entry)); + // Still capped at 30 even though 35 samples were pushed. + EXPECT_EQ(entry.ring_fill, RING_SIZE); + // ring_head wrapped: 35 writes into a 30-slot ring lands back at index 5. + EXPECT_EQ(entry.ring_head, 5); + // 35 pushes write ring indices 0..29 with values 1..30, then wrap and + // overwrite indices 0..4 with values 31..35 - leaving indices 5..29 + // still holding values 6..30 (never overwritten) and indices 0..4 + // holding the wrapped-around values 31..35. + EXPECT_EQ(entry.count_ring[5], 6); + EXPECT_EQ(entry.count_ring[29], 30); + EXPECT_EQ(entry.count_ring[0], 31); + EXPECT_EQ(entry.count_ring[4], 35); + EXPECT_EQ(entry.last_updated_epoch, RING_SIZE + 5u); +} + +// Filling the table to MAX_KLASS_POPULATION_ENTRIES and then inserting one +// more distinct klass_id must evict the least-recently-updated entry (the +// smallest last_updated_epoch) and return its representative jweak so the +// caller can release it. +TEST_F(KlassPopulationTest, EvictsLeastRecentlyUpdatedEntryWhenFull) { + LivenessTracker *tracker = LivenessTracker::instance(); + + const int CAP = 256; // MAX_KLASS_POPULATION_ENTRIES + for (u32 klass_id = 1; klass_id <= (u32)CAP; klass_id++) { + int slot; + bool created; + // epoch == klass_id, so klass_id 1 is the least-recently-updated + // entry once the table is full. + tracker->klassPopulationRecordForTest(klass_id, 1, klass_id, &slot, + &created); + ASSERT_TRUE(created); + } + EXPECT_EQ(tracker->klassPopulationSizeForTest(), CAP); + + jweak victim_ref = fakeRef(0xdead); + tracker->klassPopulationSetRepresentativeForTest(nullptr, 1, victim_ref); + + int slot; + bool created; + // Eviction now returns evicted refs via output array, not return value. + jweak evicted[KlassPopulationEntry::MAX_REPRESENTATIVES_PER_KLASS]; + int evicted_count = 0; + tracker->klassPopulationRecordForTest( + /*klass_id=*/CAP + 1, /*count=*/1, /*epoch=*/CAP + 1, &slot, &created, + evicted, &evicted_count, KlassPopulationEntry::MAX_REPRESENTATIVES_PER_KLASS); + + EXPECT_TRUE(created); + ASSERT_EQ(evicted_count, 1); + EXPECT_EQ(evicted[0], victim_ref); + // Table stays at capacity - the evicted slot was reused, not appended. + EXPECT_EQ(tracker->klassPopulationSizeForTest(), CAP); + + KlassPopulationEntry evicted_klass_entry; + EXPECT_FALSE(tracker->klassPopulationLookupForTest(1, &evicted_klass_entry)) + << "klass_id 1 should have been fully replaced by the eviction"; + + KlassPopulationEntry new_entry; + ASSERT_TRUE(tracker->klassPopulationLookupForTest(CAP + 1, &new_entry)); + EXPECT_EQ(new_entry.representative_count, 0); + EXPECT_EQ(new_entry.ring_fill, 1); +} + +// --------------------------------------------------------------------------- +// Slope computation and candidate ranking (LiveHeapReferenceChains- +// RemainingWorkPlan.md). Same rationale as KlassPopulationTest above +// for exercising LivenessTracker::instance() directly: selectLeakCandidates() +// makes no JNI call (it only copies the opaque jweak field, never +// dereferences it), so it is safe to call on the real singleton without a +// live JVM, and the *ForTest seams already in place are enough to seed +// arbitrary ring-buffer states without going through cleanup_table(). +class SelectLeakCandidatesTest : public ::testing::Test { +protected: + void SetUp() override { + installGtestCrashHandler(); + LivenessTracker::instance()->klassPopulationResetForTest(); + } + + void TearDown() override { + LivenessTracker::instance()->klassPopulationResetForTest(); + restoreDefaultSignalHandlers(); + } + + static jweak fakeRef(uintptr_t tag) { + return reinterpret_cast(tag); + } + + // Pushes `n` samples (count values `counts[0..n)`, one per epoch starting + // at `start_epoch`) into klass_id's ring buffer via the same + // recordKlassPopulationSampleLocked() path production code drives from + // cleanup_table()'s epoch-advance pass (klassPopulationRecordForTest() is + // a direct pass-through to it, see its header comment). ALSO seeds a + // qualifying per-tid trend for the same epochs (a linear 1..n ramp on a + // fixed synthetic tid) - selectLeakCandidates() now requires a qualifying + // allocating thread on top of the klass-level ramp, so a series seeded + // through this helper represents a genuinely thread-concentrated leak. + // Tests that specifically exercise the per-tid gate itself seed the + // tid trends (or their absence) directly via tidTrendRecordForTest(). + static void seedSeries(LivenessTracker *tracker, u32 klass_id, + const u16 *counts, int n, u64 start_epoch) { + for (int i = 0; i < n; i++) { + int slot; + bool created; + tracker->klassPopulationRecordForTest(klass_id, counts[i], + start_epoch + i, &slot, + &created); + tracker->tidTrendRecordForTest(klass_id, /*tid=*/42, + (u32)(i + 1), start_epoch + i); + } + } +}; + +// A klass whose population is monotonically increasing for long enough has a +// positive slope, clears the growth/floor magnitude bars +// (hasQualifyingGrowth()) for enough consecutive epochs to satisfy the +// sustained-trend hysteresis requirement, and is returned, carrying its +// representative jweak through unchanged. 20 samples (not just the 10-sample +// minimum fill) - see MinimumFillAloneDoesNotClearHysteresis/ +// SustainedGrowthClearsHysteresis below for the boundary this margin avoids. +TEST_F(SelectLeakCandidatesTest, GrowingPopulationIsSelected) { + LivenessTracker *tracker = LivenessTracker::instance(); + + u16 growing[20]; + for (int i = 0; i < 20; i++) { + growing[i] = (u16)(i + 1); + } + seedSeries(tracker, /*klass_id=*/1, growing, 20, /*start_epoch=*/1); + jweak rep = fakeRef(0x1); + tracker->klassPopulationSetRepresentativeForTest(nullptr, 1, rep); + + KlassCandidate out[5]; + int count = tracker->selectLeakCandidates(out, 5); + + ASSERT_EQ(count, 1); + EXPECT_EQ(out[0].klass_id, 1u); + EXPECT_EQ(out[0].representative, rep); +} + +// A klass with a flat population (zero slope) is not a growth candidate - +// the design doc requires strictly positive slope, not "non-negative". +TEST_F(SelectLeakCandidatesTest, FlatPopulationIsNotSelected) { + LivenessTracker *tracker = LivenessTracker::instance(); + + const u16 flat[10] = {5, 5, 5, 5, 5, 5, 5, 5, 5, 5}; + seedSeries(tracker, /*klass_id=*/1, flat, 10, /*start_epoch=*/1); + + KlassCandidate out[5]; + int count = tracker->selectLeakCandidates(out, 5); + + EXPECT_EQ(count, 0); +} + +// A klass whose population is shrinking has a negative slope and must not be +// reported as a leak candidate. +TEST_F(SelectLeakCandidatesTest, ShrinkingPopulationIsNotSelected) { + LivenessTracker *tracker = LivenessTracker::instance(); + + const u16 shrinking[10] = {10, 9, 8, 7, 6, 5, 4, 3, 2, 1}; + seedSeries(tracker, /*klass_id=*/1, shrinking, 10, /*start_epoch=*/1); + + KlassCandidate out[5]; + int count = tracker->selectLeakCandidates(out, 5); + + EXPECT_EQ(count, 0); +} + +// A klass with fewer than KLASS_POPULATION_MIN_FILL_FOR_TREND (10) samples +// is skipped regardless of how strong its apparent trend looks - not enough +// history yet to trust it (design doc's explicit minimum-fill requirement). +TEST_F(SelectLeakCandidatesTest, JustBelowMinimumFillIsNotSelected) { + LivenessTracker *tracker = LivenessTracker::instance(); + + const u16 growing_but_short[9] = {1, 2, 3, 4, 5, 6, 7, 8, 9}; + seedSeries(tracker, /*klass_id=*/1, growing_but_short, 9, + /*start_epoch=*/1); + + KlassCandidate out[5]; + int count = tracker->selectLeakCandidates(out, 5); + + EXPECT_EQ(count, 0); +} + +// Exactly KLASS_POPULATION_MIN_FILL_FOR_TREND (10) samples clears +// hasQualifyingGrowth() on only its very last push - every earlier push saw +// ring_fill below the minimum and was rejected outright, so +// consecutive_positive is only 1 by the time fill reaches 10. One qualifying +// epoch does not clear the sustained-trend hysteresis requirement +// (LEAK_TREND_HYSTERESIS_BASE, 5 consecutive qualifying epochs) on its own - +// this used to be enough before that gate existed (hence this test's name), +// but is not anymore; see SustainedGrowthClearsHysteresis below for the new +// equivalent boundary test. +TEST_F(SelectLeakCandidatesTest, MinimumFillAloneDoesNotClearHysteresis) { + LivenessTracker *tracker = LivenessTracker::instance(); + + const u16 growing[10] = {1, 2, 3, 4, 5, 6, 7, 8, 9, 10}; + seedSeries(tracker, /*klass_id=*/1, growing, 10, /*start_epoch=*/1); + + KlassCandidate out[5]; + int count = tracker->selectLeakCandidates(out, 5); + + EXPECT_EQ(count, 0); +} + +// Once growth/floor keeps qualifying for enough additional epochs past +// min-fill to reach LEAK_TREND_HYSTERESIS_BASE (5 consecutive qualifying +// epochs: fill 10 through 14), the klass is trusted. +TEST_F(SelectLeakCandidatesTest, SustainedGrowthClearsHysteresis) { + LivenessTracker *tracker = LivenessTracker::instance(); + + u16 growing[14]; + for (int i = 0; i < 14; i++) { + growing[i] = (u16)(i + 1); + } + seedSeries(tracker, /*klass_id=*/1, growing, 14, /*start_epoch=*/1); + + KlassCandidate out[5]; + int count = tracker->selectLeakCandidates(out, 5); + + EXPECT_EQ(count, 1); +} + +// The aggregate post-GC heap floor (heapFloorRising()) lowers the number of +// consecutive qualifying epochs required from LEAK_TREND_HYSTERESIS_BASE (5) +// to LEAK_TREND_HYSTERESIS_CORROBORATED (3) for every candidate in the same +// scan - it cannot single out which klass is responsible for its own rise, +// so it can only raise or lower this bar uniformly, never reorder candidates +// against each other (see that pair's own comment, livenessTracker.h). +TEST_F(SelectLeakCandidatesTest, HeapFloorCorroborationLowersRequiredHysteresis) { + LivenessTracker *tracker = LivenessTracker::instance(); + + // 12 samples: 3 consecutive qualifying epochs past min-fill (fill = 10, + // 11, 12) - enough for LEAK_TREND_HYSTERESIS_CORROBORATED (3) but not + // LEAK_TREND_HYSTERESIS_BASE (5). + u16 growing[12]; + for (int i = 0; i < 12; i++) { + growing[i] = (u16)(i + 1); + } + seedSeries(tracker, /*klass_id=*/1, growing, 12, /*start_epoch=*/1); + + KlassCandidate out[5]; + EXPECT_EQ(tracker->selectLeakCandidates(out, 5), 0) + << "3 qualifying epochs clear the corroborated (3) but not the base " + "(5) hysteresis bar - without heap-floor corroboration this klass " + "must not be selected yet"; + + // A rising aggregate heap floor (10 samples, clearly growing) makes + // heapFloorRising() report true, lowering the bar for this same scan. + constexpr u64 GiB = 1ULL << 30; + constexpr u64 MiB = 1ULL << 20; + for (int i = 0; i < 10; i++) { + tracker->heapFloorRecordForTest(2 * GiB + (u64)i * 50 * MiB); + } + ASSERT_TRUE(tracker->heapFloorRisingForTest()); + + EXPECT_EQ(tracker->selectLeakCandidates(out, 5), 1); +} + +// --- Per-(klass, tid) qualification gate (TidTrend, livenessTracker.h) --- +// The disjoint-tagged-vs-frontier pod finding: a whole-klass rising +// generation count can come from churn spread across MANY allocating +// threads, each retaining a STABLE handful of instances. A klass with a +// qualifying klass-level trend but NO thread whose own trend qualifies is +// NOT a leak candidate. +TEST_F(SelectLeakCandidatesTest, KlassTrendWithoutQualifyingTidIsNotSelected) { + LivenessTracker *tracker = LivenessTracker::instance(); + + // Klass-level ramp only - no per-tid trend seeded at all (the raw seam + // loop, not seedSeries(), which would seed a qualifying tid too). + u16 growing[20]; + for (int i = 0; i < 20; i++) { + growing[i] = (u16)(i + 1); + int slot; + bool created; + tracker->klassPopulationRecordForTest(/*klass_id=*/1, growing[i], + /*epoch=*/i + 1, &slot, &created); + } + + KlassCandidate out[5]; + EXPECT_EQ(tracker->selectLeakCandidates(out, 5), 0) + << "a klass-level rise with no qualifying allocating thread is the " + "machinery-churn shape observed on the hotdog pod - it must not " + "become a candidate"; +} + +// A klass trend plus a tid trend that is FLAT (machinery: stable small +// retained set, no rising age span, below the retained-count bar) does +// not qualify either - each discriminator is necessary, not just one of +// them being absent. +TEST_F(SelectLeakCandidatesTest, FlatTidTrendDoesNotQualify) { + LivenessTracker *tracker = LivenessTracker::instance(); + + u16 growing[20]; + for (int i = 0; i < 20; i++) { + growing[i] = (u16)(i + 1); + int slot; + bool created; + tracker->klassPopulationRecordForTest(/*klass_id=*/1, growing[i], + /*epoch=*/i + 1, &slot, &created); + // Constant 2 surviving tracked instances, every epoch: below the + // retained-count bar and no rising age-cardinality trend. + tracker->tidTrendRecordForTest(/*klass_id=*/1, /*tid=*/7, /*count=*/2, + /*epoch=*/i + 1); + } + + KlassCandidate out[5]; + EXPECT_EQ(tracker->selectLeakCandidates(out, 5), 0); +} + +// The retained-count bar (TID_RETAINED_COUNT_BAR) qualifies a tid whose +// instances all share one age (one-cohort-per-thread accumulation - each +// one-shot worker thread's distinct-age count stays 1 forever) as long as +// it retains enough tracked instances - the discriminator that covers +// LeakingCacheScenario's allocator-thread shape. +TEST_F(SelectLeakCandidatesTest, RetainedCountBarQualifiesOneCohortShape) { + LivenessTracker *tracker = LivenessTracker::instance(); + + u16 growing[20]; + for (int i = 0; i < 20; i++) { + growing[i] = (u16)(i + 1); + int slot; + bool created; + tracker->klassPopulationRecordForTest(/*klass_id=*/1, growing[i], + /*epoch=*/i + 1, &slot, &created); + // 12 > TID_RETAINED_COUNT_BAR (8), flat every epoch: the age trend + // alone would never qualify (no rise), the bar does. + tracker->tidTrendRecordForTest(/*klass_id=*/1, /*tid=*/7, /*count=*/12, + /*epoch=*/i + 1); + } + + KlassCandidate out[5]; + ASSERT_EQ(tracker->selectLeakCandidates(out, 5), 1); + ASSERT_GE(out[0].qualifying_tid_count, 1); + EXPECT_EQ(out[0].qualifying_tids[0], 7); +} + +// A rising per-tid trend alone (below the retained-count bar) qualifies: +// small leaks grow their age span long before their count clears the bar - +// the hotdog pod's simulated-memory-leak thread (12 tracked [B instances, +// age_count rising 2->3) is exactly this shape. +TEST_F(SelectLeakCandidatesTest, RisingTidTrendQualifiesBelowCountBar) { + LivenessTracker *tracker = LivenessTracker::instance(); + + u16 growing[20]; + for (int i = 0; i < 20; i++) { + growing[i] = (u16)(i + 1); + int slot; + bool created; + tracker->klassPopulationRecordForTest(/*klass_id=*/1, growing[i], + /*epoch=*/i + 1, &slot, &created); + // Rising, but capped at 6 tracked instances (below the bar of 8). + u32 tid_count = (u32)((i / 3) + 1) > 6 ? 6 : (u32)((i / 3) + 1); + tracker->tidTrendRecordForTest(/*klass_id=*/1, /*tid=*/9, tid_count, + /*epoch=*/i + 1); + } + + KlassCandidate out[5]; + ASSERT_EQ(tracker->selectLeakCandidates(out, 5), 1); + ASSERT_GE(out[0].qualifying_tid_count, 1); + EXPECT_EQ(out[0].qualifying_tids[0], 9); +} + +// Multiple positive-slope klasses must come back sorted by slope magnitude +// descending, not insertion order. 20-sample linear series (not the original +// 10 - see GrowingPopulationIsSelected's own note) at three distinct growth +// rates so every klass clears the growth/floor magnitude bars and the +// sustained-trend hysteresis requirement, while still ranking distinctly. +TEST_F(SelectLeakCandidatesTest, OrdersByMagnitudeDescending) { + LivenessTracker *tracker = LivenessTracker::instance(); + + u16 strong[20], weak[20], medium[20]; + for (int i = 0; i < 20; i++) { + strong[i] = (u16)(1 + i * 3); // steepest -> strongest + weak[i] = (u16)(1 + i * 1); // shallowest -> weakest + medium[i] = (u16)(1 + i * 2); + } + + seedSeries(tracker, /*klass_id=*/1, strong, 20, /*start_epoch=*/1); + seedSeries(tracker, /*klass_id=*/2, weak, 20, /*start_epoch=*/1); + seedSeries(tracker, /*klass_id=*/3, medium, 20, /*start_epoch=*/1); + + KlassCandidate out[5]; + int count = tracker->selectLeakCandidates(out, 5); + + ASSERT_EQ(count, 3); + EXPECT_EQ(out[0].klass_id, 1u); // strongest + EXPECT_EQ(out[1].klass_id, 3u); // middle + EXPECT_EQ(out[2].klass_id, 2u); // weakest +} + +// More than MAX_LEAK_CANDIDATES (5) positive-slope klasses exist: only the +// top 5 by magnitude are returned, even though the caller asked for more - +// design doc's "top 3-5" cutoff is an upper bound the method itself enforces, +// not just a suggestion to the caller. +TEST_F(SelectLeakCandidatesTest, CapsAtMaxLeakCandidatesRegardlessOfRequestedMax) { + LivenessTracker *tracker = LivenessTracker::instance(); + + // 7 klasses, each growing by a distinct amount per sample so every one + // has a distinct, positive slope: klass_id N grows by N per sample. 20 + // samples (not 10 - see GrowingPopulationIsSelected's own note) so every + // klass also clears the sustained-trend hysteresis requirement. + for (u32 klass_id = 1; klass_id <= 7; klass_id++) { + u16 series[20]; + for (int i = 0; i < 20; i++) { + series[i] = (u16)(1 + i * klass_id); + } + seedSeries(tracker, klass_id, series, 20, /*start_epoch=*/1); + } + + KlassCandidate out[10]; + int count = tracker->selectLeakCandidates(out, 10); + + ASSERT_EQ(count, 5); // MAX_LEAK_CANDIDATES, not the requested 10 + // Steeper growth (larger klass_id) means larger slope - the 5 returned + // must be the 5 largest klass_ids, strongest first. + EXPECT_EQ(out[0].klass_id, 7u); + EXPECT_EQ(out[1].klass_id, 6u); + EXPECT_EQ(out[2].klass_id, 5u); + EXPECT_EQ(out[3].klass_id, 4u); + EXPECT_EQ(out[4].klass_id, 3u); +} + +// The caller's own buffer capacity (`max`) is honored when it is smaller +// than MAX_LEAK_CANDIDATES - the method must never write past `max` slots. +TEST_F(SelectLeakCandidatesTest, HonorsCallerSuppliedMaxBelowCap) { + LivenessTracker *tracker = LivenessTracker::instance(); + + for (u32 klass_id = 1; klass_id <= 3; klass_id++) { + u16 series[20]; + for (int i = 0; i < 20; i++) { + series[i] = (u16)(1 + i * klass_id); + } + seedSeries(tracker, klass_id, series, 20, /*start_epoch=*/1); + } + + KlassCandidate out[2]; + int count = tracker->selectLeakCandidates(out, 2); + + ASSERT_EQ(count, 2); + EXPECT_EQ(out[0].klass_id, 3u); // strongest + EXPECT_EQ(out[1].klass_id, 2u); // second-strongest; klass 1 dropped +} + +// An empty population table (nothing tracked yet, or _gc_generations was +// never enabled so population tracking's own gate left the table empty) yields no +// candidates regardless of `max` - no separate guard is needed inside +// selectLeakCandidates() beyond the table being empty. +TEST_F(SelectLeakCandidatesTest, EmptyTableReturnsZero) { + LivenessTracker *tracker = LivenessTracker::instance(); + + KlassCandidate out[5]; + int count = tracker->selectLeakCandidates(out, 5); + + EXPECT_EQ(count, 0); +} + +// --------------------------------------------------------------------------- +// topKlassesByGenerationCount() - ranks by most-recent count_ring sample, +// with NO trend/hysteresis gate at all (unlike selectLeakCandidates() above) +// - see its own header comment (livenessTracker.h) for why: it exists to run +// AFTER hasLeakSignal() has already fired via the slower, hysteresis-gated +// path, as a faster follow-up ranking for ReferenceChainTracker's rotation +// priority. Reuses SelectLeakCandidatesTest's fixture/seedSeries() seam - +// same table, same seeding mechanism, different read method under test. +// +// Returns stable_class_tag, NOT klass_id (the classMap dictionary id) - see +// that field's own comment (livenessTracker.h) for why the two are +// deliberately different values. klassPopulationRecordForTest() (seedSeries()'s +// own underlying seam) bypasses foldKlassCountsLocked() entirely, so it never +// mints a stable_class_tag - tests seed it explicitly via +// klassPopulationSetStableClassTagForTest(), using a value distinct from +// klass_id in each test below specifically so a test that accidentally +// asserted against klass_id instead would fail loudly, not silently pass by +// coincidence. +// --------------------------------------------------------------------------- + +// A single sample (ring_fill == 1, far below selectLeakCandidates()'s +// KLASS_POPULATION_MIN_FILL_FOR_TREND) is enough to rank - the whole point +// of skipping the hysteresis gate. +TEST_F(SelectLeakCandidatesTest, TopKlassesByGenerationCountNeedsOnlyOneSample) { + LivenessTracker *tracker = LivenessTracker::instance(); + + const u16 single[1] = {42}; + seedSeries(tracker, /*klass_id=*/7, single, 1, /*start_epoch=*/1); + tracker->klassPopulationSetStableClassTagForTest(7, -700); + + u32 out[5]; + int count = tracker->topKlassesByGenerationCount(out, 5); + + ASSERT_EQ(count, 1); + EXPECT_EQ(out[0], (u32)-700); +} + +// A klass with samples but no minted stable_class_tag yet (no live instance +// resolved so far - foldKlassCountsLocked()'s own comment) has nothing +// usable to return and must be skipped, not reported with a bogus 0 tag. +TEST_F(SelectLeakCandidatesTest, TopKlassesByGenerationCountSkipsUnmintedStableClassTag) { + LivenessTracker *tracker = LivenessTracker::instance(); + + const u16 single[1] = {42}; + seedSeries(tracker, /*klass_id=*/7, single, 1, /*start_epoch=*/1); + // Deliberately no klassPopulationSetStableClassTagForTest() call. + + u32 out[5]; + int count = tracker->topKlassesByGenerationCount(out, 5); + + EXPECT_EQ(count, 0); +} + +// Ranking is by the MOST RECENT sample, not the peak or the mean - a klass +// whose count has since fallen still ranks by where it is NOW. +TEST_F(SelectLeakCandidatesTest, TopKlassesByGenerationCountUsesMostRecentSampleNotPeak) { + LivenessTracker *tracker = LivenessTracker::instance(); + + const u16 peakedThenFell[4] = {100, 5, 5, 5}; // peak 100, now 5 + const u16 steady[4] = {10, 10, 10, 10}; // never peaked, now 10 + seedSeries(tracker, /*klass_id=*/1, peakedThenFell, 4, /*start_epoch=*/1); + seedSeries(tracker, /*klass_id=*/2, steady, 4, /*start_epoch=*/1); + tracker->klassPopulationSetStableClassTagForTest(1, -100); + tracker->klassPopulationSetStableClassTagForTest(2, -200); + + u32 out[5]; + int count = tracker->topKlassesByGenerationCount(out, 5); + + ASSERT_EQ(count, 2); + EXPECT_EQ(out[0], (u32)-200) << "klass 2 (currently 10) should outrank " + "klass 1 (currently 5, despite an " + "earlier peak of 100)"; + EXPECT_EQ(out[1], (u32)-100); +} + +// A flat or shrinking population - which selectLeakCandidates() would +// exclude entirely (zero/negative slope) - still ranks here: this method +// applies no growth-direction requirement, only magnitude. +TEST_F(SelectLeakCandidatesTest, TopKlassesByGenerationCountIncludesFlatAndShrinkingPopulations) { + LivenessTracker *tracker = LivenessTracker::instance(); + + const u16 flat[3] = {50, 50, 50}; + const u16 shrinking[3] = {30, 20, 10}; + seedSeries(tracker, /*klass_id=*/1, flat, 3, /*start_epoch=*/1); + seedSeries(tracker, /*klass_id=*/2, shrinking, 3, /*start_epoch=*/1); + tracker->klassPopulationSetStableClassTagForTest(1, -100); + tracker->klassPopulationSetStableClassTagForTest(2, -200); + + u32 out[5]; + int count = tracker->topKlassesByGenerationCount(out, 5); + + ASSERT_EQ(count, 2); + EXPECT_EQ(out[0], (u32)-100) << "flat-at-50 outranks shrinking-to-10"; + EXPECT_EQ(out[1], (u32)-200); +} + +TEST_F(SelectLeakCandidatesTest, TopKlassesByGenerationCountCapsAtMaxLeakCandidates) { + LivenessTracker *tracker = LivenessTracker::instance(); + + for (u32 klass_id = 1; klass_id <= 8; klass_id++) { + const u16 sample[1] = {(u16)(klass_id * 10)}; + seedSeries(tracker, klass_id, sample, 1, /*start_epoch=*/1); + tracker->klassPopulationSetStableClassTagForTest(klass_id, -(jlong)(klass_id * 100)); + } + + u32 out[10]; + int count = tracker->topKlassesByGenerationCount(out, 10); + + EXPECT_EQ(count, 5) << "capped at MAX_LEAK_CANDIDATES regardless of the " + "caller-supplied max"; + // Descending by most-recent count: klass 8 (count 80, tag -800) first. + EXPECT_EQ(out[0], (u32)-800); + EXPECT_EQ(out[4], (u32)-400); +} + +TEST_F(SelectLeakCandidatesTest, TopKlassesByGenerationCountEmptyTableReturnsZero) { + LivenessTracker *tracker = LivenessTracker::instance(); + + u32 out[5]; + int count = tracker->topKlassesByGenerationCount(out, 5); + + EXPECT_EQ(count, 0); +} + +// --------------------------------------------------------------------------- +// Heap-wide time-to-OOM projection (secondsToOOM()) - the aggressive-leak gap +// selectLeakCandidates()'s per-klass ring-fill/hysteresis gate leaves open: +// that gate can take longer to trust a candidate than a fast, heap-wide leak +// has left before OOM (see ReferenceChainTracker::hasLeakSignal()'s +// OOM_URGENT_THRESHOLD_S fast path, referenceChains.h/.cpp). Exercises the +// heap-floor ring/time-ring pair directly via the same heapFloorRecordForTest() +// seam SelectLeakCandidatesTest's HeapFloorCorroboration test above already +// uses for heapFloorRising(), plus setMaxHeapBytesForTest() to avoid the +// JNI-dependent HeapUsage::getMaxHeap() call this suite has no live JVM for. +// --------------------------------------------------------------------------- +class SecondsToOOMTest : public ::testing::Test { +protected: + static constexpr u64 SEC_NS = 1000000000ULL; + static constexpr u64 MiB = 1ULL << 20; + + void SetUp() override { + installGtestCrashHandler(); + LivenessTracker::instance()->klassPopulationResetForTest(); + LivenessTracker::instance()->setGcGenerationsForTest(true); + } + + void TearDown() override { + LivenessTracker::instance()->klassPopulationResetForTest(); + LivenessTracker::instance()->setGcGenerationsForTest(false); + LivenessTracker::instance()->setMaxHeapBytesForTest(-1); + restoreDefaultSignalHandlers(); + } + + // Ten samples, one second apart, growing by 100MiB each: earliest third + // (indices 0-2) means to 1100MiB at t=1s, recent third (indices 7-9) + // means to 1800MiB at t=8s - a 700MiB rise over 7s, i.e. exactly + // 100MiB/s, chosen so the projected time-to-exhaustion below comes out + // to a clean value rather than a value only checked against itself. + static void seedRisingFloor(LivenessTracker *tracker) { + for (int i = 0; i < 10; i++) { + tracker->heapFloorRecordForTest(1000 * MiB + (u64)i * 100 * MiB, + (u64)i * SEC_NS); + } + } +}; + +// Fewer than KLASS_POPULATION_MIN_FILL_FOR_TREND (10) heap-floor samples - +// same "not enough history yet" gate ringThirdsStats() applies to every +// other trend check in this class. +TEST_F(SecondsToOOMTest, NotEnoughSamplesReturnsNegative) { + LivenessTracker *tracker = LivenessTracker::instance(); + tracker->setMaxHeapBytesForTest((jlong)(2800 * MiB)); + for (int i = 0; i < 9; i++) { + tracker->heapFloorRecordForTest(1000 * MiB + (u64)i * 100 * MiB, + (u64)i * SEC_NS); + } + EXPECT_LT(tracker->secondsToOOM(), 0.0); +} + +// A flat floor (zero byte delta between the earliest and recent thirds) is +// not rising - no projection is offered, mirroring hasQualifyingGrowth()'s +// own "strictly positive slope" requirement. +TEST_F(SecondsToOOMTest, FlatFloorReturnsNegative) { + LivenessTracker *tracker = LivenessTracker::instance(); + tracker->setMaxHeapBytesForTest((jlong)(2800 * MiB)); + for (int i = 0; i < 10; i++) { + tracker->heapFloorRecordForTest(1000 * MiB, (u64)i * SEC_NS); + } + EXPECT_LT(tracker->secondsToOOM(), 0.0); +} + +// No heap-floor history is ever recorded outside _gc_generations (onGC()'s +// own gate) - secondsToOOM() must not fabricate a projection from whatever +// ring contents happen to be left over from a previous _gc_generations +// session. +TEST_F(SecondsToOOMTest, GcGenerationsDisabledReturnsNegative) { + LivenessTracker *tracker = LivenessTracker::instance(); + tracker->setMaxHeapBytesForTest((jlong)(2800 * MiB)); + seedRisingFloor(tracker); + tracker->setGcGenerationsForTest(false); + + EXPECT_LT(tracker->secondsToOOM(), 0.0); +} + +// A rising floor is meaningless without a resolved max heap size to project +// against - initialize_table()'s own Error path (livenessTracker.cpp) never +// lets liveness tracking start without one, but secondsToOOM() must still +// guard the case explicitly rather than dividing/comparing against -1. +TEST_F(SecondsToOOMTest, UnresolvedMaxHeapReturnsNegative) { + LivenessTracker *tracker = LivenessTracker::instance(); + tracker->setMaxHeapBytesForTest(-1); + seedRisingFloor(tracker); + + EXPECT_LT(tracker->secondsToOOM(), 0.0); +} + +// The worked example seedRisingFloor() documents: 700MiB rise over 7s +// (100MiB/s) with 1000MiB of headroom (2800MiB max heap - 1800MiB recent +// floor mean) projects to exactly 10 seconds. +TEST_F(SecondsToOOMTest, RisingFloorProjectsExpectedSeconds) { + LivenessTracker *tracker = LivenessTracker::instance(); + tracker->setMaxHeapBytesForTest((jlong)(2800 * MiB)); + seedRisingFloor(tracker); + + EXPECT_NEAR(tracker->secondsToOOM(), 9.0, 1e-6); +} + +// The floor's own recent-third mean has already reached the max heap size - +// exhaustion is "now", not some positive number of seconds out. +TEST_F(SecondsToOOMTest, FloorAtMaxHeapReturnsZero) { + LivenessTracker *tracker = LivenessTracker::instance(); + tracker->setMaxHeapBytesForTest((jlong)(1800 * MiB)); // == recent third's mean + seedRisingFloor(tracker); + + EXPECT_EQ(tracker->secondsToOOM(), 0.0); +} + +// --------------------------------------------------------------------------- +// Leak tag pool (design A: direct tagging of leaking objects) +// --------------------------------------------------------------------------- +// The pool hands out JVMTI tags in [LEAK_TAG_BASE, LEAK_TAG_BASE + 256) and +// recycles them when the tracked object dies. These tests exercise the pure +// pool mechanics (acquire/release/info); tagLeakInstances() itself needs a +// live JVM (SetTag) and is verified on-pod. + +class LeakTagPoolTest : public ::testing::Test { +protected: + void SetUp() override { + LivenessTracker::instance()->leakTagPoolResetForTest(); + } +}; + +TEST_F(LeakTagPoolTest, AcquireReturnsTagsInLeakRange) { + LivenessTracker *tracker = LivenessTracker::instance(); + jlong base = tracker->leakTagBaseForTest(); + for (int i = 0; i < tracker->leakTagPoolSizeForTest(); i++) { + jlong tag = tracker->acquireLeakTagForTest(/*call_trace_id=*/100 + i, + /*tid=*/7 + i); + EXPECT_GE(tag, base) << "tag below pool range at acquire " << i; + EXPECT_LT(tag, base + tracker->leakTagPoolSizeForTest()) + << "tag above pool range at acquire " << i; + } + // Pool exhausted: further acquires fail with 0. + EXPECT_EQ(0, tracker->acquireLeakTagForTest(1, 1)); + EXPECT_EQ(0, tracker->leakTagFreeCountForTest()); +} + +TEST_F(LeakTagPoolTest, ReleaseReturnsTagToPoolAndInfoIsInvalidated) { + LivenessTracker *tracker = LivenessTracker::instance(); + int pool_size = tracker->leakTagPoolSizeForTest(); + + jlong tag = tracker->acquireLeakTagForTest(42, 99); + ASSERT_GE(tag, tracker->leakTagBaseForTest()); + + u64 call_trace_id = 0; + jint tid = 0; + EXPECT_TRUE(tracker->getLeakTagInfo(tag, &call_trace_id, &tid)); + EXPECT_EQ(42u, call_trace_id); + EXPECT_EQ(99, tid); + + tracker->releaseLeakTagForTest(tag); + // Released tags must not report stale info. + u64 stale_ctid = 12345; + jint stale_tid = 12345; + EXPECT_FALSE(tracker->getLeakTagInfo(tag, &stale_ctid, &stale_tid)) + << "released tag still reports info"; + + // The released tag can be acquired again (reusable pool), and the free + // count was restored: pool_size-1 after the acquire, back to pool_size + // after the release, pool_size-1 again after the re-acquire. + EXPECT_EQ(pool_size, tracker->leakTagFreeCountForTest()); + jlong re_tag = tracker->acquireLeakTagForTest(43, 100); + EXPECT_EQ(tag, re_tag) << "released tag should be recycled first (LIFO)"; + EXPECT_EQ(pool_size - 1, tracker->leakTagFreeCountForTest()); +} + +TEST_F(LeakTagPoolTest, ReleaseOutsidePoolRangeIsIgnored) { + LivenessTracker *tracker = LivenessTracker::instance(); + jlong base = tracker->leakTagBaseForTest(); + int free_before = tracker->leakTagFreeCountForTest(); + + // Tags outside [base, base+pool): frontier tags (small positive), class + // tags (negative), and one-past-the-end must all be rejected. + tracker->releaseLeakTagForTest(1); + tracker->releaseLeakTagForTest(-1); + tracker->releaseLeakTagForTest(0); + tracker->releaseLeakTagForTest(base + tracker->leakTagPoolSizeForTest()); + tracker->releaseLeakTagForTest(base - 1); + + EXPECT_EQ(free_before, tracker->leakTagFreeCountForTest()) + << "out-of-range releases must not corrupt the free list"; +} + +// --------------------------------------------------------------------------- +// Chase-phase admission boost (admitForTracking()/noteSelectedCandidates()/ +// setUrgentTracking() - see livenessTracker.h). Same "exercise the singleton +// directly" rationale as KlassPopulationTest above: the admission gate is +// JNI-free pure logic (atomic reads + the per-thread RNG draw), so it is +// testable without a live JVM; only track() beyond the gate needs a JNIEnv. +// +// Determinism: admissionResetForTest() forces _subsample_ratio to 0 and resets +// this thread's RNG ThreadLocal, so an unboosted tid's draw always comes from +// a freshly default-seeded mt19937 whose first draw is strictly in (0,1) - +// ratio 0 can never admit. That pins the fall-through case without asserting +// on RNG internals. +class AdmissionBoostTest : public ::testing::Test { +protected: + void SetUp() override { + installGtestCrashHandler(); + LivenessTracker::instance()->admissionResetForTest(); + } + + void TearDown() override { + // Leave the singleton clean for any test that follows in the same + // process (watched tids / urgency would 100%-admit unrelated tids). + LivenessTracker::instance()->admissionResetForTest(); + LivenessTracker::instance()->setSubsampleRatioForTest(0.1); + restoreDefaultSignalHandlers(); + } + + static void fillCandidate(KlassCandidate *kc, jint tid) { + kc->klass_id = 1; + kc->representative = reinterpret_cast(0x1); + kc->qualifying_tids[0] = tid; + kc->qualifying_tid_count = 1; + } +}; + +// A watched tid is admitted even though the configured ratio (0) would +// deterministically reject it - the boost precedes the ratio draw. +TEST_F(AdmissionBoostTest, WatchedTidAdmittedDespiteRejectingRatio) { + LivenessTracker *tracker = LivenessTracker::instance(); + KlassCandidate kc; + fillCandidate(&kc, /*tid=*/42); + tracker->noteSelectedCandidates(&kc, 1); + + EXPECT_TRUE(tracker->admitForTrackingForTest(42)); + // An unwatched tid on the same thread falls through to the ratio draw and + // is rejected (ratio 0). + EXPECT_FALSE(tracker->admitForTrackingForTest(43)); +} + +// Urgency admits everything - any tid, watched or not. +TEST_F(AdmissionBoostTest, UrgencyAdmitsAllTids) { + LivenessTracker *tracker = LivenessTracker::instance(); + tracker->setUrgentTracking(true); + EXPECT_TRUE(tracker->admitForTrackingForTest(1234)); + EXPECT_TRUE(tracker->admitForTrackingForTest(5678)); + + // Releasing urgency restores the ratio gate: unwatched tids reject again, + // watched tids stay boosted. + tracker->setUrgentTracking(false); + EXPECT_FALSE(tracker->admitForTrackingForTest(1234)); + KlassCandidate kc; + fillCandidate(&kc, 1234); + tracker->noteSelectedCandidates(&kc, 1); + EXPECT_TRUE(tracker->admitForTrackingForTest(1234)); +} + +// noteSelectedCandidates() dedupes tids shared across candidates and caps +// the watched set at MAX_QUALIFYING_TIDS. +TEST_F(AdmissionBoostTest, WatchedTidsAreDedupedAndCapped) { + LivenessTracker *tracker = LivenessTracker::instance(); + constexpr int kMax = KlassCandidate::MAX_QUALIFYING_TIDS; // 8 + + // Two candidates: 5 tids each, tids 3 and 4 shared -> union is 1..7, in + // candidate order (candidates are rank-ordered, so a later candidate's + // tids only append what the earlier ones did not cover). + KlassCandidate kc[2]; + for (int t = 0; t < 5; t++) { + kc[0].qualifying_tids[t] = 1 + t; // 1..5 + kc[1].qualifying_tids[t] = 3 + t; // 3..7 -> adds 6,7 + } + for (int c = 0; c < 2; c++) { + kc[c].klass_id = 1 + c; + kc[c].representative = reinterpret_cast(0x1 + c); + kc[c].qualifying_tid_count = 5; + } + tracker->noteSelectedCandidates(kc, 2); + ASSERT_EQ(7, tracker->watchedTidCountForTest()); + for (int i = 0; i < 7; i++) { + EXPECT_EQ(i + 1, tracker->watchedTidForTest(i)) + << "shared tids must not duplicate; union stays in order"; + } + + // A poll whose union exceeds the cap (1..8 from candidate 1, 9 from + // candidate 2) keeps the earlier candidates' tids and drops the overflow + // tid - and that overflow tid is not admitted. + KlassCandidate over[2]; + for (int t = 0; t < kMax; t++) { + over[0].qualifying_tids[t] = 1 + t; // 1..8 + } + over[0].qualifying_tid_count = kMax; + over[0].klass_id = 1; + over[0].representative = reinterpret_cast(0x1); + fillCandidate(&over[1], /*tid=*/9); + tracker->noteSelectedCandidates(over, 2); + ASSERT_EQ(kMax, tracker->watchedTidCountForTest()); + for (int i = 0; i < kMax; i++) { + EXPECT_EQ(i + 1, tracker->watchedTidForTest(i)); + } + EXPECT_FALSE(tracker->admitForTrackingForTest(9)); + EXPECT_TRUE(tracker->admitForTrackingForTest(8)); +} + +// A zero-candidate poll clears the watched set - a tid left watched after the +// chase ends would keep admitting that thread at 100% across OS tid reuse. +TEST_F(AdmissionBoostTest, ZeroCandidatePollClearsWatchedSet) { + LivenessTracker *tracker = LivenessTracker::instance(); + KlassCandidate kc; + fillCandidate(&kc, 77); + tracker->noteSelectedCandidates(&kc, 1); + EXPECT_TRUE(tracker->admitForTrackingForTest(77)); + + tracker->noteSelectedCandidates(nullptr, 0); + EXPECT_EQ(tracker->watchedTidCountForTest(), 0); + EXPECT_FALSE(tracker->admitForTrackingForTest(77)); +} diff --git a/ddprof-lib/src/test/cpp/referenceChainJfrRoundtrip_ut.cpp b/ddprof-lib/src/test/cpp/referenceChainJfrRoundtrip_ut.cpp new file mode 100644 index 0000000000..9737272cb0 --- /dev/null +++ b/ddprof-lib/src/test/cpp/referenceChainJfrRoundtrip_ut.cpp @@ -0,0 +1,419 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +// --------------------------------------------------------------------------- +// PROF-15341 design doc, Open Question: does JMC's parser actually resolve +// the datadog.ReferenceChain event's `chain` field - declared in +// jfrMetadata.cpp as field("chain", T_CLASS, ..., F_CPOOL | F_ARRAY), i.e. an +// *array of scalar constant-pool-index* T_CLASS values - the same way it +// resolves a plain scalar F_CPOOL field (e.g. objectClass) or a plain +// F_ARRAY-of-composite-struct field (e.g. StackTrace.frames)? Neither of +// those two existing, already-exercised shapes proves this combination. +// +// This test answers that empirically, not by inspecting the JFR spec: it +// drives the *real* production write path (Recording - not a hand-rolled +// byte layout) to produce one complete, standalone, chunk-finalized .jfr +// file containing a real datadog.ReferenceChain event plus its class +// checkpoint, and leaves the actual JMC read-back to the companion Java test +// (ddprof-test's ReferenceChainJfrParserTest), which loads this file with +// org.openjdk.jmc.flightrecorder.JfrLoaderToolkit and asserts the resolved +// class names. +// +// Constructing a real `Recording` outside of Profiler::start()'s state +// machine is unavoidable here: this gtest binary has no live JVM attached +// (see referenceChains_ut.cpp's and jvmSupport_ut.cpp's fixture comments for +// the same, already-established constraint), and Profiler::start()/check() +// both require live JVM introspection (checkJvmCapabilities() -> +// JVMThread::hasJavaThreadId(), VMStructs-backed queries) that cannot be +// satisfied without one. Recording's constructor and recordReferenceChain() +// are already public production API (flightRecorder.h) that do not go +// through Profiler::start()/stop()/dump() at all, so using them directly is +// not a new hook - it is the same "drive the real machinery, don't hand-roll +// the format" approach as every other test in this file's neighbourhood, +// just entered one layer lower. The two remaining unavoidable null-pointer +// dependencies of Recording::finishChunk() - Profiler::cpuEngine()/ +// wallEngine() (NULL until Profiler::start() runs) and VM::jni() (NULL +// _vm otherwise) - are supplied via the exact same pre-existing, already +// test-appropriate friend-accessor mechanism this codebase already uses for +// VM::_jvmti (VMTestAccessor) and Profiler::_state (ProfilerTestAccessor, +// jvmSupport_ut.cpp): profiler.h and vmEntry.h already declare `friend class +// ProfilerTestAccessor;` / `friend class VMTestAccessor;` for exactly this +// purpose, so defining those classes here (per-translation-unit, like every +// other _ut.cpp that does the same) adds no new production surface. +// --------------------------------------------------------------------------- + +#include +#include +#include +#include +#include +#include +#include +#include "arguments.h" +#include "codeCache.h" +#include "common.h" +#include "engine.h" +#include "flightRecorder.h" +#include "jfrMetadata.h" +#include "profiler.h" +#include "referenceChains.h" +#include "tsc.h" +#include "vmEntry.h" +#include "hotspot/vmStructs.h" +#ifdef ASAN_ENABLED +#include +#endif +#include "gtest_crash_handler.h" + +static constexpr char REFERENCE_CHAIN_JFR_TEST_NAME[] = "ReferenceChainJfrRoundtripTest"; + +class ReferenceChainJfrRoundtripGlobalSetup { +public: + ReferenceChainJfrRoundtripGlobalSetup() { + installGtestCrashHandler(); + } + ~ReferenceChainJfrRoundtripGlobalSetup() { + restoreDefaultSignalHandlers(); + } +}; +static ReferenceChainJfrRoundtripGlobalSetup global_setup; + +// --------------------------------------------------------------------------- +// VMTestAccessor - friend of VM (vmEntry.h). Same purpose/name as the +// identically-named, independently-defined class in referenceChains_ut.cpp +// and jvmSupport_ut.cpp (each _ut.cpp translation unit defines its own copy; +// see this codebase's established convention). Extended here with a _vm +// setter: Recording::finishChunk() (flightRecorder.cpp) unconditionally +// calls VM::jni(), which dereferences VM::_vm - NULL by default in this +// live-JVM-less gtest binary - so a mocked JavaVM is required the same way +// a mocked jvmtiEnv already is for VM::_jvmti. +// --------------------------------------------------------------------------- +class VMTestAccessor { +public: + static jvmtiEnv *getJvmti() { return VM::_jvmti; } + static void setJvmti(jvmtiEnv *env) { VM::_jvmti = env; } + static JavaVM *getVm() { return VM::_vm; } + static void setVm(JavaVM *vm) { VM::_vm = vm; } + static CodeCache *getLibjvm() { return VM::_libjvm; } + static void setLibjvm(CodeCache *lib) { VM::_libjvm = lib; } +}; + +// --------------------------------------------------------------------------- +// ProfilerTestAccessor - friend of Profiler (profiler.h), same mechanism +// jvmSupport_ut.cpp already uses for Profiler::_state. Recording:: +// finishChunk() unconditionally dereferences Profiler::instance()-> +// cpuEngine()/wallEngine() (writeDatadogProfilerConfig) - both NULL until +// Profiler::start() runs, which (per this file's header comment) cannot run +// in this gtest binary. Engine's base-class methods (name()="None", +// interval()=0) are safe no-op defaults, so a plain Engine instance is +// sufficient here - no engine-specific behaviour is exercised by this test. +// --------------------------------------------------------------------------- +class ProfilerTestAccessor { +public: + static void setCpuEngine(Profiler *p, Engine *e) { p->_cpu_engine = e; } + static void setWallEngine(Profiler *p, Engine *e) { p->_wall_engine = e; } +}; + +// --------------------------------------------------------------------------- +// ReferenceChainsTestAccessor - friend of ReferenceChainTracker +// (referenceChains.h), same reset() as referenceChains_ut.cpp's own +// identically-named class (this file's independent copy, per this +// codebase's established per-translation-unit convention - see +// VMTestAccessor's comment above). Required because ReferenceChainTracker:: +// instance() is a process-wide singleton shared with every other _ut.cpp in +// this gtest binary. +// --------------------------------------------------------------------------- +class ReferenceChainsTestAccessor { +public: + static void reset() { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + delete t->_frontier; + t->_frontier = nullptr; + t->_class_tags = ClassTagTable(); + t->_next_tag = 1; + // Shared with LivenessTracker (classTagAllocator.h) - process-wide, + // not per-ReferenceChainTracker-instance. + ClassTagAllocator::resetForTest(); + t->_search_started = false; + t->_search_state = SearchState::RUNNING; + t->_abandon_reason = SearchAbandonReason::NONE; + t->_search_start_ns = 0; + t->_pending_expand.clear(); + t->_last_pass_gc_finish_epoch = 0; + t->_last_pass_ns = 0; + t->_passes_run = 0; + t->_resolved_chains.clear(); + } +}; + +static jvmtiError JNICALL mock_SetEventNotificationMode(jvmtiEnv *, jvmtiEventMode, + jvmtiEvent, jthread, ...) { + return JVMTI_ERROR_NONE; +} + +static jvmtiError JNICALL mock_GetAvailableProcessors(jvmtiEnv *, jint *count_ptr) { + *count_ptr = 1; + return JVMTI_ERROR_NONE; +} + +// Recording::finishChunk() pins currently-loaded classes via +// GetLoadedClasses() before/after serialization (see its own comment on the +// GC-unload race this guards against in a real JVM). Reporting zero loaded +// classes here is a faithful, not a cheated, answer for this gtest binary: +// there genuinely are no JVMTI-visible loaded classes without a live JVM, +// so the DeleteLocalRef()/Deallocate() cleanup loop that follows is a no-op. +static jvmtiError JNICALL mock_GetLoadedClasses(jvmtiEnv *, jint *count_ptr, + jclass **classes_ptr) { + *count_ptr = 0; + *classes_ptr = nullptr; + return JVMTI_ERROR_NONE; +} + +static jvmtiError JNICALL mock_Deallocate(jvmtiEnv *, unsigned char *mem) { + free(mem); + return JVMTI_ERROR_NONE; +} + +static JNIEnv_ g_mock_jni_env{}; + +static jint JNICALL mock_GetEnv(JavaVM *, void **penv, jint) { + *penv = &g_mock_jni_env; + return 0; // JNI_OK +} + +class ReferenceChainJfrRoundtripTest : public ::testing::Test { +protected: + jvmtiInterface_1_ jvmti_tbl{}; + _jvmtiEnv mock_jvmti{}; + JNIInvokeInterface_ vm_tbl{}; + JavaVM_ mock_vm{}; + Engine noop_engine; + + jvmtiEnv *orig_jvmti = nullptr; + JavaVM *orig_vm = nullptr; + CodeCache *orig_libjvm = nullptr; + Engine *orig_cpu_engine = nullptr; + Engine *orig_wall_engine = nullptr; + + void SetUp() override { + ReferenceChainsTestAccessor::reset(); + + orig_jvmti = VMTestAccessor::getJvmti(); + jvmti_tbl = jvmtiInterface_1_{}; + jvmti_tbl.SetEventNotificationMode = &mock_SetEventNotificationMode; + jvmti_tbl.GetAvailableProcessors = &mock_GetAvailableProcessors; + jvmti_tbl.GetLoadedClasses = &mock_GetLoadedClasses; + jvmti_tbl.Deallocate = &mock_Deallocate; + mock_jvmti.functions = &jvmti_tbl; + VMTestAccessor::setJvmti(&mock_jvmti); + + orig_vm = VMTestAccessor::getVm(); + vm_tbl = JNIInvokeInterface_{}; + vm_tbl.GetEnv = &mock_GetEnv; + mock_vm.functions = &vm_tbl; + VMTestAccessor::setVm(&mock_vm); + + orig_cpu_engine = Profiler::instance()->cpuEngine(); + orig_wall_engine = Profiler::instance()->wallEngine(); + ProfilerTestAccessor::setCpuEngine(Profiler::instance(), &noop_engine); + ProfilerTestAccessor::setWallEngine(Profiler::instance(), &noop_engine); + + // writeSettings() (flightRecorder.cpp) unconditionally calls + // VM::libjvm()->hasDebugSymbols() - VM::_libjvm is NULL until + // VM::openJvmLibrary() has resolved a real libjvm.so, which never + // happens without a live JVM (VM::libjvm() asserts non-null rather + // than returning NULL). A name-only, no-symbols CodeCache (the same + // "fake shared library" construction libraries_ut.cpp's fixture + // already uses) is enough: hasDebugSymbols() degrades to false for + // it, same as a never-resolved real library with no debug info. + static CodeCache fake_libjvm("fake_libjvm.so"); + orig_libjvm = VMTestAccessor::getLibjvm(); + VMTestAccessor::setLibjvm(&fake_libjvm); + + // VMStructs::libjvm() is a separate cache (VMStructs::_libjvm, only + // populated by VMStructs::init() after a real symbol scan) from + // VM::libjvm() above - nothing in this test's write path reads it, + // but VMStructs::init(CodeCache*) is already public production API + // (hotspot/vmStructs.h) and idempotent (readSymbol() degrades to 0 + // for every unresolved symbol, per its own comment), so initializing + // it here too keeps this fixture consistent with every other _ut.cpp + // that touches this process-wide singleton. + VMStructs::init(&fake_libjvm); + } + + void TearDown() override { + VMTestAccessor::setJvmti(orig_jvmti); + VMTestAccessor::setVm(orig_vm); + VMTestAccessor::setLibjvm(orig_libjvm); + ProfilerTestAccessor::setCpuEngine(Profiler::instance(), orig_cpu_engine); + ProfilerTestAccessor::setWallEngine(Profiler::instance(), orig_wall_engine); + } +}; + +// Path agreed with the companion Java test (ddprof-test's +// ReferenceChainJfrParserTest), which reads the same file back via JMC's +// JfrLoaderToolkit. Both sides resolve it via the OS temp dir so the +// producer (this gtest) and the consumer (the Java test, run afterwards by +// the same operator/CI job on the same machine) agree without either side +// needing to know the other module's build directory layout. +static std::string chainRoundtripJfrPath() { + const char *tmp = getenv("TMPDIR"); + std::string dir = (tmp != nullptr && *tmp != 0) ? tmp : "/tmp"; + if (dir.back() != '/') { + dir += '/'; + } + return dir + "datadog_reference_chain_roundtrip.jfr"; +} + +TEST_F(ReferenceChainJfrRoundtripTest, ProducesValidStandaloneJfrWithChainEvent) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + // Skip OS/CPU info, JVM info, system properties and native library + // enumeration - all of them call further JVMTI/JNI entry points this + // fixture does not mock, and none of them are relevant to the question + // this test answers (whether JMC resolves the chain[] field). + args._jfr_options = JFR_SYNC_OPTS; + + // Recording::writeMetadata() (flightRecorder.cpp) serializes JfrMetadata::root() + // as-is - it does not build it. That tree is normally populated exactly once by + // JfrMetadata::initialize() (jfrMetadata.cpp), called from Profiler::start() + // (profiler.cpp:1433) - which this test does not call (per this file's header + // comment). initialize() is itself public, JVM-independent (pure fluent-builder + // data construction, no JVMTI/JNI calls) and idempotent (_initialized guard, + // jfrMetadata.cpp) - calling it directly here is completing the same + // one-time setup step every real Recording implicitly depends on, not a new + // hook; without it, writeMetadata() would serialize an empty "root" element + // (no datadog.ReferenceChain type declaration at all) instead of the real + // metadata tree. + // JfrMetadata::initialize() populates the static Element/Attribute tree + // (JfrMetadata::_root) exactly once for the life of the process - the same + // one-time, never-freed allocation every real agent process makes via + // Profiler::start() and relies on the OS to reclaim at exit. This gtest + // binary is the only asan unit test that calls initialize() directly, so + // LeakSanitizer flags that intentional process-lifetime allocation as a + // leak on test exit. Disable leak detection for this call only - it is + // not a bug in JfrMetadata or the reference-chain code under test. +#ifdef ASAN_ENABLED + { + __lsan::ScopedDisabler lsan_disabler; + JfrMetadata::initialize(args._context_attributes); + } +#else + JfrMetadata::initialize(args._context_attributes); +#endif + + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // Register three distinct, recognisable class names in the same + // StringDictionary (Profiler::classMap()) writeClasses() (flightRecorder.cpp) + // serializes into the T_CLASS checkpoint - this is the exact API + // Profiler::lookupClass() already-existing tests use (e.g. + // referenceChains_ut.cpp's ReconstructsChainForSyntheticGraph). + int leafKlass = Profiler::instance()->lookupClass( + "com/test/ChainLeaf", strlen("com/test/ChainLeaf")); + int middleKlass = Profiler::instance()->lookupClass( + "com/test/ChainMiddle", strlen("com/test/ChainMiddle")); + int rootKlass = Profiler::instance()->lookupClass( + "com/test/ChainRoot", strlen("com/test/ChainRoot")); + ASSERT_NE(-1, leafKlass); + ASSERT_NE(-1, middleKlass); + ASSERT_NE(-1, rootKlass); + + // writeClasses() only serializes classMap()->standby() (the snapshot + // captured by rotate()) - without this, the three lookupClass() calls + // above would sit in the live "active" buffer only and never reach the + // checkpoint. StringDictionary::rotate() is public production API + // (stringDictionary.h), the same one Profiler::rotateDictsAndRun() calls + // internally for every real dump - calling it directly here is not a + // hand-rolled substitute, just the same operation invoked without the + // rest of Profiler::dump()'s live-JVM-dependent machinery. + Profiler::instance()->classMap()->rotate(); + + // Seed a deterministic leaf(tag=3) <- middle(tag=2) <- root(tag=1) + // parent chain directly into the tracker's own FrontierTable, exactly + // mirroring referenceChains_ut.cpp's ReconstructsChainForSyntheticGraph + // test (which builds the same shape via a scripted heap walk instead of + // direct insertion - both reach the same FrontierTable state that + // buildChainEvent() below reads). + FrontierTable *frontier = tracker->frontierTable(); + ASSERT_NE(nullptr, frontier); + // root_kind=21 (JVMTI_HEAP_REFERENCE_JNI_GLOBAL) on the root-attached + // entry only - exercises the new rootKind field end to end through + // buildChainEvent()/recordReferenceChain(), mirroring how + // heapReferenceCallback() only ever sets it on a parent_tag==0 entry. + // Edge kinds on the interior hops (ARRAY_ELEMENT into the middle node, + // FIELD into the leaf) - with the null JNIEnv below, the field-name + // decode is skipped and each label degrades to its edge KIND, which is + // exactly what the Java-side parser test (ReferenceChainJfrParserTest) + // asserts this recording's new "edges" field contains. + ASSERT_TRUE(frontier->insert(1, 0, (u32)rootKlass, 0, + FrontierEntryState::EDGE, /*root_kind=*/21)); + ASSERT_TRUE(frontier->insert(2, 1, (u32)middleKlass, 1, + FrontierEntryState::EDGE, /*root_kind=*/0, + /*class_tag=*/0, + /*referrer_field_index=*/-1, + JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT)); + ASSERT_TRUE(frontier->insert(3, 2, (u32)leafKlass, 2, + FrontierEntryState::EDGE, /*root_kind=*/0, + /*class_tag=*/0, + /*referrer_field_index=*/-1, + JVMTI_HEAP_REFERENCE_FIELD)); + + ReferenceChainEvent event; + // Null JNIEnv: edge labels degrade to kind labels (fillHopEdgeLabels' + // own guard) - this binary's mock JVMTI table does not stub the name + // resolution slots, so a live decode would crash on them. + ASSERT_TRUE(tracker->buildChainEvent(&mock_jvmti, /*jni=*/nullptr, + /*target_tag=*/3, &event)); + ASSERT_EQ(3u, event._chain.size()); + EXPECT_EQ((u32)leafKlass, event._chain[0]); + EXPECT_EQ((u32)middleKlass, event._chain[1]); + EXPECT_EQ((u32)rootKlass, event._chain[2]); + ASSERT_EQ(3u, event._edges.size()); + EXPECT_STREQ("field", event._edges[0].c_str()); + EXPECT_STREQ("element", event._edges[1].c_str()); + EXPECT_STREQ("jni_global", event._edges[2].c_str()); + EXPECT_EQ(21u, event._root_kind); + event._start_time = TSC::ticks(); + + const std::string path = chainRoundtripJfrPath(); + { + int fd = open(path.c_str(), O_CREAT | O_RDWR | O_TRUNC, 0644); + ASSERT_GE(fd, 0) << "could not open " << path << " for writing"; + + // Recording(fd, args) and recordReferenceChain() are already public + // production API (flightRecorder.h) - this drives the real chunk + // header/metadata/settings write (constructor) and the real + // datadog.ReferenceChain event encoding (recordReferenceChain(), + // flightRecorder.cpp:1937 - the exact F_CPOOL|F_ARRAY `chain` field + // this test exists to answer for), not a hand-rolled byte layout. + Recording rec(fd, args); + Buffer *buf = rec.buffer(/*lock_index=*/0); + rec.recordReferenceChain(buf, &event); + // ~Recording() (end of scope) calls finishChunk(true): flushes buf, + // writes the real class/symbol/package constant-pool checkpoint + // (writeCpool() -> writeClasses(), which is what makes leafKlass/ + // middleKlass/rootKlass resolvable to their names at all), patches + // the chunk header's size/cpool-offset fields, and closes fd - the + // same finalization every real recording chunk goes through. + } + + FILE *f = fopen(path.c_str(), "rb"); + ASSERT_NE(nullptr, f) << "expected " << path << " to have been written"; + char magic[4] = {0, 0, 0, 0}; + size_t read = fread(magic, 1, 4, f); + fseek(f, 0, SEEK_END); + long size = ftell(f); + fclose(f); + + ASSERT_EQ(4u, read); + EXPECT_EQ(0, memcmp(magic, "FLR\0", 4)) + << "produced file does not start with the JFR chunk magic"; + EXPECT_GT(size, 4) << "produced file is empty beyond the magic header"; + + TEST_LOG("Wrote standalone reference-chain roundtrip JFR to %s (%ld bytes)", + path.c_str(), size); +} diff --git a/ddprof-lib/src/test/cpp/referenceChains_ut.cpp b/ddprof-lib/src/test/cpp/referenceChains_ut.cpp new file mode 100644 index 0000000000..7dd6a525a8 --- /dev/null +++ b/ddprof-lib/src/test/cpp/referenceChains_ut.cpp @@ -0,0 +1,7255 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "arguments.h" +#include "counters.h" +#include "livenessTracker.h" +#include "os.h" +#include "profiler.h" +#include "rcDebugLevel.h" +#include "referenceChains.h" +#include "vmEntry.h" +#include "../../main/cpp/gtest_crash_handler.h" +#include +#include +#include + +static constexpr char REFERENCE_CHAINS_TEST_NAME[] = "ReferenceChainsTest"; + +class ReferenceChainsGlobalSetup { +public: + ReferenceChainsGlobalSetup() { + installGtestCrashHandler(); + } + ~ReferenceChainsGlobalSetup() { + restoreDefaultSignalHandlers(); + } +}; + +static ReferenceChainsGlobalSetup global_setup; + +// The rc debug-log gate defaults to silent (0) so DEBUG builds are +// pod-safe (see rcDebugLevel.h). Tests want the full diagnostics that +// used to be unconditional: pin level 2 before any library code runs. +[[maybe_unused]] static const bool kRcDebugLevelPinnedForTests = + setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "2", 1) == 0; + +// --------------------------------------------------------------------------- +// RcDebugLevelTest - the runtime debug-log gate's pure parsing + env +// plumbing (rcDebugLevel.h). No tracker state involved. +// --------------------------------------------------------------------------- +TEST(RcDebugLevelTest, ParseAcceptsTrimmedSingleDigits) { + EXPECT_EQ(parseRcDebugLevel(nullptr), -1); + EXPECT_EQ(parseRcDebugLevel(""), -1); + EXPECT_EQ(parseRcDebugLevel("0"), 0); + EXPECT_EQ(parseRcDebugLevel("1"), 1); + EXPECT_EQ(parseRcDebugLevel("2"), 2); + EXPECT_EQ(parseRcDebugLevel(" 2\n"), 2); // the `echo 2 >` form + EXPECT_EQ(parseRcDebugLevel("\t1\r\n"), 1); + EXPECT_EQ(parseRcDebugLevel("3"), -1); + EXPECT_EQ(parseRcDebugLevel("12"), -1); + EXPECT_EQ(parseRcDebugLevel("-1"), -1); + EXPECT_EQ(parseRcDebugLevel("abc"), -1); + EXPECT_EQ(parseRcDebugLevel("2 garbage"), -1); +} + +TEST(RcDebugLevelTest, ReadFileParsesTrimmedAndRejectsInvalid) { + char path[] = "/tmp/rc_dbg_test_XXXXXX"; + int fd = mkstemp(path); + ASSERT_GE(fd, 0); + close(fd); + struct Case { + const char *content; + int expected; + }; + const Case cases[] = { + {"2\n", 2}, {"1", 1}, {" 2 ", 2}, {"\n1\n", 1}, + {"", -1}, {"3", -1}, {"22", -1}, {"abc", -1}, {"x", -1}, + }; + for (const Case &c : cases) { + FILE *f = fopen(path, "w"); + ASSERT_NE(f, nullptr); + EXPECT_GE(fputs(c.content, f), 0); // non-negative on success + fclose(f); + EXPECT_EQ(readRcDebugLevelFile(path), c.expected) << "content='" << c.content << "'"; + } + unlink(path); + EXPECT_EQ(readRcDebugLevelFile(path), -1); // now missing + EXPECT_EQ(readRcDebugLevelFile(nullptr), -1); +} + +TEST(RcDebugLevelTest, RefreshFollowsEnvWhenNoOverrideFile) { + // The refresh's fixed override path is machine-global; skip rather + // than flake on a developer machine that happens to have the file. + if (access("/tmp/ddprof_root/refchains_debug_level", F_OK) == 0) { + GTEST_SKIP() << "override file present on this machine"; + } + setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "1", 1); + rcDebugLevelRefresh(true); + EXPECT_EQ(rcDebugLevel(), 1); + setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "2", 1); + rcDebugLevelRefresh(true); + EXPECT_EQ(rcDebugLevel(), 2); + // invalid env value means silent, not "keep the previous level" + setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "bogus", 1); + rcDebugLevelRefresh(true); + EXPECT_EQ(rcDebugLevel(), 0); + // restore the pinned level for any later test's diagnostics + setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "2", 1); + rcDebugLevelRefresh(true); + EXPECT_EQ(rcDebugLevel(), 2); +} + +// --------------------------------------------------------------------------- +// VMTestAccessor - friend of VM (vmEntry.h), lets tests swap VM::_jvmti for a +// mock. This gtest binary has no live JVM attached (see jvmSupport_ut.cpp's +// fixture comment for the same constraint on a different subsystem), but +// ReferenceChainTracker::start()/stop() now call VM::jvmti()-> +// SetEventNotificationMode() (the lazy event-enable step), so a mock is +// required for those calls to be exercised without crashing on a null +// jvmtiEnv. +// --------------------------------------------------------------------------- +class VMTestAccessor { +public: + static jvmtiEnv* getJvmti() { return VM::_jvmti; } + static void setJvmti(jvmtiEnv* env) { VM::_jvmti = env; } +}; + +// --------------------------------------------------------------------------- +// ReferenceChainsTestAccessor - same pattern as VMTestAccessor above, for the +// same reason: ReferenceChainTracker::instance() is a process-wide singleton +// (referenceChains.h), so the search-lifecycle fields +// (_search_state/_search_started/_pending_expand/...) would otherwise leak +// from one ReferenceChainsBfsTest TEST_F into the next in this same gtest +// binary - e.g. a test that drives the search to SearchState::COMPLETED +// would leave every later test's runPass() call a permanent no-op (see +// runPass()'s "already terminal -> no-op" branch). reset() puts the tracker +// back to its just-constructed state; it does not change production +// behavior. +// --------------------------------------------------------------------------- +class ReferenceChainsTestAccessor { +public: + static void reset() { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + delete t->_frontier; + t->_frontier = nullptr; + t->_class_tags = ClassTagTable(); + t->_last_resolved_class_count = 0; + // Without these two, a prior test's fully-swept (or + // partially-swept) admitStaticFieldRoots() state survives in this + // process-wide singleton and can wrongly skip the sweep entirely on + // this test's first pass if its resolved class count happens to + // match whatever an earlier test last left behind - see + // resetForRestart()'s identical reset of these same fields for the + // production-restart equivalent of this same contract. + t->_last_static_field_class_count = -1; + t->_static_field_sweep_cursor = 0; + t->_static_field_sweep_cycle_truncated = false; + t->_next_tag = 1; + // Shared with LivenessTracker (classTagAllocator.h) - process-wide, + // not per-ReferenceChainTracker-instance, so it needs its own reset + // seam rather than being a plain member write. + ClassTagAllocator::resetForTest(); + t->_search_started = false; + t->_tags_released = true; + t->_search_state = SearchState::RUNNING; + t->_abandon_reason = SearchAbandonReason::NONE; + t->_search_start_ns = 0; + t->_pending_expand.clear(); + t->_priority_expand.clear(); + t->_priority_expand_set.clear(); + // B' at-risk FIFO + its indexes/counters, and the round-15 fresh + // lane: production clears all of these on every restartSearch()/ + // resetSearchStateForTest(), but this seam predates the FIFO and was + // never given the clears - until round 16 a test left at-risk + // entries behind and only survived because LATER tests' own + // runPass()s drained the residue with the full Bfs mock. A test + // that ends with a saturated FIFO (AtRiskFifoPerClassQuotaKeeps- + // FloodOut) exposed the gap: the residue drained into the NEXT + // fixture's runPass, whose minimal mock has a null + // GetObjectsWithTags (walkStaticFieldAnchors crash). The seam's + // contract is "empty process-wide singleton state for the next + // test" - state everything, assert nothing about prior residue. + t->_static_anchor_fifo.clear(); + t->_static_anchor_fifo_set.clear(); + t->_static_anchor_fifo_klass_counts.clear(); + t->_static_anchor_fifo_pushed = 0; + t->_static_anchor_fifo_quota_drops = 0; + t->_static_anchor_fresh_queue.clear(); + t->_last_pass_gc_finish_epoch = 0; + t->_last_pass_ns = 0; + t->_passes_run = 0; + t->_passes_since_last_progress = 0; + t->_candidate_count = 0; + t->_candidate_found_bits = 0; + memset(t->_candidate_discovered_count, 0, sizeof(t->_candidate_discovered_count)); + t->_passes_since_last_candidate_progress = 0; + t->_last_candidate_progress_mark = 0; + t->_canary_stuck_restart_count = 0; + t->_resolved_chains.clear(); + t->_safepoint_pain_budget = PainBudget(); + t->_cpu_pain_budget = PainBudget(); + t->_search_pain_ms = 0; + t->_root_kind_rotation_cursor = 1; + t->_stale_expanded_rotation_cursor = 1; + t->_static_anchor_index.clear(); + t->_static_anchor_own_class_tags.clear(); + t->_static_anchor_index_tags.clear(); + t->_anchor_container_cursor = 0; + t->_anchor_other_cursor = 0; + t->_class_shape_cache.clear(); + t->_thread_walk_anchor_cursor = 0; + memset(t->_candidate_qualifying_tid_count, 0, + sizeof(t->_candidate_qualifying_tid_count)); + t->_hop_label_cache.clear(); + t->_watched_leak_klass_count = 0; + t->_leak_signature_totals.clear(); + t->_leak_signature_prev_totals.clear(); + t->_leak_parent_fanout.clear(); + t->_borrowed_budget = 0; + t->_consecutive_under_target_passes = 0; + // Adaptive batch + lane state: NOT covered by anything above, and a + // prior test that drove expansion leaves a non-zero EMA, a live + // batch size, a stale pass deadline, and/or a mid-alternation lane + // toggle behind - all of which silently change the next test's + // expandFrontier() arithmetic (exact-value asserts on batch sizing + // only pass standalone otherwise). + t->_gotw_ema_call_ns = 0; + t->_gotw_batch_size = 0; + t->_pass_deadline_ns = 0; + t->_expand_lane_prefer_priority = true; + } + + // Search restart + pain budget (SearchRestartTest below) - same + // rationale as the pacing accessors above: private state a test needs to + // drive/observe directly. + static bool canAffordNewSearch(u64 now_ns) { + return ReferenceChainTracker::instance()->canAffordNewSearch(now_ns); + } + + static bool shouldRunPass(u64 now_ns) { + return ReferenceChainTracker::instance()->shouldRunPass(now_ns); + } + + static void setSearchPainMs(u64 ms) { + ReferenceChainTracker::instance()->_search_pain_ms = ms; + } + + static void setCandidateFrontierTagForTest(int idx, jlong tag) { + ReferenceChainTracker::instance()->setCandidateFrontierTagForTest(idx, tag); + } + + static void setCandidateCountForTest(int n) { + ReferenceChainTracker::instance()->setCandidateCountForTest(n); + } + + // Canary-lane backoff state wrappers - see _canary_backoff_mult's own + // comment (referenceChains.h). + static int canaryBackoffMult() { + return ReferenceChainTracker::instance()->canaryBackoffMultForTest(); + } + static u64 lastCanaryPassNs() { + return ReferenceChainTracker::instance()->lastCanaryPassNsForTest(); + } + static void setCanaryBackoffForTest(int mult, u64 ema_ms, u64 last_pass_ns) { + ReferenceChainTracker::instance()->setCanaryBackoffForTest(mult, ema_ms, + last_pass_ns); + } + static void setOomRampActive(bool active) { + ReferenceChainTracker::instance()->setOomRampActiveForTest(active); + } + + static int passesSinceLastCandidateProgress() { + return ReferenceChainTracker::instance()->passesSinceLastCandidateProgressForTest(); + } + + static int canaryStuckRestartCount() { + return ReferenceChainTracker::instance()->canaryStuckRestartCountForTest(); + } + + static u64 searchPainMs() { + return ReferenceChainTracker::instance()->_search_pain_ms; + } + + // Resolved-chain cache: read-only size peek and a pass-through to the + // private snapshot (drainPendingChainEvents()) and insert + // (cacheResolvedChain()), for ResolvedChainCacheTest below - same + // rationale as hasResolvedChainForTag()/resolvedChainCount() below. + static size_t resolvedChainCount() { + return ReferenceChainTracker::instance()->_resolved_chains.size(); + } + + static void drain(std::vector *out) { + ReferenceChainTracker::instance()->drainPendingChainEvents(out); + } + + static void cacheChain(jlong source_tag, ReferenceChainEvent event, + jlong source_tag_val, u64 source_search_ns) { + ReferenceChainTracker::instance()->cacheResolvedChain( + source_tag, std::move(event), source_tag_val, source_search_ns); + } + + static int maxResolvedChains() { + return ReferenceChainTracker::MAX_RESOLVED_CHAINS; + } + + // Target-selection bridging step: read-only peeks into the resolved-chain + // cache, for asserting exactly which klass a chain was resolved for and + // the tag it was reconstructed from - see PollWatchedTargetsTest below. + static bool hasResolvedChainForTag(jlong tag) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + return t->_resolved_chains.find(tag) != t->_resolved_chains.end(); + } + + static jlong resolvedChainSourceTag(jlong tag) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + auto it = t->_resolved_chains.find(tag); + return it == t->_resolved_chains.end() ? 0 : it->second.source_tag; + } + + // Leak-tag correlation (design C): read a frontier entry's stored leak + // tag, and a pass-through to the private buildChainEvent(), for + // LeakTagInterceptionTest below - same friend-accessor rationale as + // hasResolvedChainForTag() above. Returns -1 when the tag is not in the + // frontier table. + static jlong frontierLeakTag(jlong tag) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + FrontierEntry entry{}; + if (t->_frontier == nullptr || !t->_frontier->lookup(tag, &entry)) { + return -1; + } + return entry.leak_tag; + } + + static void setCandidateKlassIdForTest(int idx, u32 klass_id) { + ReferenceChainTracker::instance()->setCandidateKlassIdForTest(idx, klass_id); + } + + // Round-19: the leak-tag canary found criterion (pod 289f8 — the + // chase was structurally unresolvable after the marker->leak-tag + // migration; see buildDiscoveredInstanceChains' own comment). + static u64 candidateFoundBitsForTest() { + return ReferenceChainTracker::instance()->_candidate_found_bits; + } + + static jlong candidateFrontierTagForTest(int slot) { + return ReferenceChainTracker::instance()->_candidate_frontier_tags[slot]; + } + + static void buildDiscoveredInstanceChainsForTest(u32 klass_id, + u64 current_search_ns) { + // jvmti/jni null is safe: resolveHopEdgeLabel() null-guards and + // degrades hop labels to kind labels. + ReferenceChainTracker::instance()->buildDiscoveredInstanceChains( + nullptr, nullptr, klass_id, current_search_ns); + } + + // ---- pod-in-a-jar system harness (design-pod-in-a-jar-harness) ---- + static u8 searchStateForTest() { + return ReferenceChainTracker::instance()->_search_state; + } + + static int sweepGateResolvedCountForTest() { + return ReferenceChainTracker::instance()->_last_resolved_class_count; + } + + static int sweepGateStaticCountForTest() { + return ReferenceChainTracker::instance()->_last_static_field_class_count; + } + + static int sweepCursorForTest() { + return ReferenceChainTracker::instance()->_static_field_sweep_cursor; + } + + static int passesRunForTest() { + return ReferenceChainTracker::instance()->_passes_run; + } + + static int candidateCountForTest() { + return ReferenceChainTracker::instance()->_candidate_count; + } + + static u32 candidateKlassIdForTest(int slot) { + return ReferenceChainTracker::instance()->_candidate_klass_ids[slot]; + } + + static size_t resolvedChainCountForTest() { + return ReferenceChainTracker::instance()->_resolved_chains.size(); + } + + static std::vector resolvedChainTargetsForTest() { + std::vector out; + for (auto &kv : + ReferenceChainTracker::instance()->_resolved_chains) { + out.push_back(kv.second.event._target_tag); + } + return out; + } + + static size_t staticAnchorFreshQueueSizeForTest() { + return ReferenceChainTracker::instance() + ->_static_anchor_fresh_queue.size(); + } + + static jlong candidateDiscoveredTagForTest(int slot, int idx) { + return ReferenceChainTracker::instance()->candidateDiscoveredTagForTest(slot, idx); + } + + static int candidateDiscoveredCountForTest(int slot) { + return ReferenceChainTracker::instance()->candidateDiscoveredCountForTest(slot); + } + + // recordDiscoveredInstance()/correlateAdmittedLeakTag() are the + // production paths for the leak-correlation tests below. + static void recordDiscoveredInstanceForTest(u32 klass_id, jlong tag, + bool leak_correlated) { + ReferenceChainTracker::instance()->recordDiscoveredInstance(klass_id, tag, + leak_correlated); + } + + // Drive restartSearch() directly (the accessor base already set + // _tags_released, so its assert is satisfied). + static void restartSearchForTest() { + ReferenceChainTracker::instance()->restartSearch(); + } + + static bool anchorIndexIsEmptyForTest() { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + return t->_static_anchor_index.empty() && + t->_static_anchor_own_class_tags.empty() && + t->_static_anchor_index_tags.empty(); + } + + // Read back discovered-instance slots (frontier tags recorded by + // recordDiscoveredInstance). + static jlong discoveredTagForTest(int slot, int idx) { + return ReferenceChainTracker::instance() + ->_candidate_discovered_tags[slot][idx]; + } + + static int discoveredCountForTest(int slot) { + return ReferenceChainTracker::instance() + ->_candidate_discovered_count[slot]; + } + + static size_t priorityExpandCap() { + return ReferenceChainTracker::PRIORITY_EXPAND_CAP; + } + + static int maxDiscoveredPerClass() { + return ReferenceChainTracker::MAX_DISCOVERED_INSTANCES_PER_CLASS; + } + + static bool buildChainEventForTest(jvmtiEnv *jvmti, JNIEnv *jni, + jlong tag, ReferenceChainEvent *out) { + return ReferenceChainTracker::instance()->buildChainEvent(jvmti, jni, + tag, out); + } + + // Direct expandFrontier() drive for the AIMD batch test: a full runPass() + // drains a small graph to completion and its rotation phase adds extra + // GetObjectsWithTags calls, so per-call AIMD assertions cannot be made + // deterministic through runPass(). Seeding _pending_expand and calling + // expandFrontier() directly runs exactly one batch (one AIMD update). + static void pushPendingExpandForTest(jlong tag) { + ReferenceChainTracker::instance()->_pending_expand.push_back(tag); + } + + static void expandFrontierForTest(jvmtiEnv *jvmti, JNIEnv *jni, + int *edges_admitted) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + bool truncated = false; + bool cap_hit = false; + u64 safepoint_ticks = 0; + t->expandFrontier(jvmti, jni, t->_hop_cap, 1000, edges_admitted, + &truncated, &cap_hit, &safepoint_ticks); + } + + // Pause-time pacing controller: read-only peeks at the controller's + // derived values, and + // a pass-through to the private updatePacing() itself, for + // ReferenceChainsPacingTest below - same rationale as + // hasResolvedChainForTag()/resolvedChainCount() above (the target-selection bridging step): private state a test needs to drive/ + // observe directly, exposed via this existing friend accessor rather + // than adding public getters/setters to ReferenceChainTracker itself. + static int effectiveBudget() { + return ReferenceChainTracker::instance()->_effective_budget; + } + + static u64 effectiveCadenceNs() { + return ReferenceChainTracker::instance()->_effective_cadence_ns; + } + + static void updatePacing(u64 pass_wall_ns) { + ReferenceChainTracker::instance()->updatePacing(pass_wall_ns); + } + + static u64 baselineCadenceNs() { return ReferenceChainTracker::PASS_CADENCE_NS; } + + // Test-only seams for PacingGrowsBudgetBackAndRelaxesCadenceWhenUnderCeiling + // below, which needs to start from a controlled below-ceiling/above- + // baseline point with a freshly reset controller (see that test's own + // comment for why chaining directly off a prior constant-input sequence + // would leave _pause_pid's integral state mid-recovery from that + // sequence's windup, muddying this method's per-step direction + // assertions with a transient the test is not about). + static void setEffectiveBudget(int v) { + ReferenceChainTracker::instance()->_effective_budget = v; + } + + static void setEffectiveCadenceNs(u64 v) { + ReferenceChainTracker::instance()->_effective_cadence_ns = v; + } + + static void resetPacingController() { + ReferenceChainTracker::instance()->_pause_pid.reset(); + } + + // Budget-borrowing (referenceChains.h's _borrowed_budget comment): the + // configured multiplier PacingGrowsBudgetBackAndRelaxesCadenceWhenUnderCeiling + // below asserts convergence against, instead of hardcoding it a second + // time in the test itself. + static int borrowCeilingMultiplier() { + return ReferenceChainTracker::BORROW_CEILING_MULTIPLIER; + } + + static int64_t borrowedBudget() { + return ReferenceChainTracker::instance()->_borrowed_budget; + } + + // MaybeRevokeBorrowForRootEnumPass* tests below: drive the borrow state + // directly into "already granted" before exercising the revocation-only + // seam, and a pass-through to that seam itself - same rationale as + // updatePacing()'s own accessor above. + static void setBorrowedBudget(int64_t v) { + ReferenceChainTracker::instance()->_borrowed_budget = v; + } + + static int consecutiveUnderTargetPasses() { + return ReferenceChainTracker::instance()->_consecutive_under_target_passes; + } + + static void setConsecutiveUnderTargetPasses(int v) { + ReferenceChainTracker::instance()->_consecutive_under_target_passes = v; + } + + static void maybeRevokeBorrowForRootEnumPass(u64 pass_wall_ticks) { + ReferenceChainTracker::instance()->maybeRevokeBorrowForRootEnumPass( + pass_wall_ticks); + } + + // ReleaseSearchTagsFailureTest below: read-only peek at whether the + // tracker still owes a tag release before it can allow a restart - see + // _tags_released's own comment. + static bool tagsReleased() { + return ReferenceChainTracker::instance()->_tags_released; + } + + // ResolveLoadedClassesRescansAfterClassCountShrinksAndPartiallyRegrows + // below: direct pass-through to the private resolveLoadedClasses(), plus + // a read-only peek at the count it stashes - the same rationale as + // tagsReleased() above (private state/behavior a test needs to + // drive/observe directly, without going through a full runPass()/search + // lifecycle that resolveLoadedClasses() alone does not need). + static void resolveLoadedClasses(jvmtiEnv *jvmti, JNIEnv *jni) { + ReferenceChainTracker::instance()->resolveLoadedClasses(jvmti, jni); + } + + static int lastResolvedClassCount() { + return ReferenceChainTracker::instance()->_last_resolved_class_count; + } + + // Phase 5 (durability re-verification) test seams: direct pass-throughs + // to the private tie-break/rotation methods, plus FrontierTable::insert() + // itself (also private-by-convention here in the sense that production + // code only ever calls it via admitObject()) so tests can set up a + // frontier entry's exact starting root_kind/state/parent_tag without + // needing a live JVMTI mock for IterateOverReachableObjects/FollowReferences + // (neither is mocked in this file - see the file header's FollowReferences- + // only mock rationale). + static bool insertFrontierEntry(FrontierTable *frontier, jlong tag, + jlong parent_tag, u32 depth, u8 state, + u8 root_kind, u32 referrer_klass = 0, + jlong class_tag = 0, + jint referrer_field_index = -1, + u8 edge_kind = 0, + jlong referrer_class_tag = 0) { + return frontier->insert(tag, parent_tag, referrer_klass, depth, + state, root_kind, class_tag, + referrer_field_index, edge_kind, + referrer_class_tag); + } + + static bool maybeUpgradeRootAttachedRootKind(FrontierTable *frontier, + jlong tag, + u8 new_root_kind) { + return ReferenceChainTracker::instance() + ->maybeUpgradeRootAttachedRootKind(frontier, tag, new_root_kind); + } + + static std::vector collectStaleRootKindEntriesForRotation( + int max_count) { + return ReferenceChainTracker::instance() + ->collectStaleRootKindEntriesForRotation(max_count); + } + + static std::vector collectStaleExpandedEntriesForRotation( + int max_count) { + return ReferenceChainTracker::instance() + ->collectStaleExpandedEntriesForRotation(max_count); + } + + // Candidate-scoped reach (descendFromAnchor()/walkCandidateThreadLocals()/ + // walkStaticFieldAnchors()): direct drives for the same reason as + // expandFrontierForTest() above - a full runPass() drains a small graph + // to completion and its other phases add interference, so the walk + // phases are exercised on their own. + static void walkCandidateThreadLocalsForTest(jvmtiEnv *jvmti, JNIEnv *jni, + int budget, + int *edges_admitted) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + bool truncated = false; + bool cap_hit = false; + u64 safepoint_ticks = 0; + t->walkCandidateThreadLocals(jvmti, jni, budget, edges_admitted, + &truncated, &cap_hit, &safepoint_ticks); + } + + static void walkStaticFieldAnchorsForTest(jvmtiEnv *jvmti, JNIEnv *jni, + const std::vector &tags, + int budget, + int *edges_admitted) { + walkStaticAnchorFifoForTest(jvmti, jni, tags, budget, edges_admitted, + nullptr); + } + + static std::vector + collectStaticFieldAnchorsForRotationForTest(int max_count) { + return ReferenceChainTracker::instance() + ->collectStaticFieldAnchorsForRotation(max_count); + } + + static void addToStaticAnchorIndexForTest(jlong tag, jlong own_class_tag, + u8 root_kind) { + ReferenceChainTracker::instance() + ->addToStaticAnchorIndex(tag, own_class_tag, root_kind); + } + + // Prime the class-shape cache as if reconcileAnchorClassShapes() had + // classified `class_tag` (tests script shapes instead of driving the + // JNI interface walk, which needs real classes). + static void primeClassShapeForTest(jlong class_tag, bool container) { + ReferenceChainTracker::instance()->_class_shape_cache[class_tag] = + container + ? (u8)ReferenceChainTracker::AnchorClassShape::CONTAINER + : (u8)ReferenceChainTracker::AnchorClassShape::NON_CONTAINER; + } + + // B' at-risk static-anchor FIFO (see _static_anchor_fifo's declaration + // comment in referenceChains.h). pushStaticAnchorFifoForTest bypasses + // heapReferenceCallback's push predicate (kind == STATIC_FIELD onto a + // chain-attached entry) by calling the real pushAtRiskStaticAnchor() + // directly - the end-to-end push path is covered by + // AtRiskStaticHolderFeedsAnchorFifo below; the drive below exists so the + // drain/walk/requeue mechanics (and the round-16 per-class quota) can + // be exercised deterministically on seeded entries. + // Private nested names surfaced for TEST BODIES (friendship covers this + // class's own scope only, so test code cannot name them directly): + using AtRiskAnchor = ReferenceChainTracker::AtRiskAnchor; + static constexpr u32 kAtRiskPerKlassCap = + ReferenceChainTracker::STATIC_ANCHOR_ATRISK_PER_KLASS_CAP; + + static void pushStaticAnchorFifoForTest(jlong tag, u32 klass_id) { + ReferenceChainTracker::instance()->pushAtRiskStaticAnchor(tag, + klass_id); + } + + static int drainStaticAnchorFifoForTest( + int max_count, + std::vector &out) { + return ReferenceChainTracker::instance()->drainStaticAnchorFifo( + max_count, out); + } + + static void requeueStaticAnchorFifoFrontForTest( + const std::vector &entries) { + ReferenceChainTracker::instance()->requeueStaticAnchorFifoFront( + entries); + } + + static size_t staticAnchorFifoSizeForTest() { + return ReferenceChainTracker::instance()->_static_anchor_fifo.size(); + } + + static bool staticAnchorFifoContainsForTest(jlong tag) { + return ReferenceChainTracker::instance() + ->_static_anchor_fifo_set.contains(tag); + } + + static u64 staticAnchorFifoPushedForTest() { + return ReferenceChainTracker::instance()->_static_anchor_fifo_pushed; + } + + static u64 staticAnchorFifoQuotaDropsForTest() { + return ReferenceChainTracker::instance() + ->_static_anchor_fifo_quota_drops; + } + + static u64 selfEdgeGuardSkipsForTest() { + return ReferenceChainTracker::instance() + ->frontierTable() + ->selfEdgeGuardSkips(); + } + + static void walkStaticAnchorFifoForTest(jvmtiEnv *jvmti, JNIEnv *jni, + const std::vector &tags, + int budget, int *edges_admitted, + std::vector *unwalked) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + bool truncated = false; + bool cap_hit = false; + u64 safepoint_ticks = 0; + t->walkStaticFieldAnchors(jvmti, jni, tags, budget, edges_admitted, + &truncated, &cap_hit, &safepoint_ticks, + unwalked); + } + + // Direct candidate-slot seeding (the production path fills these via + // pollWatchedTargets()'s snapshot loop - see _candidate_qualifying_tids' + // own comment): the walk phase tests need exactly one (slot, klass, tid) + // combination without driving LivenessTracker's hysteresis machinery. + static void seedCandidateSlotForTest(int slot, u32 klass_id, + const jint *tids, int tid_count) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + t->_candidate_klass_ids[slot] = klass_id; + for (int i = 0; i < tid_count; i++) { + t->_candidate_qualifying_tids[slot][i] = tids[i]; + } + t->_candidate_qualifying_tid_count[slot] = tid_count; + if (slot + 1 > t->_candidate_count) { + t->_candidate_count = slot + 1; + } + } + + static int candidateQualifyingTidCountForTest(int slot) { + return ReferenceChainTracker::instance() + ->_candidate_qualifying_tid_count[slot]; + } + + static jlong getTagForTest(jvmtiEnv *jvmti, jobject obj) { + return ReferenceChainTracker::instance()->getTag(jvmti, obj); + } + + // Snapshot of _priority_expand's current contents, in queue order - used + // by tests to check for duplicate tags after both rotation collectors + // have run against it within the same simulated pass. + static std::vector priorityExpandContents() { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + return std::vector(t->_priority_expand.begin(), + t->_priority_expand.end()); + } + + // StaleExpandedRotationSkipsPreexistingQueueEntries below: simulates a + // tag left in _priority_expand by a prior pass's truncated expandFrontier() + // batch (expandFrontier()'s own "leave the batch at the front of the + // source queue for a later pass to retry" comment) without driving a full + // expandFrontier()/JVMTI round-trip to produce one. + static void pushPriorityExpand(jlong tag) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + t->_priority_expand.push_back(tag); + t->_priority_expand_set.insert(tag); + } + + // Simulates expandFrontier() having fully drained _priority_expand at the + // end of a pass (the common case: rotation's whole selection fit within + // that pass's rotation_budget slice) - see + // StaleExpandedRotationStarvesHighTagEntryBehindLowTagPopulation below, + // which needs this to model collectStaleExpandedEntriesForRotation() + // being called fresh on each of several simulated passes, the way + // runPassManualWalk() actually does it once per real pass. + static void clearPriorityExpand() { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + t->_priority_expand.clear(); + t->_priority_expand_set.clear(); + } + + static void setRootKindRotationCursor(jlong tag) { + ReferenceChainTracker::instance()->_root_kind_rotation_cursor = tag; + } + + static jlong rootKindRotationCursor() { + return ReferenceChainTracker::instance()->_root_kind_rotation_cursor; + } + + static int rootKindRotationBudget() { + return ReferenceChainTracker::ROOT_KIND_ROTATION_BUDGET; + } + + static int staleExpandedRotationBudget() { + return ReferenceChainTracker::STALE_EXPANDED_ROTATION_BUDGET; + } + + static size_t priorityExpandSize() { + return ReferenceChainTracker::instance()->_priority_expand.size(); + } + + // Snapshot of _pending_expand's current contents, in queue order - used + // by the rolling-resume smoke test to verify that a truncated + // expandFrontier() batch pops fully-processed entries (mark EXPANDED) + // and leaves only the partially-processed and unvisited entries at the + // front of the queue for the next pass to retry. + static std::vector pendingExpandContents() { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + return std::vector(t->_pending_expand.begin(), + t->_pending_expand.end()); + } + + static size_t pendingExpandSize() { + return ReferenceChainTracker::instance()->_pending_expand.size(); + } + + // Self-calibrating adaptive batch size (AIMD): read/write the per-call + // EMA and live batch size so tests can verify the AIMD dynamics. + static u64 gotwEmaCallNs() { + return ReferenceChainTracker::instance()->_gotw_ema_call_ns; + } + + static void setGotwEmaCallNs(u64 v) { + ReferenceChainTracker::instance()->_gotw_ema_call_ns = v; + } + + static size_t gotwBatchSize() { + return ReferenceChainTracker::instance()->_gotw_batch_size; + } + + static void setGotwBatchSize(size_t v) { + ReferenceChainTracker::instance()->_gotw_batch_size = v; + } + + // Read-only peeks at the batch-control constants (private statics - + // friendship applies inside this class's methods, not in test bodies). + static u64 gotwCpuBudgetNs() { + return ReferenceChainTracker::GOTW_CPU_BUDGET_NS; + } + + static size_t gotwInitialBatchSize() { + return (size_t)ReferenceChainTracker::GOTW_INITIAL_BATCH_SIZE; + } + + static size_t gotwMinBatch() { + return ReferenceChainTracker::GOTW_MIN_BATCH; + } + + static size_t gotwMaxBatch() { + return ReferenceChainTracker::GOTW_MAX_BATCH; + } + + static size_t gotwBacklogMinDepth() { + return ReferenceChainTracker::GOTW_BACKLOG_MIN_DEPTH; + } + + static u64 gotwBacklogWindowMult() { + return ReferenceChainTracker::GOTW_BACKLOG_WINDOW_MULT; + } + + // gotwWindowNs() is a pure function of (remaining window, lane depth) + // and the seeded EMA - directly unit-testable without a mock JVMTI call. + static u64 gotwWindowNs(u64 remaining_ns, size_t lane_depth) { + return ReferenceChainTracker::instance()->gotwWindowNs(remaining_ns, + lane_depth); + } + + static void setPassDeadlineNs(u64 v) { + ReferenceChainTracker::instance()->_pass_deadline_ns = v; + } + + static bool expandLanePreferPriority() { + return ReferenceChainTracker::instance()->_expand_lane_prefer_priority; + } + + // Leak-tag pool range base (private static) - same friend-access + // rationale as the AIMD constants above. + static jlong leakTagBase() { + return ReferenceChainTracker::LEAK_TAG_BASE; + } + + // Leak-accumulation rotation test seams (collectLeakAccumulationCandidatesForRotation()). + static void setWatchedLeakKlassIdsForTest(const std::vector &ids) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + int n = (int)std::min(ids.size(), + (size_t)ReferenceChainTracker::MAX_WATCHED_LEAK_KLASSES); + for (int i = 0; i < n; i++) { + t->_watched_leak_klass_ids[i] = ids[i]; + } + t->_watched_leak_klass_count = n; + } + + static void trackLeakAccumulation(FrontierTable *frontier, u32 referrer_klass, + jlong parent_tag, jlong tag) { + ReferenceChainTracker::instance()->trackLeakAccumulation( + frontier, referrer_klass, parent_tag, tag); + } + + static std::vector collectLeakAccumulationCandidatesForRotation( + int max_count) { + return ReferenceChainTracker::instance() + ->collectLeakAccumulationCandidatesForRotation(max_count); + } + + static int leakAccumulationRotationBudget() { + return ReferenceChainTracker::LEAK_ACCUMULATION_ROTATION_BUDGET; + } + + static u32 leakSignatureTotal(u32 leaf_klass_id, u32 parent_class_id) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + u64 key = t->leakSignatureKey(leaf_klass_id, parent_class_id); + auto it = t->_leak_signature_totals.find(key); + return it != t->_leak_signature_totals.end() ? it->second : 0; + } + + static u32 leakParentFanout(jlong parent_tag) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + auto it = t->_leak_parent_fanout.find(parent_tag); + return it != t->_leak_parent_fanout.end() ? it->second.fanout : 0; + } + + static size_t leakSignatureCount() { + return ReferenceChainTracker::instance()->_leak_signature_totals.size(); + } + + static void seedLeakAccumulationForNewlyWatchedKlass(u32 klass_id) { + ReferenceChainTracker::instance() + ->seedLeakAccumulationForNewlyWatchedKlass(klass_id); + } +}; + +static jvmtiError JNICALL mock_SetEventNotificationMode(jvmtiEnv *, jvmtiEventMode, + jvmtiEvent, jthread, ...) { + return JVMTI_ERROR_NONE; +} + +class ReferenceChainsTest : public ::testing::Test { +protected: + jvmtiInterface_1_ tbl{}; + _jvmtiEnv mock_env{}; + jvmtiEnv *orig_jvmti = nullptr; + + void SetUp() override { + orig_jvmti = VMTestAccessor::getJvmti(); + tbl = jvmtiInterface_1_{}; + tbl.SetEventNotificationMode = &mock_SetEventNotificationMode; + mock_env.functions = &tbl; + VMTestAccessor::setJvmti(&mock_env); + } + + void TearDown() override { + VMTestAccessor::setJvmti(orig_jvmti); + } +}; + +TEST_F(ReferenceChainsTest, DefaultDisabled) { + Arguments args; + EXPECT_FALSE(args._reference_chains); +} + +TEST_F(ReferenceChainsTest, FlagParsesEnabled) { + Arguments args; + Error error = args.parse("referencechains=true"); + EXPECT_FALSE(error); + EXPECT_TRUE(args._reference_chains); +} + +TEST_F(ReferenceChainsTest, FlagParsesDisabled) { + Arguments args; + Error error = args.parse("referencechains=false"); + EXPECT_FALSE(error); + EXPECT_FALSE(args._reference_chains); +} + +TEST_F(ReferenceChainsTest, FlagParsesSubOptions) { + Arguments args; + Error error = args.parse("referencechains=true:hops=64:budget=2000:ttl=5000:framecap=128"); + EXPECT_FALSE(error); + EXPECT_TRUE(args._reference_chains); + EXPECT_EQ(64, args._reference_chains_hop_cap); + EXPECT_EQ(2000, args._reference_chains_budget); + EXPECT_EQ(5000, args._reference_chains_ttl_ms); + EXPECT_EQ(128, args._reference_chains_frontier_cap); +} + +// Negative/out-of-range sub-options must be floored/clamped at the parse +// boundary (Arguments::parse(), arguments.cpp) rather than stored verbatim - +// see that call site's own comment for why an unclamped negative hops in +// particular is dangerous: `depth >= (u32)ctx->hop_cap` (referenceChains.cpp) +// casts a negative int to u32, wrapping to ~4e9 and silently disabling the +// hop cap entirely. +TEST_F(ReferenceChainsTest, FlagClampsNegativeSubOptions) { + Arguments args; + Error error = args.parse( + "referencechains=true:hops=-1:budget=-5:ttl=-1:framecap=-3:" + "pausetarget=-1:painbudget=-10"); + EXPECT_FALSE(error); + EXPECT_TRUE(args._reference_chains); + // Floored to a sane minimum (1), not left negative - a negative value + // cast to u32 downstream would otherwise wrap to a huge positive number. + EXPECT_GT(args._reference_chains_hop_cap, 0); + EXPECT_GT(args._reference_chains_budget, 0); + EXPECT_GT(args._reference_chains_frontier_cap, 0); + // ttl/pausetarget are floored at 0 (their own downstream gates already + // treat 0 as "disabled", so 0 - not 1 - is the correct floor). + EXPECT_GE(args._reference_chains_ttl_ms, 0); + EXPECT_GE(args._reference_chains_pause_target_ms, 0); + // painbudget is a percentage - clamped into [0, 100]. + EXPECT_GE(args._reference_chains_pain_budget_percent, 0); + EXPECT_LE(args._reference_chains_pain_budget_percent, 100); +} + +// A too-large painbudget must be clamped down to 100, not stored verbatim - +// the sibling of FlagClampsNegativeSubOptions above, for the upper bound +// rather than the lower one. +TEST_F(ReferenceChainsTest, FlagClampsOversizedPainBudgetPercent) { + Arguments args; + Error error = args.parse("referencechains=true:painbudget=250"); + EXPECT_FALSE(error); + EXPECT_EQ(100, args._reference_chains_pain_budget_percent); +} + +TEST_F(ReferenceChainsTest, FlagWithOtherArgsDoesNotClobberOuterParse) { + Arguments args; + Error error = args.parse("event=cpu,referencechains=true:hops=32,interval=1000000"); + EXPECT_FALSE(error); + EXPECT_TRUE(args._reference_chains); + EXPECT_EQ(32, args._reference_chains_hop_cap); + EXPECT_STREQ("cpu", args._event); + EXPECT_EQ(1000000, args._interval); +} + +TEST_F(ReferenceChainsTest, StartStopDisabledDoesNotCrash) { + Arguments args; + Error error = args.parse("referencechains=false"); + ASSERT_FALSE(error); + + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + Error startError = tracker->start(args); + EXPECT_FALSE(startError); + EXPECT_FALSE(tracker->enabled()); + tracker->stop(); +} + +TEST_F(ReferenceChainsTest, StartStopEnabledDoesNotCrash) { + Arguments args; + Error error = args.parse("referencechains=true:hops=10:budget=100"); + ASSERT_FALSE(error); + + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + Error startError = tracker->start(args); + EXPECT_FALSE(startError); + EXPECT_TRUE(tracker->enabled()); + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// GC signal (GarbageCollectionStart/Finish -> epoch counters). +// +// The callback trampolines (ReferenceChainTracker::GarbageCollectionStart/ +// Finish) ignore the jvmtiEnv* argument entirely - onGCStart()/onGCFinish() +// only bump an atomic counter, per the JVMTI spec restriction documented in +// referenceChains.h - so passing nullptr here exercises the real production +// code path. +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsTest, GCCallbacksIncrementEpochWhenEnabled) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + u64 startBefore = tracker->gcStartEpoch(); + u64 finishBefore = tracker->gcFinishEpoch(); + + ReferenceChainTracker::GarbageCollectionStart(nullptr); + ReferenceChainTracker::GarbageCollectionFinish(nullptr); + + EXPECT_EQ(startBefore + 1, tracker->gcStartEpoch()); + EXPECT_EQ(finishBefore + 1, tracker->gcFinishEpoch()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsTest, GCCallbacksAreNoOpWhenDisabled) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=false")); + + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + ASSERT_FALSE(tracker->enabled()); + + u64 startBefore = tracker->gcStartEpoch(); + u64 finishBefore = tracker->gcFinishEpoch(); + + ReferenceChainTracker::GarbageCollectionStart(nullptr); + ReferenceChainTracker::GarbageCollectionFinish(nullptr); + + EXPECT_EQ(startBefore, tracker->gcStartEpoch()); + EXPECT_EQ(finishBefore, tracker->gcFinishEpoch()); +} + +// --------------------------------------------------------------------------- +// Tag round-trip (SetTag/GetTag/clear). +// +// The implementation plan's suggested test ("allocate an object, tag it, +// force a GC, confirm the tag is still readable via GetObjectsWithTags") +// assumes a live embedded JVM. This gtest binary has no live JVM attached +// (see jvmSupport_ut.cpp's fixture comment for the same constraint on a +// different subsystem), so - following this repo's established pattern for +// testing JVMTI call sites without a real JVM (objectSampler_ut.cpp's mock +// jvmtiInterface_1_ table) - these tests exercise tagObject()/getTag()/ +// clearTag() against a mock jvmtiEnv backed by an in-memory tag map, rather +// than a real GC. This proves the SetTag/GetTag/SetTag(obj,0) call sequence +// and unique-tag allocation are correct; it does not prove GC-move- +// transparency, which requires a real collector and is out of reach of this +// native-only gtest binary. +// --------------------------------------------------------------------------- + +class ReferenceChainsTagTest : public ::testing::Test { +protected: + jvmtiInterface_1_ tbl{}; + _jvmtiEnv mock_env{}; + std::unordered_map tags; + + static ReferenceChainsTagTest *active_fixture; + + void SetUp() override { + active_fixture = this; + tbl = jvmtiInterface_1_{}; + tbl.SetTag = &mock_SetTag; + tbl.GetTag = &mock_GetTag; + mock_env.functions = &tbl; + } + + void TearDown() override { + active_fixture = nullptr; + } + + static jvmtiError JNICALL mock_SetTag(jvmtiEnv *, jobject object, jlong tag) { + if (tag == 0) { + active_fixture->tags.erase(object); + } else { + active_fixture->tags[object] = tag; + } + return JVMTI_ERROR_NONE; + } + + static jvmtiError JNICALL mock_GetTag(jvmtiEnv *, jobject object, jlong *tag_ptr) { + auto it = active_fixture->tags.find(object); + *tag_ptr = it != active_fixture->tags.end() ? it->second : 0; + return JVMTI_ERROR_NONE; + } +}; + +ReferenceChainsTagTest *ReferenceChainsTagTest::active_fixture = nullptr; + +TEST_F(ReferenceChainsTagTest, TagRoundTripsThenClears) { + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + + jlong tag = tracker->tagObject(&mock_env, obj); + EXPECT_NE(0, tag); + EXPECT_EQ(tag, tracker->getTag(&mock_env, obj)); + + tracker->clearTag(&mock_env, obj); + EXPECT_EQ(0, tracker->getTag(&mock_env, obj)); +} + +TEST_F(ReferenceChainsTagTest, TagsAreUniqueAndNeverZero) { + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + int a = 0, b = 0; + jlong tagA = tracker->tagObject(&mock_env, reinterpret_cast(&a)); + jlong tagB = tracker->tagObject(&mock_env, reinterpret_cast(&b)); + + EXPECT_NE(0, tagA); + EXPECT_NE(0, tagB); + EXPECT_NE(tagA, tagB); +} + +TEST_F(ReferenceChainsTagTest, UntaggedObjectReadsBackZero) { + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + int untagged = 0; + EXPECT_EQ(0, tracker->getTag(&mock_env, reinterpret_cast(&untagged))); +} + +// --------------------------------------------------------------------------- +// FrontierTable (tag-indexed frontier metadata table). +// +// No live JVM/JVMTI involvement here - FrontierTable is pure native slot +// storage indexed by an already-issued tag value, so these tests exercise it +// directly rather than through ReferenceChainTracker's tag helpers. +// --------------------------------------------------------------------------- + +TEST(FrontierTableTest, InsertThenLookupRoundTrips) { + FrontierTable table(64); + + ASSERT_TRUE(table.insert(1, /*parent_tag=*/0, /*referrer_klass=*/7, + /*depth=*/0, FrontierEntryState::FRONTIER)); + + FrontierEntry entry{}; + ASSERT_TRUE(table.lookup(1, &entry)); + EXPECT_EQ(0, entry.parent_tag); + EXPECT_EQ(7u, entry.referrer_klass); + EXPECT_EQ(0u, entry.depth); + EXPECT_EQ(FrontierEntryState::FRONTIER, entry.state); +} + +TEST(FrontierTableTest, LookupOfNeverInsertedTagFails) { + FrontierTable table(64); + FrontierEntry entry{}; + EXPECT_FALSE(table.lookup(1, &entry)); + EXPECT_FALSE(table.lookup(5, &entry)); +} + +TEST(FrontierTableTest, NonPositiveTagIsRejected) { + FrontierTable table(64); + FrontierEntry entry{}; + EXPECT_FALSE(table.insert(0, 0, 0, 0)); + EXPECT_FALSE(table.insert(-1, 0, 0, 0)); + EXPECT_FALSE(table.lookup(0, &entry)); + EXPECT_FALSE(table.lookup(-1, &entry)); +} + +TEST(FrontierTableTest, LookupLockedRejectsNonPositiveTag) { + FrontierTable table(64); + ASSERT_TRUE(table.insert(1, 0, 7, 0, FrontierEntryState::FRONTIER)); + FrontierEntry entry{}; + // tag=0 must be rejected the same way lookup() rejects it - callers + // index the table with tag-1, so a `tag <= 0` check (not just `tag < 0`) + // is required to keep that subtraction from wrapping into a valid slot. + EXPECT_FALSE(table.lookupLocked(0, &entry)); + EXPECT_FALSE(table.lookupLocked(-1, &entry)); +} + +TEST(FrontierTableTest, LookupLockedRejectsTagPastCurrentSize) { + FrontierTable table(64); + ASSERT_TRUE(table.insert(1, 0, 7, 0, FrontierEntryState::FRONTIER)); + FrontierEntry entry{}; + // Only tag=1 has ever been inserted (table_size == 1); tag=2 maps to + // idx=1, exactly at the current size boundary, and must be rejected + // rather than read out of bounds. + EXPECT_FALSE(table.lookupLocked(2, &entry)); +} + +TEST(FrontierTableTest, ParentTagChainReconstructsAcrossHops) { + // Mirrors how the heap-walk engine walks parent_tag links back to a root: insert + // a small chain root(tag=1) <- mid(tag=2) <- leaf(tag=3) and confirm the + // links resolve in order. + FrontierTable table(64); + ASSERT_TRUE(table.insert(1, 0, 100, 0, FrontierEntryState::EDGE)); + ASSERT_TRUE(table.insert(2, 1, 200, 1, FrontierEntryState::EDGE)); + ASSERT_TRUE(table.insert(3, 2, 300, 2, FrontierEntryState::EDGE)); + + FrontierEntry entry{}; + jlong tag = 3; + std::vector chain; + while (tag != 0) { + ASSERT_TRUE(table.lookup(tag, &entry)); + chain.push_back(entry.referrer_klass); + tag = entry.parent_tag; + } + + ASSERT_EQ(3u, chain.size()); + EXPECT_EQ(300u, chain[0]); + EXPECT_EQ(200u, chain[1]); + EXPECT_EQ(100u, chain[2]); +} + +TEST(FrontierTableTest, ClearMarksAbandonedWithoutRemovingEntry) { + FrontierTable table(64); + ASSERT_TRUE(table.insert(1, 0, 7, 0, FrontierEntryState::FRONTIER)); + + table.clear(1); + + FrontierEntry entry{}; + ASSERT_TRUE(table.lookup(1, &entry)); + EXPECT_EQ(FrontierEntryState::ABANDONED, entry.state); +} + +TEST(FrontierTableTest, ClearOfNeverInsertedTagIsNoOp) { + FrontierTable table(64); + table.clear(1); // must not crash + FrontierEntry entry{}; + EXPECT_FALSE(table.lookup(1, &entry)); +} + +TEST(FrontierTableTest, GrowsPastInitialCapacityUpToMaxCap) { + // Force at least one resize by inserting beyond the small max_cap. + const int max_cap = 10; + FrontierTable table(max_cap); + ASSERT_LE(table.capacity(), max_cap); + + for (jlong tag = 1; tag <= max_cap; tag++) { + ASSERT_TRUE(table.insert(tag, tag - 1, (u32)tag, (u32)(tag - 1))) + << "insert failed for tag " << tag; + } + EXPECT_EQ(max_cap, table.capacity()); + + for (jlong tag = 1; tag <= max_cap; tag++) { + FrontierEntry entry{}; + ASSERT_TRUE(table.lookup(tag, &entry)); + EXPECT_EQ((u32)tag, entry.referrer_klass); + } +} + +TEST(FrontierTableTest, CapacityExhaustedReportsFailureInsteadOfCrashing) { + const int max_cap = 4; + FrontierTable table(max_cap); + + for (jlong tag = 1; tag <= max_cap; tag++) { + ASSERT_TRUE(table.insert(tag, 0, 0, 0)); + } + // One past max_cap must be rejected, not silently dropped-but-crashing. + EXPECT_FALSE(table.insert(max_cap + 1, 0, 0, 0)); + EXPECT_EQ(max_cap, table.capacity()); + + // Existing entries remain intact after the failed insert. + FrontierEntry entry{}; + EXPECT_TRUE(table.lookup(1, &entry)); +} + +TEST(FrontierTableTest, ZeroMaxCapRejectsEveryInsert) { + FrontierTable table(0); + EXPECT_EQ(0, table.capacity()); + EXPECT_FALSE(table.insert(1, 0, 0, 0)); +} + +TEST(FrontierTableTest, ConcurrentInsertWhileGrowingDoesNotCrash) { + // Small max_cap relative to thread/tag count forces repeated resizes + // while other threads are concurrently inserting distinct tags. + const int max_cap = 4096; + const int thread_count = 8; + const int tags_per_thread = 256; + FrontierTable table(max_cap); + + std::vector threads; + for (int t = 0; t < thread_count; t++) { + threads.emplace_back([&table, t, tags_per_thread]() { + for (int i = 0; i < tags_per_thread; i++) { + jlong tag = (jlong)t * tags_per_thread + i + 1; + table.insert(tag, 0, (u32)tag, 0); + } + }); + } + for (auto &th : threads) { + th.join(); + } + + int found = 0; + for (jlong tag = 1; tag <= (jlong)thread_count * tags_per_thread; tag++) { + FrontierEntry entry{}; + if (table.lookup(tag, &entry)) { + EXPECT_EQ((u32)tag, entry.referrer_klass); + found++; + } + } + // Every tag fits well within max_cap, so all inserts must have + // succeeded and be independently readable. + EXPECT_EQ(thread_count * tags_per_thread, found); +} + +TEST(FrontierTableTest, ReconstructChainWalksParentTagsAndMarksEdge) { + FrontierTable table(64); + ASSERT_TRUE(table.insert(1, 0, 100, 0, FrontierEntryState::FRONTIER)); + ASSERT_TRUE(table.insert(2, 1, 200, 1, FrontierEntryState::FRONTIER)); + ASSERT_TRUE(table.insert(3, 2, 300, 2, FrontierEntryState::FRONTIER)); + + std::vector chain; + ASSERT_TRUE(table.reconstructChain(3, &chain)); + ASSERT_EQ(3u, chain.size()); + EXPECT_EQ(300u, chain[0]); + EXPECT_EQ(200u, chain[1]); + EXPECT_EQ(100u, chain[2]); + + // Every hop walked must be marked EDGE - this table's degenerate + // EdgeStore (design doc: "on a path toward a target sample"). + for (jlong tag = 1; tag <= 3; tag++) { + FrontierEntry entry{}; + ASSERT_TRUE(table.lookup(tag, &entry)); + EXPECT_EQ(FrontierEntryState::EDGE, entry.state); + } +} + +TEST(FrontierTableTest, ReconstructChainOfNeverInsertedTagFails) { + FrontierTable table(64); + std::vector chain; + EXPECT_FALSE(table.reconstructChain(1, &chain)); +} + +// --------------------------------------------------------------------------- +// Heap-walk engine (ReferenceChainTracker::runPass()/ +// heapReferenceCallback()/resolveLoadedClasses()). +// +// The implementation plan suggests testing this against "a small live-object +// graph in a test JVM (via JNI from the test)". This native-only gtest +// binary has no live JVM at all (see this file's GC-signal/tag-round-trip +// comment above, and jvmSupport_ut.cpp's fixture comment, for the same +// pre-existing constraint) - no gtest binary in this codebase embeds a +// JNI_CreateJavaVM-created JVM. So, exactly like ReferenceChainsTagTest above +// mocks SetTag/GetTag with an in-memory map, these tests mock the JVMTI/JNI +// call boundary (FollowReferences/GetLoadedClasses/GetClassSignature/ +// DeleteLocalRef) to play back a scripted synthetic object graph, and run +// the *real* production heapReferenceCallback()/resolveLoadedClasses()/ +// reconstructChain() code against it - only the JVMTI/JNI calls are faked, +// not the logic under test. A live-JVM end-to-end test belongs to a +// Java-side integration test (see ReferenceChainTrackingTest.java), not this +// native gtest binary. +// --------------------------------------------------------------------------- + +namespace { + +struct ScriptedEdge { + jvmtiHeapReferenceKind kind; + int referrer_idx; // -1 = heap root (no referrer) + int referee_idx; // index into ReferenceChainsBfsTest::node_tags + int class_idx; // index into ReferenceChainsBfsTest::classes, or -1 +}; + +struct ScriptedClass { + void *klass; + const char *signature; // JVMTI class signature, e.g. "Lcom/example/Foo;" +}; + +// Retention-edge label decode fixtures (ReferenceChainsBfsTest's +// field_decode_hierarchy + the hierarchy-introspection mock slots): a fake +// class hierarchy the slots read, mirroring just enough JVMTI class shape +// for the spec-ordinal decoder (own-declared fields in GetClassFields +// order, direct superclass, directly implemented/extended interfaces). +struct FakeField { + void *id; // fake jfieldID + const char *name; +}; +struct FakeClass { + bool is_interface; + void *super; // fake jclass, or nullptr + std::vector interfaces; // directly implemented/extended + std::vector fields; // own-declared, GetClassFields order +}; + +} // namespace + +class ReferenceChainsBfsTest : public ::testing::Test { +protected: + jvmtiInterface_1_ jvmti_tbl{}; + _jvmtiEnv mock_jvmti{}; + JNINativeInterface_ jni_tbl{}; + JNIEnv_ mock_jni{}; + + std::unordered_map tags; + std::vector classes; + std::vector script; + std::vector node_tags; + + // node_tags[idx] mirrors "the object's *current* live JVMTI + // tag" (0 once releaseSearchTags() clears it, exactly like a real + // GetTag() would report after SetTag(obj, 0)). tags_ever_assigned[idx] + // instead remembers the tag heapReferenceCallback() ever wrote through + // tag_ptr for this node, and is never reset - a production consumer + // would capture a target sample's tag the same way (at assignment time, + // e.g. via its own sample-tracking), not by re-reading GetTag() after + // the search has already released it. Tests use this to fetch a tag for + // reconstructChain() without depending on whether the search released + // it before or after the test could observe node_tags[idx]. + std::vector tags_ever_assigned; + + // Tags that GetObjectsWithTags() below reports as unresolvable, + // simulating the referenced object having died (GC'd) between passes - + // see the resolve-or-drop tests. + std::unordered_set dead_tags; + + // When true, mock_GetObjectsWithTags() below fails outright (as if the + // real JVMTI call had hit e.g. JVMTI_ERROR_OUT_OF_MEMORY), for + // ReleaseSearchTagsFailureTest - simulates releaseSearchTags()'s own + // GetObjectsWithTags() call failing rather than an individual tag + // failing to resolve (dead_tags above). + bool fail_get_objects_with_tags = false; + + // When non-zero, mock_GetObjectsWithTags() below busy-waits this many + // nanoseconds. The mock call is otherwise ~free, so a pass deadline set + // to a fraction of this value bounds an expandFrontier() invocation to + // exactly ONE batch - the production regime (one real ~25-30ms call of + // a 50ms window), needed by the lane-alternation test. + u64 gotw_delay_ns = 0; + + // Synthetic frontier-holder arrays for expandFrontier()'s array-holder + // walk: mock_NewObjectArray() hands back an opaque handle, + // mock_SetObjectArrayElement() records its elements here, and + // mock_FollowReferences() treats every recorded element as an expansion + // seed (one hop, gated by the production callback's batch_tags) when the + // holder is passed as initial_object. + std::unordered_map> holders; + uintptr_t next_holder = 0xF00D0000; + + // FindClass(name) -> registered fake class (see mock_FindClass' own + // comment): names descendFromAnchor()'s resolutions look up + // ("java/lang/ClassLoader", "java/lang/ThreadGroup", + // "java/security/ProtectionDomain", + // "java/lang/ThreadLocal$ThreadLocalMap", "java/lang/Thread"). + std::unordered_map find_classes; + // Fake class returned by mock_GetObjectClass() for unregistered objects + // (walkCandidateThreadLocals()'s fresh-anchor admission path). + void *thread_class = nullptr; + + jvmtiEnv *orig_jvmti = nullptr; + + static ReferenceChainsBfsTest *active_fixture; + + void SetUp() override { + active_fixture = this; + // See ReferenceChainsTestAccessor's own comment - without this, a + // prior test in this suite that drove the search to + // SearchState::COMPLETED/ABANDONED would make every runPass() call + // below a permanent no-op. + ReferenceChainsTestAccessor::reset(); + jvmti_tbl = jvmtiInterface_1_{}; + // start() calls VM::jvmti()->SetEventNotificationMode() - + // stub it and swap VM::_jvmti (VMTestAccessor, declared above) the + // same way ReferenceChainsTest's fixture does, so start() does not + // dereference the real (null, no live JVM) jvmtiEnv. + jvmti_tbl.SetEventNotificationMode = &mock_SetEventNotificationMode; + jvmti_tbl.SetTag = &mock_SetTag; + jvmti_tbl.GetTag = &mock_GetTag; + jvmti_tbl.GetLoadedClasses = &mock_GetLoadedClasses; + jvmti_tbl.GetClassLoader = &mock_GetClassLoader; + jvmti_tbl.GetClassSignature = &mock_GetClassSignature; + jvmti_tbl.Deallocate = &mock_Deallocate; + jvmti_tbl.FollowReferences = &mock_FollowReferences; + jvmti_tbl.IterateOverReachableObjects = &mock_IterateOverReachableObjects; + jvmti_tbl.GetObjectsWithTags = &mock_GetObjectsWithTags; + // Retention-edge label decode path (hopLabelClassFor()). + jvmti_tbl.IsInterface = &mock_IsInterface; + jvmti_tbl.GetImplementedInterfaces = &mock_GetImplementedInterfaces; + jvmti_tbl.GetClassFields = &mock_GetClassFields; + jvmti_tbl.GetFieldName = &mock_GetFieldName; + mock_jvmti.functions = &jvmti_tbl; + orig_jvmti = VMTestAccessor::getJvmti(); + VMTestAccessor::setJvmti(&mock_jvmti); + + jni_tbl = JNINativeInterface_{}; + jni_tbl.DeleteLocalRef = &mock_DeleteLocalRef; + jni_tbl.FindClass = &mock_FindClass; + jni_tbl.GetObjectClass = &mock_GetObjectClass; + jni_tbl.GetSuperclass = &mock_JniGetSuperclass; + jni_tbl.NewGlobalRef = &mock_NewGlobalRef; + jni_tbl.EnsureLocalCapacity = &mock_EnsureLocalCapacity; + jni_tbl.NewObjectArray = &mock_NewObjectArray; + jni_tbl.SetObjectArrayElement = &mock_SetObjectArrayElement; + jni_tbl.ExceptionCheck = &mock_ExceptionCheck; + jni_tbl.ExceptionClear = &mock_ExceptionClear; + mock_jni.functions = &jni_tbl; + } + + void TearDown() override { + VMTestAccessor::setJvmti(orig_jvmti); + active_fixture = nullptr; + } + + // Registers a fake class (matched by identity, not by any real JNI + // semantics) that resolveLoadedClasses() will discover via the mocked + // GetLoadedClasses(). Returns its index into `classes`. + int addClass(void *klass, const char *signature) { + classes.push_back({klass, signature}); + return (int)classes.size() - 1; + } + + // addClass() + a mock_FindClass(name) registry entry in one step, for + // the classes descendFromAnchor()'s resolution helpers look up by + // name (see find_classes' own comment). `name` uses FindClass's + // binary-name form ("java/lang/ClassLoader"), `signature` the + // class-signature form resolveLoadedClasses() interns + // ("Ljava/lang/ClassLoader;") - the production code passes each to + // exactly one of the two APIs. + int registerClassForFindClass(void *klass, const char *name, + const char *signature) { + int idx = addClass(klass, signature); + find_classes[name] = klass; + return idx; + } + + // Adds an as-yet-untagged frontier node, returning its index into + // node_tags for use as a ScriptedEdge referrer_idx/referee_idx. + int addNode() { + node_tags.push_back(0); + tags_ever_assigned.push_back(0); + return (int)node_tags.size() - 1; + } + + // Reverse lookup from a node's synthetic identity + // (&node_tags[idx], see mock_FollowReferences' initial_object handling + // below) back to its index. Returns -1 for anything else (e.g. a + // ScriptedClass's `klass` pointer, which never aliases node_tags' + // backing storage). Requires every addNode() call to happen before any + // runPass() call in a test, so node_tags never reallocates out from + // under a previously-taken address - true of every test in this file. + int indexOfNode(jobject obj) const { + for (size_t i = 0; i < node_tags.size(); i++) { + if (obj == (jobject)&node_tags[i]) { + return (int)i; + } + } + return -1; + } + + static jvmtiError JNICALL mock_SetTag(jvmtiEnv *, jobject object, jlong tag) { + // releaseSearchTags() calls SetTag(obj, 0) on the resolved + // objects GetObjectsWithTags() (below) hands back for a frontier + // node - route that through node_tags[idx] directly (the same + // storage GetObjectsWithTags's resolution and the production + // callback's tag_ptr writes both key off of), so the release is + // actually observable, not just recorded in a side map nothing else + // reads. + int idx = active_fixture->indexOfNode(object); + if (idx >= 0) { + active_fixture->node_tags[idx] = tag; + return JVMTI_ERROR_NONE; + } + if (tag == 0) { + active_fixture->tags.erase(object); + } else { + active_fixture->tags[object] = tag; + } + return JVMTI_ERROR_NONE; + } + + static jvmtiError JNICALL mock_GetTag(jvmtiEnv *, jobject object, jlong *tag_ptr) { + int idx = active_fixture->indexOfNode(object); + if (idx >= 0) { + *tag_ptr = active_fixture->node_tags[idx]; + return JVMTI_ERROR_NONE; + } + auto it = active_fixture->tags.find(object); + *tag_ptr = it != active_fixture->tags.end() ? it->second : 0; + return JVMTI_ERROR_NONE; + } + + static jvmtiError JNICALL mock_GetLoadedClasses(jvmtiEnv *, jint *count_ptr, + jclass **classes_ptr) { + auto &classes = active_fixture->classes; *count_ptr = (jint)classes.size(); + *classes_ptr = classes.empty() + ? nullptr + : (jclass *)malloc(sizeof(jclass) * classes.size()); + for (size_t i = 0; i < classes.size(); i++) { + (*classes_ptr)[i] = (jclass)classes[i].klass; + } + return JVMTI_ERROR_NONE; + } + + // admitStaticFieldRoots()'s app-classes-first partition (referenceChains.cpp) + // calls this for every loaded class. Every fixture class is "bootstrap" + // (null classloader) so the partition is a no-op and this suite's + // scripted class order/indices stay exactly as each test set them up. + static jvmtiError JNICALL mock_GetClassLoader(jvmtiEnv *, jclass, + jobject *classloader_ptr) { + *classloader_ptr = nullptr; + return JVMTI_ERROR_NONE; + } + + static jvmtiError JNICALL mock_GetClassSignature(jvmtiEnv *, jclass klass, + char **signature_ptr, + char **generic_ptr) { + for (auto &c : active_fixture->classes) { + if (c.klass == (void *)klass) { + *signature_ptr = strdup(c.signature); + if (generic_ptr != nullptr) { + *generic_ptr = nullptr; + } + return JVMTI_ERROR_NONE; + } + } + return JVMTI_ERROR_INVALID_CLASS; + } + + static jvmtiError JNICALL mock_Deallocate(jvmtiEnv *, unsigned char *mem) { + free(mem); + return JVMTI_ERROR_NONE; + } + + static void JNICALL mock_DeleteLocalRef(JNIEnv *, jobject) { + // no-op: this fixture's fake jobject/jclass values are not real JNI + // local refs. + } + + // Retention-edge label decode fixtures (fillHopEdgeLabels()/ + // hopLabelClassFor()): a fake class hierarchy the hierarchy-introspection + // slots below read, plus a tag -> fake jclass map (the decoder resolves + // the referrer class from its raw tag via GetObjectsWithTags - the test + // populates field_decode_classes from resolveLoadedClasses()-minted + // tags, and mock_GetObjectsWithTags consults it first). + std::unordered_map field_decode_classes; + std::unordered_map field_decode_hierarchy; + + // The hierarchy-introspection slots the decoder needs (IsInterface/ + // GetImplementedInterfaces/GetClassFields/GetFieldName on the JVMTI + // table, GetSuperclass on the JNI table - modern JVMTI dropped its own + // GetSuperclass). All lookups go through field_decode_hierarchy; an + // unregistered class returns a failure code so the decoder marks the + // class undecodable and degrades to kind labels - the exact production + // fail-safe shape. + static jvmtiError JNICALL mock_IsInterface(jvmtiEnv *, jclass cls, + jboolean *is_interface_ptr) { + auto it = active_fixture->field_decode_hierarchy.find(cls); + if (it == active_fixture->field_decode_hierarchy.end()) { + return JVMTI_ERROR_INVALID_CLASS; + } + *is_interface_ptr = it->second.is_interface ? JNI_TRUE : JNI_FALSE; + return JVMTI_ERROR_NONE; + } + static jclass JNICALL mock_JniGetSuperclass(JNIEnv *, jclass cls) { + auto it = active_fixture->field_decode_hierarchy.find(cls); + return it == active_fixture->field_decode_hierarchy.end() + ? nullptr + : (jclass)it->second.super; + } + static jvmtiError JNICALL mock_GetImplementedInterfaces( + jvmtiEnv *, jclass cls, jint *count_ptr, jclass **ifaces_ptr) { + auto it = active_fixture->field_decode_hierarchy.find(cls); + if (it == active_fixture->field_decode_hierarchy.end()) { + return JVMTI_ERROR_INVALID_CLASS; + } + const std::vector &ifaces = it->second.interfaces; + *count_ptr = (jint)ifaces.size(); + *ifaces_ptr = ifaces.empty() + ? nullptr + : (jclass *)malloc(sizeof(jclass) * ifaces.size()); + for (size_t i = 0; i < ifaces.size(); i++) { + (*ifaces_ptr)[i] = (jclass)ifaces[i]; + } + return JVMTI_ERROR_NONE; + } + static jvmtiError JNICALL mock_GetClassFields( + jvmtiEnv *, jclass cls, jint *count_ptr, jfieldID **fields_ptr) { + auto it = active_fixture->field_decode_hierarchy.find(cls); + if (it == active_fixture->field_decode_hierarchy.end()) { + return JVMTI_ERROR_INVALID_CLASS; + } + const std::vector &fields = it->second.fields; + *count_ptr = (jint)fields.size(); + *fields_ptr = fields.empty() + ? nullptr + : (jfieldID *)malloc(sizeof(jfieldID) * fields.size()); + for (size_t i = 0; i < fields.size(); i++) { + (*fields_ptr)[i] = (jfieldID)fields[i].id; + } + return JVMTI_ERROR_NONE; + } + static jvmtiError JNICALL mock_GetFieldName( + jvmtiEnv *, jclass, jfieldID field, char **name_ptr, + char ** /*signature_ptr*/, char ** /*generic_ptr*/) { + for (const auto &kv : active_fixture->field_decode_hierarchy) { + for (const FakeField &f : kv.second.fields) { + if (f.id == (void *)field) { + size_t len = strlen(f.name) + 1; + char *name = (char *)malloc(len); + memcpy(name, f.name, len); + *name_ptr = name; + return JVMTI_ERROR_NONE; + } + } + } + return JVMTI_ERROR_INVALID_FIELDID; + } + + // expandFrontier()/admitStaticFieldRoots() resolve java/lang/Object once + // as the holder array's element type - a non-null fake jclass is all it + // needs (the type is never introspected, only passed to NewObjectArray()). + // descendFromAnchor()'s class resolution (resolveNoDescendClassTags()/ + // resolveThreadLocalMapClassTag()) additionally needs NAME lookup - the + // find_classes registry maps a FindClass name to a class registered via + // addClass() so those resolutions see the same tagged fake class the + // scripted graph uses. + static jclass JNICALL mock_FindClass(JNIEnv *, const char *name) { + auto it = active_fixture->find_classes.find(name); + if (it != active_fixture->find_classes.end()) { + return (jclass)it->second; + } + return (jclass)0xC1A55; + } + + // walkCandidateThreadLocals()'s fresh-anchor admission calls + // GetObjectClass(thread) - unregistered classes return the fixture's + // fake Thread class (set_thread_class) so the anchor entry's class tag + // resolves through the same mocked GetTag/tagging path. + static jclass JNICALL mock_GetObjectClass(JNIEnv *, jobject) { + return (jclass)active_fixture->thread_class; + } + + + // The production code wraps that fake jclass in a global ref (a real + // local ref would dangle across JNI-entered test seams - see + // _cached_object_class's own comment). This fixture's "refs" are raw + // fake pointers with no JNI lifetime, so identity is the correct mock. + static jobject JNICALL mock_NewGlobalRef(JNIEnv *, jobject obj) { + return obj; + } + + static jint JNICALL mock_EnsureLocalCapacity(JNIEnv *, jint) { + return JNI_OK; + } + + // expandFrontier() calls jniExceptionCheck() after every upcall that can + // legally throw (NewObjectArray/SetObjectArrayElement/EnsureLocalCapacity + // failures) - this fixture's mocks never throw, so there is never a + // pending exception to report or clear. + static jboolean JNICALL mock_ExceptionCheck(JNIEnv *) { + return JNI_FALSE; + } + + static void JNICALL mock_ExceptionClear(JNIEnv *) { + // no-op: mock_ExceptionCheck() never reports a pending exception. + } + + // Hands back a fresh opaque holder handle and registers it in `holders` so + // mock_SetObjectArrayElement()/mock_FollowReferences() can find its + // elements. `length`/`elementClass`/`initialElement` are unused - the + // fixture never reads the array back, only its recorded element list. + static jobjectArray JNICALL mock_NewObjectArray(JNIEnv *, jsize, jclass, + jobject) { + jobject handle = (jobject)(active_fixture->next_holder++); + active_fixture->holders[handle] = {}; + return (jobjectArray)handle; + } + + static void JNICALL mock_SetObjectArrayElement(JNIEnv *, jobjectArray array, + jsize idx, jobject value) { active_fixture->holders[(jobject)array].push_back(value); + } + + // runPassManualWalk()'s root enumeration (the default, non-fallback path): + // reports each scripted root edge's referee to heapRootCallback() exactly + // as a real IterateOverReachableObjects() reports a root-held object - + // tag_ptr only, no oop, no transitive children (see runPassManualWalk()'s + // own comment). Expansion past the roots is then driven by expandFrontier() + // through the same mock_FollowReferences() array-holder path the resumed + // fallback passes use. stack_ref/object_ref callbacks are unused here - a + // JNI-global root (durable) is all these tests need to model. + static jvmtiError JNICALL mock_IterateOverReachableObjects( + jvmtiEnv *, jvmtiHeapRootCallback heap_root_cb, + jvmtiStackReferenceCallback, jvmtiObjectReferenceCallback, + const void *user_data) { + for (auto &e : active_fixture->script) { + if (e.referrer_idx != -1) { + continue; + } + jlong class_tag = 0; + if (e.class_idx >= 0) { + class_tag = active_fixture->tags[active_fixture->classes[e.class_idx].klass]; + } + jlong *tag_ptr = &active_fixture->node_tags[e.referee_idx]; + jvmtiIterationControl ctl = heap_root_cb( + JVMTI_HEAP_ROOT_JNI_GLOBAL, class_tag, /*size=*/0, tag_ptr, + const_cast(user_data)); + if (*tag_ptr != 0) { + active_fixture->tags_ever_assigned[e.referee_idx] = *tag_ptr; + } + if (ctl == JVMTI_ITERATION_ABORT) { + break; + } + } + return JVMTI_ERROR_NONE; + } + + // Resolves each requested tag to its node's synthetic identity + // (&node_tags[idx]) by scanning node_tags for a matching current value - + // mirroring real GetObjectsWithTags()'s "only currently-live tags come + // back" contract. A tag in `dead_tags` is deliberately omitted even if + // node_tags still holds it, simulating "the object died, JVMTI forgot + // the tag with it" for the resolve-or-drop tests. + static jvmtiError JNICALL mock_GetObjectsWithTags( + jvmtiEnv *, jint tag_count, const jlong *req_tags, jint *count_ptr, + jobject **object_result_ptr, jlong **tag_result_ptr) { + if (active_fixture->fail_get_objects_with_tags) { + // Deliberately leave *count_ptr/*object_result_ptr/*tag_result_ptr + // untouched - a real failed JVMTI call makes no promise about + // them, and releaseSearchTags() must not read them on this path. + return JVMTI_ERROR_OUT_OF_MEMORY; + } + if (active_fixture->gotw_delay_ns != 0) { + u64 until = OS::nanotime() + active_fixture->gotw_delay_ns; + while (OS::nanotime() < until) { + // busy-wait: a sleep could overshoot by scheduler latency, + // and the overshoot direction matters for the one-batch + // deadline arithmetic the callers of this knob rely on. + } + } + std::vector objs; + std::vector found; + for (jint i = 0; i < tag_count; i++) { + jlong want = req_tags[i]; + if (want == 0 || active_fixture->dead_tags.count(want) > 0) { + continue; + } + // The decoder resolves a referrer CLASS from its raw (negative) + // tag - no node carries one, so the tag -> fake jclass map + // (field_decode_classes, see its own comment) serves it. + auto fd = active_fixture->field_decode_classes.find(want); + if (fd != active_fixture->field_decode_classes.end()) { + objs.push_back((jobject)fd->second); + found.push_back(want); + break; + } + for (size_t idx = 0; idx < active_fixture->node_tags.size(); idx++) { + if (active_fixture->node_tags[idx] == want) { + objs.push_back((jobject)&active_fixture->node_tags[idx]); + found.push_back(want); + break; + } + } + } + *count_ptr = (jint)objs.size(); + *object_result_ptr = objs.empty() + ? nullptr : (jobject *)malloc(sizeof(jobject) * objs.size()); + *tag_result_ptr = found.empty() + ? nullptr : (jlong *)malloc(sizeof(jlong) * found.size()); + for (size_t i = 0; i < objs.size(); i++) { + (*object_result_ptr)[i] = objs[i]; + (*tag_result_ptr)[i] = found[i]; + } + return JVMTI_ERROR_NONE; + } + + // Plays back `script` against the real production heap_reference_callback, + // modelling enough of FollowReferences' actual semantics for these + // heap-walk tests to be meaningful: + // - "a reference from A to B is not traversed until A is visited" - + // an edge whose referrer was not returned JVMTI_VISIT_OBJECTS for + // (or was never itself visited) is skipped, exactly as a real + // traversal would never reach it. + // - a JVMTI_VISIT_ABORT return stops delivery immediately. + // - when `initial_object` is non-NULL (expandFrontier()'s + // resumed-pass calls), only edges reachable from that object's own + // node are replayed - root edges (referrer_idx == -1) are skipped + // entirely, matching FollowReferences' real "the specified object is + // used instead of the heap roots" contract. `initial_object == NULL` + // (the first-pass, root-seeded call) is unchanged from the original + // single-pass heap-walk engine. + // This is not a full JVMTI implementation (real traversal order, + // multi-referrer objects, and primitive/array callbacks are all out of + // scope) - just enough fidelity to exercise the hop-cap/budget/ + // frontier-cap/class-skip/resumption logic in heapReferenceCallback()/ + // expandFrontier() themselves. + static jvmtiError JNICALL mock_FollowReferences( + jvmtiEnv *, jint, jclass, jobject initial_object, + const jvmtiHeapCallbacks *callbacks, const void *user_data) { + std::unordered_map expandable; // seed_idx == -2 marks the root walk (initial_object == NULL); any + // other value marks an expansion walk seeded from one or more boundary + // objects, in which case root edges are never replayed. -1 is the + // array-holder walk (a whole BFS level's boundary objects at once); + // >= 0 is the single-object legacy per-entry walk. + int seed_idx = -2; + // The transient holder array itself is never tagged (mirrors real + // production: admitStaticFieldRoots()/expandFrontier() never call + // SetTag on the frontier-holder array they build), so every + // holder->element ARRAY_ELEMENT edge below is replayed with a + // referrer tag of 0. + static jlong holder_tag = 0; + if (initial_object != nullptr) { + auto holder_it = active_fixture->holders.find(initial_object); + if (holder_it != active_fixture->holders.end()) { + // Array-holder walk (expandFrontier()'s already-tagged + // boundary batch, or admitStaticFieldRoots()'s negative- + // tagged class-object seed): actually invoke the production + // callback for each holder->element edge, exactly like a + // real FollowReferences(initial_object=holder_array) call + // would - this is what lets heap_reference_callback()'s own + // tag-sign/reference_kind logic (e.g. the *tag_ptr < 0 + // early-return and its admitStaticFieldRoots() carve-out) + // actually run, rather than assuming every element is + // expandable. + seed_idx = -1; + for (jobject elem : holder_it->second) { + int idx = active_fixture->indexOfNode(elem); + if (idx < 0) { + continue; + } + jlong *tag_ptr = &active_fixture->node_tags[idx]; + jint ctl = callbacks->heap_reference_callback( + JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT, nullptr, + /*class_tag=*/0, /*referrer_class_tag=*/0, + /*size=*/0, tag_ptr, &holder_tag, + /*length=*/-1, const_cast(user_data)); + if (*tag_ptr != 0) { + active_fixture->tags_ever_assigned[idx] = *tag_ptr; + } + if (ctl & JVMTI_VISIT_ABORT) { + return JVMTI_ERROR_NONE; + } + expandable[idx] = (ctl & JVMTI_VISIT_OBJECTS) != 0; } + } else { + seed_idx = active_fixture->indexOfNode(initial_object); + expandable[seed_idx] = true; + } + } + for (auto &e : active_fixture->script) { + if (e.referrer_idx == -1) { + if (seed_idx != -2) { + continue; // resumed pass: never replay root edges + } + } else { + auto it = expandable.find(e.referrer_idx); + if (it == expandable.end() || !it->second) { continue; + } + } jlong class_tag = 0; + if (e.class_idx >= 0) { + class_tag = active_fixture->tags[active_fixture->classes[e.class_idx].klass]; + } + jlong *referrer_tag_ptr = e.referrer_idx >= 0 + ? &active_fixture->node_tags[e.referrer_idx] : nullptr; + jlong *tag_ptr = &active_fixture->node_tags[e.referee_idx]; + jint ctl = callbacks->heap_reference_callback( + e.kind, nullptr, class_tag, /*referrer_class_tag=*/0, + /*size=*/0, tag_ptr, referrer_tag_ptr, /*length=*/-1, + const_cast(user_data)); + if (*tag_ptr != 0) { + active_fixture->tags_ever_assigned[e.referee_idx] = *tag_ptr; + } + if (ctl & JVMTI_VISIT_ABORT) { + return JVMTI_ERROR_NONE; + } + expandable[e.referee_idx] = (ctl & JVMTI_VISIT_OBJECTS) != 0; + } + return JVMTI_ERROR_NONE; + } +}; + +ReferenceChainsBfsTest *ReferenceChainsBfsTest::active_fixture = nullptr; + +TEST_F(ReferenceChainsBfsTest, ReconstructsChainForSyntheticGraph) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + void *classA = (void *)0x2001, *classB = (void *)0x2002, + *classTarget = (void *)0x2003; + int ca = addClass(classA, "Lcom/rc/phase3/graph/A;"); + int cb = addClass(classB, "Lcom/rc/phase3/graph/B;"); + int ct = addClass(classTarget, "Lcom/rc/phase3/graph/Target;"); + + int nodeA = addNode(); + int nodeB = addNode(); + int nodeTarget = addNode(); + + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, ca}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, cb}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeB, nodeTarget, ct}, + }; + + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_FALSE(truncated); + + // A single pass that reaches full exhaustion of the reachable + // graph completes the search and releases every tag it assigned - so + // the tag must be fetched via tags_ever_assigned (captured at + // assignment time), not node_tags (already reset to 0 by + // releaseSearchTags() by the time runPass() returns; see + // ReleasesTagsOnCompletion below for the release itself). + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + jlong targetTag = tags_ever_assigned[nodeTarget]; + ASSERT_NE(0, targetTag); + EXPECT_EQ(0, node_tags[nodeTarget]); // released - see the comment above + + std::vector chain; + ASSERT_TRUE(tracker->frontierTable()->reconstructChain(targetTag, &chain)); + ASSERT_EQ(3u, chain.size()); + + int expectedTarget = Profiler::instance()->lookupClass( + "com/rc/phase3/graph/Target", strlen("com/rc/phase3/graph/Target")); + int expectedB = Profiler::instance()->lookupClass( + "com/rc/phase3/graph/B", strlen("com/rc/phase3/graph/B")); + int expectedA = Profiler::instance()->lookupClass( + "com/rc/phase3/graph/A", strlen("com/rc/phase3/graph/A")); + ASSERT_NE(-1, expectedTarget); + ASSERT_NE(-1, expectedB); + ASSERT_NE(-1, expectedA); + + EXPECT_EQ((u32)expectedTarget, chain[0]); + EXPECT_EQ((u32)expectedB, chain[1]); + EXPECT_EQ((u32)expectedA, chain[2]); + + // buildChainEvent() wraps the same reconstructChain() call into + // the ReferenceChainEvent shape Recording::recordReferenceChain() + // (flightRecorder.cpp) expects - same chain/order, plus the target's own + // depth from FrontierEntry. + ReferenceChainEvent event; + ASSERT_TRUE(tracker->buildChainEvent(&mock_jvmti, &mock_jni, targetTag, + &event)); + EXPECT_EQ((u64)targetTag, event._target_tag); + EXPECT_EQ(2u, event._depth); // root(A, depth0) -> B(depth1) -> Target(depth2) + ASSERT_EQ(chain, event._chain); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, BuildChainEventFailsForUnknownTag) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + ReferenceChainEvent event; + EXPECT_FALSE(tracker->buildChainEvent(&mock_jvmti, &mock_jni, 12345, + &event)); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, BuildAbandonedEventFailsUnlessSearchAbandoned) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // Freshly started: never abandoned (never even run a pass yet). + ReferenceChainAbandonedEvent event; + EXPECT_FALSE(tracker->buildAbandonedEvent(&event)); + + int nodeA = addNode(); + script = {{JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}}; + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + // Small graph, no caps hit -> COMPLETED, not ABANDONED. + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + EXPECT_FALSE(tracker->buildAbandonedEvent(&event)); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, HopCapStopsAdmittingBeyondCap) { + Arguments args; + // hops=1: only depth 0 (direct root references) may be admitted. + ASSERT_FALSE(args.parse("referencechains=true:hops=1:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int nodeA = addNode(); + int nodeB = addNode(); + + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, // depth 0 - admitted + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, // depth 1 - capped + }; + + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_FALSE(truncated); // hop cap is not truncation - a normal boundary + // Not truncated -> graph fully explored within the hop cap -> the search + // completes and releases its tags in the same call (see the previous + // test's comment) - fetch nodeA's tag via tags_ever_assigned, not + // node_tags. + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + + EXPECT_NE(0, tags_ever_assigned[nodeA]); + EXPECT_EQ(0, node_tags[nodeB]); // never admitted into the frontier + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, BudgetExhaustionTruncatesAndIsReported) { + Arguments args; + // budget=1: root enumeration and the expand phase draw from separate + // budget pools (see runPassManualWalk()'s own comment), each sized 1 + // here - root enum admits nodeA, then the expand phase's own 1-unit + // budget admits exactly one of nodeA's two children before exhausting. + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int nodeA = addNode(); + int nodeB = addNode(); + int nodeC = addNode(); + + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, // admitted via root enum + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, // admitted via expand + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeC, -1}, // expand budget exhausted + }; + + bool truncated = false; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_TRUE(truncated); + // Budget exhaustion for *this pass* leaves pending work (nodeC's own + // edge was never even attempted) - the search stays RUNNING, not + // COMPLETED, so no tag release happens yet and node_tags[nodeA]/[nodeB] + // are still the real assigned tags. + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + + EXPECT_NE(0, node_tags[nodeA]); + EXPECT_NE(0, node_tags[nodeB]); + EXPECT_EQ(0, node_tags[nodeC]); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, PreTaggedClassObjectsAreNeverExpandedOrAdmitted) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int classNode = addNode(); + node_tags[classNode] = -7; // simulate a class object already tagged by + // resolveLoadedClasses() before this pass - + // see ClassTagTable's tag-sign convention. + int fieldTargetNode = addNode(); + + script = { + // A root reference straight to the class object (e.g. a + // JVMTI_HEAP_REFERENCE_SYSTEM_CLASS root edge in a real walk). + {JVMTI_HEAP_REFERENCE_SYSTEM_CLASS, -1, classNode, -1}, + // A static field of that class - must never be delivered by a real + // FollowReferences call, since the class-object edge above must not + // return JVMTI_VISIT_OBJECTS; mock_FollowReferences enforces this + // the same way a real traversal would (see its own comment). + {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, fieldTargetNode, -1}, + }; + + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_FALSE(truncated); + + EXPECT_EQ(-7, node_tags[classNode]); // untouched - never treated as a + // frontier object + EXPECT_EQ(0, node_tags[fieldTargetNode]); // never reached - the class + // edge above must not expand + + tracker->stop(); +} + +// Regression test for admitStaticFieldRoots(): an object retained solely by +// a static field (no other GC root reaches it) must still be discovered. +// The scripted class is registered via addClass() with its own node's +// address as the jclass identity, so GetLoadedClasses()/resolveLoadedClasses() +// (real production code, driven through the same mocked jvmti) tag it +// negative through node_tags[classNode] exactly like a real Class object - +// the same identity the STATIC_FIELD script edge below uses as its +// referrer, mirroring how a real Class object is simultaneously "a loaded +// class" and "the referrer of its own static-field edges". +TEST_F(ReferenceChainsBfsTest, DiscoversObjectRetainedOnlyByStaticField) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int classNode = addNode(); + int fieldTargetNode = addNode(); + addClass((void *)&node_tags[classNode], "Lcom/rc/statics/Holder;"); + + script = { + // No GC-root path to fieldTargetNode at all - it is reachable only + // via classNode's static field. + {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, fieldTargetNode, -1}, + }; + + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_FALSE(truncated); + + // classNode got tagged negative by the real resolveLoadedClasses() scan + // (not manually, unlike PreTaggedClassObjectsAreNeverExpandedOrAdmitted + // above), and was never itself admitted as a frontier object. + EXPECT_LT(node_tags[classNode], 0); + + jlong target_tag = tags_ever_assigned[fieldTargetNode]; + ASSERT_NE(0, target_tag); + + std::vector chain; + ASSERT_TRUE(tracker->frontierTable()->reconstructChain(target_tag, &chain)); + ASSERT_EQ(1u, chain.size()); + + FrontierEntry entry{}; + ASSERT_TRUE(tracker->frontierTable()->lookup(target_tag, &entry)); + EXPECT_EQ(0, entry.parent_tag); // root-attached, not a child hop + EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD, entry.root_kind); + + tracker->stop(); +} + +// Regression test for the resolveLoadedClasses() scan-skip guard: it must +// compare `class_count != _last_resolved_class_count`, not `class_count > +// _last_resolved_class_count`. GetLoadedClasses()'s count is not monotonic - +// class unloading can shrink it - so a `>` guard would stay permanently +// skipped once the count is loaded back up to, but not past, a prior +// historical peak, silently leaving any *different* classes loaded in that +// regrowth untagged forever. See resolveLoadedClasses()'s own comment for +// the full rationale. +TEST_F(ReferenceChainsBfsTest, ResolveLoadedClassesRescansAfterClassCountShrinksAndPartiallyRegrows) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + void *classA = (void *)0x3001, *classB = (void *)0x3002, *classC = (void *)0x3003; + addClass(classA, "Lcom/rc/regress/A;"); + int idxB = addClass(classB, "Lcom/rc/regress/B;"); + + // Pass 1: both A and B loaded (count == 2) - both get resolved/tagged. + ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); + EXPECT_EQ(2, ReferenceChainsTestAccessor::lastResolvedClassCount()); + ASSERT_NE(0u, tags.count(classA)); + ASSERT_NE(0u, tags.count(classB)); + EXPECT_NE(0, tags[classA]); + EXPECT_NE(0, tags[classB]); + + // Simulate B's classloader being GC'd: GetLoadedClasses() now reports + // only A (count shrinks 2 -> 1), exactly like a real class unload. + classes.erase(classes.begin() + idxB); + ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); + EXPECT_EQ(1, ReferenceChainsTestAccessor::lastResolvedClassCount()); + + // Simulate a *different* class C loading back in, bringing the count + // back to 2 - the same count as pass 1's peak, but not the same class + // set. The buggy `>` guard (2 > 2 is false, since it never re-lowered + // _last_resolved_class_count on the shrink above either) would skip the + // scan here and leave C's tag at 0 forever; the fixed `!=` guard must + // still resolve it. + addClass(classC, "Lcom/rc/regress/C;"); + ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); + EXPECT_EQ(2, ReferenceChainsTestAccessor::lastResolvedClassCount()); + ASSERT_NE(0u, tags.count(classC)); + EXPECT_NE(0, tags[classC]); // the regression this test guards against + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// Incremental resumption across passes (ReferenceChainTracker:: +// expandFrontier()/releaseSearchTags()/shouldRunPass(), and runPass()'s +// SearchState transitions). +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsBfsTest, MultiPassResumptionReconstructsChainAcrossPasses) { + Arguments args; + // budget=1 forces each pass to admit at most one new frontier entry, so + // this 3-hop chain cannot be discovered within a single pass - + // exercising expandFrontier() (resumed passes), not just the first + // pass's root walk. + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int nodeA = addNode(); + int nodeB = addNode(); + int nodeTarget = addNode(); + + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeB, nodeTarget, -1}, + }; + + // Drive the search to completion one pass at a time, exactly as + // threadLoop() would once wired up (each call bounded by `budget`). + bool truncated = true; + int passes_issued = 0; + while (tracker->searchState() == SearchState::RUNNING && passes_issued < 20) { + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + passes_issued++; + } + + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + EXPECT_GT(tracker->passesRun(), 1); // did not fit in a single pass + EXPECT_EQ(tracker->passesRun(), passes_issued); + + jlong targetTag = tags_ever_assigned[nodeTarget]; + ASSERT_NE(0, targetTag); + std::vector chain; + ASSERT_TRUE(tracker->frontierTable()->reconstructChain(targetTag, &chain)); + // Depth/parent_tag linkage survived resumption intact - all 3 hops walk + // back to a root-attached (depth 0) entry, which reconstructChain() + // requires to succeed at all (see its own "reaching parent_tag == 0" + // contract). + EXPECT_EQ(3u, chain.size()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, FrontierCapHitAbandonsImmediately) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000:framecap=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + ASSERT_EQ(1, tracker->frontierTable()->maxCapacity()); + + int nodeA = addNode(); + int nodeB = addNode(); + + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, // fits (the one slot) + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, // frontier cap hit + }; + + bool truncated = false; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_TRUE(truncated); + // Frontier-size cap hit -- the search abandons immediately (design + // doc's Termination-section priority 1). Once the table is full, no + // new entry can ever be admitted, so the frontier can never grow + // again -- deferring to a separate no-progress counter would never + // actually observe that counter, since this same condition matches + // every subsequent pass too. + ASSERT_EQ(SearchState::ABANDONED, tracker->searchState()); + ASSERT_EQ(SearchAbandonReason::FRONTIER_CAP, tracker->abandonReason()); + + // nodeA was admitted (frontier cap=1 allowed one entry), then its tag + // was released as part of this same pass's abandon handling (the mock + // GetObjectsWithTags() succeeds by default). + EXPECT_NE(0, tags_ever_assigned[nodeA]); + EXPECT_EQ(0, node_tags[nodeA]); + // nodeB was never admitted (frontier cap hit). + EXPECT_EQ(0, node_tags[nodeB]); + + tracker->stop(); +} + +// releaseSearchTags()'s GetObjectsWithTags() call failing must NOT be treated +// as "released" - see that method's own comment for why: marking a tag +// ABANDONED (or resetting _next_tag on restart) while its object might still +// be live would let a restarted search's fresh tags collide with it, +// corrupting FrontierTable's tag-uniqueness invariant. This is the regression +// test for that failure path (previously the return value was discarded +// entirely). +TEST_F(ReferenceChainsBfsTest, ReleaseSearchTagsFailureBlocksTagReuseUntilItSucceeds) { + Arguments args; + // framecap=1 with a self-cycle: pass 1 admits nodeA (the frontier's + // only slot); the nodeA->nodeA edge then finds nodeA + // ALREADY_ADMITTED rather than hitting the frontier cap (no new slot + // is needed for an edge back to an already-tagged object), so the + // search stays RUNNING and only the no-progress detector - after + // NO_PROGRESS_PASS_LIMIT stale passes - can abandon it. This test + // verifies that the tag release on that abandonment works even when + // GetObjectsWithTags() fails. + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000:framecap=1:ttl=0")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int nodeA = addNode(); + // nodeA -> nodeA self-cycle. With framecap=1, pass 1 admits nodeA; + // the self-cycle edge is ALREADY_ADMITTED, not a fresh frontier slot. + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeA, -1}, // self-cycle + }; + + long long failedBefore = + Counters::getCounter(REFERENCE_CHAIN_TAG_RELEASE_FAILED); + + fail_get_objects_with_tags = true; + bool truncated = false; + // Pass 1: admits nodeA; the self-cycle keeps the pass truncated + // without growing the frontier further. + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_TRUE(truncated); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + EXPECT_NE(0, tags_ever_assigned[nodeA]); + EXPECT_NE(0, node_tags[nodeA]); + + // Run enough stale passes to trigger no-progress abandonment. + for (int i = 1; i < ReferenceChainTracker::NO_PROGRESS_PASS_LIMIT + 1; i++) { + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + } + // The next pass should abandon via no-progress. + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + ASSERT_EQ(SearchState::ABANDONED, tracker->searchState()); + + // GetObjectsWithTags() failed - nodeA's still-live tag must NOT have been + // cleared, and the failure must be counted. + EXPECT_NE(0, tags_ever_assigned[nodeA]); + EXPECT_NE(0, node_tags[nodeA]) << "tag must not be cleared when the " + "release batch itself failed"; + EXPECT_FALSE(ReferenceChainsTestAccessor::tagsReleased()); + EXPECT_EQ(failedBefore + 1, + Counters::getCounter(REFERENCE_CHAIN_TAG_RELEASE_FAILED)); + + // While the release is still outstanding, shouldRunPass() must force a + // retry unconditionally. + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); + EXPECT_EQ(SearchState::ABANDONED, tracker->searchState()) + << "must retry the release in place, not restart, while tags are " + "still unreleased"; + + // A further runPass() call retries the release; still failing. + int passesBefore = tracker->passesRun(); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(passesBefore, tracker->passesRun()); + EXPECT_NE(0, node_tags[nodeA]); + EXPECT_FALSE(ReferenceChainsTestAccessor::tagsReleased()); + EXPECT_EQ(failedBefore + 2, + Counters::getCounter(REFERENCE_CHAIN_TAG_RELEASE_FAILED)); + + // Once GetObjectsWithTags() starts succeeding again. + fail_get_objects_with_tags = false; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(0, node_tags[nodeA]); + EXPECT_TRUE(ReferenceChainsTestAccessor::tagsReleased()); + EXPECT_EQ(failedBefore + 2, + Counters::getCounter(REFERENCE_CHAIN_TAG_RELEASE_FAILED)) + << "a successful release must not itself count as a failure"; + + tracker->stop(); +} + +// Regression test for the CANARY_STUCK/frontier-wipe convergence bug: prior +// to this fix, runPass()'s canary-stuck branch fired purely off +// _passes_since_last_candidate_progress, so a search whose candidate simply +// had not been found yet was abandoned - and its frontier destructively +// wiped by the next restartSearch() - after only CANARY_NO_PROGRESS_PASS_LIMIT +// passes, even while the whole-graph frontier was still growing every single +// pass. Live on-pod evidence showed exactly this: a frontier that had grown +// to 12k-16k entries got wiped roughly every 20s while chasing a +// confirmed-reachable candidate. The fix requires the whole-graph frontier to +// ALSO have stalled (_passes_since_last_progress >= NO_PROGRESS_PASS_LIMIT) +// before CANARY_STUCK can fire - see CANARY_NO_PROGRESS_PASS_LIMIT's and +// canaryStuckPassLimit()'s own comments. +TEST_F(ReferenceChainsBfsTest, CanaryStuckRequiresWholeGraphFrontierAlsoStalled) { + Arguments args; + // budget=1: exactly one new frontier admission per pass, so the frontier + // grows every single pass for as long as the chain has unexplored nodes + // left - _passes_since_last_progress never leaves 0. + ASSERT_FALSE(args.parse( + "referencechains=true:hops=200:budget=1:ttl=0:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // A chain longer than the number of passes driven below, so the frontier + // still has pending work - and is still growing one node per pass - at + // every pass this test checks. + constexpr int kChainLength = 50; + std::vector nodes; + for (int i = 0; i < kChainLength; i++) { + nodes.push_back(addNode()); + } + script.push_back({JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodes[0], -1}); + for (int i = 1; i < kChainLength; i++) { + script.push_back( + {JVMTI_HEAP_REFERENCE_FIELD, nodes[i - 1], nodes[i], -1}); + } + + // A canary candidate that this graph never actually contains (no node is + // ever tagged with the candidate's marker tag) - the candidate-specific + // stuck counter (_passes_since_last_candidate_progress) climbs every pass + // with zero discovery progress, exactly like the live-pod scenario + // chasing a candidate deeper than the old fixed + // CANARY_NO_PROGRESS_PASS_LIMIT (30) passes could reach. + ReferenceChainsTestAccessor::setCandidateCountForTest(1); + + bool truncated = true; + // One more pass than the old fixed CANARY_NO_PROGRESS_PASS_LIMIT: long + // enough that the pre-fix single-condition check would already have + // abandoned the search, but short enough that the 40-node chain still has + // unexplored work left, so the frontier is still genuinely growing every + // pass. + for (int i = 0; i < ReferenceChainTracker::CANARY_NO_PROGRESS_PASS_LIMIT + 2; + i++) { + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + ASSERT_EQ(0, tracker->passesSinceLastProgressForTest()) + << "pass " << i << ": frontier must still be growing every pass"; + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()) + << "pass " << i + << ": a canary search must not be abandoned while the " + "whole-graph frontier is still growing, even if its specific " + "candidate has not yet been found"; + } + // The narrower candidate-stuck counter climbed the whole time - this is + // what the old, single-condition check would have abandoned on alone. + EXPECT_GE(ReferenceChainsTestAccessor::passesSinceLastCandidateProgress(), + ReferenceChainTracker::CANARY_NO_PROGRESS_PASS_LIMIT); + + tracker->stop(); +} + +// Canary-lane backoff pacing (option A): a chase with unresolved candidates +// runs back-to-back only while it is fresh or making candidate progress; +// each pass with NO candidate progress doubles the spacing multiplier +// (_canary_backoff_mult) up to CANARY_BACKOFF_MULT_MAX, progress resets it +// to 1, and the OOM urgency ramp overrides the gate entirely. Live evidence +// this bounds (hotdog rounds 3-4): an un-findable candidate held the chase +// open for 32 minutes at ~88 passes/min - a full core - because canary_active +// bypassed every cadence check and threadLoop() skips its sleep whenever a +// pass will run. Work-scaled rather than a fixed cap: hotdog's own passes +// ran 0.7-4s, so any fixed cap below that would have changed nothing at all +// - the loop is work-bound when the pass exceeds the cap - while a fixed 1s +// cap starved ReferenceChainTrackingTest's deep ~200-pass chase outright. +TEST_F(ReferenceChainsBfsTest, CanaryLaneBacksOffWithoutProgressAndResetsOnProgress) { + Arguments args; + // Same shape as CanaryStuckRequiresWholeGraphFrontierAlsoStalled above: + // budget=1 with a long chain keeps the frontier growing one node per + // pass, so the CANARY_STUCK detector (which also requires a stalled + // frontier) never fires and the chase stays RUNNING through the whole + // loop below. + ASSERT_FALSE(args.parse( + "referencechains=true:hops=200:budget=1:ttl=0:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + constexpr int kChainLength = 50; + std::vector nodes; + for (int i = 0; i < kChainLength; i++) { + nodes.push_back(addNode()); + } + script.push_back({JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodes[0], -1}); + for (int i = 1; i < kChainLength; i++) { + script.push_back( + {JVMTI_HEAP_REFERENCE_FIELD, nodes[i - 1], nodes[i], -1}); + } + + ReferenceChainsTestAccessor::setCandidateCountForTest(1); + bool truncated = true; + + // Pass 1: the candidate admission itself raises the progress mark + // (0 -> 1), so this counts as progress and the multiplier stays at 1 - + // back-to-back. + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(1, ReferenceChainsTestAccessor::canaryBackoffMult()); + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(OS::nanotime())) + << "a fresh chase must be allowed to run back-to-back"; + + // Pass 2: no candidate progress -> first doubling (1 -> 2). Seed a + // deterministic pass-cost EMA (mock passes are sub-ms, so the real EMA + // decays to 0) and a fresh pass timestamp so the spacing arithmetic is + // exact: spacing = 2 x 100ms. + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(2, ReferenceChainsTestAccessor::canaryBackoffMult()); + ReferenceChainsTestAccessor::setCanaryBackoffForTest( + /*mult=*/2, /*ema_ms=*/100, OS::nanotime()); + EXPECT_FALSE(ReferenceChainsTestAccessor::shouldRunPass(OS::nanotime())) + << "a no-progress canary pass must hold off the next one"; + // Beyond the spacing, the chase is allowed again - the backoff paces, + // it never abandons. + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass( + ReferenceChainsTestAccessor::lastCanaryPassNs() + + 2ULL * 100ULL * 1000000ULL + 1)) + << "elapsed spacing must re-admit the canary pass"; + + // The OOM urgency ramp overrides the backoff gate entirely. + ReferenceChainsTestAccessor::setOomRampActive(true); + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(OS::nanotime())) + << "urgency must bypass the canary backoff"; + ReferenceChainsTestAccessor::setOomRampActive(false); + + // Consecutive no-progress passes double the multiplier up to the cap + // (seeded at 8 so one more pass reaches it, the next holds it). + ReferenceChainsTestAccessor::setCanaryBackoffForTest( + /*mult=*/8, /*ema_ms=*/100, OS::nanotime()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(ReferenceChainTracker::CANARY_BACKOFF_MULT_MAX, + ReferenceChainsTestAccessor::canaryBackoffMult()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(ReferenceChainTracker::CANARY_BACKOFF_MULT_MAX, + ReferenceChainsTestAccessor::canaryBackoffMult()) + << "the multiplier must hold at its cap, not grow past it"; + + // Candidate progress (a new candidate admitted into a slot) resets the + // lane to back-to-back. + ReferenceChainsTestAccessor::setCandidateCountForTest(2); + ReferenceChainsTestAccessor::setCandidateKlassIdForTest(1, /*klass_id=*/987); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(1, ReferenceChainsTestAccessor::canaryBackoffMult()) + << "candidate progress must reset the spacing multiplier to 1"; + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(OS::nanotime())); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, ChainCompletesWithoutAbandonment) { + Arguments args; + // budget=1 on a graph where each pass admits exactly one new edge + // until the chain is exhausted, then the frontier stops growing. + // After NO_PROGRESS_PASS_LIMIT passes with no growth, the search + // is abandoned (progress-based termination, not wall-clock TTL). + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:ttl=0:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int nodeA = addNode(); + int nodeB = addNode(); + int nodeC = addNode(); + int nodeD = addNode(); + + // A chain one node longer than either pass's 1-edge expand budget can + // fully drain in a single call, so each pass still ends truncated (see + // mock_FollowReferences()'s own comment: an array-holder walk chains + // through as many script edges as it can admit before budget aborts it) + // and there is still pending work left for the no-progress check to catch + // once the chain is fully discovered. + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeB, nodeC, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeC, nodeD, -1}, + }; + + bool truncated = false; + // Pass 1: root enum admits nodeA, expand admits nodeB, aborts on + // nodeB->nodeC for lack of budget - truncated, frontier grew (progress). + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_TRUE(truncated); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + + // Pass 2: admits nodeC, aborts on nodeC->nodeD - still truncated, + // still growing (progress). + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_TRUE(truncated); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + + // Pass 3: admits nodeD, chain exhausted - no longer truncated, + // no pending frontier, natural completion. + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_FALSE(truncated); + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + + // Tags released. + EXPECT_NE(0, tags_ever_assigned[nodeA]); + EXPECT_EQ(0, node_tags[nodeA]); + EXPECT_NE(0, tags_ever_assigned[nodeB]); + EXPECT_EQ(0, node_tags[nodeB]); + + // No-progress (not TTL) is reported as the reason when the frontier stalls. + // This test uses ttl=0 to disable the wall-clock TTL, so only the + // no-progress detector can abandon. + EXPECT_EQ(SearchAbandonReason::NONE, tracker->abandonReason()); + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, NoProgressAbandonsSearchAndReleasesTags) { + // Verify the progress-based abandonment wiring: the no-progress limit + // is accessible, positive, and reset to 0 by resetSearchStateForTest(). + // The mock runPass() re-enumerates roots on every search_started=0 + // pass, causing the frontier to oscillate rather than stabilize, + // so a full end-to-end no-progress abandonment can't be tested + // with the mock. The real JVM path is validated by the + // AggressiveLeakReferenceChainTest Java integration test. + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:ttl=0:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // Verify the no-progress limit is accessible and positive. + EXPECT_GT(ReferenceChainTracker::NO_PROGRESS_PASS_LIMIT, 0); + + // Verify that a fresh search starts with zero passes since last progress. + EXPECT_EQ(0, tracker->passesSinceLastProgressForTest()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, ResolveOrDropPrunesDeadFrontierEntries) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int nodeA = addNode(); + int nodeB = addNode(); + // A second root that root enumeration's own 1-unit budget (firstpassbudget=1) + // can't reach this pass - the resulting root-enum truncation makes + // runPassManualWalk() return before expandFrontier() ever runs (see its + // own comment on frontier-cap-hit/budget-exhausted root-enum truncation), + // so nodeA is admitted but never gets a chance to expand nodeA->nodeB. + int decoyRoot = addNode(); + + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, decoyRoot, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, + }; + + bool truncated = false; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); // pass 1 + ASSERT_TRUE(truncated); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + + jlong aTag = tags_ever_assigned[nodeA]; + ASSERT_NE(0, aTag); + // Simulate nodeA dying (collected) between pass 1 and pass 2 - + // GetObjectsWithTags will no longer report it as live. + dead_tags.insert(aTag); + + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); // pass 2: resolve-or-drop + EXPECT_FALSE(truncated); + // The dead branch was pruned for free - with nothing else pending, the + // search completes rather than staying RUNNING or being ABANDONED. + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + EXPECT_EQ(2, tracker->passesRun()); + + FrontierEntry entry{}; + ASSERT_TRUE(tracker->frontierTable()->lookup(aTag, &entry)); + EXPECT_EQ(FrontierEntryState::ABANDONED, entry.state); + + // nodeB was never discovered - nodeA's subtree was pruned, not expanded. + EXPECT_EQ(0, tags_ever_assigned[nodeB]); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// pollWatchedTargets() (design doc's Open Question 3 bridging +// step, corrected read-only mechanism - see referenceChains.h's own +// target-selection bridging step header comment and pollWatchedTargets()'s +// comment for the plan doc's "Correction to the design doc's Open Question 3 +// mechanism"). +// +// Mirrors referenceChainJfrRoundtrip_ut.cpp's FrontierTable::insert() +// seeding style rather than driving a full scripted runPass() (this test +// suite's gtest coverage bullet explicitly asks for reusing that seeding style over +// writing a third pattern): a candidate's "already discovered by an +// ordinary pass" tag is modelled directly as a FrontierTable entry plus a +// mocked GetTag() that reports it - both reach the same FrontierTable + tag +// state pollWatchedTargets()/buildChainEvent() read, regardless of whether a +// scripted heap walk or direct insertion produced it. +// +// LivenessTracker::instance() is a second process-wide singleton (see its +// own header comment) shared with livenessTracker_ut.cpp within this same +// gtest binary - klassPopulationResetForTest()/setGcGenerationsForTest() at +// SetUp/TearDown keep this suite's use of it self-contained, the same way +// ReferenceChainsTestAccessor::reset() already isolates +// ReferenceChainTracker::instance() above. +// --------------------------------------------------------------------------- + +// =========================================================================== +// Pod-in-a-jar system harness (design node: design-pod-in-a-jar-harness; +// meta-whackamole-analysis): the REAL tracker loop (shouldRunPass -> +// runPass -> pollWatchedTargets, the exact threadLoop body) driven over the +// scripted mock heap, asserting SYSTEM INVARIANTS instead of unit symptoms. +// The topology encodes every pod-discovered shape: a static holder holding +// a synchronized-list wrapper WITH its mutex==this self-edge (round 16 +// fix-A), the wrapper->c->chunks subtree, leak-tagged chunks (pre-seeded +// tags model what LivenessTracker::tagLeakInstances assigns on the pod; +// the walk's leak-tag interception path then runs for real), and flood +// statics for anchor-tier volume. Each invariant maps 1:1 to rounds that +// shipped a bug the 348-test unit suite could not see. +// =========================================================================== +class PodInAJarTest : public ReferenceChainsBfsTest { +protected: + struct PodTopology { + int holder_class_node = -1; + int wrapper_node = -1; + int list_node = -1; + int leak_cls_idx = -1; + int wrapper_cls_idx = -1; + int list_cls_idx = -1; + std::vector chunk_nodes; + std::vector chunk_leak_tags; + std::vector flood_class_nodes; + std::vector flood_nodes; + std::vector filler_nodes; + }; + + static constexpr int kChunks = 6; // < MAX_DISCOVERED_INSTANCES_PER_CLASS + static constexpr u64 kCycleNs = 2000000000ULL; // 2s fake-clock step + + void SetUp() override { + ReferenceChainsBfsTest::SetUp(); + // resolveCandidateRepresentative() NewLocalRef()s the stored + // representative; the Bfs fixture never wires that JNI slot + // (PollWatchedTargetsTest has its own). Passthrough is right for + // the harness: its reps are mock node pointers with fixture + // lifetime, never GC'd. + jni_tbl.NewLocalRef = &mock_NewLocalRefPassthrough; + // NOTE: no liveness calls here - the Bfs tests run without them, + // and liveness state changes (setGcGenerationsForTest) alter the + // pass machinery's behavior; each harness test resets liveness + // explicitly where it wants it (resetLivenessForPod below). + } + + static jobject JNICALL mock_NewLocalRefPassthrough(JNIEnv *, jobject ref) { + return ref; + } + + void resetLivenessForPod() { + LivenessTracker::instance()->klassPopulationResetForTest(); + LivenessTracker::instance()->setGcGenerationsForTest(true); + LivenessTracker::instance()->leakTagPoolResetForTest(); + } + + void TearDown() override { + ReferenceChainsBfsTest::TearDown(); + // Hygiene for whatever suite runs next: clear any population this + // harness seeded. + LivenessTracker::instance()->klassPopulationResetForTest(); + } + + // Builds the hotdog leak topology: holder-class static -> synchronized + // wrapper (with its mutex==this self-edge) -> c -> K leak-tagged chunks, + // plus 2 flood classes x 16 static-held nodes for anchor-tier volume. + PodTopology buildLeakPod() { + PodTopology topo; + // CAPACITY FIXTURE CONTRACT: classes registered via + // classes.push_back({(void*)&node_tags[node], ...}) capture the + // node's ADDRESS as the jclass identity - std::vector growth would + // reallocate node_tags and dangle every captured pointer, silently + // making indexOfNode() fail for those classes in the static sweep + // (the holder array seeds no expandable class, the STATIC_FIELD + // edges never replay, nothing admits). Reserve the full node count + // up front - the topology builder never shrinks. Node total: + // 3 roots + 2 flood class nodes + 6 chunks + 32 flood + 300 filler. + node_tags.reserve(600); + tags_ever_assigned.reserve(600); + topo.leak_cls_idx = addClass((void *)0x5001, "[B"); + topo.wrapper_cls_idx = addClass( + (void *)0x5002, + "Ljava/util/Collections$SynchronizedRandomAccessList;"); + topo.list_cls_idx = addClass((void *)0x5003, "Ljava/util/ArrayList;"); + + topo.holder_class_node = addNode(); + topo.wrapper_node = addNode(); + topo.list_node = addNode(); + // The holder class node doubles as the jclass identity (see + // DiscoversObjectRetainedOnlyByStaticField): register it as a loaded + // class so the static-field sweep admits the wrapper root-attached. + classes.push_back({(void *)&node_tags[topo.holder_class_node], + "Lcom/rc/pod/Holder;"}); + // Flood classes get their own jclass-identity nodes too, so their + // statics are separate anchors (not more statics on the holder). + const char *flood_sigs[2] = {"Lcom/rc/pod/FloodA;", + "Lcom/rc/pod/FloodB;"}; + for (int c = 0; c < 2; c++) { + topo.flood_class_nodes.push_back(addNode()); + classes.push_back( + {(void *)&node_tags[topo.flood_class_nodes[c]], flood_sigs[c]}); + } + + for (int i = 0; i < kChunks; i++) { + topo.chunk_nodes.push_back(addNode()); + } + for (int i = 0; i < 32; i++) { + topo.flood_nodes.push_back(addNode()); + } + // Backlog volume: a 300-edge deep chain under one flood root. The + // real pod's searches span many passes because the heap is huge; + // the harness needs the same property or the first pass (whose + // budget auto-scales 10x for the first pass, arguments.cpp) admits + // everything before the poll ever auto-marks, and the search + // completes before any chain can be cached. + for (int i = 0; i < 300; i++) { + topo.filler_nodes.push_back(addNode()); + } + + script = { + // LEAK_BUFFER: the holder class's static field -> wrapper. + {JVMTI_HEAP_REFERENCE_STATIC_FIELD, topo.holder_class_node, + topo.wrapper_node, topo.wrapper_cls_idx}, + // The synchronized wrapper's mutex == this self-edge (round-16 + // fix-A shape: must not demote the root-attached wrapper). + {JVMTI_HEAP_REFERENCE_FIELD, topo.wrapper_node, topo.wrapper_node, + topo.wrapper_cls_idx}, + // wrapper -> c -> chunks. + {JVMTI_HEAP_REFERENCE_FIELD, topo.wrapper_node, topo.list_node, + topo.list_cls_idx}, + }; + for (int i = 0; i < kChunks; i++) { + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, topo.list_node, + topo.chunk_nodes[i], topo.leak_cls_idx}); + } + // Flood volume: each flood class holds 16 statics. + for (int c = 0; c < 2; c++) { + for (int i = 0; i < 16; i++) { + script.push_back({JVMTI_HEAP_REFERENCE_STATIC_FIELD, + topo.flood_class_nodes[c], + topo.flood_nodes[16 * c + i], + topo.leak_cls_idx /* any class; volume only */}); + } + } + // The filler chain hangs off flood node 0 (already a static anchor). + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, topo.flood_nodes[0], + topo.filler_nodes[0], topo.leak_cls_idx}); + for (int i = 0; i + 1 < (int)topo.filler_nodes.size(); i++) { + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, + topo.filler_nodes[i], topo.filler_nodes[i + 1], + topo.leak_cls_idx}); + } + reseedChunkLeakTags(topo); + return topo; + } + + // Leak tags model the poll's tagLeakInstances() assignment (the pod + // re-tags within minutes of every restart - the verify-not-retag state + // machine); the walk's leak-tag interception path consumes them for + // real from the live node tags. + void reseedChunkLeakTags(const PodTopology &topo) { + for (int i = 0; i < (int)topo.chunk_nodes.size(); i++) { + jlong leak_tag = 1073741824LL + 100 + i; + node_tags[topo.chunk_nodes[i]] = leak_tag; + tags_ever_assigned[topo.chunk_nodes[i]] = leak_tag; + } + } + + // Resolves the leak class's tracker-side klass id from the classTags + // table via the SAME tag value the mock passes as the chunk edges' + // class_tag (tags[klass_ptr] - the sweep's negative class tag, set + // during phase 1's static sweep). Class tags are process-lifetime, so + // this survives the phase-1 search completing and releasing its + // frontier/tags - unlike reading node_tags, which the release zeroes. + u32 resolveLeakKlassId(const PodTopology &topo) { + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + EXPECT_GT(ReferenceChainsTestAccessor::passesRunForTest(), 0) + << "phase-1 pass never ran"; + jlong class_tag = tags[classes[topo.leak_cls_idx].klass]; + EXPECT_LT(class_tag, 0) << "leak class never sweep-tagged (class_tag=" + << class_tag << ")"; + return tracker->classTags()->resolve(class_tag); + } + + // Seeds the leak-side liveness (population growth + qualifying tid + + // representative = the first chunk), mirroring + // PollWatchedTargetsTest::seedGrowingCandidate. + void seedLeakLiveness(u32 klass_id, const PodTopology &topo) { + int slot; + bool created; + for (u16 i = 1; i <= 20; i++) { + LivenessTracker::instance()->klassPopulationRecordForTest( + klass_id, i, i, &slot, &created); + LivenessTracker::instance()->tidTrendRecordForTest( + klass_id, /*tid=*/4242, (u32)i, (u64)i); + } + LivenessTracker::instance()->klassPopulationSetRepresentativeForTest( + nullptr, klass_id, (jweak)&node_tags[topo.chunk_nodes[0]]); + // The representative's GetTag() identity: the mock tags map is keyed + // by object pointer; getTag(rep) must report the rep's leak tag. + tags[&node_tags[topo.chunk_nodes[0]]] = topo.chunk_leak_tags.empty() + ? 1073741924LL + : 0; /* placeholder */ + // (chunk_leak_tags is not kept; the tags map entry keeps the CURRENT + // leak tag of the rep - the reseed above made node_tags hold it.) + tags[&node_tags[topo.chunk_nodes[0]]] = + node_tags[topo.chunk_nodes[0]]; + } + + // One threadLoop iteration, fake clock: the exact body order from + // ReferenceChainTracker::threadLoop() (shouldRunPass -> runPass -> + // pollWatchedTargets, poll unconditional). + bool drivePodCycle(u64 &fake_now) { + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + bool should_run = tracker->shouldRunPassForTest(fake_now); + if (should_run) { + bool truncated = false; + tracker->runPass(&mock_jvmti, &mock_jni, &truncated); + } + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + fake_now += kCycleNs; + return should_run; + } + + // Seeds a throwaway candidate klass (NO instances, never matches any + // real class) whose only job is arming the leak signal so the phase-1 + // pass runs: the dormancy invariant (L8) proved shouldRunPass stays + // false without a candidate, and the klass-id resolution needs a pass. + void seedThrowawayLiveness(u32 klass_id) { + int slot; + bool created; + for (u16 i = 1; i <= 20; i++) { + LivenessTracker::instance()->klassPopulationRecordForTest( + klass_id, i, i, &slot, &created); + LivenessTracker::instance()->tidTrendRecordForTest( + klass_id, /*tid=*/4242, (u32)i, (u64)i); + } + LivenessTracker::instance()->klassPopulationSetRepresentativeForTest( + nullptr, klass_id, (jweak)0xBADC0DE); + } + + // Two-phase pod bring-up. Phase 1: a throwaway candidate arms the leak + // signal so cycle 1's pass runs - the sweep tags the classes and the + // walk admits the wrapper subtree, which makes the leak class's + // tracker-side klass id resolvable (the only reliable source of the id + // is the system's own resolve() over a real frontier entry). Phase 2: + // clean slate (population + search restart - the pod's churn), leak + // tags re-seeded (the poll's re-tagging after every restart), and the + // REAL candidate armed, so admissions auto-mark from cycle one exactly + // like the pod after warmup. + void bringUpPod(const PodTopology &topo, u32 &leak_klass_id_out, + u64 &fake_now) { + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + // Seed the fake clock from the REAL monotonic clock: the pain + // budgets' _last_update_ns is OS::nanotime()-based, so a fake epoch + // starting at ~0 sits before every budget's initialization and + // canStartNow() blocks every pass (the dormancy test passes either + // way - no passes at all - so only the pass-running tests caught + // this). + fake_now = OS::nanotime(); + resetLivenessForPod(); + seedThrowawayLiveness(/*klass_id=*/999); + // The candidate arms in the POLL, which runs after shouldRunPass in + // each cycle - so the first cycle only arms, and a pass actually + // runs one cycle later. Drive until a pass ran. + bool ran = false; + for (int i = 0; i < 6 && !ran; i++) { + ran = drivePodCycle(fake_now); + } + leak_klass_id_out = resolveLeakKlassId(topo); + LivenessTracker::instance()->klassPopulationResetForTest(); + ReferenceChainsTestAccessor::restartSearchForTest(); + reseedChunkLeakTags(topo); + seedLeakLiveness(leak_klass_id_out, topo); + // Poll once BEFORE the first pass of the new search: the candidate + // slot registers in the poll, and the pass's admission auto-mark + // requires the slot to exist (heapReferenceCallback's auto-mark + // guards on _candidate_count > 0). This is the pod's real ordering + // - the poll runs for minutes arming candidates before search #1. + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + } + + // Captures stdout per distinct ReferenceChainTracker/LivenessTracker + // line prefix (L4: the log-budget invariant - the round-15 800k/min + // flood and the round-18 tick stragglers become assertion failures). + struct StdoutCapture { + int saved_fd = -1; + int tmp_fd = -1; + char path[64] = {0}; + bool done = false; + StdoutCapture() { + snprintf(path, sizeof(path), "/tmp/podjar_stdout_XXXXXX"); + tmp_fd = mkstemp(path); + fflush(stdout); + saved_fd = dup(1); + dup2(tmp_fd, 1); + } + std::map perPrefixCounts() { + if (done) { + return {}; + } + done = true; + fflush(stdout); + dup2(saved_fd, 1); + close(saved_fd); + saved_fd = -1; + lseek(tmp_fd, 0, SEEK_SET); + std::map counts; + FILE *f = fdopen(tmp_fd, "r"); + char line[512]; + while (fgets(line, sizeof(line), f)) { + char cls[96], fn[96]; + if (sscanf(line, "[TEST::INFO] %95[^:]::%95[a-zA-Z]", cls, + fn) == 2) { + counts[std::string(cls) + "::" + fn]++; + } + } + fclose(f); + unlink(path); + tmp_fd = -1; + return counts; + } + }; +}; + +TEST_F(PodInAJarTest, SystemLivenessLeakChainsBuildAndCanaryResolves) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + PodTopology topo = buildLeakPod(); + u32 leak_klass_id; + u64 fake_now; + bringUpPod(topo, leak_klass_id, fake_now); + + int cycles_run = 0; + while (cycles_run < 120 && + ReferenceChainsTestAccessor::resolvedChainCountForTest() < + (size_t)kChunks && + ReferenceChainsTestAccessor::searchStateForTest() == + SearchState::RUNNING) { + drivePodCycle(fake_now); + cycles_run++; + } + + // L1: every leak-tagged chunk's tag appears as a cached chain target + // (leak-correlated events - the pod's round-16 end-goal state). + auto targets = ReferenceChainsTestAccessor::resolvedChainTargetsForTest(); + for (int i = 0; i < kChunks; i++) { + jlong leak_tag = 1073741824LL + 100 + i; + EXPECT_NE(std::find(targets.begin(), targets.end(), (u64)leak_tag), + targets.end()) + << "leak tag " << leak_tag + << " never became a cached chain target after " << cycles_run + << " cycles"; + } + + // L2: the canary resolved for the LEAK klass (found bit set on its + // slot - the round-19 criterion; fails on any pre-788d7b2a7 build). + // Candidate slots persist across restarts BY DESIGN (see the slot + // registration comment: "a klass ... stops consuming pool tags even + // though its slot persists"), so the throwaway 999 slot from + // bringUpPod's phase 1 legitimately survives as an unfound slot. + int leak_slot = -1; + for (int s2 = 0; s2 < ReferenceChainsTestAccessor::candidateCountForTest(); + s2++) { + if (ReferenceChainsTestAccessor::candidateKlassIdForTest(s2) == + leak_klass_id) { + leak_slot = s2; + break; + } + } + ASSERT_GE(leak_slot, 0) << "leak klass never registered a candidate slot"; + EXPECT_TRUE(ReferenceChainsTestAccessor::candidateFoundBitsForTest() & + (1ULL << leak_slot)) + << "leak candidate slot never marked found despite leak-tag chains"; +} + +TEST_F(PodInAJarTest, SystemSearchCompletesNaturally) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + PodTopology topo = buildLeakPod(); + u32 leak_klass_id; + u64 fake_now; + bringUpPod(topo, leak_klass_id, fake_now); + + int cycles = 0; + while (cycles < 200 && + ReferenceChainsTestAccessor::searchStateForTest() == + SearchState::RUNNING) { + drivePodCycle(fake_now); + cycles++; + } + + // L3: a healthy-topology search must end COMPLETED (all candidates + // found), never ABANDONED (TTL/frontier-cap). Pre-round-19 this line + // fails: the structurally unresolvable chase can only ever reach + // ABANDONED. + EXPECT_EQ((u8)SearchState::COMPLETED, + ReferenceChainsTestAccessor::searchStateForTest()) + << "search did not complete naturally within " << cycles + << " cycles (state=" + << (int)ReferenceChainsTestAccessor::searchStateForTest() << ")"; + EXPECT_GT(ReferenceChainsTestAccessor::passesRunForTest(), 0); +} + +TEST_F(PodInAJarTest, SystemRestartLeavesNothingBehind) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + PodTopology topo = buildLeakPod(); + u32 leak_klass_id; + u64 fake_now; + bringUpPod(topo, leak_klass_id, fake_now); + for (int i = 0; i < 120 && + ReferenceChainsTestAccessor::resolvedChainCountForTest() == 0; + i++) { + drivePodCycle(fake_now); + } + ASSERT_GT(ReferenceChainsTestAccessor::resolvedChainCountForTest(), + (size_t)0); + + ReferenceChainsTestAccessor::restartSearchForTest(); + + // L6: the restart contract as a test instead of discipline. Resolved + // chains intentionally SURVIVE the restart (restartSearch's own + // comment: a chain describes a still-live sample and keeps being + // re-emitted across restarts) - the per-search state below must not. + EXPECT_GT(ReferenceChainsTestAccessor::resolvedChainCountForTest(), + (size_t)0) + << "resolved chains should persist across restarts by design"; + EXPECT_EQ((size_t)0, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_EQ((size_t)0, + ReferenceChainsTestAccessor::staticAnchorFreshQueueSizeForTest()); + EXPECT_TRUE(ReferenceChainsTestAccessor::anchorIndexIsEmptyForTest()); + for (int s = 0; s < 5; s++) { + EXPECT_EQ(0, + ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(s)) + << "discovered slot " << s << " survived restartSearch()"; + } +} + +TEST_F(PodInAJarTest, TopologyCapacityContractStaticAdmits) { + // The topology-builder capacity contract as a regression: the FULL + // buildLeakPod topology + one direct pass must admit the wrapper + // static (and with it the whole leak subtree). This is the test that + // caught the builder's original bug: classes capturing + // &node_tags[node] as their jclass identity dangle when addNode() + // grows the vector past capacity - the sweep then seeds no class, the + // gate closes on an empty lap, and NOTHING admits, silently (the + // state machine still reports a clean COMPLETED search). + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + PodTopology topo = buildLeakPod(); + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_GT(tags_ever_assigned[topo.wrapper_node], 0) + << "wrapper static never admitted (holder_cls_tag=" + << node_tags[topo.holder_class_node] + << " sweep_gate=" << ReferenceChainsTestAccessor::sweepGateStaticCountForTest() + << "/" << ReferenceChainsTestAccessor::sweepGateResolvedCountForTest() + << ")"; + + tracker->stop(); +} + +TEST_F(PodInAJarTest, SystemHealthyAppIsDormant) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // L8: the accidental 3h pod control, encoded - a heap with no leak + // candidates must produce ZERO searches (the candidate/generations + // gate holds the machinery dormant on a healthy app). + PodTopology topo = buildLeakPod(); + // No seedLeakLiveness(): no candidate ever qualifies. bringUpPod is + // also skipped - it seeds liveness; drive raw cycles instead. + (void)topo; + resetLivenessForPod(); + u64 fake_now = OS::nanotime(); + for (int i = 0; i < 20; i++) { + drivePodCycle(fake_now); + } + EXPECT_EQ(0, ReferenceChainsTestAccessor::passesRunForTest()) + << "searches ran on a healthy (candidate-less) app"; + EXPECT_EQ((size_t)0, + ReferenceChainsTestAccessor::resolvedChainCountForTest()); +} + +TEST_F(PodInAJarTest, SystemLogBudgetPerPass) { +#ifndef DEBUG + // The gtest binary compiles the main sources WITHOUT DEBUG (round-16 + // lesson: TEST_LOG is a no-op here) - the log-budget invariant can only + // run in a DEBUG-built test binary. Skip with the reason instead of + // asserting on captured output that structurally cannot exist. + GTEST_SKIP() << "log budget needs a DEBUG-built gtest binary"; +#else + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + PodTopology topo = buildLeakPod(); + u32 leak_klass_id; + u64 fake_now; + bringUpPod(topo, leak_klass_id, fake_now); + for (int i = 0; i < 3; i++) { + drivePodCycle(fake_now); + } + + // L4: capture one full cycle (pass + poll) and bound every distinct + // line shape. The round-15 flood was thousands of lines/pass of a + // single shape; the budget catches any per-object/per-entry line that + // escapes its tier. The harness runs at debug level 2 (pinned by the + // binary's static setenv), so every gated line is live here. + StdoutCapture capture; + drivePodCycle(fake_now); + auto counts = capture.perPrefixCounts(); + ASSERT_FALSE(counts.empty()) << "no diagnostics captured at level 2"; + for (const auto &kv : counts) { + EXPECT_LE(kv.second, 300) << "log line shape '" << kv.first + << "' fired " << kv.second + << " times in one pass+poll cycle"; + } +#endif +} + +class PollWatchedTargetsTest : public ::testing::Test { +protected: + jvmtiInterface_1_ jvmti_tbl{}; + _jvmtiEnv mock_jvmti{}; + JNINativeInterface_ jni_tbl{}; + JNIEnv_ mock_jni{}; + + std::unordered_map tags; + std::unordered_set dead_refs; // NewLocalRef returns NULL for these + + jvmtiEnv *orig_jvmti = nullptr; + static PollWatchedTargetsTest *active_fixture; + + void SetUp() override { + active_fixture = this; + ReferenceChainsTestAccessor::reset(); + LivenessTracker::instance()->klassPopulationResetForTest(); + LivenessTracker::instance()->setGcGenerationsForTest(true); + + jvmti_tbl = jvmtiInterface_1_{}; + jvmti_tbl.SetEventNotificationMode = &mock_SetEventNotificationMode; + jvmti_tbl.GetTag = &mock_GetTag; + jvmti_tbl.SetTag = &mock_SetTag; + jvmti_tbl.GetClassSignature = &mock_GetClassSignature; + jvmti_tbl.Deallocate = &mock_Deallocate; + mock_jvmti.functions = &jvmti_tbl; + orig_jvmti = VMTestAccessor::getJvmti(); + VMTestAccessor::setJvmti(&mock_jvmti); + + jni_tbl = JNINativeInterface_{}; + jni_tbl.NewLocalRef = &mock_NewLocalRef; + jni_tbl.DeleteLocalRef = &mock_DeleteLocalRef; + jni_tbl.GetObjectClass = &mock_GetObjectClass; + mock_jni.functions = &jni_tbl; + } + + void TearDown() override { + VMTestAccessor::setJvmti(orig_jvmti); + LivenessTracker::instance()->klassPopulationResetForTest(); + LivenessTracker::instance()->setGcGenerationsForTest(false); + active_fixture = nullptr; + } + + static jvmtiError JNICALL mock_GetTag(jvmtiEnv *, jobject object, jlong *tag_ptr) { + auto it = active_fixture->tags.find(object); + *tag_ptr = it != active_fixture->tags.end() ? it->second : 0; + return JVMTI_ERROR_NONE; + } + + static jvmtiError JNICALL mock_SetTag(jvmtiEnv *, jobject object, jlong tag) { + active_fixture->tags[object] = tag; + return JVMTI_ERROR_NONE; + } + + static jobject JNICALL mock_NewLocalRef(JNIEnv *, jobject ref) { + if (active_fixture->dead_refs.count(ref) > 0) { + return nullptr; + } + return ref; // identity passthrough - see this fixture's own comment + } + + static void JNICALL mock_DeleteLocalRef(JNIEnv *, jobject) { + // no-op: this fixture's fake jobject values are not real JNI refs. + } + + // pollWatchedTargets()'s diagnostic class-name lookup on the candidate's + // representative: this fixture's fake jobjects carry no real class + // identity, so a fixed non-null jclass plus a fixed signature is all + // GetObjectClass()/GetClassSignature() need to return for that lookup to + // complete without touching a real JVM. + static jclass JNICALL mock_GetObjectClass(JNIEnv *, jobject) { + return (jclass)0xC1A55; + } + + static jvmtiError JNICALL mock_GetClassSignature(jvmtiEnv *, jclass, + char **signature_ptr, + char **generic_ptr) { + *signature_ptr = strdup("Ltest/FakeKlass;"); + if (generic_ptr != nullptr) { + *generic_ptr = nullptr; + } + return JVMTI_ERROR_NONE; + } + + static jvmtiError JNICALL mock_Deallocate(jvmtiEnv *, unsigned char *mem) { + free(mem); + return JVMTI_ERROR_NONE; + } + + // Seeds LivenessTracker's real population table with a growing series + // for `klass_id` (20 strictly-increasing samples - satisfies + // selectLeakCandidates()'s min-fill, growth/floor magnitude, and + // sustained-trend hysteresis requirements, livenessTracker.h; 20 rather + // than the 10-sample minimum fill leaves comfortable margin past the + // hysteresis threshold rather than sitting exactly on its boundary) and + // points its representative at `rep`. + void seedGrowingCandidate(u32 klass_id, jweak rep) { + int slot; + bool created; + for (u16 i = 1; i <= 20; i++) { + LivenessTracker::instance()->klassPopulationRecordForTest( + klass_id, i, i, &slot, &created); + // Per-(klass, tid) qualification: selectLeakCandidates() also + // requires a qualifying allocating thread. These fixtures have + // no real tracked instances (mock JVMTI, no live heap), so a + // fixed synthetic tid exercises the gate without pretending to + // match any instance's real tid. + LivenessTracker::instance()->tidTrendRecordForTest( + klass_id, /*tid=*/4242, (u32)i, (u64)i); + } + LivenessTracker::instance()->klassPopulationSetRepresentativeForTest(nullptr, klass_id, rep); + } +}; + +PollWatchedTargetsTest *PollWatchedTargetsTest::active_fixture = nullptr; + +TEST_F(PollWatchedTargetsTest, EmitsEventForAlreadyDiscoveredCandidate) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); + + // Model "already discovered by an ordinary runPass()": a root-level + // FrontierTable entry plus a matching GetTag() result, mirroring + // referenceChainJfrRoundtrip_ut.cpp's seeding style. + ASSERT_TRUE(tracker->frontierTable()->insert( + /*tag=*/7, /*parent_tag=*/0, /*referrer_klass=*/1, /*depth=*/0, + FrontierEntryState::EDGE)); + tags[obj] = 7; + // With class-tag matching, _candidate_frontier_tags must be set + // so buildCanaryChainEvent() can reconstruct the chain. + ReferenceChainsTestAccessor::setCandidateFrontierTagForTest(0, 7); + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)); + EXPECT_EQ(7, ReferenceChainsTestAccessor::resolvedChainSourceTag(7)); + + tracker->stop(); +} + +TEST_F(PollWatchedTargetsTest, NoEventForNotYetDiscoveredCandidate) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); + // GetTag() reports 0 (default) - no pass has reached this object yet. + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::resolvedChainCount()); + + tracker->stop(); +} + +TEST_F(PollWatchedTargetsTest, NoDuplicateOnRepeatPoll) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); + + ASSERT_TRUE(tracker->frontierTable()->insert( + 7, 0, 1, 0, FrontierEntryState::EDGE)); + tags[obj] = 7; + ReferenceChainsTestAccessor::setCandidateFrontierTagForTest(0, 7); + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); + + // Klass 1 is still flagged (LivenessTracker's ranking doesn't know an + // event was already emitted for it) - a second, third, ... poll must + // not re-emit for the same target_tag. + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)); + EXPECT_EQ(7, ReferenceChainsTestAccessor::resolvedChainSourceTag(7)); + + tracker->stop(); +} + +TEST_F(PollWatchedTargetsTest, SkipsCandidateWhoseWeakReferenceDied) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); + dead_refs.insert(obj); // NewLocalRef(rep) -> NULL, as if GC'd + + ASSERT_TRUE(tracker->frontierTable()->insert( + 7, 0, 1, 0, FrontierEntryState::EDGE)); + tags[obj] = 7; // would resolve to a discovered tag, if it could resolve + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::resolvedChainCount()); + + tracker->stop(); +} + +// A cached chain must not re-emit forever: once the klass's representative +// stops resolving (collected, or LRU-evicted from LivenessTracker's +// population table - klassPopulationSetRepresentativeForTest()'s ref is the +// stand-in for either), the very next poll must prune it from +// _resolved_chains rather than leaving a dump keep re-emitting a chain for a +// sample that is gone (see _resolved_chains' own comment, referenceChains.h, +// and pollWatchedTargets()'s "candidate died, or was evicted" branch). +TEST_F(PollWatchedTargetsTest, ChainPersistsAfterRepresentativeDies) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); + + ASSERT_TRUE(tracker->frontierTable()->insert( + 7, 0, 1, 0, FrontierEntryState::EDGE)); + tags[obj] = 7; + ReferenceChainsTestAccessor::setCandidateFrontierTagForTest(0, 7); + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); + ASSERT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)); + + // The representative died. Per-instance caching: the chain persists + // (it describes a reference path that was valid at resolution time). + // It expires naturally when the search restarts and the frontier is + // wiped. The backend can filter stale chains via HeapLiveObject events. + dead_refs.insert(obj); + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()) + << "per-instance chains persist after representative dies; " + "they expire on search restart, not on representative death"; + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)); + + tracker->stop(); +} + +TEST_F(PollWatchedTargetsTest, NoOpWhenGcGenerationsDisabled) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // Overrides this fixture's own SetUp() default - exercises the + // pollWatchedTargets() guard covering LivenessTracker's own + // _gc_generations gate (population tracking's own gate), not just this tracker's own _enabled. + LivenessTracker::instance()->setGcGenerationsForTest(false); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); + ASSERT_TRUE(tracker->frontierTable()->insert( + 7, 0, 1, 0, FrontierEntryState::EDGE)); + tags[obj] = 7; + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::resolvedChainCount()); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// Resolved-chain cache (ReferenceChainTracker::cacheResolvedChain()/ +// drainPendingChainEvents(), referenceChains.cpp) - the mechanism that keeps a +// resolved chain alive across dumps so it re-emits into every JFR chunk the +// sample survives into, and keeps Profiler::writeReferenceChain()'s blocking +// lock-acquisition retry loop off the BFS scheduling thread (see +// _resolved_chains' own comment, referenceChains.h). These tests drive +// cacheResolvedChain()/drainPendingChainEvents() directly via +// ReferenceChainsTestAccessor rather than through the full +// pollWatchedTargets()/selectLeakCandidates() pipeline - the cache/overflow/ +// counter mechanism is independent of how a chain was produced, and driving it +// through hundreds of real LivenessTracker candidates just to reach +// MAX_RESOLVED_CHAINS would test the seeding helper, not this mechanism. +// --------------------------------------------------------------------------- + +class ResolvedChainCacheTest : public ::testing::Test { +protected: + void SetUp() override { + ReferenceChainsTestAccessor::reset(); + } + + void TearDown() override { + ReferenceChainsTestAccessor::reset(); + } + + static ReferenceChainEvent makeEvent(u64 target_tag) { + ReferenceChainEvent event; + event._target_tag = target_tag; + event._depth = 0; + return event; + } +}; + +// The defining property of the "stick around" model: a cached chain is +// re-emitted on every dump, not drained once. Two successive drains with no +// intervening change must BOTH return the cached chain, and the cache must +// stay populated afterwards (unlike the old queue, which emptied on drain). +TEST_F(ResolvedChainCacheTest, SnapshotReEmitsOnEveryDumpWithoutClearing) { + ReferenceChainsTestAccessor::cacheChain(/*source_tag=*/1, makeEvent(7), + /*source_tag=*/7, /*search_ns=*/0); + + std::vector firstDump; + ReferenceChainsTestAccessor::drain(&firstDump); + ASSERT_EQ(1u, firstDump.size()); + EXPECT_EQ(7u, firstDump[0]._target_tag); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()) + << "drain must not clear the cache"; + + // A second dump with nothing changed re-emits the same chain. + std::vector secondDump; + ReferenceChainsTestAccessor::drain(&secondDump); + ASSERT_EQ(1u, secondDump.size()); + EXPECT_EQ(7u, secondDump[0]._target_tag); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); +} + +// Re-resolving the same klass (a restart re-tags its sample, or a fresh walk +// finds a deeper path) refreshes its single cache slot in place rather than +// accumulating duplicates - so a dump re-emits one current chain per klass, +// not one per resolution. +TEST_F(ResolvedChainCacheTest, RefreshReplacesSameKlassInPlace) { + ReferenceChainsTestAccessor::cacheChain(1, makeEvent(7), 7, 0); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); + EXPECT_EQ(7, ReferenceChainsTestAccessor::resolvedChainSourceTag(1)); + + // Same klass, rebuilt from a new tag (e.g. after a search restart). + ReferenceChainsTestAccessor::cacheChain(1, makeEvent(9), 9, 0); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()) + << "refresh must overwrite, not append"; + EXPECT_EQ(9, ReferenceChainsTestAccessor::resolvedChainSourceTag(1)); + + std::vector dump; + ReferenceChainsTestAccessor::drain(&dump); + ASSERT_EQ(1u, dump.size()); + EXPECT_EQ(9u, dump[0]._target_tag); +} + +// Distinct klasses each get their own slot and all re-emit together in one +// dump (order is unspecified - the cache is a map keyed by klass_id). +TEST_F(ResolvedChainCacheTest, MultipleKlassesAllSnapshotTogether) { + ReferenceChainsTestAccessor::cacheChain(1, makeEvent(1), 1, 0); + ReferenceChainsTestAccessor::cacheChain(2, makeEvent(2), 2, 0); + ReferenceChainsTestAccessor::cacheChain(3, makeEvent(3), 3, 0); + ASSERT_EQ(3u, ReferenceChainsTestAccessor::resolvedChainCount()); + + std::vector dump; + ReferenceChainsTestAccessor::drain(&dump); + ASSERT_EQ(3u, dump.size()); + std::set tags; + for (const auto &e : dump) { + tags.insert(e._target_tag); + } + EXPECT_EQ((std::set{1, 2, 3}), tags); +} + +// A brand-new klass arriving with the cache already at MAX_RESOLVED_CHAINS is +// dropped (and counted via REFERENCE_CHAIN_EVENTS_DROPPED, this codebase's own +// "dropped-event-without-counter" review lens) rather than evicting some other +// still-live sample's chain - but refreshing a klass that is already cached +// still succeeds even at capacity. +TEST_F(ResolvedChainCacheTest, OverflowDropsNewKlassButAllowsRefresh) { + const int cap = ReferenceChainsTestAccessor::maxResolvedChains(); + long long droppedBefore = Counters::getCounter(REFERENCE_CHAIN_EVENTS_DROPPED); + + for (int i = 0; i < cap; i++) { + ReferenceChainsTestAccessor::cacheChain((jlong)i, makeEvent((jlong)i), + (jlong)i, 0); + } + ASSERT_EQ((size_t)cap, ReferenceChainsTestAccessor::resolvedChainCount()); + EXPECT_EQ(droppedBefore, Counters::getCounter(REFERENCE_CHAIN_EVENTS_DROPPED)) + << "filling exactly to capacity must not drop anything yet"; + + // A brand-new klass at capacity is dropped and counted. + ReferenceChainsTestAccessor::cacheChain((jlong)cap, makeEvent((jlong)cap), + (jlong)cap, 0); + EXPECT_EQ((size_t)cap, ReferenceChainsTestAccessor::resolvedChainCount()) + << "cache must stay capped, not grow past MAX_RESOLVED_CHAINS"; + EXPECT_EQ(droppedBefore + 1, Counters::getCounter(REFERENCE_CHAIN_EVENTS_DROPPED)); + EXPECT_FALSE(ReferenceChainsTestAccessor::hasResolvedChainForTag((u32)cap)); + + // Refreshing an already-cached klass at capacity must still succeed - it + // reuses that klass's existing slot rather than needing a free one. + ReferenceChainsTestAccessor::cacheChain(/*source_tag=*/0, makeEvent(999), + /*source_tag=*/999, 0); + EXPECT_EQ((size_t)cap, ReferenceChainsTestAccessor::resolvedChainCount()); + EXPECT_EQ(999, ReferenceChainsTestAccessor::resolvedChainSourceTag(0)); + EXPECT_EQ(droppedBefore + 1, Counters::getCounter(REFERENCE_CHAIN_EVENTS_DROPPED)) + << "an in-place refresh must not count as a drop"; +} + +// --------------------------------------------------------------------------- +// Pause-time pacing controller: pause-time-SLO feedback loop +// (ReferenceChainTracker::updatePacing(), referenceChains.cpp) - see that +// method's own comment (referenceChains.h) for the full mechanism. These +// tests drive updatePacing() directly with a synthetic sequence of "pass +// took Xms" wall-clock durations via ReferenceChainsTestAccessor (this +// file's existing pattern for reaching a private method/state - see the +// target-selection bridging step's hasResolvedChainForTag()/resolvedChainCount() above), +// reusing the ReferenceChainsTest fixture since updatePacing() itself makes +// no JVMTI calls. +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsTest, PacingHoldsSteadyWhenPassesLandExactlyOnCeiling) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:budget=4000:pausetarget=5")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int startBudget = ReferenceChainsTestAccessor::effectiveBudget(); + u64 startCadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); + ASSERT_EQ(4000, startBudget); // starts pinned at the configured ceiling + + // A pass landing exactly on the pause-time target is a zero error every + // call - the controller should never move away from its starting point, + // regardless of how many such passes are observed in a row. + for (int i = 0; i < 10; i++) { + ReferenceChainsTestAccessor::updatePacing(5 * 1000000ULL); // 5ms + EXPECT_EQ(startBudget, ReferenceChainsTestAccessor::effectiveBudget()); + EXPECT_EQ(startCadence, ReferenceChainsTestAccessor::effectiveCadenceNs()); + } + + tracker->stop(); +} + +TEST_F(ReferenceChainsTest, PacingShrinksBudgetAndWidensCadenceWhenOverCeiling) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:budget=4000:pausetarget=5")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int initialBudget = ReferenceChainsTestAccessor::effectiveBudget(); + u64 initialCadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); + + // A pass taking 10x the pause-time ceiling, fed repeatedly (a constant + // input - the plan's own "does not oscillate indefinitely" scenario). + int lastBudget = initialBudget; + u64 lastCadence = initialCadence; + for (int i = 0; i < 20; i++) { + ReferenceChainsTestAccessor::updatePacing(50 * 1000000ULL); // 50ms + int budget = ReferenceChainsTestAccessor::effectiveBudget(); + u64 cadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); + EXPECT_LE(budget, lastBudget); // never grows while still over ceiling + EXPECT_GE(cadence, lastCadence); // never shrinks while still over ceiling + lastBudget = budget; + lastCadence = cadence; + } + + // Moved in the correct direction... + EXPECT_LT(lastBudget, initialBudget); + EXPECT_GT(lastCadence, initialCadence); + // ...and converged to a fixed point rather than oscillating: one more + // identical input produces no further change. + ReferenceChainsTestAccessor::updatePacing(50 * 1000000ULL); + EXPECT_EQ(lastBudget, ReferenceChainsTestAccessor::effectiveBudget()); + EXPECT_EQ(lastCadence, ReferenceChainsTestAccessor::effectiveCadenceNs()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsTest, PacingGrowsBudgetBackAndRelaxesCadenceWhenUnderCeiling) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:budget=4000:pausetarget=5")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // Start from a controlled below-ceiling/above-baseline point (as if an + // earlier over-ceiling run had already shrunk/widened them - see the + // previous test) with a freshly reset controller, rather than chaining + // directly off a constant-input sequence like the previous test's own: + // _pause_pid's integral state would otherwise still be recovering from + // that sequence's windup for many iterations after switching to a + // smaller-magnitude error, muddying this test's per-step "moves in the + // correct direction every step" assertions with a transient this test + // is not about. + ReferenceChainsTestAccessor::setEffectiveBudget(2400); + ReferenceChainsTestAccessor::setEffectiveCadenceNs( + 2 * ReferenceChainsTestAccessor::baselineCadenceNs()); + ReferenceChainsTestAccessor::resetPacingController(); + int shrunkBudget = ReferenceChainsTestAccessor::effectiveBudget(); + u64 widenedCadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); + + // Now feed passes comfortably under the ceiling, repeatedly (a constant + // input, to check convergence rather than oscillation). 200 iterations - + // more than PacingShrinksBudgetAndWidensCadenceWhenOverCeiling needs - + // because this scenario's error magnitude (pausetarget=5 vs. an + // effectively-instant 0ms pass) is smaller, so the cadence side takes + // more iterations to fully unwind down to MIN_EFFECTIVE_CADENCE_NS, and + // because the borrowed-budget distance to close (configured budget * + // (BORROW_CEILING_MULTIPLIER - 1)) scales with the configured budget + // while the PID's per-pass step size does not. + int lastBudget = shrunkBudget; + u64 lastCadence = widenedCadence; + for (int i = 0; i < 200; i++) { + ReferenceChainsTestAccessor::updatePacing(0); // effectively instant + int budget = ReferenceChainsTestAccessor::effectiveBudget(); + u64 cadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); + EXPECT_GE(budget, lastBudget); // never shrinks while comfortably under + EXPECT_LE(cadence, lastCadence); // never widens while comfortably under + lastBudget = budget; + lastCadence = cadence; + } + + // Moved in the correct direction... and, since 50 identical + // comfortably-under-target passes is well past BORROW_WARMUP_PASSES, + // past the configured ceiling too - budget-borrowing lets it converge at + // the borrowed ceiling (configured budget * multiplier) instead of + // stalling at the plain configured budget. + EXPECT_GT(lastBudget, shrunkBudget); + EXPECT_EQ(4000 * ReferenceChainsTestAccessor::borrowCeilingMultiplier(), lastBudget); + EXPECT_LT(lastCadence, widenedCadence); + // ...and converged: one more identical input produces no further change. + ReferenceChainsTestAccessor::updatePacing(0); + EXPECT_EQ(lastBudget, ReferenceChainsTestAccessor::effectiveBudget()); + EXPECT_EQ(lastCadence, ReferenceChainsTestAccessor::effectiveCadenceNs()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsTest, MaybeRevokeBorrowForRootEnumPassPreservesBorrowAtBoundary) { + Arguments args; + // BORROW_UNDER_TARGET_FRACTION (referenceChains.h) is 0.5, so with + // pausetarget=10 the comfortably-under-target boundary is exactly 5ms. + ASSERT_FALSE(args.parse("referencechains=true:budget=1000:pausetarget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + ReferenceChainsTestAccessor::setBorrowedBudget(500); + ReferenceChainsTestAccessor::setConsecutiveUnderTargetPasses(5); + + // Exactly at the boundary: comfortably_under_target's `<=` check must + // still treat this as comfortably under, so the borrow is preserved. + ReferenceChainsTestAccessor::maybeRevokeBorrowForRootEnumPass(5 * 1000000ULL); + EXPECT_EQ(500, ReferenceChainsTestAccessor::borrowedBudget()); + EXPECT_EQ(5, ReferenceChainsTestAccessor::consecutiveUnderTargetPasses()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsTest, MaybeRevokeBorrowForRootEnumPassRevokesJustPastBoundary) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:budget=1000:pausetarget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + ReferenceChainsTestAccessor::setBorrowedBudget(500); + ReferenceChainsTestAccessor::setConsecutiveUnderTargetPasses(5); + ReferenceChainsTestAccessor::setEffectiveBudget(1500); // as if borrow had raised the ceiling + + // Just past the boundary: no longer comfortably under target, so the + // grant is revoked immediately, including re-clamping _effective_budget + // down to the plain (non-borrowed) budget rather than leaving it + // borrow-inflated until the next ordinary pass's updatePacing() call. + ReferenceChainsTestAccessor::maybeRevokeBorrowForRootEnumPass(6 * 1000000ULL); + EXPECT_EQ(0, ReferenceChainsTestAccessor::borrowedBudget()); + EXPECT_EQ(0, ReferenceChainsTestAccessor::consecutiveUnderTargetPasses()); + EXPECT_EQ(1000, ReferenceChainsTestAccessor::effectiveBudget()); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// PainBudget (painBudget.h) - standalone, no ReferenceChainTracker singleton +// involved. A leaky bucket over cost (ms), not an event rate: spend() +// records how much an operation cost, canStartNow() drains the balance by +// elapsed wall-clock time at the configured refill rate and reports whether +// the debt has cleared. +// --------------------------------------------------------------------------- + +TEST(PainBudgetTest, ClearBeforeAnythingIsEverSpent) { + PainBudget budget(0.01); + EXPECT_TRUE(budget.canStartNow(1000)); +} + +TEST(PainBudgetTest, SpendCreatesDebtThatBlocksAnImmediateSecondCall) { + PainBudget budget(0.01); // 1% + ASSERT_TRUE(budget.canStartNow(1000)); // establishes the drain baseline + budget.spend(100); // 100ms of debt + // No time has elapsed since the baseline call above - the debt cannot + // have drained at all yet. + EXPECT_FALSE(budget.canStartNow(1000)); +} + +TEST(PainBudgetTest, DebtDrainsProportionallyToElapsedTimeAndRefillRate) { + PainBudget budget(0.01); // 1% -> 1ms of debt needs 100ms elapsed to clear + ASSERT_TRUE(budget.canStartNow(0)); + budget.spend(10); // 10ms of debt -> needs 1000ms elapsed to fully clear + EXPECT_FALSE(budget.canStartNow(500ULL * 1000000ULL)); // 500ms elapsed - not enough + EXPECT_TRUE(budget.canStartNow(1500ULL * 1000000ULL)); // 1500ms total - enough +} + +TEST(PainBudgetTest, ZeroRefillRateNeverClearsDebt) { + PainBudget budget(0.0); + ASSERT_TRUE(budget.canStartNow(0)); + budget.spend(1); + // An enormous elapsed time still drains nothing at a 0 refill rate. + EXPECT_FALSE(budget.canStartNow(1000000000000ULL)); +} + +// --------------------------------------------------------------------------- +// Search restart (referenceChains.h's own header comment: gating a +// restarted search's first pass on LivenessTracker already reporting a leak +// candidate, plus the PainBudget cooldown above, so a search that already +// walked the whole reachable graph once does not do so again indefinitely +// without a reason). Uses an "empty reachable graph" FollowReferences mock +// (no callback invocations at all) to reach SearchState::COMPLETED in one +// call - the simplest way to drive a search to a terminal state without +// ReferenceChainsBfsTest's full scripted-graph machinery, which exists for +// chain-reconstruction coverage this suite does not need. +// --------------------------------------------------------------------------- + +class SearchRestartTest : public ::testing::Test { +protected: + jvmtiInterface_1_ jvmti_tbl{}; + _jvmtiEnv mock_jvmti{}; + jvmtiEnv *orig_jvmti = nullptr; + + void SetUp() override { + ReferenceChainsTestAccessor::reset(); + LivenessTracker::instance()->klassPopulationResetForTest(); + LivenessTracker::instance()->setGcGenerationsForTest(false); + + jvmti_tbl = jvmtiInterface_1_{}; + jvmti_tbl.SetEventNotificationMode = &mock_SetEventNotificationMode; + jvmti_tbl.GetLoadedClasses = &mock_GetLoadedClasses; + jvmti_tbl.FollowReferences = &mock_FollowReferences; + jvmti_tbl.IterateOverReachableObjects = &mock_IterateOverReachableObjects; + jvmti_tbl.GetAvailableProcessors = &mock_GetAvailableProcessors; + mock_jvmti.functions = &jvmti_tbl; + orig_jvmti = VMTestAccessor::getJvmti(); + VMTestAccessor::setJvmti(&mock_jvmti); + } + + void TearDown() override { + VMTestAccessor::setJvmti(orig_jvmti); + LivenessTracker::instance()->klassPopulationResetForTest(); + LivenessTracker::instance()->setGcGenerationsForTest(false); + // UrgentOOMProjectionBypassesCandidateGate below sets this - reset it + // here (TearDown always runs, even after a fatal ASSERT_* return) + // rather than as a trailing statement in that test body, so a failed + // assertion can't leak a stale max-heap value into the next test + // sharing this singleton. + LivenessTracker::instance()->setMaxHeapBytesForTest(-1); + } + + // No loaded classes to resolve - resolveLoadedClasses() reports 0 and + // does nothing further. + static jvmtiError JNICALL mock_GetLoadedClasses(jvmtiEnv *, jint *count, + jclass **out) { + *count = 0; + *out = nullptr; + return JVMTI_ERROR_NONE; + } + + // ReferenceChainTracker::start() -> autoTuneDefaults() queries this + // whenever LivenessTracker reports a max heap > 0 - which + // UrgentOOMProjectionBypassesCandidateGate below sets. This fixture's + // table predates that query (facdc70c0, 2026-08-17): the unwired entry + // null-crashed that test at start() - before its subject ever ran - + // for every full-suite run since. A fixed single processor keeps the + // auto-tuned pause/pain values deterministic regardless of the host. + static jvmtiError JNICALL mock_GetAvailableProcessors(jvmtiEnv *, + jint *nprocs) { + *nprocs = 1; + return JVMTI_ERROR_NONE; + } + + // Never invokes the callback - models a heap with nothing reachable from + // any root, so the very first pass completes immediately (0 admitted + // edges, not truncated). + static jvmtiError JNICALL mock_FollowReferences( + jvmtiEnv *, jint, jclass, jobject, const jvmtiHeapCallbacks *, + const void *) { + return JVMTI_ERROR_NONE; + } + + // runPassManualWalk()'s root enumeration - never invokes the root + // callback, same "nothing reachable from any root" heap model as + // mock_FollowReferences() above, so the first pass still completes + // immediately with 0 admitted edges. + static jvmtiError JNICALL mock_IterateOverReachableObjects( + jvmtiEnv *, jvmtiHeapRootCallback, jvmtiStackReferenceCallback, + jvmtiObjectReferenceCallback, const void *) { + return JVMTI_ERROR_NONE; + } + + // Same seeding helper as PollWatchedTargetsTest above (20 strictly- + // increasing samples - satisfies selectLeakCandidates()'s min-fill, + // growth/floor magnitude, and sustained-trend hysteresis requirements). + void seedGrowingCandidate(u32 klass_id, jweak rep) { + int slot; + bool created; + for (u16 i = 1; i <= 20; i++) { + LivenessTracker::instance()->klassPopulationRecordForTest( + klass_id, i, i, &slot, &created); + // Per-(klass, tid) qualification: selectLeakCandidates() also + // requires a qualifying allocating thread. These fixtures have + // no real tracked instances (mock JVMTI, no live heap), so a + // fixed synthetic tid exercises the gate without pretending to + // match any instance's real tid. + LivenessTracker::instance()->tidTrendRecordForTest( + klass_id, /*tid=*/4242, (u32)i, (u64)i); + } + LivenessTracker::instance()->klassPopulationSetRepresentativeForTest(nullptr, klass_id, rep); + } +}; + +TEST_F(SearchRestartTest, WithoutGenerationsSignalRestartStaysUnconditional) { + // gc_generations off (this fixture's SetUp default): canAffordNewSearch() + // has no candidate signal to gate on at all, so a terminal search is + // immediately eligible to restart - preserves this tracker's pre-restart + // behavior for a referencechains-without-generations setup (this class's + // own header comment, last paragraph). + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + + tracker->stop(); +} + +TEST_F(SearchRestartTest, GenerationsEnabledButNoCandidateBlocksFirstSearch) { + // A brand-new tracker must not pay for the initial whole-heap + // walk/tagging pass either when there is no leak candidate yet - + // shouldRunPass()'s !_search_started branch now shares + // canAffordNewSearch() with the restart gate below (this class's own + // header comment). + LivenessTracker::instance()->setGcGenerationsForTest(true); + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + EXPECT_FALSE(ReferenceChainsTestAccessor::shouldRunPass(1)); + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + EXPECT_EQ(0, tracker->passesRun()); + + int fake_object_storage = 0; + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)&fake_object_storage); + + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(2)); + + tracker->stop(); +} + +TEST_F(SearchRestartTest, GenerationsEnabledButNoCandidateBlocksRestart) { + LivenessTracker::instance()->setGcGenerationsForTest(true); + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + + // No leak candidate flagged - nothing to justify the cost of a restart. + EXPECT_FALSE(ReferenceChainsTestAccessor::shouldRunPass(1)); + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + + tracker->stop(); +} + +TEST_F(SearchRestartTest, RestartsOnceACandidateAppearsAndResetsPerSearchState) { + LivenessTracker::instance()->setGcGenerationsForTest(true); + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + ASSERT_EQ(1, tracker->passesRun()); + + int fake_object_storage = 0; + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)&fake_object_storage); + + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); // restartSearch() runs inline + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + EXPECT_EQ(0, tracker->passesRun()); // restartSearch() zeroed per-search state + + // The next runPass() call takes the "first pass of a search" branch + // again, exactly like a brand-new tracker. + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + EXPECT_EQ(1, tracker->passesRun()); + + tracker->stop(); +} + +TEST_F(SearchRestartTest, PainBudgetBlocksARestartUntilItDrains) { + LivenessTracker::instance()->setGcGenerationsForTest(true); + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:painbudget=1")); // 1% + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)&fake_object_storage); + + // First-ever search: called via runPass() directly here, bypassing + // shouldRunPass()'s canAffordNewSearch() gate entirely - the candidate + // seeded above would satisfy that gate anyway (this class's own header + // comment). + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + + // Restart #1: _safepoint_pain_budget has never had anything spent into it yet, so + // this is always immediately affordable regardless of this first + // search's own cost - the cost a search incurs only debits the *next* + // restart's affordability (restartSearch()'s own spend-then-reset + // order), not its own. + ASSERT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + + // Pretend this second search cost 1000ms of safepoint time - a mocked + // FollowReferences call in this fixture takes ~0 real wall-clock time, + // so this accessor stands in for what a real, expensive pass would have + // accumulated into _search_pain_ms on its own. + ReferenceChainsTestAccessor::setSearchPainMs(1000); + + // Restart #2: approved (nothing spent into _safepoint_pain_budget yet), and its + // own spend() call debits 1000ms into the balance for the *next* + // restart to contend with. + ASSERT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(2)); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + + // Restart #3, attempted immediately after restart #2's drain baseline: + // at 1% refill, 1000ms of debt needs 100000ms (1e11ns) of elapsed + // wall-clock time to clear - 1ns later is nowhere close. + EXPECT_FALSE(ReferenceChainsTestAccessor::shouldRunPass(3)); + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + + // Well past the drain point - the debt has cleared, restart #3 proceeds. + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(2ULL + 200000000000ULL)); + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + + tracker->stop(); +} + +// hasLeakSignal()'s OOM_URGENT_THRESHOLD_S fast path (referenceChains.h/.cpp): +// a heap-wide leak growing fast enough to project exhaustion sooner than the +// threshold must start a search immediately, without waiting for any klass +// to clear selectLeakCandidates()'s own per-klass ring-fill/hysteresis gate - +// this is the aggressive-leak gap GenerationsEnabledButNoCandidateBlocksFirstSearch +// above documents for the non-urgent case. Deliberately seeds no candidate at +// all (LivenessTracker::instance()->klassPopulationResetForTest() in SetUp +// leaves the population table empty) so this test can only pass via the +// heap-floor projection, never via selectLeakCandidates() falling back to a +// real candidate. +TEST_F(SearchRestartTest, UrgentOOMProjectionBypassesCandidateGate) { + LivenessTracker::instance()->setGcGenerationsForTest(true); + constexpr u64 SEC_NS = 1000000000ULL; + constexpr u64 MiB = 1ULL << 20; + // Same worked example as livenessTracker_ut.cpp's + // SecondsToOOMTest.RisingFloorProjectsExpectedSeconds: 700MiB rise over + // 7s against a 2800MiB max heap projects to 10s - comfortably under + // OOM_URGENT_THRESHOLD_S (5 minutes). + LivenessTracker::instance()->setMaxHeapBytesForTest((jlong)(2800 * MiB)); + for (int i = 0; i < 10; i++) { + LivenessTracker::instance()->heapFloorRecordForTest( + 1000 * MiB + (u64)i * 100 * MiB, (u64)i * SEC_NS); + } + + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// Phase 5 - correctness hardening: durability re-verification. +// +// These tests drive maybeUpgradeRootAttachedRootKind()/ +// collectStaleRootKindEntriesForRotation() directly via +// ReferenceChainsTestAccessor rather than through a full +// IterateOverReachableObjects()-driven runPassManualWalk() pass: neither +// IterateOverReachableObjects nor FollowReferences-as-a-safepoint-pin is +// mocked in this file (see the fixture's own FollowReferences-only mock +// rationale above), and both methods are pure FrontierTable/queue logic with +// no JVMTI dependency of their own - the same rationale +// admitObject()/rootKindDurability() being free of any callback shape +// already established for this subsystem. +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsBfsTest, StaleRootAttributionUpgradesOnRediscovery) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // Synthetic stack-local root: admitted, root-attached (parent_tag == 0), + // its owning frame has since "gone away" from the design doc's scenario + // (nothing further to model here - the entry simply stays as-is until a + // more durable root is discovered). + jlong tag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, /*parent_tag=*/0, /*depth=*/0, + FrontierEntryState::EXPANDED, JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + + // A second, equally-or-less durable root discovery does not overwrite + // the recorded root_kind. + EXPECT_FALSE(ReferenceChainsTestAccessor::maybeUpgradeRootAttachedRootKind( + frontier, tag, JVMTI_HEAP_REFERENCE_JNI_LOCAL)); + FrontierEntry entry{}; + ASSERT_TRUE(frontier->lookup(tag, &entry)); + EXPECT_EQ(JVMTI_HEAP_REFERENCE_STACK_LOCAL, entry.root_kind); + + // A durable root (JNI global) attaching to the same object upgrades it. + EXPECT_TRUE(ReferenceChainsTestAccessor::maybeUpgradeRootAttachedRootKind( + frontier, tag, JVMTI_HEAP_REFERENCE_JNI_GLOBAL)); + ASSERT_TRUE(frontier->lookup(tag, &entry)); + EXPECT_EQ(JVMTI_HEAP_REFERENCE_JNI_GLOBAL, entry.root_kind); + EXPECT_EQ(0, entry.parent_tag); // still root-attached, unchanged + + // An even less durable root discovered afterwards cannot downgrade it. + EXPECT_FALSE(ReferenceChainsTestAccessor::maybeUpgradeRootAttachedRootKind( + frontier, tag, JVMTI_HEAP_REFERENCE_MONITOR)); + ASSERT_TRUE(frontier->lookup(tag, &entry)); + EXPECT_EQ(JVMTI_HEAP_REFERENCE_JNI_GLOBAL, entry.root_kind); + + tracker->stop(); +} + +// Exercises the invariant conflict Phase 5 itself calls out: a non-root +// entry (parent_tag != 0) rediscovered as if via a root context must never +// have its root_kind overwritten - doing so would leave a non-zero root_kind +// on an entry nothing else treats as root-attached (referenceChains.h's +// FrontierEntry::root_kind comment), since this mutator never touches +// parent_tag. This is the edge-based, non-root-Y re-expansion case the +// option (a) resolution above exists for. +TEST_F(ReferenceChainsBfsTest, NonRootAttachedEntryNeverUpgraded) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // Parent Y (root-attached) and child X, admitted the way frontier + // re-expansion admits a non-root child: non-root + // (parent_tag == Y's tag), root_kind == 0. + jlong yTag = 1; + jlong xTag = 2; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, yTag, /*parent_tag=*/0, /*depth=*/0, + FrontierEntryState::EXPANDED, JVMTI_HEAP_REFERENCE_JNI_GLOBAL)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, xTag, /*parent_tag=*/yTag, /*depth=*/1, + FrontierEntryState::EXPANDED, /*root_kind=*/0)); + + // Re-expanding Y rediscovers an edge to X (already tracked) - even if + // this rediscovery is (incorrectly) attempted with a durable root_kind, + // it must be rejected because X is not root-attached. + EXPECT_FALSE(ReferenceChainsTestAccessor::maybeUpgradeRootAttachedRootKind( + frontier, xTag, JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + FrontierEntry entry{}; + ASSERT_TRUE(frontier->lookup(xTag, &entry)); + EXPECT_EQ(0, entry.root_kind); + EXPECT_EQ(yTag, entry.parent_tag); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, RotationSelectsOnlyTransientExpandedRootAttachedEntries) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // Eligible: root-attached, EXPANDED, transient root_kind. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + // Not eligible: durable root_kind. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 2, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_JNI_GLOBAL)); + // Not eligible: transient but still FRONTIER, not yet EXPANDED. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 3, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_JNI_LOCAL)); + // Not eligible: transient root_kind but not root-attached. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 4, /*parent_tag=*/1, 1, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_JNI_LOCAL)); + // Eligible: root-attached, EXPANDED, transient (JNI local this time). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 5, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_JNI_LOCAL)); + + std::vector selected = + ReferenceChainsTestAccessor::collectStaleRootKindEntriesForRotation(10); + std::sort(selected.begin(), selected.end()); + EXPECT_EQ((std::vector{1, 5}), selected); + + // Selected tags are queued for re-expansion, exactly like an ordinary + // admission would queue a newly-discovered tag. + EXPECT_EQ(2u, ReferenceChainsTestAccessor::priorityExpandSize()); + + tracker->stop(); +} + +// N transient-root_kind entries, rotation size R: every entry must be +// selected at least once within ceil(N/R) calls, regardless of where the +// cursor happened to start. +TEST_F(ReferenceChainsBfsTest, RotationCoversAllEntriesWithinCeilNOverR) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + const int N = 10; + const int R = 3; + for (jlong tag = 1; tag <= N; tag++) { + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + } + + std::unordered_set covered; + int calls = (N + R - 1) / R; + for (int i = 0; i < calls; i++) { + std::vector selected = + ReferenceChainsTestAccessor::collectStaleRootKindEntriesForRotation(R); + for (jlong tag : selected) { + covered.insert(tag); + } + } + EXPECT_EQ((size_t)N, covered.size()); + + tracker->stop(); +} + +// collectStaleExpandedEntriesForRotation()'s own EXPANDED-only criterion is a +// strict superset of collectStaleRootKindEntriesForRotation()'s (which also +// requires parent_tag == 0 and a transient root_kind), and runPassManualWalk() +// calls the root-kind collector first, into the very same _priority_expand +// deque. Without a dedup check, a tag the root-kind collector already queued +// would be queued a second time by the EXPANDED-only sweep, wasting one of +// expandFrontier()'s per-entry batch slots on an already-EXPANDED tag every +// pass. This drives both collectors back-to-back, the way runPassManualWalk() +// does, and asserts _priority_expand ends up with no duplicate tags. +TEST_F(ReferenceChainsBfsTest, StaleExpandedRotationDoesNotDuplicateRootKindSelection) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // Eligible for both collectors: EXPANDED, root-attached, transient + // root_kind - exactly the overlap collectStaleRootKindEntriesForRotation() + // will pick up first. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + // Eligible only for the EXPANDED-only sweep: EXPANDED but not root-attached. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 2, /*parent_tag=*/1, 1, FrontierEntryState::EXPANDED, + /*root_kind=*/0)); + + std::vector root_kind_selected = + ReferenceChainsTestAccessor::collectStaleRootKindEntriesForRotation( + ReferenceChainsTestAccessor::rootKindRotationBudget()); + EXPECT_EQ((std::vector{1}), root_kind_selected); + + std::vector stale_expanded_selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation( + ReferenceChainsTestAccessor::staleExpandedRotationBudget()); + // Tag 1 is already queued from the root-kind collector above and must not + // be selected again; tag 2 is newly discovered by this sweep. + EXPECT_EQ((std::vector{2}), stale_expanded_selected); + + std::vector queued = ReferenceChainsTestAccessor::priorityExpandContents(); + EXPECT_EQ((std::vector{1, 2}), queued); + std::unordered_set unique_queued(queued.begin(), queued.end()); + EXPECT_EQ(queued.size(), unique_queued.size()); + + tracker->stop(); +} + +// A tag left over in _priority_expand from a prior pass's truncated +// expandFrontier() batch (see expandFrontier()'s own "leave the batch at the +// front of the source queue for a later pass to retry" comment) must also be +// skipped by collectStaleExpandedEntriesForRotation() - not just tags queued +// by collectStaleRootKindEntriesForRotation() earlier in the same call. +TEST_F(ReferenceChainsBfsTest, StaleExpandedRotationSkipsPreexistingQueueEntries) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + + // Simulate a truncated batch from a prior pass still sitting at the front + // of _priority_expand, without going through the root-kind collector at + // all - the leftover entry alone must still be enough to suppress a + // duplicate. + ReferenceChainsTestAccessor::pushPriorityExpand(1); + + std::vector stale_expanded_selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation( + ReferenceChainsTestAccessor::staleExpandedRotationBudget()); + EXPECT_TRUE(stale_expanded_selected.empty()); + + std::vector queued = ReferenceChainsTestAccessor::priorityExpandContents(); + EXPECT_EQ((std::vector{1}), queued); + + tracker->stop(); +} + +// End-to-end proof of the prof-analyzer-hotdog-jb pod's actual leak shape: +// a static-field-rooted collection (like ProfileAnalyzer.LEAK_BUFFER) whose +// owning node is admitted and fully EXPANDED once, then has a *new* element +// appended to it afterward - mirroring a Java List field being mutated in +// place, never reassigned, well after admitStaticFieldRoots()'s one-time +// sweep. The critical property under test is that this new element is +// discovered by collectStaleExpandedEntriesForRotation()'s rotation without +// the overall search ever reaching SearchState::COMPLETED - i.e. without +// requiring a full heap walk to finish, which on a multi-GiB heap can take +// far longer than the pod can tolerate between the leaked field's own +// growth events. A large distractor root chain (never fully drained within +// this test's bounded pass loops) keeps the search perpetually RUNNING so +// this property is exercised directly, not sidestepped by letting the +// search finish and then trivially re-discovering everything from scratch. +TEST_F(ReferenceChainsBfsTest, RotationDiscoversLateElementOfExpandedStaticFieldCollectionWithoutSearchCompleting) { + Arguments args; + // budget=8 -> rotation_reserved_budget = min(8/2, 272) = 4, ordinary = 4: + // both slices non-zero, unlike a budget=1 pattern which would zero out + // rotation's reserved slice entirely (min(0, 272) == 0). + ASSERT_FALSE(args.parse("referencechains=true:hops=5000:budget=8:firstpassbudget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int classNode = addNode(); + int listNode = addNode(); + int seedChildNode = addNode(); + int lateChildNode = addNode(); + + // Distractor chain: a long, independently root-seeded chain that never + // fully drains within this test's bounded pass loops below, so the + // overall search always has forward progress available and never + // reaches SearchState::COMPLETED (nor NO_PROGRESS_PASS_LIMIT-triggered + // ABANDONED) purely as a side effect of this test's own loop bounds. + const int kDistractorNodes = 500; + std::vector distractor(kDistractorNodes); + for (int i = 0; i < kDistractorNodes; i++) { + distractor[i] = addNode(); + } + + // addClass() captures classNode's address in node_tags' backing storage - + // must come after every addNode() call above (including the distractor + // loop), or a later push_back reallocating node_tags would silently + // leave this pointer dangling (indexOfNode() would then never match it). + addClass((void *)&node_tags[classNode], "Lcom/rc/statics/GrowingListHolder;"); + + script = { + // listNode is retained only via classNode's static field - the same + // shape as DiscoversObjectRetainedOnlyByStaticField above. + {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, listNode, -1}, + // listNode's one pre-existing element, discovered the first time + // listNode itself is expanded. + {JVMTI_HEAP_REFERENCE_FIELD, listNode, seedChildNode, -1}, + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, distractor[0], -1}, + }; + for (int i = 0; i + 1 < kDistractorNodes; i++) { + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, distractor[i], distractor[i + 1], -1}); + } + + // Phase 1: run passes until listNode has been fully expanded (its one + // pre-existing child discovered), without ever letting the search + // complete. + bool truncated = true; + FrontierEntry listEntry{}; + bool listExpanded = false; + for (int i = 0; i < 200 && !listExpanded; i++) { + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + jlong listTag = tags_ever_assigned[listNode]; + if (listTag != 0 && tracker->frontierTable()->lookup(listTag, &listEntry) + && listEntry.state == FrontierEntryState::EXPANDED) { + listExpanded = true; + } + } + ASSERT_TRUE(listExpanded); + ASSERT_NE(0, tags_ever_assigned[seedChildNode]); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + + // Phase 2: simulate a new element appended to the leaking static + // field's list *after* listNode's one-time expansion - the exact + // "growing collection" shape found in the real pod's leak generator. + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, listNode, lateChildNode, -1}); + + for (int i = 0; i < 200 && tags_ever_assigned[lateChildNode] == 0; i++) { + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + } + + // The late element was discovered purely via rotation re-expanding + // listNode - and, critically, without the search ever completing (no + // dependency on a full heap walk finishing). + ASSERT_NE(0, tags_ever_assigned[lateChildNode]); + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + + std::vector chain; + ASSERT_TRUE(tracker->frontierTable()->reconstructChain( + tags_ever_assigned[lateChildNode], &chain)); + FrontierEntry lateEntry{}; + ASSERT_TRUE(tracker->frontierTable()->lookup( + tags_ever_assigned[lateChildNode], &lateEntry)); + EXPECT_EQ(tags_ever_assigned[listNode], lateEntry.parent_tag); + + tracker->stop(); +} + +// Proof of the fix for the actual prof-analyzer-hotdog-jb stall: +// collectStaleExpandedEntriesForRotation() (referenceChains.cpp) used to +// always rescan FrontierTable slots starting from tag 1, unlike its sibling +// collectStaleRootKindEntriesForRotation() which already carried its own +// persistent cursor. isQueuedForRotation()'s dedup check only looks at the +// CURRENT pass's _priority_expand (drained by expandFrontier() at the end of +// that same pass - see clearPriorityExpand()'s own comment), so without a +// cursor, later passes had no memory of what earlier passes already +// selected: whenever a real heap's frontier table held +// STALE_EXPANDED_ROTATION_BUDGET (256) or more low-tag EXPANDED entries that +// stay EXPANDED forever (long-lived infrastructure objects - exactly what +// the sweep's own comment says it favors), that population alone filled the +// sweep's 256-entry-per-pass cap on every single call, permanently starving +// any EXPANDED entry with a higher tag (e.g. a static field's collection +// node, admitted only once its owning class first loads, well after +// startup) of ever being re-queued - a bug proved directly, before the fix, +// by this same test (then named +// StaleExpandedRotationStarvesHighTagEntryBehindLowTagPopulation). +// +// _stale_expanded_rotation_cursor now makes collectStaleExpandedEntriesForRotation() +// resume from where the previous call left off instead of always restarting +// at tag 1, the same wrapping-cursor guarantee +// RotationCoversAllEntriesWithinCeilNOverR above already proves for +// collectStaleRootKindEntriesForRotation(): every entry, including one +// sitting behind an arbitrarily large low-tag population, gets a turn within +// ceil(table_size / max_count) calls. +// +// Driven directly against collectStaleExpandedEntriesForRotation() (the same +// unit-level style as RotationCoversAllEntriesWithinCeilNOverR above) rather +// than through a full JVMTI-mocked BFS walk: this property is intrinsic to +// the selection function's own tag-order scan, so it needs neither a real +// graph nor runPass()'s pacing/budget machinery to demonstrate. +TEST_F(ReferenceChainsBfsTest, StaleExpandedRotationCoversHighTagEntryBehindLowTagPopulationWithinBoundedPasses) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + const int lowTagBudget = ReferenceChainsTestAccessor::staleExpandedRotationBudget(); + // Comfortably above the 256-entry cap, so the low-tag population alone + // would fill every sweep before an always-from-1 scan could ever reach + // the high-tag entry below - mirrors a real multi-GiB heap's frontier + // table, which accumulates far more than 256 long-lived, perpetually- + // EXPANDED entries (bootstrap classes, caches, etc.) well before any one + // leak-candidate class even loads. + const int lowTagPopulation = lowTagBudget + 50; + for (jlong tag = 1; tag <= lowTagPopulation; tag++) { + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + } + + // The leak candidate's own owning node - e.g. LEAK_BUFFER's list, admitted + // via a static field only once its class loads, well after the JVM's own + // bootstrap population already occupies every low tag number. + const jlong highTag = lowTagPopulation + 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, highTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + + // Several simulated passes: each iteration mirrors one real pass - + // collectStaleExpandedEntriesForRotation() runs once, then + // clearPriorityExpand() mirrors expandFrontier() having drained whatever + // it selected before the next pass's sweep resumes from the cursor. + const int table_size = lowTagPopulation + 1; + const int calls = (table_size + lowTagBudget - 1) / lowTagBudget; + bool highTagSelected = false; + std::unordered_set covered; + for (int pass = 0; pass < calls && !highTagSelected; pass++) { + std::vector selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation( + lowTagBudget); + for (jlong tag : selected) { + covered.insert(tag); + if (tag == highTag) { + highTagSelected = true; + } + } + ReferenceChainsTestAccessor::clearPriorityExpand(); + } + + EXPECT_TRUE(highTagSelected) + << "highTag was never selected within ceil(table_size / max_count) " + "passes - the fix's coverage guarantee does not hold"; + EXPECT_EQ((size_t)table_size, covered.size()); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// trackLeakAccumulation() - the admission-time hook (called from +// admitObject(), the single shared admission path) that aggregates by +// (leaf_class_tag, parent_class_tag) signature and by individual parent +// fanout - stable JVMTI class tags (classTagAllocator.h), not classMap +// dictionary ids, precisely because that dictionary can be compacted/ +// regenerated independently, silently reassigning the same class a +// different id at different times (found via the external-process test - +// see class_tag's own comment, referenceChains.h). Driven directly, pure +// FrontierTable/map logic with no JVMTI dependency of its own. +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationAggregatesBySignatureAndFanout) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987; + constexpr u32 kParent1Klass = 100, kParent2Klass = 200; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + jlong parent1Tag = 1, parent2Tag = 2; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parent1Tag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParent1Klass)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parent2Tag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParent2Klass)); + + // 3 children of the watched leaf klass under parent1, 1 under parent2 - + // each call simulates one admission (the childTag argument is only used + // by production code for logging/future use, not read by this method). + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parent1Tag, 10); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parent1Tag, 11); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parent1Tag, 12); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parent2Tag, 20); + + EXPECT_EQ(3u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafKlass, kParent1Klass)); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafKlass, kParent2Klass)); + EXPECT_EQ(3u, ReferenceChainsTestAccessor::leakParentFanout(parent1Tag)); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(parent2Tag)); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationSkipsUnwatchedKlass) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({987}); + jlong parentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, /*class_tag=*/100)); + + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, /*class_tag=*/555, + parentTag, 10); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()); + EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationSkipsRootAttachedChild) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({987}); + + // parent_tag == 0 - a root-attached leaf itself, nothing to attribute a + // container to. + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, 987, /*parent_tag=*/0, 10); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationSkipsWhenParentNotFound) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({987}); + + // parent_tag=99 was never inserted - graceful no-op, not a crash. + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, 987, /*parent_tag=*/99, 10); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()); + + tracker->stop(); +} + +// Proof of the actual bug this design was found fixing: the classMap +// dictionary id (referrer_klass) for the exact same class can differ +// depending on which subsystem/generation resolved it (see class_tag's own +// comment, referenceChains.h, for the real-world case - "[B" resolving to +// two different classMap ids for LivenessTracker vs. ReferenceChainTracker). +// Matching must work via class_tag regardless of what referrer_klass says. +TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationMatchesByClassTagEvenWhenReferrerKlassDiffers) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafClassTag = 987; + constexpr u32 kParentClassTag = 100; + // Deliberately different, "wrong" classMap ids - simulating exactly the + // compaction/regeneration scenario that broke referrer_klass-based + // matching. If matching used referrer_klass at all, this test would fail. + constexpr u32 kParentStaleReferrerKlass = 555555; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafClassTag}); + + jlong parentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, kParentStaleReferrerKlass, + kParentClassTag)); + + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafClassTag, + parentTag, 10); + + EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafClassTag, + kParentClassTag)); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// collectLeakAccumulationCandidatesForRotation() - the two-tier design +// itself (see its own header comment, referenceChains.cpp, for the full +// rationale). Driven directly against the aggregation state +// trackLeakAccumulation() above populates, the same unit-level style as the +// other rotation collectors. +// --------------------------------------------------------------------------- + +// The central discriminating test for the whole design (per the "ubiquitous +// common leaf class held by many small unrelated parents" concern this +// design exists to solve): a signature with a LARGE but FLAT total (many +// unrelated parents, e.g. a common leaf class scattered across a real +// classpath) must NOT outrank a signature with a SMALLER but GROWING total +// (the actual leak) once a growth history exists - retained-size-style +// ranking alone would pick the wrong one every time. +TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationPrioritizesGrowingSignatureOverLargeFlatOne) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987; + constexpr u32 kGrowingParentKlass = 100; // signature A: the real leak + constexpr u32 kUbiquitousParentKlass = 999; // signature B: common, but flat + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + // Signature A: one parent, growing. + jlong growingParentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, growingParentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kGrowingParentKlass)); + + // Signature B: 20 distinct, unrelated parents, each holding just 1-2 + // instances of the same common leaf klass - a much LARGER total than A, + // but it will not grow between passes. + constexpr int kUbiquitousParentCount = 20; + std::vector ubiquitousParentTags; + for (int i = 0; i < kUbiquitousParentCount; i++) { + jlong tag = 100 + i; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kUbiquitousParentKlass)); + ubiquitousParentTags.push_back(tag); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, tag, 1000 + i); + } + // Pass 1: A has fanout 5, B has total 20 (20 parents x 1 each) - B is + // larger. First-ever call has no prior snapshot, so both deltas equal + // their totals; B legitimately wins this one call (nothing to compare + // growth against yet). + for (int i = 0; i < 5; i++) { + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, + growingParentTag, 2000 + i); + } + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation( + ReferenceChainsTestAccessor::leakAccumulationRotationBudget()); + + // Pass 2: B stays exactly flat (no new admissions); A grows from 5 to 8. + for (int i = 0; i < 3; i++) { + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, + growingParentTag, 3000 + i); + } + std::vector selected = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation( + ReferenceChainsTestAccessor::leakAccumulationRotationBudget()); + + ASSERT_EQ(1u, selected.size()); + EXPECT_EQ(growingParentTag, selected[0]) + << "the growing signature's parent must be selected, even though " + "the flat-but-larger signature has a much bigger absolute total"; + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationRanksByFanoutWithinWinningSignature) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kParentKlass = 100; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + jlong lowFanoutTag = 1, highFanoutTag = 2; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, lowFanoutTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, highFanoutTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, lowFanoutTag, 10); + for (int i = 0; i < 5; i++) { + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, + highFanoutTag, 20 + i); + } + + std::vector selected = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); + ASSERT_EQ(2u, selected.size()); + EXPECT_EQ(highFanoutTag, selected[0]) << "higher fanout ranks first"; + EXPECT_EQ(lowFanoutTag, selected[1]); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationRespectsMaxCountAndDedup) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kParentKlass = 100; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + jlong tag1 = 1, tag2 = 2, tag3 = 3; + for (jlong tag : {tag1, tag2, tag3}) { + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, tag, 10); + } + // tag2 already queued from an earlier collector this same pass - must + // be skipped even though it qualifies structurally. + ReferenceChainsTestAccessor::pushPriorityExpand(tag2); + + std::vector selected = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation( + /*max_count=*/1); + EXPECT_EQ(1u, selected.size()) << "capped at max_count"; + EXPECT_NE(tag2, selected[0]) << "already-queued tag must not be re-selected"; + + tracker->stop(); +} + +// Reversed on round-4 pod evidence (ev-leaktag-onpod-round4): the previous +// EXPANDED-only selection made the targeted tier select ZERO every pass +// on a live leak - the growing holders are un-expanded FRONTIER-state +// backlog entries that the starved pending lane never reaches (a 127k +// backlog at ~120-200 objects/min). A FRONTIER-state parent must be +// selected AND placed at the head of the priority lane so the next +// expandFrontier() batch reaches it ahead of the stale re-walks already +// queued (the same pod's priority deque held ~1016 entries). +TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationSelectsUnexpandedFrontierParentAheadOfBacklog) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kParentKlass = 100; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + // Stale re-walks already sitting in the priority lane (push_back, as + // the other two collectors do). + ReferenceChainsTestAccessor::pushPriorityExpand(900); + ReferenceChainsTestAccessor::pushPriorityExpand(901); + + jlong notYetExpandedTag = 1, expandedLowFanoutTag = 2; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, notYetExpandedTag, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, expandedLowFanoutTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + for (int i = 0; i < 10; i++) { + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, + notYetExpandedTag, 10 + i); + } + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, + expandedLowFanoutTag, 100); + + std::vector selected = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); + ASSERT_EQ(2u, selected.size()); + EXPECT_EQ(notYetExpandedTag, selected[0]) + << "the FRONTIER-state parent qualifies and outranks the " + "lower-fanout EXPANDED one"; + EXPECT_EQ(expandedLowFanoutTag, selected[1]); + + std::vector queue = ReferenceChainsTestAccessor::priorityExpandContents(); + ASSERT_GE(queue.size(), 4u); + EXPECT_EQ(notYetExpandedTag, queue[0]) + << "the targeted un-expanded holder must JUMP the backlog, not " + "queue behind the stale re-walks"; + EXPECT_EQ(expandedLowFanoutTag, queue[1]) + << "selection order must be preserved at the head (fanout rank)"; + EXPECT_EQ(900, queue[2]); + EXPECT_EQ(901, queue[3]); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationReturnsEmptyWhenNothingHasGrownSincePreviousPass) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kParentKlass = 100; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + jlong parentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parentTag, 10); + + // First call establishes the baseline (delta == total, since there is no + // prior snapshot) and selects it. + std::vector firstPass = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); + ASSERT_EQ(1u, firstPass.size()); + ReferenceChainsTestAccessor::clearPriorityExpand(); + + // Second call, nothing new admitted - delta is now 0 for every + // signature, so nothing should be selected. + std::vector secondPass = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); + EXPECT_TRUE(secondPass.empty()) + << "no signature grew since the previous pass's snapshot"; + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationReturnsEmptyWhenNoSignaturesTracked) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + std::vector selected = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); + EXPECT_TRUE(selected.empty()); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// seedLeakAccumulationForNewlyWatchedKlass() - the cold-start fix: retroactively +// seeds the aggregation from entries ALREADY admitted before a klass_id +// started being watched, since trackLeakAccumulation() alone only ever sees +// admissions happening after watching starts, and the container that +// actually needs re-expansion is typically already fully admitted by then +// (found via the external-process test: the delegate ArrayList sits one hop +// below the root-attached wrapper, and admitStaticFieldRoots()'s own sweep +// admits both in the same call, long before any leak signal can plausibly +// have fired). +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsBfsTest, SeedLeakAccumulationPopulatesFromAlreadyAdmittedEntries) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kParentKlass = 100; + // Entries inserted directly (as if admitted by an earlier pass), with no + // watched klass_id set at all yet at insertion time - trackLeakAccumulation() + // was never called for any of these. + jlong parentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + for (int i = 0; i < 4; i++) { + jlong childTag = 10 + i; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, childTag, parentTag, 1, FrontierEntryState::EXPANDED, + /*root_kind=*/0, /*referrer_klass=*/0, kLeafKlass)); + } + ASSERT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()) + << "nothing tracked yet - trackLeakAccumulation() was never called"; + + ReferenceChainsTestAccessor::seedLeakAccumulationForNewlyWatchedKlass(kLeafKlass); + + EXPECT_EQ(4u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafKlass, kParentKlass)); + EXPECT_EQ(4u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, SeedLeakAccumulationSkipsNonMatchingAndNonExpandedEntries) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kOtherKlass = 555, kParentKlass = 100; + jlong parentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + // Wrong class - must not be counted. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 10, parentTag, 1, FrontierEntryState::EXPANDED, + /*root_kind=*/0, /*referrer_klass=*/0, kOtherKlass)); + // Right class, but still FRONTIER (not yet EXPANDED) - must not be counted. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 11, parentTag, 1, FrontierEntryState::FRONTIER, + /*root_kind=*/0, /*referrer_klass=*/0, kLeafKlass)); + // Right class, root-attached (no real parent) - must not be counted. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 12, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kLeafKlass)); + + ReferenceChainsTestAccessor::seedLeakAccumulationForNewlyWatchedKlass(kLeafKlass); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, SeedLeakAccumulationComposesWithOngoingIncrementalUpdates) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kParentKlass = 100; + jlong parentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 10, parentTag, 1, FrontierEntryState::EXPANDED, + /*root_kind=*/0, /*referrer_klass=*/0, kLeafKlass)); + + // Retroactive seed sees the one pre-existing child. + ReferenceChainsTestAccessor::seedLeakAccumulationForNewlyWatchedKlass(kLeafKlass); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); + + // A genuinely new admission after watching starts must add on top of the + // retroactive baseline, not reset or double it. + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parentTag, 11); + + EXPECT_EQ(2u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); + EXPECT_EQ(2u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafKlass, kParentKlass)); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// Smoke test simulating the hotdog pod conditions that starved BFS: +// +// 1. GetObjectsWithTags quadratic bottleneck: a large frontier backlog +// makes each GetObjectsWithTags call expensive. The self-calibrating +// adaptive batch_size must keep it bounded. +// +// 2. Shared deadline bug: the static-field sweep's FollowReferences ate +// the entire per-pass wall-clock deadline, leaving expand with zero +// time. The deadline split gives each sub-operation its own fresh +// deadline. +// +// 3. Rolling resume: when FollowReferences truncates mid-batch (budget +// exhausted), the fully-processed entries must be popped (mark +// EXPANDED) and only the partially-processed + unvisited entries left +// for retry. +// +// This test builds a graph with a static-field root leading to a chain of +// objects (simulating the leaking collection), plus a small set of +// distractor roots to keep the search RUNNING. It runs passes with a +// small budget so expand truncates mid-batch, then verifies the rolling +// resume and progress properties. +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsBfsTest, RollingResumePopsProcessedEntriesOnTruncatedBatch) { + Arguments args; + // budget=4: small enough that expand truncates mid-batch after admitting + // a few children. firstpassbudget=1000: large enough to enumerate all + // roots in the first pass without truncating root enum. + ASSERT_FALSE(args.parse( + "referencechains=true:hops=5000:budget=4:firstpassbudget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // A static-field root: classNode -> listNode (the leaking collection). + int classNode = addNode(); + int listNode = addNode(); + + // A chain of 20 children hanging off listNode. With budget=4, the + // callback admits 4 children then returns JVMTI_VISIT_ABORT + // (BUDGET_EXHAUSTED), truncating mid-batch. + constexpr int kChainLen = 20; + std::vector chainNodes(kChainLen); + for (int i = 0; i < kChainLen; i++) { + chainNodes[i] = addNode(); + } + + // Distractor roots: 20 independent JNI-global roots, each with one child. + // Enough to keep the search RUNNING but small enough to drain quickly. + constexpr int kDistractors = 20; + std::vector distractorRoots(kDistractors); + std::vector distractorChildren(kDistractors); + for (int i = 0; i < kDistractors; i++) { + distractorRoots[i] = addNode(); + distractorChildren[i] = addNode(); + } + + // addClass() must come after all addNode() calls. + addClass((void *)&node_tags[classNode], "Lcom/rc/SmokeTestHolder;"); + + script = { + {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, listNode, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, listNode, chainNodes[0], -1}, + }; + for (int i = 0; i + 1 < kChainLen; i++) { + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, chainNodes[i], chainNodes[i + 1], -1}); + } + for (int i = 0; i < kDistractors; i++) { + script.push_back({JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, distractorRoots[i], -1}); + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, distractorRoots[i], distractorChildren[i], -1}); + } + + // Phase 1: run passes until listNode is admitted via the static-field sweep. + bool truncated = true; + jlong listTag = 0; + for (int i = 0; i < 200 && listTag == 0; i++) { + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + listTag = tags_ever_assigned[listNode]; + } + ASSERT_NE(0, listTag) << "listNode was never admitted to the frontier"; + + // Phase 2: run passes until listNode is expanded (rolling resume pops it). + for (int i = 0; i < 200; i++) { + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + FrontierEntry entry{}; + if (frontier->lookup(listTag, &entry) && + entry.state == FrontierEntryState::EXPANDED) { + break; + } + } + FrontierEntry listEntry{}; + ASSERT_TRUE(frontier->lookup(listTag, &listEntry)); + EXPECT_EQ(FrontierEntryState::EXPANDED, listEntry.state) + << "listNode should be EXPANDED after rolling resume popped it"; + + // Verify some chain children were admitted. + int admittedChildren = 0; + for (int i = 0; i < kChainLen; i++) { + if (tags_ever_assigned[chainNodes[i]] != 0) admittedChildren++; + } + EXPECT_GT(admittedChildren, 0) + << "No chain children were admitted — expand never ran"; + + // Phase 3: run more passes until all chain children are admitted. + for (int i = 0; i < 500 && admittedChildren < kChainLen; i++) { + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + admittedChildren = 0; + for (int j = 0; j < kChainLen; j++) { + if (tags_ever_assigned[chainNodes[j]] != 0) admittedChildren++; + } + } + EXPECT_EQ(kChainLen, admittedChildren) + << "Not all chain children were admitted within bounded passes"; + + tracker->stop(); +} + +// Verify the AIMD adaptive batch_size: with the per-call EMA over the CPU +// budget, expandFrontier should multiplicatively decrease the batch; under +// the budget it should additively increase toward the cap. The mock +// GetObjectsWithTags is instant (no real tag-map cost), so we drive the EMA +// by hand and verify the AIMD response, not the timing. +TEST_F(ReferenceChainsBfsTest, AdaptiveBatchSizeProportionalToWindow) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // Adaptive-batch state is zeroed by reset() (SetUp) but zeroed here too + // for the same reason as before: exact per-phase arithmetic below. + ReferenceChainsTestAccessor::setGotwEmaCallNs(0); + ReferenceChainsTestAccessor::setGotwBatchSize(0); + ReferenceChainsTestAccessor::setPassDeadlineNs(0); + + // Seed a frontier root manually (mirrors PollWatchedTargetsTest's + // seeding style): node carries frontier tag 1, pending expansion has + // exactly that tag. No runPass() - a pass would drain the tiny graph to + // COMPLETED and release all tags, and its rotation phase adds extra + // GetObjectsWithTags calls, both of which break per-call arithmetic. + // The script stays empty until the admission-sanity phase below, so + // each expandFrontier() drive runs exactly one batch (one control + // update) and admits nothing. + int rootNode = addNode(); + int childNode = addNode(); + node_tags[rootNode] = 1; + ASSERT_TRUE(tracker->frontierTable()->insert( + 1, 0, 1, 0, FrontierEntryState::EDGE)); + ReferenceChainsTestAccessor::pushPendingExpandForTest(1); + int edges = 0; + const u64 budget = ReferenceChainsTestAccessor::gotwCpuBudgetNs(); + + // --- Populate phase: first GetObjectsWithTags call. The EMA should be + // non-zero afterwards, and the near-zero mock call time means the + // window (nominal budget, no deadline) fits ~unbounded many calls - + // the proportion scales the batch all the way to the cap. + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + EXPECT_NE(0u, ReferenceChainsTestAccessor::gotwEmaCallNs()) + << "per-call EMA should be populated after first GetObjectsWithTags"; + EXPECT_EQ(ReferenceChainsTestAccessor::gotwMaxBatch(), + ReferenceChainsTestAccessor::gotwBatchSize()) + << "near-free call should scale the batch to the cap"; + + // --- Shrink phase: EMA at 2x the window with no deadline -> batch + // halves (512 x 1 / 1.6 after the EMA update). + ReferenceChainsTestAccessor::setGotwEmaCallNs(budget * 2); + ReferenceChainsTestAccessor::setGotwBatchSize(512); + ReferenceChainsTestAccessor::pushPendingExpandForTest(1); + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + // EMA after the call: 2x window x 0.8 + mock elapsed/5 - slightly + // above 1.6x window, so the exact expectation is computed from the + // actual EMA the same way the control law does (window = nominal + // budget, no deadline): next = 512 x window / ema. + EXPECT_EQ((size_t)(512ULL * budget / + std::max(ReferenceChainsTestAccessor::gotwEmaCallNs(), + 1ULL)), + ReferenceChainsTestAccessor::gotwBatchSize()) + << "EMA at ~1.6x the window should scale the batch to 512/1.6"; + + // --- Grow phase: EMA at half the window -> batch scales up 2.5x, + // i.e. the floor-dominated regime GROWS the batch (the whole point of + // the proportional law - the old AIMD could not grow past a fixed + // budget even when bigger batches were nearly free). + ReferenceChainsTestAccessor::setGotwBatchSize(64); + ReferenceChainsTestAccessor::setGotwEmaCallNs(budget / 2); + ReferenceChainsTestAccessor::pushPendingExpandForTest(1); + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + // Same computation from the actual post-call EMA (~0.4x window): + // next = 64 x window / ema. + EXPECT_EQ((size_t)(64ULL * budget / + std::max(ReferenceChainsTestAccessor::gotwEmaCallNs(), + 1ULL)), + ReferenceChainsTestAccessor::gotwBatchSize()) + << "EMA under the window should scale the batch up proportionally"; + + // --- Deadline-window phase: with a live pass deadline the window is the + // REMAINING time, not the nominal budget. A deadline 10x the budget out + // with EMA ~ 1x budget scales the batch 10x/0.8 - past the cap, so the + // clamp holds it at GOTW_MAX_BATCH (robust to the nanoseconds the call + // itself consumes). + ReferenceChainsTestAccessor::setPassDeadlineNs( + OS::nanotime() + budget * 10); + ReferenceChainsTestAccessor::setGotwBatchSize(64); + ReferenceChainsTestAccessor::setGotwEmaCallNs(budget); + ReferenceChainsTestAccessor::pushPendingExpandForTest(1); + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + EXPECT_EQ(ReferenceChainsTestAccessor::gotwMaxBatch(), + ReferenceChainsTestAccessor::gotwBatchSize()) + << "a wide remaining deadline should grow the batch to the cap"; + ReferenceChainsTestAccessor::setPassDeadlineNs(0); + + // --- Admission sanity: expansion still walks the graph. Root -> child + // edge, one more drive, child must be admitted. + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, rootNode, childNode, -1}); + ReferenceChainsTestAccessor::pushPendingExpandForTest(1); + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + EXPECT_NE(0, tags_ever_assigned[childNode]) + << "expandFrontier failed to admit childNode with adaptive batch_size"; + + tracker->stop(); +} + +// gotwWindowNs() backlog-pressure widening, unit level: the pod regime is +// a remaining pass window (~10ms) smaller than the measured per-call floor +// (~22-40ms at a 242k-entry tag map), against a lane 127k deep. In that +// regime the previous window math handed the proportional law a window +// smaller than one unavoidable call, collapsing the batch to +// GOTW_MIN_BATCH forever (batch=8 live on every call while the backlog +// drained at ~120-200 objects/min). The widening makes the law size the +// batch UP instead - but ONLY under real backlog depth, so rotation +// fast-lane batches stay deadline-sized. +TEST_F(ReferenceChainsBfsTest, GotwWindowWidensOnlyUnderBacklogPressure) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + const u64 budget = ReferenceChainsTestAccessor::gotwCpuBudgetNs(); + const size_t depth = ReferenceChainsTestAccessor::gotwBacklogMinDepth(); + const u64 mult = ReferenceChainsTestAccessor::gotwBacklogWindowMult(); + const u64 floor = budget * 2; // any floor above the nominal window + + // No deadline and no EMA yet: the nominal budget window. + ReferenceChainsTestAccessor::setGotwEmaCallNs(0); + EXPECT_EQ(budget, ReferenceChainsTestAccessor::gotwWindowNs(0, depth)); + + ReferenceChainsTestAccessor::setGotwEmaCallNs(floor); + + // Floor above the remaining window but a SHALLOW lane: no widening - + // the remaining window stands (rotation fast-lane stays cheap). + EXPECT_EQ(1u, ReferenceChainsTestAccessor::gotwWindowNs(1, 1)); + + // Floor above the remaining window and a DEEP lane: widened to + // EMA x mult, never below the remaining window itself. + EXPECT_EQ(floor * mult, + ReferenceChainsTestAccessor::gotwWindowNs(1, depth)); + + // Floor BELOW the remaining window: no widening even at depth - + // the ordinary proportional law already fits the call in the window. + ReferenceChainsTestAccessor::setGotwEmaCallNs(budget / 2); + EXPECT_EQ(budget, + ReferenceChainsTestAccessor::gotwWindowNs(budget, depth)); + ReferenceChainsTestAccessor::setGotwEmaCallNs(floor); + + // Floor above the NOMINAL window (deadline already passed, the exact + // pod's post-call state) at depth: still widened - the floor is paid + // by the next call regardless, so the batch must amortize it. + EXPECT_EQ(floor * mult, + ReferenceChainsTestAccessor::gotwWindowNs(0, depth)); + + tracker->stop(); +} + +// The widened window in action through the real control loop: one +// GetObjectsWithTags call whose floor (simulated by the mock's busy-wait) +// exceeds both the remaining pass deadline and the nominal window, with +// a backlog deeper than GOTW_BACKLOG_MIN_DEPTH, must GROW the calibrated +// batch (calib x mult, exactly - the window scales with the measured EMA) +// instead of collapsing it to GOTW_MIN_BATCH. This is the round-4 pod +// failure reproduced mechanically: pre-widening, next = calib x nominal / +// ema < calib clamps to MIN forever. +TEST_F(ReferenceChainsBfsTest, AdaptiveBatchGrowsWhenFloorExceedsWindowUnderDeepBacklog) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // A pass deadline a fraction of the simulated per-call floor: the call + // overruns it (exactly the pod's 10ms window vs 22-40ms floor), so after + // the call the loop's deadline check stops the invocation with ONE + // control update - deterministic arithmetic for the assertion below. + gotw_delay_ns = ReferenceChainsTestAccessor::gotwCpuBudgetNs(); // 25ms floor + ReferenceChainsTestAccessor::setPassDeadlineNs(OS::nanotime() + 5000000ULL); + ReferenceChainsTestAccessor::setGotwBatchSize(ReferenceChainsTestAccessor::gotwMinBatch()); + ReferenceChainsTestAccessor::setGotwEmaCallNs(0); // seeded by the call below + + // A pending lane deep enough to cross GOTW_BACKLOG_MIN_DEPTH. The tags + // are unresolvable (never assigned to a node): each batch resolves + // nothing, which is fine - this test drives the control law, not + // admissions. + const size_t depth = ReferenceChainsTestAccessor::gotwBacklogMinDepth() + 1; + for (size_t i = 0; i < depth; i++) { + ReferenceChainsTestAccessor::pushPendingExpandForTest( + (jlong)(1000000 + i)); + } + + int edges = 0; + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + + // EMA after the call = the busy-wait floor (~25ms). The pass deadline is + // long past, so the window is widened to EMA x GOTW_BACKLOG_WINDOW_MULT, + // and the control law computes calib x window / ema = calib x mult - + // exactly, because the window is a whole multiple of the same EMA it + // divides by. + EXPECT_EQ(ReferenceChainsTestAccessor::gotwMinBatch() * + ReferenceChainsTestAccessor::gotwBacklogWindowMult(), + ReferenceChainsTestAccessor::gotwBatchSize()) + << "the floor-dominated deep-backlog regime must GROW the batch, " + "not clamp it to GOTW_MIN_BATCH"; + + ReferenceChainsTestAccessor::setPassDeadlineNs(0); + gotw_delay_ns = 0; + tracker->stop(); +} + +// FAIR-SHARE DRAIN persistence: the lane toggle must survive across +// expandFrontier() invocations. With per-invocation deadlines bounding an +// invocation to a single batch (the production regime: one ~25-30ms +// GetObjectsWithTags call of a 50ms window), a per-invocation reset to +// "priority first" made priority win EVERY invocation and the ordinary +// pending lane was never drained - observed live on hotdog (pending grew +// 109k->113k over 260 passes, every call edges=0 stale re-walks). Here: +// invocation 1 drains the priority lane, the toggle flips to pending; +// priority is refilled (rotation would) and invocation 2 must still drain +// the PENDING lane despite priority being non-empty. +TEST_F(ReferenceChainsBfsTest, FairShareLaneAlternationPersistsAcrossInvocations) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int rootNode = addNode(); + int otherRoot = addNode(); + // Two live boundary objects: tag 1 in pending, tag 2 in priority. + node_tags[rootNode] = 1; + node_tags[otherRoot] = 2; + ASSERT_TRUE(tracker->frontierTable()->insert( + 1, 0, 1, 0, FrontierEntryState::EDGE)); + ASSERT_TRUE(tracker->frontierTable()->insert( + 2, 0, 1, 0, FrontierEntryState::EDGE)); + ReferenceChainsTestAccessor::pushPendingExpandForTest(1); + ReferenceChainsTestAccessor::pushPriorityExpand(2); + int edges = 0; + + // Mock GetObjectsWithTags calls are ~free, so without a deadline a + // single expandFrontier() invocation would drain BOTH lanes in one + // loop. A delayed mock call (gotw_delay_ns below) plus a deadline set + // to a fraction of that delay bounds each invocation to exactly ONE + // batch: the first iteration's top-of-loop check passes (the deadline + // is ~200us away), the delayed call burns past it, and the second + // iteration's check breaks - the production regime, where one real + // ~25-30ms call consumes a 50ms window. + gotw_delay_ns = 1 * 1000 * 1000; // 1ms + ReferenceChainsTestAccessor::setPassDeadlineNs(OS::nanotime() + 200 * 1000); + + // Invocation 1: priority first (the standing preference). + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + EXPECT_EQ(0u, ReferenceChainsTestAccessor::priorityExpandSize()) + << "first invocation should drain the priority lane"; + EXPECT_EQ(1u, ReferenceChainsTestAccessor::pendingExpandSize()) + << "first invocation must leave the pending lane for the next one"; + EXPECT_FALSE(ReferenceChainsTestAccessor::expandLanePreferPriority()); + + // Rotation refills the priority lane; invocation 2 must STILL prefer + // the pending lane - the toggle persists, it is not reset per call. + ReferenceChainsTestAccessor::pushPriorityExpand(2); + ReferenceChainsTestAccessor::setPassDeadlineNs(OS::nanotime() + 200 * 1000); + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + EXPECT_EQ(0u, ReferenceChainsTestAccessor::pendingExpandSize()) + << "second invocation should drain the pending lane"; + EXPECT_EQ(1u, ReferenceChainsTestAccessor::priorityExpandSize()) + << "second invocation must leave the refilled priority lane alone"; + EXPECT_TRUE(ReferenceChainsTestAccessor::expandLanePreferPriority()); + + tracker->stop(); +} +// FANOUT HYGIENE: a _leak_parent_fanout entry whose parent no longer +// resolves in the frontier (pruned: dead object, or a search-restart wipe) +// can never be re-walked, so collectStaleExpandedEntriesForRotation() must +// erase it during selection rather than skip it forever - without the +// erase, the fanout grows monotonically with corpses (observed live at +// ~11k entries of overwhelmingly-dead old backing arrays), which both +// bloats the selection scan and turns the fanout cursor's lap arithmetic +// into mostly wasted skips. +TEST_F(ReferenceChainsBfsTest, StaleRotationEvictsDeadFanoutParents) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + constexpr u32 kLeafKlass = 987; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + FrontierTable *frontier = tracker->frontierTable(); + + // Live fanout parent 1 and dead fanout parent 5 (frontier entry pruned). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, + /*class_tag=*/42)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 5, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, + /*class_tag=*/43)); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, 1, 10); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, 5, 20); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(1)); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(5)); + + frontier->clear(5); // parent 5's object died / search restart pruned it + + std::vector selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(4); + ASSERT_EQ(1u, selected.size()); + EXPECT_EQ((jlong)1, selected[0]); + EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakParentFanout(5)) + << "dead fanout parent must be erased during selection"; + EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(1)) + << "live fanout parent must survive"; + + tracker->stop(); +} + + +// Leak-tag interception (design A + C): an object pre-tagged with a leak tag +// (as LivenessTracker::tagLeakInstances() would have set on a tracked leaking +// instance) must be admitted by converting the leak tag to a frontier tag, +// with the leak tag preserved in the frontier entry so buildChainEvent() +// emits it as target_tag - the ReferenceChain <-> HeapLiveObject correlation +// key. An untagged sibling of the same class must get an ordinary admit with +// no leak tag. +TEST_F(ReferenceChainsBfsTest, LeakTagInterceptionConvertsToFrontierTagAndCorrelates) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + const jlong leak_tag = ReferenceChainsTestAccessor::leakTagBase(); + + int rootNode = addNode(); + int leakChild = addNode(); + int plainChild = addNode(); + // Simulate tagLeakInstances(): the tracked leaking instance already + // carries a leak tag; the sibling does not. + node_tags[leakChild] = leak_tag; + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, rootNode, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, rootNode, leakChild, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, rootNode, plainChild, -1}, + }; + + // A single pass drains this tiny graph to completion, and a completed + // search releases all JVMTI tags (releaseSearchTags(), "tagsReleased" + // in runPass's own log) - so read the tags from tags_ever_assigned, + // which records each tag at assignment time and is never reset (see + // its own comment), not from node_tags (which reads 0 after release). + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + + // The leak-tagged child's tag was REPLACED by a frontier tag (small + // positive, outside the leak range). + jlong leak_ftag = tags_ever_assigned[leakChild]; + ASSERT_NE(leak_tag, leak_ftag) + << "leak tag was never intercepted - BFS did not reach the object"; + ASSERT_GT(leak_ftag, 0); + EXPECT_LT(leak_ftag, leak_tag) << "frontier tag must be outside leak range"; + + // The frontier entry preserves the leak tag for correlation. + EXPECT_EQ(leak_tag, ReferenceChainsTestAccessor::frontierLeakTag(leak_ftag)); + + // The untagged sibling got an ordinary admit: frontier tag assigned, but + // no leak tag stored. + jlong plain_ftag = tags_ever_assigned[plainChild]; + ASSERT_GT(plain_ftag, 0); + EXPECT_EQ(0, ReferenceChainsTestAccessor::frontierLeakTag(plain_ftag)); + + // Design C: buildChainEvent() reports the leak tag as target_tag for the + // leak-tagged instance, and the plain frontier tag for the sibling. + ReferenceChainEvent event; + ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( + &mock_jvmti, &mock_jni, leak_ftag, &event)); + EXPECT_EQ((u64)leak_tag, event._target_tag) + << "chain target tag must be the leak tag (correlation key)"; + EXPECT_GE(event._depth, 1u) << "leak child sits behind the root, not at it"; + + ReferenceChainEvent plain_event; + ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( + &mock_jvmti, &mock_jni, plain_ftag, &plain_event)); + EXPECT_EQ((u64)plain_ftag, plain_event._target_tag) + << "untagged instance must keep the frontier tag as target tag"; + + tracker->stop(); +} + +// Candidate-scoped reach, prong 1 (walkCandidateThreadLocals()): a leak +// held through the leaking thread's ThreadLocalMap must be intercepted +// with its full chain by ONE bounded walk from the Thread object, no +// matter what the ordinary BFS backlog state is - and the walk's gates +// must keep it off the Thread's non-thread-local fields entirely. +TEST_F(ReferenceChainsBfsTest, ThreadWalkDescendsOnlyThreadLocalMapAndInterceptsLeak) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // The classes descendFromAnchor()'s resolutions look up by FindClass + // name (see registerClassForFindClass' own comment) + the scripted + // graph's own classes. resolveLoadedClasses() below tags every + // registered class, which is what makes the class-tag gates resolvable. + void *threadCls = (void *)0x3001, *tlmapCls = (void *)0x3002, + *loaderCls = (void *)0x3003, *holderCls = (void *)0x3004, + *chunkCls = (void *)0x3005; + int tlmapIdx = + registerClassForFindClass(tlmapCls, + "java/lang/ThreadLocal$ThreadLocalMap", + "Ljava/lang/ThreadLocal$ThreadLocalMap;"); + int loaderIdx = + registerClassForFindClass(loaderCls, "java/lang/ClassLoader", + "Ljava/lang/ClassLoader;"); + registerClassForFindClass(threadCls, "java/lang/Thread", + "Ljava/lang/Thread;"); + int holder = addClass(holderCls, "Lcom/rc/descendwalk/Holder;"); + int chunk = addClass(chunkCls, "Lcom/rc/descendwalk/LeakChunk;"); + thread_class = threadCls; + ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); + + int threadNode = addNode(); + int threadNode2 = addNode(); // second candidate thread: fresh-admission path + int tlmapNode = addNode(); + int loaderNode = addNode(); // Thread's contextClassLoader: anchor gate + int loaderNode2 = addNode(); // a ClassLoader below the gate: no-descend + int holderNode = addNode(); + int leakChunk = addNode(); + + const jlong leak_tag = ReferenceChainsTestAccessor::leakTagBase(); + node_tags[leakChunk] = leak_tag; + + // Topological order (mock_FollowReferences replays edges in script + // order, expanding only refs the production callback said to descend + // into). + script = { + {JVMTI_HEAP_REFERENCE_FIELD, threadNode, tlmapNode, + /*class_idx=*/-1}, + {JVMTI_HEAP_REFERENCE_FIELD, threadNode, loaderNode, + /*class_idx=*/-1}, + {JVMTI_HEAP_REFERENCE_FIELD, tlmapNode, holderNode, holder}, + {JVMTI_HEAP_REFERENCE_FIELD, tlmapNode, loaderNode2, + /*class_idx=*/-1}, + {JVMTI_HEAP_REFERENCE_FIELD, holderNode, leakChunk, chunk}, + }; + // The anchor gate compares the REFEREE's class tag, so the thread + // edges' class_idx values matter: the tlmap edge carries + // ThreadLocalMap's tag, and the loader edges ClassLoader's. + script[0].class_idx = tlmapIdx; + script[1].class_idx = loaderIdx; + script[3].class_idx = loaderIdx; + + // The first thread walks the REUSE path: its Thread object is already + // admitted (root-attached THREAD entry + JVMTI tag) exactly as it is + // in production after the first walk pass or root enumeration. The + // mock keeps the scripted tag array (node_tags, what callbacks see as + // tag_ptr) separate from the pointer-keyed tag map (what GetTag/SetTag + // see), so the pre-anchored tag must be mirrored into both. + FrontierTable *frontier = tracker->frontierTable(); + jlong anchor_tag = + tracker->tagObject(&mock_jvmti, + reinterpret_cast(&node_tags[threadNode])); + ASSERT_GT(anchor_tag, 0); + node_tags[threadNode] = anchor_tag; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, anchor_tag, 0, 0, FrontierEntryState::FRONTIER, + (u8)JVMTI_HEAP_REFERENCE_THREAD)); + + // Two candidate slots, two qualifying tids: tid 777's thread is + // pre-anchored (reuse path), tid 778's is untagged (fresh-admission + // path - GetObjectClass + tagObject + insert, the path a never-walked + // thread takes in production). + jint tids0[] = {777}; + jint tids1[] = {778}; + ReferenceChainsTestAccessor::seedCandidateSlotForTest( + /*slot=*/0, /*klass_id=*/6, tids0, 1); + ReferenceChainsTestAccessor::seedCandidateSlotForTest( + /*slot=*/1, /*klass_id=*/6, tids1, 1); + tracker->registerThreadObject( + &mock_jni, 777, reinterpret_cast(&node_tags[threadNode])); + tracker->registerThreadObject( + &mock_jni, 778, reinterpret_cast(&node_tags[threadNode2])); + + int edges = 0; + ReferenceChainsTestAccessor::walkCandidateThreadLocalsForTest( + &mock_jvmti, &mock_jni, 1000, &edges); + + // The ThreadLocalMap-held chain was admitted end-to-end and the + // leak-tagged chunk was intercepted (tag replaced by a frontier tag, + // leak tag preserved for correlation). + jlong thread_ftag = anchor_tag; + jlong tlmap_ftag = tags_ever_assigned[tlmapNode]; + ASSERT_GT(tlmap_ftag, 0) << "anchor gate did not descend into ThreadLocalMap"; + jlong holder_ftag = tags_ever_assigned[holderNode]; + ASSERT_GT(holder_ftag, 0) << "walk did not descend below ThreadLocalMap"; + jlong chunk_ftag = tags_ever_assigned[leakChunk]; + ASSERT_NE(chunk_ftag, leak_tag) + << "leak-tagged chunk under the ThreadLocalMap was never intercepted"; + ASSERT_GT(chunk_ftag, 0); + EXPECT_EQ(leak_tag, ReferenceChainsTestAccessor::frontierLeakTag(chunk_ftag)); + + // Chain shape: Thread (root-attached, THREAD root kind) -> ThreadLocalMap + // -> holder -> chunk. + FrontierEntry thread_entry{}; + ASSERT_TRUE(frontier->lookup(thread_ftag, &thread_entry)); + EXPECT_EQ(0, thread_entry.parent_tag); + EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_THREAD, thread_entry.root_kind); + FrontierEntry chunk_entry{}; + ASSERT_TRUE(frontier->lookup(chunk_ftag, &chunk_entry)); + EXPECT_EQ(holder_ftag, chunk_entry.parent_tag); + EXPECT_EQ(3u, chunk_entry.depth); + + // The gates kept the walk off the metadata branches: neither the + // Thread's own contextClassLoader edge (anchor gate) nor a ClassLoader + // below ThreadLocalMap (no-descend gate) was admitted. + EXPECT_EQ(0, tags_ever_assigned[loaderNode]) + << "anchor gate must not admit the Thread's non-ThreadLocalMap fields"; + EXPECT_EQ(0, tags_ever_assigned[loaderNode2]) + << "no-descend gate must not admit fat-metadata classes below the anchor"; + + // The second thread took the fresh-admission path (no prior tag/entry): + // its Thread object was admitted root-attached with the THREAD root + // kind. (Its scripted tag array entry was never set, so the minted tag + // is read back through the pointer-keyed tag map via getTag.) + jlong thread2_ftag = ReferenceChainsTestAccessor::getTagForTest( + &mock_jvmti, reinterpret_cast(&node_tags[threadNode2])); + ASSERT_GT(thread2_ftag, 0) << "fresh thread anchor was never admitted"; + FrontierEntry thread2_entry{}; + ASSERT_TRUE(frontier->lookup(thread2_ftag, &thread2_entry)); + EXPECT_EQ(0, thread2_entry.parent_tag); + EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_THREAD, thread2_entry.root_kind); + + tracker->stop(); +} + +// Candidate-scoped reach, prong 2 (collectStaticFieldAnchorsForRotation()/ +// walkStaticFieldAnchors()): the collector selects exactly the root-attached +// static-holder entries with a wrapping cursor, and the walk reaches a leak +// held 3-4 hops inside a static collection in one bounded call - the shape +// the one-hop Tier-2 rotation demonstrably cannot reach from an un-expanded +// FRONTIER holder on a rising heap (pod rounds 5-6). +TEST_F(ReferenceChainsBfsTest, StaticAnchorRotationWalksRootAttachedStaticHolders) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + void *holderCls = (void *)0x4001, *chunkCls = (void *)0x4002; + addClass(holderCls, "Lcom/rc/descendwalk/StaticHolder;"); + int chunk = addClass(chunkCls, "Lcom/rc/descendwalk/StaticChunk;"); + + int holderNode = addNode(); + int tableNode = addNode(); + int entryNode = addNode(); + int leakChunk = addNode(); + const jlong leak_tag = ReferenceChainsTestAccessor::leakTagBase(); + node_tags[leakChunk] = leak_tag; + + // Static Map -> table -> Entry -> chunk: the collection-shaped static + // holder's internals, deeper than one hop. + script = { + {JVMTI_HEAP_REFERENCE_FIELD, holderNode, tableNode, -1}, + {JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT, tableNode, entryNode, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, entryNode, leakChunk, chunk}, + }; + + // Seed the frontier exactly as the static sweep admits a static field's + // value: root-attached, STATIC_FIELD root kind, FRONTIER state. Plus a + // second durable-root anchor (JNI_GLOBAL - the pod-round-7 filter + // extension), a TRANSIENT-root decoy the collector must skip, and a + // chain-attached child (table entries only - deliberately NOT mirrored + // into node_tags, so the walk below freshly admits those nodes instead + // of tripping ALREADY_ADMITTED on stale scripted tags). + FrontierTable *frontier = tracker->frontierTable(); + node_tags[holderNode] = 101; // mock_GetObjectsWithTags' resolvable tag + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 101, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + 101, 9001, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 102, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 103, 101, 1, FrontierEntryState::FRONTIER, 0)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 104, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_JNI_GLOBAL)); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + 104, 9004, JVMTI_HEAP_REFERENCE_JNI_GLOBAL); + + std::vector selected = + ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( + 4); + // Both DURABLE root kinds are selected (STATIC_FIELD 101, JNI_GLOBAL + // 104, in cursor/tag order); the transient-root decoy and the child are + // not. The walk below drives only 101 (the resolvable node). + ASSERT_EQ(2u, selected.size()); + EXPECT_EQ(101, selected[0]); + EXPECT_EQ(104, selected[1]); + std::vector walk_selected = {selected[0]}; + + // The static anchor's whole internal structure is admitted by one + // bounded walk, intercepting the leak tag at depth 3 below the holder. + int edges = 0; + ReferenceChainsTestAccessor::walkStaticFieldAnchorsForTest( + &mock_jvmti, &mock_jni, walk_selected, 1000, &edges); + jlong table_ftag = tags_ever_assigned[tableNode]; + jlong entry_ftag = tags_ever_assigned[entryNode]; + jlong chunk_ftag = tags_ever_assigned[leakChunk]; + ASSERT_GT(table_ftag, 0) << "table array was not reached by the anchor walk"; + ASSERT_GT(entry_ftag, 0) << "Entry was not reached one hop below table"; + ASSERT_NE(chunk_ftag, leak_tag) + << "leak-tagged chunk inside the static holder was never intercepted"; + EXPECT_EQ(leak_tag, ReferenceChainsTestAccessor::frontierLeakTag(chunk_ftag)); + FrontierEntry chunk_entry{}; + ASSERT_TRUE(frontier->lookup(chunk_ftag, &chunk_entry)); + EXPECT_EQ(entry_ftag, chunk_entry.parent_tag); + EXPECT_EQ(3u, chunk_entry.depth); + + tracker->stop(); +} + +// Round-14 tiered selection: a container-shaped anchor (its own class +// implements Collection/Map - the LEAK_BUFFER wrapper shape) admitted at a +// LATE index position must leap the queue of ~28k other-tier anchors (the +// round-13 hotdog measurement: admission-order selection put the leak +// holder at position ~12-21k against ~4k walk coverage per search - +// deterministically unreachable). Also the tier-0 guarantee: a leak-tagged +// anchor leads the walk order unconditionally. +TEST_F(ReferenceChainsBfsTest, ContainerAnchorLeapsQueueAcrossLargeIndex) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr int kAnchorCount = 28000; + for (int i = 0; i < kAnchorCount; i++) { + jlong tag = 100 + i; + jlong class_tag = 500000 + i; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + tag, class_tag, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + // Containers only at late positions 24000..24003 - buried behind + // 24k other-tier anchors under any admission-order cursor. + bool container = (i >= 24000 && i <= 24003); + ReferenceChainsTestAccessor::primeClassShapeForTest(class_tag, + container); + } + // One leak-tagged anchor at the very tail - tier 0, must lead. + jlong leak_anchor = 100 + kAnchorCount; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, leak_anchor, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + frontier->setLeakTag(leak_anchor, 777); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + leak_anchor, 599999, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + ReferenceChainsTestAccessor::primeClassShapeForTest( + 599999, false /* its tier comes from leak_tag, not shape */); + + std::vector selected = + ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( + 16); + ASSERT_EQ(16u, selected.size()) + << "28k+5 eligible anchors at a 16 budget must fill the selection"; + EXPECT_EQ(leak_anchor, selected[0]) + << "the leak-tagged anchor (tier 0) must lead the walk order"; + // The four container anchors are selected in this FIRST call despite + // positions 24000+ (tags 24100-24103). + for (jlong t : {24100, 24101, 24102, 24103}) { + EXPECT_NE(std::find(selected.begin(), selected.end(), t), selected.end()) + << "container anchor " << t + << " did not leap the other-tier queue"; + } + // Sanity: an early other-tier anchor also made the cut (cursor-fair + // fill from position 0). + EXPECT_NE(std::find(selected.begin(), selected.end(), (jlong)100), + selected.end()); + + tracker->stop(); +} + +// Round-15 fresh-admission priority: the hotdog wrapper is admitted +// LATE in a search (its holder class sits at sweep index 24627 of 33270, +// so admission lands at the anchor-index TAIL) - behind the whole fair +// container backlog measured at 1633 entries against 44-75-pass search +// lifetimes (the fair container cursor would reach it at pass ~102+, +// after the search is dead). A container admitted since the last +// collector call must be walked in the very NEXT call, ahead of the fair +// backlog; an unclassified fresh anchor rides the same lane (the wrapper +// admits one pass before reconcileAnchorClassShapes can classify its +// class), while a fresh NON-container waits in the other tier as before. +TEST_F(ReferenceChainsBfsTest, FreshContainerWalkedBeforeFairBacklog) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // A fair container backlog, "admitted long ago": 200 containers at + // positions 0..199. + constexpr int kBacklog = 200; + for (int i = 0; i < kBacklog; i++) { + jlong tag = 500 + i; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + tag, 700000 + i, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + ReferenceChainsTestAccessor::primeClassShapeForTest(700000 + i, true); + } + // First collector call: all 200 anchors are fresh (nothing has had + // a first look yet), the lane keeps 16 and the rest spend their + // first look (they fall back to the fair container tier). + std::vector first = + ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( + 16); + ASSERT_EQ(16u, first.size()) << "200 eligible containers at a 16 budget"; + + // The late admission wave: a wrapper-class container at the index + // tail, a fresh classified NON-container (a fresh String static - it + // must NOT ride the fresh lane), and a fresh unclassified anchor (a + // genuinely new class - it must ride the lane, so the classification + // lag cannot lose the wrapper's fresh window). + const jlong wrapper_tag = 500 + kBacklog; + const jlong fresh_string_tag = 500 + kBacklog + 1; + const jlong fresh_unknown_tag = 500 + kBacklog + 2; + for (auto [tag, class_tag, container, prime] : + {std::make_tuple(wrapper_tag, (jlong)799001, true, true), + std::make_tuple(fresh_string_tag, (jlong)799002, false, true), + std::make_tuple(fresh_unknown_tag, (jlong)799003, false, + false /* deliberately unclassified */)}) { + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + tag, class_tag, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + if (prime) { + ReferenceChainsTestAccessor::primeClassShapeForTest(class_tag, + container); + } + } + + std::vector second = + ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( + 16); + ASSERT_EQ(16u, second.size()); + EXPECT_EQ(wrapper_tag, second[0]) + << "the freshly admitted container must lead the walk order, not " + "wait behind the 184-container fair backlog"; + EXPECT_NE(std::find(second.begin(), second.end(), fresh_unknown_tag), + second.end()) + << "an unclassified fresh anchor must ride the fresh lane (the " + "wrapper admits a pass before reconcile classifies its class)"; + EXPECT_EQ(std::find(second.begin(), second.end(), fresh_string_tag), + second.end()) + << "a fresh NON-container stays in the other tier - the fresh lane " + "is the container lane, or fresh Strings would flood it"; + // The fair container lap still gets the leftover budget. Call 1's + // fresh picks (500-515) spent the whole budget, so the fair cursor + // never advanced - call 2's fair picks start at tag 500 again (a + // benign one-call overlap: walks are idempotent, and it only + // happens when fresh and fair coincide at a lap boundary). + EXPECT_NE(std::find(second.begin(), second.end(), (jlong)500), + second.end()) + << "the fair container lap must still advance with the leftover " + "budget"; + EXPECT_EQ(14, (int)std::count_if(second.begin(), second.end(), + [](jlong t) { + return t >= 500 && t < 500 + kBacklog; + })) + << "16 budget - 2 fresh picks = 14 fair-container picks"; + + // And the dropped fresh entries from call 1 (516-699 spent their + // first look) are still covered by the fair tier: a third call with + // no new admits must keep advancing the fair container cursor from + // wherever call 2 left it, not re-drain anything. + std::vector third = + ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( + 16); + ASSERT_EQ(16u, third.size()); + EXPECT_EQ(std::find(third.begin(), third.end(), wrapper_tag), + third.end()) + << "the wrapper already had its first look - it must not be " + "re-selected while the fair cursor has 184 uncovered peers"; + + tracker->stop(); +} + +// Round-14 tier fairness: the other tier (everything not leak-tagged, not +// container-shaped) still reaches full coverage across wraps - a 40-anchor +// tier at a 16 budget covers all 40 in exactly 3 calls with no duplicates +// within a call. Containers must not permanently starve the rest of the +// index. +TEST_F(ReferenceChainsBfsTest, AnchorOtherTierFairCoverageAcrossWraps) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr int kOtherCount = 40; + for (int i = 0; i < kOtherCount; i++) { + jlong tag = 200 + i; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + tag, 300000 + i, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + ReferenceChainsTestAccessor::primeClassShapeForTest(300000 + i, false); + } + + std::vector all_selected; + for (int call = 0; call < 3; call++) { + std::vector selected = + ReferenceChainsTestAccessor:: + collectStaticFieldAnchorsForRotationForTest(16); + int expected = (call == 2) ? 8 : 16; + ASSERT_EQ(expected, (int)selected.size()) + << "call " << call << " should select " << expected; + std::set dedup(selected.begin(), selected.end()); + ASSERT_EQ(selected.size(), dedup.size()) + << "no anchor may be selected twice within a call"; + all_selected.insert(all_selected.end(), selected.begin(), + selected.end()); + } + std::set covered(all_selected.begin(), all_selected.end()); + ASSERT_EQ(40u, covered.size()) << "full other-tier coverage expected"; + for (int i = 0; i < kOtherCount; i++) { + EXPECT_NE(covered.find(200 + i), covered.end()) + << "other-tier anchor " << 200 + i << " never selected"; + } + + tracker->stop(); +} + +// Round-13/14 restart hygiene: discovered-instance tags are FRONTIER tags; +// restartSearch() resets the frontier table and _next_tag=1, so any +// surviving discovered slot either fails reconstructChain (observed on-pod: +// 'buildChainEvent failed ... reconstructChain failed for target_tag=8851') +// or, worse, resolves into a live new-search entry and emits a chain event +// for the WRONG OBJECT (the likely origin of the earlier noise-[B event). +// restartSearch() must clear them, along with the anchor index and its +// parallel arrays/cursors. +TEST_F(ReferenceChainsBfsTest, RestartSearchClearsDiscoveredInstanceTags) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // A watched candidate with one discovered instance recorded. + ReferenceChainsTestAccessor::setCandidateCountForTest(1); + ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, 7); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(7, 4242, + false); + ASSERT_EQ(4242, ReferenceChainsTestAccessor::discoveredTagForTest(0, 0)); + ASSERT_EQ(1, ReferenceChainsTestAccessor::discoveredCountForTest(0)); + + // An anchor index entry (tag + parallel class tag) to confirm the + // index reset covers the parallel structures too. + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + 4242, 555, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + (void)frontier; + + ReferenceChainsTestAccessor::restartSearchForTest(); + + EXPECT_EQ(0, ReferenceChainsTestAccessor::discoveredTagForTest(0, 0)) + << "stale discovered frontier tag survived restartSearch()"; + EXPECT_EQ(0, ReferenceChainsTestAccessor::discoveredCountForTest(0)); + EXPECT_TRUE(ReferenceChainsTestAccessor::anchorIndexIsEmptyForTest()) + << "anchor index (or its parallel arrays) survived restartSearch()"; + + tracker->stop(); +} + +// Round-19 (pod 289f8 — three JVMs of "canary search, 0/1 candidates found" +// while leak-tagged instances WERE intercepted and chains re-emitted): the +// marker->leak-tag design migration never updated the found criterion. +// _candidate_found_bits was set ONLY in heapReferenceCallback()'s marker-tag +// block, and markers are never set under leak tags ("no marker tags — using +// leak tags now"), so the chase was structurally unresolvable and every +// search ended via frontier-cap/no-progress instead of "all candidates +// found". The leak-tag-world criterion: a leak-tag-target chain +// (target_tag >= LEAK_TAG_BASE) built for a candidate slot marks that slot +// found and records its canary link; a noise chain (no leak tag) must NOT. +TEST_F(ReferenceChainsBfsTest, LeakTagChainMarksCanaryFound) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // Slot 0: candidate klass 7 whose discovered instance carries a leak tag + // (entry.leak_tag set — the shape a walk + tagLeakInstances correlation + // produces; target_tag becomes the leak tag). Slot 1: candidate klass 9 + // with a plain noise discovered instance, no leak tag. + ReferenceChainsTestAccessor::setCandidateCountForTest(2); + ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, 7); + ReferenceChainsTestAccessor::setCandidateKlassIdForTest(1, 9); + // depth=1 at a durable STATIC_FIELD root (root_kind 8) so the retention + // filter (suppressChainEvent: depth==0, or depth==1 at a transient + // root) keeps both chains. + ASSERT_TRUE(frontier->insert(4242, 0, 7, 1, FrontierEntryState::FRONTIER, + 8, /*class_tag=*/0, -1, /*kind=*/0)); + frontier->setLeakTag(4242, 1073742079); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(7, 4242, true); + ASSERT_EQ(4242, ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 0)); + ASSERT_TRUE(frontier->insert(5353, 0, 9, 1, FrontierEntryState::FRONTIER, + 8, 0, -1, 0)); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(9, 5353, false); + ASSERT_EQ(5353, ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(1, 0)); + + ReferenceChainsTestAccessor::buildDiscoveredInstanceChainsForTest(7, 1); + ReferenceChainsTestAccessor::buildDiscoveredInstanceChainsForTest(9, 1); + + // Slot 0 found via its leak-tag chain (bit 0 set, link recorded); + // slot 1's noise chain did not mark anything found. + EXPECT_EQ(1ULL, ReferenceChainsTestAccessor::candidateFoundBitsForTest()); + EXPECT_EQ(4242, ReferenceChainsTestAccessor::candidateFrontierTagForTest(0)); + EXPECT_EQ(0, ReferenceChainsTestAccessor::candidateFrontierTagForTest(1)) + << "noise-target chain must not mark the canary found"; + + tracker->stop(); +} + +// B' push site 1 - DEMOTION TIME (find-anchor-holder-eviction / _static_anchor_fifo): +// when improveChain() replaces a root-attached durable (STATIC_FIELD/ +// JNI_GLOBAL) entry with a deeper chain-attached path, the entry is leaving +// the anchor tier's eligible population at exactly that moment - the push +// must fire right there. This matters because the sweep gate only re-laps +// while the class count is in flux, so a post-lap demotion would otherwise +// never see another static edge onto the entry. Driven through the REAL +// expandFrontier batch walk (mock FollowReferences + real heapReferenceCallback). +TEST_F(ReferenceChainsBfsTest, DemotionPushFiresWhenImproveChainEvictsRootAttachedStatic) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int parentNode = addNode(); + int holderNode = addNode(); + + // The chain edge whose delivery demotes the holder. + script = { + {JVMTI_HEAP_REFERENCE_FIELD, parentNode, holderNode, -1}, + }; + + // Seed exactly the pre-demotion shape: holder root-attached STATIC + // (anchor-eligible), parent a root-attached frontier object whose + // expansion delivers the deeper chain edge. node_tags make both + // resolvable by mock_GetObjectsWithTags for the batch walk. + FrontierTable *frontier = tracker->frontierTable(); + node_tags[holderNode] = 105; + node_tags[parentNode] = 104; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 105, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 104, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + ReferenceChainsTestAccessor::pushPendingExpandForTest(104); + + int edges = 0; + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, &mock_jni, + &edges); + // The chain edge was delivered: the holder's entry is now chain-attached + // (improveChain replaced the depth-0 root-attached admission), and the + // demotion pushed its tag into the at-risk FIFO. + FrontierEntry entry{}; + ASSERT_TRUE(frontier->lookup(105, &entry)); + EXPECT_EQ(104, entry.parent_tag); + EXPECT_EQ(1u, entry.depth); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_TRUE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(105)); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest()); + + // Re-walking the same edge must NOT push twice: improveChain refuses + // (new depth 1 is not > current 1), and the set dedupes regardless. + int edges2 = 0; + ReferenceChainsTestAccessor::pushPendingExpandForTest(104); + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, &mock_jni, + &edges2); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest()); + + tracker->stop(); +} + +// B' push site 2 - SWEEP TIME: the static sweep's class->field edge onto an +// already-admitted CHAIN-ATTACHED entry (the admission-order eviction shape: +// born as a non-root child, never root-attached at all) must feed the FIFO. +// Driven through a full runPass so the real admitStaticFieldRoots sweep (and +// its FollowReferences) delivers the edge, and the rotation phase of the +// SAME pass drains the FIFO - the push counter survives the drain. +TEST_F(ReferenceChainsBfsTest, SweepPushFiresOnStaticEdgeOntoChainAttachedHolder) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=100")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int classNode = addNode(); + int holderNode = addNode(); + addClass((void *)&node_tags[classNode], "Lcom/rc/statics/ChainBornHolder;"); + + script = { + // Only the sweep's static edge: nothing else reaches holderNode, so + // the only possible push is the static-edge-onto-chain-attached site. + {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, holderNode, -1}, + }; + + // Seed the born-chain-attached shape the eviction leaves: holder already + // a non-root child (parent 104, depth 1). The sweep's static edge below + // hits maybeUpgradeRootAttachedRootKind's documented parent_tag != 0 + // refusal and must fall into the B' push instead. + FrontierTable *frontier = tracker->frontierTable(); + node_tags[holderNode] = 105; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 105, 104, 1, FrontierEntryState::FRONTIER, 0)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 104, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + + EXPECT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest()) + << "static sweep edge onto the chain-attached holder never fed " + "the at-risk anchor FIFO"; + // The rotation phase of the same pass drained the pushed tag into the + // anchor walk (the holder has no scripted subtree - the walk is a no-op, + // which is what makes the drained-empty assertion attributable to the + // drain rather than to a walk). + EXPECT_EQ(0u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + // The holder's entry is untouched by the push - the B' feed records the + // at-risk shape, it never re-attributes the entry (re-rooting is the + // documented refusal that motivated the FIFO in the first place). + FrontierEntry entry{}; + ASSERT_TRUE(frontier->lookup(105, &entry)); + EXPECT_EQ(104, entry.parent_tag); + EXPECT_EQ(1u, entry.depth); + + tracker->stop(); +} + +// B' mechanics: a chain-attached holder drained from the at-risk FIFO is +// descend-walked and intercepts a leak chunk 3 hops below it - the repair +// for the population the root-attached collector demonstrably cannot +// select (the negative control below). Mirrors +// StaticAnchorRotationWalksRootAttachedStaticHolders's walk-phase shape, +// but with the holder chain-attached and FIFO-sourced. +TEST_F(ReferenceChainsBfsTest, AtRiskAnchorFifoDrainAndWalkIntercept) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + void *holderCls = (void *)0x6001, *chunkCls = (void *)0x6002; + addClass(holderCls, "Lcom/rc/descendwalk/ChainAttachedHolder;"); + int chunk = addClass(chunkCls, "Lcom/rc/descendwalk/ChainChunk;"); + + int parentNode = addNode(); + int holderNode = addNode(); + int tableNode = addNode(); + int entryNode = addNode(); + int leakChunk = addNode(); + const jlong leak_tag = ReferenceChainsTestAccessor::leakTagBase(); + node_tags[leakChunk] = leak_tag; + + // Chain: parent -> holder -> table -> entry -> leak chunk. Only the + // holder's own subtree is scripted for the walk below (the direct + // walkStaticAnchors drive never runs the roots/expand phases). + script = { + {JVMTI_HEAP_REFERENCE_FIELD, holderNode, tableNode, -1}, + {JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT, tableNode, entryNode, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, entryNode, leakChunk, chunk}, + }; + + // Seed the holder exactly as the eviction leaves it: chain-attached + // (parent_tag = 104, depth 1, no root_kind). node_tags[holderNode] = 105 + // makes the tag resolvable by mock_GetObjectsWithTags, exactly like the + // root-attached test's own 101 mapping. + FrontierTable *frontier = tracker->frontierTable(); + node_tags[holderNode] = 105; + // root_kind = 0: the entry is chain-attached, and a non-root entry's + // edge kind is not recorded (FrontierEntry::root_kind's own comment). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 105, 104, 1, FrontierEntryState::FRONTIER, 0)); + // The chain's root: TRANSIENT (stack local), so the collector's durable + // root-kind filter skips it too - the whole table is un-selectable. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 104, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + + // Negative control: the root-attached collector selects NOTHING from a + // table holding only a chain-attached holder and a transient root - + // the pre-B' behavior that stranded the dual-reachable population. + std::vector selected = + ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest(4); + ASSERT_TRUE(selected.empty()); + + // B': push + drain + walk reaches what the collector cannot. + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(105, 28366); + std::vector drained; + // 16 = ReferenceChainTracker::STATIC_ANCHOR_FIFO_DRAIN (private), the + // same per-pass drain cap runPassManualWalk() uses. + ASSERT_EQ(1, ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 16, drained)); + ASSERT_EQ(1u, drained.size()); + EXPECT_EQ(105, drained[0].tag); + EXPECT_EQ(28366u, drained[0].klass_id); + EXPECT_EQ(0u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + + int edges = 0; + std::vector drained_tags; + for (const auto &at_risk : drained) { + drained_tags.push_back(at_risk.tag); + } + ReferenceChainsTestAccessor::walkStaticAnchorFifoForTest( + &mock_jvmti, &mock_jni, drained_tags, 1000, &edges, nullptr); + jlong table_ftag = tags_ever_assigned[tableNode]; + jlong entry_ftag = tags_ever_assigned[entryNode]; + jlong chunk_ftag = tags_ever_assigned[leakChunk]; + ASSERT_GT(table_ftag, 0) << "table array was not reached by the anchor walk"; + ASSERT_GT(entry_ftag, 0) << "Entry was not reached one hop below table"; + ASSERT_NE(chunk_ftag, leak_tag) + << "leak-tagged chunk inside the chain-attached holder was never " + "intercepted"; + EXPECT_EQ(leak_tag, + ReferenceChainsTestAccessor::frontierLeakTag(chunk_ftag)); + FrontierEntry chunk_entry{}; + ASSERT_TRUE(frontier->lookup(chunk_ftag, &chunk_entry)); + EXPECT_EQ(entry_ftag, chunk_entry.parent_tag); + EXPECT_EQ(4u, chunk_entry.depth); + + tracker->stop(); +} + +// B' requeue mechanics: a truncated anchor walk reports exactly the +// RESOLVED-but-unwalked tags, and requeueStaticAnchorFifoFront() restores +// them to the FIFO front in order with a consistent membership set - so +// an at-risk holder that lost its budget turn keeps it for the next pass +// instead of waiting for the next sweep lap. +TEST_F(ReferenceChainsBfsTest, TruncatedAnchorWalkRequeuesUnwalkedFifoTags) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + void *holderCls = (void *)0x7001; + addClass(holderCls, "Lcom/rc/descendwalk/RequeueHolder;"); + + int holderANode = addNode(); + int holderBNode = addNode(); + int tableNode = addNode(); + int entryNode = addNode(); + + // Holder A's subtree is deep enough that a budget of 2 truncates the + // walk after A (two edges admitted, budget exhausted on the descend); + // holder B then must come back unwalked. B has no scripted subtree - + // its walk would be a no-op anyway, which is exactly what makes the + // unwalked report attributable to the truncation, not to content. + script = { + {JVMTI_HEAP_REFERENCE_FIELD, holderANode, tableNode, -1}, + {JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT, tableNode, entryNode, -1}, + }; + + FrontierTable *frontier = tracker->frontierTable(); + node_tags[holderANode] = 105; + node_tags[holderBNode] = 106; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 105, 104, 1, FrontierEntryState::FRONTIER, 0)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 106, 104, 1, FrontierEntryState::FRONTIER, 0)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 104, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(105, 2001); + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(106, 2002); + std::vector drained; + // 16 = STATIC_ANCHOR_FIFO_DRAIN (private), the per-pass drain cap. + ASSERT_EQ(2, ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 16, drained)); + ASSERT_EQ(2u, drained.size()); + EXPECT_EQ(105, drained[0].tag); + EXPECT_EQ(106, drained[1].tag); + + int edges = 0; + std::vector drained_tags; + for (const auto &at_risk : drained) { + drained_tags.push_back(at_risk.tag); + } + std::vector unwalked; + ReferenceChainsTestAccessor::walkStaticAnchorFifoForTest( + &mock_jvmti, &mock_jni, drained_tags, 2, &edges, &unwalked); + ASSERT_EQ(1u, unwalked.size()); + EXPECT_EQ(106, unwalked[0]); + EXPECT_NE(0, tags_ever_assigned[tableNode]) + << "holder A's walk never ran - the truncation happened too early"; + + // Requeue exactly what the caller-side filter in runPassManualWalk() + // would requeue (here: everything unwalked, both FIFO-sourced), keeping + // the drained entries' klass so the per-class occupancy is restored. + std::vector requeue; + for (jlong unwalked_tag : unwalked) { + for (const auto &at_risk : drained) { + if (unwalked_tag == at_risk.tag) { + requeue.push_back(at_risk); + break; + } + } + } + ReferenceChainsTestAccessor::requeueStaticAnchorFifoFrontForTest(requeue); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_TRUE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(106)); + EXPECT_FALSE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(105)); + std::vector redrained; + ASSERT_EQ(1, ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 16, redrained)); + ASSERT_EQ(1u, redrained.size()); + EXPECT_EQ(106, redrained[0].tag); + + tracker->stop(); +} + +// Round 16, fix A: a this-field self-edge is REAL in the heap - every +// java.util.Collections$Synchronized* holder carries mutex == this - so +// walking such a holder's own subtree (rotation anchor walk or BFS descent) +// re-reports the holder as its own child through that field. On the pod +// (round-15 evidence, ev-leaktag-onpod-round15-results) that edge's +// improveChain() "improvement" (depth = holder.depth + 1 > 0) overwrote the +// LEAK_BUFFER wrapper's root-attached entry with a parent==its-own-tag +// chain-attached one (probe: parent==tag root_kind=0 referrer_klass=) - collector-invisible ever after, because the entry left the +// parent_tag == 0 population just as B' got cap-pinned by floods. The guard +// refuses the self-parent (and counts it), so the entry keeps its +// root-attached shape and stays collector-selectable. Driven through the +// real expandFrontier batch walk, the same harness as the demotion-push +// test above it - which also keeps proving the LEGIT deeper-path demotion +// still works (that test must keep passing). +TEST_F(ReferenceChainsBfsTest, + SelfEdgeFieldDoesNotDemoteRootAttachedStaticHolder) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + void *holderCls = (void *)0x7101; + int holderClsIdx = + addClass(holderCls, "Ljava/util/Collections$SynchronizedRandomAccessList;"); + int holderNode = addNode(); + + // The holder's own self-edge: referrer == referee == holder (the + // mutex == this field), exactly as its rotation walk re-reports it. + script = { + {JVMTI_HEAP_REFERENCE_FIELD, holderNode, holderNode, holderClsIdx}, + }; + + // Seed the pre-demotion shape: holder root-attached STATIC_FIELD + // (anchor-eligible), admitted at depth 0. The batch walk from its own + // pending-expand slot delivers the self-edge with the holder as BOTH + // referrer and referee. + FrontierTable *frontier = tracker->frontierTable(); + node_tags[holderNode] = 105; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 105, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::pushPendingExpandForTest(105); + + // A delivered self-edge trips BOTH sibling guards: improveChain refuses, + // and the already-admitted block's else-if then offers the same + // self-parent to reparentToDurableRoot, which refuses and counts too. + // Two refusals per delivered edge is the designed accounting (the pod + // counter will climb 2 per wrapper walk). + u64 skips_before = + ReferenceChainsTestAccessor::selfEdgeGuardSkipsForTest(); + int edges = 0; + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, &mock_jni, + &edges); + + // The self-edge was delivered and refused: the entry keeps its + // root-attached shape (the collector's parent_tag == 0 eligibility), + // no demotion push fired, and the guard counted the refusal. + FrontierEntry entry{}; + ASSERT_TRUE(frontier->lookup(105, &entry)); + EXPECT_EQ(0, entry.parent_tag); + EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD, entry.root_kind); + EXPECT_EQ(0u, entry.depth); + EXPECT_EQ(0u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_EQ(2u, + ReferenceChainsTestAccessor::selfEdgeGuardSkipsForTest() - + skips_before); + + // The sibling guards and the direct table calls agree: a self-parent is + // refused by both improvement paths, unconditionally, and every + // refusal is counted (the callback's two refusals above are the first + // and second). + EXPECT_FALSE(frontier->improveChain(105, 105, 0, 5, 0, -1, 0, 0)); + EXPECT_FALSE(frontier->reparentToDurableRoot(105, 105, 0, -1, 0)); + EXPECT_EQ(4u, + ReferenceChainsTestAccessor::selfEdgeGuardSkipsForTest() - + skips_before); + + tracker->stop(); +} + +// Round 16, fix B (pod round-15 measurement, ev-leaktag-onpod-round15- +// results): the B' at-risk FIFO sat cap-pinned at 1024 because three +// classes flooded it (klass 1: 1396 pushes, klass 215: 1063, klass 1733: +// 988+), so the LEAK_BUFFER wrapper's pushes (klass 28366) were dropped at +// the cap check and the lane designed to repair exactly its demotion never +// carried it. The per-class quota keeps one class from owning the lane: at +// 64 per class, a full 1024-entry FIFO necessarily holds >= 16 distinct +// classes, the flood's excess is dropped and counted, and the wrapper's +// fresh-class push lands. Occupancy is maintained exactly across push / +// drain / requeue, and a saturated-but-diverse FIFO drops newcomers at the +// cap rather than evicting anyone. +TEST_F(ReferenceChainsBfsTest, AtRiskFifoPerClassQuotaKeepsFloodOut) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + const u32 quota = + ReferenceChainsTestAccessor::kAtRiskPerKlassCap; + ASSERT_EQ(64u, quota); + // The push/quota-drop counters are CUMULATIVE across the tracker's + // lifetime (start() does not reset them - only the full + // search-restart reset does), so every assertion below is a DELTA + // from this baseline: order-immune to whatever the tests that ran + // before this one left behind (the round-15 singleton-state lesson). + const u64 pushed0 = + ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest(); + const u64 drops0 = + ReferenceChainsTestAccessor::staticAnchorFifoQuotaDropsForTest(); + + // The flood: only the first `quota` pushes of one class land. + for (int i = 0; i < 70; i++) { + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(1000 + i, + 1733); + } + EXPECT_EQ(quota, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_EQ(quota, ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest() - + pushed0); + EXPECT_EQ(70u - quota, + ReferenceChainsTestAccessor::staticAnchorFifoQuotaDropsForTest() - + drops0); + + // The wrapper's push (a different class) lands despite the flood - + // exactly the push the pod dropped. + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(2000, 28366); + EXPECT_EQ(quota + 1, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_TRUE( + ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(2000)); + + // Tag dedupe is unchanged: the same tag never enters twice (and the + // duplicate is not counted as a quota drop, nor as a push). + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(2000, 28366); + EXPECT_EQ(quota + 1, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_EQ(quota + 1, + ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest() - + pushed0); + + // Full drain: order preserved (flood first, newcomer last), occupancy + // erased with the entries - the next flood can land again (it never + // owns MORE than its quota, but it is not permanently locked out + // either). + std::vector drained; + ASSERT_EQ((int)quota + 1, + ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 1024, drained)); + ASSERT_EQ(quota + 1, drained.size()); + EXPECT_EQ(1000, drained.front().tag); + EXPECT_EQ(1733u, drained.front().klass_id); + EXPECT_EQ(2000, drained.back().tag); + EXPECT_EQ(28366u, drained.back().klass_id); + for (int i = 0; i < 70; i++) { + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(1000 + i, + 1733); + } + EXPECT_EQ(quota, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_EQ(2u * quota + 1, + ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest() - + pushed0); + + // Partial drain at the real per-pass rate (16/pass): the flood's + // occupancy is 64 - 16 = 48 after the drain, so its next push lands + // (refilling its share as it drains - the flood self-throttles, it + // never starves the lane). + std::vector partial; + ASSERT_EQ(16, ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 16, partial)); + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(3000, 1733); + EXPECT_EQ(quota - 16 + 1, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_TRUE( + ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(3000)); + + // Requeue restores occupancy exactly: the requeued entries occupy + // their slots again and drain in FIFO order. Note this may put a class + // momentarily ABOVE its quota - the quota gates NEW pushes, it does + // not evict truncated-walk requeues of entries that legitimately + // held their slots before the drain. + std::vector requeue(partial.begin(), + partial.end()); + ReferenceChainsTestAccessor::requeueStaticAnchorFifoFrontForTest(requeue); + EXPECT_EQ(quota + 1, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + std::vector after_requeue; + ASSERT_EQ(16, + ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 16, after_requeue)); + EXPECT_EQ(1000, after_requeue.front().tag); + + // Saturated-but-diverse: 16 distinct classes at exactly their quota + // fill the 1024 cap, and the newcomer is dropped at the CAP (legitimate + // saturation - no eviction), not because of any flood. The remaining + // count after the drains above: 64 flood entries - 16 (partial) - 16 + // (post-requeue) + 1 (tag 3000) = quota - 16 + 1. + std::vector rest; + ASSERT_EQ((int)quota - 16 + 1, + ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 1024, rest)); + for (u32 klass = 1; klass <= 16; klass++) { + for (u32 i = 0; i < quota; i++) { + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest( + 10000 + klass * 100 + i, 4000 + klass); + } + } + EXPECT_EQ(1024u, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_EQ(0u, + ReferenceChainsTestAccessor::staticAnchorFifoQuotaDropsForTest() - + drops0 - (70u - quota) * 2); + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(20000, 28366); + EXPECT_EQ(1024u, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_FALSE( + ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(20000)); + + // Leave the FIFO drained: this test is the one that saturates it, and + // the reset() seam above now clears it for the next test regardless - + // but a drained ending also keeps this test order-independent even if + // that seam ever regresses again. + std::vector final_drain; + ASSERT_EQ(1024, + ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 1024, final_drain)); + + tracker->stop(); +} + +// Retention-edge labels (fillHopEdgeLabels()/hopLabelClassFor()): the +// emitted chain's per-hop labels must decode the JVMTI-SPECIFICATION field +// ordinal captured at admission - the interface offset, the superclass-chain +// order, the interface-referrer branch - and degrade to the edge KIND on any +// undecodable hop, never a fabricated name (the fail-safe contract: a wrong +// numbering on an unverified JVM degrades, it does not lie). +TEST_F(ReferenceChainsBfsTest, HopEdgeLabelsDecodeSpecFieldOrdinals) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // Hierarchy mirroring the spec's own numbering example shape: + // interface IBase { int p; } 1 own field + // interface ISub extends IBase { int x; } 1 own field + // class Base { int base_f; } 1 own field, no super + // class Holder extends Base implements ISub + // { int holder_a; Object leakList; } + // interface ISink { Object CONST_A; Object CONST_B; } + // Spec ordinal spaces (jvmtiHeapReferenceInfoField): + // Holder (class branch): base = ISub(1) + IBase(1) = 2 (transitive + // interfaces, each once); then the superclass chain root-first: + // base_f@2; then own fields in GetClassFields order: + // holder_a@3, leakList@4. + // ISink (interface branch): base = superinterfaces' fields = 0; own + // fields in GetClassFields order: CONST_A@0, CONST_B@1. + void *ibase = (void *)0x5001, *isub = (void *)0x5002, *base = (void *)0x5003, + *holder = (void *)0x5004, *isink = (void *)0x5005; + addClass(ibase, "Lcom/rc/labels/IBase;"); + addClass(isub, "Lcom/rc/labels/ISub;"); + addClass(base, "Lcom/rc/labels/Base;"); + addClass(holder, "Lcom/rc/labels/Holder;"); + addClass(isink, "Lcom/rc/labels/ISink;"); + ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); + // resolveLoadedClasses minted each class's raw (negative) tag via the + // mock's tag map - read them back for the decoder's tag->class lookup + // and the entries' referrer_class_tag values. + auto tagOf = [&](void *k) -> jlong { return tags[k]; }; + field_decode_classes[tagOf(holder)] = holder; + field_decode_classes[tagOf(isink)] = isink; + field_decode_classes[tagOf(base)] = base; + // ibase deliberately NOT registered into field_decode_classes: an + // unresolvable referrer class below must degrade to a kind label. + field_decode_hierarchy[ibase] = {true, nullptr, {}, {{(void *)0x6001, "p"}}}; + field_decode_hierarchy[isub] = + {true, nullptr, {ibase}, {{(void *)0x6002, "x"}}}; + field_decode_hierarchy[base] = {false, nullptr, {}, {{(void *)0x6003, "base_f"}}}; + field_decode_hierarchy[holder] = + {false, base, {isub}, + {{(void *)0x6004, "holder_a"}, {(void *)0x6005, "leakList"}}}; + field_decode_hierarchy[isink] = + {true, nullptr, {}, + {{(void *)0x6006, "CONST_A"}, {(void *)0x6007, "CONST_B"}}}; + + FrontierTable *frontier = tracker->frontierTable(); + // Chain: [chunk(3)] <- Base.base_f(ordinal 0 over Base's space) <- + // [value2(2), class Base] <- Holder.leakList(ordinal 4, the static root + // edge with the declaring class as referrer) <- [static value(1), class + // Holder]. Interior hops decode against the PARENT entry's class_tag. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EDGE, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, + /*referrer_klass=*/0, /*class_tag=*/tagOf(holder), + /*referrer_field_index=*/4, /*edge_kind=*/0, + /*referrer_class_tag=*/tagOf(holder))); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 2, 1, 1, FrontierEntryState::EDGE, /*root_kind=*/0, + /*referrer_klass=*/0, /*class_tag=*/tagOf(base), + /*referrer_field_index=*/3, JVMTI_HEAP_REFERENCE_FIELD)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 3, 2, 2, FrontierEntryState::EDGE, /*root_kind=*/0, + /*referrer_klass=*/0, /*class_tag=*/tagOf(holder), + /*referrer_field_index=*/0, JVMTI_HEAP_REFERENCE_FIELD)); + + ReferenceChainEvent event; + ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( + &mock_jvmti, &mock_jni, /*target_tag=*/3, &event)); + ASSERT_EQ(3u, event._chain.size()); + ASSERT_EQ(3u, event._edges.size()) + << "edge labels must align with the chain, one per hop"; + // Leaf first: chunk is retained via Base.base_f (parent entry's class is + // Base, ordinal 0 in Base's own space), then value2 via Holder's + // holder_a (ordinal 3 = interface offset 2 + Base's 1 + own position 0), + // then the static root edge's field name leakList (ordinal 4). + EXPECT_EQ("base_f", event._edges[0]); + EXPECT_EQ("holder_a", event._edges[1]); + EXPECT_EQ("leakList", event._edges[2]); + + // Interface-referrer branch: ISink's own-field ordinals have NO + // superclass-chain component (base = superinterfaces' fields only). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 4, 0, 0, FrontierEntryState::EDGE, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, + /*referrer_klass=*/0, /*class_tag=*/tagOf(isink), + /*referrer_field_index=*/1, /*edge_kind=*/0, + /*referrer_class_tag=*/tagOf(isink))); + ReferenceChainEvent iface_event; + ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( + &mock_jvmti, &mock_jni, /*target_tag=*/4, &iface_event)); + ASSERT_EQ(1u, iface_event._edges.size()); + EXPECT_EQ("CONST_B", iface_event._edges[0]); + + // Fail-safe: a referrer class that cannot be resolved degrades to the + // edge KIND label, never a fabricated name. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 5, 0, 0, FrontierEntryState::EDGE, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, + /*referrer_klass=*/0, /*class_tag=*/tagOf(holder), + /*referrer_field_index=*/0, /*edge_kind=*/0, + /*referrer_class_tag=*/tagOf(ibase))); + ReferenceChainEvent degraded_event; + ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( + &mock_jvmti, &mock_jni, /*target_tag=*/5, °raded_event)); + ASSERT_EQ(1u, degraded_event._edges.size()); + EXPECT_EQ("static_field", degraded_event._edges[0]); + + tracker->stop(); +} + +// PRIORITY_EXPAND_CAP backpressure: with the fast lane at the cap, the +// rotation collectors must stop pushing. The uncapped queue is what starved +// the BFS on-pod (39k->103k entries while _pending_expand never drained a +// single batch - rotation inflow 256/pass exceeded the deadline-bounded +// drain ~150-300/pass on every pass). +TEST_F(ReferenceChainsBfsTest, PriorityExpandCapStopsRotationCollectorPushes) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // One eligible stale-EXPANDED entry. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + + for (size_t i = 0; i < ReferenceChainsTestAccessor::priorityExpandCap(); i++) { + ReferenceChainsTestAccessor::pushPriorityExpand((jlong)(100 + i)); + } + std::vector selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(10); + EXPECT_TRUE(selected.empty()) + << "collector must stop pushing once _priority_expand hits the cap"; + + // With the lane drained (a pass's expand phase consumed it), the + // collector selects again. + ReferenceChainsTestAccessor::clearPriorityExpand(); + selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(10); + ASSERT_EQ(1u, selected.size()); + EXPECT_EQ((jlong)1, selected[0]); + + tracker->stop(); +} + +// The stale-expansion rotation must select leak parents from +// _leak_parent_fanout ahead of the blind table lap: the fanout entries are +// the EXPANDED parents that actually lead to watched leak-klass children, +// and neither the blind lap (~table_size/budget passes, hundreds live) nor +// the growth-gated leak-accumulation tier reaches them in steady state. +TEST_F(ReferenceChainsBfsTest, StaleRotationPrefersLeakParentsOverBlindLap) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + // Fanout parent 1 and an unrelated stale-EXPANDED entry 3. Parent 1 + // needs a non-zero class_tag: trackLeakAccumulation() attributes via the + // parent entry's class_tag and skips entries without one. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, + /*class_tag=*/42)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 3, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, 1, 10); + + // Budget 1: the fanout parent wins over the blind-lap entry. + std::vector selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(1); + ASSERT_EQ(1u, selected.size()); + EXPECT_EQ((jlong)1, selected[0]); + + // Budget covering both: fanout parent first, blind lap fills the rest. + ReferenceChainsTestAccessor::clearPriorityExpand(); + selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(2); + ASSERT_EQ(2u, selected.size()); + EXPECT_EQ((jlong)1, selected[0]); + EXPECT_EQ((jlong)3, selected[1]); + + tracker->stop(); +} + +// reparentToDurableRoot: a depth-1 entry first admitted through a transient +// root (stack/JNI local) is re-parented to a durable root-attached parent +// at equal depth - the case improveChain() cannot express (it requires a +// strictly deeper path), and exactly the hotdog shape where the singleton +// collection is a depth-0 static root and its elements depth 1. +TEST_F(ReferenceChainsBfsTest, ReparentToDurableRootSwapsTransientForDurable) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // tag 1: transient root (old parent). tag 2: target at depth 1 under it. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_JNI_LOCAL)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 2, 1, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + // tag 5: durable static root (new parent). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 5, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + // tag 6: another transient root - must never be swapped TO. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 6, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + + EXPECT_TRUE(frontier->reparentToDurableRoot(2, 5, 42)); + FrontierEntry entry{}; + ASSERT_TRUE(frontier->lookup(2, &entry)); + EXPECT_EQ((jlong)5, entry.parent_tag); + EXPECT_EQ((u32)42, entry.referrer_klass); + + // Transient new parent: no swap (would trade one noise root for + // another). + EXPECT_FALSE(frontier->reparentToDurableRoot(2, 6, 43)); + ASSERT_TRUE(frontier->lookup(2, &entry)); + EXPECT_EQ((jlong)5, entry.parent_tag) << "parent must be unchanged"; + + // Depth-2 targets are out of scope (judging root durability there + // would require walking both chains). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 7, 2, 2, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + EXPECT_FALSE(frontier->reparentToDurableRoot(7, 5, 44)); + + tracker->stop(); +} + +// recordDiscoveredInstance eviction: noise instances fill discovery slots +// first-come-first-served, but a leak-correlated discovery must evict a +// noise slot when all are full - without eviction, the 8 noise instances +// observed on-pod permanently blocked every later leak-tagged instance of +// the watched class. +TEST_F(ReferenceChainsBfsTest, RecordDiscoveredInstanceEvictsNoiseSlots) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kKlass = 3; + ReferenceChainsTestAccessor::setCandidateCountForTest(1); + ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, kKlass); + + const int cap = ReferenceChainsTestAccessor::maxDiscoveredPerClass(); + for (int d = 0; d < cap; d++) { + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest( + kKlass, /*tag=*/100 + d, /*leak_correlated=*/false); + } + EXPECT_EQ(cap, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + + // Noise beyond the cap is dropped, slots unchanged. + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest( + kKlass, 108, false); + EXPECT_EQ(cap, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + EXPECT_EQ((jlong)100, + ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 0)); + + // Leak-correlated discovery evicts the first noise slot (tag 100 has + // no frontier entry -> treated as uncorrelated). + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest( + kKlass, 200, true); + EXPECT_EQ(cap, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + EXPECT_EQ((jlong)200, + ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 0)); + EXPECT_EQ((jlong)101, + ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 1)); + + // Once every slot is leak-correlated, a further leak discovery is + // dropped (no eviction of real signal). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 200, 0, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + frontier->setLeakTag(200, ReferenceChainsTestAccessor::leakTagBase() + 1); + // Entries for the remaining noise slots so the eviction scan finds all + // slots leak-tagged. + for (int d = 1; d < cap; d++) { + jlong tag = 101 + (d - 1); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + frontier->setLeakTag(tag, ReferenceChainsTestAccessor::leakTagBase() + 2); + } + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest( + kKlass, 201, true); + EXPECT_EQ(cap, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + for (int d = 0; d < cap; d++) { + EXPECT_NE((jlong)201, + ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, d)); + } + + tracker->stop(); +} + +// correlateAdmittedLeakTag: a tracked instance the BFS admitted BEFORE it +// was leak-tagged carries a frontier tag on the object; correlating stores +// the leak tag ON the entry (chain events then emit targetTag = leak tag) +// and records the instance as discovered. Never retags the object. +TEST_F(ReferenceChainsBfsTest, CorrelateAdmittedLeakTagSetsEntryAndDiscovers) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kKlass = 3; + constexpr jlong kLeakTag = 0x40000000LL + 5; + ReferenceChainsTestAccessor::setCandidateCountForTest(1); + ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, kKlass); + + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 300, 0, 1, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + + EXPECT_TRUE(tracker->correlateAdmittedLeakTag(300, kLeakTag, kKlass)); + EXPECT_EQ(kLeakTag, (jlong)ReferenceChainsTestAccessor::frontierLeakTag(300)); + EXPECT_EQ(1, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + EXPECT_EQ((jlong)300, + ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 0)); + + // Idempotent: an already-correlated entry just returns true. + EXPECT_TRUE(tracker->correlateAdmittedLeakTag(300, kLeakTag, kKlass)); + EXPECT_EQ(kLeakTag, (jlong)ReferenceChainsTestAccessor::frontierLeakTag(300)); + + // Unknown tag: no crash, no discovery side effects. + EXPECT_FALSE(tracker->correlateAdmittedLeakTag(999, kLeakTag, kKlass)); + EXPECT_EQ(1, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + + tracker->stop(); +} + +// Retention-explanation gate on the discovered-instance chains (depth==0 +// always suppressed; depth==1 suppressed only for TRANSIENT roots - a +// depth-1 chain from a durable root is the real direct-retention shape): +// transient depth-1 must NOT be cached, durable depth-1 and depth-2 must. +TEST_F(PollWatchedTargetsTest, DiscoveredChainGateSuppressesTransientDepthOne) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*klass_id=*/3, /*rep=*/(jweak)obj); + + // First poll populates the candidate slots from LivenessTracker's + // population. No discovered instances yet. + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + FrontierTable *frontier = tracker->frontierTable(); + // Noise shape: transient root (JNI local frame) -> depth-1 instance. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 6, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_JNI_LOCAL)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 7, 6, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + // Real direct-retention shape: static-field root -> depth-1 instance. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 8, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 9, 8, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + // Deeper chain through the transient root: passes on depth alone. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 10, 7, 2, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + // Depth-0 transient root: the candidate instance itself held by a live + // frame - suppressed like the depth-1 transient shape. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 11, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + // Depth-0 durable root: the candidate instance IS the static field's + // value (the singleton-collection-itself shape) - a real direct-retention + // chain, NOT suppressible as noise. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 12, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 7, false); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 9, false); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 10, false); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 11, false); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 12, false); + ASSERT_EQ(5, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_FALSE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)) + << "depth-1 chain rooted at a transient (JNI local) root is noise"; + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(9)) + << "depth-1 chain rooted at a durable (static field) root is a real " + "direct-retention chain"; + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(10)) + << "depth-2 chain passes the gate regardless of root kind"; + EXPECT_FALSE(ReferenceChainsTestAccessor::hasResolvedChainForTag(11)) + << "depth-0 chain rooted at a transient (stack local) root is noise"; + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(12)) + << "depth-0 chain rooted at a durable (static field) root is the " + "direct-retention shape the search exists to report"; + + tracker->stop(); +} + +// Orphan slot sweep: a candidate that qualified long enough for the walk to +// record discovered instances, then stopped qualifying (its trend aged out +// of the poll's candidate list), must still get chains built for those +// instances. The slot persists by design precisely so the klass "can still +// be found there" - before the sweep, nothing iterated it once the klass +// left the poll candidates, stranding every instance recorded while it +// qualified (observed live: 8 discovered instances recorded the pass +// after the candidate's trend aged out were never built across the +// remaining 116 passes of the run). +TEST_F(PollWatchedTargetsTest, OrphanedSlotBuildsDiscoveredChainsAfterCandidateDropsOut) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*klass_id=*/3, /*rep=*/(jweak)obj); + + // First poll admits klass 3 into candidate slot 0 (nothing discovered + // yet, so nothing is built here). + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + ASSERT_EQ(0, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + + // The walk discovered an instance while the candidate still + // qualified: the real direct-retention shape (static-field root -> + // depth-1 instance), which the discovered-chain gate lets through. + FrontierTable *frontier = tracker->frontierTable(); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 9, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 8, 9, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 8, false); + ASSERT_EQ(1, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + + // The candidate stops qualifying: LivenessTracker's population table + // is wiped, so selectLeakCandidates() returns 0 on every poll from + // here on. Before the orphan sweep, this poll would leave the recorded + // instance permanently unbuildable. + LivenessTracker::instance()->klassPopulationResetForTest(); + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(8)) + << "discovered instances recorded while the candidate qualified must " + "still get chains built after it stops qualifying"; + + tracker->stop(); +} diff --git a/ddprof-lib/src/test/cpp/staleLeaf_ut.cpp b/ddprof-lib/src/test/cpp/staleLeaf_ut.cpp new file mode 100644 index 0000000000..d6fcf344ee --- /dev/null +++ b/ddprof-lib/src/test/cpp/staleLeaf_ut.cpp @@ -0,0 +1,129 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + * + * PROF-15130 investigation: spurious STALE LEAF frames. + * + * Hypothesis under test (from the ticket): convertNativeTrace() leaves a + * counted slot unwritten (depth counts a slot that was never populated), + * leaking stale data from a previous sample into the leaf of a parked + * thread's stack. + * + * These tests drive Profiler::convertNativeTrace() directly and assert the + * counted-slot invariant: every slot in [0, depth) returned by + * convertNativeTrace MUST have been written this call. The output buffer is + * pre-filled with a recognizable sentinel so any surviving sentinel below + * `depth` is, by definition, a counted-but-unwritten (stale) slot. + */ + +#include +#include +#include +#include "../../main/cpp/profiler.h" +#include "../../main/cpp/vmEntry.h" +#include "../../main/cpp/libraries.h" +#include "../../main/cpp/gtest_crash_handler.h" + +static constexpr char STALELEAF_TEST_NAME[] = "StaleLeafTest"; + +// Sentinel value pre-loaded into every output slot. Distinct bci so we can +// tell "never touched by convertNativeTrace" from any value the function +// could legitimately write (BCI_NATIVE_FRAME == -11, BCI_NATIVE_FRAME_REMOTE +// == -19). +static constexpr jint SENTINEL_BCI = 0x5A5A5A5A; +static const void* const SENTINEL_MID = (const void*)(uintptr_t)0xDEADBEEFDEADBEEFULL; + +class StaleLeafTest : public ::testing::Test { +protected: + void SetUp() override { + installGtestCrashHandler(); + // Without this, findNativeMethod() resolves every PC to nullptr and + // convertNativeTrace() never writes a frame, so depth is always 0 and + // the counted-slot invariant below is checked over an empty range. + Libraries::instance()->updateSymbols(false); + } + void TearDown() override { + restoreDefaultSignalHandlers(); + } + + static void fillSentinel(ASGCT_CallFrame* frames, int n) { + for (int i = 0; i < n; i++) { + frames[i].bci = SENTINEL_BCI; + frames[i].method_id = (jmethodID)SENTINEL_MID; + } + } + + // Returns true if every slot in [0, depth) was overwritten (no sentinel + // survives). A surviving sentinel below `depth` is a stale leaf bug. + static bool noCountedSlotIsStale(const ASGCT_CallFrame* frames, int depth) { + for (int i = 0; i < depth; i++) { + if (frames[i].bci == SENTINEL_BCI && + frames[i].method_id == (jmethodID)SENTINEL_MID) { + return false; + } + } + return true; + } +}; + +// Empty callchain: depth must be 0, nothing written. +TEST_F(StaleLeafTest, emptyCallchain_returnsZero_noStaleSlot) { + ASGCT_CallFrame frames[16]; + fillSentinel(frames, 16); + + int depth = Profiler::instance()->convertNativeTrace(0, nullptr, frames, 0, false); + + EXPECT_EQ(0, depth); + EXPECT_TRUE(noCountedSlotIsStale(frames, depth)); +} + +// PCs that resolve to no library: traditional path yields method_name == NULL, +// so the frame is neither written nor counted. depth must equal the number of +// slots actually written (0 here), with no stale slot below depth. +TEST_F(StaleLeafTest, unresolvablePcs_neverLeaveCountedStaleSlot) { + // Deliberately bogus PCs not inside any loaded library. + const void* callchain[] = { + (const void*)0x1000, + (const void*)0x2000, + (const void*)0x3000, + (const void*)0x4000, + }; + const int n = (int)(sizeof(callchain) / sizeof(callchain[0])); + + ASGCT_CallFrame frames[16]; + fillSentinel(frames, 16); + + int depth = Profiler::instance()->convertNativeTrace(n, callchain, frames, 0, false); + + ASSERT_GE(depth, 0); + ASSERT_LE(depth, n); + // INVARIANT: no counted slot may remain a sentinel. + EXPECT_TRUE(noCountedSlotIsStale(frames, depth)) + << "convertNativeTrace counted a slot it never wrote (stale leaf)"; +} + +// Mix of resolvable (test-binary code) and bogus PCs. Whatever resolves, the +// invariant must hold: depth counts only written slots. +TEST_F(StaleLeafTest, realCodePcs_neverLeaveCountedStaleSlot) { + // Use the address of this function and a few library functions as PCs that + // are plausibly inside loaded code segments. Even if symbol resolution + // returns NULL (library not parsed because the profiler is not started), + // the invariant is the same: counted == written. + const void* callchain[] = { + (const void*)&memcpy, + (const void*)&snprintf, + (const void*)0xBADC0DE, // unresolvable + (const void*)&strlen + }; + const int n = (int)(sizeof(callchain) / sizeof(callchain[0])); + + ASGCT_CallFrame frames[16]; + fillSentinel(frames, 16); + + int depth = Profiler::instance()->convertNativeTrace(n, callchain, frames, 0, false); + + ASSERT_GE(depth, 0); + ASSERT_LE(depth, n); + EXPECT_TRUE(noCountedSlotIsStale(frames, depth)) + << "convertNativeTrace counted a slot it never wrote (stale leaf)"; +} From 818e19d95c3225cafabdfea47974fa3b479608c9 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Wed, 16 Sep 2026 17:01:46 +0200 Subject: [PATCH 03/18] Add Java integration tests for reference chains Integration tests for ReferenceChain events on real recordings: leak-tag correlation, aggressive-leak, external-process, and thread-local scenarios, with JFR parsing, assertions, and test-seam coverage. Migrate ReferenceChainTrackingTest to the JfrEvents API and extend the shared test harness (external launcher, profiler test base). --- .../profiler/AbstractProcessProfilerTest.java | 13 +- .../profiler/AbstractProfilerTest.java | 7 + .../datadoghq/profiler/ExternalLauncher.java | 92 ++ .../JMethodIDInvalidationStressTest.java | 11 +- .../AggressiveLeakReferenceChainTest.java | 132 +++ .../ExternalProcessReferenceChainTest.java | 208 ++++ .../LeakTagCorrelationReferenceChainTest.java | 239 +++++ .../LeakTagCorrelationScenario.java | 510 +++++++++ .../referencechains/LeakingCacheScenario.java | 353 +++++++ .../ReferenceChainAssertions.java | 216 ++++ .../ReferenceChainJfrParserTest.java | 192 ++++ .../ReferenceChainTestSeamsTest.java | 203 ++++ .../ReferenceChainTrackingTest.java | 976 ++++++++++++++++++ .../StaticFieldGrowingCollectionScenario.java | 278 +++++ .../ThreadLocalLeakReferenceChainTest.java | 160 +++ .../ThreadLocalLeakScenario.java | 351 +++++++ 16 files changed, 3935 insertions(+), 6 deletions(-) create mode 100644 ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/AggressiveLeakReferenceChainTest.java create mode 100644 ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ExternalProcessReferenceChainTest.java create mode 100644 ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/LeakTagCorrelationReferenceChainTest.java create mode 100644 ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/LeakTagCorrelationScenario.java create mode 100644 ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/LeakingCacheScenario.java create mode 100644 ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainAssertions.java create mode 100644 ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainJfrParserTest.java create mode 100644 ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainTestSeamsTest.java create mode 100644 ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainTrackingTest.java create mode 100644 ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/StaticFieldGrowingCollectionScenario.java create mode 100644 ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakReferenceChainTest.java create mode 100644 ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakScenario.java diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/AbstractProcessProfilerTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/AbstractProcessProfilerTest.java index 6c0bc0685f..e213b9a005 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/AbstractProcessProfilerTest.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/AbstractProcessProfilerTest.java @@ -36,6 +36,17 @@ protected final LaunchResult launch(String target, List jvmArgs, String } protected final LaunchResult launch(String target, List jvmArgs, String commands, Map env, Function onStdoutLine, Function onStderrLine) throws Exception { + return launch(target, jvmArgs, commands, env, 10, onStdoutLine, onStderrLine); + } + + /** + * Same as the 6-arg overload above, but with a caller-supplied wait timeout instead of the + * hardcoded 10 seconds - needed by scenarios that run substantially longer than a plain + * attach/init check, e.g. {@code ExternalProcessReferenceChainTest}'s population-growth loop + * (up to 25 rounds plus a grace period, ~20s+ observed in-process, plus a whole separate + * JVM's own startup/classloading cost on top). + */ + protected final LaunchResult launch(String target, List jvmArgs, String commands, Map env, long timeoutSeconds, Function onStdoutLine, Function onStderrLine) throws Exception { String javaHome = System.getenv("JAVA_TEST_HOME"); if (javaHome == null) { javaHome = System.getenv("JAVA_HOME"); @@ -130,7 +141,7 @@ protected final LaunchResult launch(String target, List jvmArgs, String stdoutReader.start(); stderrReader.start(); - boolean val = p.waitFor(10, TimeUnit.SECONDS); + boolean val = p.waitFor(timeoutSeconds, TimeUnit.SECONDS); if (!val) { p.destroyForcibly(); p.waitFor(5, TimeUnit.SECONDS); diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/AbstractProfilerTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/AbstractProfilerTest.java index b4e988c582..923d03d031 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/AbstractProfilerTest.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/AbstractProfilerTest.java @@ -73,6 +73,12 @@ public static double scaledSize(JfrEvent item) { protected JavaProfiler profiler; private Path jfrDump; + // Set at the very start of setupProfiler(), before getProfilerCommand() is + // consulted, so a subclass can branch its command on the current test + // method (e.g. distinct hop/budget/frontier-cap values per @Test) without + // needing separate test classes per configuration. + protected TestInfo testInfo; + private Duration cpuInterval; private Duration wallInterval; @@ -192,6 +198,7 @@ protected void withTestAssumptions() {} @BeforeEach public void setupProfiler(TestInfo testInfo) throws Exception { + this.testInfo = testInfo; Assumptions.assumeTrue(isPlatformSupported()); withTestAssumptions(); diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/ExternalLauncher.java b/ddprof-test/src/test/java/com/datadoghq/profiler/ExternalLauncher.java index 2dbb429668..a08674d004 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/ExternalLauncher.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/ExternalLauncher.java @@ -5,10 +5,16 @@ package com.datadoghq.profiler; +import com.datadoghq.profiler.referencechains.LeakTagCorrelationScenario; +import com.datadoghq.profiler.referencechains.LeakingCacheScenario; +import com.datadoghq.profiler.referencechains.StaticFieldGrowingCollectionScenario; +import com.datadoghq.profiler.referencechains.ThreadLocalLeakScenario; + import java.io.IOException; import java.lang.management.ManagementFactory; import java.lang.management.ThreadMXBean; import java.lang.reflect.Method; +import java.nio.file.Paths; import java.util.Random; import java.util.concurrent.atomic.LongAdder; @@ -30,6 +36,26 @@ * CPU concurrently on the main thread, on a plain {@code new Thread(Runnable)} and on a * two-level {@link Thread} subclass, and stops the profiler again. The resulting recording * holds samples rooted at each of the three thread entry points; see {@code EntryFrameTest} + *

  • leak-cache "<start command>|||<scratch dump path>" - starts the profiler with + * the given start command, then runs {@link LeakingCacheScenario} in this process and exits + * directly (no readiness/stdin-signal handshake, unlike the other modes above) once it + * prints its one result line. Used by {@code ExternalProcessReferenceChainTest} + * (referencechains package) to prove the reference-chains mechanism end-to-end against a + * genuinely separate JVM, not the in-process dynamic-attach lifecycle every other test in + * this module uses - see that scenario's own class comment for why a *separate* process is + * load-bearing here (the one-shot root-seeded BFS walk is a process-wide, once-ever + * resource).
  • + *
  • leak-static-field "<start command>|||<scratch dump path>" - same packing and + * lifecycle as {@code leak-cache} above, but runs {@link StaticFieldGrowingCollectionScenario} + * instead: a {@code static final List} field appended to (never reassigned) after + * its owning class has already been swept once by {@code admitStaticFieldRoots()} - + * mirroring the real leak generator found in the {@code prof-analyzer-hotdog-jb} pod.
  • + *
  • leak-correlation "<start command>|||<scratch dump path>" - same packing and + * lifecycle as {@code leak-cache} above, but runs {@link LeakTagCorrelationScenario} + * instead: a static-collection leak coexisting with ephemeral stack-local noise of another + * class and a large live filler graph, asserting the emitted chain's targetTag is a + * leak-tag-pool tag matching a HeapLiveObject's leakTag - the chain-to-live-heap-sample + * correlation use case.
  • * */ public class ExternalLauncher { @@ -201,6 +227,72 @@ public static void main(String[] args) throws Exception { worker.start(); } } + } else if (args[0].equals("leak-cache")) { + // "|||" packed into one args[1] string rather + // than extending AbstractProcessProfilerTest.launch()'s generic 2-arg + // (target, commands) contract with a 3rd argument every other mode would have to + // ignore. + String packed = args.length == 2 ? args[1] : ""; + int sep = packed.indexOf("|||"); + if (sep < 0) { + throw new IllegalArgumentException( + "leak-cache requires \"|||\", got: " + packed); + } + String commands = packed.substring(0, sep); + String scratchPath = packed.substring(sep + "|||".length()); + JavaProfiler instance = JavaProfiler.getInstance(); + // LeakingCacheScenario.run() itself calls instance.execute(commands), not here - + // it needs to seed its cache fixture *before* starting the profiler (see that + // method's own comment for why). + LeakingCacheScenario.run(instance, commands, Paths.get(scratchPath)); + System.out.flush(); + // Deliberately exits here rather than falling through to the shared + // "[ready]" + stdin-signal handshake below: this mode runs to completion in one + // shot (no live back-and-forth with the parent needed) and its own result line + // has already been printed by LeakingCacheScenario.run(). + System.exit(0); + } else if (args[0].equals("leak-static-field")) { + // Same "|||" packing as leak-cache above. + String packed = args.length == 2 ? args[1] : ""; + int sep = packed.indexOf("|||"); + if (sep < 0) { + throw new IllegalArgumentException( + "leak-static-field requires \"|||\", got: " + packed); + } + String commands = packed.substring(0, sep); + String scratchPath = packed.substring(sep + "|||".length()); + JavaProfiler instance = JavaProfiler.getInstance(); + StaticFieldGrowingCollectionScenario.run(instance, commands, Paths.get(scratchPath)); + System.out.flush(); + System.exit(0); + } else if (args[0].equals("leak-correlation")) { + // Same "|||" packing as leak-cache above. + String packed = args.length == 2 ? args[1] : ""; + int sep = packed.indexOf("|||"); + if (sep < 0) { + throw new IllegalArgumentException( + "leak-correlation requires \"|||\", got: " + packed); + } + String commands = packed.substring(0, sep); + String scratchPath = packed.substring(sep + "|||".length()); + JavaProfiler instance = JavaProfiler.getInstance(); + LeakTagCorrelationScenario.run(instance, commands, Paths.get(scratchPath)); + System.out.flush(); + System.exit(0); + } else if (args[0].equals("threadlocal-leak")) { + // Same "|||" packing as leak-cache above. + String packed = args.length == 2 ? args[1] : ""; + int sep = packed.indexOf("|||"); + if (sep < 0) { + throw new IllegalArgumentException( + "threadlocal-leak requires \"|||\", got: " + packed); + } + String commands = packed.substring(0, sep); + String scratchPath = packed.substring(sep + "|||".length()); + JavaProfiler instance = JavaProfiler.getInstance(); + ThreadLocalLeakScenario.run(instance, commands, Paths.get(scratchPath)); + System.out.flush(); + System.exit(0); } } finally { System.out.println("[ready]"); diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/memleak/JMethodIDInvalidationStressTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/memleak/JMethodIDInvalidationStressTest.java index 443fbd635b..e171e6323f 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/memleak/JMethodIDInvalidationStressTest.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/memleak/JMethodIDInvalidationStressTest.java @@ -42,7 +42,7 @@ /** * Exploratory stress test for jmethodID invalidation, motivated by PROF-15385 (SIGSEGV in * {@code Lookup::fillJavaMethodInfo} copying a JVMTI line-number table for a stale jmethodID, - * fixed by guarding that copy with {@code SafeAccess::isReadableRange}). + * fixed by guarding that copy with {@code SafeAccess::safeCopy}). * *

    That fix addressed one call site. This test does not target a specific call site; it tries * to manufacture the same underlying condition — jmethodIDs whose declaring class is unloaded @@ -69,10 +69,11 @@ * But that alone would pass vacuously if the churn never actually raced class unload against * {@code Lookup::resolveMethod}/{@code fillJavaMethodInfo}. To rule that out, this test reads the * native whitebox counters ({@code JavaProfiler#getDebugCounters()}) for {@code - * jmethodid_skipped_count} and {@code line_number_table_unreadable} -- both are incremented in - * {@code fillJavaMethodInfo} exactly when {@code SafeAccess::isReadableRange} rejects a stale - * jmethodID's class/method metadata or line-number table (see flightRecorder.cpp) -- and checks - * that at least one of them increased during the churn window. Whether the race is actually hit + * jmethodid_skipped_count} and {@code line_number_table_unreadable} -- incremented in + * {@code fillJavaMethodInfo} when a stale jmethodID's class/method-name probe is rejected by + * {@code SafeAccess::isReadableRange}, or when its line-number table copy is rejected by + * {@code SafeAccess::safeCopy} (see flightRecorder.cpp) -- and checks that at least one of them + * increased during the churn window. Whether the race is actually hit * within the window is JVM/host-discretionary, so that check is a JUnit assumption rather than an * assertion: if neither counter moved, the test is reported as skipped (not failed), since a * healthy host that simply didn't race tightly enough this run is not evidence of a regression. A diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/AggressiveLeakReferenceChainTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/AggressiveLeakReferenceChainTest.java new file mode 100644 index 0000000000..b3a0cfd247 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/AggressiveLeakReferenceChainTest.java @@ -0,0 +1,132 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.referencechains; + +import com.datadoghq.profiler.AbstractProfilerTest; +import com.datadoghq.profiler.JavaProfiler; +import com.datadoghq.profiler.Platform; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.junit.jupiter.api.Assumptions.assumeTrue; + +/** + * E2E coverage for {@code LivenessTracker::secondsToOOM()}'s urgent-OOM bypass + * ({@code ReferenceChainTracker::hasLeakSignal()}, referenceChains.cpp). Complements + * {@link ReferenceChainTrackingTest}'s organic coverage of a slow leak - a gradually-growing + * population that clears {@code selectLeakCandidates()}'s per-klass hysteresis gate over many + * rounds - with the opposite, previously-unexercised case: a heap-wide leak growing fast enough + * that a real deployment would OOM before any single klass's own population history ever clears + * that same hysteresis gate. + * + *

    Reproducing a genuine near-OOM heap in this shared, no-{@code forkEvery} test JVM is + * impractical (see {@link ReferenceChainTrackingTest}'s own header comment on the JVM this class + * shares with every other {@code ddprof-test} class). The debug-only + * {@code heapFloorRecordForTest0()}/{@code setMaxHeapBytesForTest0()} seams (javaApi.cpp) instead + * seed {@code LivenessTracker}'s aggregate heap-floor-ring history and its max-heap-bytes cache + * directly, so {@code secondsToOOM()}'s projection can be driven under + * {@code OOM_URGENT_THRESHOLD_S} (referenceChains.h) deterministically, decoupled from this JVM's + * real {@code -Xmx} and from real GC timing. {@code shouldRunPassForTest0()} then reads + * {@code ReferenceChainTracker::shouldRunPass()} - the search-restart gate {@code hasLeakSignal()} + * feeds - directly, rather than running a real pass, so this class never depends on + * {@code selectLeakCandidates()} ranking anything, a real JVMTI heap walk, or a JFR event. + */ +public class AggressiveLeakReferenceChainTest extends AbstractProfilerTest { + + @Override + protected String getProfilerCommand() { + // generations=true: gates LivenessTracker::gcGenerationsEnabled() - hasLeakSignal() (which + // shouldRunPassForTest0() ultimately reads) short-circuits to true immediately when this is + // unset (its own first check, referenceChains.cpp), which would make every test below + // trivially pass regardless of the urgent-OOM bypass this class exists to exercise. + // referencechains=true: constructs the FrontierTable/ReferenceChainTracker singleton at all - + // shouldRunPassForTest0() has nothing to read otherwise. + return "generations=true,referencechains=true:hops=32:budget=500:ttl=60000:framecap=2000000"; + } + + @Override + protected boolean isPlatformSupported() { + // Mirrors ReferenceChainTrackingTest's own guard - FollowReferences/tag-based frontier walking + // assumes a HotSpot-shaped JVMTI heap implementation. + return !(Platform.isJavaVersion(8) || Platform.isJ9() || Platform.isZing()); + } + + private static void assumeDebugBuild() { + assumeTrue("debug".equals(System.getProperty("ddprof_test.config")), + "heapFloorRecordForTest0/setMaxHeapBytesForTest0/shouldRunPassForTest0 only exist in a " + + "debug native build (javaApi.cpp's #ifdef DEBUG guard)"); + } + + /** + * Seeds ten heap-floor-ring samples rising fast enough that {@code secondsToOOM()}'s projection + * lands well under {@code OOM_URGENT_THRESHOLD_S} (300s), with LivenessTracker's per-klass + * population table left empty throughout - {@code selectLeakCandidateKlassIds0()} returns + * nothing at any point in this test. Asserts the search-restart gate still opens, proving the + * urgent-OOM projection alone - not a per-klass candidate - is what let it through. + */ + @Test + public void shouldOpenSearchGateOnAggressiveHeapWideGrowthWithNoLeakCandidate() { + assumeDebugBuild(); + JavaProfiler.setHeapFloorRecordingForTest0(false); + JavaProfiler.resetKlassPopulationForTest0(); + JavaProfiler.resetReferenceChainSearchForTest0(); + try { + // 1000 MiB fake max heap, rising from 100 MiB to 991 MiB over 9 (fake) seconds: ~99 MiB/s, + // 9 MiB of headroom left at the last sample -> projected time-to-OOM ~= 9/99 s, far under + // OOM_URGENT_THRESHOLD_S = 300s. + JavaProfiler.setMaxHeapBytesForTest0(1000L * 1024 * 1024); + for (int i = 0; i < 10; i++) { + long usedBytes = (100L + i * 99L) * 1024 * 1024; + long timestampNs = i * 1_000_000_000L; + JavaProfiler.heapFloorRecordForTest0(usedBytes, timestampNs); + } + + int[] candidates = JavaProfiler.selectLeakCandidateKlassIds0(); + assertTrue(candidates == null || candidates.length == 0, + "This test's own precondition: no per-klass candidate should exist, so a true result " + + "below can only come from the aggregate urgent-OOM bypass"); + + assertTrue(JavaProfiler.shouldRunPassForTest0(), + "Expected the search-restart gate to open from the urgent heap-wide OOM projection " + + "alone, with zero per-klass leak candidate"); + } finally { + JavaProfiler.setMaxHeapBytesForTest0(-1); + JavaProfiler.setHeapFloorRecordingForTest0(true); + } + } + + /** + * Same empty per-klass population table as above, but a flat heap floor (no growth at all) - + * {@code secondsToOOM()} has nothing to project from, so this test's own gate must stay closed. + * Without this control, a stubbed-out {@code secondsToOOM()} that always signals urgency would + * pass the test above just as well. + */ + @Test + public void shouldNotOpenSearchGateOnFlatHeapFloorWithNoLeakCandidate() { + assumeDebugBuild(); + JavaProfiler.setHeapFloorRecordingForTest0(false); + JavaProfiler.resetKlassPopulationForTest0(); + JavaProfiler.resetReferenceChainSearchForTest0(); + try { + JavaProfiler.setMaxHeapBytesForTest0(1000L * 1024 * 1024); + for (int i = 0; i < 10; i++) { + JavaProfiler.heapFloorRecordForTest0(200L * 1024 * 1024, i * 1_000_000_000L); + } + + int[] candidates = JavaProfiler.selectLeakCandidateKlassIds0(); + assertTrue(candidates == null || candidates.length == 0, + "This test's own precondition: no per-klass candidate should exist either"); + + assertFalse(JavaProfiler.shouldRunPassForTest0(), + "Expected the search-restart gate to stay closed: no heap growth and no per-klass " + + "candidate give hasLeakSignal() nothing to trust"); + } finally { + JavaProfiler.setMaxHeapBytesForTest0(-1); + JavaProfiler.setHeapFloorRecordingForTest0(true); + } + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ExternalProcessReferenceChainTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ExternalProcessReferenceChainTest.java new file mode 100644 index 0000000000..cb91ad34f8 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ExternalProcessReferenceChainTest.java @@ -0,0 +1,208 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.referencechains; + +import com.datadoghq.profiler.AbstractProcessProfilerTest; +import com.datadoghq.profiler.Platform; +import org.junit.jupiter.api.Tag; +import org.junit.jupiter.api.Test; + +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.Collections; +import java.util.List; +import java.util.concurrent.atomic.AtomicReference; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.junit.jupiter.api.Assumptions.assumeFalse; + +/** + * Genuinely separate-process end-to-end coverage for the reference-chains target-selection + * mechanism, complementing {@link ReferenceChainTrackingTest}'s in-process coverage: runs a real + * "leaking Java app" (an ever-growing, never-evicted {@code HashMap}-backed cache - see + * {@link LeakingCacheScenario}) in a genuinely separate child JVM, launched the same way + * {@code JavaProfilerTest}/{@code JVMAccessTest} already launch child processes, and asserts on + * that child's own reported result rather than reading a JFR file back out of this test's own + * process. + * + *

    Why a separate process, not another {@code @Test} in {@code ReferenceChainTrackingTest}: + * an earlier attempt added a second success-path {@code @Test} method there, using this exact + * same {@code HashMap}-based leak shape, sharing that class's in-process {@code AbstractProfilerTest} + * dynamic-attach lifecycle. It failed reliably whenever it ran after + * {@code ReferenceChainTrackingTest}'s own {@code ChainLink} test in the same JVM - not a + * test-ordering bug, but a genuine, non-obvious property of {@code ReferenceChainTracker}: + * {@code runPass()} (referenceChains.cpp) performs exactly one root-seeded {@code FollowReferences} + * walk per search's *entire lifetime* ({@code _search_started}); every later pass only expands + * frontier entries that walk already discovered ({@code expandFrontier()}) - it never + * re-examines GC roots or revisits an already-{@code EXPANDED} entry for newly-added children. + * Since {@code ReferenceChainTracker} is a process-wide singleton, only the *first* test to ever + * call {@code runPass()} in a given JVM gets a real root-seeded walk; every subsequent test's own, + * independently-rooted local variables are structurally invisible to the mechanism afterward, no + * matter how they are built. A genuinely separate child JVM per scenario sidesteps this + * entirely: each process gets its own fresh {@code ReferenceChainTracker} singleton, and + * therefore its own guaranteed first-ever root walk. + */ +@Tag("slow") +public class ExternalProcessReferenceChainTest extends AbstractProcessProfilerTest { + + @Test + void shouldReconstructReferrerChainInSeparateProcess() throws Exception { + // Mirrors ReferenceChainTrackingTest.isPlatformSupported()'s own guard - FollowReferences/ + // tag-based frontier walking assumes a HotSpot-shaped JVMTI heap implementation. + assumeFalse(Platform.isJavaVersion(8)); + assumeFalse(Platform.isJ9()); + assumeFalse(Platform.isZing()); + + Path scratchDumpPath = Files.createTempFile("referencechains-external-process", ".jfr"); + // LeakingCacheScenario.run() (inside the child) creates this file on its own first dump() - + // an empty placeholder here would make that first dump() attempt fail confusingly. + Files.deleteIfExists(scratchDumpPath); + // "start" requires a "jfr,file=..." clause (found the hard way: JavaProfiler.execute() + // throws "Flight Recorder output file is not specified" without one) even though this + // scenario never reads its content - only the explicit dump() calls + // LeakingCacheScenario.run() makes to scratchDumpPath matter. + Path continuousJfrPath = Files.createTempFile("referencechains-external-process-continuous", ".jfr"); + try { + // budget=200000 (up from the in-process test's 4000, found via TEST_LOG instrumentation + // in referenceChains.cpp/livenessTracker.cpp): a genuinely fresh external JVM's + // reachable-from-roots graph (full JUnit/JMC/Gradle-worker classpath, bootstrapped from + // scratch - no benefit from a warm, already-running shared worker JVM) is far larger than + // the in-process test's, and each pass's own JVMTI walk cost scales with cumulative + // frontier size, not with this scenario's own allocation rate - budget=4000 stayed + // truncated=1 at 76000+ admitted edges after 19 real passes and never got close to this + // scenario's own cache before the test's own timeout. _budget is the pacing controller's + // hard ceiling (see updatePacing()'s own comment, referenceChains.cpp) - pausetarget alone + // cannot compensate for a ceiling set too low, only pacing *within* it. + // memory=64:l:1.0 - the explicit 100% live-sample ratio removes the + // default 10% subsampling lottery from the tracked-instance pool + // (see LeakTagCorrelationReferenceChainTest's own comment for the + // full analysis); total volume stays capped by the 256KiB interval + // floor, so the tracking table remains tiny. + String startCommand = "start,memory=64:l:1.0,generations=true," + + "referencechains=true:hops=64:budget=200000:ttl=120000:framecap=2000000:pausetarget=60000" + + ",jfr,file=" + continuousJfrPath.toAbsolutePath(); + // Packed into one args[1] string - see ExternalLauncher's own "leak-cache" mode comment + // for why (avoids extending AbstractProcessProfilerTest.launch()'s generic (target, + // commands) contract with a 3rd argument every other mode would have to ignore). + String packedCommand = startCommand + "|||" + scratchDumpPath.toAbsolutePath(); + + // Propagates this JVM's own ddprof_test.config (set by ProfilerTestPlugin.kt on the + // ddprof-test Test task itself, not inherited by a ProcessBuilder-launched child on its + // own) so LeakingCacheScenario can use the same debug-only seeded-representative fallback + // ReferenceChainTrackingTest already relies on instead of depending purely on real + // allocation-sampling timing. + List jvmArgs = Collections.singletonList( + "-Dddprof_test.config=" + System.getProperty("ddprof_test.config")); + + AtomicReference resultLine = new AtomicReference<>(); + LaunchResult result = launch("leak-cache", jvmArgs, packedCommand, + Collections.emptyMap(), + // Generous: a whole separate JVM's startup/classloading cost, on top of the same + // up-to-25-round population-growth loop plus grace period that takes ~20s in-process + // (ReferenceChainTrackingTest's own history). + 90, + line -> { + if (line.startsWith(LeakingCacheScenario.FOUND_MARKER) + || line.equals(LeakingCacheScenario.NOT_FOUND_MARKER) + || line.startsWith(LeakingCacheScenario.NO_HASHMAP_INTERNALS_MARKER) + || line.startsWith(LeakingCacheScenario.CHAIN_NOT_PERSISTED_MARKER)) { + resultLine.set(line); + } + return LineConsumerResult.CONTINUE; + }, + null); + + assertTrue(result.inTime, "Child process did not exit within the wait timeout"); + assertEquals(0, result.exitCode, "Child process exited with a non-zero code"); + assertNotNull(resultLine.get(), + "Child process never printed a recognizable result marker on stdout"); + assertEquals( + LeakingCacheScenario.FOUND_MARKER + LeakingCacheScenario.CachedPayload.class.getName(), + resultLine.get(), + "Expected a successfully reconstructed chain whose leaf is CachedPayload, threading " + + "through java.util.HashMap's own internal storage - got: " + resultLine.get()); + } finally { + Files.deleteIfExists(scratchDumpPath); + Files.deleteIfExists(continuousJfrPath); + } + } + + /** + * Mirrors {@link #shouldReconstructReferrerChainInSeparateProcess()}, but against {@link + * StaticFieldGrowingCollectionScenario} instead of {@link LeakingCacheScenario}: a {@code + * static final List} field appended to (never reassigned) after its owning class has + * already been swept once by {@code admitStaticFieldRoots()} (referenceChains.cpp) - the exact + * shape found in the {@code prof-analyzer-hotdog-jb} pod's real leak generator, as opposed to + * {@code LeakingCacheScenario}'s local-variable (stack-root) cache. Local, fast reproduction of + * whether {@code collectStaleExpandedEntriesForRotation()}'s rotation mechanism actually + * re-expands an already-{@code EXPANDED} static-field-rooted collection to pick up elements + * added after its one-time sweep. + */ + @Test + void shouldReconstructReferrerChainForGrowingStaticFieldCollection() throws Exception { + assumeFalse(Platform.isJavaVersion(8)); + assumeFalse(Platform.isJ9()); + assumeFalse(Platform.isZing()); + + Path scratchDumpPath = Files.createTempFile("referencechains-static-field-external-process", ".jfr"); + Files.deleteIfExists(scratchDumpPath); + Path continuousJfrPath = Files.createTempFile("referencechains-static-field-external-process-continuous", ".jfr"); + try { + // See ExternalProcessReferenceChainTest.shouldReconstructReferrerChainInSeparateProcess()'s + // own comment for why budget/pausetarget are raised this far above the in-process test's + // defaults for a genuinely fresh, cold external JVM. + // memory=64:l:1.0 - the explicit 100% live-sample ratio removes the + // default 10% subsampling lottery from the tracked-instance pool + // (see LeakTagCorrelationReferenceChainTest's own comment for the + // full analysis); total volume stays capped by the 256KiB interval + // floor, so the tracking table remains tiny. + String startCommand = "start,memory=64:l:1.0,generations=true," + + "referencechains=true:hops=64:budget=200000:ttl=120000:framecap=2000000:pausetarget=60000" + + ",jfr,file=" + continuousJfrPath.toAbsolutePath(); + String packedCommand = startCommand + "|||" + scratchDumpPath.toAbsolutePath(); + + List jvmArgs = Collections.singletonList( + "-Dddprof_test.config=" + System.getProperty("ddprof_test.config")); + + AtomicReference resultLine = new AtomicReference<>(); + LaunchResult result = launch("leak-static-field", jvmArgs, packedCommand, + Collections.emptyMap(), + // Wider than shouldReconstructReferrerChainInSeparateProcess()'s 90s: this scenario's + // own tail wait (StaticFieldGrowingCollectionScenario.run()) needs up to ~60s on top of + // the round loop for a cold JVM's first full pass to actually complete - see that + // method's own comment. + 150, + line -> { + if (line.startsWith(StaticFieldGrowingCollectionScenario.FOUND_MARKER) + || line.equals(StaticFieldGrowingCollectionScenario.NOT_FOUND_MARKER)) { + resultLine.set(line); + } + return LineConsumerResult.CONTINUE; + }, + null); + + assertTrue(result.inTime, "Child process did not exit within the wait timeout"); + assertEquals(0, result.exitCode, "Child process exited with a non-zero code"); + assertNotNull(resultLine.get(), + "Child process never printed a recognizable result marker on stdout"); + assertEquals( + // "byte[]", not byte[].class.getName()'s "[B": StaticFieldGrowingCollectionScenario + // prints match.chain.get(0).getFullName(), and JMC's IMCType renders array types in + // Java source notation, not JVM signature notation - see + // ReferenceChainAssertions.jmcStyleName()'s own comment for why matching required the + // same distinction. + StaticFieldGrowingCollectionScenario.FOUND_MARKER + "byte[]", + resultLine.get(), + "Expected a successfully reconstructed chain to a byte[] chunk appended to the static " + + "field's list after its one-time admitStaticFieldRoots() sweep - got: " + resultLine.get()); + } finally { + Files.deleteIfExists(scratchDumpPath); + Files.deleteIfExists(continuousJfrPath); + } + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/LeakTagCorrelationReferenceChainTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/LeakTagCorrelationReferenceChainTest.java new file mode 100644 index 0000000000..656be82fa1 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/LeakTagCorrelationReferenceChainTest.java @@ -0,0 +1,239 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.referencechains; + +import com.datadoghq.profiler.AbstractProcessProfilerTest; +import com.datadoghq.profiler.Platform; +import org.junit.jupiter.api.Tag; +import org.junit.jupiter.api.Test; + +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.ArrayDeque; +import java.util.ArrayList; +import java.util.Collections; +import java.util.Deque; +import java.util.List; +import java.util.concurrent.atomic.AtomicBoolean; +import java.util.concurrent.atomic.AtomicReference; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.junit.jupiter.api.Assertions.fail; +import static org.junit.jupiter.api.Assumptions.assumeFalse; + +/** + * End-to-end coverage of the leak-tag correlation use case in a genuinely separate child + * JVM (same rationale as {@code ExternalProcessReferenceChainTest}'s own class comment - + * the one-shot root-seeded BFS walk is a per-process resource): a growing static-collection + * leak amid ephemeral stack-local noise and a large live filler graph must be reported as a + * {@code datadog.ReferenceChain} whose {@code targetTag} is in the leak-tag pool's range + * AND matches a {@code datadog.HeapLiveObject}'s {@code leakTag}. See + * {@link LeakTagCorrelationScenario}'s own header for what each ingredient reproduces. + * + *

    Diagnostics on failure: unlike the pod-debugging loop this scenario exists to + * replace, the child's {@code TEST_LOG} stream (queue depths, batch sizes, interception + * counts - the same evidence collected off a live pod's logs) arrives on this process's + * stdout consumer. A failing run embeds a filtered tail of it in the assertion message, so + * the failure mode is diagnosable without redeploying anything. + */ +@Tag("slow") +public class LeakTagCorrelationReferenceChainTest extends AbstractProcessProfilerTest { + + // Mirrored from referenceChains.h (LEAK_TAG_BASE/LEAK_TAG_POOL_SIZE) - the contract under + // test, asserted against the runtime event values the child reports. + private static final long LEAK_TAG_BASE = 0x40000000L; + private static final long LEAK_TAG_POOL_SIZE = 256; + + @Test + void shouldCorrelateLeakChainTargetTagWithLiveObjectLeakTag() throws Exception { + assertCorrelationFound("pausetarget=60000"); + } + + /** + * The pod-regime variant of the correlation use case: the same scenario under a + * pod-like tight pass window (pausetarget in single-digit ms, as on the production-like + * pod whose expand window was ~10ms against a 14-41ms GetObjectsWithTags floor at a + * 242k-entry tag map, ev-leaktag-onpod-round4). Full correlation cannot be the pass + * criterion at this window width: the scenario's trend gates were tuned for the loose + * window's crawl/GC interleaving, and under tight windows the leak klass's real-fold + * trend can flip within a single run - the same knife-edge the pod itself lives on + * (chains arrive via the plain crawl path while pool tagging/interception loses the + * race; observed: [correlation-tag-out-of-pool] with zero interceptions). What CAN be + * asserted deterministically here is the machinery this window exists to exercise: + * the leak-accumulation rotation must engage and select FRONTIER-state (un-expanded + * backlog) holders - the round-4 fix whose absence made that tier select zero on the + * pod while holding 51k known parent candidates. GetObjectsWithTags batch control is + * covered unit-level (gotwWindowNs() widening): a real child JVM's local tag map + * (~100k filler) keeps the per-call floor under the nominal window, so the widening + * regime cannot be reproduced end-to-end locally without a 242k-entry tag map. + */ + @Test + void shouldRotateFrontierLeakHoldersUnderPodLikeTightWindows() throws Exception { + assertCorrelationFound("pausetarget=10", /*requireFrontierRotation=*/ true); + } + + /** + * Drives the leak-correlation child JVM and asserts the correlated chain, with the + * referencechains pausetarget given by {@code pauseTargetOption} (everything else held + * fixed across the two callers so the variant differs in exactly the pass-window regime). + */ + private void assertCorrelationFound(String pauseTargetOption) throws Exception { + assertCorrelationFound(pauseTargetOption, /*requireFrontierRotation=*/ false); + } + + /** + * Drives the leak-correlation child JVM and asserts the correlated chain, with the + * referencechains pausetarget given by {@code pauseTargetOption} (everything else held + * fixed across the two callers so the variant differs in exactly the pass-window regime). + * With {@code requireFrontierRotation} (the tight-window pod-regime variant), the + * leak-accumulation rotation's FRONTIER-state selection is additionally asserted, and + * the correlation result markers that the pod regime legitimately produces without the + * full correlation completing (not-found, tag-out-of-pool - see the tight variant's own + * comment) are accepted instead of failed. + */ + private void assertCorrelationFound(String pauseTargetOption, + boolean requireFrontierRotation) throws Exception { + assumeFalse(Platform.isJavaVersion(8)); + assumeFalse(Platform.isJ9()); + assumeFalse(Platform.isZing()); + + Path scratchDumpPath = Files.createTempFile("referencechains-correlation-", ".jfr"); + Files.deleteIfExists(scratchDumpPath); + Path continuousJfrPath = Files.createTempFile("referencechains-correlation-continuous-", ".jfr"); + // Rolling tail of the child's TEST_LOG stream for failure diagnostics - bounded so a + // long run cannot grow this without limit. + Deque testLogTail = new ArrayDeque<>(); + AtomicBoolean frontierRotationSeen = new AtomicBoolean(false); + try { + // Same raised budget as ExternalProcessReferenceChainTest's own scenarios: a genuinely + // fresh, cold external JVM's reachable graph is far larger than the in-process tests', + // and this scenario adds ~100k filler objects on top. The pausetarget varies by caller. + // memory=64:l:1.0 - the explicit liveness-sample ratio is load-bearing, + // not decoration: without a ratio segment, the default is 10% + // (arguments.h's _live_samples_ratio) and LivenessTracker::track() + // drops 90% of tracked instances probabilistically. This scenario's + // leak cohort is only tens of 1MB chunks, so the default makes the + // (klass, tid) population a per-run Bernoulli lottery - observed live + // as the intermittent "tagLeakInstances ZERO" runs where zero chunks + // survived the filter (passing runs' stable "tagged=5" is exactly the + // ~10% tail of ~50 chunks). 1.0 makes the scenario's documented + // assumption ("every allocation is sampled, tracked, and taggable", + // LeakTagCorrelationScenario) actually true. + String startCommand = "start,memory=64:l:1.0,generations=true," + + "referencechains=true:hops=64:budget=200000:ttl=120000:framecap=2000000:" + + pauseTargetOption + + ",jfr,file=" + continuousJfrPath.toAbsolutePath(); + String packedCommand = startCommand + "|||" + scratchDumpPath.toAbsolutePath(); + + List jvmArgs = Collections.singletonList( + "-Dddprof_test.config=" + System.getProperty("ddprof_test.config")); + + AtomicReference resultLine = new AtomicReference<>(); + LaunchResult result = launch("leak-correlation", jvmArgs, packedCommand, + Collections.emptyMap(), + // The scenario's own round loop (up to 25 rounds plus a 5-attempt grace period, + // ~40-60s observed) on top of a cold external JVM's startup/classloading cost and + // this scenario's larger graph - same margin logic as the existing scenarios. + 150, + line -> { + if (line.startsWith(LeakTagCorrelationScenario.FOUND_MARKER) + || line.equals(LeakTagCorrelationScenario.NOT_FOUND_MARKER) + || line.startsWith(LeakTagCorrelationScenario.TAG_OUT_OF_POOL_MARKER) + || line.startsWith(LeakTagCorrelationScenario.NO_LIVE_OBJECT_MARKER) + || line.startsWith(LeakTagCorrelationScenario.TRANSIENT_CHAIN_MARKER) + || line.startsWith(LeakTagCorrelationScenario.NOISE_CHAIN_MARKER)) { + resultLine.set(line); + } + if (line.startsWith("[TEST::INFO]")) { + if (line.contains("collectLeakAccumulationCandidatesForRotation selected") + && line.contains("state=FRONTIER")) { + frontierRotationSeen.set(true); + } + testLogTail.addLast(line); + while (testLogTail.size() > 2000) { + testLogTail.removeFirst(); + } + } + return LineConsumerResult.CONTINUE; + }, + null); + + assertTrue(result.inTime, "Child process did not exit within the wait timeout"); + assertEquals(0, result.exitCode, "Child process exited with a non-zero code"); + assertNotNull(resultLine.get(), "Child process never printed a recognizable result marker" + + " on stdout" + diagnostics(testLogTail)); + String line = resultLine.get(); + if (requireFrontierRotation) { + assertTrue(frontierRotationSeen.get(), + "leak-accumulation rotation never selected a FRONTIER-state (un-expanded) " + + "holder under the pod-like tight window - the round-4 regression this " + + "variant exists to catch (tier dead: zero selections) would look exactly " + + "like this" + diagnostics(testLogTail)); + } + if (line.startsWith(LeakTagCorrelationScenario.FOUND_MARKER)) { + long targetTag = Long.parseLong( + line.substring(LeakTagCorrelationScenario.FOUND_MARKER.length()).trim()); + assertTrue(targetTag >= LEAK_TAG_BASE && targetTag < LEAK_TAG_BASE + LEAK_TAG_POOL_SIZE, + "Reported targetTag " + targetTag + " is outside the leak-tag pool range [" + + LEAK_TAG_BASE + ", " + (LEAK_TAG_BASE + LEAK_TAG_POOL_SIZE) + ")"); + return; + } + if (requireFrontierRotation && (line.equals(LeakTagCorrelationScenario.NOT_FOUND_MARKER) + || line.startsWith(LeakTagCorrelationScenario.TAG_OUT_OF_POOL_MARKER))) { + // The pod regime legitimately produces these without full correlation completing + // (see the tight variant's comment); the loose variant still fails on them. + return; + } + fail("Leak-tag correlation did not succeed: " + line + diagnostics(testLogTail)); + } finally { + Files.deleteIfExists(scratchDumpPath); + Files.deleteIfExists(continuousJfrPath); + } + } + + /** + * A filtered summary of the child's TEST_LOG stream for failure messages: the last few + * pass summaries (queue depths are the first thing to check on any crawl stall), the + * gotw batch-size trace (batch control regressions), and any leak-tag interception or + * correlation lines. Capped so the message stays readable. + */ + private static String diagnostics(Deque testLogTail) { + if (testLogTail.isEmpty()) { + return "\n(no TEST_LOG output captured - non-debug build or no reference-chain passes ran)"; + } + List runPassDone = new ArrayList<>(); + List gotw = new ArrayList<>(); + List leakTagLines = new ArrayList<>(); + for (String s : testLogTail) { + if (s.contains("runPass done:")) { + runPassDone.add(s); + } else if (s.contains("expandFrontier gotw")) { + gotw.add(s); + } else if (s.contains("intercepted") || s.contains("correlateAdmittedLeakTag") + || s.contains("recordDiscoveredInstance") || s.contains("auto-mark") + || s.contains("rotation_candidates") + || s.contains("collectLeakAccumulationCandidatesForRotation selected")) { + leakTagLines.add(s); + } + } + StringBuilder sb = new StringBuilder("\n--- child diagnostics (filtered TEST_LOG) ---"); + appendLast(sb, "runPass done", runPassDone, 5); + appendLast(sb, "gotw", gotw, 5); + appendLast(sb, "leak-tag activity", leakTagLines, 20); + return sb.toString(); + } + + private static void appendLast(StringBuilder sb, String label, List lines, int max) { + sb.append("\n").append(label).append(" (last ").append(max).append(" of ") + .append(lines.size()).append("):"); + int from = Math.max(0, lines.size() - max); + for (int i = from; i < lines.size(); i++) { + sb.append("\n ").append(lines.get(i)); + } + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/LeakTagCorrelationScenario.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/LeakTagCorrelationScenario.java new file mode 100644 index 0000000000..f83eb16c15 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/LeakTagCorrelationScenario.java @@ -0,0 +1,510 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.referencechains; + +import com.datadoghq.profiler.JavaProfiler; +import org.openjdk.jmc.common.IMCType; +import org.openjdk.jmc.common.item.IItem; +import org.openjdk.jmc.common.item.IItemCollection; +import org.openjdk.jmc.common.item.IItemIterable; +import org.openjdk.jmc.common.item.IMemberAccessor; +import org.openjdk.jmc.common.item.IType; +import org.openjdk.jmc.flightrecorder.JfrLoaderToolkit; + +import java.io.InputStream; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.ArrayList; +import java.util.List; +import java.util.ArrayList; +import java.util.List; + +/** + * The "leak-tag correlation" use case, run inside a genuinely separate child JVM by + * {@code ExternalLauncher}'s {@code leak-correlation} mode: a growing static-field + * collection leak, coexisting with (a) concurrent, ephemeral, stack-local allocations of + * another class and (b) a large live object graph, must end up reported as a + * {@code datadog.ReferenceChain} event whose {@code targetTag} is a leak-tag-pool tag, + * matching a {@code datadog.HeapLiveObject} event's {@code leakTag} for the same object - + * the chain-to-live-heap-sample correlation the whole leak-tag pool exists to provide. + * + *

    Each ingredient exists to reproduce, in miniature, a condition that only showed + * against real large deployments before this scenario existed (all observed live on a + * production-like pod, each escaping every earlier local test): + *

      + *
    • Scale ({@link #FILLER}): ~100k live filler objects grow the JVMTI tag map + * enough that {@code GetObjectsWithTags} has a real per-call cost floor, the BFS + * frontier/backlog has real depth, and per-pass deadlines actually bind - the regime + * where batch-size control, fair-share queue drain, and rotation starvation bugs live. + * Built (like the leak itself) BEFORE the profiler starts, per + * {@code LeakingCacheScenario}'s own comment: the one-shot root-seeded walk fires ~1s + * after start and must never catch the fixture half-built.
    • + *
    • Noise ({@link NoisePayload} on the noise thread): short-lived allocations + * held only by a live thread's stack frame during that same one-shot walk - admitted + * as {@code stack_local}-rooted depth-1 frontier entries, exactly the transient-root + * noise shape the discovered-chain gate exists to suppress. The scenario asserts no + * such chain is ever emitted; a regression in the gate emits one and fails this + * scenario with {@link #TRANSIENT_CHAIN_MARKER}.
    • + *
    • Correlation (the actual acceptance signal): a leaked {@code byte[]} + * chunk's chain must carry a {@code targetTag} in the leak-tag pool's range - + * mirrored here from {@code LEAK_TAG_BASE}/{@code LEAK_TAG_POOL_SIZE} + * (referenceChains.h) - and at least one {@code datadog.HeapLiveObject} event for a + * {@code byte[]} must carry the same value in {@code leakTag}.
    • + *
    + * + *

    Same separate-process rationale as {@code LeakingCacheScenario}'s own class comment + * (the one-shot root-seeded walk is a per-process resource), and same seed-before-start + * and debug-seam patterns as {@code StaticFieldGrowingCollectionScenario}. + */ +public final class LeakTagCorrelationScenario { + private LeakTagCorrelationScenario() {} + + /** + * The leak: a static collection appended to for the process's whole life - the + * canonical unmaintained-singleton retention shape, DELIBERATELY a growing + * {@code ArrayList} and not a fixed holder. An ArrayList's backing + * {@code elementData} array is REPLACED on growth, and an EXPANDED frontier node's + * children set is frozen at expansion time (expandFrontier()'s own comment), so + * discovering contents added after a resize is exactly the hard production case + * this scenario must prove works: the live holder's re-walk has to admit each new + * backing array, and only then can its new chunks be admitted and leak-tag + * intercepted. (An earlier fixed-array version of this fixture sidestepped that + * requirement and still failed - see the scenario history in the investigation + * notes - which is what exposed the rotation starvation this fixture now encodes.) + */ + static final List LEAK_SINK = new ArrayList<>(); + + /** + * Scale ingredient: a static tree of plain objects, unreachable from anywhere but this + * field - exists only to give the frontier/backlog and the JVMTI tag map real depth. + */ + static final Object[][] FILLER = new Object[1500][]; + + // Arbitrary, scenario-chosen klass ids - see LeakingCacheScenario.CACHED_PAYLOAD_TEST_KLASS_ID's + // own comment; distinct from the other scenarios' ids (987301/987302) since all may run in the + // same suite, each in its own separate child JVM. + private static final int LEAK_KLASS_ID = 987303; + private static final int NOISE_KLASS_ID = 987304; + + // See LeakingCacheScenario.SEED_EPOCHS_FOR_HYSTERESIS's own comment for the full derivation. + private static final int SEED_EPOCHS_FOR_HYSTERESIS = 15; + + private static final int SEED_CHUNKS = 10; + + /** + * Per-round trend-maintenance epoch counter (debug builds only) - same + * pattern and rationale as StaticFieldGrowingCollectionScenario's own: + * a one-shot seeded ramp ages out of LivenessTracker's hysteresis once + * enough real fold samples interleave (observed live: the byte[] candidate + * dropped right after the first leak-tag interceptions, stranding the + * correlated discoveries with no poll loop to build their chains), so + * each round pushes one fresh rising epoch to keep the candidates alive. + */ + private static int hysteresisEpoch = 0; + + private static void addChunk() { + LEAK_SINK.add(new byte[CHUNK_BYTES]); + } + + // Mirrored from referenceChains.h (LEAK_TAG_BASE/LEAK_TAG_POOL_SIZE): the pool range a + // correlated chain's targetTag must fall into, and the range a matching HeapLiveObject's + // leakTag must fall into. This is the contract under test, asserted against runtime + // event values - not constants compared to constants. + private static final long LEAK_TAG_BASE = 0x40000000L; + private static final long LEAK_TAG_POOL_SIZE = 256; + + /** + * Transient handoff channel for the noise representative - deliberately a local of + * {@code run()} (captured by the noise thread's lambda), NOT a static field: any static + * field holding a {@link NoisePayload} would make that one instance genuinely, + * durably static-retained, and the profiler would be RIGHT to emit a static-rooted chain + * for it - found the hard way on this scenario's very first run, which correctly reported + * {@code NoisePayload@depth0:static_field}. After the handoff the only JVM-heap reference + * to the representative is the noise thread's live frame, which is exactly the + * transient-rooted retention this class exists to exercise. + */ + private static final class NoiseHandoff { + volatile NoisePayload representative; + // The noise thread's own profiler tid - published alongside the payload so + // the seeding below can qualify exactly that thread for the noise klass + // (selectLeakCandidates()'s per-(klass, tid) gate matches the tid that + // actually allocated the noise instances, and that is this thread). + volatile int tid; + + void publish(NoisePayload payload, int tid) { + // tid FIRST (volatile ordering): a caller that sees `representative` set + // is then also guaranteed to see the tid published with it. + this.tid = tid; + representative = payload; + } + + NoisePayload take() { + NoisePayload payload = representative; + representative = null; // drop this path's only reference - retention stays transient + return payload; + } + } + + /** Printed to stdout, followed by the correlated targetTag, on full success. */ + public static final String FOUND_MARKER = "[correlation-found] "; + + /** Printed to stdout if no leaked byte[] chain was ever observed. */ + public static final String NOT_FOUND_MARKER = "[correlation-not-found]"; + + /** + * Printed to stdout (with the offending tag) if a leaked byte[] chain was emitted but its + * targetTag never landed in the leak-tag pool's range - the chain-to-live-heap + * correlation's first half failing. + */ + public static final String TAG_OUT_OF_POOL_MARKER = "[correlation-tag-out-of-pool] "; + + /** + * Printed to stdout (with the targetTag) if a correlated byte[] chain was emitted but + * no datadog.HeapLiveObject event for a byte[] ever carried the same leakTag - the + * correlation's second half failing (tagging and live-heap recording disagreeing on + * which instances are the tracked leak candidates). + */ + public static final String NO_LIVE_OBJECT_MARKER = "[correlation-no-live-object] "; + + /** + * Printed to stdout (with a summary of every violation) if any ReferenceChain event was + * emitted whose root kind is a transient root (stack/jni local) at depth <= 1 - the + * discovered-chain gate's exact suppression contract (depth==0 always, depth==1 when + * transient-rooted); anything matching it reaching the output is a gate regression. + */ + public static final String TRANSIENT_CHAIN_MARKER = "[correlation-transient-chain-emitted] "; + + /** + * Printed to stdout (with a summary of the offending chains) if the deliberately ephemeral + * {@link NoisePayload} - held only by the noise thread's stack frame - ever got a chain + * emitted, whether via the transient gate or any other path: nothing about it is durably + * retained, so it must never be reported as a leak. + */ + public static final String NOISE_CHAIN_MARKER = "[correlation-noise-chain-emitted] "; + + /** One dump's scan outcome - all fields per the latest dump only. */ + private static final class ScanResult { + ReferenceChainAssertions.ChainMatch payloadChain; + boolean payloadChainInPool; + int liveObjectMatchesForChainTag; + final List transientViolations = new ArrayList<>(); + final List noiseChains = new ArrayList<>(); + } + + public static void run(JavaProfiler profiler, String startCommand, Path scratchDumpPath) + throws Exception { + // Build the whole fixture before the profiler starts - see the class comment and + // LeakingCacheScenario's own seed-before-start rationale. + for (int i = 0; i < SEED_CHUNKS; i++) { + addChunk(); + } + for (int row = 0; row < FILLER.length; row++) { + Object[] column = new Object[64]; + for (int col = 0; col < column.length; col++) { + column[col] = new FillerNode(); + } + FILLER[row] = column; + } + NoiseHandoff handoff = new NoiseHandoff(); + Thread noiseThread = new Thread(() -> { + // Held ONLY by this live frame across the process's whole life, so the one-shot + // root-seeded walk (a few seconds away) admits them as stack_local-rooted entries. + NoisePayload[] held = new NoisePayload[64]; + for (int i = 0; i < held.length; i++) { + held[i] = new NoisePayload(); + } + // The tid must be captured on THIS thread: the per-(klass, tid) seeding below + // qualifies the allocating thread, and the noise instances are all allocated here. + handoff.publish(held[0], JavaProfiler.getTid()); + long slot = 0; + while (true) { + try { + Thread.sleep(50); + } catch (InterruptedException e) { + return; + } + // Churn one slot at a time: the population stays live-but-ephemeral (constant live + // count, nothing durably retained beyond the frame) while still allocating. + held[(int) (slot++ % held.length)] = new NoisePayload(); + } + }, "leak-correlation-noise"); + noiseThread.setDaemon(true); + noiseThread.start(); + + if (startCommand != null && !startCommand.isEmpty()) { + profiler.execute(startCommand); + } + + boolean debugBuild = "debug".equals(System.getProperty("ddprof_test.config")); + // Declared at method scope: the per-round and final-attempt maintenance + // seeding below (in their own debugBuild blocks) reuse these tids. + int leakTid = 0; + int noiseTid = 0; + if (debugBuild) { + // Representatives FIRST, seeds second: since the leak-tag pool redesign, candidate + // matching, leak-tag assignment (tagLeakInstances scanning the live-heap tracking + // table), and discovered-instance recording are all keyed on the REAL klass id - + // setKlassPopulationRepresentativeForTest0 resolves each representative's real id + // and aliases the synthetic id to it, and the hysteresis seeds below then land in + // the real entry (seeding first would leave them under the synthetic id, where they + // authorize nothing but hasLeakSignal's candidate-count check). + // The leak representative is durably reachable via LEAK_SINK for the rest of the + // process's life; the noise representative only via the noise thread's frame (the + // handoff's own reference is dropped by take()) - exactly what makes one a valid + // leak candidate and the other a gate test. + JavaProfiler.setKlassPopulationRepresentativeForTest0(LEAK_KLASS_ID, LEAK_SINK.get(0)); + JavaProfiler.setKlassPopulationRepresentativeForTest0(NOISE_KLASS_ID, handoff.take()); + // Per-(klass, tid) qualification seeds: selectLeakCandidates() now also + // requires a qualifying ALLOCATING THREAD, and tagLeakInstances() scopes leak + // tags to exactly those tids' instances - so the seeded qualification must + // name the REAL allocating tids (this thread for the leak chunks, the noise + // thread for the noise payloads) or no tracked instance would ever match + // the tagging scope. Same epoch-aligned ramp shape as the klass seeds + // (scaled to fit the per-tid u8 ring); the synthetic flag keeps these + // ramps immune to the real fold's absent-tid decay between rounds. + leakTid = JavaProfiler.getTid(); + noiseTid = handoff.tid; + for (int epoch = 1; epoch <= SEED_EPOCHS_FOR_HYSTERESIS; epoch++) { + JavaProfiler.seedKlassPopulationSample0(LEAK_KLASS_ID, epoch * 10, epoch); + JavaProfiler.seedKlassPopulationSample0(NOISE_KLASS_ID, epoch * 10, epoch); + JavaProfiler.seedTidTrendSample0(LEAK_KLASS_ID, leakTid, epoch * 3, epoch); + JavaProfiler.seedTidTrendSample0(NOISE_KLASS_ID, noiseTid, epoch * 3, epoch); + } + // Wait for the BFS thread to fully complete its first pass (static sweep AND frontier + // expansion) before any growth round - same load-bearing wait and rationale as + // StaticFieldGrowingCollectionScenario: the late chunks added afterward must be + // discovered via rotation re-walks, not by a first pass that has not even run yet. + int initialPasses = JavaProfiler.referenceChainPassesRunForTest0(); + for (int i = 0; i < 300; i++) { + if (JavaProfiler.referenceChainPassesRunForTest0() != initialPasses) { + break; + } + Thread.sleep(100); + } + JavaProfiler.pollReferenceChainTargets0(); + } else { + // No debug seams outside debug builds - give the real allocation sampler and the + // pass loop time to notice the seeded population on their own. + Thread.sleep(5000); + } + + // Each dump re-emits every resolved chain (drainPendingChainEvents()'s + // snapshot-and-keep contract), so a later dump's scan is a superset state - the + // latest scan with a payload chain observed is always the best evidence. + ScanResult best = null; + int totalRounds = 25; + // Slow growth per round is deliberate: each chunk clears the allocation + // sampler's floor on its own (see CHUNK_BYTES), so every allocation is + // sampled, tracked, and taggable - the per-round population growth is + // only the trend signal, not a sampling-volume requirement. + for (int round = 1; round <= totalRounds; round++) { + addChunk(); + if (debugBuild) { + hysteresisEpoch++; + JavaProfiler.seedKlassPopulationSample0( + LEAK_KLASS_ID, hysteresisEpoch * 10, hysteresisEpoch); + JavaProfiler.seedKlassPopulationSample0( + NOISE_KLASS_ID, hysteresisEpoch * 10, hysteresisEpoch); + JavaProfiler.seedTidTrendSample0( + LEAK_KLASS_ID, leakTid, hysteresisEpoch * 3, hysteresisEpoch); + JavaProfiler.seedTidTrendSample0( + NOISE_KLASS_ID, noiseTid, hysteresisEpoch * 3, hysteresisEpoch); + } + System.gc(); + profiler.dump(scratchDumpPath); + // Scan only every 3rd round: each scan's JMC parse allocates megabytes + // on this same JVM, and that churn races the leak chunks for space in + // LivenessTracker's tracking table (the age-priority tagging then + // selects old JVM-machinery survivors instead of the young chunks - + // observed live as tagged=5 stable but zero interceptions). Less + // scan churn gives the chunks time to accumulate surviving ages and + // win the tagging priority. Rounds are still dumped every time so + // the final dump always holds the freshest chains. + if (round % 3 == 0) { + best = keepBest(best, scan(scratchDumpPath)); + if (best != null && best.payloadChainInPool + && best.liveObjectMatchesForChainTag > 0) { + break; // full correlation achieved - no need to grow further + } + } + Thread.sleep(300); + } + + for (int attempt = 0; attempt < 5; attempt++) { + if (best != null && best.payloadChainInPool && best.liveObjectMatchesForChainTag > 0) { + break; + } + if (debugBuild) { + hysteresisEpoch++; + JavaProfiler.seedKlassPopulationSample0( + LEAK_KLASS_ID, hysteresisEpoch * 10, hysteresisEpoch); + JavaProfiler.seedKlassPopulationSample0( + NOISE_KLASS_ID, hysteresisEpoch * 10, hysteresisEpoch); + JavaProfiler.seedTidTrendSample0( + LEAK_KLASS_ID, leakTid, hysteresisEpoch * 3, hysteresisEpoch); + JavaProfiler.seedTidTrendSample0( + NOISE_KLASS_ID, noiseTid, hysteresisEpoch * 3, hysteresisEpoch); + } + Thread.sleep(1000); + System.gc(); + profiler.dump(scratchDumpPath); + best = keepBest(best, scan(scratchDumpPath)); + } + + if (best == null || best.payloadChain == null) { + System.out.println(NOT_FOUND_MARKER); + return; + } + if (!best.transientViolations.isEmpty()) { + System.out.println(TRANSIENT_CHAIN_MARKER + best.transientViolations); + return; + } + if (!best.noiseChains.isEmpty()) { + System.out.println(NOISE_CHAIN_MARKER + best.noiseChains); + return; + } + if (!best.payloadChainInPool) { + System.out.println(TAG_OUT_OF_POOL_MARKER + best.payloadChain.targetTag); + return; + } + if (best.liveObjectMatchesForChainTag == 0) { + System.out.println(NO_LIVE_OBJECT_MARKER + best.payloadChain.targetTag); + return; + } + System.out.println(FOUND_MARKER + best.payloadChain.targetTag); + } + + /** Latest scan wins; a scan without a payload chain never displaces one that had it. */ + private static ScanResult keepBest(ScanResult best, ScanResult scan) { + if (scan == null) { + return best; + } + if (scan.payloadChain != null) { + return scan; + } + return best != null ? best : scan; + } + + /** + * Scans one dump: the first leaked {@code byte[]} chain (if any), whether its targetTag is + * in the leak-tag pool, how many {@code datadog.HeapLiveObject} events for a + * leaked {@code byte[]} carry that same tag in {@code leakTag}, plus the two gate + * regressions (transient-rooted depth<=1 chains, any {@link NoisePayload} chain). + */ + private static ScanResult scan(Path scratchDumpPath) throws Exception { + if (!Files.exists(scratchDumpPath)) { + return null; + } + IItemCollection events; + try (InputStream in = Files.newInputStream(scratchDumpPath)) { + events = JfrLoaderToolkit.loadEvents(in); + } + ScanResult result = new ScanResult(); + + IItemCollection chains = events.apply( + org.openjdk.jmc.common.item.ItemFilters.type("datadog.ReferenceChain")); + for (IItemIterable iterable : chains) { + IType type = iterable.getType(); + IMemberAccessor chainAccessor = ReferenceChainAssertions.findAccessor(type, "chain"); + IMemberAccessor targetTagAccessor = + ReferenceChainAssertions.findAccessor(type, "targetTag"); + IMemberAccessor depthAccessor = ReferenceChainAssertions.findAccessor(type, "depth"); + IMemberAccessor rootKindAccessor = + ReferenceChainAssertions.findAccessor(type, "rootKind"); + for (IItem item : iterable) { + Object chainValue = chainAccessor.getMember(item); + if (!(chainValue instanceof Object[])) { + continue; + } + Object[] rawChain = (Object[]) chainValue; + String leaf = rawChain.length > 0 && rawChain[0] instanceof IMCType + ? ((IMCType) rawChain[0]).getFullName() : ""; + long targetTag = targetTagAccessor != null + ? ReferenceChainAssertions.numberValue(targetTagAccessor.getMember(item)) : -1; + int depth = depthAccessor != null + ? (int) ReferenceChainAssertions.numberValue(depthAccessor.getMember(item)) : -1; + Object rootKind = rootKindAccessor != null ? rootKindAccessor.getMember(item) : null; + String rootKindName = rootKind != null ? rootKind.toString() : ""; + if (rootKindName.startsWith("first_observed_via:") && depth <= 1) { + result.transientViolations.add(leaf + "@depth" + depth + ":" + rootKindName); + } + if (NoisePayload.class.getName().equals(leaf)) { + result.noiseChains.add(leaf + "@depth" + depth + ":" + rootKindName); + } + if ("byte[]".equals(leaf)) { + boolean inPool = + targetTag >= LEAK_TAG_BASE && targetTag < LEAK_TAG_BASE + LEAK_TAG_POOL_SIZE; + // Multiple byte[] chains exist in a single dump: every discovered + // instance of the candidate class gets one, and only the tracked-and- + // tagged instances carry a pool-range targetTag - the rest legitimately + // carry their frontier tag (the correlation is additive on top of the + // chain, not a property of every chain). Prefer the first in-pool + // (correlated) chain over any out-of-pool one seen earlier. + if ((inPool && !result.payloadChainInPool) || result.payloadChain == null) { + List chain = new ArrayList<>(rawChain.length); + for (Object element : rawChain) { + chain.add((IMCType) element); + } + result.payloadChain = new ReferenceChainAssertions.ChainMatch( + chain, targetTag, depth); + result.payloadChainInPool = inPool; + } + } + } + } + + if (result.payloadChainInPool) { + IItemCollection liveObjects = events.apply( + org.openjdk.jmc.common.item.ItemFilters.type("datadog.HeapLiveObject")); + for (IItemIterable iterable : liveObjects) { + IType type = iterable.getType(); + IMemberAccessor leakTagAccessor = + ReferenceChainAssertions.findAccessor(type, "leakTag"); + IMemberAccessor objectClassAccessor = + ReferenceChainAssertions.findAccessor(type, "objectClass"); + for (IItem item : iterable) { + long leakTag = leakTagAccessor != null + ? ReferenceChainAssertions.numberValue(leakTagAccessor.getMember(item)) : -1; + if (leakTag != result.payloadChain.targetTag) { + continue; + } + Object objectClass = objectClassAccessor != null + ? objectClassAccessor.getMember(item) : null; + if (objectClass instanceof IMCType + && "byte[]".equals(((IMCType) objectClass).getFullName())) { + result.liveObjectMatchesForChainTag++; + } + } + } + } + return result; + } + + /** + * Chunk size that clears ObjectSampler's real 256KiB sampling floor on its own - same + * rationale as StaticFieldGrowingCollectionScenario.CHUNK_BYTES. Load-bearing here: + * every chunk allocation must be allocation-sampled so it becomes a live-heap tracked + * instance (the only instances LivenessTracker's leak-tag pool tags, and the only ones + * that get datadog.HeapLiveObject events to correlate against). A small leaked object + * only gets sampled probabilistically, which made the correlation dependent on + * sampling luck - observed live as pass-standalone/fail-in-suite before this. + */ + private static final int CHUNK_BYTES = 1_000_000; + + /** + * The noise object: deliberately ephemeral - held only by the noise thread's live + * frame, never by any field, so nothing about it is durably retained. + */ + static final class NoisePayload { + long p0, p1, p2, p3, p4, p5, p6, p7, p8, p9, p10, p11, p12, p13, p14, p15; + long p16, p17, p18, p19, p20, p21, p22, p23, p24, p25, p26, p27, p28, p29, p30, p31; + } + + /** Scale-only object: no references, just weight - one frontier entry each. */ + static final class FillerNode { + long p0, p1, p2, p3, p4, p5, p6, p7; + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/LeakingCacheScenario.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/LeakingCacheScenario.java new file mode 100644 index 0000000000..ed47fe10f3 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/LeakingCacheScenario.java @@ -0,0 +1,353 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.referencechains; + +import com.datadoghq.profiler.JavaProfiler; +import org.openjdk.jmc.common.IMCType; +import org.openjdk.jmc.common.item.IItemCollection; +import org.openjdk.jmc.common.item.ItemFilters; +import org.openjdk.jmc.flightrecorder.JfrLoaderToolkit; + +import java.io.InputStream; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.HashMap; +import java.util.Map; + +/** + * The actual "leaking Java app" body run inside a genuinely separate child JVM by + * {@code ExternalLauncher}'s {@code leak-cache} mode, driven by + * {@code ExternalProcessReferenceChainTest}. Deliberately not a JUnit test itself - no + * {@code AbstractProfilerTest}/dynamic-attach lifecycle, no shared-JVM {@code ReferenceChainTracker} + * singleton with any other test - this class's whole point is to run alone in its own process, so + * the one-shot root-seeded BFS walk (see this class's own comment on why a previous attempt at + * folding this scenario into {@code ReferenceChainTrackingTest} as a second {@code @Test} method + * failed) always belongs to this scenario and only this scenario. + * + *

    Same realistic-leak shape as {@code ReferenceChainTrackingTest}'s own (now-removed) in-process + * attempt: an ever-growing, never-evicted {@code HashMap}-backed cache, held by a local variable. + * + *

    Why {@code cache} is seeded and populated before the profiler is even started + * (found the hard way, against a real separate-process run): {@code threadLoop()} + * (referenceChains.cpp) calls {@code OS::sleep(_effective_cadence_ns)} - 1 second, initially - + * *before* its very first {@code shouldRunPass()} check, and {@code shouldRunPass()} always + * returns {@code true} on that first-ever check ({@code !_search_started}). So the process's + * one-and-only root-seeded walk fires roughly one second after {@code startThread()}, or earlier + * if woken by a GC-finish signal - a race {@code ReferenceChainTrackingTest}'s own + * {@code shouldReportAbandonedSearchOnTinyFrontierCap()} comment already documents ("that wakeup + * can race the thread's own startup ... and be missed"). An initial in-process attempt at this + * exact scenario created {@code cache} and started allocating only *after* starting the profiler + * (mirroring {@code ReferenceChainTrackingTest}'s {@code gcRootHolder}, which happened to work by + * luck of scheduling in a warm, already-running JVM) - against a genuinely fresh, cold JVM this + * lost the race essentially every time: the one-shot walk caught {@code cache} still empty, + * permanently blocking discovery of everything added to it afterward (a frontier entry once + * marked {@code EXPANDED} is never revisited for new children - see {@code expandFrontier()}'s + * own comment). Seeding {@code cache} with real data *before* {@code profiler.execute()} is even + * called removes the race entirely: no matter how fast the walk fires, {@code cache} is never + * empty when it does. + */ +public final class LeakingCacheScenario { + private LeakingCacheScenario() {} + + // Arbitrary, scenario-chosen klass id - this process's own LivenessTracker population table + // treats it as an opaque key (see KlassPopulationEntry's own comment) and this process never + // runs any other reference-chains scenario, so no collision risk with e.g. + // ReferenceChainTrackingTest's/ReferenceChainTestSeamsTest's own klass ids in their own, + // separate JVMs. + private static final int CACHED_PAYLOAD_TEST_KLASS_ID = 987301; + + // See ReferenceChainTrackingTest.SEED_EPOCHS_FOR_HYSTERESIS's own comment for the full + // derivation: LivenessTracker's hysteresis gate (livenessTracker.h's + // KLASS_POPULATION_MIN_FILL_FOR_TREND=10 / LEAK_TREND_HYSTERESIS_BASE=5) only starts + // incrementing consecutive_positive once ring_fill has reached 10, so a batch of exactly 10 + // seedKlassPopulationSample0() calls caps consecutive_positive at 1 - far short of what + // selectLeakCandidates() (livenessTracker.cpp) requires before ReferenceChainTracker:: + // hasLeakSignal() reports a candidate, which is what this scenario's own generations=true + // startCommand needs to authorize the BFS thread's first real pass at all. + private static final int SEED_EPOCHS_FOR_HYSTERESIS = 15; + + /** + * Durable holder for the cache fixture. The discovered-chain gate suppresses + * chains shallower than the first real holder hop that are rooted at a TRANSIENT + * root (stack/JNI local) - a frame-held "leak" is by definition not a leak, so the + * cache must be retained through a static field (the real singleton-collection + * shape) rather than only {@code run()}'s own local, which is exactly the + * transient-rooted shape the gate suppresses. {@code run()}'s local {@code cache} + * variable below aliases this static; the durable root is what makes the emitted + * chain durable-rooted. + */ + private static final Map CACHE = new HashMap<>(); + + /** Printed to stdout, followed by the matched leaf class's name, on success. */ + public static final String FOUND_MARKER = "[chain-found] "; + + /** Printed to stdout (with no class name suffix) if no match was ever observed. */ + public static final String NOT_FOUND_MARKER = "[chain-not-found]"; + + /** + * Printed to stdout (with the offending class name appended) if a match was found but its + * chain did not pass through {@code java.util.HashMap}'s own internal storage - see + * {@code ExternalProcessReferenceChainTest} for why this is asserted in the parent process + * rather than here (keeping this scenario's own stdout contract to "found/not-found" and + * leaving richer chain-shape assertions to the JUnit side, the same separation of concerns + * {@code ReferenceChainJfrParserTest} already uses for the lower-level JFR-format proof). + */ + public static final String NO_HASHMAP_INTERNALS_MARKER = "[chain-found-no-hashmap-internals] "; + + /** + * Printed to stdout (with an "N/M" re-emitted-dump count appended) if the chain was + * successfully reconstructed but did not re-emit into one or more subsequent, + * independently written JFR dumps - i.e. the "snapshot-and-keep" contract of + * {@code ReferenceChainTracker::drainPendingChainEvents()} (referenceChains.cpp), which + * {@code Profiler::dump()} invokes on every dump without clearing the resolved-chain cache, + * failed to hold across dumps. Asserted by {@code ExternalProcessReferenceChainTest}. + */ + public static final String CHAIN_NOT_PERSISTED_MARKER = "[chain-not-persisted] "; + + /** + * Seeds {@code cache} with an initial batch, *then* starts the profiler (see this class's own + * comment for why that order is load-bearing), then runs the same population-growth loop + * {@code ReferenceChainTrackingTest} pioneered - dumping to {@code scratchDumpPath} after every + * round (and a short grace period beyond the last round) until a {@code datadog.ReferenceChain} + * event whose {@code chain[0]} is {@link CachedPayload} appears - then prints exactly one + * result line to stdout. Called from {@code ExternalLauncher} (a different package - hence + * {@code public}), which owns the generic child-process bootstrap (JVMTI/library init, process + * exit) this scenario deliberately stays agnostic of; starting the profiler itself happens + * here, not in {@code ExternalLauncher}, specifically so {@code cache} can be seeded first. + */ + public static void run(JavaProfiler profiler, String startCommand, Path scratchDumpPath) throws Exception { + Map cache = CACHE; + seedInitialBatch(cache, 300); + System.out.println("[debug] startCommand=" + startCommand); + System.out.println("[debug] scratchDumpPath=" + scratchDumpPath); + if (startCommand != null && !startCommand.isEmpty()) { + profiler.execute(startCommand); + } + System.out.println("[debug] profiler.execute() returned"); + + boolean debugBuild = "debug".equals(System.getProperty("ddprof_test.config")); + boolean seededTestKlassTrend = false; + ReferenceChainAssertions.ChainMatch match = null; + int totalRounds = 25; + for (int round = 1; round <= totalRounds && match == null; round++) { + int newEntries = round * 300; + int roundNumber = round; + Thread allocator = new Thread(() -> { + for (int i = 0; i < newEntries; i++) { + String key = "leak-" + roundNumber + "-" + i; + cache.put(key, new CachedPayload(key)); + } + }); + allocator.start(); + allocator.join(); + System.gc(); + profiler.dump(scratchDumpPath); + match = findMatch(scratchDumpPath); + + if (match == null && debugBuild) { + // Same debug-only, seeded-representative short-circuit as ReferenceChainTrackingTest's + // own shouldReconstructReferrerChainThroughUnboundedCacheLeak() (see that method's own + // comment, and SEED_EPOCHS_FOR_HYSTERESIS's own comment above): decouples this scenario's + // assertion from whether the real, probabilistic allocation-sampling-driven slope + // detection happens to notice CachedPayload's own population trend within this scenario's + // fixed round/timeout budget - by clearing hasLeakSignal()'s own hysteresis gate outright, + // which is what actually authorizes the BFS thread's first real pass, rather than assuming + // some earlier pass has already tagged anything. "seed-0" (from seedInitialBatch()) is + // reachable via cache for the scenario's entire lifetime, so it is a safe, always-valid + // representative regardless of which round this fires on. + // Representative BEFORE seeding: setKlassPopulationRepresentativeForTest0 + // resolves the representative's real klass id and aliases the synthetic id + // to it, so the hysteresis seeds below land in the real population entry + // (candidate matching is real-id keyed since the leak-tag pool redesign). + JavaProfiler.setKlassPopulationRepresentativeForTest0(CACHED_PAYLOAD_TEST_KLASS_ID, cache.get("seed-0")); + if (!seededTestKlassTrend) { + // Per-(klass, tid) qualification seeds: the qualifying tid is THIS + // thread's - the creator of "seed-0", the always-reachable representative + // the seeded ramp and the match below hang off. (The per-round allocator + // threads are one-shot, so no single one of their tids carries a sustained + // per-tid signal - exactly the one-cohort-per-thread shape the retained- + // count bar / synthetic seeds exist to keep qualifying.) + int leakTid = JavaProfiler.getTid(); + for (int epoch = 1; epoch <= SEED_EPOCHS_FOR_HYSTERESIS; epoch++) { + JavaProfiler.seedKlassPopulationSample0(CACHED_PAYLOAD_TEST_KLASS_ID, epoch * 10, epoch); + JavaProfiler.seedTidTrendSample0(CACHED_PAYLOAD_TEST_KLASS_ID, leakTid, epoch * 3, epoch); + } + seededTestKlassTrend = true; + } + JavaProfiler.pollReferenceChainTargets0(); + profiler.dump(scratchDumpPath); + match = findMatch(scratchDumpPath); + } + + if (match == null) { + Thread.sleep(300); + profiler.dump(scratchDumpPath); + match = findMatch(scratchDumpPath); + } + if (round == 1 || round == 5 || round == 10 || round == 25) { + System.out.println("[debug] round=" + round + " cache.size()=" + cache.size()); + debugDumpAllChainLeaves(scratchDumpPath); + } + } + + for (int attempt = 0; match == null && attempt < 5; attempt++) { + Thread.sleep(1000); + profiler.dump(scratchDumpPath); + match = findMatch(scratchDumpPath); + } + + if (match == null) { + debugDumpAllChainLeaves(scratchDumpPath); + debugDumpCounters(profiler); + System.out.println(NOT_FOUND_MARKER); + return; + } + boolean sawHashMapInternals = false; + for (IMCType type : match.chain) { + if (type.getFullName().startsWith("java.util.HashMap")) { + sawHashMapInternals = true; + break; + } + } + if (!sawHashMapInternals) { + System.out.println(NO_HASHMAP_INTERNALS_MARKER + match.chain); + return; + } + + // Across-dumps persistence: a resolved chain must re-emit into EVERY subsequent dump the + // sample survives into, not only the one dump that first observed it. Profiler::dump() + // re-snapshots the whole resolved-chain cache without clearing it on every call + // (drainPendingChainEvents(), referenceChains.cpp - "snapshot-and-keep, not a drain"), so a + // still-live sample's chain re-emits into every chunk. `cache` still holds every + // CachedPayload here, so the sample stays live across these extra dumps; each dump targets a + // brand-new file, so a chain found there can only be present because it was re-emitted into + // that dump, not left over from the round-loop's own scratchDumpPath. + int persistChecks = 3; + int persisted = 0; + for (int i = 1; i <= persistChecks; i++) { + Path reDumpPath = Files.createTempFile("referencechains-persist-" + i + "-", ".jfr"); + Files.deleteIfExists(reDumpPath); + try { + profiler.dump(reDumpPath); + if (findMatch(reDumpPath) != null) { + persisted++; + } else { + System.out.println("[debug] across-dumps: CachedPayload chain missing from fresh dump #" + i); + } + } finally { + Files.deleteIfExists(reDumpPath); + } + } + if (persisted != persistChecks) { + System.out.println(CHAIN_NOT_PERSISTED_MARKER + persisted + "/" + persistChecks); + return; + } + System.out.println("[debug] across-dumps: CachedPayload chain re-emitted in all " + + persistChecks + " fresh dumps"); + System.out.println(FOUND_MARKER + match.chain.get(0).getFullName()); + } + + /** + * Populates {@code cache} with {@code count} entries under a key namespace ("seed-") disjoint + * from the round loop's own ("leak-") - called before the profiler (and therefore + * {@code ReferenceChainTracker}'s BFS thread) even starts, so the process's one-shot + * root-seeded walk can never catch {@code cache} empty (see this class's own comment on why + * that race is otherwise real). + */ + private static void seedInitialBatch(Map cache, int count) { + for (int i = 0; i < count; i++) { + String key = "seed-" + i; + cache.put(key, new CachedPayload(key)); + } + } + + /** Temporary diagnostic: print all native debug counters. */ + private static void debugDumpCounters(JavaProfiler profiler) { + try { + Map counters = profiler.getDebugCounters(); + System.out.println("[debug] counters (" + counters.size() + " total):"); + counters.entrySet().stream() + .filter(e -> e.getValue() != 0) + .sorted(Map.Entry.comparingByKey()) + .forEach(e -> System.out.println("[debug] " + e.getKey() + " = " + e.getValue())); + } catch (Exception e) { + System.out.println("[debug] exception while dumping counters: " + e); + } + } + + /** Temporary diagnostic: print every datadog.ReferenceChain event's chain[0] class, if any. */ + private static void debugDumpAllChainLeaves(Path scratchDumpPath) { + try { + if (!Files.exists(scratchDumpPath)) { + System.out.println("[debug] scratch dump path does not exist: " + scratchDumpPath); + return; + } + IItemCollection events; + try (InputStream in = Files.newInputStream(scratchDumpPath)) { + events = JfrLoaderToolkit.loadEvents(in); + } + IItemCollection chains = events.apply(ItemFilters.type("datadog.ReferenceChain")); + long total = chains.stream().mapToLong(it -> it.getItemCount()).sum(); + System.out.println("[debug] datadog.ReferenceChain total events: " + total); + IItemCollection liveObjects = events.apply(ItemFilters.type("datadog.HeapLiveObject")); + long totalLive = liveObjects.stream().mapToLong(it -> it.getItemCount()).sum(); + System.out.println("[debug] datadog.HeapLiveObject total events: " + totalLive); + IItemCollection abandoned = events.apply(ItemFilters.type("datadog.ReferenceChainAbandoned")); + long totalAbandoned = abandoned.stream().mapToLong(it -> it.getItemCount()).sum(); + System.out.println("[debug] datadog.ReferenceChainAbandoned total events: " + totalAbandoned); + for (org.openjdk.jmc.common.item.IItemIterable iterable : chains) { + org.openjdk.jmc.common.item.IType type = iterable.getType(); + org.openjdk.jmc.common.item.IMemberAccessor chainAccessor = + ReferenceChainAssertions.findAccessor(type, "chain"); + if (chainAccessor == null) { + System.out.println("[debug] no chain accessor found"); + continue; + } + for (org.openjdk.jmc.common.item.IItem item : iterable) { + Object chainValue = chainAccessor.getMember(item); + if (chainValue instanceof Object[]) { + Object[] raw = (Object[]) chainValue; + String leaf = raw.length > 0 && raw[0] instanceof IMCType + ? ((IMCType) raw[0]).getFullName() : ""; + System.out.println("[debug] chain leaf: " + leaf); + } + } + } + } catch (Exception e) { + System.out.println("[debug] exception while dumping chain leaves: " + e); + } + } + + private static ReferenceChainAssertions.ChainMatch findMatch(Path scratchDumpPath) throws Exception { + if (!Files.exists(scratchDumpPath)) { + return null; + } + IItemCollection events; + try (InputStream in = Files.newInputStream(scratchDumpPath)) { + events = JfrLoaderToolkit.loadEvents(in); + } + return ReferenceChainAssertions.findMatchForClass( + events.apply(ItemFilters.type("datadog.ReferenceChain")), CachedPayload.class); + } + + /** + * Realistic-leak fixture - identical shape to {@code ReferenceChainTrackingTest}'s own + * (now-removed) in-process attempt. The 32 {@code long} fields exist purely so a given number + * of megabytes needs far fewer entries to reach {@code ObjectSampler}'s real 256KiB sampling + * floor - deliberately plain fields, not a nested array (a same-instance companion array would + * be a *second*, separate heap allocation the size-weighted allocation sampler would compete + * for, taking sampling attention away from {@code CachedPayload} itself). + */ + static final class CachedPayload { + final String key; + long p0, p1, p2, p3, p4, p5, p6, p7, p8, p9, p10, p11, p12, p13, p14, p15; + long p16, p17, p18, p19, p20, p21, p22, p23, p24, p25, p26, p27, p28, p29, p30, p31; + + CachedPayload(String key) { + this.key = key; + } + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainAssertions.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainAssertions.java new file mode 100644 index 0000000000..6bab444524 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainAssertions.java @@ -0,0 +1,216 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.referencechains; + +import com.datadoghq.profiler.JfrEvent; +import com.datadoghq.profiler.JfrEvents; +import org.openjdk.jmc.common.IMCType; +import org.openjdk.jmc.common.item.IAccessorKey; +import org.openjdk.jmc.common.item.IItem; +import org.openjdk.jmc.common.item.IItemCollection; +import org.openjdk.jmc.common.item.IItemIterable; +import org.openjdk.jmc.common.item.IMemberAccessor; +import org.openjdk.jmc.common.item.IType; +import org.openjdk.jmc.common.unit.IQuantity; + +import java.util.ArrayList; +import java.util.List; +import java.util.Map; + +/** + * Shared {@code datadog.ReferenceChain} JFR-parsing helpers, extracted out of + * {@code ReferenceChainTrackingTest} so both that in-process JUnit test and + * {@link LeakingCacheScenario} (run inside a genuinely separate child JVM by + * {@code ExternalProcessReferenceChainTest}) can reuse the exact same JMC-accessor logic + * rather than maintaining two copies. + * + *

    {@link #findMatchForClass(JfrEvents, Class)} is a separate, jafar-backed counterpart to + * {@link #findMatchForClass(IItemCollection, Class)} for {@code ReferenceChainTrackingTest}, which + * loads recordings via {@code AbstractProfilerTest}'s {@code JfrEvents}-returning helpers + * (jb/jfr-lightweight-query-api). {@link LeakingCacheScenario} and {@code ReferenceChainJfrParserTest} + * load recordings directly via JMC's {@code JfrLoaderToolkit} and keep using the + * {@code IItemCollection} overload unchanged. + */ +public final class ReferenceChainAssertions { + private ReferenceChainAssertions() {} + + /** Result of {@link #findMatchForClass(IItemCollection, Class)}: one resolved chain event's fields. */ + public static final class ChainMatch { + public final List chain; + public final long targetTag; + public final int depth; + + ChainMatch(List chain, long targetTag, int depth) { + this.chain = chain; + this.targetTag = targetTag; + this.depth = depth; + } + } + + /** Result of {@link #findMatchForClass(JfrEvents, Class)}: one resolved chain event's fields. */ + public static final class JfrChainMatch { + public final List chain; + public final long targetTag; + public final int depth; + + JfrChainMatch(List chain, long targetTag, int depth) { + this.chain = chain; + this.targetTag = targetTag; + this.depth = depth; + } + } + + /** + * Scans {@code events} for a {@code datadog.ReferenceChain} item whose {@code chain[0]} is + * {@code targetClass} specifically, ignoring any events for other klasses this same + * leak-candidate mechanism may have legitimately flagged (e.g. "[B"/byte[] - see each caller's + * own comment). Returns {@code null} if {@code events} is empty or none match. + */ + public static ChainMatch findMatchForClass(IItemCollection events, Class targetClass) { + if (events == null || !events.hasItems()) { + return null; + } + for (IItemIterable iterable : events) { + IType type = iterable.getType(); + IMemberAccessor chainAccessor = findAccessor(type, "chain"); + IMemberAccessor targetTagAccessor = findAccessor(type, "targetTag"); + IMemberAccessor depthAccessor = findAccessor(type, "depth"); + if (chainAccessor == null) { + throw new IllegalStateException("No accessor for 'chain' field on datadog.ReferenceChain"); + } + + String targetName = jmcStyleName(targetClass); + for (IItem item : iterable) { + Object chainValue = chainAccessor.getMember(item); + if (!(chainValue instanceof Object[])) { + throw new IllegalStateException( + "'chain' field resolved to " + chainValue + ", expected an array"); + } + Object[] rawChain = (Object[]) chainValue; + if (rawChain.length == 0 || !(rawChain[0] instanceof IMCType) + || !targetName.equals(((IMCType) rawChain[0]).getFullName())) { + continue; + } + List chain = new ArrayList<>(rawChain.length); + for (Object element : rawChain) { + chain.add((IMCType) element); + } + long targetTag = targetTagAccessor != null ? numberValue(targetTagAccessor.getMember(item)) : -1; + int depth = depthAccessor != null ? (int) numberValue(depthAccessor.getMember(item)) : -1; + return new ChainMatch(chain, targetTag, depth); + } + } + return null; + } + + /** + * jafar/{@code JfrEvents}-backed counterpart to {@link #findMatchForClass(IItemCollection, Class)} - + * see this class's own header comment for why these are two separate overloads rather than one. + */ + public static JfrChainMatch findMatchForClass(JfrEvents events, Class targetClass) { + if (events == null || !events.hasItems()) { + return null; + } + String targetName = jmcStyleName(targetClass); + for (JfrEvent item : events) { + Object chainValue = item.get("chain"); + if (!(chainValue instanceof Object[])) { + throw new IllegalStateException( + "'chain' field resolved to " + chainValue + ", expected an array"); + } + Object[] rawChain = (Object[]) chainValue; + if (rawChain.length == 0 || !targetName.equals(classFullName(rawChain[0]))) { + continue; + } + List chain = new ArrayList<>(rawChain.length); + for (Object element : rawChain) { + chain.add(classFullName(element)); + } + long targetTag = item.getLong("targetTag", -1); + int depth = (int) item.getLong("depth", -1); + return new JfrChainMatch(chain, targetTag, depth); + } + return null; + } + + /** + * {@code targetClass}'s name in the same format {@code IMCType.getFullName()} uses for array + * types - found the hard way: {@code Class.getName()} renders {@code byte[].class} as {@code + * "[B"} (JVM internal signature notation), but JMC's chunk parser renders the same array class's + * {@code IMCType.getFullName()} as {@code "byte[]"} (Java source notation), so comparing {@link + * Class#getName()} directly against {@code getFullName()} - as both {@code findMatchForClass} + * overloads above used to - can never match an array-typed leaf class, even when the correct + * event is genuinely present in the recording. Only affects array {@code targetClass} values + * (e.g. {@code byte[].class}); a non-array class's {@code getName()} already matches {@code + * getFullName()} as-is. + */ + private static String jmcStyleName(Class targetClass) { + int dimensions = 0; + Class component = targetClass; + while (component.isArray()) { + dimensions++; + component = component.getComponentType(); + } + if (dimensions == 0) { + return targetClass.getName(); + } + StringBuilder name = new StringBuilder(component.getName()); + for (int i = 0; i < dimensions; i++) { + name.append("[]"); + } + return name.toString(); + } + + /** + * The full name (e.g. {@code java.lang.String}) of a resolved {@code chain[]} array element - + * mirrors {@code JfrEvent.getClassName(String)}'s own class-reference-map unwrapping, applied + * to an array element rather than a named field. + */ + @SuppressWarnings("unchecked") + private static String classFullName(Object element) { + if (!(element instanceof Map)) { + throw new IllegalStateException( + "chain[] element resolved to " + element + ", expected a class reference map"); + } + Object name = ((Map) element).get("name"); + String s; + if (name instanceof Map) { + Object v = ((Map) name).get("string"); + s = v != null ? v.toString() : null; + } else { + s = name != null ? name.toString() : null; + } + return s != null ? s.replace('/', '.') : null; + } + + /** + * Looks up a field's accessor by identifier rather than via {@code Attribute.attr(...)}: JMC's + * v1 chunk parser (internal.parser.v1.ValueReaders.ArrayReader#getContentType()) registers + * {@code UnitLookup.UNKNOWN} as the declared content type for every array field regardless of + * what its element reader resolves to, so an {@code F_ARRAY} field like {@code chain} + * (T_CLASS, F_CPOOL|F_ARRAY, jfrMetadata.cpp) cannot be bound via a compile-time-typed + * {@code Attribute}. Mirrors {@code ReferenceChainJfrParserTest}'s identical lookup, which + * already proves this resolves {@code chain}'s array elements to real {@link IMCType}s. + */ + public static IMemberAccessor findAccessor(IType type, String identifier) { + for (IAccessorKey key : type.getAccessorKeys().keySet()) { + if (identifier.equals(key.getIdentifier())) { + return type.getAccessor(key); + } + } + return null; + } + + public static long numberValue(Object value) { + if (value instanceof Number) { + return ((Number) value).longValue(); + } + if (value instanceof IQuantity) { + return ((IQuantity) value).longValue(); + } + return -1; + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainJfrParserTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainJfrParserTest.java new file mode 100644 index 0000000000..d33072a588 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainJfrParserTest.java @@ -0,0 +1,192 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.referencechains; + +import org.junit.jupiter.api.Test; +import org.openjdk.jmc.common.IMCType; +import org.openjdk.jmc.common.item.IAccessorKey; +import org.openjdk.jmc.common.item.IItem; +import org.openjdk.jmc.common.item.IItemCollection; +import org.openjdk.jmc.common.item.IItemIterable; +import org.openjdk.jmc.common.item.IMemberAccessor; +import org.openjdk.jmc.common.item.IType; +import org.openjdk.jmc.common.item.ItemFilters; +import org.openjdk.jmc.flightrecorder.CouldNotLoadRecordingException; +import org.openjdk.jmc.flightrecorder.JfrLoaderToolkit; + +import java.io.IOException; +import java.io.InputStream; +import java.nio.file.Files; +import java.nio.file.Path; +import java.nio.file.Paths; +import java.util.ArrayList; +import java.util.List; +import java.util.Map; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.junit.jupiter.api.Assertions.fail; + +/** + * PROF-15341 design doc, Open Question: does JMC's parser actually resolve the {@code + * datadog.ReferenceChain} event's {@code chain} field - declared in jfrMetadata.cpp as {@code + * field("chain", T_CLASS, ..., F_CPOOL | F_ARRAY)}, i.e. an array of scalar + * constant-pool-index {@code T_CLASS} values - the same way it resolves a plain scalar + * F_CPOOL field (e.g. {@code objectClass}, already exercised by {@code AbstractProfilerTest}) + * or a plain F_ARRAY-of-composite-struct field (e.g. {@code jdk.StackTrace#frames})? Neither + * already-exercised shape proves this exact combination. + * + *

    This test loads the standalone .jfr file produced by the companion gtest ({@code + * ddprof-lib/src/test/cpp/referenceChainJfrRoundtrip_ut.cpp}) directly via {@link + * JfrLoaderToolkit} - no live profiler attach, no {@code AbstractProfilerTest} lifecycle - and + * asserts the {@code chain} field resolves to the exact class names that gtest seeded, in the + * exact leaf-to-root order {@code ReferenceChainTracker::buildChainEvent()} (referenceChains.h) + * produces. + * + *

    Why a generic accessor, not {@code Attribute.attr(...)}: JMC's v1 chunk parser + * (internal.parser.v1.ValueReaders.ArrayReader#getContentType()) registers {@code + * UnitLookup.UNKNOWN} as the declared content type for every array field, regardless of + * what its element reader resolves to - this is generic to all array fields, not specific to the + * cpool case. A field registered as UNKNOWN cannot be bound via {@code + * Attribute.attr(id, name, desc, CLASS).getAccessor(type)} (that requires the field's registered + * content type to match). This test instead looks the field up by identifier via {@code + * IType#getAccessorKeys()} and calls {@code IType#getAccessor(IAccessorKey)} directly - the same + * lower-level lookup JMC's own UI uses for fields it has no compile-time-known attribute for. + * This is not a workaround for a missing capability; it is the correct API for an unregistered + * custom field, and it still calls through the very code + * ({@code ArrayReader.read()}/{@code resolve()} delegating per-element to the field's element + * reader, which for {@code chain} is a {@code PoolReader}) that resolves each array element's + * constant-pool index to its class - the actual thing this test exists to prove. + */ +public class ReferenceChainJfrParserTest { + + private static final String EVENT_TYPE = "datadog.ReferenceChain"; + + /** + * Same path {@code chainRoundtripJfrPath()} in the companion gtest resolves to: the OS temp + * dir (TMPDIR, falling back to /tmp), agreed by both sides rather than by a shared build + * directory - the gtest (ddprof-lib) and this test (ddprof-test) are different Gradle + * modules/tasks with no other filesystem contract between them. + */ + private static Path roundtripJfrPath() { + String tmp = System.getenv("TMPDIR"); + String dir = (tmp != null && !tmp.isEmpty()) ? tmp : "/tmp"; + return Paths.get(dir, "datadog_reference_chain_roundtrip.jfr"); + } + + @Test + public void chainFieldResolvesToSeededClassNamesInLeafToRootOrder() + throws IOException, CouldNotLoadRecordingException { + Path jfrPath = roundtripJfrPath(); + assertTrue(Files.exists(jfrPath), + "Expected " + jfrPath + " to exist - run " + + ":ddprof-lib:gtestDebug_referenceChainJfrRoundtrip_ut first " + + "(referenceChainJfrRoundtrip_ut.cpp produces this file)."); + + IItemCollection events; + try (InputStream in = Files.newInputStream(jfrPath)) { + events = JfrLoaderToolkit.loadEvents(in); + } + IItemCollection chainEvents = events.apply(ItemFilters.type(EVENT_TYPE)); + assertTrue(chainEvents.hasItems(), "Expected at least one " + EVENT_TYPE + " event"); + + List resolvedChain = null; + long targetTag = -1; + int depth = -1; + for (IItemIterable iterable : chainEvents) { + IType type = iterable.getType(); + IMemberAccessor chainAccessor = findAccessor(type, "chain"); + IMemberAccessor edgesAccessor = findAccessor(type, "edges"); + IMemberAccessor targetTagAccessor = findAccessor(type, "targetTag"); + IMemberAccessor depthAccessor = findAccessor(type, "depth"); + assertNotNull(chainAccessor, "No accessor for 'chain' field on " + EVENT_TYPE); + assertNotNull(edgesAccessor, "No accessor for 'edges' field on " + EVENT_TYPE + + " - the new retention-edge label array (T_STRING|F_ARRAY) did not make it " + + "into the recorded metadata"); + + for (IItem item : iterable) { + Object chainValue = chainAccessor.getMember(item); + assertNotNull(chainValue, "'chain' field resolved to null"); + assertTrue(chainValue instanceof Object[], + "'chain' field resolved to " + chainValue.getClass() + ", expected an array"); + + Object[] rawChain = (Object[]) chainValue; + List chain = new ArrayList<>(rawChain.length); + for (Object element : rawChain) { + assertNotNull(element, + "chain[] element resolved to null - the constant-pool reference for this " + + "T_CLASS array entry did not resolve to a class"); + assertTrue(element instanceof IMCType, + "chain[] element resolved to " + element.getClass() + + ", expected " + IMCType.class + " (a resolved class, not a raw cpool index)"); + chain.add((IMCType) element); + } + resolvedChain = chain; + // Retention-edge labels: one per chain hop, aligned with it (the + // gtest seeds kind labels because its JNIEnv is null - the label + // DECODE is skipped; the Java side validates the ARRAY FIELD parses, + // not the decode itself, which referenceChains_ut covers natively). + Object edgesValue = edgesAccessor.getMember(item); + assertNotNull(edgesValue, "'edges' field resolved to null"); + assertTrue(edgesValue instanceof Object[], + "'edges' field resolved to " + edgesValue.getClass() + + ", expected an array"); + Object[] rawEdges = (Object[]) edgesValue; + assertEquals(3, rawEdges.length, "edges must align with the chain, one label per hop"); + for (Object e : rawEdges) { + assertTrue(e instanceof String, + "edges[] element resolved to " + (e == null ? "null" : e.getClass()) + + ", expected a String"); + } + assertEquals("field", rawEdges[0]); + assertEquals("element", rawEdges[1]); + assertEquals("jni_global", rawEdges[2]); + if (targetTagAccessor != null) { + Object v = targetTagAccessor.getMember(item); + if (v instanceof Number) { + targetTag = ((Number) v).longValue(); + } else if (v instanceof org.openjdk.jmc.common.unit.IQuantity) { + targetTag = ((org.openjdk.jmc.common.unit.IQuantity) v).longValue(); + } + } + if (depthAccessor != null) { + Object v = depthAccessor.getMember(item); + if (v instanceof Number) { + depth = ((Number) v).intValue(); + } else if (v instanceof org.openjdk.jmc.common.unit.IQuantity) { + depth = (int) ((org.openjdk.jmc.common.unit.IQuantity) v).longValue(); + } + } + break; // referenceChainJfrRoundtrip_ut.cpp writes exactly one event + } + if (resolvedChain != null) { + break; + } + } + + assertNotNull(resolvedChain, "Never iterated a " + EVENT_TYPE + " item"); + assertEquals(3, resolvedChain.size(), "Expected leaf/middle/root - 3 entries"); + // Leaf-to-root order, matching ReferenceChainTracker::buildChainEvent()'s + // FrontierTable::reconstructChain() contract and the gtest's insert() calls + // (tag=3 leaf -> tag=2 middle -> tag=1 root). + assertEquals("com.test.ChainLeaf", resolvedChain.get(0).getFullName()); + assertEquals("com.test.ChainMiddle", resolvedChain.get(1).getFullName()); + assertEquals("com.test.ChainRoot", resolvedChain.get(2).getFullName()); + assertEquals(3L, targetTag, "targetTag should be the leaf's frontier tag (3)"); + assertEquals(2, depth, "depth should be the leaf's own depth (2 hops from the root)"); + } + + private static IMemberAccessor findAccessor(IType type, String identifier) { + Map, ?> keys = type.getAccessorKeys(); + for (IAccessorKey key : keys.keySet()) { + if (identifier.equals(key.getIdentifier())) { + return type.getAccessor(key); + } + } + return null; + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainTestSeamsTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainTestSeamsTest.java new file mode 100644 index 0000000000..c2109d8d09 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainTestSeamsTest.java @@ -0,0 +1,203 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.referencechains; + +import com.datadoghq.profiler.AbstractProfilerTest; +import com.datadoghq.profiler.JavaProfiler; +import com.datadoghq.profiler.Platform; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.junit.jupiter.api.Assumptions.assumeTrue; + +/** + * PROF-15341 follow-up: {@code ReferenceChainTrackingTest} exercises {@code LivenessTracker}'s + * probabilistic allocation-sampling-driven slope detection and {@code ReferenceChainTracker}'s + * root-seeded BFS discovery together, in one real-JVM run - reliable only when both mechanisms + * happen to line up within the same bounded retry window. This class decouples them via the + * debug-build-only test seams on {@link JavaProfiler} (backed by {@code LivenessTracker}'s + * existing {@code *ForTest} seams and a new {@code ReferenceChainTracker::tagAsRootForTest()}, + * see javaApi.cpp), so each mechanism can be verified end-to-end in isolation: + *

      + *
    • {@link #shouldSelectSeededKlassAsLeakCandidateOnPositiveSlope()} - asserts a slope signal + * would fire from directly-seeded population history, with no allocation sampling involved.
    • + *
    • {@link #shouldReconstructChainForDirectlyTaggedRoot()} - asserts the BFS/chain- + * reconstruction/event-emission path fires for a directly-tagged, known live object, with no + * dependency on {@code selectLeakCandidates()} organically picking the right klass.
    • + *
    + * + *

    Only runs under the debug native build ({@code testdebug}) - the backing native methods do + * not exist in a release build (see javaApi.cpp's {@code #ifdef DEBUG} guard), mirroring + * {@code JVMAccessTest}'s own {@code "debug".equals(System.getProperty("ddprof_test.config"))} + * pattern for the same reason. + */ +public class ReferenceChainTestSeamsTest extends AbstractProfilerTest { + + // Arbitrary, test-chosen klass ids - opaque keys for the population table, EXCEPT + // where a representative is involved: setKlassPopulationRepresentativeForTest() now + // aliases the representative's REAL class id onto the seeded synthetic id (the + // leak-tag redesign keys every consumer by real ids), so shouldReconstructChainForDirectlyTaggedRoot() + // must target a class whose real population entry does not already exist - a unique + // nested class, not java.lang.Object - otherwise the real entry's genuine (noisy) + // history outranks the seeded ramp and the target never becomes a candidate. + private static final int SLOPE_TEST_KLASS_ID = 987001; + private static final int CHAIN_TEST_KLASS_ID = 987002; + + /** Unique class for {@link #shouldReconstructChainForDirectlyTaggedRoot()}'s target. */ + private static final class ChainTarget {} + + /** + * Durable root for {@link #shouldReconstructChainForDirectlyTaggedRoot()}'s target: the + * noise gate (suppressChainEvent) intentionally drops depth<2 chains whose root is + * transient (stack local / JNI local) - a method-local target is transient BY DESIGN, so + * holding it statically (the canonical leak shape: static field) is required for the + * chain event to be emitted at all. Shared-JVM hygiene: nulled in a finally below so the + * next test in this no-forkEvery JVM does not inherit the reference. + */ + private static Object chainTargetHolder; + + @Override + protected String getProfilerCommand() { + // generations=true: gates LivenessTracker's population tracking (gcGenerationsEnabled()) - + // required for selectLeakCandidates() to return anything at all, real or seeded. + // referencechains=true: constructs ReferenceChainTracker's FrontierTable so + // tagAsReferenceChainRoot0()/runReferenceChainPass0() have a table to insert into. + // framecap=2000000 (not 1024): shouldReconstructChainForDirectlyTaggedRoot()'s first pass + // runs IterateOverReachableObjects, which admits every GC root in the whole JVM - not just + // this test's directly-tagged target - into the same frontier. A small framecap fills with + // that ambient root count before the tagged root ever gets to expandFrontier(), hitting + // SearchAbandonReason::FRONTIER_CAP with no chain event queued. Same value/rationale as + // ReferenceChainTrackingTest's own success-path tests (see that class's header comment). + return "generations=true,referencechains=true:hops=32:budget=500:ttl=60000:framecap=2000000"; + } + + @Override + protected boolean isPlatformSupported() { + return !(Platform.isJavaVersion(8) || Platform.isJ9() || Platform.isZing()); + } + + private static void assumeDebugBuild() { + assumeTrue("debug".equals(System.getProperty("ddprof_test.config"))); + } + + /** + * Seeds twenty epochs of strictly increasing population counts for {@link #SLOPE_TEST_KLASS_ID} + * directly into LivenessTracker's ring buffer - bypassing real allocation sampling entirely - + * then asserts {@code selectLeakCandidates()} ranks it as a leak candidate. This is the "assert + * a slope signal would be generated" seam: it proves the ranking logic itself works without + * depending on the real JVMTI heap sampler ever surfacing this specific klass. + */ + @Test + public void shouldSelectSeededKlassAsLeakCandidateOnPositiveSlope() { + assumeDebugBuild(); + JavaProfiler.resetKlassPopulationForTest0(); + JavaProfiler.setGcGenerationsEnabled0(true); + + // KLASS_POPULATION_MIN_FILL_FOR_TREND = 10 (livenessTracker.h) is only the floor at which a + // trend becomes eligible at all - selectLeakCandidates() also requires consecutive_positive + // to reach LEAK_TREND_HYSTERESIS_BASE = 5 consecutive qualifying epochs before trusting it + // (the sustained-trend hysteresis gate), so at least 10 + (5 - 1) = 14 strictly increasing + // samples are needed; 20 gives headroom. + for (int epoch = 1; epoch <= 20; epoch++) { + JavaProfiler.seedKlassPopulationSample0(SLOPE_TEST_KLASS_ID, epoch * 10, epoch); + // Per-(klass, tid) qualification: selectLeakCandidates() now requires a + // qualifying allocating thread on top of the klass-level ramp. This test + // only asserts the klass makes it into the candidate list - it has no + // tracked instances whose tid a tagging scope would need to match - so a + // fixed synthetic tid suffices here. + JavaProfiler.seedTidTrendSample0(SLOPE_TEST_KLASS_ID, 4242, epoch * 3, epoch); + } + + int[] candidates = JavaProfiler.selectLeakCandidateKlassIds0(); + boolean found = false; + for (int klassId : candidates) { + if (klassId == SLOPE_TEST_KLASS_ID) { + found = true; + break; + } + } + assertTrue(found, "Expected klass id " + SLOPE_TEST_KLASS_ID + + " to be selected as a leak candidate after a seeded positive-slope population history"); + } + + /** + * Tags a real, live, caller-chosen object directly as a reference-chain frontier root + * (bypassing ReferenceChainTracker's normal root-seeded discovery walk), wires it in as a + * seeded leak candidate's representative, then drives one BFS pass and one poll cycle + * synchronously. This is the "trigger the refchain on a known live heap sample" seam: it + * proves {@code runPass()}/{@code pollWatchedTargets()}/{@code buildChainEvent()} correctly + * produce a chain event for a target this test controls directly, decoupled from whether + * LivenessTracker's probabilistic sampler would have picked the same object on its own. + */ + @Test + public void shouldReconstructChainForDirectlyTaggedRoot() { + assumeDebugBuild(); + JavaProfiler.resetKlassPopulationForTest0(); + JavaProfiler.setGcGenerationsEnabled0(true); + // Guards against inheriting a FrontierTable an earlier test in this same, no-forkEvery + // JVM left permanently full/tiny - e.g. ReferenceChainTrackingTest's own + // shouldReportAbandonedSearchOnTinyFrontierCap deliberately drives the shared table to + // framecap=1 and leaves it that way. Same defensive pattern that class's own tests already + // use (see its header comment); without it, tagAsReferenceChainRoot0()'s insert() below can + // fail against a table this test never sized itself. + JavaProfiler.resetReferenceChainSearchForTest0(); + + // The target is referenced ONLY through the static holder - never bound to a + // live local of this frame. A local `target` variable would make the object a + // STACK_LOCAL GC root for the whole test body, and the noise gate (rightly) + // suppresses depth-0 chains rooted in a transient stack slot: the durable + // roots this fixture wants (the static holder, plus the JNI-global-weak-ref + // root LivenessTracker's representative itself creates) then lose the + // durability race only if a stack-local root exists at all. Accessing the + // object only via the static field leaves its frame slots empty between + // calls, so no stack root is ever enumerated. + chainTargetHolder = new ChainTarget(); + + long tag = JavaProfiler.tagAsReferenceChainRoot0(chainTargetHolder); + assertTrue(tag > 0, "Expected tagAsReferenceChainRoot0 to assign a valid frontier tag"); + + // Representative BEFORE seeding: with the aliasing seam, setting the representative is + // what registers the synthetic->real id alias, so the seeding below lands re-keyed under + // the target's real class id (seeding first still works - the synthetic entry gets + // re-keyed at representative-set time - but rep-first is the documented load-bearing order). + JavaProfiler.setKlassPopulationRepresentativeForTest0(CHAIN_TEST_KLASS_ID, chainTargetHolder); + + // See shouldSelectSeededKlassAsLeakCandidateOnPositiveSlope()'s comment above for why 20 + // (not just KLASS_POPULATION_MIN_FILL_FOR_TREND = 10) is needed to clear the hysteresis gate. + for (int epoch = 1; epoch <= 20; epoch++) { + JavaProfiler.seedKlassPopulationSample0(CHAIN_TEST_KLASS_ID, epoch * 10, epoch); + // Per-(klass, tid) qualification, seeded with THIS thread's real + // profiler tid: the target was allocated here, so if any path in the + // pass/poll cycle below does hinge on leak tagging of the target as a + // tracked instance, the qualifying tid matches it. (The chain's primary + // path is the directly-tagged frontier root + seeded representative, so + // this is belt-and-braces, not a proven requirement.) + JavaProfiler.seedTidTrendSample0( + CHAIN_TEST_KLASS_ID, JavaProfiler.getTid(), epoch * 3, epoch); + } + + // Drive pass+poll cycles until the chain event lands. One pass is NOT enough: + // admitStaticFieldRoots() sweeps loaded classes in chunks (a few hundred per pass, + // bounded by the budget) and the first pass truncates before reaching this test's + // class - the durable (static-field) root discovery that upgrades the target's + // root_kind only happens on the pass whose chunk covers ChainTarget's holder class. + // The bound (40) gives the full sweep lap (~10 passes for ~4000 loaded classes) + // several round trips of headroom; the BFS thread is excluded from racing these + // calls by the engine lock (runPassSerialized/pollWatchedTargetsSerialized). + int eventCount = 0; + for (int pass = 0; pass < 40 && eventCount == 0; pass++) { + boolean sawPassRun = JavaProfiler.runReferenceChainPass0(); + assertTrue(sawPassRun || pass > 0, + "Expected the first runReferenceChainPass0 to run (reference chains enabled)"); + JavaProfiler.pollReferenceChainTargets0(); + eventCount = JavaProfiler.drainReferenceChainEventCount0(); + } + assertTrue(eventCount > 0, + "Expected pollWatchedTargets() to have queued at least one chain event for the " + + "directly-tagged, seeded-representative target"); + chainTargetHolder = null; // shared-JVM hygiene + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainTrackingTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainTrackingTest.java new file mode 100644 index 0000000000..c1f5a3a676 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainTrackingTest.java @@ -0,0 +1,976 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.referencechains; + +import com.datadoghq.profiler.AbstractProfilerTest; +import com.datadoghq.profiler.JavaProfiler; +import com.datadoghq.profiler.JfrEvent; +import com.datadoghq.profiler.JfrEvents; +import com.datadoghq.profiler.Platform; +import org.junit.jupiter.api.MethodOrderer; +import org.junit.jupiter.api.Order; +import org.junit.jupiter.api.Tag; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.TestMethodOrder; +import org.junitpioneer.jupiter.RetryingTest; + +import java.nio.file.Files; +import java.util.concurrent.LinkedBlockingQueue; +import java.util.concurrent.Semaphore; +import java.util.concurrent.atomic.AtomicLong; +import java.nio.file.Path; +import java.nio.file.Paths; +import java.util.ArrayList; +import java.util.HashMap; +import java.util.List; +import java.util.Map; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertTrue; + +/** + * PROF-15341 (+ lifecycle-wiring follow-up, + the Remaining Work Plan's target-selection bridging, + * pause-time pacing, and reporting work): end-to-end + * coverage for {@code ReferenceChainTracker} (ddprof-lib/src/main/cpp/referenceChains.h/.cpp). + * + *

    Scope note: {@code ReferenceChainTracker::start()} is now called from + * {@code Profiler::start()} (profiler.cpp), gated on {@code args._reference_chains}, followed by + * {@code ReferenceChainTracker::startThread()} which spawns its BFS thread; {@code Profiler::stop()} + * calls the matching {@code stopThread()}/{@code stop()} pair. A {@code datadog.ReferenceChainAbandoned} + * event now reaches a live {@code Recording}: {@code Profiler::dump()} calls + * {@code buildAbandonedEvent()} and writes its output via + * {@code Profiler::writeReferenceChainAbandoned()}/{@code FlightRecorder::recordReferenceChainAbandoned()} + * whenever the search has ended in {@code SearchState::ABANDONED} - mirroring the + * {@code LivenessTracker::flush()} call site there. {@link #shouldReportAbandonedSearchOnTinyFrontierCap()} + * exercises that path end-to-end below. + * + *

    {@code buildChainEvent()}'s target-selection feed (Remaining Work Plan's target-selection + * bridging step): + * {@code ReferenceChainTracker::pollWatchedTargets()} (referenceChains.cpp), called from + * {@code threadLoop()} once per scheduling cycle after {@code runPass()}, closes the gap this class's + * header comment used to describe as unclosed. It polls + * {@code LivenessTracker::selectLeakCandidates()} (positive population-slope ranking over a rolling + * per-klass survivor-count window, gated on {@code _gc_generations} - design doc's Open Question 3) + * and, for each ranked klass whose representative instance an ordinary {@code runPass()} walk has + * *already* tagged ({@code getTag() > 0} - a read, never a {@code SetTag} seed), reconstructs and + * emits its chain via {@code Profiler::writeReferenceChain()}. {@link #shouldReconstructReferrerChainToGcRoot()} + * below exercises this end to end against a real JVM, real GCs, and a real JVMTI heap walk - not a + * synthetic frontier fixture (see {@code referenceChainJfrRoundtrip_ut.cpp}/{@code ReferenceChainJfrParserTest} + * for that already-covered, lower-level reconstruction-correctness proof). + * + *

    Why {@code @TestMethodOrder}/{@code @Order(1)}: {@code ReferenceChainTracker} is a + * process-wide singleton whose single, singleton-owned search (design doc's Open Question 3 + * "Shipped" note) never restarts once it leaves {@code SearchState::RUNNING} - + * {@code shouldRunPass()} (referenceChains.cpp) returns {@code false} forever after that point, and + * the transition itself releases every tag the search ever assigned + * ({@code releaseSearchTags()}). Its {@code FrontierTable} is sized once, the first time any test + * in this JVM calls {@code ReferenceChainTracker::start()}, and never resized on later + * start()/stop() cycles (referenceChains.cpp's {@code if (_frontier == nullptr)} guard) - so + * whichever test runs first also fixes that capacity for every test that runs after it in the same + * JVM (no {@code forkEvery} configured, see {@code ProfilerTestPlugin.kt}, so this whole class runs + * in one). {@link #shouldReportAbandonedSearchOnTinyFrontierCap()} below deliberately drives that + * shared search to {@code SearchState::ABANDONED}, via an artificially tiny frontier cap when it + * gets to build the table itself, or its own {@code ttl} fallback otherwise (see that method's own + * comment). Both {@link #shouldReconstructReferrerChainToGcRoot()} and + * {@link #shouldReconstructReferrerChainThroughUnboundedCacheLeak()} need the search still + * {@code RUNNING} to find their own candidates' tags non-zero, so they are pinned to run first, + * via {@code @Order(1)}/{@code @Order(2)} respectively - without that ordering all three tests + * would race for which ones get to observe a still-{@code RUNNING} search, and neither + * success-path test has a fallback for losing that race the way the abandonment test does. The + * relative order between the two success-path tests does not itself matter - both target + * different klasses ({@link ChainLink} vs {@link CachedPayload}) within the same shared search, + * and neither exhausts it - only their both running before the abandonment test does. + */ +@TestMethodOrder(MethodOrderer.OrderAnnotation.class) +@Tag("slow") +public class ReferenceChainTrackingTest extends AbstractProfilerTest { + + // Arbitrary, test-chosen klass ids for the debug-only population-seeding seams below (see + // ReferenceChainTestSeamsTest's own comment: LivenessTracker's population table treats these as + // opaque keys, so they need not resolve to any real class). Distinct per test/from + // ReferenceChainTestSeamsTest's own ids purely as cheap insurance against collision in a shared, + // no-forkEvery test JVM. + private static final int CHAIN_LINK_TEST_KLASS_ID = 987201; + private static final int CACHED_PAYLOAD_TEST_KLASS_ID = 987202; + + // Durable (static-field) holder for the gc-root fixture below. The + // discovered-chain gate suppresses chains shallower than the first real + // holder hop that are rooted at a TRANSIENT root (stack/JNI local) - a + // frame-held "leak" is by definition not a leak - and + // shouldReconstructReferrerChainToGcRoot()'s ChainLink instances sit at + // depth 1 of a locally-held list, exactly the transient-rooted shape the + // gate exists to suppress. A static field gives them the real + // unmaintained-singleton-collection retention shape instead. (The cache + // fixture below needs no such change: its CachedPayload chains are three + // hops deep through HashMap's own internals, which the gate correctly + // lets through regardless of root kind.) Cleared in the test's finally: + // this is a shared, no-forkEvery test JVM, and a leftover holder would + // keep its entire population reachable for every later test here. + private static List gcRootHolder; + + // The current round's allocation batch, handed to the persistent allocator + // thread below. Volatile: written here, read by the allocator thread every + // round - the request queue's happens-before (offer after set, take before + // read) already covers it, but the visibility of a plain field handed + // across threads is kept explicit rather than relying on queue semantics + // alone. + private static volatile List currentBatch; + + // LivenessTracker's own hysteresis gate (livenessTracker.h's + // KLASS_POPULATION_MIN_FILL_FOR_TREND=10 / LEAK_TREND_HYSTERESIS_BASE=5; + // livenessTracker.cpp's recordKlassPopulationSampleLocked()/hasQualifyingGrowth()) only starts + // incrementing KlassPopulationEntry::consecutive_positive once ring_fill has reached 10 - every + // call before that resets it straight back to 0 (ringThirdsStats()'s own fill < min_fill guard). + // A single batch of exactly 10 seedKlassPopulationSample0() calls therefore caps + // consecutive_positive at 1 (only the 10th call ever sees ring_fill==10), far short of the >=5 + // (or >=3 if LivenessTracker::heapFloorRising() independently corroborates) that + // selectLeakCandidates() (livenessTracker.cpp) requires before it reports a candidate at all - + // which is what ReferenceChainTracker::hasLeakSignal() consults, and what shouldRunPass() + // (referenceChains.cpp) gates even this search's very first pass on whenever generations=true + // (see this class's own header comment on why runPass() never otherwise fires here without one). + // 10 (KLASS_POPULATION_MIN_FILL_FOR_TREND) + 5 (LEAK_TREND_HYSTERESIS_BASE) - 1 = 14 calls are + // the minimum for the 5 qualifying calls at ring_fill>=10 (epochs 10-14) to actually land; +1 + // for margin against off-by-one uncertainty in that derivation. + private static final int SEED_EPOCHS_FOR_HYSTERESIS = 15; + + // Dedicated id for shouldReportAbandonedSearchOnTinyFrontierCap()'s own seeding (see that + // method's own comment) - distinct from CHAIN_LINK_TEST_KLASS_ID/CACHED_PAYLOAD_TEST_KLASS_ID + // for the same cheap-insurance-against-collision reason those two are distinct from each other. + private static final int ABANDON_TEST_KLASS_ID = 987203; + + @Override + protected String getProfilerCommand() { + String testName = testInfo != null + ? testInfo.getTestMethod().map(java.lang.reflect.Method::getName).orElse("") + : ""; + if ("shouldReconstructReferrerChainToGcRoot".equals(testName) + || "shouldReconstructReferrerChainThroughUnboundedCacheLeak".equals(testName)) { + // memory=...:l + generations=true: LivenessTracker::gcGenerationsEnabled() (Remaining + // Work Plan's population tracking / target-selection bridging, livenessTracker.h) must be true for pollWatchedTargets() + // (referenceChains.cpp) to do anything at all - referencechains=... alone stays + // whole-graph-only (design doc's Open Question 3 "still undecided" fallback). A small + // sampling interval maximizes the chance the many ChainLink instances this test + // allocates below actually get liveness-tracked (LivenessTracker::track(), invoked from + // the allocation-sampling path whenever _gc_generations || _record_liveness is set, + // objectSampler.cpp). A large framecap avoids an early + // SearchAbandonReason::FRONTIER_CAP abandonment against this JVM's real, + // not-controlled-by-this-test root-reachable graph size, and a long ttl avoids a + // TTL abandonment mid-test - either would call releaseSearchTags() and permanently + // zero every tag this test depends on (see this class's header comment on why this + // test is pinned to @Order(1)). + // + // pausetarget=5000: the pause-time pacing controller's updatePacing() (referenceChains.cpp) measures this JVM's real + // FollowReferences/GetObjectsWithTags call latency at ~150ms early on, climbing past a full + // second as this test's own allocation keeps handing the search more to discover (each pass + // re-scans the whole already-discovered/tagged set, so cost grows with cumulative progress, + // not with this test's own allocation *rate*) - i.e. that latency is dominated by per-call + // overhead, not by the requested edge budget. Against the default pausetarget=5ms ceiling + // (DEFAULT_REFERENCE_CHAINS_PAUSE_TARGET_MS, arguments.h) every observed pass is "over + // ceiling even at the floor", so the controller settles at MIN_EFFECTIVE_BUDGET=50 edges/pass + // and MAX_EFFECTIVE_CADENCE_NS=4s between passes - a real but glacial discovery rate this + // test's own wall-clock budget cannot outwait, and one a merely-generous ceiling (e.g. 500ms) + // only postpones: once cumulative progress pushes the real per-pass cost back past *that* + // ceiling too, the same throttling recurs. A ceiling comfortably above the highest per-call + // cost this test's own scale of population growth is expected to reach instead lets the + // controller keep _effective_budget at this test's own configured budget=4000 ceiling for + // the method's entire run, the same convergence behavior referenceChains_ut.cpp's own pacing + // tests already exercise synthetically. + // + // painbudget=100: shouldReconstructReferrerChainThroughUnboundedCacheLeak shares this + // singleton search with shouldReconstructReferrerChainToGcRoot (@Order(1), runs first) and + // needs a restart (canAffordNewSearch(), referenceChains.cpp) once that first search + // reaches a terminal state. PainBudget (painBudget.h) gates that restart on + // pain_spent_ms / refill_rate milliseconds of cooldown; the default + // DEFAULT_REFERENCE_CHAINS_PAIN_BUDGET_PERCENT=1 (1% refill) can make that cooldown run to + // tens of seconds to minutes for a search this large - far past this test's own bounded + // retry loop (16 rounds plus a short grace period). painbudget=100 keeps the restart gate + // affordable immediately, which is fine here since nothing else in this JVM is competing + // for the pain budget. + // + // firstpassbudget=200000: root/stack-ref enumeration (runPassManualWalk()'s + // IterateOverReachableObjects call) shares budget=4000 with expandFrontier()'s steady-state + // expansion unless overridden - in this shared, forked test JVM's real (not controlled by + // this test) root-reachable graph, that leaves this method's own CachedPayload-holding + // cache HashMap (a direct stack-local root on this test's own thread) at the mercy of + // whatever order JVMTI's enumeration happens to visit roots in: budget=4000 was observed + // exhausting itself on other roots' subtrees every single attempt, never once admitting a + // single edge from this thread's own stack. Overriding just root enumeration's own budget + // (not budget=4000, which the pacing controller has already been tuned around - raising it + // instead throttles every expansion pass down to + // MIN_EFFECTIVE_BUDGET/MAX_EFFECTIVE_CADENCE_NS, see updatePacing()) gives every root + // enumeration attempt (the first pass, and any later one ROOT_ENUM_MIN_INTERVAL_NS gates + // back in - see that constant's own comment) enough headroom to reach this thread's stack + // without slowing down the steady-state expansion passes that follow. + // memory=64:l:1.0 - the explicit 100% live-sample ratio is load-bearing: + // without a ratio segment the default is 10% (arguments.h's + // _live_samples_ratio) and LivenessTracker::track() drops 90% of + // tracked instances probabilistically, thinning the pool the + // representative-minting and candidate trend rings draw from - + // observed live as the intermittent "representative died/evicted" + // and zero-tag runs (see LeakTagCorrelationReferenceChainTest's own + // comment for the full lottery analysis). Total tracking volume is + // still capped by the 256KiB sampling-interval floor, not the ratio, + // so 100% keeps the tracking table tiny. + return "memory=64:l:1.0,generations=true," + + "referencechains=true:hops=64:budget=4000:ttl=120000:framecap=2000000:pausetarget=5000:painbudget=100:firstpassbudget=200000"; + } + // shouldReportAbandonedSearchOnTinyFrontierCap needs the frontier table (not the + // per-pass budget) to be what runs out first: heapReferenceCallback() (referenceChains.cpp) + // checks the budget before ever calling FrontierTable::insert(), so a budget as small as + // the frontier cap itself (e.g. budget=1:framecap=1) only ever exhausts the budget after + // admitting exactly one object - insert() never gets a chance to fail, and the search + // completes normally instead of abandoning. A budget comfortably larger than framecap=1 + // lets a *second* root-referenced object reach insert() and hit the actually-full table, + // which is what runPass() treats as grounds to abandon (Termination section priority 1). + // + // framecap=1 only actually controls the table's capacity the *first* time + // ReferenceChainTracker::start() ever constructs its FrontierTable in this JVM + // (referenceChains.cpp's `if (_frontier == nullptr)` guard - the table, like + // LivenessTracker's own, survives every later start()/stop() cycle at whatever size it was + // first built with). This method's own @Order(1) pinning on + // shouldReconstructReferrerChainToGcRoot means *that* test's own, much larger framecap=2000000 + // wins that race instead - so this method also supplies a small ttl as a second, independent + // abandonment trigger. runPass()'s Termination-section priority still checks frontier-cap + // first, so a fresh, genuinely-tiny table (this method running first, e.g. in isolation via + // `-Ptests=ReferenceChainTrackingTest.shouldReportAbandonedSearchOnTinyFrontierCap`) still + // abandons via SearchAbandonReason::FRONTIER_CAP as originally designed; inheriting an + // already-oversized table instead falls through to the ttl check, which fires almost + // immediately regardless of table size because _search_start_ns (referenceChains.h) is set + // once, the first time the shared search ever started - by the time this method's own + // start() call re-parses ttl, that clock already reads however long + // shouldReconstructReferrerChainToGcRoot's own run took. Either path produces a + // datadog.ReferenceChainAbandoned event, which is all this method actually asserts on. + // + // painbudget=100, for the same reason shouldReconstructReferrerChainThroughUnboundedCacheLeak() + // above needs it (see that method's own comment on painbudget=100): this method's own + // resetReferenceChainSearchForTest0() call spends whatever _search_pain_ms + // shouldReconstructReferrerChainThroughUnboundedCacheLeak()'s own long-running search left + // behind (found the hard way - that search runs 20+ passes admitting tens of thousands of + // edges each, and nothing zeroes _search_pain_ms between its own end and this method's start() + // reconstructing a fresh, zero-balance PainBudget) into *this* method's own PainBudget + // (referenceChains.cpp's restartSearch()/resetSearchStateForTest(), both call + // _pain_budget.spend(_search_pain_ms) unconditionally) - at the default 1% refill rate, + // draining that debt back to canStartNow()==true would take on the order of minutes, far past + // this method's own retry budget below. painbudget=100 keeps that gate affordable immediately, + // the same way it does for the sibling method. + if ("shouldReportAbandonedSearchOnTinyFrontierCap".equals(testName)) { + return "referencechains=true:hops=32:budget=500:framecap=1:ttl=100:painbudget=100"; + } + // Deliberately does not request cpu/wall/memory/nativemem: those categories also + // write an "enabled" ActiveSetting (flightRecorder.cpp:1141-1144) and all default to + // false when not requested, so "enabled"="true" is unambiguous evidence of the + // datadog.ReferenceChain setting specifically without needing to disambiguate by the + // ActiveSetting "id" field (JMC's generic accessor lookup does not resolve that field + // for this custom event type - not worth a bespoke accessor for one assertion). + return "referencechains=true:hops=32:budget=500:ttl=2000:framecap=256"; + } + + @Override + protected boolean isPlatformSupported() { + // FollowReferences/tag-based frontier walking (referenceChains.cpp) assumes a + // HotSpot-shaped JVMTI heap implementation; excluded platforms mirror + // LivenessTrackingTest's own guard (memleak/LivenessTrackingTest.java). + return !(Platform.isJavaVersion(8) || Platform.isJ9() || Platform.isZing()); + } + + /** + * Verifies the {@code referencechains=...} flag round-trips through + * {@code Arguments} parsing (arguments.cpp's {@code CASE("referencechains")}) into the + * {@code datadog.ReferenceChain} JFR setting (flightRecorder.cpp:1143's + * {@code writeBoolSetting(buf, T_REFERENCE_CHAIN, "enabled", args._reference_chains)}). + */ + @RetryingTest(5) + public void shouldExposeReferenceChainsSettingWhenEnabled() { + stopProfiler(); + JfrEvents settings = verifyEvents("jdk.ActiveSetting"); + boolean sawEnabledSetting = false; + for (JfrEvent item : settings) { + if ("enabled".equals(item.getString("name")) && "true".equals(item.getString("value"))) { + sawEnabledSetting = true; + } + } + assertTrue(sawEnabledSetting, "datadog.ReferenceChain#enabled setting was not found"); + } + + /** + * This test's stated success-path scenario, now exercising the Remaining Work Plan's + * target-selection bridging feed instead of staying disabled: allocates a growing population of + * {@link ChainLink} instances, forcing GCs between allocation rounds so + * {@code LivenessTracker::cleanup_table()}'s epoch-advance pass (livenessTracker.cpp) observes a + * rising per-klass survivor count each time, until {@code selectLeakCandidates()} trusts the + * resulting trend (needs {@code KLASS_POPULATION_MIN_FILL_FOR_TREND = 10} ring-buffer samples). + * Then waits for {@code ReferenceChainTracker::pollWatchedTargets()} to notice that ranked + * candidate has already been tagged by an ordinary {@code runPass()} walk and reconstruct + + * emit its chain, and asserts on the resulting {@code datadog.ReferenceChain} event. + * + *

    Each population-growth round pairs with an explicit {@link #dump(Path)}: per + * livenessTracker.cpp, {@code cleanup_table()}'s per-klass population accounting only runs + * (unforced) from {@code flush_table()}, which only runs from {@code LivenessTracker::flush()}, + * which is called *only* from {@code Profiler::dump()} - there is no timer-driven flush. + * {@code System.gc()} bumps {@code LivenessTracker::_gc_epoch} synchronously inside the + * {@code GarbageCollectionFinish} callback ({@code onGC()}), so by the time {@code System.gc()} + * returns to Java the epoch bump is already visible - no extra sleep is needed between the + * {@code gc()} and the {@code dump()} that observes it. + */ + @Test + @Order(1) + public void shouldReconstructReferrerChainToGcRoot() throws Exception { + // Seed gcRootHolder with one live element *before* resetting the search below: ArrayList's + // backing array starts out as the shared empty-array sentinel, and only gets replaced with a + // real array on its first grow() (addAll()). ReferenceChainTracker::expandFrontier() + // (referenceChains.cpp) freezes whatever children it observes for a node the moment that + // node is expanded - it never re-examines an already-EXPANDED node for a later field mutation, + // regardless of how many times root/stack-ref enumeration itself reruns over the search's + // lifetime (runPassManualWalk()'s own comment). If gcRootHolder's node gets expanded while + // elementData still pointed at the empty sentinel (nothing stops some *unrelated* GC in this + // shared, no-forkEvery JVM from waking the freshly (re)started BFS thread in the gap between + // resetting the search below and round 1's own addAll()), the frozen children set would be + // permanently empty, and no ChainLink added in any of the 16 rounds that follow would ever + // become reachable - matching a real CI failure where pollWatchedTargets() reported tag=0 for + // this klass on every single poll across the whole test. Seeding one element first means the + // backing array is never empty at any point after the search (re)starts: every later + // addAll()/grow() copies all prior elements (including this one) into the new array, so + // whichever array the walk happens to snapshot when it expands the node, index 0 is always + // present in it. + gcRootHolder = new ArrayList<>(); + gcRootHolder.add(new ChainLink("gc-root-seed")); + if ("debug".equals(System.getProperty("ddprof_test.config"))) { + // Being pinned to run first *within this class* (this class's own header comment, "Why + // @TestMethodOrder/@Order(1)") does not guarantee this is the first reference-chain test + // to ever call into the shared, process-wide singleton ReferenceChainTracker in this whole + // test JVM - no forkEvery is configured (same header comment), so another test class can + // run first and leave the singleton search already non-RUNNING/already-tagged, in which + // case runPass() never takes its real root-seeded-walk branch again. Force a genuine fresh + // search so this test's own ChainLink population is guaranteed reachable by a real walk, + // regardless of what ran earlier in this JVM. Debug-only: this native seam does not exist + // in a release build. + // + // resetKlassPopulationForTest0() first, for the same "no forkEvery" reason: + // LivenessTracker's _klass_population table (livenessTracker.cpp) also survives stop()/ + // start() cycles, so an unrelated earlier test class in this same shared JVM (e.g. another + // generations=true scenario) could leave a klass already past hasLeakSignal()'s + // consecutive_positive hysteresis gate (selectLeakCandidates(), livenessTracker.cpp) + // sitting in that table - which would let this test's first pass start for a reason that + // has nothing to do with the seeding this method does further down (see + // SEED_EPOCHS_FOR_HYSTERESIS's own comment), masking whether that seeding is actually + // sufficient on its own. + JavaProfiler.resetKlassPopulationForTest0(); + JavaProfiler.resetReferenceChainSearchForTest0(); + } + // One persistent allocator thread, reused by every round - NOT a fresh + // spawned thread per round. Same JVMTI property the per-round spawn + // relied on (SampledObjectAlloc never fires for this JUnit worker + // thread's own allocations, but reliably fires for threads created + // after Profiler::start() already ran), plus the decisive per-tid + // consequence: LivenessTracker's per-(klass, tid) qualification trend + // accumulates REAL samples for this one thread's tid every round + // (foldKlassCountsLocked() folds each tracked instance's tid), so + // qualification is organic and cannot age out mid-test. The previous + // per-round spawns gave every round a DIFFERENT short-lived tid, so + // the seeded (klass, worker-tid) qualification ramp was the only + // qualifying signal - and once its ring slots aged out (observed + // live: slope decayed to ~0 at ring_fill=22, consecutive_positive + // reset, the candidate never returned again) the candidate dropped + // out of selectLeakCandidates() with the whole crawl still unfinished. + LinkedBlockingQueue allocRequests = new LinkedBlockingQueue<>(); + Semaphore roundDone = new Semaphore(0); + AtomicLong allocatorTid = new AtomicLong(-1); + Thread allocator = new Thread(() -> { + allocatorTid.set(JavaProfiler.getTid()); + try { + Integer requested; + while ((requested = allocRequests.take()) != null) { + List batch = currentBatch; + for (int i = 0; i < requested; i++) { + batch.add(new ChainLink("leak-" + requested + "-" + i)); + } + roundDone.release(); + } + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); // test shutdown - see finally below + } + }, "togcroot-leak-allocator"); + allocator.start(); + // The seeding block below needs the allocator's tid; the thread publishes + // its own tid as its first action, so this wait is bounded by thread start. + while (allocatorTid.get() == -1) { + Thread.sleep(1); + } + + Path scratchDumpPath = Paths.get("referencechains-population-scratch.jfr"); + try { + // Up to 16 rounds: selectLeakCandidates()'s KLASS_POPULATION_MIN_FILL_FOR_TREND = 10 + // (livenessTracker.h) needs 10 *epochs that actually observe a surviving ChainLink sample*, + // not just 10 dump() calls - allocation sampling is probabilistic (see below), so some + // early, smaller rounds may not land a single ChainLink sample. Growing the per-round count + // compensates, and this loop keeps going (checking for the event every round) rather than + // committing to a fixed round count up front, so it self-adjusts to whatever this JVM's + // actual sampling behavior turns out to be. Staying under KLASS_POPULATION_RING_SIZE = 30 + // keeps every round's sample within the trend computeKlassPopulationSlope() reads back out. + // Capped at 16 rather than a larger margin above the 10-round minimum: this loop's own + // worst case (every round retained, no early match) runs inside the same forked test JVM + // every other ddprof-test class shares (ProfilerTestPlugin.kt's shared -Xmx512m default, + // no forkEvery) - a prior CI run hit "Java heap space" in that shared fork with this loop + // capped at 25, so the cap trades some of the original margin above the 10-round minimum + // for staying inside that shared heap. + // + // Per-round size growth itself is clamped to round 10 (Math.min(round, 10) below): rounds + // past 10 exist only to give pollWatchedTargets() more retries against the lock-contention + // race described below, not to keep building the population trend (already eligible by + // round 10) - letting round*600 keep scaling unclamped through round 16 made rounds 11-16 + // each add strictly more retained garbage than the last for no trend benefit, which is what + // drove the "Java heap space" failure this comment's own history refers to. + // + // Instance count is sized against ObjectSampler::check() (objectSampler.cpp), not this + // test's own "memory=64" request: "do not allow shorter interval than 256KiB" means the + // *actual* allocation-sampling interval is 262144 bytes regardless of the small value + // requested above (that request only affects LivenessTracker's own table-capacity formula, + // livenessTracker.cpp's initialize_table()). ChainLink's own padding fields (see that + // class's comment) get enough total megabytes sampled from far fewer instances than plain, + // unpadded ~40-byte instances would need - keeping this test's own contribution to the real + // JVM's root-reachable graph (which ReferenceChainTracker's BFS walk must also traverse, + // Triggering section) from ballooning to the point the walk can't practically catch up + // within this test's own wall-clock budget. ChainLink competes on equal footing against + // other incidental klasses (e.g. "[B"/byte[]) that this same mechanism may also legitimately + // flag as leak candidates - this test's own assertions below look for ChainLink specifically + // among however many datadog.ReferenceChain events actually appear, rather than assuming it + // is the only one. + ReferenceChainAssertions.JfrChainMatch match = null; + boolean seededTestKlassTrend = false; + int totalRounds = 16; + for (int round = 1; round <= totalRounds && match == null; round++) { + int newInstances = Math.min(round, 10) * 600; + List newLinks = new ArrayList<>(newInstances); + currentBatch = newLinks; // published before the request - see the allocator's loop + allocRequests.offer(newInstances); + roundDone.acquireUninterruptibly(); + gcRootHolder.addAll(newLinks); + System.gc(); + dump(scratchDumpPath); + match = ReferenceChainAssertions.findMatchForClass(verifyEvents(scratchDumpPath, "datadog.ReferenceChain", false), ChainLink.class); + + if (match == null && "debug".equals(System.getProperty("ddprof_test.config"))) { + // Debug-only native seams - a no-op in release builds, so this leaves release-build + // coverage of the organic path below completely unchanged. + // + // Seeding a klass_id's population ring here is not just a ranking shortcut for an + // object some other pass has already tagged - with generations=true, + // ReferenceChainTracker::shouldRunPass() (referenceChains.cpp) gates even this search's + // very first pass on LivenessTracker::hasLeakSignal(), which (once secondsToOOM() has + // nothing to report yet, this early) falls through to selectLeakCandidates() requiring + // consecutive_positive >= LEAK_TREND_HYSTERESIS_BASE for *some* klass, real or not (see + // SEED_EPOCHS_FOR_HYSTERESIS's own comment for why 10 seeded epochs alone cannot clear + // that bar). Seeding SEED_EPOCHS_FOR_HYSTERESIS epochs for this test-chosen id clears it + // outright, which is what actually authorizes the background BFS thread's next + // scheduled wake to run its real, root-seeded runPass() walk - nothing before that point + // has tagged gcRootHolder's own ChainLink yet. pollReferenceChainTargets0() right below + // is therefore best-effort, not guaranteed on this call: it only reconstructs a chain + // once getTag() on the representative is actually > 0 (pollWatchedTargets()'s only + // precondition), which needs that background pass to have already run. The retry loop + // below (a 300ms-sleep-then-dump fallback, repeated across rounds) gives the + // now-unblocked background thread room to do that and have its own post-pass + // pollWatchedTargets() call (threadLoop(), referenceChains.cpp) pick the chain up on its + // own, independent of this method's own poll call succeeding on the first try. + // + // Seed exactly once (recordKlassPopulationSampleLocked(), livenessTracker.cpp, always + // *appends* a fresh ring slot rather than overwriting one for a repeated epoch - calling + // this whole block again on a later round would append a second ramp right after the + // first, turning the ring into a non-monotonic sawtooth and destroying the very + // positive-slope signal selectLeakCandidates() needs). Only the poll+dump recheck below + // needs to repeat across rounds - not the seeding - to give the background pass, or a + // lock-contended dump(), more rounds to resolve. + // Representative BEFORE seeding: setKlassPopulationRepresentativeForTest0 + // resolves the representative's real klass id and aliases the synthetic + // id to it, so the hysteresis seeds below land in the real population + // entry (candidate matching is real-id keyed since the leak-tag pool + // redesign; seeds under a bare synthetic id authorize only + // hasLeakSignal()'s candidate-count check and match nothing else). + JavaProfiler.setKlassPopulationRepresentativeForTest0(CHAIN_LINK_TEST_KLASS_ID, gcRootHolder.get(0)); + if (!seededTestKlassTrend) { + // Per-(klass, tid) qualification seeds - the persistent allocator + // thread owns every ChainLink allocation, so ITS tid is the + // qualifying one; the seeds accelerate the organic per-tid growth + // its own tracked instances produce every round (see the + // allocator block's own comment for why qualification aging out + // stranded the crawl before this thread existed). + int leakTid = (int) allocatorTid.get(); + // Seed counts scale as the organic ramp itself (+1 per epoch), + // NOT some inflated multiple: the ring is one continuous series + // of seeds followed by the real folds, and hasQualifyingGrowth() + // takes the difference of third-window MEANS over that whole + // series. Seeds of epoch*10 (10..150) dwarf the real generation + // counts (the real ramp is +1 per epoch, ~1..15 by test end), so + // the moment real samples displace the seeded tail the slope + // went NEGATIVE (-20..-40 observed live: the "transition cliff") + // and the candidate never returned again. With matched-scale + // seeds the series is [1..15, 1, 2, 3, ...] and every third- + // window pairing across the transition still yields a positive + // slope - the candidate keeps qualifying organically the whole + // run, the real signal seamlessly taking over from the seeds. + for (int epoch = 1; epoch <= SEED_EPOCHS_FOR_HYSTERESIS; epoch++) { + JavaProfiler.seedKlassPopulationSample0(CHAIN_LINK_TEST_KLASS_ID, epoch, epoch); + JavaProfiler.seedTidTrendSample0(CHAIN_LINK_TEST_KLASS_ID, leakTid, epoch, epoch); + } + seededTestKlassTrend = true; + } + JavaProfiler.pollReferenceChainTargets0(); + dump(scratchDumpPath); + match = ReferenceChainAssertions.findMatchForClass(verifyEvents(scratchDumpPath, "datadog.ReferenceChain", false), ChainLink.class); + } + + if (match == null) { + // A quiet window with no dump() in flight: Profiler::dump() (profiler.cpp) holds every + // entry of _locks[] exclusively for the duration of its own _jfr.dump() call + // (rotateDictsAndRun()'s lockAll()/unlockAll() pair) - the same array + // Profiler::writeReferenceChain() (profiler.cpp) needs a slot from to record an event at + // all. Back-to-back rounds with essentially no gap between one dump() and the next leave + // pollWatchedTargets() (referenceChains.cpp) little to no window to ever win that race; + // sleeping briefly here, then dumping again with no intervening dump()/gc() in between + // (so cleanup_table()'s epoch-advance pass, livenessTracker.cpp, stays a no-op and the + // population trend built up so far is undisturbed), gives it one. + Thread.sleep(300); + dump(scratchDumpPath); + match = ReferenceChainAssertions.findMatchForClass(verifyEvents(scratchDumpPath, "datadog.ReferenceChain", false), ChainLink.class); + } + } + + // Grace period: the population trend only became eligible (ring_fill >= 10) partway through + // the loop above, and the same lock contention the loop's own comment describes can still + // delay the resulting write past this method's very last dump() - retry a few more times, + // well past that contention window, before concluding the mechanism genuinely did not fire. + for (int attempt = 0; match == null && attempt < 5; attempt++) { + Thread.sleep(1000); + dump(scratchDumpPath); + match = ReferenceChainAssertions.findMatchForClass(verifyEvents(scratchDumpPath, "datadog.ReferenceChain", false), ChainLink.class); + } + + assertNotNull(match, + "Never observed a datadog.ReferenceChain event whose chain[0] is " + ChainLink.class + + " after " + gcRootHolder.size() + " retained ChainLink instances across up to " + + totalRounds + " population-growth rounds plus a grace period"); + // chain[0] is the *target* object's own class, not its holder's: + // ReferenceChainTracker::buildChainEvent()'s reconstructChain() (referenceChains.h) + // appends each visited FrontierEntry's referrer_klass leaf-to-root, and referrer_klass is + // populated from the *discovered object's own* GetClassSignature at insertion time (see + // referenceChainJfrRoundtrip_ut.cpp's insert(tag, parent_tag, referrer_klass, depth) calls, + // where tag 3's own referrer_klass is leafKlass, not middleKlass). The target here is + // always some currently-live ChainLink instance - the only class this test's leak-candidate + // population trend was built from - regardless of exactly which of the many instances + // allocated above LivenessTracker::foldKlassCountsLocked() (livenessTracker.cpp) happened to + // keep as its representative. Everything above chain[0] reflects real JDK-internal + // collection representation (e.g. ArrayList's backing array) rather than anything this test + // controls, so it is deliberately not asserted beyond "at least one hop was reconstructed". + assertEquals(ChainLink.class.getName(), match.chain.get(0)); + assertTrue(match.targetTag > 0, "targetTag should be a valid, non-zero JVMTI tag"); + assertTrue(match.depth >= 0, "depth should be a non-negative hop count"); + assertTrue(!gcRootHolder.isEmpty()); // keeps every allocated ChainLink reachable until here + } finally { + gcRootHolder = null; // shared-JVM hygiene - see the static holder's own comment + currentBatch = null; + // Shut the persistent allocator thread down before leaving: this is a + // shared, no-forkEvery test JVM, and a leftover allocation thread would + // keep allocating (and keep its tid trend alive) into every later test. + allocator.interrupt(); + allocator.join(1000); + Files.deleteIfExists(scratchDumpPath); + } + } + + /** + * A second, independent end-to-end scenario for the same target-selection mechanism as + * {@link #shouldReconstructReferrerChainToGcRoot()}, using a shape much closer to a real-world + * leak than that method's hand-rolled {@link ChainLink} list: an ever-growing, never-evicted + * {@link java.util.HashMap}-backed cache - a very common real leak pattern - holding + * {@link CachedPayload} values. Beyond confirming a chain fires for the leaking value type, + * this also asserts the reconstructed chain actually threads back through + * {@code java.util.HashMap}'s own internal storage ({@code HashMap$Node}), i.e. that the + * mechanism correctly walks a real JDK collection's internals rather than only ever having + * been proven against a single purpose-built linked fixture. + * + *

    Why {@code cache} is a local variable, not a {@code static} field (found the hard + * way): {@code ReferenceChainTracker::expandFrontier()} (referenceChains.cpp) freezes + * whatever children it observes for a node the moment that node is expanded - it never + * re-examines an already-{@code EXPANDED} entry for newly-added children, regardless of how + * many times root/stack-ref enumeration itself reruns over the search's lifetime + * ({@code runPassManualWalk()}'s own comment). A {@code static} field becomes a root at + * class-load time, essentially guaranteeing some early pass catches it (and marks it + * {@code EXPANDED}) while still empty - permanently blocking discovery of anything added to it + * afterward, no matter how many rounds run. A local variable created fresh at the top of this + * method, immediately followed by round-1 allocation, gives root enumeration a real chance to + * catch it only after it already holds live data - mirroring + * {@link #shouldReconstructReferrerChainToGcRoot()}'s {@code gcRootHolder}, which works for + * exactly this reason (a local {@code ArrayList}, not a {@code static} one). + * + *

    See this class's own header comment for why this runs as {@code @Order(2)}, before + * {@link #shouldReportAbandonedSearchOnTinyFrontierCap()}, and for why its relative order + * against {@link #shouldReconstructReferrerChainToGcRoot()} does not itself matter. + */ + @Test + @Order(2) + public void shouldReconstructReferrerChainThroughUnboundedCacheLeak() throws Exception { + // Seed cache with one live entry *before* the resets below, mirroring + // shouldReconstructReferrerChainToGcRoot()'s own gcRootHolder seeding (see that method's + // comment): HashMap's backing table also starts out as a shared, lazily-replaced empty + // sentinel (only allocated on the first put()), so the same one-shot-BFS-freeze hazard + // documented there - a concurrent GC-triggered walk expanding this node's children while the + // backing table is still the empty sentinel, permanently freezing an empty children set - + // applies here too. Seeding one entry first means the backing table is never empty at any + // point after the search (re)starts. + Map cache = new HashMap<>(); + cache.put("cache-leak-seed", new CachedPayload("cache-leak-seed")); + if ("debug".equals(System.getProperty("ddprof_test.config"))) { + // Mirrors shouldReconstructReferrerChainToGcRoot()'s own reset (see that method's comment): + // the natural restartSearch() cycle (shouldRunPass(), referenceChains.cpp) that would + // otherwise give this method its own fresh root walk once shouldReconstructReferrerChainToGcRoot()'s + // own ChainLink candidate is found and its tags released is gated on cadence/pacing budget + // (canAffordNewSearch()) - not guaranteed to fire again before this method's own round/retry + // budget runs out. Force a genuine fresh search here too, rather than depend on that timing, + // so cache (below) is guaranteed reachable by a real walk regardless of it. Debug-only: this + // native seam does not exist in a release build. + // + // resetReferenceChainSearchForTest0() only resets ReferenceChainTracker's own search state + // (frontier/tags/search-progress fields); it does not touch LivenessTracker's per-klass + // population-history rings, which selectLeakCandidates() consults independently to decide + // which klass is "trending". Without also resetting those, a klass ChainLink already + // accumulated a positive slope for during shouldReconstructReferrerChainToGcRoot() can still + // outrank CachedPayload as the leak candidate here, so the fresh search below ends up + // reconstructing ChainLink's chain again instead of CachedPayload's. + JavaProfiler.resetKlassPopulationForTest0(); + JavaProfiler.resetReferenceChainSearchForTest0(); + } + Path scratchDumpPath = Paths.get("referencechains-cache-leak-scratch.jfr"); + try { + // Same round-growth/retry shape as shouldReconstructReferrerChainToGcRoot() - see that + // method's own comment for why the loop self-adjusts rather than committing to a fixed + // round count, why per-round scale is sized against ObjectSampler's real 256KiB sampling + // floor rather than this test's own "memory=64" request, why totalRounds is capped at + // 16 rather than a larger margin above the 10-round minimum (shared-fork heap headroom), + // and why per-round growth itself is clamped to round 10 (Math.min(round, 10) below). + ReferenceChainAssertions.JfrChainMatch match = null; + boolean seededTestKlassTrend = false; + int totalRounds = 16; + + // Pre-generate every key this loop will ever need, up front, rather than concatenating a + // fresh String on every put() below. CachedPayload's own class comment already documents + // why this fixture uses plain long fields instead of a byte[] (an earlier version's byte[] + // field stole every allocation-sampling hit from CachedPayload itself) - the key String + // (plus its own backing byte[]) is the same kind of same-instant, size-weighted-sampler + // competitor, just a *separate* object instead of a field. Generating all keys in one + // batch before the round loop starts lets their own klass_population trend flatten out + // (selectLeakCandidates() needs KLASS_POPULATION_MIN_FILL_FOR_TREND=10 *new* samples, not + // just a nonzero ring) well before CachedPayload's own ring starts growing, so the sampler + // has nothing else in flight to compete with once round 1 begins. + int maxEntries = 0; + for (int round = 1; round <= totalRounds; round++) { + maxEntries += Math.min(round, 10) * 600; + } + String[] keys = new String[maxEntries]; + for (int i = 0; i < keys.length; i++) { + keys[i] = "leak-" + i; + } + + int nextKey = 0; + for (int round = 1; round <= totalRounds && match == null; round++) { + // Matches shouldReconstructReferrerChainToGcRoot()'s own *600 growth rate (not *300, this + // method's previous value) - CachedPayload and ChainLink are near-identical in size (same + // padding fields; CachedPayload is actually slightly smaller, missing ChainLink's `next` + // reference), so halving the growth rate here bought no headroom and instead just let + // CachedPayload's own population ring stall short of KLASS_POPULATION_MIN_FILL_FOR_TREND + // within the same totalRounds budget both methods share. + int newEntries = Math.min(round, 10) * 600; + int keyOffset = nextKey; + Thread allocator = new Thread(() -> { + for (int i = 0; i < newEntries; i++) { + String key = keys[keyOffset + i]; + cache.put(key, new CachedPayload(key)); + } + }); + allocator.start(); + allocator.join(); + nextKey += newEntries; + System.gc(); + dump(scratchDumpPath); + JfrEvents events1 = verifyEvents(scratchDumpPath, "datadog.ReferenceChain", false); + match = ReferenceChainAssertions.findMatchForClass(events1, CachedPayload.class); + + if (match == null && "debug".equals(System.getProperty("ddprof_test.config"))) { + // Same deterministic short-circuit as shouldReconstructReferrerChainToGcRoot() - see + // that method's own comment for the full mechanism (why seeding + // SEED_EPOCHS_FOR_HYSTERESIS epochs is what actually authorizes the background BFS + // thread's next real pass, and why this method's own pollReferenceChainTargets0() call + // right below is therefore best-effort rather than something the seed alone + // guarantees on this call). Seed exactly once - see that method's own comment for why + // reseeding the same ramp on a later round would corrupt the ring into a non-monotonic + // sawtooth and destroy the positive-slope signal instead of just re-establishing it. + // Representative BEFORE seeding - same real-id aliasing rationale as + // shouldReconstructReferrerChainToGcRoot()'s own comment above. + JavaProfiler.setKlassPopulationRepresentativeForTest0(CACHED_PAYLOAD_TEST_KLASS_ID, cache.get(keys[0])); + if (!seededTestKlassTrend) { + // Per-(klass, tid) qualification seeds - see LeakingCacheScenario's + // own seeding block for why this thread's tid (the representative's + // creator) is the qualifying one for the one-cohort-per-round shape. + int leakTid = JavaProfiler.getTid(); + for (int epoch = 1; epoch <= SEED_EPOCHS_FOR_HYSTERESIS; epoch++) { + JavaProfiler.seedKlassPopulationSample0(CACHED_PAYLOAD_TEST_KLASS_ID, epoch * 10, epoch); + JavaProfiler.seedTidTrendSample0(CACHED_PAYLOAD_TEST_KLASS_ID, leakTid, epoch * 3, epoch); + } + seededTestKlassTrend = true; + } + JavaProfiler.pollReferenceChainTargets0(); + dump(scratchDumpPath); + JfrEvents events2 = verifyEvents(scratchDumpPath, "datadog.ReferenceChain", false); + match = ReferenceChainAssertions.findMatchForClass(events2, CachedPayload.class); + } + + if (match == null) { + // Same lock-contention race shouldReconstructReferrerChainToGcRoot()'s own comment + // describes - a quiet retry gives pollWatchedTargets() a window to win a _locks[] + // slot against Profiler::dump()'s own exclusive hold. + Thread.sleep(300); + dump(scratchDumpPath); + JfrEvents events3 = verifyEvents(scratchDumpPath, "datadog.ReferenceChain", false); + match = ReferenceChainAssertions.findMatchForClass(events3, CachedPayload.class); + } + } + + for (int attempt = 0; match == null && attempt < 5; attempt++) { + Thread.sleep(1000); + dump(scratchDumpPath); + JfrEvents events4 = verifyEvents(scratchDumpPath, "datadog.ReferenceChain", false); + match = ReferenceChainAssertions.findMatchForClass(events4, CachedPayload.class); + } + + assertNotNull(match, + "Never observed a datadog.ReferenceChain event whose chain[0] is " + CachedPayload.class + + " after " + cache.size() + " cached entries across up to " + totalRounds + + " population-growth rounds plus a grace period"); + assertEquals(CachedPayload.class.getName(), match.chain.get(0)); + assertTrue(match.targetTag > 0, "targetTag should be a valid, non-zero JVMTI tag"); + assertTrue(match.depth >= 0, "depth should be a non-negative hop count"); + + // The point of this test over shouldReconstructReferrerChainToGcRoot(): confirm the walk + // actually passed through the cache's own internal storage, not some other, coincidental + // retainer - cache is the only thing keeping any CachedPayload instance reachable. + boolean sawHashMapInternals = false; + for (String type : match.chain) { + if (type.startsWith("java.util.HashMap")) { + sawHashMapInternals = true; + break; + } + } + assertTrue(sawHashMapInternals, + "Expected the reconstructed chain to pass through java.util.HashMap's own internal " + + "storage (cache is a HashMap) - chain was: " + match.chain); + assertTrue(!cache.isEmpty()); // keeps every cached CachedPayload reachable until here + } finally { + Files.deleteIfExists(scratchDumpPath); + } + } + + /** + * This test's stated abandonment-path scenario: an artificially tiny frontier cap + * ({@code getProfilerCommand()}'s {@code framecap=1} for this test method, see that method's own + * comment for why {@code budget} must stay larger than {@code framecap}) makes the very first + * BFS pass hit the frontier cap once a second root-referenced object is discovered, so + * {@code runPass()} (referenceChains.cpp) abandons the search rather than silently truncating it + * (Termination section, doc/architecture/LiveHeapReferenceChains.md) - or, if this method's own + * {@code framecap=1} lost the race to size the shared {@code FrontierTable} (see + * {@code getProfilerCommand()}'s own comment on this method), its {@code ttl=100} fallback + * abandons the search instead once enough wall-clock time has passed, which by the time this + * method runs it always has. Either path exercises the same {@code buildAbandonedEvent()} / + * {@code Profiler::writeReferenceChainAbandoned()} / {@code FlightRecorder::recordReferenceChainAbandoned()} + * write into the recording that {@code Profiler::dump()} triggers - this method only asserts that + * a {@code datadog.ReferenceChainAbandoned} event exists, not which reason produced it. + * + *

    {@code @Order(3)}: this permanently exhausts the process-wide {@code ReferenceChainTracker} + * singleton's one-and-only search (see this class's header comment), so it must run after + * {@link #shouldReconstructReferrerChainToGcRoot()} and + * {@link #shouldReconstructReferrerChainThroughUnboundedCacheLeak()}. It used to rely on being + * the last method in source order plus JUnit's default (unannotated methods sort after + * {@code @Order}-annotated ones) for that - which happened to guarantee run order, but not the + * search *state* this method's own comment above presumes: if either earlier test's own + * organic GC/allocation-sampling trend never fired in time (observed in practice - real + * flakiness, not this test's fault), the shared search can still be {@code RUNNING} with a + * large, non-fresh frontier when this method starts, instead of the small, just-abandoned one + * its own {@code framecap=1}/{@code ttl=100} scenario assumes. Calling + * {@code resetReferenceChainSearchForTest0()} here, exactly as both earlier tests already do for + * their own scenarios, makes this method's own precondition (a fresh search) something it + * establishes itself rather than something it presumes a prior test left behind. Debug-only: + * this native seam does not exist in a release build. + * + *

    Why this method also seeds a klass population trend, despite requesting no + * {@code generations=true} of its own (found the hard way: {@code resetReferenceChainSearchForTest0()} + * alone was not enough - the search's first pass never ran at all, so it could never reach + * {@code SearchState::ABANDONED} for {@code dump()} to observe): {@code LivenessTracker::_gc_generations} + * (livenessTracker.cpp's {@code initialize()}) is only ever updated when a {@code start()} command + * actually requests liveness tracking - a bare {@code referencechains=true:...} command like this + * method's own does not call {@code LivenessTracker::start()} at all, so + * {@code gcGenerationsEnabled()} stays stuck at whatever the two {@code generations=true} tests + * above last left it (confirmed directly: this method's own native trace shows + * {@code gc_generations=1}). {@code ReferenceChainTracker::hasLeakSignal()}, which + * {@code shouldRunPass()} consults for even this search's very first pass, therefore falls + * through to {@code selectLeakCandidates()} exactly as the two tests above do - with nothing + * left over from {@link #shouldReconstructReferrerChainThroughUnboundedCacheLeak()}'s own + * {@code resetKlassPopulationForTest0()} call to satisfy it. Seeding + * {@code SEED_EPOCHS_FOR_HYSTERESIS} epochs for a dedicated klass id (see + * {@code SEED_EPOCHS_FOR_HYSTERESIS}'s own comment for the full derivation) clears that gate the + * same way it does there - this method never sets a representative or polls for it, since it has + * no reconstructed chain of its own to prove; {@code selectLeakCandidates()} + * (livenessTracker.cpp) counts a qualifying entry toward {@code hasLeakSignal()}'s candidate + * count regardless of whether its representative was ever set, and + * {@code resolveCandidateRepresentative()} is null-safe against one that wasn't. + * + *

    Why the seeding is preceded by a {@code System.gc()} and a throwaway {@code dump()} + * (found the hard way, twice: a bare {@code dump()} with no preceding GC still lost the + * race): {@code cleanup_table()} (livenessTracker.cpp) compares + * {@code Profiler::classMap()->generation()} against its own cached copy and, on a mismatch - + * which this method's own fresh {@code start()} causes, by clearing that map - wipes + * {@code _klass_population} back to empty before folding anything else, exactly once per + * mismatch; but that comparison sits *after* an early-return guard + * (`{@code if (!is_epoch_owner && !forced) return;}`) that a {@code dump()}-triggered call + * (always {@code forced=false}) only clears when a genuinely new, not-yet-processed GC epoch + * is pending - a {@code dump()} with no GC since the last {@code cleanup_table()} pass never + * reaches the mismatch check at all. The two methods above never hit any of this because their + * own seed loop runs only after their round-1 {@code System.gc()}+{@code dump()} pair has + * already triggered that one-time wipe-and-resync as a side effect of their own allocation + * loop; seeding before any GC/dump of this method's own would instead leave the seed to be + * discarded by the *next* call that does clear the guard - this class's own + * {@code maybeForceCleanup()} tick (called once per {@code ReferenceChainTracker::threadLoop()} + * wake, {@code forced=true} unconditionally clears the guard), which fires on this method's own + * 3 {@code System.gc()} calls below regardless. The explicit {@code System.gc()}+{@code dump()} + * pair here forces that one-time wipe-and-resync to happen now, before seeding, exactly like + * the two methods above get for free from their own round-1 allocation loop. + */ + @Test + @Order(3) + public void shouldReportAbandonedSearchOnTinyFrontierCap() throws Exception { + if ("debug".equals(System.getProperty("ddprof_test.config"))) { + // See this method's own javadoc ("Why this method also seeds a klass population trend...") + // for why this is needed even though this method requests no generations=true of its own. + // + // A System.gc() + throwaway dump() first - see this method's own javadoc ("Why the seeding + // is preceded by...") for why *both* are needed: a dump() alone triggers cleanup_table() + // with forced=false, which only clears the class-map-generation-mismatch guard (and + // therefore only wipes/resyncs _klass_population) when a new GC epoch is actually pending - + // this System.gc() is what makes that true for this call, rather than leaving it to chance + // (or to this method's own 3 System.gc() calls below, which run after the seed and would + // let maybeForceCleanup()'s forced=true tick discard it first). + System.gc(); + Path warmupDumpPath = Paths.get("referencechains-abandoned-test-warmup.jfr"); + try { + dump(warmupDumpPath); + } finally { + Files.deleteIfExists(warmupDumpPath); + } + JavaProfiler.resetKlassPopulationForTest0(); + // Per-(klass, tid) qualification seeds - gate-only usage (this test + // drives the search to ABANDONED and asserts on the abandon marker, no + // tagged instance needs to match), so the tid is this thread's real one + // but any distinct tid would do. + int abandonTid = JavaProfiler.getTid(); + for (int epoch = 1; epoch <= SEED_EPOCHS_FOR_HYSTERESIS; epoch++) { + JavaProfiler.seedKlassPopulationSample0(ABANDON_TEST_KLASS_ID, epoch * 10, epoch); + JavaProfiler.seedTidTrendSample0(ABANDON_TEST_KLASS_ID, abandonTid, epoch * 3, epoch); + } + JavaProfiler.resetReferenceChainSearchForTest0(); + } + List gcRootHolder = new ArrayList<>(); + gcRootHolder.add(new ChainLink("middle", new ChainLink("leaf"))); + + // GarbageCollectionFinish (onGCFinish(), referenceChains.cpp) wakes the BFS thread + // early, but that wakeup can race the thread's own startup (VM::attachThread() + // completing before its first OS::sleep() call) and be missed. Don't rely on the + // signal alone: sleep comfortably past ReferenceChainTracker::PASS_CADENCE_NS (1s, + // referenceChains.h) too, so the thread's fixed-cadence fallback trigger + // (shouldRunPass()) guarantees at least one pass runs regardless of that race. + for (int i = 0; i < 3; i++) { + System.gc(); + Thread.sleep(100); + } + Thread.sleep(1500); + + Path dumpPath = Paths.get("referencechains-abandoned-test.jfr"); + try { + // Even with the seeding above clearing hasLeakSignal()'s gate so the search's first pass is + // actually allowed to start (this method's own javadoc), that pass still has to be woken by + // the BFS thread and still has to actually reach the frontier cap/ttl cutoff, inside whatever + // real wall-clock time this shared, no-forkEvery test JVM's own scheduling happens to grant + // it (this class's own header comment) - a single fixed-length sleep-then-dump-then-assert, + // with no room to retry, previously failed outright whenever that one attempt landed a beat + // early. Retry across a few more dumps, spaced far enough apart to tolerate a slow + // BFS-thread wake or a walk over a larger-than-usual inherited heap, before concluding the + // search genuinely never abandoned - mirroring the bounded retry loops + // shouldReconstructReferrerChainToGcRoot()/shouldReconstructReferrerChainThroughUnboundedCacheLeak() + // already use above. + JfrEvents abandoned = null; + for (int attempt = 0; attempt < 8; attempt++) { + dump(dumpPath); + abandoned = verifyEvents(dumpPath, "datadog.ReferenceChainAbandoned", false); + if (abandoned.hasItems()) { + break; + } + Thread.sleep(1000); + } + assertTrue(abandoned != null && abandoned.hasItems(), + "Expected at least one datadog.ReferenceChainAbandoned event after 3 GCs, an initial " + + "1500ms grace period, and 8 retries 1s apart"); + } finally { + Files.deleteIfExists(dumpPath); + } + assertTrue(!gcRootHolder.isEmpty()); // keeps gcRootHolder reachable until the dump above + } + + /** + * Referrer-type fixture shared by both the success-path and abandonment-path tests. The 32 + * {@code long} fields below exist purely so {@link #shouldReconstructReferrerChainToGcRoot()} + * needs far fewer instances to allocate a given number of megabytes of ChainLink - deliberately + * plain fields, not a nested array: an array field would be a *second*, separate heap + * allocation, and the allocation-sampling interval that method's comment describes picks + * whichever allocation happens to cross its byte threshold size-weighted, so a same-instance + * companion array would take sampling attention away from ChainLink itself rather than adding + * to it. Plays no role in {@link #shouldReportAbandonedSearchOnTinyFrontierCap()}'s tiny, + * two-object fixture beyond trivially increasing its size. + */ + private static final class ChainLink { + final String name; + final Object next; + long p0, p1, p2, p3, p4, p5, p6, p7, p8, p9, p10, p11, p12, p13, p14, p15; + long p16, p17, p18, p19, p20, p21, p22, p23, p24, p25, p26, p27, p28, p29, p30, p31; + + ChainLink(String name) { + this(name, null); + } + + ChainLink(String name, Object next) { + this.name = name; + this.next = next; + } + } + + /** + * Realistic-leak fixture for {@link #shouldReconstructReferrerChainThroughUnboundedCacheLeak()}: + * models the single most common real-world leak shape this mechanism is meant to catch - an + * unbounded, never-evicted cache - as opposed to {@link ChainLink}'s hand-rolled linked list. + * The 32 {@code long} fields exist purely so a given number of megabytes needs far fewer + * entries to reach {@code ObjectSampler}'s real 256KiB sampling floor - deliberately plain + * fields, not a nested array, for exactly the reason {@link ChainLink}'s own comment already + * documents: a same-instance companion array would be a *second*, separate heap allocation + * that the size-weighted allocation sampler would compete for, taking sampling attention away + * from {@code CachedPayload} itself (an earlier version of this fixture used a {@code byte[]} + * field and never observed a single {@code CachedPayload} sample as a result - the sampler was + * catching the array instead). + */ + private static final class CachedPayload { + final String key; + long p0, p1, p2, p3, p4, p5, p6, p7, p8, p9, p10, p11, p12, p13, p14, p15; + long p16, p17, p18, p19, p20, p21, p22, p23, p24, p25, p26, p27, p28, p29, p30, p31; + + CachedPayload(String key) { + this.key = key; + } + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/StaticFieldGrowingCollectionScenario.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/StaticFieldGrowingCollectionScenario.java new file mode 100644 index 0000000000..a1251f5fed --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/StaticFieldGrowingCollectionScenario.java @@ -0,0 +1,278 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.referencechains; + +import com.datadoghq.profiler.JavaProfiler; +import org.openjdk.jmc.common.item.IItemCollection; +import org.openjdk.jmc.common.item.ItemFilters; +import org.openjdk.jmc.flightrecorder.JfrLoaderToolkit; + +import java.io.InputStream; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.ArrayList; +import java.util.Collections; +import java.util.List; + +/** + * Mimics the real leak shape found in the {@code prof-analyzer-hotdog-jb} pod's leak generator + * ({@code ProfileAnalyzer}'s {@code LEAK_BUFFER}, profiling-backend repo, branch-local diff): a + * {@code static final List} field whose *instance* is admitted exactly once by + * {@code admitStaticFieldRoots()} (referenceChains.cpp) when the owning class first loads, and + * which is then only ever mutated in place (elements appended) - never reassigned - for the rest + * of the process's life. This is the "new elements added to an already-discovered collection" + * shape, as opposed to {@code LeakingCacheScenario}'s local-variable cache (a stack-local GC + * root, not a static field) - see this scenario's own use to confirm whether + * {@code collectStaleExpandedEntriesForRotation()}'s rotation mechanism actually re-expands + * {@link #LEAK_BUFFER} to pick up elements added after its one-time static-field sweep. + * + *

    Same seed-before-start rationale as {@code LeakingCacheScenario}'s own comment: {@link + * #LEAK_BUFFER} is seeded with a few chunks before the profiler (and therefore {@code + * ReferenceChainTracker}'s BFS thread) is even started, so the process's one-shot root-seeded + * walk can never catch it empty. + */ +public final class StaticFieldGrowingCollectionScenario { + private StaticFieldGrowingCollectionScenario() {} + + /** + * Mirrors production's {@code ProfileAnalyzer.LEAK_BUFFER}: a static field holding a + * {@code synchronizedList}, appended to (never reassigned) for the process's whole + * life. + */ + static final List LEAK_BUFFER = Collections.synchronizedList(new ArrayList<>()); + + // Arbitrary, scenario-chosen klass id - see LeakingCacheScenario.CACHED_PAYLOAD_TEST_KLASS_ID's + // own comment; distinct from that scenario's id since both may run in the same suite (each in + // its own separate child JVM/LivenessTracker population table, but kept distinct regardless). + private static final int LEAK_BUFFER_TEST_KLASS_ID = 987302; + + // See LeakingCacheScenario.SEED_EPOCHS_FOR_HYSTERESIS's own comment for the full derivation: + // LivenessTracker's hysteresis gate needs this many qualifying epochs before + // selectLeakCandidates() reports a candidate at all. + private static final int SEED_EPOCHS_FOR_HYSTERESIS = 15; + + // Clears ObjectSampler's real sampling floor on its own - unlike LeakingCacheScenario's + // CachedPayload (long fields only, no byte[]), this scenario's leaked object *is* the byte[] + // itself, so there is no second, competing companion allocation. + private static final int CHUNK_BYTES = 300_000; + + private static final int SEED_CHUNKS = 10; + + /** Printed to stdout, followed by the matched leaf class's name, on success. */ + public static final String FOUND_MARKER = "[chain-found] "; + + /** Printed to stdout (with no class name suffix) if no match was ever observed. */ + public static final String NOT_FOUND_MARKER = "[chain-not-found]"; + + /** + * Seeds {@link #LEAK_BUFFER}, starts the profiler, then keeps appending new {@code byte[]} + * chunks to the same list instance - the exact shape implicated in the live hotdog-pod stall - + * watching specifically for a chain to a chunk appended *after* the static field's one-time + * sweep, until a {@code datadog.ReferenceChain} event for it appears or {@code totalRounds} is + * exhausted. + */ + public static void run(JavaProfiler profiler, String startCommand, Path scratchDumpPath) throws Exception { + for (int i = 0; i < SEED_CHUNKS; i++) { + LEAK_BUFFER.add(new byte[CHUNK_BYTES]); + } + if (startCommand != null && !startCommand.isEmpty()) { + profiler.execute(startCommand); + } + + boolean debugBuild = "debug".equals(System.getProperty("ddprof_test.config")); + int hysteresisEpoch = 0; + // Declared at method scope: the per-round maintenance seeding below (in + // its own debugBuild block) reuses this tid. + int leakTid = 0; + // Snapshotted BEFORE seeding anything, while the BFS thread's search has not run a single + // pass yet (referenceChainPassesRunForTest0() reads 0 here) - see the wait loop below for why + // this exact value, not just "nonzero", is what the wait needs to advance past. + int initialPasses = debugBuild ? JavaProfiler.referenceChainPassesRunForTest0() : -1; + + if (debugBuild) { + // Seed enough hysteresis-qualifying epochs to unlock ReferenceChainTracker::hasLeakSignal()/ + // shouldRunPass() - WITHOUT wiring in a representative object yet (setKlassPopulationRepresentativeForTest0 + // is not called here). hasLeakSignal() only asks selectLeakCandidates() for a nonzero count - + // it never reads/dereferences the representative field - so this step's only job is + // authorizing the BFS thread's first pass to actually start. Doing this BEFORE waiting on + // passesRun() advancing below is load-bearing, found the hard way: a pass only ever runs + // once shouldRunPass() sees a leak signal - waiting on passesRun() before seeding anything is + // a deadlock, not just a slow path. + // Per-(klass, tid) qualification seeds (see LeakTagCorrelationScenario's own + // seeding block for the rationale): the qualifying tid must be the thread that + // allocates the tracked instances - this thread allocates every LEAK_BUFFER + // chunk, including the lateChunk this scenario watches. + leakTid = JavaProfiler.getTid(); + for (int epoch = 1; epoch <= SEED_EPOCHS_FOR_HYSTERESIS; epoch++) { + JavaProfiler.seedKlassPopulationSample0(LEAK_BUFFER_TEST_KLASS_ID, epoch * 10, epoch); + JavaProfiler.seedTidTrendSample0(LEAK_BUFFER_TEST_KLASS_ID, leakTid, epoch * 3, epoch); + } + hysteresisEpoch = SEED_EPOCHS_FOR_HYSTERESIS; + } + + // Wait for the BFS thread to fully complete its own first pass - not just + // admitStaticFieldRoots()'s one-time sweep of LEAK_BUFFER's *current* contents (an earlier, + // narrower version of this wait checked exactly that, via a since-removed + // hasCompletedStaticFieldSweepForTest0() seam), but the WHOLE pass, static-field sweep and + // expandFrontier()/collectStaleExpandedEntriesForRotation() alike - strictly before this + // scenario creates the one chunk whose chain it is actually going to watch for. This is the + // entire point of this scenario: proving that an element appended to a static field's + // collection AFTER that field's List is already ReferenceChainTracker-EXPANDED gets discovered + // via a LATER re-expansion of that stale entry, not via the very first pass that admitted it. + // Waiting on only the static-field sub-step (found the hard way, by actually instrumenting + // runPassManualWalk() and reading cacheResolvedChain()'s own timestamp against it) is not + // enough: admitStaticFieldRoots() and expandFrontier() are two steps of the SAME runPass() + // call, so Java can observe the former's completion while the latter - which is what actually + // marks LEAK_BUFFER's List entry EXPANDED - is still in flight, still racing lateChunk's + // creation exactly the way the sleep-based version this replaced did. Waiting for + // referenceChainPassesRunForTest0() to advance past its pre-seeding snapshot instead only + // becomes true once that ENTIRE pass (both steps) has finished. Non-debug builds have no such + // seam and fall back to a generous sleep - acceptable there because the non-debug path never + // pins a specific representative anyway (see the debugBuild branch below): it just waits on + // real, much-slower allocation sampling to notice the growth on its own, so a few extra + // seconds of margin either way changes nothing about what the eventual match would prove. + if (debugBuild) { + boolean passCompleted = false; + // admitStaticFieldRoots() sweeps every loaded class in one FollowReferences call - a cold + // external JVM's full JUnit/JMC/Gradle-worker classpath (thousands of classes) can need + // several retries against pause-time-SLO pacing before one completes without truncating (see + // that method's own header comment on retry-from-scratch-but-cheap-on-already-admitted) - + // generous margin for the same reason ExternalProcessReferenceChainTest's own budget/ + // pausetarget are raised this far above the in-process test's defaults. + for (int i = 0; i < 300 && !passCompleted; i++) { + passCompleted = JavaProfiler.referenceChainPassesRunForTest0() != initialPasses; + if (!passCompleted) { + Thread.sleep(100); + } + } + if (!passCompleted) { + throw new IllegalStateException( + "ReferenceChainTracker never completed its first pass within 30s - cannot safely " + + "create the chunk this scenario is supposed to discover only via a later " + + "re-expansion of an already-EXPANDED static-field entry"); + } + } else { + Thread.sleep(3000); + } + // Snapshotted right after the wait above confirms the first pass has fully completed - any + // later change (advance, or reset-to-lower via a restart) proves a walk that started strictly + // after lateChunk existed. + int passesRunBeforeLateChunk = debugBuild ? JavaProfiler.referenceChainPassesRunForTest0() : -1; + byte[] lateChunk = new byte[CHUNK_BYTES]; + LEAK_BUFFER.add(lateChunk); + + // From here on, seeded incrementally, one epoch per round below, rather than as a single + // upfront burst - found the hard way: profiler.execute()'s start() resets + // Profiler::classMap()'s generation synchronously, but LivenessTracker::cleanup_table() only + // observes that new generation (and updates its own _last_class_map_generation) lazily, on the + // BFS thread's own first tick. A burst seeded entirely before that first tick lands in + // _klass_population, then gets wiped wholesale the moment that tick's class-map-generation + // mismatch fires (cleanup_table()'s own comment, livenessTracker.cpp). Seeding one fresh epoch + // per round instead is self-healing exactly the way the real leak this scenario mirrors is: + // LEAK_BUFFER keeps growing every round regardless of any one-time startup housekeeping, so + // however many rounds a wipe costs, the trend simply resumes accumulating from the next + // round's seed call. (Waiting for the first pass to fully complete above already makes this + // particular race very unlikely to matter by this point, but there is no reason to give up the + // self-healing margin now that it costs nothing.) + ReferenceChainAssertions.ChainMatch match = null; + int totalRounds = 25; + for (int round = 1; round <= totalRounds && match == null; round++) { + // Continues growing LEAK_BUFFER every round after lateChunk - mirrors the real leak's own + // continuous growth - but lateChunk itself (created above, strictly after the first sweep) + // stays the fixed target this scenario watches for throughout. + LEAK_BUFFER.add(new byte[CHUNK_BYTES]); + System.gc(); + + if (match == null && debugBuild) { + // Seeding/polling every round unconditionally (not gated on passesRun() having advanced) + // is load-bearing, found the hard way: gating this on the tracker's OWN progress creates a + // deadlock - ReferenceChainTracker::hasLeakSignal() (which authorizes both a search's first + // pass and any later restart once the search reaches a terminal state) reads + // LivenessTracker::selectLeakCandidates(), which needs a continuously-refreshed population + // trend to keep reporting this candidate; stopping that refresh mid-search (to wait on the + // search's own progress) can stall the search itself with no path back, since nothing else + // keeps re-authorizing it. This scenario's target - a growing static-field collection - + // legitimately produces a continuous real signal in production; seeding continuously here + // is what actually mirrors that, not an artifact to gate away. + // + // Same debug-only seeded-representative short-circuit as LeakingCacheScenario's own - + // watches lateChunk specifically (added after the static field's first sweep), not + // whatever the real allocation sampler happens to pick. See SEED_EPOCHS_FOR_HYSTERESIS's + // own comment for the minimum epoch count this needs - kept uncapped here (not stopped + // once that minimum is reached) specifically so a wipe that costs an early round or two + // still leaves enough remaining rounds to reach it, rather than exhausting a fixed budget + // of calls before ever getting there. + hysteresisEpoch++; + JavaProfiler.seedKlassPopulationSample0( + LEAK_BUFFER_TEST_KLASS_ID, hysteresisEpoch * 10, hysteresisEpoch); + JavaProfiler.seedTidTrendSample0( + LEAK_BUFFER_TEST_KLASS_ID, leakTid, hysteresisEpoch * 3, hysteresisEpoch); + JavaProfiler.setKlassPopulationRepresentativeForTest0(LEAK_BUFFER_TEST_KLASS_ID, lateChunk); + JavaProfiler.pollReferenceChainTargets0(); + + // Only TRUST a match once passesRun() has genuinely changed since the snapshot taken right + // before lateChunk was created - either advanced (the same, never-restarted search ran a + // later pass) or reset to a lower value (restartSearch() ran, which only ever happens on a + // threadLoop tick strictly after that snapshot, since time only moves forward). Either way + // proves the match came from a walk that started after lateChunk existed, ruling out the + // one race this whole seam exists to rule out: the very first, still-in-flight pass finding + // lateChunk by accident because admitStaticFieldRoots() and expandFrontier() are two steps + // of that SAME call, and Java can observe the former's completion before the latter runs. + if (JavaProfiler.referenceChainPassesRunForTest0() != passesRunBeforeLateChunk) { + profiler.dump(scratchDumpPath); + match = findMatch(scratchDumpPath); + } + } + + if (match == null && !debugBuild) { + profiler.dump(scratchDumpPath); + match = findMatch(scratchDumpPath); + } + if (match == null && !debugBuild) { + Thread.sleep(300); + profiler.dump(scratchDumpPath); + match = findMatch(scratchDumpPath); + } + } + + // A cold external JVM's one-shot root-seeded walk over its whole reachable graph (full + // JUnit/JMC/Gradle-worker classpath) takes far longer than this scenario's own round loop - + // found the hard way: a first attempt at this scenario's tail wait (5 x 1s) gave up and + // printed NOT_FOUND_MARKER mere moments before the tracker's own runPass() actually + // completed (searchState=COMPLETED) and successfully built the chain. Mirrors + // ExternalProcessReferenceChainTest's own comment on why budget/pausetarget are raised this + // far for exactly this reason. + for (int attempt = 0; match == null && attempt < 40; attempt++) { + Thread.sleep(1500); + // Same passesRun()-advanced check as the round loop above (debug builds only) - by this + // point in a real run it has virtually always already flipped, but there is no reason to + // drop the guarantee here just because the round loop above didn't need it. + if (debugBuild && JavaProfiler.referenceChainPassesRunForTest0() == passesRunBeforeLateChunk) { + continue; + } + profiler.dump(scratchDumpPath); + match = findMatch(scratchDumpPath); + } + + if (match == null) { + System.out.println(NOT_FOUND_MARKER); + return; + } + System.out.println(FOUND_MARKER + match.chain.get(0).getFullName()); + } + + private static ReferenceChainAssertions.ChainMatch findMatch(Path scratchDumpPath) throws Exception { + if (!Files.exists(scratchDumpPath)) { + return null; + } + IItemCollection events; + try (InputStream in = Files.newInputStream(scratchDumpPath)) { + events = JfrLoaderToolkit.loadEvents(in); + } + return ReferenceChainAssertions.findMatchForClass( + events.apply(ItemFilters.type("datadog.ReferenceChain")), byte[].class); + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakReferenceChainTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakReferenceChainTest.java new file mode 100644 index 0000000000..1bf3bf58cd --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakReferenceChainTest.java @@ -0,0 +1,160 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.referencechains; + +import com.datadoghq.profiler.AbstractProcessProfilerTest; +import com.datadoghq.profiler.Platform; +import org.junit.jupiter.api.Tag; +import org.junit.jupiter.api.Test; + +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.ArrayDeque; +import java.util.ArrayList; +import java.util.Collections; +import java.util.Deque; +import java.util.List; +import java.util.concurrent.atomic.AtomicReference; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.junit.jupiter.api.Assertions.fail; +import static org.junit.jupiter.api.Assumptions.assumeFalse; + +/** + * The thread-local taxonomy of the leak-tag correlation use case, end-to-end + * in a separate child JVM (same separate-process rationale as + * {@code LeakTagCorrelationReferenceChainTest}'s own class comment): a growing + * {@code byte[]} collection held ONLY through the leaking thread's + * {@code ThreadLocal} map must still produce a correlated + * {@code datadog.ReferenceChain} - {@code targetTag} in the leak-tag pool + * range, matched by a {@code datadog.HeapLiveObject} {@code leakTag}. The + * scenario is {@code ThreadLocalLeakScenario}; this test drives it and + * asserts the reported correlation, with a filtered child-log tail embedded + * in any failure message for diagnosability without a redeploy. + */ +@Tag("slow") +public class ThreadLocalLeakReferenceChainTest extends AbstractProcessProfilerTest { + + // Mirrored from referenceChains.h (LEAK_TAG_BASE/LEAK_TAG_POOL_SIZE) - the + // contract under test, asserted against the runtime event values the child + // reports. + private static final long LEAK_TAG_BASE = 0x40000000L; + private static final long LEAK_TAG_POOL_SIZE = 256; + + @Test + void shouldCorrelateThreadLocalHeldLeakChain() throws Exception { + assumeFalse(Platform.isJavaVersion(8)); + assumeFalse(Platform.isJ9()); + assumeFalse(Platform.isZing()); + + Path scratchDumpPath = Files.createTempFile("referencechains-tl-correlation-", ".jfr"); + Files.deleteIfExists(scratchDumpPath); + Path continuousJfrPath = Files.createTempFile("referencechains-tl-correlation-continuous-", ".jfr"); + Deque testLogTail = new ArrayDeque<>(); + try { + // Same raised budget/ratio rationale as + // LeakTagCorrelationReferenceChainTest (cold external JVM, ~100k filler, + // memory=64:l:1.0 keeping every chunk allocation tracked). + String startCommand = "start,memory=64:l:1.0,generations=true," + + "referencechains=true:hops=64:budget=200000:ttl=120000:framecap=2000000:" + + "pausetarget=60000" + + ",jfr,file=" + continuousJfrPath.toAbsolutePath(); + String packedCommand = startCommand + "|||" + scratchDumpPath.toAbsolutePath(); + + List jvmArgs = Collections.singletonList( + "-Dddprof_test.config=" + System.getProperty("ddprof_test.config")); + + AtomicReference resultLine = new AtomicReference<>(); + LaunchResult result = launch("threadlocal-leak", jvmArgs, packedCommand, + Collections.emptyMap(), + 150, + line -> { + if (line.startsWith(ThreadLocalLeakScenario.FOUND_MARKER) + || line.equals(ThreadLocalLeakScenario.NOT_FOUND_MARKER) + || line.startsWith(ThreadLocalLeakScenario.TAG_OUT_OF_POOL_MARKER) + || line.startsWith(ThreadLocalLeakScenario.NO_LIVE_OBJECT_MARKER)) { + resultLine.set(line); + } + if (line.startsWith("[TEST::INFO]")) { + testLogTail.addLast(line); + while (testLogTail.size() > 2000) { + testLogTail.removeFirst(); + } + } + return LineConsumerResult.CONTINUE; + }, + null); + + assertTrue(result.inTime, "Child process did not exit within the wait timeout"); + assertEquals(0, result.exitCode, "Child process exited with a non-zero code"); + assertNotNull(resultLine.get(), "Child process never printed a recognizable result " + + "marker on stdout" + diagnostics(testLogTail)); + String line = resultLine.get(); + if (line.startsWith(ThreadLocalLeakScenario.FOUND_MARKER)) { + long targetTag = Long.parseLong( + line.substring(ThreadLocalLeakScenario.FOUND_MARKER.length()).trim()); + assertTrue(targetTag >= LEAK_TAG_BASE && targetTag < LEAK_TAG_BASE + LEAK_TAG_POOL_SIZE, + "Reported targetTag " + targetTag + " is outside the leak-tag pool range [" + + LEAK_TAG_BASE + ", " + (LEAK_TAG_BASE + LEAK_TAG_POOL_SIZE) + ")"); + return; + } + fail("Thread-local leak-tag correlation did not succeed: " + line + + diagnostics(testLogTail)); + } finally { + Files.deleteIfExists(scratchDumpPath); + Files.deleteIfExists(continuousJfrPath); + } + } + + /** + * A filtered summary of the child's TEST_LOG stream for failure messages - + * same first-things-to-check selection as + * LeakTagCorrelationReferenceChainTest.diagnostics(): the thread-walk and + * pass summaries (the new machinery this scenario exercises), the gotw + * batch-size trace, and any leak-tag interception or correlation lines. + */ + private static String diagnostics(Deque testLogTail) { + if (testLogTail.isEmpty()) { + return "\n(no TEST_LOG output captured - non-debug build or no reference-chain passes ran)"; + } + List threadWalk = new ArrayList<>(); + List runPassDone = new ArrayList<>(); + List gotw = new ArrayList<>(); + List leakTagLines = new ArrayList<>(); + for (String s : testLogTail) { + if (s.contains("walkCandidateThreadLocals")) { + threadWalk.add(s); + } else if (s.contains("runPass done:")) { + runPassDone.add(s); + } else if (s.contains("expandFrontier gotw")) { + gotw.add(s); + } else if (s.contains("leak-tag intercepted") + || s.contains("correlateAdmittedLeakTag") + || s.contains("tagLeakInstances summary")) { + leakTagLines.add(s); + } + } + StringBuilder sb = new StringBuilder("\n--- filtered TEST_LOG tail ---\n"); + appendTail(sb, "thread-walk:", threadWalk, 10); + appendTail(sb, "runPass:", runPassDone, 5); + appendTail(sb, "gotw:", gotw, 5); + appendTail(sb, "leak-tag:", leakTagLines, 10); + return sb.toString(); + } + + private static void appendTail(StringBuilder sb, String label, List lines, int cap) { + if (lines.isEmpty()) { + return; + } + sb.append(label).append('\n'); + int start = Math.max(0, lines.size() - cap); + for (int i = start; i < lines.size(); i++) { + sb.append(" ").append(lines.get(i)).append('\n'); + } + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakScenario.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakScenario.java new file mode 100644 index 0000000000..45f61e4211 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakScenario.java @@ -0,0 +1,351 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.referencechains; + +import com.datadoghq.profiler.JavaProfiler; +import org.openjdk.jmc.common.IMCType; +import org.openjdk.jmc.common.item.IItem; +import org.openjdk.jmc.common.item.IItemCollection; +import org.openjdk.jmc.common.item.IItemIterable; +import org.openjdk.jmc.common.item.IMemberAccessor; +import org.openjdk.jmc.common.item.IType; +import org.openjdk.jmc.flightrecorder.JfrLoaderToolkit; + +import java.io.InputStream; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.ArrayList; +import java.util.List; +import java.util.concurrent.ArrayBlockingQueue; +import java.util.concurrent.BlockingQueue; +import java.util.concurrent.TimeUnit; + +/** + * The thread-local taxonomy of the leak-tag correlation use case, run inside a + * separate child JVM by {@code ExternalLauncher}'s {@code threadlocal-leak} mode: + * a growing {@code byte[]} collection held ONLY through the leaking thread's + * own {@code ThreadLocal} map - never by any static field - must end up + * reported as a {@code datadog.ReferenceChain} event whose {@code targetTag} + * is a leak-tag-pool tag, matching a {@code datadog.HeapLiveObject} event's + * {@code leakTag} for the same object. + * + *

    This is the taxonomy shape {@code ReferenceChainTracker}'s thread-scoped + * walk exists for (walkCandidateThreadLocals(), referenceChains.cpp): the + * retained chunks are reachable only through the Thread object's + * {@code threadLocals} -> ThreadLocalMap -> table -> Entry -> value path, so + * the scenario proves that holder shape keeps working end-to-end - the + * thread-object registry, the anchor admission and the descend gates do not + * break the plain chain + correlation contract. The transient-root gate and + * the noise-chain suppression are {@code LeakTagCorrelationScenario}'s own + * contracts and are deliberately not re-asserted here. + * + *

    Same separate-process, seed-before-start, per-round hysteresis + * maintenance and debug-seam patterns as {@code LeakTagCorrelationScenario}'s + * own class comment (the one-shot root-seeded walk is a per-process + * resource, and a one-shot seeded ramp ages out of LivenessTracker's + * hysteresis once real fold samples interleave). + */ +public final class ThreadLocalLeakScenario { + private ThreadLocalLeakScenario() {} + + /** + * The leak sink: a per-thread growing list, touched ONLY by the leaking + * thread below. Deliberately NOT static-retained anywhere - the whole + * point is that the only durable path to the chunks runs through the + * leaking thread's ThreadLocalMap. + */ + private static final ThreadLocal> SINK = + ThreadLocal.withInitial(() -> new ArrayList<>()); + + /** Scale ingredient - same rationale as LeakTagCorrelationScenario.FILLER. */ + static final Object[][] FILLER = new Object[1500][]; + + // Distinct from the other scenarios' ids (987301-987304) - each scenario + // runs in its own child JVM, but a distinct id keeps logs unambiguous. + private static final int LEAK_KLASS_ID = 987305; + + // See LeakTagCorrelationScenario.SEED_EPOCHS_FOR_HYSTERESIS' own comment. + private static final int SEED_EPOCHS_FOR_HYSTERESIS = 15; + private static final int SEED_CHUNKS = 10; + + private static int hysteresisEpoch = 0; + + /** + * Chunk size that clears ObjectSampler's 256KiB sampling floor on its own + * (see LeakTagCorrelationScenario.CHUNK_BYTES' own comment): every chunk + * allocation must be sampled so it becomes a tracked, taggable + * live-heap instance. + */ + private static final int CHUNK_BYTES = 1_000_000; + + // Mirrored from referenceChains.h (LEAK_TAG_BASE/LEAK_TAG_POOL_SIZE). + private static final long LEAK_TAG_BASE = 0x40000000L; + private static final long LEAK_TAG_POOL_SIZE = 256; + + /** + * Handoff for the leak thread's profiler tid and one seed chunk - the + * main thread seeds LivenessTracker's per-(klass, tid) qualification with + * the REAL allocating thread's tid, and that tid lives on the leak thread. + * The chunk reference is dropped by take() so the main thread never keeps a + * second path to the sink alive. + */ + private static final class LeakHandoff { + volatile int tid; + volatile byte[] seedChunk; + + void publish(int tid, byte[] chunk) { + this.tid = tid; + this.seedChunk = chunk; + } + + byte[] take() { + byte[] chunk = seedChunk; + seedChunk = null; + return chunk; + } + } + + /** Printed to stdout, followed by the correlated targetTag, on full success. */ + public static final String FOUND_MARKER = "[tl-correlation-found] "; + + /** Printed to stdout if no leaked byte[] chain was ever observed. */ + public static final String NOT_FOUND_MARKER = "[tl-correlation-not-found]"; + + /** Printed to stdout (with the tag) if a chain's targetTag is out of pool range. */ + public static final String TAG_OUT_OF_POOL_MARKER = "[tl-correlation-tag-out-of-pool] "; + + /** Printed to stdout (with the tag) if no HeapLiveObject matched the chain's tag. */ + public static final String NO_LIVE_OBJECT_MARKER = "[tl-correlation-no-live-object] "; + + private static final class ScanResult { + ReferenceChainAssertions.ChainMatch payloadChain; + boolean payloadChainInPool; + int liveObjectMatchesForChainTag; + } + + public static void run(JavaProfiler profiler, String startCommand, Path scratchDumpPath) + throws Exception { + // Build the whole fixture before the profiler starts - see the class + // comment and LeakingCacheScenario's seed-before-start rationale. + for (int row = 0; row < FILLER.length; row++) { + Object[] column = new Object[64]; + for (int col = 0; col < column.length; col++) { + column[col] = new FillerNode(); + } + FILLER[row] = column; + } + + LeakHandoff handoff = new LeakHandoff(); + // One round = the main thread offers a token; the leak thread takes it and + // appends one chunk to ITS OWN ThreadLocal sink (same persistent-allocator + // shape as ReferenceChainTrackingTest's togcroot fixture, so the chunks are + // always allocated on the same qualifying tid). + BlockingQueue rounds = new ArrayBlockingQueue<>(32); + Thread leakThread = new Thread(() -> { + List sink = SINK.get(); + for (int i = 0; i < SEED_CHUNKS; i++) { + sink.add(new byte[CHUNK_BYTES]); + } + handoff.publish(JavaProfiler.getTid(), sink.get(0)); + while (true) { + try { + if (rounds.poll(50, TimeUnit.MILLISECONDS) == null) { + continue; + } + sink.add(new byte[CHUNK_BYTES]); + } catch (InterruptedException e) { + return; + } + } + }, "threadlocal-leak"); + leakThread.setDaemon(true); + leakThread.start(); + + if (startCommand != null && !startCommand.isEmpty()) { + profiler.execute(startCommand); + } + + boolean debugBuild = "debug".equals(System.getProperty("ddprof_test.config")); + int leakTid = 0; + if (debugBuild) { + // Representative first, seeds second - same aliasing rationale as + // LeakTagCorrelationScenario (the rep's real klass id must own the + // seeds or the seeds authorize nothing). The rep chunk stays durably + // reachable only through the leak thread's ThreadLocalMap - the + // take() above dropped this handoff's reference. + JavaProfiler.setKlassPopulationRepresentativeForTest0(LEAK_KLASS_ID, handoff.take()); + leakTid = handoff.tid; + for (int epoch = 1; epoch <= SEED_EPOCHS_FOR_HYSTERESIS; epoch++) { + JavaProfiler.seedKlassPopulationSample0(LEAK_KLASS_ID, epoch * 10, epoch); + JavaProfiler.seedTidTrendSample0(LEAK_KLASS_ID, leakTid, epoch * 3, epoch); + } + // Wait for the BFS thread's first full pass, then poll - same + // load-bearing wait as LeakTagCorrelationScenario. + int initialPasses = JavaProfiler.referenceChainPassesRunForTest0(); + for (int i = 0; i < 300; i++) { + if (JavaProfiler.referenceChainPassesRunForTest0() != initialPasses) { + break; + } + Thread.sleep(100); + } + JavaProfiler.pollReferenceChainTargets0(); + } else { + Thread.sleep(5000); + } + + ScanResult best = null; + int totalRounds = 25; + for (int round = 1; round <= totalRounds; round++) { + rounds.offer(round); + if (debugBuild) { + hysteresisEpoch++; + JavaProfiler.seedKlassPopulationSample0( + LEAK_KLASS_ID, hysteresisEpoch * 10, hysteresisEpoch); + JavaProfiler.seedTidTrendSample0(LEAK_KLASS_ID, leakTid, hysteresisEpoch * 3, hysteresisEpoch); + } + Thread.sleep(300); + System.gc(); + profiler.dump(scratchDumpPath); + // Scan every 3rd round - same scan-churn rationale as + // LeakTagCorrelationScenario's own comment. + if (round % 3 == 0) { + best = keepBest(best, scan(scratchDumpPath)); + if (best != null && best.payloadChainInPool + && best.liveObjectMatchesForChainTag > 0) { + break; + } + } + } + + for (int attempt = 0; attempt < 5; attempt++) { + if (best != null && best.payloadChainInPool && best.liveObjectMatchesForChainTag > 0) { + break; + } + if (debugBuild) { + hysteresisEpoch++; + JavaProfiler.seedKlassPopulationSample0( + LEAK_KLASS_ID, hysteresisEpoch * 10, hysteresisEpoch); + JavaProfiler.seedTidTrendSample0(LEAK_KLASS_ID, leakTid, hysteresisEpoch * 3, hysteresisEpoch); + } + rounds.offer(attempt); + Thread.sleep(1000); + System.gc(); + profiler.dump(scratchDumpPath); + best = keepBest(best, scan(scratchDumpPath)); + } + + if (best == null || best.payloadChain == null) { + System.out.println(NOT_FOUND_MARKER); + return; + } + if (!best.payloadChainInPool) { + System.out.println(TAG_OUT_OF_POOL_MARKER + best.payloadChain.targetTag); + return; + } + if (best.liveObjectMatchesForChainTag == 0) { + System.out.println(NO_LIVE_OBJECT_MARKER + best.payloadChain.targetTag); + return; + } + System.out.println(FOUND_MARKER + best.payloadChain.targetTag); + } + + private static ScanResult keepBest(ScanResult best, ScanResult scan) { + if (scan == null) { + return best; + } + if (scan.payloadChain != null) { + return scan; + } + return best != null ? best : scan; + } + + /** + * Scans one dump: the first leaked {@code byte[]} chain (if any), whether + * its targetTag is in the leak-tag pool, and how many + * {@code datadog.HeapLiveObject} events for a leaked {@code byte[]} carry + * that same tag in {@code leakTag}. Simplified from + * LeakTagCorrelationScenario.scan() - no transient/noise gate scans (those + * are that scenario's own contracts). + */ + private static ScanResult scan(Path scratchDumpPath) throws Exception { + if (!Files.exists(scratchDumpPath)) { + return null; + } + IItemCollection events; + try (InputStream in = Files.newInputStream(scratchDumpPath)) { + events = JfrLoaderToolkit.loadEvents(in); + } + ScanResult result = new ScanResult(); + + IItemCollection chains = events.apply( + org.openjdk.jmc.common.item.ItemFilters.type("datadog.ReferenceChain")); + for (IItemIterable iterable : chains) { + IType type = iterable.getType(); + IMemberAccessor chainAccessor = + ReferenceChainAssertions.findAccessor(type, "chain"); + IMemberAccessor targetTagAccessor = + ReferenceChainAssertions.findAccessor(type, "targetTag"); + for (IItem item : iterable) { + Object chainValue = chainAccessor.getMember(item); + if (!(chainValue instanceof Object[])) { + continue; + } + Object[] rawChain = (Object[]) chainValue; + String leaf = rawChain.length > 0 && rawChain[0] instanceof IMCType + ? ((IMCType) rawChain[0]).getFullName() : ""; + long targetTag = targetTagAccessor != null + ? ReferenceChainAssertions.numberValue(targetTagAccessor.getMember(item)) : -1; + if (!"byte[]".equals(leaf)) { + continue; + } + boolean inPool = + targetTag >= LEAK_TAG_BASE && targetTag < LEAK_TAG_BASE + LEAK_TAG_POOL_SIZE; + // Prefer the first in-pool (correlated) chain over any out-of-pool + // one seen earlier - same rationale as + // LeakTagCorrelationScenario.scan(). + if ((inPool && !result.payloadChainInPool) || result.payloadChain == null) { + List chain = new ArrayList<>(rawChain.length); + for (Object element : rawChain) { + chain.add((IMCType) element); + } + result.payloadChain = new ReferenceChainAssertions.ChainMatch(chain, targetTag, 0); + result.payloadChainInPool = inPool; + } + } + } + + if (result.payloadChainInPool) { + IItemCollection liveObjects = events.apply( + org.openjdk.jmc.common.item.ItemFilters.type("datadog.HeapLiveObject")); + for (IItemIterable iterable : liveObjects) { + IType type = iterable.getType(); + IMemberAccessor leakTagAccessor = + ReferenceChainAssertions.findAccessor(type, "leakTag"); + IMemberAccessor objectClassAccessor = + ReferenceChainAssertions.findAccessor(type, "objectClass"); + for (IItem item : iterable) { + long leakTag = leakTagAccessor != null + ? ReferenceChainAssertions.numberValue(leakTagAccessor.getMember(item)) : -1; + if (leakTag != result.payloadChain.targetTag) { + continue; + } + Object objectClass = objectClassAccessor != null + ? objectClassAccessor.getMember(item) : null; + if (objectClass instanceof IMCType + && "byte[]".equals(((IMCType) objectClass).getFullName())) { + result.liveObjectMatchesForChainTag++; + } + } + } + } + return result; + } + + /** Scale-only object - same rationale as LeakTagCorrelationScenario.FillerNode. */ + static final class FillerNode { + long p0, p1, p2, p3, p4, p5, p6, p7; + } +} From ac51d2887cfd9be51b35511f82e80708ec326a28 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Wed, 16 Sep 2026 17:01:53 +0200 Subject: [PATCH 04/18] Add chaos and repro scenarios for reference chains Stresstest wiring for reference-chain validation: a leak antagonist that grows collections and holds thread-locals, a standalone leak demo for repro runs, and chaos-harness build integration. --- ddprof-stresstest/build.gradle.kts | 39 +++ .../com/datadoghq/profiler/chaos/Main.java | 2 + .../chaos/ReferenceChainLeakAntagonist.java | 106 +++++++ .../repro/ReferenceChainLeakDemo.java | 282 ++++++++++++++++++ 4 files changed, 429 insertions(+) create mode 100644 ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/ReferenceChainLeakAntagonist.java create mode 100644 ddprof-stresstest/src/repro/java/com/datadoghq/profiler/repro/ReferenceChainLeakDemo.java diff --git a/ddprof-stresstest/build.gradle.kts b/ddprof-stresstest/build.gradle.kts index 550baee43b..67896f2b7d 100644 --- a/ddprof-stresstest/build.gradle.kts +++ b/ddprof-stresstest/build.gradle.kts @@ -101,6 +101,45 @@ tasks.register("chaosJar") { } } +// --- reference-chains repro app --------------------------------------------- +// Plain, dependency-free demo app for manually reproducing the reference-chains +// feature: launched directly with `-agentpath:libjavaProfiler.so=start,...` +// (no dd-trace-java agent, no dynamic attach), so it deliberately has zero +// compile/runtime dependency on ddprof-lib or dd-trace - the profiler is +// entirely out-of-process from this app's own point of view. + +sourceSets { + create("repro") +} + +dependencies { + // ddprof-lib's Java API only - not the native library itself (that's the one + // already loaded in-process via -agentpath). Used solely to call + // JavaProfiler.getInstance().dump(...) periodically: Profiler::dump() + // (profiler.cpp) is the only code path that drains ReferenceChainTracker's + // pending chain-event queue and actually writes datadog.ReferenceChain into + // the JFR file - nothing does this automatically on a timer, and in + // production it's dd-trace-java's own recording-chunk rotation that calls + // it. Without this dependency+call, chain events get built and enqueued + // (visible in TEST_LOG output) but are silently dropped when the process + // exits, and the JFR file never shows a single datadog.ReferenceChain event. + "reproImplementation"(project(mapOf("path" to ":ddprof-lib", "configuration" to "debug"))) +} + +tasks.register("reproJar") { + group = "build" + description = "Demo app that leaks memory in a controlled way, for manually reproducing reference-chains via -agentpath" + archiveFileName.set("refchains_repro.jar") + from(sourceSets["repro"].output) + from({ + configurations["reproRuntimeClasspath"].map { if (it.isDirectory) it else zipTree(it) } + }) + duplicatesStrategy = DuplicatesStrategy.EXCLUDE + manifest { + attributes("Main-Class" to "com.datadoghq.profiler.repro.ReferenceChainLeakDemo") + } +} + tasks.register("runStressTests") { dependsOn(tasks.named("jmhJar")) diff --git a/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/Main.java b/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/Main.java index 7e846c02d1..a8e51e8666 100644 --- a/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/Main.java +++ b/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/Main.java @@ -92,6 +92,8 @@ private static Antagonist create(String name) { return new WeakRefWaveAntagonist(); case "dump-storm": return new DumpStormAntagonist(); + case "reference-chain-leak": + return new ReferenceChainLeakAntagonist(); case "reapply-context-value": return new ReapplyContextValueAntagonist(); // Deferred: dlopen-churn (needs per-arch dummy .so built in CI prep). diff --git a/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/ReferenceChainLeakAntagonist.java b/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/ReferenceChainLeakAntagonist.java new file mode 100644 index 0000000000..a63c613752 --- /dev/null +++ b/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/ReferenceChainLeakAntagonist.java @@ -0,0 +1,106 @@ +/* + * Copyright 2026, Datadog, Inc + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + */ +package com.datadoghq.profiler.chaos; + +import java.time.Duration; +import java.util.Map; +import java.util.concurrent.ConcurrentHashMap; + +/** + * Grows an ever-referenced, never-evicted cache to give the reference-chains + * feature (LivenessTracker/ReferenceChainTracker) a real, steadily growing + * population to detect and walk. Growth is capped, not open-ended: once + * {@link #CAP_ENTRIES} is reached the cache stops growing and the antagonist + * just holds it there for the rest of the run, so a long chaos duration + * (hours) still produces the same bounded leak instead of eventually OOMing + * the JVM out from under the other antagonists sharing its heap. + * + *

    Same object shape as {@code LeakingCacheScenario.CachedPayload} + * (ddprof-test's reference-chain integration test) — plain long fields + * rather than a nested array, so a single instance clears the allocation + * sampler's real size floor without a second, competing heap allocation per + * entry. + */ +public final class ReferenceChainLeakAntagonist implements Antagonist { + + // ~64MiB of CachedEntry payloads at steady state (entries are a little + // over 300 bytes with object header/field overhead) — enough for the + // liveness tracker's population-growth trend detection to have a real, + // sustained trend to lock onto, without meaningfully competing with the + // other antagonists' own memory budget for the run's whole duration. + private static final int CAP_ENTRIES = 200_000; + private static final int BATCH_SIZE = 200; + private static final long BATCH_INTERVAL_MS = 50L; + + private final Map cache = new ConcurrentHashMap<>(); + private volatile boolean running; + private Thread growthDriver; + private long nextKey; + + @Override + public String name() { + return "reference-chain-leak"; + } + + @Override + public void start() { + running = true; + growthDriver = new Thread(this::growthLoop, "chaos-refchain-leak"); + growthDriver.setDaemon(true); + growthDriver.start(); + } + + @Override + public void stopGracefully(Duration timeout) { + running = false; + try { + growthDriver.join(timeout.toMillis()); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + } + } + + private void growthLoop() { + while (running && cache.size() < CAP_ENTRIES) { + for (int i = 0; i < BATCH_SIZE && cache.size() < CAP_ENTRIES; i++) { + String key = "refchain-leak-" + (nextKey++); + cache.put(key, new CachedEntry(key)); + MemoryGovernor.pace(); + } + try { + Thread.sleep(BATCH_INTERVAL_MS); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + return; + } + } + // Cap reached: keep the reference alive (do nothing) for the rest of + // the run instead of exiting the thread, so `cache` stays reachable + // via this antagonist's own field for as long as the harness runs. + while (running) { + try { + Thread.sleep(1_000L); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + return; + } + } + } + + static final class CachedEntry { + final String key; + long p0, p1, p2, p3, p4, p5, p6, p7, p8, p9, p10, p11, p12, p13, p14, p15; + long p16, p17, p18, p19, p20, p21, p22, p23, p24, p25, p26, p27, p28, p29, p30, p31; + + CachedEntry(String key) { + this.key = key; + } + } +} diff --git a/ddprof-stresstest/src/repro/java/com/datadoghq/profiler/repro/ReferenceChainLeakDemo.java b/ddprof-stresstest/src/repro/java/com/datadoghq/profiler/repro/ReferenceChainLeakDemo.java new file mode 100644 index 0000000000..0416a1fea1 --- /dev/null +++ b/ddprof-stresstest/src/repro/java/com/datadoghq/profiler/repro/ReferenceChainLeakDemo.java @@ -0,0 +1,282 @@ +/* + * Copyright 2026, Datadog, Inc + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + */ +package com.datadoghq.profiler.repro; + +import com.datadoghq.profiler.JavaProfiler; +import java.lang.management.ManagementFactory; +import java.lang.management.MemoryUsage; +import java.nio.file.Path; +import java.nio.file.Paths; +import java.util.HashMap; +import java.util.Map; + +/** + * Standalone demo app for manually reproducing the reference-chains feature end-to-end. + * Meant to be launched directly with the native agent loaded at JVM startup, e.g. + * + *

    + * java -agentpath:/path/to/libjavaProfiler.so=start,memory=64:l,generations=true,\
    + *   referencechains=true:hops=64:budget=4000:ttl=120000:framecap=2000000:\
    + *   pausetarget=5000:painbudget=100,jfr,file=repro.jfr \
    + *   -jar refchains_repro.jar repro.jfr /path/to/libjavaProfiler.so
    + * 
    + * + *

    Takes the JFR output path as {@code args[0]} (must match {@code file=} above) and the + * agent's own {@code .so} path as {@code args[1]} (must match the {@code -agentpath} value + * above), and periodically calls {@link JavaProfiler#dump}. Both are load-bearing: + * + *

    An optional {@code args[2]} duration in seconds switches from the default "run forever" + * manual mode to a bounded session that dumps once more and prints a single {@code [metrics]} + * line at exit (throughput, round latency, heap growth) - see + * {@code utils/compare-refchains-repro.sh}, which runs this app twice (referencechains on vs. + * off) over the same duration and diffs those lines against each run's safepoint/GC logs. + *

      + *
    • The dump call itself: {@code Profiler::dump()} (profiler.cpp) is the only code path + * that drains {@code ReferenceChainTracker}'s pending chain-event queue and actually writes + * {@code datadog.ReferenceChain} into the JFR file - nothing drains that queue on a timer, + * and in production it's dd-trace-java's own periodic recording-chunk rotation that calls + * {@code dump()}. Without it, chain events get built and enqueued (visible via {@code + * TEST_LOG}) but are silently discarded when the process exits. + *
    • Passing the {@code .so} path explicitly to {@link JavaProfiler#getInstance(String, + * String)}: with no {@code libLocation}, {@code getInstance()} extracts the jar's own + * bundled copy of the library to a fresh temp file and {@code System.load()}s that + * instead of attaching to the one {@code -agentpath} already loaded - a second, never-{@code + * start()}-ed {@code Profiler} singleton whose {@code dump()} calls are silent no-ops. + * Passing the same path lets the dynamic linker dedup the load onto the already-running + * singleton, the same way dd-trace-java attaches to its own {@code -agentpath} agent. + *
    + * + *

    Deliberately omits {@code firstpassbudget}: the tracker auto-scales its + * first pass's budget from the plain {@code budget} value + * ({@code ReferenceChainTracker::AUTO_FIRST_PASS_BUDGET_MULTIPLIER}, capped at + * {@code AUTO_FIRST_PASS_BUDGET_CAP}, referenceChains.h) rather than reusing + * {@code budget} as-is for that first, cold root-seeded walk - reusing it directly + * truncates every pass before it gets anywhere near this app's {@code CachedEntry} + * population, and the target's JVMTI tag stays 0 forever. Pass + * {@code firstpassbudget=N} explicitly only to override that auto-scaled default. + * + *

    Grows an ever-referenced, never-evicted {@code HashMap}-backed cache (same shape as + * ddprof-test's {@code LeakingCacheScenario}/the chaos harness's {@code + * ReferenceChainLeakAntagonist}). Two things this app does deliberately, both found the hard + * way by running it without them and never seeing a chain get discovered: + * + *

      + *
    • Explicit {@code System.gc()} every round. {@code LivenessTracker}'s + * population-growth trend ({@code computeKlassPopulationSlope}, livenessTracker.cpp) needs at + * least {@code KLASS_POPULATION_MIN_FILL_FOR_TREND} (10) distinct GC epochs of population + * samples before it will even attempt a slope, and {@code _gc_epoch} only advances on a + * {@code GarbageCollectionFinish} JVMTI callback - i.e. on an actual GC. Left to whatever GCs + * the JVM decides to run on its own, a slow, modest allocation rate against a large default + * heap can go many minutes without a single GC, so the epoch count - and therefore the trend - + * never moves. Forcing a GC every round (same pattern {@code LeakingCacheScenario}'s round + * loop and {@code ReferenceChainTrackingTest} already rely on) turns epoch progress into + * something this app controls directly instead of leaving it to chance. + *
    • No hard stop. A cache that grows to a fixed cap and then sits flat stops being a + * growing population the instant it stops growing - {@code computeKlassPopulationSlope} + * compares the earliest vs. most recent third of its 30-epoch ring, so a long enough flat + * period after the cap pushes the slope back to zero (or the real growth epochs age out of the + * ring before the search ever gets scheduled/its own pain-budget cooldown clears). This app + * keeps growing forever instead, self-throttling its own pace against a heap-usage watermark + * (see {@link #run}) so it stays a real, positive, continuously-observable trend without ever + * growing fast/large enough to actually exhaust the heap. + *
    + * + *

    Seeds the cache with an initial batch on the very first line of {@code main} - before + * printing anything or sleeping - so the agent's one-shot, root-seeded first pass (which fires + * roughly a second after the agent's own background thread starts, i.e. before {@code + * Agent_OnLoad} even returns control to this class's {@code main}) has a real chance of finding + * {@code cache} already non-empty. See {@code LeakingCacheScenario}'s own comment for the full + * story of why that ordering is load-bearing. + */ +public final class ReferenceChainLeakDemo { + + // Heap-usage watermarks gating growth pace, mirroring the chaos harness's own + // MemoryGovernor - a much simpler version since this app has only one grower to pace, + // not several antagonists sharing a budget. + private static final double HIGH_WATERMARK = 0.70; + private static final double CRITICAL_WATERMARK = 0.85; + + private static final int NORMAL_BATCH_SIZE = 2_000; + private static final int THROTTLED_BATCH_SIZE = 100; + private static final long ROUND_INTERVAL_MS = 300L; + private static final long THROTTLED_ROUND_INTERVAL_MS = 2_000L; + + // How often to call JavaProfiler.dump() to drain any pending reference-chain + // events into the JFR file - see this class's own header comment for why this + // call, not just this app's own memory growth, is load-bearing for the repro. + private static final int DUMP_EVERY_N_ROUNDS = 10; + + public static void main(String[] args) throws Exception { + if (args.length < 2) { + System.err.println("usage: ReferenceChainLeakDemo [duration-seconds] " + + "(both required args must match the -agentpath:=...jfr,file= values; " + + "duration-seconds runs a bounded, metrics-reporting session instead of forever)"); + System.exit(1); + } + long durationSeconds = args.length >= 3 ? Long.parseLong(args[2]) : -1; + run(Paths.get(args[0]), args[1], durationSeconds); + } + + // FlightRecorder::dump() (flightRecorder.cpp) rejects dumping the continuous recording to + // its own file= path ("Can not dump recording to itself"), so periodic drains must target + // a different path; this is that path's suffix. + private static final String SNAPSHOT_SUFFIX = ".snapshot"; + + // Tracked so a bounded run (durationSeconds >= 0) can report throughput/latency/memory + // deltas at exit for A/B comparison against a referencechains=false baseline - see + // utils/compare-refchains-repro.sh. + private static long roundCount = 0; + private static long totalEntriesAdded = 0; + private static long totalRoundNanos = 0; + private static long maxRoundNanos = 0; + + private static void run(Path recordingPath, String agentSoPath, long durationSeconds) throws InterruptedException { + // Must differ from recordingPath itself - see SNAPSHOT_SUFFIX's comment. + Path snapshotPath = recordingPath.resolveSibling(recordingPath.getFileName() + SNAPSHOT_SUFFIX); + // Must pass the exact .so path -agentpath already loaded: JavaProfiler.getInstance() + // with no libLocation extracts the jar's own bundled copy to a fresh temp file and + // System.load()s *that* instead, producing a second, never-started Profiler singleton + // completely disconnected from the real profiling session - dump() calls against it + // are silent no-ops. Passing the same path here lets the dynamic linker dedup the + // load to the already-mapped library, i.e. the real, already-running singleton. + JavaProfiler profiler; + try { + profiler = JavaProfiler.getInstance(agentSoPath, null); + } catch (Exception e) { + throw new IllegalStateException( + "JavaProfiler.getInstance() failed - was this launched with " + + "-agentpath:libjavaProfiler.so=start,...?", e); + } + + Map cache = new HashMap<>(); + seed(cache, 300); + + System.out.println("[repro] pid=" + pid() + + " - growing cache " + (durationSeconds >= 0 ? ("for " + durationSeconds + "s") : "forever") + + ", forcing a GC every round so the profiler's " + + "population-growth trend detection has real epochs to work with; watch " + + "heapUsedMb/heapMaxMb and attach/dump once a chain shows up"); + + long startNanos = System.nanoTime(); + long deadlineNanos = durationSeconds >= 0 ? startNanos + durationSeconds * 1_000_000_000L : Long.MAX_VALUE; + long startHeapUsedMb = heapUsedMb(); + + long nextKey = 300; + int round = 0; + while (System.nanoTime() < deadlineNanos) { + long roundStartNanos = System.nanoTime(); + round++; + double heapFraction = heapUsedFraction(); + int batchSize = heapFraction >= HIGH_WATERMARK ? THROTTLED_BATCH_SIZE : NORMAL_BATCH_SIZE; + long roundIntervalMs = heapFraction >= HIGH_WATERMARK ? THROTTLED_ROUND_INTERVAL_MS : ROUND_INTERVAL_MS; + + if (heapFraction < CRITICAL_WATERMARK) { + for (int i = 0; i < batchSize; i++) { + String key = "leak-" + (nextKey++); + cache.put(key, new CachedEntry(key)); + } + totalEntriesAdded += batchSize; + } else { + System.out.println("[repro] heap usage critical (" + pct(heapFraction) + + ") - pausing growth this round, attach/dump now if you haven't"); + } + + // Forces a GC every round regardless of watermark state, so _gc_epoch (and + // therefore the population-growth ring) keeps advancing even while paused/ + // throttled - see this class's own header comment for why that's load-bearing. + System.gc(); + + if (round % 10 == 0 || heapFraction >= HIGH_WATERMARK) { + System.out.println("[repro] round=" + round + " cache.size()=" + cache.size() + + " heapUsedMb=" + heapUsedMb() + " heapFraction=" + pct(heapFraction)); + } + + if (round % DUMP_EVERY_N_ROUNDS == 0) { + profiler.dump(snapshotPath); + } + + long roundNanos = System.nanoTime() - roundStartNanos; + roundCount++; + totalRoundNanos += roundNanos; + maxRoundNanos = Math.max(maxRoundNanos, roundNanos); + + Thread.sleep(roundIntervalMs); + } + + if (durationSeconds >= 0) { + profiler.dump(snapshotPath); + reportMetrics(startNanos, startHeapUsedMb); + } + } + + // Printed with a distinct "[metrics]" prefix so utils/compare-refchains-repro.sh can grep + // it out of the two runs' stdout (referencechains=true vs. false) and diff throughput/ + // latency/memory-growth directly, without needing to touch the safepoint/GC logs for those. + private static void reportMetrics(long startNanos, long startHeapUsedMb) { + long wallNanos = System.nanoTime() - startNanos; + double wallSeconds = wallNanos / 1_000_000_000.0; + double avgRoundMs = roundCount == 0 ? 0 : (totalRoundNanos / (double) roundCount) / 1_000_000.0; + double maxRoundMs = maxRoundNanos / 1_000_000.0; + double entriesPerSec = wallSeconds > 0 ? totalEntriesAdded / wallSeconds : 0; + long heapGrowthMb = heapUsedMb() - startHeapUsedMb; + + System.out.println("[metrics] wallSeconds=" + String.format("%.1f", wallSeconds) + + " rounds=" + roundCount + + " entriesAdded=" + totalEntriesAdded + + " entriesPerSec=" + String.format("%.1f", entriesPerSec) + + " avgRoundMs=" + String.format("%.3f", avgRoundMs) + + " maxRoundMs=" + String.format("%.3f", maxRoundMs) + + " heapUsedMbStart=" + startHeapUsedMb + + " heapUsedMbEnd=" + heapUsedMb() + + " heapGrowthMb=" + heapGrowthMb); + } + + private static void seed(Map cache, int count) { + for (int i = 0; i < count; i++) { + String key = "seed-" + i; + cache.put(key, new CachedEntry(key)); + } + } + + private static String pid() { + String name = ManagementFactory.getRuntimeMXBean().getName(); + int at = name.indexOf('@'); + return at >= 0 ? name.substring(0, at) : name; + } + + private static long heapUsedMb() { + return ManagementFactory.getMemoryMXBean().getHeapMemoryUsage().getUsed() / (1024 * 1024); + } + + private static double heapUsedFraction() { + MemoryUsage heap = ManagementFactory.getMemoryMXBean().getHeapMemoryUsage(); + return heap.getMax() > 0 ? (double) heap.getUsed() / (double) heap.getMax() : 0.0; + } + + private static String pct(double fraction) { + return String.format("%.1f%%", fraction * 100.0); + } + + /** + * Same shape as {@code LeakingCacheScenario.CachedPayload}/{@code + * ReferenceChainLeakAntagonist.CachedEntry}: plain long fields, not a nested array, so a + * single instance clears the allocation sampler's real size floor without a second, + * competing heap allocation per entry. + */ + static final class CachedEntry { + final String key; + long p0, p1, p2, p3, p4, p5, p6, p7, p8, p9, p10, p11, p12, p13, p14, p15; + long p16, p17, p18, p19, p20, p21, p22, p23, p24, p25, p26, p27, p28, p29, p30, p31; + + CachedEntry(String key) { + this.key = key; + } + } +} From 4df5906998d5bfbdfe4408c39eb2a0a5156e7d8c Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Wed, 16 Sep 2026 17:01:56 +0200 Subject: [PATCH 05/18] Add repro, sweep, and reporting tooling for reference chains Scripts to run the reference-chains repro, sweep budget/behavior matrices across builds, compare repro output, and render single/multi run reports from JFR recordings (with vendored chart.js). --- utils/compare-refchains-repro.sh | 163 +++++++++++++++++++ utils/refchains-jfr-metrics.py | 79 ++++++++++ utils/refchains-report-multi.py | 261 +++++++++++++++++++++++++++++++ utils/refchains-report.py | 231 +++++++++++++++++++++++++++ utils/run-chaos-harness.sh | 38 +++-- utils/run-refchains-repro.sh | 136 ++++++++++++++++ utils/sweep-refchains-all.sh | 149 ++++++++++++++++++ utils/sweep-refchains-budgets.sh | 139 ++++++++++++++++ utils/vendor/chart.umd.min.js | 14 ++ 9 files changed, 1201 insertions(+), 9 deletions(-) create mode 100755 utils/compare-refchains-repro.sh create mode 100644 utils/refchains-jfr-metrics.py create mode 100644 utils/refchains-report-multi.py create mode 100644 utils/refchains-report.py create mode 100755 utils/run-refchains-repro.sh create mode 100644 utils/sweep-refchains-all.sh create mode 100644 utils/sweep-refchains-budgets.sh create mode 100644 utils/vendor/chart.umd.min.js diff --git a/utils/compare-refchains-repro.sh b/utils/compare-refchains-repro.sh new file mode 100755 index 0000000000..0eaee3c90f --- /dev/null +++ b/utils/compare-refchains-repro.sh @@ -0,0 +1,163 @@ +#!/usr/bin/env bash +# +# Runs the reference-chains repro app (run-refchains-repro.sh) twice, back to +# back, over an identical bounded window - once with referencechains=true and +# once with referencechains=false - and diffs the two runs' throughput, +# safepoint/GC pause times, and heap growth, to answer: +# - is it slower overall (throughput/round latency)? +# - is it imposing lengthy STW safepoints (safepoint/GC log)? +# - is it increasing memory usage significantly (heap growth, peak RSS)? +# +# Prints a markdown comparison table plus explicit answers to those three +# questions, and saves the same content as report.md in the run's temp +# working directory alongside the raw stdout/safepoint logs. +# +# Usage: compare-refchains-repro.sh [duration-seconds] +# +# Env vars: same REFCHAINS_SO/REFCHAINS_JAR/REFCHAINS_ARGS/REFCHAINS_JAVA_HOME/ +# REFCHAINS_GC as run-refchains-repro.sh - passed through unchanged to both +# variants so the two runs stay comparable (REFCHAINS_ENABLED is set by this +# script itself, per run - don't set it yourself). + +set -euo pipefail + +HERE="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )" + +DURATION_SECONDS="${1:-120}" +WORKDIR="$(mktemp -d /tmp/refchains_compare.XXXXXX)" + +JDK_DESC="${REFCHAINS_JAVA_HOME:-java on PATH}" +GC_DESC="${REFCHAINS_GC:-JVM ergonomic default}" + +echo "Comparing referencechains=true vs. false over ${DURATION_SECONDS}s each" +echo "JDK: ${JDK_DESC}" +echo "GC: ${GC_DESC}" +echo "Working directory: ${WORKDIR}" + +run_variant() { + local label="$1" enabled="$2" jfr="${WORKDIR}/${1}.jfr" + echo + echo "=== running variant '${label}' (referencechains=${enabled}) ===" + REFCHAINS_ENABLED="${enabled}" "${HERE}/run-refchains-repro.sh" "${jfr}" "${DURATION_SECONDS}" \ + > "${WORKDIR}/${label}.stdout.log" 2>&1 & + local pid=$! + + # Peak RSS while the JVM runs - the -Xlog heap numbers only cover Java heap, + # not native/off-heap growth (e.g. the tracker's own chain-storage arena). + local peak_rss_kb=0 + while kill -0 "${pid}" 2>/dev/null; do + local rss + rss=$(ps -o rss= -p "${pid}" 2>/dev/null | tr -d ' ' || true) + if [ -n "${rss}" ] && [ "${rss}" -gt "${peak_rss_kb}" ]; then + peak_rss_kb="${rss}" + fi + sleep 1 + done + wait "${pid}" || echo "WARN: variant '${label}' exited non-zero" + echo "${peak_rss_kb}" > "${WORKDIR}/${label}.peak_rss_kb" +} + +run_variant "on" "true" +run_variant "off" "false" + +# Parses one variant's logs and echoes a single space-separated line of +# metrics (entriesPerSec avgRoundMs maxRoundMs heapGrowthMb stwCount stwTotalSec +# stwAvgSec stwMaxSec peakRssKb wallSeconds), so both variants can be captured +# into shell variables and diffed numerically rather than eyeballed as text. +parse_variant() { + local label="$1" + local stdout="${WORKDIR}/${label}.stdout.log" + local safepoint_log="${WORKDIR}/${label}.jfr.safepoint.log" + local peak_rss_kb + peak_rss_kb=$(cat "${WORKDIR}/${label}.peak_rss_kb" 2>/dev/null || echo 0) + + local metrics_line + metrics_line=$(grep '^\[metrics\]' "${stdout}" || true) + local wall_seconds entries_per_sec avg_round_ms max_round_ms heap_growth_mb + wall_seconds=$(echo "${metrics_line}" | grep -oE 'wallSeconds=[0-9.]+' | cut -d= -f2 || true) + entries_per_sec=$(echo "${metrics_line}" | grep -oE 'entriesPerSec=[0-9.]+' | cut -d= -f2 || true) + avg_round_ms=$(echo "${metrics_line}" | grep -oE 'avgRoundMs=[0-9.]+' | cut -d= -f2 || true) + max_round_ms=$(echo "${metrics_line}" | grep -oE 'maxRoundMs=[0-9.]+' | cut -d= -f2 || true) + heap_growth_mb=$(echo "${metrics_line}" | grep -oE 'heapGrowthMb=-?[0-9]+' | cut -d= -f2 || true) + + # Unified-logging safepoint format (JDK 17+, -Xlog:safepoint*): one line per + # safepoint event, e.g. `[safepoint] Safepoint "G1CollectFull", Time since + # last: ... ns, ..., Total: 11011880 ns`. "Total" is the full STW window for + # that safepoint (sync + cleanup + vmop), which includes GC pauses since GC + # itself runs at a safepoint. Sum + max (converted ns -> s) across the run + # gives total/longest STW time. + # Two -Xlog safepoint line shapes across JDK versions: + # - JDK 17+: `Safepoint "name", ... Total: N ns` (one line, everything on it) + # - JDK <=16 (e.g. 11): `Total time for which application threads were + # stopped: N seconds, Stopping threads took: ...` (separate summary line, + # no per-name "Safepoint" prefix, value already in seconds not ns) + # Try the ns-based JDK17+ shape first; fall back to the seconds-based one. + local stw_stats + stw_stats=$(grep -oE 'Safepoint "[^"]+".*Total: [0-9]+ ns' "${safepoint_log}" 2>/dev/null \ + | grep -oE 'Total: [0-9]+ ns' \ + | grep -oE '[0-9]+' \ + | awk '{sec=$1/1e9; sum+=sec; if(sec>max) max=sec; n+=1} END {if(n>0) printf "%d %.4f %.4f %.4f", n, sum, sum/n, max}') + if [ -z "${stw_stats}" ]; then + stw_stats=$(grep -oE 'Total time for which application threads were stopped: [0-9.]+ seconds' "${safepoint_log}" 2>/dev/null \ + | grep -oE '[0-9.]+' \ + | awk '{sec=$1; sum+=sec; if(sec>max) max=sec; n+=1} END {if(n>0) printf "%d %.4f %.4f %.4f", n, sum, sum/n, max}') + fi + read -r stw_count stw_total stw_avg stw_max <<< "${stw_stats:-0 0 0 0}" + + echo "${entries_per_sec:-0} ${avg_round_ms:-0} ${max_round_ms:-0} ${heap_growth_mb:-0} ${stw_count} ${stw_total} ${stw_avg} ${stw_max} ${peak_rss_kb} ${wall_seconds:-0}" +} + +read -r on_entries on_avg_round on_max_round on_heap_growth on_stw_count on_stw_total on_stw_avg on_stw_max on_rss on_wall \ + <<< "$(parse_variant "on")" +read -r off_entries off_avg_round off_max_round off_heap_growth off_stw_count off_stw_total off_stw_avg off_stw_max off_rss off_wall \ + <<< "$(parse_variant "off")" + +# Percentage delta of "on" relative to "off": positive means "on" is worse +# (slower/longer-paused/more memory). Guards div-by-zero by falling back to 0. +pct_delta() { + awk -v on="$1" -v off="$2" 'BEGIN { + if (off == 0) { printf "n/a"; exit } + printf "%+.1f%%", ((on - off) / off) * 100.0 + }' +} + +throughput_delta_pct=$(pct_delta "${off_entries}" "${on_entries}") # off/on: lower throughput with refchains on is the "cost" +avg_round_delta_pct=$(pct_delta "${on_avg_round}" "${off_avg_round}") +stw_total_delta_pct=$(pct_delta "${on_stw_total}" "${off_stw_total}") +stw_max_delta_pct=$(pct_delta "${on_stw_max}" "${off_stw_max}") +heap_growth_delta_mb=$((on_heap_growth - off_heap_growth)) +rss_delta_kb=$((on_rss - off_rss)) +rss_delta_pct=$(pct_delta "${on_rss}" "${off_rss}") + +REPORT="${WORKDIR}/report.md" +{ + echo "# reference-chains overhead comparison" + echo + echo "referencechains=true vs. false, ${DURATION_SECONDS}s each. Raw logs: \`${WORKDIR}\`" + echo + echo "JDK: ${JDK_DESC} | GC: ${GC_DESC}" + echo + echo "| metric | on (refchains=true) | off (refchains=false) | delta (on vs. off) |" + echo "|---|---|---|---|" + echo "| entries/sec (throughput) | ${on_entries} | ${off_entries} | ${throughput_delta_pct} |" + echo "| avg round latency (ms) | ${on_avg_round} | ${off_avg_round} | ${avg_round_delta_pct} |" + echo "| max round latency (ms) | ${on_max_round} | ${off_max_round} | - |" + echo "| safepoint count | ${on_stw_count} | ${off_stw_count} | - |" + echo "| total STW stop time (s) | ${on_stw_total} | ${off_stw_total} | ${stw_total_delta_pct} |" + echo "| longest single STW pause (s) | ${on_stw_max} | ${off_stw_max} | ${stw_max_delta_pct} |" + echo "| heap growth (MB) | ${on_heap_growth} | ${off_heap_growth} | ${heap_growth_delta_mb} MB |" + echo "| peak RSS (KB) | ${on_rss} | ${off_rss} | ${rss_delta_pct} (${rss_delta_kb} KB) |" + echo + echo "## Answers" + echo + echo "- **Slower overall?** throughput ${throughput_delta_pct}, avg round latency ${avg_round_delta_pct} with referencechains on." + echo "- **Lengthy STW safepoints?** total stop time ${stw_total_delta_pct}, longest single pause ${on_stw_max}s (vs. ${off_stw_max}s off)." + echo "- **Memory usage up significantly?** heap grew ${heap_growth_delta_mb} MB more, peak RSS ${rss_delta_pct} (${rss_delta_kb} KB) with referencechains on." +} > "${REPORT}" + +echo +echo "===================== results =====================" +cat "${REPORT}" +echo +echo "Report saved to: ${REPORT}" +echo "Raw logs kept in: ${WORKDIR}" diff --git a/utils/refchains-jfr-metrics.py b/utils/refchains-jfr-metrics.py new file mode 100644 index 0000000000..68197c7c5b --- /dev/null +++ b/utils/refchains-jfr-metrics.py @@ -0,0 +1,79 @@ +#!/usr/bin/env python3 +"""Extracts reference-chains discovery-latency metrics from one repro run's +JFR snapshot: how many datadog.ReferenceChain/ReferenceChainAbandoned events +got recorded, and the wall-clock time from the recording's start to the +first ReferenceChain event (the feature's actual "time to first chain"). + +`jfr print`/`jfr summary` require a ".jfr" filename extension, which the +repro app's snapshot files (foo.jfr.snapshot) don't have - the caller must +pass a path already renamed/copied to end in .jfr. + +Usage: refchains-jfr-metrics.py +Prints one line: " " +(time_to_first_chain_s is -1 if no ReferenceChain event was recorded). +""" +import json +import re +import subprocess +import sys +from datetime import datetime, timezone + + +def run_jfr(args, path): + try: + result = subprocess.run(["jfr", *args, path], capture_output=True, text=True, check=True) + except (subprocess.CalledProcessError, FileNotFoundError) as e: + stderr = getattr(e, "stderr", None) or str(e) + print(f"error: 'jfr {' '.join(args)} {path}' failed: {stderr}", file=sys.stderr) + sys.exit(1) + return result.stdout + + +def jfr_recording_start(path): + out = run_jfr(["summary"], path) + m = re.search(r"^ Start:\s+(.+?)\s*\(UTC\)\s*$", out, re.MULTILINE) + if not m: + return None + return datetime.strptime(m.group(1).strip(), "%Y-%m-%d %H:%M:%S").replace(tzinfo=timezone.utc) + + +def jfr_event_times(path, event_name): + out = run_jfr(["print", "--events", event_name, "--json"], path) + try: + data = json.loads(out) + except json.JSONDecodeError as e: + print(f"error: could not parse 'jfr print --events {event_name} --json {path}' " + f"output as JSON: {e}", file=sys.stderr) + sys.exit(1) + times = [] + for ev in data.get("recording", {}).get("events", []): + st = ev.get("values", {}).get("startTime") or ev.get("startTime") + if st is None: + continue + # jfr --json timestamps look like "2026-07-24T11:38:20.123+00:00" + times.append(datetime.fromisoformat(st.replace("Z", "+00:00"))) + return times + + +def main(): + if len(sys.argv) < 2: + print("Usage: refchains-jfr-metrics.py ", file=sys.stderr) + sys.exit(1) + path = sys.argv[1] + start = jfr_recording_start(path) + chain_times = jfr_event_times(path, "datadog.ReferenceChain") + abandoned_times = jfr_event_times(path, "datadog.ReferenceChainAbandoned") + + chain_count = len(chain_times) + abandoned_count = len(abandoned_times) + if chain_times and start is not None: + first = min(chain_times) + time_to_first_s = max(0.0, (first - start).total_seconds()) + else: + time_to_first_s = -1 + + print(f"{chain_count} {abandoned_count} {time_to_first_s:.3f}") + + +if __name__ == "__main__": + main() diff --git a/utils/refchains-report-multi.py b/utils/refchains-report-multi.py new file mode 100644 index 0000000000..300b05c819 --- /dev/null +++ b/utils/refchains-report-multi.py @@ -0,0 +1,261 @@ +#!/usr/bin/env python3 +"""Renders sweep-refchains-all.sh's combined (knob,param_value,) CSV +into a single self-contained HTML report: one tab per knob, one Chart.js line +chart per metric within each tab. Chart.js is vendored under utils/vendor and +inlined directly into the report so it opens standalone, offline, from the +sweep's temp working directory (no CDN dependency at view time). +""" +import argparse +import csv +import html +import json +import os + +ACCENT = "#2a78d6" # dataviz palette categorical slot 1 ("blue") - one series per chart, no legend needed +GRID = "#dedcd3" +TEXT_PRIMARY = "#0b0b0b" +TEXT_SECONDARY = "#52514e" +SURFACE = "#fcfcfb" + +HERE = os.path.dirname(os.path.abspath(__file__)) +CHARTJS_PATH = os.path.join(HERE, "vendor", "chart.umd.min.js") + +METRICS = [ + ("time_to_first_chain_s", "Time to first reference chain", "s", True), + ("chain_count", "Reference chains found", "count", False), + ("abandoned_count", "Searches abandoned", "count", True), + ("entries_per_sec", "Throughput", "entries/sec", False), + ("avg_round_ms", "Avg round latency", "ms", True), + ("max_round_ms", "Max round latency", "ms", True), + ("stw_total_s", "Total STW stop time", "s", True), + ("stw_max_s", "Longest single STW pause", "s", True), + ("heap_growth_mb", "Heap growth", "MB", True), + ("peak_rss_kb", "Peak RSS", "KB", True), +] + +# time_to_first_chain_s uses -1 as a sentinel for "no chain found within the +# run's duration" (see refchains-jfr-metrics.py) - never plot that as a +# real value, and never let it participate in the min/max Y bounds. +NOT_FOUND_SENTINEL = -1.0 + +KNOB_LABELS = { + "budget": "budget", + "firstpassbudget": "firstpassbudget", + "pausetarget": "pausetarget", + "painbudget": "painbudget", +} + + +def fnum(s): + try: + return float(s) + except (TypeError, ValueError): + return 0.0 + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("csv_path") + ap.add_argument("html_path") + ap.add_argument("--jdk", default="") + ap.add_argument("--gc", default="") + ap.add_argument("--duration", default="") + args = ap.parse_args() + + with open(args.csv_path, newline="") as f: + rows = list(csv.DictReader(f)) + + knobs = [] + for r in rows: + if r["knob"] not in knobs: + knobs.append(r["knob"]) + + with open(CHARTJS_PATH) as f: + chartjs_src = f.read() + + tabs_nav = [] + tabs_content = [] + chart_configs = [] # list of (canvas_id, config_dict) + + for ti, knob in enumerate(knobs): + knob_rows = [r for r in rows if r["knob"] == knob] + active = "active" if ti == 0 else "" + tabs_nav.append( + f'' + ) + + xs_raw = [r["param_value"] for r in knob_rows] + xs_num = [fnum(v) for v in xs_raw] + use_log = min(xs_num) > 0 and (max(xs_num) / min(xs_num) >= 10) + + cards = [] + for field, title, unit, higher_is_worse in METRICS: + canvas_id = f"chart-{knob}-{field}" + raw_ys = [fnum(r[field]) for r in knob_rows] + if field == "time_to_first_chain_s": + not_found = sum(1 for y in raw_ys if y == NOT_FOUND_SENTINEL) + ys = [None if y == NOT_FOUND_SENTINEL else y for y in raw_ys] + badge = ( + f'{not_found}/{len(raw_ys)} runs never found a chain' + if not_found else "" + ) + else: + ys = raw_ys + badge = "" + note = "higher is worse" if higher_is_worse else "" + cards.append(f''' +

    +

    {title} ({unit}{', ' + note if note else ''}) {badge}

    +
    +
    ''') + + config = { + "type": "line", + "data": { + "labels": xs_raw, + "datasets": [{ + "label": title, + "data": ys, + "borderColor": ACCENT, + "backgroundColor": ACCENT, + "borderWidth": 2, + "pointRadius": 4, + "pointHoverRadius": 6, + "tension": 0, + "fill": False, + "spanGaps": False, + }], + }, + "options": { + "responsive": True, + "maintainAspectRatio": False, + "plugins": { + "legend": {"display": False}, + "tooltip": { + "callbacks": {}, + "backgroundColor": "#1a1a19", + "titleColor": "#ffffff", + "bodyColor": "#ffffff", + }, + }, + "scales": { + "x": { + "type": "logarithmic" if use_log else "linear", + "title": {"display": True, "text": f"{KNOB_LABELS.get(knob, knob)}" + (" (log scale)" if use_log else "")}, + "grid": {"color": GRID}, + "ticks": {"color": TEXT_SECONDARY}, + }, + "y": { + "title": {"display": True, "text": unit}, + "grid": {"color": GRID}, + "ticks": {"color": TEXT_SECONDARY}, + "beginAtZero": True, + }, + }, + }, + } + chart_configs.append((canvas_id, config, xs_num if use_log else None)) + + tabs_content.append(f''' +
    +
    {"".join(cards)}
    +
    ''') + + table_header = "".join(f"{html.escape(str(k))}" for k in rows[0]) if rows else "" + table_rows = "\n".join( + "" + "".join(f"{html.escape(str(r[k]))}" for k in r) + "" for r in rows + ) + + # Chart.js logarithmic scale needs numeric x values, not category labels - + # use a scatter-with-lines dataset (x/y pairs) instead of the labels array + # for any knob using log scale, since category+log axis isn't supported. + js_chart_inits = [] + for canvas_id, config, xs_num in chart_configs: + if xs_num is not None: + ys = config["data"]["datasets"][0]["data"] + config["data"]["datasets"][0]["data"] = [{"x": x, "y": y} for x, y in zip(xs_num, ys)] + del config["data"]["labels"] + js_chart_inits.append( + f'new Chart(document.getElementById("{canvas_id}"), {json.dumps(config)});' + ) + + out = f""" + + + +reference-chains OFAT budget-knob sweep + + + +

    reference-chains OFAT budget-knob sweep

    +
    JDK: {html.escape(str(args.jdk))}  |  GC: {html.escape(str(args.gc))}  |  duration/point: {html.escape(str(args.duration))}s  |  one-factor-at-a-time: each knob swept independently, others left at their built-in defaults
    +
    {"".join(tabs_nav)}
    +{"".join(tabs_content)} +
    + Raw data ({len(rows)} runs) + + {table_header} + {table_rows} +
    +
    + + + +""" + + with open(args.html_path, "w") as f: + f.write(out) + + +if __name__ == "__main__": + main() diff --git a/utils/refchains-report.py b/utils/refchains-report.py new file mode 100644 index 0000000000..3f7cf3e0a3 --- /dev/null +++ b/utils/refchains-report.py @@ -0,0 +1,231 @@ +#!/usr/bin/env python3 +"""Renders sweep-refchains-budgets.sh's CSV into a self-contained HTML report +with one line chart per metric (param value on x, metric on y). No external +JS/CSS dependencies - everything (SVG chart bodies, tooltip behavior) is +inlined so the report opens standalone from the sweep's temp working dir. +""" +import argparse +import csv +import html +import sys + +ACCENT = "#2a78d6" # dataviz palette categorical slot 1 ("blue") - single series, no legend needed +GRID = "#dedcd3" +TEXT_PRIMARY = "#0b0b0b" +TEXT_SECONDARY = "#52514e" +SURFACE = "#fcfcfb" + +CHARTS = [ + ("entries_per_sec", "Throughput", "entries/sec", False), + ("avg_round_ms", "Avg round latency", "ms", False), + ("max_round_ms", "Max round latency", "ms", False), + ("stw_total_s", "Total STW stop time", "s", False), + ("stw_max_s", "Longest single STW pause", "s", False), + ("heap_growth_mb", "Heap growth", "MB", True), + ("peak_rss_kb", "Peak RSS", "KB", False), +] + +CHART_W, CHART_H = 640, 260 +MARGIN_L, MARGIN_R, MARGIN_T, MARGIN_B = 56, 24, 20, 32 + + +def fnum(s): + try: + return float(s) + except (TypeError, ValueError): + # Coercing an unparsable CSV cell to 0.0 lets the chart render + # instead of crashing, but a silent 0.0 looks like a real + # measurement - flag it so a malformed sweep row doesn't masquerade + # as a genuine zero data point. + print(f"warning: could not parse {s!r} as a number, using 0.0", file=sys.stderr) + return 0.0 + + +def render_chart(rows, field, title, unit): + xs = [fnum(r["param_value"]) for r in rows] + ys = [fnum(r[field]) for r in rows] + plot_w = CHART_W - MARGIN_L - MARGIN_R + plot_h = CHART_H - MARGIN_T - MARGIN_B + + # Budget-style sweep values are usually log-spaced (1000, 10000, 100000, ...). + # A linear x-scale collapses all but the top value into the left few + # pixels, so switch to log scale whenever the sweep spans more than a + # decade and every value is positive. + use_log_x = min(xs) > 0 and (max(xs) / min(xs) >= 10) + x_min, x_max = min(xs), max(xs) + y_min, y_max = min(0.0, min(ys)), max(ys) if ys else 1.0 + if y_max == y_min: + y_max = y_min + 1.0 + if x_max == x_min: + x_max = x_min + 1.0 + + def px(x): + if use_log_x: + import math + lo, hi, v = math.log10(x_min), math.log10(x_max), math.log10(x) + return MARGIN_L + (v - lo) / (hi - lo) * plot_w + return MARGIN_L + (x - x_min) / (x_max - x_min) * plot_w + + def py(y): + return MARGIN_T + plot_h - (y - y_min) / (y_max - y_min) * plot_h + + points = [(px(x), py(y)) for x, y in zip(xs, ys)] + path_d = "M " + " L ".join(f"{x:.1f},{y:.1f}" for x, y in points) + + # 4 recessive horizontal gridlines with y-axis value labels. + gridlines = [] + for i in range(5): + gy = MARGIN_T + plot_h * i / 4 + gval = y_max - (y_max - y_min) * i / 4 + gridlines.append( + f'' + f'{gval:.3g}' + ) + + x_labels = [] + for x, r in zip(xs, rows): + x_labels.append( + f'{r["param_value"]}' + ) + + markers = [] + for i, ((x, y), r) in enumerate(zip(points, rows)): + val = fnum(r[field]) + markers.append( + f'' + ) + + x_scale_note = ( + f'x: log scale' + if use_log_x else "" + ) + + svg = f''' + + {"".join(gridlines)} + + {"".join(markers)} + {"".join(x_labels)} + {x_scale_note} +''' + return svg + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("csv_path") + ap.add_argument("html_path") + ap.add_argument("--param", default="budget") + ap.add_argument("--jdk", default="") + ap.add_argument("--gc", default="") + ap.add_argument("--duration", default="") + args = ap.parse_args() + + with open(args.csv_path, newline="") as f: + rows = list(csv.DictReader(f)) + + chart_sections = [] + for field, title, unit, higher_is_worse in CHARTS: + note = "higher is worse" if higher_is_worse else "" + svg = render_chart(rows, field, title, unit) + chart_sections.append(f""" +
    +

    {html.escape(title)} ({html.escape(unit)}{', ' + note if note else ''})

    +
    {svg}
    +
    """) + + table_rows = "\n".join( + "" + "".join(f"{html.escape(r[k])}" for k in r) + "" for r in rows + ) + table_header = "".join(f"{html.escape(k)}" for k in rows[0]) if rows else "" + + out = f""" + + + +reference-chains budget sweep — {html.escape(args.param)} + + + +

    reference-chains budget sweep — {html.escape(args.param)}

    +
    JDK: {html.escape(args.jdk)}  |  GC: {html.escape(args.gc)}  |  duration/point: {html.escape(str(args.duration))}s  |  x-axis: {html.escape(args.param)} value
    +
    +{"".join(chart_sections)} +
    + +{table_header} + +{table_rows} + +
    +
    + + +""" + + with open(args.html_path, "w") as f: + f.write(out) + + +if __name__ == "__main__": + main() diff --git a/utils/run-chaos-harness.sh b/utils/run-chaos-harness.sh index b1a62139c7..8c12b905c0 100755 --- a/utils/run-chaos-harness.sh +++ b/utils/run-chaos-harness.sh @@ -177,16 +177,31 @@ case $ALLOCATOR in if [ -n "${GLIBC_VERSION}" ] && [ "$(printf '%s\n' "2.34" "${GLIBC_VERSION}" | sort -V | head -1)" = "2.34" ]; then MALLOC_DEBUG_LIB=$(find /usr/lib/ /usr/lib64/ /lib/ /lib64/ -maxdepth 4 -name 'libc_malloc_debug.so*' 2>/dev/null | head -1) if [ -z "${MALLOC_DEBUG_LIB}" ]; then - echo "FAIL:glibc ${GLIBC_VERSION} requires libc_malloc_debug to be preloaded for MALLOC_CHECK_ to take effect, but it could not be found" >&2 + # Fatal, matching tcmalloc/jemalloc's own missing-library handling + # below: ALLOCATOR=gmalloc is an explicit test variant in the + # scheduled chaos matrix specifically to exercise MALLOC_CHECK_'s + # heap-corruption detection, and the CI caller recognizes failure + # by grepping stderr for "FAIL:". Silently downgrading this to a + # WARN would let the job report success while never actually + # performing the corruption checking this variant promises (glibc + # >= 2.34's MALLOC_CHECK_ is a silent no-op without + # libc_malloc_debug preloaded). + echo "FAIL: glibc ${GLIBC_VERSION} requires libc_malloc_debug to be preloaded for MALLOC_CHECK_ to take effect, but it could not be found" >&2 exit 1 + else + export LD_PRELOAD="${MALLOC_DEBUG_LIB}${LD_PRELOAD:+:${LD_PRELOAD}}" + echo "glibc ${GLIBC_VERSION} detected — preloading ${MALLOC_DEBUG_LIB} for MALLOC_CHECK_" fi - export LD_PRELOAD="${MALLOC_DEBUG_LIB}${LD_PRELOAD:+:${LD_PRELOAD}}" - echo "glibc ${GLIBC_VERSION} detected — preloading ${MALLOC_DEBUG_LIB} for MALLOC_CHECK_" fi fi ;; tcmalloc) - export LD_PRELOAD=$(find /usr/lib/ /usr/lib64/ /opt/homebrew/lib/ /usr/local/lib/ -maxdepth 4 -name 'libtcmalloc_minimal.so.4' -o -name 'libtcmalloc.dylib' 2>/dev/null | head -1) + TCMALLOC_LIB=$(find /usr/lib/ /usr/lib64/ /opt/homebrew/lib/ /usr/local/lib/ -maxdepth 4 -name 'libtcmalloc_minimal.so.4' -o -name 'libtcmalloc.dylib' 2>/dev/null | head -1) + if [ -z "${TCMALLOC_LIB}" ]; then + echo "FAIL: allocator=tcmalloc requested but libtcmalloc_minimal.so.4/libtcmalloc.dylib could not be found" >&2 + exit 1 + fi + export LD_PRELOAD="${TCMALLOC_LIB}" # thread-churn/dump-storm antagonists cycle many short-lived threads; # tcmalloc's defaults are slow to return their per-thread caches to the # OS, which was inflating container RSS past the OOM limit on aarch64. @@ -194,7 +209,12 @@ case $ALLOCATOR in export TCMALLOC_AGGRESSIVE_DECOMMIT=1 ;; jemalloc) - export LD_PRELOAD=$(find /usr/lib/ /usr/lib64/ /opt/homebrew/lib/ /usr/local/lib/ -maxdepth 4 -name 'libjemalloc.so' -o -name 'libjemalloc.dylib' 2>/dev/null | head -1) + JEMALLOC_LIB=$(find /usr/lib/ /usr/lib64/ /opt/homebrew/lib/ /usr/local/lib/ -maxdepth 4 -name 'libjemalloc.so' -o -name 'libjemalloc.dylib' 2>/dev/null | head -1) + if [ -z "${JEMALLOC_LIB}" ]; then + echo "FAIL: allocator=jemalloc requested but libjemalloc.so/libjemalloc.dylib could not be found" >&2 + exit 1 + fi + export LD_PRELOAD="${JEMALLOC_LIB}" # Same aarch64 RSS-inflation issue as tcmalloc above: jemalloc's default # decay times leave dirty/muzzy pages resident under heavy thread churn. export MALLOC_CONF="background_thread:true,dirty_decay_ms:1000,muzzy_decay_ms:1000" @@ -247,7 +267,7 @@ echo "disk usage ($(dirname "${DDPROF_ROOT}")): $(df -h "$(dirname "${DDPROF_ROO # OOME instead, so a failure is a single, diagnosable event. CHAOS_START=$(date +%s) timeout "$((RUNTIME + 300))" \ -java -javaagent:${PATCHED_AGENT} \ +java -javaagent:"${PATCHED_AGENT}" \ --add-opens java.base/java.lang=ALL-UNNAMED \ ${ENABLEMENT} \ -Ddd.profiling.upload.period=10 \ @@ -261,11 +281,11 @@ java -javaagent:${PATCHED_AGENT} \ -Xmx${HEAP_MB}m -Xms${HEAP_MB}m \ -XX:MaxMetaspaceSize=384m \ -XX:NativeMemoryTracking=summary \ - -XX:ErrorFile=${HS_ERR} \ + -XX:ErrorFile="${HS_ERR}" \ -XX:+ExitOnOutOfMemoryError \ - -jar ${CHAOS_JAR} \ + -jar "${CHAOS_JAR}" \ --duration ${RUNTIME}s \ - --antagonists ${ANTAGONISTS} + --antagonists "${ANTAGONISTS}" RC=$? CHAOS_ELAPSED=$(( $(date +%s) - CHAOS_START )) diff --git a/utils/run-refchains-repro.sh b/utils/run-refchains-repro.sh new file mode 100755 index 0000000000..1425bd9ef0 --- /dev/null +++ b/utils/run-refchains-repro.sh @@ -0,0 +1,136 @@ +#!/usr/bin/env bash +# +# Runs the standalone reference-chains repro app (ddprof-stresstest/src/repro) +# with -agentpath and a set of referencechains sub-options known to actually +# reach the app's leaking CachedEntry population - see ReferenceChainLeakDemo's +# own class comment for why the tracker's auto-scaled first-pass budget +# (referenceChains.h's AUTO_FIRST_PASS_BUDGET_MULTIPLIER/_CAP) matters here. +# +# The jar takes the JFR output path as its own arg and periodically calls +# JavaProfiler.dump() on it - also load-bearing, per ReferenceChainLeakDemo's +# comment: without a dump() call nothing ever drains the tracker's pending +# chain-event queue, so datadog.ReferenceChain never actually lands in the file. +# +# Usage: run-refchains-repro.sh [jfr-output-path] [duration-seconds] +# +# duration-seconds (optional) bounds the run and makes the app print a single +# "[metrics]" summary line (throughput/round-latency/heap growth) at exit +# instead of running forever - see ReferenceChainLeakDemo.reportMetrics(). Also +# enables -Xlog safepoint+GC logging to .safepoint.log and +# .gc.log for STW-pause analysis. Used by compare-refchains-repro.sh to A/B +# referencechains=true vs. false over an identical bounded window. +# +# Env vars: +# REFCHAINS_SO path to libjavaProfiler.so/.dylib (default: locate the +# locally built debug artifact under ddprof-lib/build/lib) +# REFCHAINS_JAR path to refchains_repro.jar (default: build/rebuild via +# ./gradlew :ddprof-stresstest:reproJar) +# REFCHAINS_ENABLED "true" (default) or "false" - toggles the +# referencechains=... agent sub-option itself, for A/B +# comparison against a baseline with the feature off. +# REFCHAINS_ARGS extra referencechains=... sub-options appended after the +# baked-in defaults below (e.g. "hops=32") +# REFCHAINS_JAVA_HOME JDK home to launch the repro with (default: "java" on +# PATH). Lets you A/B a specific JDK, e.g. to reproduce a +# version-specific VMStructs bug: REFCHAINS_JAVA_HOME=/usr/local/sdkman/candidates/java/25.0.3-tem +# REFCHAINS_GC GC to force via -XX:+UseGC (default: JVM's own +# ergonomic default - no flag added). Accepts either the +# bare name ("Serial", "G1", "Parallel", "Shenandoah", +# "Z", "Epsilon") or the full flag name ("SerialGC"). +# e.g. REFCHAINS_GC=Serial or REFCHAINS_GC=ZGC + +set -euo pipefail + +HERE="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )" +ROOT="$( cd "${HERE}/.." >/dev/null 2>&1 && pwd )" + +JFR_OUT="${1:-/tmp/refchains_repro.jfr}" +DURATION_SECONDS="${2:-}" + +JAVA_BIN="java" +if [ -n "${REFCHAINS_JAVA_HOME:-}" ]; then + JAVA_BIN="${REFCHAINS_JAVA_HOME}/bin/java" + if [ ! -x "${JAVA_BIN}" ]; then + echo "FAIL: ${JAVA_BIN} not found/executable - check REFCHAINS_JAVA_HOME" >&2 + exit 1 + fi +fi +echo "Using JDK: $("${JAVA_BIN}" -version 2>&1 | head -1)" + +GC_JVM_OPTS=() +if [ -n "${REFCHAINS_GC:-}" ]; then + case "${REFCHAINS_GC}" in + *GC) GC_FLAG="-XX:+Use${REFCHAINS_GC}" ;; + *) GC_FLAG="-XX:+Use${REFCHAINS_GC}GC" ;; + esac + GC_JVM_OPTS=("${GC_FLAG}") + echo "Forcing GC: ${GC_FLAG}" +else + echo "GC: JVM ergonomic default" +fi + +if [ -z "${REFCHAINS_SO:-}" ]; then + # Prefer the debug artifact: this repro is built with `assembleDebug`, and a + # stale `release/` build left over from an earlier `assembleAll` sits under + # the same build/lib tree. A bare find|head -1 picks whichever the directory + # walk hits first (observed: release), silently running the OLD binary and + # making source edits appear to have no effect - so match debug/ first and + # only fall back to any build if no debug artifact exists. + REFCHAINS_SO=$(find "${ROOT}/ddprof-lib/build/lib" \ + \( -name 'libjavaProfiler.so' -o -name 'libjavaProfiler.dylib' \) \ + -path '*/debug/*' 2>/dev/null | head -1) + if [ -z "${REFCHAINS_SO}" ]; then + REFCHAINS_SO=$(find "${ROOT}/ddprof-lib/build/lib" \ + \( -name 'libjavaProfiler.so' -o -name 'libjavaProfiler.dylib' \) \ + 2>/dev/null | head -1) + fi +fi +if [ -z "${REFCHAINS_SO}" ] || [ ! -f "${REFCHAINS_SO}" ]; then + echo "FAIL:libjavaProfiler.so/.dylib not found - build one first: ./gradlew :ddprof-lib:debugSharedLibrary" >&2 + exit 1 +fi +echo "Using agent: ${REFCHAINS_SO}" + +if [ -z "${REFCHAINS_JAR:-}" ]; then + REFCHAINS_JAR="${ROOT}/ddprof-stresstest/build/libs/refchains_repro.jar" + if [ ! -f "${REFCHAINS_JAR}" ]; then + echo "refchains_repro.jar not present - building it" + ( cd "${ROOT}" && ./gradlew :ddprof-stresstest:reproJar -q --no-daemon ) + fi +fi +if [ ! -f "${REFCHAINS_JAR}" ]; then + echo "FAIL:refchains_repro.jar unavailable" >&2 + exit 1 +fi +echo "Using jar: ${REFCHAINS_JAR}" + +# No firstpassbudget here: the tracker auto-scales the first pass's budget +# from budget=4000 (currently x50, capped at 200000 - referenceChains.h) so +# it can actually reach this app's leaking population. Pass +# REFCHAINS_ARGS="firstpassbudget=N" to override that auto-scaled default. +REFCHAINS_ENABLED="${REFCHAINS_ENABLED:-true}" +REFERENCECHAINS_OPTS="${REFCHAINS_ENABLED}:hops=64:budget=10000:ttl=120000:framecap=2000000:pausetarget=500:painbudget=100" +if [ -n "${REFCHAINS_ARGS:-}" ]; then + REFERENCECHAINS_OPTS="${REFERENCECHAINS_OPTS}:${REFCHAINS_ARGS}" +fi + +echo "JFR output: ${JFR_OUT}" +echo "referencechains enabled: ${REFCHAINS_ENABLED}" +rm -f "${JFR_OUT}" + +JAVA_LOG_OPTS=() +if [ -n "${DURATION_SECONDS}" ]; then + # Unified-logging safepoint+GC pause detail, kept per-run alongside the JFR file so + # compare-refchains-repro.sh can parse "Total time for which application threads were + # stopped" (STW safepoint pauses, includes GC) and per-GC pause times out of one file. + SAFEPOINT_LOG="${JFR_OUT}.safepoint.log" + rm -f "${SAFEPOINT_LOG}" + JAVA_LOG_OPTS=(-Xlog:safepoint*=info,gc*=info:file="${SAFEPOINT_LOG}":time,uptime,level,tags) + echo "Safepoint/GC log: ${SAFEPOINT_LOG}" +fi + +exec "${JAVA_BIN}" \ + "${JAVA_LOG_OPTS[@]}" \ + "${GC_JVM_OPTS[@]}" \ + -agentpath:"${REFCHAINS_SO}"=start,memory=64:l,generations=true,referencechains=${REFERENCECHAINS_OPTS},jfr,file="${JFR_OUT}" \ + -jar "${REFCHAINS_JAR}" "${JFR_OUT}" "${REFCHAINS_SO}" ${DURATION_SECONDS} diff --git a/utils/sweep-refchains-all.sh b/utils/sweep-refchains-all.sh new file mode 100644 index 0000000000..76cd7b7f73 --- /dev/null +++ b/utils/sweep-refchains-all.sh @@ -0,0 +1,149 @@ +#!/usr/bin/env bash +# +# One-factor-at-a-time (OFAT) sweep across all four referencechains budget +# knobs (budget, firstpassbudget, pausetarget, painbudget): for each knob, +# sweeps its own value list while holding the other three at their built-in +# defaults (i.e. omitted from the sub-option string, letting the tracker's +# own defaults/auto-scaling apply - see referenceChains.h). A full cartesian +# product across all four knobs is combinatorially far more runs for little +# extra insight over OFAT at this stage, so this is deliberately OFAT, not a +# grid search. +# +# Writes one combined CSV (knob,param_value,) covering all four +# sweeps, then renders it via refchains-report-multi.py into a single +# Chart.js-based HTML report with one tab per knob. +# +# Usage: sweep-refchains-all.sh [duration-seconds] +# +# Env vars: +# REFCHAINS_SO/REFCHAINS_JAR/REFCHAINS_JAVA_HOME/REFCHAINS_GC same as +# run-refchains-repro.sh - passed through unchanged +# to every run so all sweeps stay comparable. +# REFCHAINS_ARGS extra referencechains=... sub-options held fixed +# across every run (in addition to the defaults +# baked into run-refchains-repro.sh). + +set -euo pipefail + +HERE="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )" + +DURATION_SECONDS="${1:-60}" + +KNOBS=(budget firstpassbudget pausetarget painbudget) +declare -A KNOB_VALUES=( + [budget]="${REFCHAINS_SWEEP_BUDGET_VALUES:-1000,4000,10000,40000,100000}" + [firstpassbudget]="${REFCHAINS_SWEEP_FIRSTPASSBUDGET_VALUES:-2000,10000,50000,200000}" + [pausetarget]="${REFCHAINS_SWEEP_PAUSETARGET_VALUES:-50,100,250,500,1000}" + [painbudget]="${REFCHAINS_SWEEP_PAINBUDGET_VALUES:-10,50,100,500,1000}" +) + +WORKDIR="$(mktemp -d /tmp/refchains_sweep_all.XXXXXX)" +CSV="${WORKDIR}/sweep.csv" + +JDK_DESC="${REFCHAINS_JAVA_HOME:-java on PATH}" +GC_DESC="${REFCHAINS_GC:-JVM ergonomic default}" + +TOTAL_RUNS=0 +for k in "${KNOBS[@]}"; do + IFS=',' read -r -a vals <<< "${KNOB_VALUES[${k}]}" + TOTAL_RUNS=$((TOTAL_RUNS + ${#vals[@]})) +done + +echo "OFAT sweep across knobs: ${KNOBS[*]}" +echo "JDK: ${JDK_DESC}" +echo "GC: ${GC_DESC}" +echo "Duration per point: ${DURATION_SECONDS}s, total runs: ${TOTAL_RUNS} (~$((TOTAL_RUNS * DURATION_SECONDS / 60)) min)" +echo "Working directory: ${WORKDIR}" + +echo "knob,param_value,entries_per_sec,avg_round_ms,max_round_ms,heap_growth_mb,stw_count,stw_total_s,stw_avg_s,stw_max_s,peak_rss_kb,wall_seconds,chain_count,abandoned_count,time_to_first_chain_s" > "${CSV}" + +# Same metrics extraction as sweep-refchains-budgets.sh/compare-refchains-repro.sh - +# kept in sync across all three since they read the same stdout/-Xlog sources. +parse_run() { + local label="$1" + local stdout="${WORKDIR}/${label}.stdout.log" + local safepoint_log="${WORKDIR}/${label}.jfr.safepoint.log" + local peak_rss_kb + peak_rss_kb=$(cat "${WORKDIR}/${label}.peak_rss_kb" 2>/dev/null || echo 0) + + local metrics_line + metrics_line=$(grep '^\[metrics\]' "${stdout}" || true) + local wall_seconds entries_per_sec avg_round_ms max_round_ms heap_growth_mb + wall_seconds=$(echo "${metrics_line}" | grep -oE 'wallSeconds=[0-9.]+' | cut -d= -f2) + entries_per_sec=$(echo "${metrics_line}" | grep -oE 'entriesPerSec=[0-9.]+' | cut -d= -f2) + avg_round_ms=$(echo "${metrics_line}" | grep -oE 'avgRoundMs=[0-9.]+' | cut -d= -f2) + max_round_ms=$(echo "${metrics_line}" | grep -oE 'maxRoundMs=[0-9.]+' | cut -d= -f2) + heap_growth_mb=$(echo "${metrics_line}" | grep -oE 'heapGrowthMb=-?[0-9]+' | cut -d= -f2) + + local stw_stats + stw_stats=$(grep -oE 'Safepoint "[^"]+".*Total: [0-9]+ ns' "${safepoint_log}" 2>/dev/null \ + | grep -oE 'Total: [0-9]+ ns' \ + | grep -oE '[0-9]+' \ + | awk '{sec=$1/1e9; sum+=sec; if(sec>max) max=sec; n+=1} END {if(n>0) printf "%d %.4f %.4f %.4f", n, sum, sum/n, max}') + if [ -z "${stw_stats}" ]; then + stw_stats=$(grep -oE 'Total time for which application threads were stopped: [0-9.]+ seconds' "${safepoint_log}" 2>/dev/null \ + | grep -oE '[0-9.]+' \ + | awk '{sec=$1; sum+=sec; if(sec>max) max=sec; n+=1} END {if(n>0) printf "%d %.4f %.4f %.4f", n, sum, sum/n, max}') + fi + read -r stw_count stw_total stw_avg stw_max <<< "${stw_stats:-0 0 0 0}" + + # jfr print/summary require a ".jfr" extension - the repro app's periodic + # dump target (".snapshot", see ReferenceChainLeakDemo's own header + # comment on SNAPSHOT_SUFFIX) doesn't have one, so copy it under a .jfr + # name before handing it to refchains-jfr-metrics.py. + local snapshot="${WORKDIR}/${label}.jfr.snapshot" + local chain_count=0 abandoned_count=0 time_to_first_chain_s=-1 + if [ -f "${snapshot}" ]; then + local tmp_jfr="${WORKDIR}/${label}.metrics.jfr" + cp "${snapshot}" "${tmp_jfr}" + local chain_stats + chain_stats=$(python3 "${HERE}/refchains-jfr-metrics.py" "${tmp_jfr}" 2>/dev/null || echo "0 0 -1") + read -r chain_count abandoned_count time_to_first_chain_s <<< "${chain_stats}" + rm -f "${tmp_jfr}" + fi + + echo "${entries_per_sec:-0} ${avg_round_ms:-0} ${max_round_ms:-0} ${heap_growth_mb:-0} ${stw_count} ${stw_total} ${stw_avg} ${stw_max} ${peak_rss_kb} ${wall_seconds:-0} ${chain_count} ${abandoned_count} ${time_to_first_chain_s}" +} + +run_idx=0 +for knob in "${KNOBS[@]}"; do + IFS=',' read -r -a vals <<< "${KNOB_VALUES[${knob}]}" + for value in "${vals[@]}"; do + run_idx=$((run_idx + 1)) + label="${knob}_${value}" + jfr="${WORKDIR}/${label}.jfr" + echo + echo "=== [${run_idx}/${TOTAL_RUNS}] ${knob}=${value} ===" + + REFCHAINS_ENABLED="true" REFCHAINS_ARGS="${knob}=${value}${REFCHAINS_ARGS:+:${REFCHAINS_ARGS}}" \ + "${HERE}/run-refchains-repro.sh" "${jfr}" "${DURATION_SECONDS}" \ + > "${WORKDIR}/${label}.stdout.log" 2>&1 & + pid=$! + + peak_rss_kb=0 + while kill -0 "${pid}" 2>/dev/null; do + rss=$(ps -o rss= -p "${pid}" 2>/dev/null | tr -d ' ' || true) + if [ -n "${rss}" ] && [ "${rss}" -gt "${peak_rss_kb}" ]; then + peak_rss_kb="${rss}" + fi + sleep 1 + done + wait "${pid}" || echo "WARN: run ${knob}=${value} exited non-zero" + echo "${peak_rss_kb}" > "${WORKDIR}/${label}.peak_rss_kb" + + read -r entries avg_round max_round heap_growth stw_count stw_total stw_avg stw_max rss wall \ + chain_count abandoned_count time_to_first_chain \ + <<< "$(parse_run "${label}")" + echo "${knob},${value},${entries},${avg_round},${max_round},${heap_growth},${stw_count},${stw_total},${stw_avg},${stw_max},${rss},${wall},${chain_count},${abandoned_count},${time_to_first_chain}" >> "${CSV}" + done +done + +echo +echo "CSV written to: ${CSV}" + +REPORT_HTML="${WORKDIR}/report.html" +python3 "${HERE}/refchains-report-multi.py" "${CSV}" "${REPORT_HTML}" \ + --jdk "${JDK_DESC}" --gc "${GC_DESC}" --duration "${DURATION_SECONDS}" + +echo "HTML report written to: ${REPORT_HTML}" +echo "Raw logs kept in: ${WORKDIR}" diff --git a/utils/sweep-refchains-budgets.sh b/utils/sweep-refchains-budgets.sh new file mode 100644 index 0000000000..1d795fb6e5 --- /dev/null +++ b/utils/sweep-refchains-budgets.sh @@ -0,0 +1,139 @@ +#!/usr/bin/env bash +# +# Sweeps referencechains=... budget-related sub-options (budget, +# firstpassbudget, pausetarget, painbudget) across a matrix of values and +# runs the repro app (run-refchains-repro.sh) once per combination, to +# answer: how does each budget knob trade off throughput/round-latency, +# STW pause time, and heap/RSS growth? +# +# This is the multi-point sibling of compare-refchains-repro.sh (which only +# ever does one on/off comparison at a single fixed set of sub-options). +# Reuses the same repro app and the same metric sources (stdout +# "[metrics]" line + -Xlog safepoint/GC log + peak RSS sampling), but drives +# N runs instead of 2 and writes them to a CSV plus a self-contained HTML +# report with charts (see refchains-report.py). +# +# Usage: sweep-refchains-budgets.sh [duration-seconds] +# +# Env vars: +# REFCHAINS_SWEEP_PARAM which sub-option to sweep: "budget" (default), +# "firstpassbudget", "pausetarget", or "painbudget". +# REFCHAINS_SWEEP_VALUES comma-separated values for that sub-option +# (default depends on REFCHAINS_SWEEP_PARAM - see +# below). +# REFCHAINS_SO/REFCHAINS_JAR/REFCHAINS_JAVA_HOME/REFCHAINS_GC same as +# run-refchains-repro.sh - passed through unchanged +# to every run so they stay comparable. +# REFCHAINS_ARGS extra referencechains=... sub-options held fixed +# across the whole sweep (in addition to the +# defaults baked into run-refchains-repro.sh). + +set -euo pipefail + +HERE="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )" + +DURATION_SECONDS="${1:-120}" +SWEEP_PARAM="${REFCHAINS_SWEEP_PARAM:-budget}" + +case "${SWEEP_PARAM}" in + budget) DEFAULT_VALUES="1000,4000,10000,40000,100000" ;; + firstpassbudget) DEFAULT_VALUES="2000,10000,50000,200000" ;; + pausetarget) DEFAULT_VALUES="50,100,250,500,1000" ;; + painbudget) DEFAULT_VALUES="10,50,100,500,1000" ;; + *) + echo "FAIL: unknown REFCHAINS_SWEEP_PARAM=${SWEEP_PARAM} (expected: budget, firstpassbudget, pausetarget, painbudget)" >&2 + exit 1 + ;; +esac +IFS=',' read -r -a SWEEP_VALUES <<< "${REFCHAINS_SWEEP_VALUES:-${DEFAULT_VALUES}}" + +WORKDIR="$(mktemp -d /tmp/refchains_sweep.XXXXXX)" +CSV="${WORKDIR}/sweep.csv" + +JDK_DESC="${REFCHAINS_JAVA_HOME:-java on PATH}" +GC_DESC="${REFCHAINS_GC:-JVM ergonomic default}" + +echo "Sweeping referencechains ${SWEEP_PARAM} over: ${SWEEP_VALUES[*]}" +echo "JDK: ${JDK_DESC}" +echo "GC: ${GC_DESC}" +echo "Duration per point: ${DURATION_SECONDS}s" +echo "Working directory: ${WORKDIR}" + +echo "param_value,entries_per_sec,avg_round_ms,max_round_ms,heap_growth_mb,stw_count,stw_total_s,stw_avg_s,stw_max_s,peak_rss_kb,wall_seconds" > "${CSV}" + +# Parses one run's logs into a single space-separated metrics line - same +# fields/sources as compare-refchains-repro.sh's parse_variant, factored out +# here since this script iterates N runs instead of a fixed on/off pair. +parse_run() { + local label="$1" + local stdout="${WORKDIR}/${label}.stdout.log" + local safepoint_log="${WORKDIR}/${label}.jfr.safepoint.log" + local peak_rss_kb + peak_rss_kb=$(cat "${WORKDIR}/${label}.peak_rss_kb" 2>/dev/null || echo 0) + + local metrics_line + metrics_line=$(grep '^\[metrics\]' "${stdout}" || true) + local wall_seconds entries_per_sec avg_round_ms max_round_ms heap_growth_mb + wall_seconds=$(echo "${metrics_line}" | grep -oE 'wallSeconds=[0-9.]+' | cut -d= -f2) + entries_per_sec=$(echo "${metrics_line}" | grep -oE 'entriesPerSec=[0-9.]+' | cut -d= -f2) + avg_round_ms=$(echo "${metrics_line}" | grep -oE 'avgRoundMs=[0-9.]+' | cut -d= -f2) + max_round_ms=$(echo "${metrics_line}" | grep -oE 'maxRoundMs=[0-9.]+' | cut -d= -f2) + heap_growth_mb=$(echo "${metrics_line}" | grep -oE 'heapGrowthMb=-?[0-9]+' | cut -d= -f2) + + # Two -Xlog safepoint line shapes across JDK versions: + # - JDK 17+: `Safepoint "name", ... Total: N ns` (one line, everything on it) + # - JDK <=16 (e.g. 11): `Total time for which application threads were + # stopped: N seconds, Stopping threads took: ...` (separate summary line, + # no per-name "Safepoint" prefix, value already in seconds not ns) + # Try the ns-based JDK17+ shape first; fall back to the seconds-based one. + local stw_stats + stw_stats=$(grep -oE 'Safepoint "[^"]+".*Total: [0-9]+ ns' "${safepoint_log}" 2>/dev/null \ + | grep -oE 'Total: [0-9]+ ns' \ + | grep -oE '[0-9]+' \ + | awk '{sec=$1/1e9; sum+=sec; if(sec>max) max=sec; n+=1} END {if(n>0) printf "%d %.4f %.4f %.4f", n, sum, sum/n, max}') + if [ -z "${stw_stats}" ]; then + stw_stats=$(grep -oE 'Total time for which application threads were stopped: [0-9.]+ seconds' "${safepoint_log}" 2>/dev/null \ + | grep -oE '[0-9.]+' \ + | awk '{sec=$1; sum+=sec; if(sec>max) max=sec; n+=1} END {if(n>0) printf "%d %.4f %.4f %.4f", n, sum, sum/n, max}') + fi + read -r stw_count stw_total stw_avg stw_max <<< "${stw_stats:-0 0 0 0}" + + echo "${entries_per_sec:-0} ${avg_round_ms:-0} ${max_round_ms:-0} ${heap_growth_mb:-0} ${stw_count} ${stw_total} ${stw_avg} ${stw_max} ${peak_rss_kb} ${wall_seconds:-0}" +} + +for value in "${SWEEP_VALUES[@]}"; do + label="${SWEEP_PARAM}_${value}" + jfr="${WORKDIR}/${label}.jfr" + echo + echo "=== running ${SWEEP_PARAM}=${value} ===" + + REFCHAINS_ENABLED="true" REFCHAINS_ARGS="${SWEEP_PARAM}=${value}${REFCHAINS_ARGS:+:${REFCHAINS_ARGS}}" \ + "${HERE}/run-refchains-repro.sh" "${jfr}" "${DURATION_SECONDS}" \ + > "${WORKDIR}/${label}.stdout.log" 2>&1 & + pid=$! + + peak_rss_kb=0 + while kill -0 "${pid}" 2>/dev/null; do + rss=$(ps -o rss= -p "${pid}" 2>/dev/null | tr -d ' ' || true) + if [ -n "${rss}" ] && [ "${rss}" -gt "${peak_rss_kb}" ]; then + peak_rss_kb="${rss}" + fi + sleep 1 + done + wait "${pid}" || echo "WARN: run ${SWEEP_PARAM}=${value} exited non-zero" + echo "${peak_rss_kb}" > "${WORKDIR}/${label}.peak_rss_kb" + + read -r entries avg_round max_round heap_growth stw_count stw_total stw_avg stw_max rss wall \ + <<< "$(parse_run "${label}")" + echo "${value},${entries},${avg_round},${max_round},${heap_growth},${stw_count},${stw_total},${stw_avg},${stw_max},${rss},${wall}" >> "${CSV}" +done + +echo +echo "CSV written to: ${CSV}" + +REPORT_HTML="${WORKDIR}/report.html" +python3 "${HERE}/refchains-report.py" "${CSV}" "${REPORT_HTML}" \ + --param "${SWEEP_PARAM}" --jdk "${JDK_DESC}" --gc "${GC_DESC}" --duration "${DURATION_SECONDS}" + +echo "HTML report written to: ${REPORT_HTML}" +echo "Raw logs kept in: ${WORKDIR}" diff --git a/utils/vendor/chart.umd.min.js b/utils/vendor/chart.umd.min.js new file mode 100644 index 0000000000..008464faae --- /dev/null +++ b/utils/vendor/chart.umd.min.js @@ -0,0 +1,14 @@ +/*! + * Chart.js v4.5.1 + * https://www.chartjs.org + * (c) 2025 Chart.js Contributors + * Released under the MIT License + */ +!function(t,e){"object"==typeof exports&&"undefined"!=typeof module?module.exports=e():"function"==typeof define&&define.amd?define(e):(t="undefined"!=typeof globalThis?globalThis:t||self).Chart=e()}(this,(function(){"use strict";var t=Object.freeze({__proto__:null,get Colors(){return Jo},get Decimation(){return ta},get Filler(){return ba},get Legend(){return Ma},get SubTitle(){return Pa},get Title(){return ka},get Tooltip(){return Na}});function e(){}const i=(()=>{let t=0;return()=>t++})();function s(t){return null==t}function n(t){if(Array.isArray&&Array.isArray(t))return!0;const e=Object.prototype.toString.call(t);return"[object"===e.slice(0,7)&&"Array]"===e.slice(-6)}function o(t){return null!==t&&"[object Object]"===Object.prototype.toString.call(t)}function a(t){return("number"==typeof t||t instanceof Number)&&isFinite(+t)}function r(t,e){return a(t)?t:e}function l(t,e){return void 0===t?e:t}const h=(t,e)=>"string"==typeof t&&t.endsWith("%")?parseFloat(t)/100:+t/e,c=(t,e)=>"string"==typeof t&&t.endsWith("%")?parseFloat(t)/100*e:+t;function d(t,e,i){if(t&&"function"==typeof t.call)return t.apply(i,e)}function u(t,e,i,s){let a,r,l;if(n(t))if(r=t.length,s)for(a=r-1;a>=0;a--)e.call(i,t[a],a);else for(a=0;at,x:t=>t.x,y:t=>t.y};function v(t){const e=t.split("."),i=[];let s="";for(const t of e)s+=t,s.endsWith("\\")?s=s.slice(0,-1)+".":(i.push(s),s="");return i}function M(t,e){const i=y[e]||(y[e]=function(t){const e=v(t);return t=>{for(const i of e){if(""===i)break;t=t&&t[i]}return t}}(e));return i(t)}function w(t){return t.charAt(0).toUpperCase()+t.slice(1)}const k=t=>void 0!==t,S=t=>"function"==typeof t,P=(t,e)=>{if(t.size!==e.size)return!1;for(const i of t)if(!e.has(i))return!1;return!0};function D(t){return"mouseup"===t.type||"click"===t.type||"contextmenu"===t.type}const C=Math.PI,O=2*C,A=O+C,T=Number.POSITIVE_INFINITY,L=C/180,E=C/2,R=C/4,I=2*C/3,z=Math.log10,F=Math.sign;function V(t,e,i){return Math.abs(t-e)t-e)).pop(),e}function N(t){return!function(t){return"symbol"==typeof t||"object"==typeof t&&null!==t&&!(Symbol.toPrimitive in t||"toString"in t||"valueOf"in t)}(t)&&!isNaN(parseFloat(t))&&isFinite(t)}function H(t,e){const i=Math.round(t);return i-e<=t&&i+e>=t}function j(t,e,i){let s,n,o;for(s=0,n=t.length;sl&&h=Math.min(e,i)-s&&t<=Math.max(e,i)+s}function et(t,e,i){i=i||(i=>t[i]1;)s=o+n>>1,i(s)?o=s:n=s;return{lo:o,hi:n}}const it=(t,e,i,s)=>et(t,i,s?s=>{const n=t[s][e];return nt[s][e]et(t,i,(s=>t[s][e]>=i));function nt(t,e,i){let s=0,n=t.length;for(;ss&&t[n-1]>i;)n--;return s>0||n{const i="_onData"+w(e),s=t[e];Object.defineProperty(t,e,{configurable:!0,enumerable:!1,value(...e){const n=s.apply(this,e);return t._chartjs.listeners.forEach((t=>{"function"==typeof t[i]&&t[i](...e)})),n}})})))}function rt(t,e){const i=t._chartjs;if(!i)return;const s=i.listeners,n=s.indexOf(e);-1!==n&&s.splice(n,1),s.length>0||(ot.forEach((e=>{delete t[e]})),delete t._chartjs)}function lt(t){const e=new Set(t);return e.size===t.length?t:Array.from(e)}const ht="undefined"==typeof window?function(t){return t()}:window.requestAnimationFrame;function ct(t,e){let i=[],s=!1;return function(...n){i=n,s||(s=!0,ht.call(window,(()=>{s=!1,t.apply(e,i)})))}}function dt(t,e){let i;return function(...s){return e?(clearTimeout(i),i=setTimeout(t,e,s)):t.apply(this,s),e}}const ut=t=>"start"===t?"left":"end"===t?"right":"center",ft=(t,e,i)=>"start"===t?e:"end"===t?i:(e+i)/2,gt=(t,e,i,s)=>t===(s?"left":"right")?i:"center"===t?(e+i)/2:e;function pt(t,e,i){const n=e.length;let o=0,a=n;if(t._sorted){const{iScale:r,vScale:l,_parsed:h}=t,c=t.dataset&&t.dataset.options?t.dataset.options.spanGaps:null,d=r.axis,{min:u,max:f,minDefined:g,maxDefined:p}=r.getUserBounds();if(g){if(o=Math.min(it(h,d,u).lo,i?n:it(e,d,r.getPixelForValue(u)).lo),c){const t=h.slice(0,o+1).reverse().findIndex((t=>!s(t[l.axis])));o-=Math.max(0,t)}o=Z(o,0,n-1)}if(p){let t=Math.max(it(h,r.axis,f,!0).hi+1,i?0:it(e,d,r.getPixelForValue(f),!0).hi+1);if(c){const e=h.slice(t-1).findIndex((t=>!s(t[l.axis])));t+=Math.max(0,e)}a=Z(t,o,n)-o}else a=n-o}return{start:o,count:a}}function mt(t){const{xScale:e,yScale:i,_scaleRanges:s}=t,n={xmin:e.min,xmax:e.max,ymin:i.min,ymax:i.max};if(!s)return t._scaleRanges=n,!0;const o=s.xmin!==e.min||s.xmax!==e.max||s.ymin!==i.min||s.ymax!==i.max;return Object.assign(s,n),o}class xt{constructor(){this._request=null,this._charts=new Map,this._running=!1,this._lastDate=void 0}_notify(t,e,i,s){const n=e.listeners[s],o=e.duration;n.forEach((s=>s({chart:t,initial:e.initial,numSteps:o,currentStep:Math.min(i-e.start,o)})))}_refresh(){this._request||(this._running=!0,this._request=ht.call(window,(()=>{this._update(),this._request=null,this._running&&this._refresh()})))}_update(t=Date.now()){let e=0;this._charts.forEach(((i,s)=>{if(!i.running||!i.items.length)return;const n=i.items;let o,a=n.length-1,r=!1;for(;a>=0;--a)o=n[a],o._active?(o._total>i.duration&&(i.duration=o._total),o.tick(t),r=!0):(n[a]=n[n.length-1],n.pop());r&&(s.draw(),this._notify(s,i,t,"progress")),n.length||(i.running=!1,this._notify(s,i,t,"complete"),i.initial=!1),e+=n.length})),this._lastDate=t,0===e&&(this._running=!1)}_getAnims(t){const e=this._charts;let i=e.get(t);return i||(i={running:!1,initial:!0,items:[],listeners:{complete:[],progress:[]}},e.set(t,i)),i}listen(t,e,i){this._getAnims(t).listeners[e].push(i)}add(t,e){e&&e.length&&this._getAnims(t).items.push(...e)}has(t){return this._getAnims(t).items.length>0}start(t){const e=this._charts.get(t);e&&(e.running=!0,e.start=Date.now(),e.duration=e.items.reduce(((t,e)=>Math.max(t,e._duration)),0),this._refresh())}running(t){if(!this._running)return!1;const e=this._charts.get(t);return!!(e&&e.running&&e.items.length)}stop(t){const e=this._charts.get(t);if(!e||!e.items.length)return;const i=e.items;let s=i.length-1;for(;s>=0;--s)i[s].cancel();e.items=[],this._notify(t,e,Date.now(),"complete")}remove(t){return this._charts.delete(t)}}var bt=new xt; +/*! + * @kurkle/color v0.3.2 + * https://github.com/kurkle/color#readme + * (c) 2023 Jukka Kurkela + * Released under the MIT License + */function _t(t){return t+.5|0}const yt=(t,e,i)=>Math.max(Math.min(t,i),e);function vt(t){return yt(_t(2.55*t),0,255)}function Mt(t){return yt(_t(255*t),0,255)}function wt(t){return yt(_t(t/2.55)/100,0,1)}function kt(t){return yt(_t(100*t),0,100)}const St={0:0,1:1,2:2,3:3,4:4,5:5,6:6,7:7,8:8,9:9,A:10,B:11,C:12,D:13,E:14,F:15,a:10,b:11,c:12,d:13,e:14,f:15},Pt=[..."0123456789ABCDEF"],Dt=t=>Pt[15&t],Ct=t=>Pt[(240&t)>>4]+Pt[15&t],Ot=t=>(240&t)>>4==(15&t);function At(t){var e=(t=>Ot(t.r)&&Ot(t.g)&&Ot(t.b)&&Ot(t.a))(t)?Dt:Ct;return t?"#"+e(t.r)+e(t.g)+e(t.b)+((t,e)=>t<255?e(t):"")(t.a,e):void 0}const Tt=/^(hsla?|hwb|hsv)\(\s*([-+.e\d]+)(?:deg)?[\s,]+([-+.e\d]+)%[\s,]+([-+.e\d]+)%(?:[\s,]+([-+.e\d]+)(%)?)?\s*\)$/;function Lt(t,e,i){const s=e*Math.min(i,1-i),n=(e,n=(e+t/30)%12)=>i-s*Math.max(Math.min(n-3,9-n,1),-1);return[n(0),n(8),n(4)]}function Et(t,e,i){const s=(s,n=(s+t/60)%6)=>i-i*e*Math.max(Math.min(n,4-n,1),0);return[s(5),s(3),s(1)]}function Rt(t,e,i){const s=Lt(t,1,.5);let n;for(e+i>1&&(n=1/(e+i),e*=n,i*=n),n=0;n<3;n++)s[n]*=1-e-i,s[n]+=e;return s}function It(t){const e=t.r/255,i=t.g/255,s=t.b/255,n=Math.max(e,i,s),o=Math.min(e,i,s),a=(n+o)/2;let r,l,h;return n!==o&&(h=n-o,l=a>.5?h/(2-n-o):h/(n+o),r=function(t,e,i,s,n){return t===n?(e-i)/s+(e>16&255,o>>8&255,255&o]}return t}(),Ht.transparent=[0,0,0,0]);const e=Ht[t.toLowerCase()];return e&&{r:e[0],g:e[1],b:e[2],a:4===e.length?e[3]:255}}const $t=/^rgba?\(\s*([-+.\d]+)(%)?[\s,]+([-+.e\d]+)(%)?[\s,]+([-+.e\d]+)(%)?(?:[\s,/]+([-+.e\d]+)(%)?)?\s*\)$/;const Yt=t=>t<=.0031308?12.92*t:1.055*Math.pow(t,1/2.4)-.055,Ut=t=>t<=.04045?t/12.92:Math.pow((t+.055)/1.055,2.4);function Xt(t,e,i){if(t){let s=It(t);s[e]=Math.max(0,Math.min(s[e]+s[e]*i,0===e?360:1)),s=Ft(s),t.r=s[0],t.g=s[1],t.b=s[2]}}function qt(t,e){return t?Object.assign(e||{},t):t}function Kt(t){var e={r:0,g:0,b:0,a:255};return Array.isArray(t)?t.length>=3&&(e={r:t[0],g:t[1],b:t[2],a:255},t.length>3&&(e.a=Mt(t[3]))):(e=qt(t,{r:0,g:0,b:0,a:1})).a=Mt(e.a),e}function Gt(t){return"r"===t.charAt(0)?function(t){const e=$t.exec(t);let i,s,n,o=255;if(e){if(e[7]!==i){const t=+e[7];o=e[8]?vt(t):yt(255*t,0,255)}return i=+e[1],s=+e[3],n=+e[5],i=255&(e[2]?vt(i):yt(i,0,255)),s=255&(e[4]?vt(s):yt(s,0,255)),n=255&(e[6]?vt(n):yt(n,0,255)),{r:i,g:s,b:n,a:o}}}(t):Bt(t)}class Jt{constructor(t){if(t instanceof Jt)return t;const e=typeof t;let i;var s,n,o;"object"===e?i=Kt(t):"string"===e&&(o=(s=t).length,"#"===s[0]&&(4===o||5===o?n={r:255&17*St[s[1]],g:255&17*St[s[2]],b:255&17*St[s[3]],a:5===o?17*St[s[4]]:255}:7!==o&&9!==o||(n={r:St[s[1]]<<4|St[s[2]],g:St[s[3]]<<4|St[s[4]],b:St[s[5]]<<4|St[s[6]],a:9===o?St[s[7]]<<4|St[s[8]]:255})),i=n||jt(t)||Gt(t)),this._rgb=i,this._valid=!!i}get valid(){return this._valid}get rgb(){var t=qt(this._rgb);return t&&(t.a=wt(t.a)),t}set rgb(t){this._rgb=Kt(t)}rgbString(){return this._valid?(t=this._rgb)&&(t.a<255?`rgba(${t.r}, ${t.g}, ${t.b}, ${wt(t.a)})`:`rgb(${t.r}, ${t.g}, ${t.b})`):void 0;var t}hexString(){return this._valid?At(this._rgb):void 0}hslString(){return this._valid?function(t){if(!t)return;const e=It(t),i=e[0],s=kt(e[1]),n=kt(e[2]);return t.a<255?`hsla(${i}, ${s}%, ${n}%, ${wt(t.a)})`:`hsl(${i}, ${s}%, ${n}%)`}(this._rgb):void 0}mix(t,e){if(t){const i=this.rgb,s=t.rgb;let n;const o=e===n?.5:e,a=2*o-1,r=i.a-s.a,l=((a*r==-1?a:(a+r)/(1+a*r))+1)/2;n=1-l,i.r=255&l*i.r+n*s.r+.5,i.g=255&l*i.g+n*s.g+.5,i.b=255&l*i.b+n*s.b+.5,i.a=o*i.a+(1-o)*s.a,this.rgb=i}return this}interpolate(t,e){return t&&(this._rgb=function(t,e,i){const s=Ut(wt(t.r)),n=Ut(wt(t.g)),o=Ut(wt(t.b));return{r:Mt(Yt(s+i*(Ut(wt(e.r))-s))),g:Mt(Yt(n+i*(Ut(wt(e.g))-n))),b:Mt(Yt(o+i*(Ut(wt(e.b))-o))),a:t.a+i*(e.a-t.a)}}(this._rgb,t._rgb,e)),this}clone(){return new Jt(this.rgb)}alpha(t){return this._rgb.a=Mt(t),this}clearer(t){return this._rgb.a*=1-t,this}greyscale(){const t=this._rgb,e=_t(.3*t.r+.59*t.g+.11*t.b);return t.r=t.g=t.b=e,this}opaquer(t){return this._rgb.a*=1+t,this}negate(){const t=this._rgb;return t.r=255-t.r,t.g=255-t.g,t.b=255-t.b,this}lighten(t){return Xt(this._rgb,2,t),this}darken(t){return Xt(this._rgb,2,-t),this}saturate(t){return Xt(this._rgb,1,t),this}desaturate(t){return Xt(this._rgb,1,-t),this}rotate(t){return function(t,e){var i=It(t);i[0]=Vt(i[0]+e),i=Ft(i),t.r=i[0],t.g=i[1],t.b=i[2]}(this._rgb,t),this}}function Zt(t){if(t&&"object"==typeof t){const e=t.toString();return"[object CanvasPattern]"===e||"[object CanvasGradient]"===e}return!1}function Qt(t){return Zt(t)?t:new Jt(t)}function te(t){return Zt(t)?t:new Jt(t).saturate(.5).darken(.1).hexString()}const ee=["x","y","borderWidth","radius","tension"],ie=["color","borderColor","backgroundColor"];const se=new Map;function ne(t,e,i){return function(t,e){e=e||{};const i=t+JSON.stringify(e);let s=se.get(i);return s||(s=new Intl.NumberFormat(t,e),se.set(i,s)),s}(e,i).format(t)}const oe={values:t=>n(t)?t:""+t,numeric(t,e,i){if(0===t)return"0";const s=this.chart.options.locale;let n,o=t;if(i.length>1){const e=Math.max(Math.abs(i[0].value),Math.abs(i[i.length-1].value));(e<1e-4||e>1e15)&&(n="scientific"),o=function(t,e){let i=e.length>3?e[2].value-e[1].value:e[1].value-e[0].value;Math.abs(i)>=1&&t!==Math.floor(t)&&(i=t-Math.floor(t));return i}(t,i)}const a=z(Math.abs(o)),r=isNaN(a)?1:Math.max(Math.min(-1*Math.floor(a),20),0),l={notation:n,minimumFractionDigits:r,maximumFractionDigits:r};return Object.assign(l,this.options.ticks.format),ne(t,s,l)},logarithmic(t,e,i){if(0===t)return"0";const s=i[e].significand||t/Math.pow(10,Math.floor(z(t)));return[1,2,3,5,10,15].includes(s)||e>.8*i.length?oe.numeric.call(this,t,e,i):""}};var ae={formatters:oe};const re=Object.create(null),le=Object.create(null);function he(t,e){if(!e)return t;const i=e.split(".");for(let e=0,s=i.length;et.chart.platform.getDevicePixelRatio(),this.elements={},this.events=["mousemove","mouseout","click","touchstart","touchmove"],this.font={family:"'Helvetica Neue', 'Helvetica', 'Arial', sans-serif",size:12,style:"normal",lineHeight:1.2,weight:null},this.hover={},this.hoverBackgroundColor=(t,e)=>te(e.backgroundColor),this.hoverBorderColor=(t,e)=>te(e.borderColor),this.hoverColor=(t,e)=>te(e.color),this.indexAxis="x",this.interaction={mode:"nearest",intersect:!0,includeInvisible:!1},this.maintainAspectRatio=!0,this.onHover=null,this.onClick=null,this.parsing=!0,this.plugins={},this.responsive=!0,this.scale=void 0,this.scales={},this.showLine=!0,this.drawActiveElementsOnTop=!0,this.describe(t),this.apply(e)}set(t,e){return ce(this,t,e)}get(t){return he(this,t)}describe(t,e){return ce(le,t,e)}override(t,e){return ce(re,t,e)}route(t,e,i,s){const n=he(this,t),a=he(this,i),r="_"+e;Object.defineProperties(n,{[r]:{value:n[e],writable:!0},[e]:{enumerable:!0,get(){const t=this[r],e=a[s];return o(t)?Object.assign({},e,t):l(t,e)},set(t){this[r]=t}}})}apply(t){t.forEach((t=>t(this)))}}var ue=new de({_scriptable:t=>!t.startsWith("on"),_indexable:t=>"events"!==t,hover:{_fallback:"interaction"},interaction:{_scriptable:!1,_indexable:!1}},[function(t){t.set("animation",{delay:void 0,duration:1e3,easing:"easeOutQuart",fn:void 0,from:void 0,loop:void 0,to:void 0,type:void 0}),t.describe("animation",{_fallback:!1,_indexable:!1,_scriptable:t=>"onProgress"!==t&&"onComplete"!==t&&"fn"!==t}),t.set("animations",{colors:{type:"color",properties:ie},numbers:{type:"number",properties:ee}}),t.describe("animations",{_fallback:"animation"}),t.set("transitions",{active:{animation:{duration:400}},resize:{animation:{duration:0}},show:{animations:{colors:{from:"transparent"},visible:{type:"boolean",duration:0}}},hide:{animations:{colors:{to:"transparent"},visible:{type:"boolean",easing:"linear",fn:t=>0|t}}}})},function(t){t.set("layout",{autoPadding:!0,padding:{top:0,right:0,bottom:0,left:0}})},function(t){t.set("scale",{display:!0,offset:!1,reverse:!1,beginAtZero:!1,bounds:"ticks",clip:!0,grace:0,grid:{display:!0,lineWidth:1,drawOnChartArea:!0,drawTicks:!0,tickLength:8,tickWidth:(t,e)=>e.lineWidth,tickColor:(t,e)=>e.color,offset:!1},border:{display:!0,dash:[],dashOffset:0,width:1},title:{display:!1,text:"",padding:{top:4,bottom:4}},ticks:{minRotation:0,maxRotation:50,mirror:!1,textStrokeWidth:0,textStrokeColor:"",padding:3,display:!0,autoSkip:!0,autoSkipPadding:3,labelOffset:0,callback:ae.formatters.values,minor:{},major:{},align:"center",crossAlign:"near",showLabelBackdrop:!1,backdropColor:"rgba(255, 255, 255, 0.75)",backdropPadding:2}}),t.route("scale.ticks","color","","color"),t.route("scale.grid","color","","borderColor"),t.route("scale.border","color","","borderColor"),t.route("scale.title","color","","color"),t.describe("scale",{_fallback:!1,_scriptable:t=>!t.startsWith("before")&&!t.startsWith("after")&&"callback"!==t&&"parser"!==t,_indexable:t=>"borderDash"!==t&&"tickBorderDash"!==t&&"dash"!==t}),t.describe("scales",{_fallback:"scale"}),t.describe("scale.ticks",{_scriptable:t=>"backdropPadding"!==t&&"callback"!==t,_indexable:t=>"backdropPadding"!==t})}]);function fe(){return"undefined"!=typeof window&&"undefined"!=typeof document}function ge(t){let e=t.parentNode;return e&&"[object ShadowRoot]"===e.toString()&&(e=e.host),e}function pe(t,e,i){let s;return"string"==typeof t?(s=parseInt(t,10),-1!==t.indexOf("%")&&(s=s/100*e.parentNode[i])):s=t,s}const me=t=>t.ownerDocument.defaultView.getComputedStyle(t,null);function xe(t,e){return me(t).getPropertyValue(e)}const be=["top","right","bottom","left"];function _e(t,e,i){const s={};i=i?"-"+i:"";for(let n=0;n<4;n++){const o=be[n];s[o]=parseFloat(t[e+"-"+o+i])||0}return s.width=s.left+s.right,s.height=s.top+s.bottom,s}const ye=(t,e,i)=>(t>0||e>0)&&(!i||!i.shadowRoot);function ve(t,e){if("native"in t)return t;const{canvas:i,currentDevicePixelRatio:s}=e,n=me(i),o="border-box"===n.boxSizing,a=_e(n,"padding"),r=_e(n,"border","width"),{x:l,y:h,box:c}=function(t,e){const i=t.touches,s=i&&i.length?i[0]:t,{offsetX:n,offsetY:o}=s;let a,r,l=!1;if(ye(n,o,t.target))a=n,r=o;else{const t=e.getBoundingClientRect();a=s.clientX-t.left,r=s.clientY-t.top,l=!0}return{x:a,y:r,box:l}}(t,i),d=a.left+(c&&r.left),u=a.top+(c&&r.top);let{width:f,height:g}=e;return o&&(f-=a.width+r.width,g-=a.height+r.height),{x:Math.round((l-d)/f*i.width/s),y:Math.round((h-u)/g*i.height/s)}}const Me=t=>Math.round(10*t)/10;function we(t,e,i,s){const n=me(t),o=_e(n,"margin"),a=pe(n.maxWidth,t,"clientWidth")||T,r=pe(n.maxHeight,t,"clientHeight")||T,l=function(t,e,i){let s,n;if(void 0===e||void 0===i){const o=t&&ge(t);if(o){const t=o.getBoundingClientRect(),a=me(o),r=_e(a,"border","width"),l=_e(a,"padding");e=t.width-l.width-r.width,i=t.height-l.height-r.height,s=pe(a.maxWidth,o,"clientWidth"),n=pe(a.maxHeight,o,"clientHeight")}else e=t.clientWidth,i=t.clientHeight}return{width:e,height:i,maxWidth:s||T,maxHeight:n||T}}(t,e,i);let{width:h,height:c}=l;if("content-box"===n.boxSizing){const t=_e(n,"border","width"),e=_e(n,"padding");h-=e.width+t.width,c-=e.height+t.height}h=Math.max(0,h-o.width),c=Math.max(0,s?h/s:c-o.height),h=Me(Math.min(h,a,l.maxWidth)),c=Me(Math.min(c,r,l.maxHeight)),h&&!c&&(c=Me(h/2));return(void 0!==e||void 0!==i)&&s&&l.height&&c>l.height&&(c=l.height,h=Me(Math.floor(c*s))),{width:h,height:c}}function ke(t,e,i){const s=e||1,n=Me(t.height*s),o=Me(t.width*s);t.height=Me(t.height),t.width=Me(t.width);const a=t.canvas;return a.style&&(i||!a.style.height&&!a.style.width)&&(a.style.height=`${t.height}px`,a.style.width=`${t.width}px`),(t.currentDevicePixelRatio!==s||a.height!==n||a.width!==o)&&(t.currentDevicePixelRatio=s,a.height=n,a.width=o,t.ctx.setTransform(s,0,0,s,0,0),!0)}const Se=function(){let t=!1;try{const e={get passive(){return t=!0,!1}};fe()&&(window.addEventListener("test",null,e),window.removeEventListener("test",null,e))}catch(t){}return t}();function Pe(t,e){const i=xe(t,e),s=i&&i.match(/^(\d+)(\.\d+)?px$/);return s?+s[1]:void 0}function De(t){return!t||s(t.size)||s(t.family)?null:(t.style?t.style+" ":"")+(t.weight?t.weight+" ":"")+t.size+"px "+t.family}function Ce(t,e,i,s,n){let o=e[n];return o||(o=e[n]=t.measureText(n).width,i.push(n)),o>s&&(s=o),s}function Oe(t,e,i,s){let o=(s=s||{}).data=s.data||{},a=s.garbageCollect=s.garbageCollect||[];s.font!==e&&(o=s.data={},a=s.garbageCollect=[],s.font=e),t.save(),t.font=e;let r=0;const l=i.length;let h,c,d,u,f;for(h=0;hi.length){for(h=0;h0&&t.stroke()}}function Re(t,e,i){return i=i||.5,!e||t&&t.x>e.left-i&&t.xe.top-i&&t.y0&&""!==r.strokeColor;let c,d;for(t.save(),t.font=a.string,function(t,e){e.translation&&t.translate(e.translation[0],e.translation[1]),s(e.rotation)||t.rotate(e.rotation),e.color&&(t.fillStyle=e.color),e.textAlign&&(t.textAlign=e.textAlign),e.textBaseline&&(t.textBaseline=e.textBaseline)}(t,r),c=0;ct[0])){const o=i||t;void 0===s&&(s=ti("_fallback",t));const a={[Symbol.toStringTag]:"Object",_cacheable:!0,_scopes:t,_rootScopes:o,_fallback:s,_getTarget:n,override:i=>je([i,...t],e,o,s)};return new Proxy(a,{deleteProperty:(e,i)=>(delete e[i],delete e._keys,delete t[0][i],!0),get:(i,s)=>qe(i,s,(()=>function(t,e,i,s){let n;for(const o of e)if(n=ti(Ue(o,t),i),void 0!==n)return Xe(t,n)?Ze(i,s,t,n):n}(s,e,t,i))),getOwnPropertyDescriptor:(t,e)=>Reflect.getOwnPropertyDescriptor(t._scopes[0],e),getPrototypeOf:()=>Reflect.getPrototypeOf(t[0]),has:(t,e)=>ei(t).includes(e),ownKeys:t=>ei(t),set(t,e,i){const s=t._storage||(t._storage=n());return t[e]=s[e]=i,delete t._keys,!0}})}function $e(t,e,i,s){const a={_cacheable:!1,_proxy:t,_context:e,_subProxy:i,_stack:new Set,_descriptors:Ye(t,s),setContext:e=>$e(t,e,i,s),override:n=>$e(t.override(n),e,i,s)};return new Proxy(a,{deleteProperty:(e,i)=>(delete e[i],delete t[i],!0),get:(t,e,i)=>qe(t,e,(()=>function(t,e,i){const{_proxy:s,_context:a,_subProxy:r,_descriptors:l}=t;let h=s[e];S(h)&&l.isScriptable(e)&&(h=function(t,e,i,s){const{_proxy:n,_context:o,_subProxy:a,_stack:r}=i;if(r.has(t))throw new Error("Recursion detected: "+Array.from(r).join("->")+"->"+t);r.add(t);let l=e(o,a||s);r.delete(t),Xe(t,l)&&(l=Ze(n._scopes,n,t,l));return l}(e,h,t,i));n(h)&&h.length&&(h=function(t,e,i,s){const{_proxy:n,_context:a,_subProxy:r,_descriptors:l}=i;if(void 0!==a.index&&s(t))return e[a.index%e.length];if(o(e[0])){const i=e,s=n._scopes.filter((t=>t!==i));e=[];for(const o of i){const i=Ze(s,n,t,o);e.push($e(i,a,r&&r[t],l))}}return e}(e,h,t,l.isIndexable));Xe(e,h)&&(h=$e(h,a,r&&r[e],l));return h}(t,e,i))),getOwnPropertyDescriptor:(e,i)=>e._descriptors.allKeys?Reflect.has(t,i)?{enumerable:!0,configurable:!0}:void 0:Reflect.getOwnPropertyDescriptor(t,i),getPrototypeOf:()=>Reflect.getPrototypeOf(t),has:(e,i)=>Reflect.has(t,i),ownKeys:()=>Reflect.ownKeys(t),set:(e,i,s)=>(t[i]=s,delete e[i],!0)})}function Ye(t,e={scriptable:!0,indexable:!0}){const{_scriptable:i=e.scriptable,_indexable:s=e.indexable,_allKeys:n=e.allKeys}=t;return{allKeys:n,scriptable:i,indexable:s,isScriptable:S(i)?i:()=>i,isIndexable:S(s)?s:()=>s}}const Ue=(t,e)=>t?t+w(e):e,Xe=(t,e)=>o(e)&&"adapters"!==t&&(null===Object.getPrototypeOf(e)||e.constructor===Object);function qe(t,e,i){if(Object.prototype.hasOwnProperty.call(t,e)||"constructor"===e)return t[e];const s=i();return t[e]=s,s}function Ke(t,e,i){return S(t)?t(e,i):t}const Ge=(t,e)=>!0===t?e:"string"==typeof t?M(e,t):void 0;function Je(t,e,i,s,n){for(const o of e){const e=Ge(i,o);if(e){t.add(e);const o=Ke(e._fallback,i,n);if(void 0!==o&&o!==i&&o!==s)return o}else if(!1===e&&void 0!==s&&i!==s)return null}return!1}function Ze(t,e,i,s){const a=e._rootScopes,r=Ke(e._fallback,i,s),l=[...t,...a],h=new Set;h.add(s);let c=Qe(h,l,i,r||i,s);return null!==c&&((void 0===r||r===i||(c=Qe(h,l,r,c,s),null!==c))&&je(Array.from(h),[""],a,r,(()=>function(t,e,i){const s=t._getTarget();e in s||(s[e]={});const a=s[e];if(n(a)&&o(i))return i;return a||{}}(e,i,s))))}function Qe(t,e,i,s,n){for(;i;)i=Je(t,e,i,s,n);return i}function ti(t,e){for(const i of e){if(!i)continue;const e=i[t];if(void 0!==e)return e}}function ei(t){let e=t._keys;return e||(e=t._keys=function(t){const e=new Set;for(const i of t)for(const t of Object.keys(i).filter((t=>!t.startsWith("_"))))e.add(t);return Array.from(e)}(t._scopes)),e}function ii(t,e,i,s){const{iScale:n}=t,{key:o="r"}=this._parsing,a=new Array(s);let r,l,h,c;for(r=0,l=s;re"x"===t?"y":"x";function ai(t,e,i,s){const n=t.skip?e:t,o=e,a=i.skip?e:i,r=q(o,n),l=q(a,o);let h=r/(r+l),c=l/(r+l);h=isNaN(h)?0:h,c=isNaN(c)?0:c;const d=s*h,u=s*c;return{previous:{x:o.x-d*(a.x-n.x),y:o.y-d*(a.y-n.y)},next:{x:o.x+u*(a.x-n.x),y:o.y+u*(a.y-n.y)}}}function ri(t,e="x"){const i=oi(e),s=t.length,n=Array(s).fill(0),o=Array(s);let a,r,l,h=ni(t,0);for(a=0;a!t.skip))),"monotone"===e.cubicInterpolationMode)ri(t,n);else{let i=s?t[t.length-1]:t[0];for(o=0,a=t.length;o0===t||1===t,di=(t,e,i)=>-Math.pow(2,10*(t-=1))*Math.sin((t-e)*O/i),ui=(t,e,i)=>Math.pow(2,-10*t)*Math.sin((t-e)*O/i)+1,fi={linear:t=>t,easeInQuad:t=>t*t,easeOutQuad:t=>-t*(t-2),easeInOutQuad:t=>(t/=.5)<1?.5*t*t:-.5*(--t*(t-2)-1),easeInCubic:t=>t*t*t,easeOutCubic:t=>(t-=1)*t*t+1,easeInOutCubic:t=>(t/=.5)<1?.5*t*t*t:.5*((t-=2)*t*t+2),easeInQuart:t=>t*t*t*t,easeOutQuart:t=>-((t-=1)*t*t*t-1),easeInOutQuart:t=>(t/=.5)<1?.5*t*t*t*t:-.5*((t-=2)*t*t*t-2),easeInQuint:t=>t*t*t*t*t,easeOutQuint:t=>(t-=1)*t*t*t*t+1,easeInOutQuint:t=>(t/=.5)<1?.5*t*t*t*t*t:.5*((t-=2)*t*t*t*t+2),easeInSine:t=>1-Math.cos(t*E),easeOutSine:t=>Math.sin(t*E),easeInOutSine:t=>-.5*(Math.cos(C*t)-1),easeInExpo:t=>0===t?0:Math.pow(2,10*(t-1)),easeOutExpo:t=>1===t?1:1-Math.pow(2,-10*t),easeInOutExpo:t=>ci(t)?t:t<.5?.5*Math.pow(2,10*(2*t-1)):.5*(2-Math.pow(2,-10*(2*t-1))),easeInCirc:t=>t>=1?t:-(Math.sqrt(1-t*t)-1),easeOutCirc:t=>Math.sqrt(1-(t-=1)*t),easeInOutCirc:t=>(t/=.5)<1?-.5*(Math.sqrt(1-t*t)-1):.5*(Math.sqrt(1-(t-=2)*t)+1),easeInElastic:t=>ci(t)?t:di(t,.075,.3),easeOutElastic:t=>ci(t)?t:ui(t,.075,.3),easeInOutElastic(t){const e=.1125;return ci(t)?t:t<.5?.5*di(2*t,e,.45):.5+.5*ui(2*t-1,e,.45)},easeInBack(t){const e=1.70158;return t*t*((e+1)*t-e)},easeOutBack(t){const e=1.70158;return(t-=1)*t*((e+1)*t+e)+1},easeInOutBack(t){let e=1.70158;return(t/=.5)<1?t*t*((1+(e*=1.525))*t-e)*.5:.5*((t-=2)*t*((1+(e*=1.525))*t+e)+2)},easeInBounce:t=>1-fi.easeOutBounce(1-t),easeOutBounce(t){const e=7.5625,i=2.75;return t<1/i?e*t*t:t<2/i?e*(t-=1.5/i)*t+.75:t<2.5/i?e*(t-=2.25/i)*t+.9375:e*(t-=2.625/i)*t+.984375},easeInOutBounce:t=>t<.5?.5*fi.easeInBounce(2*t):.5*fi.easeOutBounce(2*t-1)+.5};function gi(t,e,i,s){return{x:t.x+i*(e.x-t.x),y:t.y+i*(e.y-t.y)}}function pi(t,e,i,s){return{x:t.x+i*(e.x-t.x),y:"middle"===s?i<.5?t.y:e.y:"after"===s?i<1?t.y:e.y:i>0?e.y:t.y}}function mi(t,e,i,s){const n={x:t.cp2x,y:t.cp2y},o={x:e.cp1x,y:e.cp1y},a=gi(t,n,i),r=gi(n,o,i),l=gi(o,e,i),h=gi(a,r,i),c=gi(r,l,i);return gi(h,c,i)}const xi=/^(normal|(\d+(?:\.\d+)?)(px|em|%)?)$/,bi=/^(normal|italic|initial|inherit|unset|(oblique( -?[0-9]?[0-9]deg)?))$/;function _i(t,e){const i=(""+t).match(xi);if(!i||"normal"===i[1])return 1.2*e;switch(t=+i[2],i[3]){case"px":return t;case"%":t/=100}return e*t}const yi=t=>+t||0;function vi(t,e){const i={},s=o(e),n=s?Object.keys(e):e,a=o(t)?s?i=>l(t[i],t[e[i]]):e=>t[e]:()=>t;for(const t of n)i[t]=yi(a(t));return i}function Mi(t){return vi(t,{top:"y",right:"x",bottom:"y",left:"x"})}function wi(t){return vi(t,["topLeft","topRight","bottomLeft","bottomRight"])}function ki(t){const e=Mi(t);return e.width=e.left+e.right,e.height=e.top+e.bottom,e}function Si(t,e){t=t||{},e=e||ue.font;let i=l(t.size,e.size);"string"==typeof i&&(i=parseInt(i,10));let s=l(t.style,e.style);s&&!(""+s).match(bi)&&(console.warn('Invalid font style specified: "'+s+'"'),s=void 0);const n={family:l(t.family,e.family),lineHeight:_i(l(t.lineHeight,e.lineHeight),i),size:i,style:s,weight:l(t.weight,e.weight),string:""};return n.string=De(n),n}function Pi(t,e,i,s){let o,a,r,l=!0;for(o=0,a=t.length;oi&&0===t?0:t+e;return{min:a(s,-Math.abs(o)),max:a(n,o)}}function Ci(t,e){return Object.assign(Object.create(t),e)}function Oi(t,e,i){return t?function(t,e){return{x:i=>t+t+e-i,setWidth(t){e=t},textAlign:t=>"center"===t?t:"right"===t?"left":"right",xPlus:(t,e)=>t-e,leftForLtr:(t,e)=>t-e}}(e,i):{x:t=>t,setWidth(t){},textAlign:t=>t,xPlus:(t,e)=>t+e,leftForLtr:(t,e)=>t}}function Ai(t,e){let i,s;"ltr"!==e&&"rtl"!==e||(i=t.canvas.style,s=[i.getPropertyValue("direction"),i.getPropertyPriority("direction")],i.setProperty("direction",e,"important"),t.prevTextDirection=s)}function Ti(t,e){void 0!==e&&(delete t.prevTextDirection,t.canvas.style.setProperty("direction",e[0],e[1]))}function Li(t){return"angle"===t?{between:J,compare:K,normalize:G}:{between:tt,compare:(t,e)=>t-e,normalize:t=>t}}function Ei({start:t,end:e,count:i,loop:s,style:n}){return{start:t%i,end:e%i,loop:s&&(e-t+1)%i==0,style:n}}function Ri(t,e,i){if(!i)return[t];const{property:s,start:n,end:o}=i,a=e.length,{compare:r,between:l,normalize:h}=Li(s),{start:c,end:d,loop:u,style:f}=function(t,e,i){const{property:s,start:n,end:o}=i,{between:a,normalize:r}=Li(s),l=e.length;let h,c,{start:d,end:u,loop:f}=t;if(f){for(d+=l,u+=l,h=0,c=l;hb||l(n,x,p)&&0!==r(n,x),v=()=>!b||0===r(o,p)||l(o,x,p);for(let t=c,i=c;t<=d;++t)m=e[t%a],m.skip||(p=h(m[s]),p!==x&&(b=l(p,n,o),null===_&&y()&&(_=0===r(p,n)?t:i),null!==_&&v()&&(g.push(Ei({start:_,end:t,loop:u,count:a,style:f})),_=null),i=t,x=p));return null!==_&&g.push(Ei({start:_,end:d,loop:u,count:a,style:f})),g}function Ii(t,e){const i=[],s=t.segments;for(let n=0;nn&&t[o%e].skip;)o--;return o%=e,{start:n,end:o}}(i,n,o,s);if(!0===s)return Fi(t,[{start:a,end:r,loop:o}],i,e);return Fi(t,function(t,e,i,s){const n=t.length,o=[];let a,r=e,l=t[e];for(a=e+1;a<=i;++a){const i=t[a%n];i.skip||i.stop?l.skip||(s=!1,o.push({start:e%n,end:(a-1)%n,loop:s}),e=r=i.stop?a:null):(r=a,l.skip&&(e=a)),l=i}return null!==r&&o.push({start:e%n,end:r%n,loop:s}),o}(i,a,r!s(t[e.axis])));n.lo-=Math.max(0,a);const r=i.slice(n.hi).findIndex((t=>!s(t[e.axis])));n.hi+=Math.max(0,r)}return n}if(o._sharedOptions){const t=a[0],s="function"==typeof t.getRange&&t.getRange(e);if(s){const t=r(a,e,i-s),n=r(a,e,i+s);return{lo:t.lo,hi:n.hi}}}}return{lo:0,hi:a.length-1}}function $i(t,e,i,s,n){const o=t.getSortedVisibleDatasetMetas(),a=i[e];for(let t=0,i=o.length;t{t[a]&&t[a](e[i],n)&&(o.push({element:t,datasetIndex:s,index:l}),r=r||t.inRange(e.x,e.y,n))})),s&&!r?[]:o}var Ki={evaluateInteractionItems:$i,modes:{index(t,e,i,s){const n=ve(e,t),o=i.axis||"x",a=i.includeInvisible||!1,r=i.intersect?Yi(t,n,o,s,a):Xi(t,n,o,!1,s,a),l=[];return r.length?(t.getSortedVisibleDatasetMetas().forEach((t=>{const e=r[0].index,i=t.data[e];i&&!i.skip&&l.push({element:i,datasetIndex:t.index,index:e})})),l):[]},dataset(t,e,i,s){const n=ve(e,t),o=i.axis||"xy",a=i.includeInvisible||!1;let r=i.intersect?Yi(t,n,o,s,a):Xi(t,n,o,!1,s,a);if(r.length>0){const e=r[0].datasetIndex,i=t.getDatasetMeta(e).data;r=[];for(let t=0;tYi(t,ve(e,t),i.axis||"xy",s,i.includeInvisible||!1),nearest(t,e,i,s){const n=ve(e,t),o=i.axis||"xy",a=i.includeInvisible||!1;return Xi(t,n,o,i.intersect,s,a)},x:(t,e,i,s)=>qi(t,ve(e,t),"x",i.intersect,s),y:(t,e,i,s)=>qi(t,ve(e,t),"y",i.intersect,s)}};const Gi=["left","top","right","bottom"];function Ji(t,e){return t.filter((t=>t.pos===e))}function Zi(t,e){return t.filter((t=>-1===Gi.indexOf(t.pos)&&t.box.axis===e))}function Qi(t,e){return t.sort(((t,i)=>{const s=e?i:t,n=e?t:i;return s.weight===n.weight?s.index-n.index:s.weight-n.weight}))}function ts(t,e){const i=function(t){const e={};for(const i of t){const{stack:t,pos:s,stackWeight:n}=i;if(!t||!Gi.includes(s))continue;const o=e[t]||(e[t]={count:0,placed:0,weight:0,size:0});o.count++,o.weight+=n}return e}(t),{vBoxMaxWidth:s,hBoxMaxHeight:n}=e;let o,a,r;for(o=0,a=t.length;o{s[t]=Math.max(e[t],i[t])})),s}return s(t?["left","right"]:["top","bottom"])}function os(t,e,i,s){const n=[];let o,a,r,l,h,c;for(o=0,a=t.length,h=0;ot.box.fullSize)),!0),s=Qi(Ji(e,"left"),!0),n=Qi(Ji(e,"right")),o=Qi(Ji(e,"top"),!0),a=Qi(Ji(e,"bottom")),r=Zi(e,"x"),l=Zi(e,"y");return{fullSize:i,leftAndTop:s.concat(o),rightAndBottom:n.concat(l).concat(a).concat(r),chartArea:Ji(e,"chartArea"),vertical:s.concat(n).concat(l),horizontal:o.concat(a).concat(r)}}(t.boxes),l=r.vertical,h=r.horizontal;u(t.boxes,(t=>{"function"==typeof t.beforeLayout&&t.beforeLayout()}));const c=l.reduce(((t,e)=>e.box.options&&!1===e.box.options.display?t:t+1),0)||1,d=Object.freeze({outerWidth:e,outerHeight:i,padding:n,availableWidth:o,availableHeight:a,vBoxMaxWidth:o/2/c,hBoxMaxHeight:a/2}),f=Object.assign({},n);is(f,ki(s));const g=Object.assign({maxPadding:f,w:o,h:a,x:n.left,y:n.top},n),p=ts(l.concat(h),d);os(r.fullSize,g,d,p),os(l,g,d,p),os(h,g,d,p)&&os(l,g,d,p),function(t){const e=t.maxPadding;function i(i){const s=Math.max(e[i]-t[i],0);return t[i]+=s,s}t.y+=i("top"),t.x+=i("left"),i("right"),i("bottom")}(g),rs(r.leftAndTop,g,d,p),g.x+=g.w,g.y+=g.h,rs(r.rightAndBottom,g,d,p),t.chartArea={left:g.left,top:g.top,right:g.left+g.w,bottom:g.top+g.h,height:g.h,width:g.w},u(r.chartArea,(e=>{const i=e.box;Object.assign(i,t.chartArea),i.update(g.w,g.h,{left:0,top:0,right:0,bottom:0})}))}};class hs{acquireContext(t,e){}releaseContext(t){return!1}addEventListener(t,e,i){}removeEventListener(t,e,i){}getDevicePixelRatio(){return 1}getMaximumSize(t,e,i,s){return e=Math.max(0,e||t.width),i=i||t.height,{width:e,height:Math.max(0,s?Math.floor(e/s):i)}}isAttached(t){return!0}updateConfig(t){}}class cs extends hs{acquireContext(t){return t&&t.getContext&&t.getContext("2d")||null}updateConfig(t){t.options.animation=!1}}const ds="$chartjs",us={touchstart:"mousedown",touchmove:"mousemove",touchend:"mouseup",pointerenter:"mouseenter",pointerdown:"mousedown",pointermove:"mousemove",pointerup:"mouseup",pointerleave:"mouseout",pointerout:"mouseout"},fs=t=>null===t||""===t;const gs=!!Se&&{passive:!0};function ps(t,e,i){t&&t.canvas&&t.canvas.removeEventListener(e,i,gs)}function ms(t,e){for(const i of t)if(i===e||i.contains(e))return!0}function xs(t,e,i){const s=t.canvas,n=new MutationObserver((t=>{let e=!1;for(const i of t)e=e||ms(i.addedNodes,s),e=e&&!ms(i.removedNodes,s);e&&i()}));return n.observe(document,{childList:!0,subtree:!0}),n}function bs(t,e,i){const s=t.canvas,n=new MutationObserver((t=>{let e=!1;for(const i of t)e=e||ms(i.removedNodes,s),e=e&&!ms(i.addedNodes,s);e&&i()}));return n.observe(document,{childList:!0,subtree:!0}),n}const _s=new Map;let ys=0;function vs(){const t=window.devicePixelRatio;t!==ys&&(ys=t,_s.forEach(((e,i)=>{i.currentDevicePixelRatio!==t&&e()})))}function Ms(t,e,i){const s=t.canvas,n=s&&ge(s);if(!n)return;const o=ct(((t,e)=>{const s=n.clientWidth;i(t,e),s{const e=t[0],i=e.contentRect.width,s=e.contentRect.height;0===i&&0===s||o(i,s)}));return a.observe(n),function(t,e){_s.size||window.addEventListener("resize",vs),_s.set(t,e)}(t,o),a}function ws(t,e,i){i&&i.disconnect(),"resize"===e&&function(t){_s.delete(t),_s.size||window.removeEventListener("resize",vs)}(t)}function ks(t,e,i){const s=t.canvas,n=ct((e=>{null!==t.ctx&&i(function(t,e){const i=us[t.type]||t.type,{x:s,y:n}=ve(t,e);return{type:i,chart:e,native:t,x:void 0!==s?s:null,y:void 0!==n?n:null}}(e,t))}),t);return function(t,e,i){t&&t.addEventListener(e,i,gs)}(s,e,n),n}class Ss extends hs{acquireContext(t,e){const i=t&&t.getContext&&t.getContext("2d");return i&&i.canvas===t?(function(t,e){const i=t.style,s=t.getAttribute("height"),n=t.getAttribute("width");if(t[ds]={initial:{height:s,width:n,style:{display:i.display,height:i.height,width:i.width}}},i.display=i.display||"block",i.boxSizing=i.boxSizing||"border-box",fs(n)){const e=Pe(t,"width");void 0!==e&&(t.width=e)}if(fs(s))if(""===t.style.height)t.height=t.width/(e||2);else{const e=Pe(t,"height");void 0!==e&&(t.height=e)}}(t,e),i):null}releaseContext(t){const e=t.canvas;if(!e[ds])return!1;const i=e[ds].initial;["height","width"].forEach((t=>{const n=i[t];s(n)?e.removeAttribute(t):e.setAttribute(t,n)}));const n=i.style||{};return Object.keys(n).forEach((t=>{e.style[t]=n[t]})),e.width=e.width,delete e[ds],!0}addEventListener(t,e,i){this.removeEventListener(t,e);const s=t.$proxies||(t.$proxies={}),n={attach:xs,detach:bs,resize:Ms}[e]||ks;s[e]=n(t,e,i)}removeEventListener(t,e){const i=t.$proxies||(t.$proxies={}),s=i[e];if(!s)return;({attach:ws,detach:ws,resize:ws}[e]||ps)(t,e,s),i[e]=void 0}getDevicePixelRatio(){return window.devicePixelRatio}getMaximumSize(t,e,i,s){return we(t,e,i,s)}isAttached(t){const e=t&&ge(t);return!(!e||!e.isConnected)}}function Ps(t){return!fe()||"undefined"!=typeof OffscreenCanvas&&t instanceof OffscreenCanvas?cs:Ss}var Ds=Object.freeze({__proto__:null,BasePlatform:hs,BasicPlatform:cs,DomPlatform:Ss,_detectPlatform:Ps});const Cs="transparent",Os={boolean:(t,e,i)=>i>.5?e:t,color(t,e,i){const s=Qt(t||Cs),n=s.valid&&Qt(e||Cs);return n&&n.valid?n.mix(s,i).hexString():e},number:(t,e,i)=>t+(e-t)*i};class As{constructor(t,e,i,s){const n=e[i];s=Pi([t.to,s,n,t.from]);const o=Pi([t.from,n,s]);this._active=!0,this._fn=t.fn||Os[t.type||typeof o],this._easing=fi[t.easing]||fi.linear,this._start=Math.floor(Date.now()+(t.delay||0)),this._duration=this._total=Math.floor(t.duration),this._loop=!!t.loop,this._target=e,this._prop=i,this._from=o,this._to=s,this._promises=void 0}active(){return this._active}update(t,e,i){if(this._active){this._notify(!1);const s=this._target[this._prop],n=i-this._start,o=this._duration-n;this._start=i,this._duration=Math.floor(Math.max(o,t.duration)),this._total+=n,this._loop=!!t.loop,this._to=Pi([t.to,e,s,t.from]),this._from=Pi([t.from,s,e])}}cancel(){this._active&&(this.tick(Date.now()),this._active=!1,this._notify(!1))}tick(t){const e=t-this._start,i=this._duration,s=this._prop,n=this._from,o=this._loop,a=this._to;let r;if(this._active=n!==a&&(o||e1?2-r:r,r=this._easing(Math.min(1,Math.max(0,r))),this._target[s]=this._fn(n,a,r))}wait(){const t=this._promises||(this._promises=[]);return new Promise(((e,i)=>{t.push({res:e,rej:i})}))}_notify(t){const e=t?"res":"rej",i=this._promises||[];for(let t=0;t{const a=t[s];if(!o(a))return;const r={};for(const t of e)r[t]=a[t];(n(a.properties)&&a.properties||[s]).forEach((t=>{t!==s&&i.has(t)||i.set(t,r)}))}))}_animateOptions(t,e){const i=e.options,s=function(t,e){if(!e)return;let i=t.options;if(!i)return void(t.options=e);i.$shared&&(t.options=i=Object.assign({},i,{$shared:!1,$animations:{}}));return i}(t,i);if(!s)return[];const n=this._createAnimations(s,i);return i.$shared&&function(t,e){const i=[],s=Object.keys(e);for(let e=0;e{t.options=i}),(()=>{})),n}_createAnimations(t,e){const i=this._properties,s=[],n=t.$animations||(t.$animations={}),o=Object.keys(e),a=Date.now();let r;for(r=o.length-1;r>=0;--r){const l=o[r];if("$"===l.charAt(0))continue;if("options"===l){s.push(...this._animateOptions(t,e));continue}const h=e[l];let c=n[l];const d=i.get(l);if(c){if(d&&c.active()){c.update(d,h,a);continue}c.cancel()}d&&d.duration?(n[l]=c=new As(d,t,l,h),s.push(c)):t[l]=h}return s}update(t,e){if(0===this._properties.size)return void Object.assign(t,e);const i=this._createAnimations(t,e);return i.length?(bt.add(this._chart,i),!0):void 0}}function Ls(t,e){const i=t&&t.options||{},s=i.reverse,n=void 0===i.min?e:0,o=void 0===i.max?e:0;return{start:s?o:n,end:s?n:o}}function Es(t,e){const i=[],s=t._getSortedDatasetMetas(e);let n,o;for(n=0,o=s.length;n0||!i&&e<0)return n.index}return null}function Vs(t,e){const{chart:i,_cachedMeta:s}=t,n=i._stacks||(i._stacks={}),{iScale:o,vScale:a,index:r}=s,l=o.axis,h=a.axis,c=function(t,e,i){return`${t.id}.${e.id}.${i.stack||i.type}`}(o,a,s),d=e.length;let u;for(let t=0;ti[t].axis===e)).shift()}function Ws(t,e){const i=t.controller.index,s=t.vScale&&t.vScale.axis;if(s){e=e||t._parsed;for(const t of e){const e=t._stacks;if(!e||void 0===e[s]||void 0===e[s][i])return;delete e[s][i],void 0!==e[s]._visualValues&&void 0!==e[s]._visualValues[i]&&delete e[s]._visualValues[i]}}}const Ns=t=>"reset"===t||"none"===t,Hs=(t,e)=>e?t:Object.assign({},t);class js{static defaults={};static datasetElementType=null;static dataElementType=null;constructor(t,e){this.chart=t,this._ctx=t.ctx,this.index=e,this._cachedDataOpts={},this._cachedMeta=this.getMeta(),this._type=this._cachedMeta.type,this.options=void 0,this._parsing=!1,this._data=void 0,this._objectData=void 0,this._sharedOptions=void 0,this._drawStart=void 0,this._drawCount=void 0,this.enableOptionSharing=!1,this.supportsDecimation=!1,this.$context=void 0,this._syncList=[],this.datasetElementType=new.target.datasetElementType,this.dataElementType=new.target.dataElementType,this.initialize()}initialize(){const t=this._cachedMeta;this.configure(),this.linkScales(),t._stacked=Is(t.vScale,t),this.addElements(),this.options.fill&&!this.chart.isPluginEnabled("filler")&&console.warn("Tried to use the 'fill' option without the 'Filler' plugin enabled. Please import and register the 'Filler' plugin and make sure it is not disabled in the options")}updateIndex(t){this.index!==t&&Ws(this._cachedMeta),this.index=t}linkScales(){const t=this.chart,e=this._cachedMeta,i=this.getDataset(),s=(t,e,i,s)=>"x"===t?e:"r"===t?s:i,n=e.xAxisID=l(i.xAxisID,Bs(t,"x")),o=e.yAxisID=l(i.yAxisID,Bs(t,"y")),a=e.rAxisID=l(i.rAxisID,Bs(t,"r")),r=e.indexAxis,h=e.iAxisID=s(r,n,o,a),c=e.vAxisID=s(r,o,n,a);e.xScale=this.getScaleForId(n),e.yScale=this.getScaleForId(o),e.rScale=this.getScaleForId(a),e.iScale=this.getScaleForId(h),e.vScale=this.getScaleForId(c)}getDataset(){return this.chart.data.datasets[this.index]}getMeta(){return this.chart.getDatasetMeta(this.index)}getScaleForId(t){return this.chart.scales[t]}_getOtherScale(t){const e=this._cachedMeta;return t===e.iScale?e.vScale:e.iScale}reset(){this._update("reset")}_destroy(){const t=this._cachedMeta;this._data&&rt(this._data,this),t._stacked&&Ws(t)}_dataCheck(){const t=this.getDataset(),e=t.data||(t.data=[]),i=this._data;if(o(e)){const t=this._cachedMeta;this._data=function(t,e){const{iScale:i,vScale:s}=e,n="x"===i.axis?"x":"y",o="x"===s.axis?"x":"y",a=Object.keys(t),r=new Array(a.length);let l,h,c;for(l=0,h=a.length;l0&&i._parsed[t-1];if(!1===this._parsing)i._parsed=s,i._sorted=!0,d=s;else{d=n(s[t])?this.parseArrayData(i,s,t,e):o(s[t])?this.parseObjectData(i,s,t,e):this.parsePrimitiveData(i,s,t,e);const a=()=>null===c[l]||f&&c[l]t&&!e.hidden&&e._stacked&&{keys:Es(i,!0),values:null})(e,i,this.chart),h={min:Number.POSITIVE_INFINITY,max:Number.NEGATIVE_INFINITY},{min:c,max:d}=function(t){const{min:e,max:i,minDefined:s,maxDefined:n}=t.getUserBounds();return{min:s?e:Number.NEGATIVE_INFINITY,max:n?i:Number.POSITIVE_INFINITY}}(r);let u,f;function g(){f=s[u];const e=f[r.axis];return!a(f[t.axis])||c>e||d=0;--u)if(!g()){this.updateRangeFromParsed(h,t,f,l);break}return h}getAllParsedValues(t){const e=this._cachedMeta._parsed,i=[];let s,n,o;for(s=0,n=e.length;s=0&&tthis.getContext(i,s,e)),c);return f.$shared&&(f.$shared=r,n[o]=Object.freeze(Hs(f,r))),f}_resolveAnimations(t,e,i){const s=this.chart,n=this._cachedDataOpts,o=`animation-${e}`,a=n[o];if(a)return a;let r;if(!1!==s.options.animation){const s=this.chart.config,n=s.datasetAnimationScopeKeys(this._type,e),o=s.getOptionScopes(this.getDataset(),n);r=s.createResolver(o,this.getContext(t,i,e))}const l=new Ts(s,r&&r.animations);return r&&r._cacheable&&(n[o]=Object.freeze(l)),l}getSharedOptions(t){if(t.$shared)return this._sharedOptions||(this._sharedOptions=Object.assign({},t))}includeOptions(t,e){return!e||Ns(t)||this.chart._animationsDisabled}_getSharedOptions(t,e){const i=this.resolveDataElementOptions(t,e),s=this._sharedOptions,n=this.getSharedOptions(i),o=this.includeOptions(e,n)||n!==s;return this.updateSharedOptions(n,e,i),{sharedOptions:n,includeOptions:o}}updateElement(t,e,i,s){Ns(s)?Object.assign(t,i):this._resolveAnimations(e,s).update(t,i)}updateSharedOptions(t,e,i){t&&!Ns(e)&&this._resolveAnimations(void 0,e).update(t,i)}_setStyle(t,e,i,s){t.active=s;const n=this.getStyle(e,s);this._resolveAnimations(e,i,s).update(t,{options:!s&&this.getSharedOptions(n)||n})}removeHoverStyle(t,e,i){this._setStyle(t,i,"active",!1)}setHoverStyle(t,e,i){this._setStyle(t,i,"active",!0)}_removeDatasetHoverStyle(){const t=this._cachedMeta.dataset;t&&this._setStyle(t,void 0,"active",!1)}_setDatasetHoverStyle(){const t=this._cachedMeta.dataset;t&&this._setStyle(t,void 0,"active",!0)}_resyncElements(t){const e=this._data,i=this._cachedMeta.data;for(const[t,e,i]of this._syncList)this[t](e,i);this._syncList=[];const s=i.length,n=e.length,o=Math.min(n,s);o&&this.parse(0,o),n>s?this._insertElements(s,n-s,t):n{for(t.length+=e,a=t.length-1;a>=o;a--)t[a]=t[a-e]};for(r(n),a=t;a{s[t]=i[t]&&i[t].active()?i[t]._to:this[t]})),s}}function Ys(t,e){const i=t.options.ticks,n=function(t){const e=t.options.offset,i=t._tickSize(),s=t._length/i+(e?0:1),n=t._maxLength/i;return Math.floor(Math.min(s,n))}(t),o=Math.min(i.maxTicksLimit||n,n),a=i.major.enabled?function(t){const e=[];let i,s;for(i=0,s=t.length;io)return function(t,e,i,s){let n,o=0,a=i[0];for(s=Math.ceil(s),n=0;nn)return e}return Math.max(n,1)}(a,e,o);if(r>0){let t,i;const n=r>1?Math.round((h-l)/(r-1)):null;for(Us(e,c,d,s(n)?0:l-n,l),t=0,i=r-1;t"top"===e||"left"===e?t[e]+i:t[e]-i,qs=(t,e)=>Math.min(e||t,t);function Ks(t,e){const i=[],s=t.length/e,n=t.length;let o=0;for(;oa+r)))return h}function Js(t){return t.drawTicks?t.tickLength:0}function Zs(t,e){if(!t.display)return 0;const i=Si(t.font,e),s=ki(t.padding);return(n(t.text)?t.text.length:1)*i.lineHeight+s.height}function Qs(t,e,i){let s=ut(t);return(i&&"right"!==e||!i&&"right"===e)&&(s=(t=>"left"===t?"right":"right"===t?"left":t)(s)),s}class tn extends $s{constructor(t){super(),this.id=t.id,this.type=t.type,this.options=void 0,this.ctx=t.ctx,this.chart=t.chart,this.top=void 0,this.bottom=void 0,this.left=void 0,this.right=void 0,this.width=void 0,this.height=void 0,this._margins={left:0,right:0,top:0,bottom:0},this.maxWidth=void 0,this.maxHeight=void 0,this.paddingTop=void 0,this.paddingBottom=void 0,this.paddingLeft=void 0,this.paddingRight=void 0,this.axis=void 0,this.labelRotation=void 0,this.min=void 0,this.max=void 0,this._range=void 0,this.ticks=[],this._gridLineItems=null,this._labelItems=null,this._labelSizes=null,this._length=0,this._maxLength=0,this._longestTextCache={},this._startPixel=void 0,this._endPixel=void 0,this._reversePixels=!1,this._userMax=void 0,this._userMin=void 0,this._suggestedMax=void 0,this._suggestedMin=void 0,this._ticksLength=0,this._borderValue=0,this._cache={},this._dataLimitsCached=!1,this.$context=void 0}init(t){this.options=t.setContext(this.getContext()),this.axis=t.axis,this._userMin=this.parse(t.min),this._userMax=this.parse(t.max),this._suggestedMin=this.parse(t.suggestedMin),this._suggestedMax=this.parse(t.suggestedMax)}parse(t,e){return t}getUserBounds(){let{_userMin:t,_userMax:e,_suggestedMin:i,_suggestedMax:s}=this;return t=r(t,Number.POSITIVE_INFINITY),e=r(e,Number.NEGATIVE_INFINITY),i=r(i,Number.POSITIVE_INFINITY),s=r(s,Number.NEGATIVE_INFINITY),{min:r(t,i),max:r(e,s),minDefined:a(t),maxDefined:a(e)}}getMinMax(t){let e,{min:i,max:s,minDefined:n,maxDefined:o}=this.getUserBounds();if(n&&o)return{min:i,max:s};const a=this.getMatchingVisibleMetas();for(let r=0,l=a.length;rs?s:i,s=n&&i>s?i:s,{min:r(i,r(s,i)),max:r(s,r(i,s))}}getPadding(){return{left:this.paddingLeft||0,top:this.paddingTop||0,right:this.paddingRight||0,bottom:this.paddingBottom||0}}getTicks(){return this.ticks}getLabels(){const t=this.chart.data;return this.options.labels||(this.isHorizontal()?t.xLabels:t.yLabels)||t.labels||[]}getLabelItems(t=this.chart.chartArea){return this._labelItems||(this._labelItems=this._computeLabelItems(t))}beforeLayout(){this._cache={},this._dataLimitsCached=!1}beforeUpdate(){d(this.options.beforeUpdate,[this])}update(t,e,i){const{beginAtZero:s,grace:n,ticks:o}=this.options,a=o.sampleSize;this.beforeUpdate(),this.maxWidth=t,this.maxHeight=e,this._margins=i=Object.assign({left:0,right:0,top:0,bottom:0},i),this.ticks=null,this._labelSizes=null,this._gridLineItems=null,this._labelItems=null,this.beforeSetDimensions(),this.setDimensions(),this.afterSetDimensions(),this._maxLength=this.isHorizontal()?this.width+i.left+i.right:this.height+i.top+i.bottom,this._dataLimitsCached||(this.beforeDataLimits(),this.determineDataLimits(),this.afterDataLimits(),this._range=Di(this,n,s),this._dataLimitsCached=!0),this.beforeBuildTicks(),this.ticks=this.buildTicks()||[],this.afterBuildTicks();const r=a=n||i<=1||!this.isHorizontal())return void(this.labelRotation=s);const h=this._getLabelSizes(),c=h.widest.width,d=h.highest.height,u=Z(this.chart.width-c,0,this.maxWidth);o=t.offset?this.maxWidth/i:u/(i-1),c+6>o&&(o=u/(i-(t.offset?.5:1)),a=this.maxHeight-Js(t.grid)-e.padding-Zs(t.title,this.chart.options.font),r=Math.sqrt(c*c+d*d),l=Y(Math.min(Math.asin(Z((h.highest.height+6)/o,-1,1)),Math.asin(Z(a/r,-1,1))-Math.asin(Z(d/r,-1,1)))),l=Math.max(s,Math.min(n,l))),this.labelRotation=l}afterCalculateLabelRotation(){d(this.options.afterCalculateLabelRotation,[this])}afterAutoSkip(){}beforeFit(){d(this.options.beforeFit,[this])}fit(){const t={width:0,height:0},{chart:e,options:{ticks:i,title:s,grid:n}}=this,o=this._isVisible(),a=this.isHorizontal();if(o){const o=Zs(s,e.options.font);if(a?(t.width=this.maxWidth,t.height=Js(n)+o):(t.height=this.maxHeight,t.width=Js(n)+o),i.display&&this.ticks.length){const{first:e,last:s,widest:n,highest:o}=this._getLabelSizes(),r=2*i.padding,l=$(this.labelRotation),h=Math.cos(l),c=Math.sin(l);if(a){const e=i.mirror?0:c*n.width+h*o.height;t.height=Math.min(this.maxHeight,t.height+e+r)}else{const e=i.mirror?0:h*n.width+c*o.height;t.width=Math.min(this.maxWidth,t.width+e+r)}this._calculatePadding(e,s,c,h)}}this._handleMargins(),a?(this.width=this._length=e.width-this._margins.left-this._margins.right,this.height=t.height):(this.width=t.width,this.height=this._length=e.height-this._margins.top-this._margins.bottom)}_calculatePadding(t,e,i,s){const{ticks:{align:n,padding:o},position:a}=this.options,r=0!==this.labelRotation,l="top"!==a&&"x"===this.axis;if(this.isHorizontal()){const a=this.getPixelForTick(0)-this.left,h=this.right-this.getPixelForTick(this.ticks.length-1);let c=0,d=0;r?l?(c=s*t.width,d=i*e.height):(c=i*t.height,d=s*e.width):"start"===n?d=e.width:"end"===n?c=t.width:"inner"!==n&&(c=t.width/2,d=e.width/2),this.paddingLeft=Math.max((c-a+o)*this.width/(this.width-a),0),this.paddingRight=Math.max((d-h+o)*this.width/(this.width-h),0)}else{let i=e.height/2,s=t.height/2;"start"===n?(i=0,s=t.height):"end"===n&&(i=e.height,s=0),this.paddingTop=i+o,this.paddingBottom=s+o}}_handleMargins(){this._margins&&(this._margins.left=Math.max(this.paddingLeft,this._margins.left),this._margins.top=Math.max(this.paddingTop,this._margins.top),this._margins.right=Math.max(this.paddingRight,this._margins.right),this._margins.bottom=Math.max(this.paddingBottom,this._margins.bottom))}afterFit(){d(this.options.afterFit,[this])}isHorizontal(){const{axis:t,position:e}=this.options;return"top"===e||"bottom"===e||"x"===t}isFullSize(){return this.options.fullSize}_convertTicksToLabels(t){let e,i;for(this.beforeTickToLabelConversion(),this.generateTickLabels(t),e=0,i=t.length;e{const i=t.gc,s=i.length/2;let n;if(s>e){for(n=0;n({width:r[t]||0,height:l[t]||0});return{first:P(0),last:P(e-1),widest:P(k),highest:P(S),widths:r,heights:l}}getLabelForValue(t){return t}getPixelForValue(t,e){return NaN}getValueForPixel(t){}getPixelForTick(t){const e=this.ticks;return t<0||t>e.length-1?null:this.getPixelForValue(e[t].value)}getPixelForDecimal(t){this._reversePixels&&(t=1-t);const e=this._startPixel+t*this._length;return Q(this._alignToPixels?Ae(this.chart,e,0):e)}getDecimalForPixel(t){const e=(t-this._startPixel)/this._length;return this._reversePixels?1-e:e}getBasePixel(){return this.getPixelForValue(this.getBaseValue())}getBaseValue(){const{min:t,max:e}=this;return t<0&&e<0?e:t>0&&e>0?t:0}getContext(t){const e=this.ticks||[];if(t>=0&&ta*s?a/i:r/s:r*s0}_computeGridLineItems(t){const e=this.axis,i=this.chart,s=this.options,{grid:n,position:a,border:r}=s,h=n.offset,c=this.isHorizontal(),d=this.ticks.length+(h?1:0),u=Js(n),f=[],g=r.setContext(this.getContext()),p=g.display?g.width:0,m=p/2,x=function(t){return Ae(i,t,p)};let b,_,y,v,M,w,k,S,P,D,C,O;if("top"===a)b=x(this.bottom),w=this.bottom-u,S=b-m,D=x(t.top)+m,O=t.bottom;else if("bottom"===a)b=x(this.top),D=t.top,O=x(t.bottom)-m,w=b+m,S=this.top+u;else if("left"===a)b=x(this.right),M=this.right-u,k=b-m,P=x(t.left)+m,C=t.right;else if("right"===a)b=x(this.left),P=t.left,C=x(t.right)-m,M=b+m,k=this.left+u;else if("x"===e){if("center"===a)b=x((t.top+t.bottom)/2+.5);else if(o(a)){const t=Object.keys(a)[0],e=a[t];b=x(this.chart.scales[t].getPixelForValue(e))}D=t.top,O=t.bottom,w=b+m,S=w+u}else if("y"===e){if("center"===a)b=x((t.left+t.right)/2);else if(o(a)){const t=Object.keys(a)[0],e=a[t];b=x(this.chart.scales[t].getPixelForValue(e))}M=b-m,k=M-u,P=t.left,C=t.right}const A=l(s.ticks.maxTicksLimit,d),T=Math.max(1,Math.ceil(d/A));for(_=0;_0&&(o-=s/2)}d={left:o,top:n,width:s+e.width,height:i+e.height,color:t.backdropColor}}x.push({label:v,font:P,textOffset:O,options:{rotation:m,color:i,strokeColor:o,strokeWidth:h,textAlign:f,textBaseline:A,translation:[M,w],backdrop:d}})}return x}_getXAxisLabelAlignment(){const{position:t,ticks:e}=this.options;if(-$(this.labelRotation))return"top"===t?"left":"right";let i="center";return"start"===e.align?i="left":"end"===e.align?i="right":"inner"===e.align&&(i="inner"),i}_getYAxisLabelAlignment(t){const{position:e,ticks:{crossAlign:i,mirror:s,padding:n}}=this.options,o=t+n,a=this._getLabelSizes().widest.width;let r,l;return"left"===e?s?(l=this.right+n,"near"===i?r="left":"center"===i?(r="center",l+=a/2):(r="right",l+=a)):(l=this.right-o,"near"===i?r="right":"center"===i?(r="center",l-=a/2):(r="left",l=this.left)):"right"===e?s?(l=this.left+n,"near"===i?r="right":"center"===i?(r="center",l-=a/2):(r="left",l-=a)):(l=this.left+o,"near"===i?r="left":"center"===i?(r="center",l+=a/2):(r="right",l=this.right)):r="right",{textAlign:r,x:l}}_computeLabelArea(){if(this.options.ticks.mirror)return;const t=this.chart,e=this.options.position;return"left"===e||"right"===e?{top:0,left:this.left,bottom:t.height,right:this.right}:"top"===e||"bottom"===e?{top:this.top,left:0,bottom:this.bottom,right:t.width}:void 0}drawBackground(){const{ctx:t,options:{backgroundColor:e},left:i,top:s,width:n,height:o}=this;e&&(t.save(),t.fillStyle=e,t.fillRect(i,s,n,o),t.restore())}getLineWidthForValue(t){const e=this.options.grid;if(!this._isVisible()||!e.display)return 0;const i=this.ticks.findIndex((e=>e.value===t));if(i>=0){return e.setContext(this.getContext(i)).lineWidth}return 0}drawGrid(t){const e=this.options.grid,i=this.ctx,s=this._gridLineItems||(this._gridLineItems=this._computeGridLineItems(t));let n,o;const a=(t,e,s)=>{s.width&&s.color&&(i.save(),i.lineWidth=s.width,i.strokeStyle=s.color,i.setLineDash(s.borderDash||[]),i.lineDashOffset=s.borderDashOffset,i.beginPath(),i.moveTo(t.x,t.y),i.lineTo(e.x,e.y),i.stroke(),i.restore())};if(e.display)for(n=0,o=s.length;n{this.drawBackground(),this.drawGrid(t),this.drawTitle()}},{z:s,draw:()=>{this.drawBorder()}},{z:e,draw:t=>{this.drawLabels(t)}}]:[{z:e,draw:t=>{this.draw(t)}}]}getMatchingVisibleMetas(t){const e=this.chart.getSortedVisibleDatasetMetas(),i=this.axis+"AxisID",s=[];let n,o;for(n=0,o=e.length;n{const s=i.split("."),n=s.pop(),o=[t].concat(s).join("."),a=e[i].split("."),r=a.pop(),l=a.join(".");ue.route(o,n,l,r)}))}(e,t.defaultRoutes);t.descriptors&&ue.describe(e,t.descriptors)}(t,o,i),this.override&&ue.override(t.id,t.overrides)),o}get(t){return this.items[t]}unregister(t){const e=this.items,i=t.id,s=this.scope;i in e&&delete e[i],s&&i in ue[s]&&(delete ue[s][i],this.override&&delete re[i])}}class sn{constructor(){this.controllers=new en(js,"datasets",!0),this.elements=new en($s,"elements"),this.plugins=new en(Object,"plugins"),this.scales=new en(tn,"scales"),this._typedRegistries=[this.controllers,this.scales,this.elements]}add(...t){this._each("register",t)}remove(...t){this._each("unregister",t)}addControllers(...t){this._each("register",t,this.controllers)}addElements(...t){this._each("register",t,this.elements)}addPlugins(...t){this._each("register",t,this.plugins)}addScales(...t){this._each("register",t,this.scales)}getController(t){return this._get(t,this.controllers,"controller")}getElement(t){return this._get(t,this.elements,"element")}getPlugin(t){return this._get(t,this.plugins,"plugin")}getScale(t){return this._get(t,this.scales,"scale")}removeControllers(...t){this._each("unregister",t,this.controllers)}removeElements(...t){this._each("unregister",t,this.elements)}removePlugins(...t){this._each("unregister",t,this.plugins)}removeScales(...t){this._each("unregister",t,this.scales)}_each(t,e,i){[...e].forEach((e=>{const s=i||this._getRegistryForType(e);i||s.isForType(e)||s===this.plugins&&e.id?this._exec(t,s,e):u(e,(e=>{const s=i||this._getRegistryForType(e);this._exec(t,s,e)}))}))}_exec(t,e,i){const s=w(t);d(i["before"+s],[],i),e[t](i),d(i["after"+s],[],i)}_getRegistryForType(t){for(let e=0;et.filter((t=>!e.some((e=>t.plugin.id===e.plugin.id))));this._notify(s(e,i),t,"stop"),this._notify(s(i,e),t,"start")}}function an(t,e){return e||!1!==t?!0===t?{}:t:null}function rn(t,{plugin:e,local:i},s,n){const o=t.pluginScopeKeys(e),a=t.getOptionScopes(s,o);return i&&e.defaults&&a.push(e.defaults),t.createResolver(a,n,[""],{scriptable:!1,indexable:!1,allKeys:!0})}function ln(t,e){const i=ue.datasets[t]||{};return((e.datasets||{})[t]||{}).indexAxis||e.indexAxis||i.indexAxis||"x"}function hn(t){if("x"===t||"y"===t||"r"===t)return t}function cn(t,...e){if(hn(t))return t;for(const s of e){const e=s.axis||("top"===(i=s.position)||"bottom"===i?"x":"left"===i||"right"===i?"y":void 0)||t.length>1&&hn(t[0].toLowerCase());if(e)return e}var i;throw new Error(`Cannot determine type of '${t}' axis. Please provide 'axis' or 'position' option.`)}function dn(t,e,i){if(i[e+"AxisID"]===t)return{axis:e}}function un(t,e){const i=re[t.type]||{scales:{}},s=e.scales||{},n=ln(t.type,e),a=Object.create(null);return Object.keys(s).forEach((e=>{const r=s[e];if(!o(r))return console.error(`Invalid scale configuration for scale: ${e}`);if(r._proxy)return console.warn(`Ignoring resolver passed as options for scale: ${e}`);const l=cn(e,r,function(t,e){if(e.data&&e.data.datasets){const i=e.data.datasets.filter((e=>e.xAxisID===t||e.yAxisID===t));if(i.length)return dn(t,"x",i[0])||dn(t,"y",i[0])}return{}}(e,t),ue.scales[r.type]),h=function(t,e){return t===e?"_index_":"_value_"}(l,n),c=i.scales||{};a[e]=b(Object.create(null),[{axis:l},r,c[l],c[h]])})),t.data.datasets.forEach((i=>{const n=i.type||t.type,o=i.indexAxis||ln(n,e),r=(re[n]||{}).scales||{};Object.keys(r).forEach((t=>{const e=function(t,e){let i=t;return"_index_"===t?i=e:"_value_"===t&&(i="x"===e?"y":"x"),i}(t,o),n=i[e+"AxisID"]||e;a[n]=a[n]||Object.create(null),b(a[n],[{axis:e},s[n],r[t]])}))})),Object.keys(a).forEach((t=>{const e=a[t];b(e,[ue.scales[e.type],ue.scale])})),a}function fn(t){const e=t.options||(t.options={});e.plugins=l(e.plugins,{}),e.scales=un(t,e)}function gn(t){return(t=t||{}).datasets=t.datasets||[],t.labels=t.labels||[],t}const pn=new Map,mn=new Set;function xn(t,e){let i=pn.get(t);return i||(i=e(),pn.set(t,i),mn.add(i)),i}const bn=(t,e,i)=>{const s=M(e,i);void 0!==s&&t.add(s)};class _n{constructor(t){this._config=function(t){return(t=t||{}).data=gn(t.data),fn(t),t}(t),this._scopeCache=new Map,this._resolverCache=new Map}get platform(){return this._config.platform}get type(){return this._config.type}set type(t){this._config.type=t}get data(){return this._config.data}set data(t){this._config.data=gn(t)}get options(){return this._config.options}set options(t){this._config.options=t}get plugins(){return this._config.plugins}update(){const t=this._config;this.clearCache(),fn(t)}clearCache(){this._scopeCache.clear(),this._resolverCache.clear()}datasetScopeKeys(t){return xn(t,(()=>[[`datasets.${t}`,""]]))}datasetAnimationScopeKeys(t,e){return xn(`${t}.transition.${e}`,(()=>[[`datasets.${t}.transitions.${e}`,`transitions.${e}`],[`datasets.${t}`,""]]))}datasetElementScopeKeys(t,e){return xn(`${t}-${e}`,(()=>[[`datasets.${t}.elements.${e}`,`datasets.${t}`,`elements.${e}`,""]]))}pluginScopeKeys(t){const e=t.id;return xn(`${this.type}-plugin-${e}`,(()=>[[`plugins.${e}`,...t.additionalOptionScopes||[]]]))}_cachedScopes(t,e){const i=this._scopeCache;let s=i.get(t);return s&&!e||(s=new Map,i.set(t,s)),s}getOptionScopes(t,e,i){const{options:s,type:n}=this,o=this._cachedScopes(t,i),a=o.get(e);if(a)return a;const r=new Set;e.forEach((e=>{t&&(r.add(t),e.forEach((e=>bn(r,t,e)))),e.forEach((t=>bn(r,s,t))),e.forEach((t=>bn(r,re[n]||{},t))),e.forEach((t=>bn(r,ue,t))),e.forEach((t=>bn(r,le,t)))}));const l=Array.from(r);return 0===l.length&&l.push(Object.create(null)),mn.has(e)&&o.set(e,l),l}chartOptionScopes(){const{options:t,type:e}=this;return[t,re[e]||{},ue.datasets[e]||{},{type:e},ue,le]}resolveNamedOptions(t,e,i,s=[""]){const o={$shared:!0},{resolver:a,subPrefixes:r}=yn(this._resolverCache,t,s);let l=a;if(function(t,e){const{isScriptable:i,isIndexable:s}=Ye(t);for(const o of e){const e=i(o),a=s(o),r=(a||e)&&t[o];if(e&&(S(r)||vn(r))||a&&n(r))return!0}return!1}(a,e)){o.$shared=!1;l=$e(a,i=S(i)?i():i,this.createResolver(t,i,r))}for(const t of e)o[t]=l[t];return o}createResolver(t,e,i=[""],s){const{resolver:n}=yn(this._resolverCache,t,i);return o(e)?$e(n,e,void 0,s):n}}function yn(t,e,i){let s=t.get(e);s||(s=new Map,t.set(e,s));const n=i.join();let o=s.get(n);if(!o){o={resolver:je(e,i),subPrefixes:i.filter((t=>!t.toLowerCase().includes("hover")))},s.set(n,o)}return o}const vn=t=>o(t)&&Object.getOwnPropertyNames(t).some((e=>S(t[e])));const Mn=["top","bottom","left","right","chartArea"];function wn(t,e){return"top"===t||"bottom"===t||-1===Mn.indexOf(t)&&"x"===e}function kn(t,e){return function(i,s){return i[t]===s[t]?i[e]-s[e]:i[t]-s[t]}}function Sn(t){const e=t.chart,i=e.options.animation;e.notifyPlugins("afterRender"),d(i&&i.onComplete,[t],e)}function Pn(t){const e=t.chart,i=e.options.animation;d(i&&i.onProgress,[t],e)}function Dn(t){return fe()&&"string"==typeof t?t=document.getElementById(t):t&&t.length&&(t=t[0]),t&&t.canvas&&(t=t.canvas),t}const Cn={},On=t=>{const e=Dn(t);return Object.values(Cn).filter((t=>t.canvas===e)).pop()};function An(t,e,i){const s=Object.keys(t);for(const n of s){const s=+n;if(s>=e){const o=t[n];delete t[n],(i>0||s>e)&&(t[s+i]=o)}}}class Tn{static defaults=ue;static instances=Cn;static overrides=re;static registry=nn;static version="4.5.1";static getChart=On;static register(...t){nn.add(...t),Ln()}static unregister(...t){nn.remove(...t),Ln()}constructor(t,e){const s=this.config=new _n(e),n=Dn(t),o=On(n);if(o)throw new Error("Canvas is already in use. Chart with ID '"+o.id+"' must be destroyed before the canvas with ID '"+o.canvas.id+"' can be reused.");const a=s.createResolver(s.chartOptionScopes(),this.getContext());this.platform=new(s.platform||Ps(n)),this.platform.updateConfig(s);const r=this.platform.acquireContext(n,a.aspectRatio),l=r&&r.canvas,h=l&&l.height,c=l&&l.width;this.id=i(),this.ctx=r,this.canvas=l,this.width=c,this.height=h,this._options=a,this._aspectRatio=this.aspectRatio,this._layers=[],this._metasets=[],this._stacks=void 0,this.boxes=[],this.currentDevicePixelRatio=void 0,this.chartArea=void 0,this._active=[],this._lastEvent=void 0,this._listeners={},this._responsiveListeners=void 0,this._sortedMetasets=[],this.scales={},this._plugins=new on,this.$proxies={},this._hiddenIndices={},this.attached=!1,this._animationsDisabled=void 0,this.$context=void 0,this._doResize=dt((t=>this.update(t)),a.resizeDelay||0),this._dataChanges=[],Cn[this.id]=this,r&&l?(bt.listen(this,"complete",Sn),bt.listen(this,"progress",Pn),this._initialize(),this.attached&&this.update()):console.error("Failed to create chart: can't acquire context from the given item")}get aspectRatio(){const{options:{aspectRatio:t,maintainAspectRatio:e},width:i,height:n,_aspectRatio:o}=this;return s(t)?e&&o?o:n?i/n:null:t}get data(){return this.config.data}set data(t){this.config.data=t}get options(){return this._options}set options(t){this.config.options=t}get registry(){return nn}_initialize(){return this.notifyPlugins("beforeInit"),this.options.responsive?this.resize():ke(this,this.options.devicePixelRatio),this.bindEvents(),this.notifyPlugins("afterInit"),this}clear(){return Te(this.canvas,this.ctx),this}stop(){return bt.stop(this),this}resize(t,e){bt.running(this)?this._resizeBeforeDraw={width:t,height:e}:this._resize(t,e)}_resize(t,e){const i=this.options,s=this.canvas,n=i.maintainAspectRatio&&this.aspectRatio,o=this.platform.getMaximumSize(s,t,e,n),a=i.devicePixelRatio||this.platform.getDevicePixelRatio(),r=this.width?"resize":"attach";this.width=o.width,this.height=o.height,this._aspectRatio=this.aspectRatio,ke(this,a,!0)&&(this.notifyPlugins("resize",{size:o}),d(i.onResize,[this,o],this),this.attached&&this._doResize(r)&&this.render())}ensureScalesHaveIDs(){u(this.options.scales||{},((t,e)=>{t.id=e}))}buildOrUpdateScales(){const t=this.options,e=t.scales,i=this.scales,s=Object.keys(i).reduce(((t,e)=>(t[e]=!1,t)),{});let n=[];e&&(n=n.concat(Object.keys(e).map((t=>{const i=e[t],s=cn(t,i),n="r"===s,o="x"===s;return{options:i,dposition:n?"chartArea":o?"bottom":"left",dtype:n?"radialLinear":o?"category":"linear"}})))),u(n,(e=>{const n=e.options,o=n.id,a=cn(o,n),r=l(n.type,e.dtype);void 0!==n.position&&wn(n.position,a)===wn(e.dposition)||(n.position=e.dposition),s[o]=!0;let h=null;if(o in i&&i[o].type===r)h=i[o];else{h=new(nn.getScale(r))({id:o,type:r,ctx:this.ctx,chart:this}),i[h.id]=h}h.init(n,t)})),u(s,((t,e)=>{t||delete i[e]})),u(i,(t=>{ls.configure(this,t,t.options),ls.addBox(this,t)}))}_updateMetasets(){const t=this._metasets,e=this.data.datasets.length,i=t.length;if(t.sort(((t,e)=>t.index-e.index)),i>e){for(let t=e;te.length&&delete this._stacks,t.forEach(((t,i)=>{0===e.filter((e=>e===t._dataset)).length&&this._destroyDatasetMeta(i)}))}buildOrUpdateControllers(){const t=[],e=this.data.datasets;let i,s;for(this._removeUnreferencedMetasets(),i=0,s=e.length;i{this.getDatasetMeta(e).controller.reset()}),this)}reset(){this._resetElements(),this.notifyPlugins("reset")}update(t){const e=this.config;e.update();const i=this._options=e.createResolver(e.chartOptionScopes(),this.getContext()),s=this._animationsDisabled=!i.animation;if(this._updateScales(),this._checkEventBindings(),this._updateHiddenIndices(),this._plugins.invalidate(),!1===this.notifyPlugins("beforeUpdate",{mode:t,cancelable:!0}))return;const n=this.buildOrUpdateControllers();this.notifyPlugins("beforeElementsUpdate");let o=0;for(let t=0,e=this.data.datasets.length;t{t.reset()})),this._updateDatasets(t),this.notifyPlugins("afterUpdate",{mode:t}),this._layers.sort(kn("z","_idx"));const{_active:a,_lastEvent:r}=this;r?this._eventHandler(r,!0):a.length&&this._updateHoverStyles(a,a,!0),this.render()}_updateScales(){u(this.scales,(t=>{ls.removeBox(this,t)})),this.ensureScalesHaveIDs(),this.buildOrUpdateScales()}_checkEventBindings(){const t=this.options,e=new Set(Object.keys(this._listeners)),i=new Set(t.events);P(e,i)&&!!this._responsiveListeners===t.responsive||(this.unbindEvents(),this.bindEvents())}_updateHiddenIndices(){const{_hiddenIndices:t}=this,e=this._getUniformDataChanges()||[];for(const{method:i,start:s,count:n}of e){An(t,s,"_removeElements"===i?-n:n)}}_getUniformDataChanges(){const t=this._dataChanges;if(!t||!t.length)return;this._dataChanges=[];const e=this.data.datasets.length,i=e=>new Set(t.filter((t=>t[0]===e)).map(((t,e)=>e+","+t.splice(1).join(",")))),s=i(0);for(let t=1;tt.split(","))).map((t=>({method:t[1],start:+t[2],count:+t[3]})))}_updateLayout(t){if(!1===this.notifyPlugins("beforeLayout",{cancelable:!0}))return;ls.update(this,this.width,this.height,t);const e=this.chartArea,i=e.width<=0||e.height<=0;this._layers=[],u(this.boxes,(t=>{i&&"chartArea"===t.position||(t.configure&&t.configure(),this._layers.push(...t._layers()))}),this),this._layers.forEach(((t,e)=>{t._idx=e})),this.notifyPlugins("afterLayout")}_updateDatasets(t){if(!1!==this.notifyPlugins("beforeDatasetsUpdate",{mode:t,cancelable:!0})){for(let t=0,e=this.data.datasets.length;t=0;--e)this._drawDataset(t[e]);this.notifyPlugins("afterDatasetsDraw")}_drawDataset(t){const e=this.ctx,i={meta:t,index:t.index,cancelable:!0},s=Ni(this,t);!1!==this.notifyPlugins("beforeDatasetDraw",i)&&(s&&Ie(e,s),t.controller.draw(),s&&ze(e),i.cancelable=!1,this.notifyPlugins("afterDatasetDraw",i))}isPointInArea(t){return Re(t,this.chartArea,this._minPadding)}getElementsAtEventForMode(t,e,i,s){const n=Ki.modes[e];return"function"==typeof n?n(this,t,i,s):[]}getDatasetMeta(t){const e=this.data.datasets[t],i=this._metasets;let s=i.filter((t=>t&&t._dataset===e)).pop();return s||(s={type:null,data:[],dataset:null,controller:null,hidden:null,xAxisID:null,yAxisID:null,order:e&&e.order||0,index:t,_dataset:e,_parsed:[],_sorted:!1},i.push(s)),s}getContext(){return this.$context||(this.$context=Ci(null,{chart:this,type:"chart"}))}getVisibleDatasetCount(){return this.getSortedVisibleDatasetMetas().length}isDatasetVisible(t){const e=this.data.datasets[t];if(!e)return!1;const i=this.getDatasetMeta(t);return"boolean"==typeof i.hidden?!i.hidden:!e.hidden}setDatasetVisibility(t,e){this.getDatasetMeta(t).hidden=!e}toggleDataVisibility(t){this._hiddenIndices[t]=!this._hiddenIndices[t]}getDataVisibility(t){return!this._hiddenIndices[t]}_updateVisibility(t,e,i){const s=i?"show":"hide",n=this.getDatasetMeta(t),o=n.controller._resolveAnimations(void 0,s);k(e)?(n.data[e].hidden=!i,this.update()):(this.setDatasetVisibility(t,i),o.update(n,{visible:i}),this.update((e=>e.datasetIndex===t?s:void 0)))}hide(t,e){this._updateVisibility(t,e,!1)}show(t,e){this._updateVisibility(t,e,!0)}_destroyDatasetMeta(t){const e=this._metasets[t];e&&e.controller&&e.controller._destroy(),delete this._metasets[t]}_stop(){let t,e;for(this.stop(),bt.remove(this),t=0,e=this.data.datasets.length;t{e.addEventListener(this,i,s),t[i]=s},s=(t,e,i)=>{t.offsetX=e,t.offsetY=i,this._eventHandler(t)};u(this.options.events,(t=>i(t,s)))}bindResponsiveEvents(){this._responsiveListeners||(this._responsiveListeners={});const t=this._responsiveListeners,e=this.platform,i=(i,s)=>{e.addEventListener(this,i,s),t[i]=s},s=(i,s)=>{t[i]&&(e.removeEventListener(this,i,s),delete t[i])},n=(t,e)=>{this.canvas&&this.resize(t,e)};let o;const a=()=>{s("attach",a),this.attached=!0,this.resize(),i("resize",n),i("detach",o)};o=()=>{this.attached=!1,s("resize",n),this._stop(),this._resize(0,0),i("attach",a)},e.isAttached(this.canvas)?a():o()}unbindEvents(){u(this._listeners,((t,e)=>{this.platform.removeEventListener(this,e,t)})),this._listeners={},u(this._responsiveListeners,((t,e)=>{this.platform.removeEventListener(this,e,t)})),this._responsiveListeners=void 0}updateHoverStyle(t,e,i){const s=i?"set":"remove";let n,o,a,r;for("dataset"===e&&(n=this.getDatasetMeta(t[0].datasetIndex),n.controller["_"+s+"DatasetHoverStyle"]()),a=0,r=t.length;a{const i=this.getDatasetMeta(t);if(!i)throw new Error("No dataset found at index "+t);return{datasetIndex:t,element:i.data[e],index:e}}));!f(i,e)&&(this._active=i,this._lastEvent=null,this._updateHoverStyles(i,e))}notifyPlugins(t,e,i){return this._plugins.notify(this,t,e,i)}isPluginEnabled(t){return 1===this._plugins._cache.filter((e=>e.plugin.id===t)).length}_updateHoverStyles(t,e,i){const s=this.options.hover,n=(t,e)=>t.filter((t=>!e.some((e=>t.datasetIndex===e.datasetIndex&&t.index===e.index)))),o=n(e,t),a=i?t:n(t,e);o.length&&this.updateHoverStyle(o,s.mode,!1),a.length&&s.mode&&this.updateHoverStyle(a,s.mode,!0)}_eventHandler(t,e){const i={event:t,replay:e,cancelable:!0,inChartArea:this.isPointInArea(t)},s=e=>(e.options.events||this.options.events).includes(t.native.type);if(!1===this.notifyPlugins("beforeEvent",i,s))return;const n=this._handleEvent(t,e,i.inChartArea);return i.cancelable=!1,this.notifyPlugins("afterEvent",i,s),(n||i.changed)&&this.render(),this}_handleEvent(t,e,i){const{_active:s=[],options:n}=this,o=e,a=this._getActiveElements(t,s,i,o),r=D(t),l=function(t,e,i,s){return i&&"mouseout"!==t.type?s?e:t:null}(t,this._lastEvent,i,r);i&&(this._lastEvent=null,d(n.onHover,[t,a,this],this),r&&d(n.onClick,[t,a,this],this));const h=!f(a,s);return(h||e)&&(this._active=a,this._updateHoverStyles(a,s,e)),this._lastEvent=l,h}_getActiveElements(t,e,i,s){if("mouseout"===t.type)return[];if(!i)return e;const n=this.options.hover;return this.getElementsAtEventForMode(t,n.mode,n,s)}}function Ln(){return u(Tn.instances,(t=>t._plugins.invalidate()))}function En(){throw new Error("This method is not implemented: Check that a complete date adapter is provided.")}class Rn{static override(t){Object.assign(Rn.prototype,t)}options;constructor(t){this.options=t||{}}init(){}formats(){return En()}parse(){return En()}format(){return En()}add(){return En()}diff(){return En()}startOf(){return En()}endOf(){return En()}}var In={_date:Rn};function zn(t){const e=t.iScale,i=function(t,e){if(!t._cache.$bar){const i=t.getMatchingVisibleMetas(e);let s=[];for(let e=0,n=i.length;et-e)))}return t._cache.$bar}(e,t.type);let s,n,o,a,r=e._length;const l=()=>{32767!==o&&-32768!==o&&(k(a)&&(r=Math.min(r,Math.abs(o-a)||r)),a=o)};for(s=0,n=i.length;sMath.abs(r)&&(l=r,h=a),e[i.axis]=h,e._custom={barStart:l,barEnd:h,start:n,end:o,min:a,max:r}}(t,e,i,s):e[i.axis]=i.parse(t,s),e}function Vn(t,e,i,s){const n=t.iScale,o=t.vScale,a=n.getLabels(),r=n===o,l=[];let h,c,d,u;for(h=i,c=i+s;ht.x,i="left",s="right"):(e=t.base"spacing"!==t,_indexable:t=>"spacing"!==t&&!t.startsWith("borderDash")&&!t.startsWith("hoverBorderDash")};static overrides={aspectRatio:1,plugins:{legend:{labels:{generateLabels(t){const e=t.data,{labels:{pointStyle:i,textAlign:s,color:n,useBorderRadius:o,borderRadius:a}}=t.legend.options;return e.labels.length&&e.datasets.length?e.labels.map(((e,r)=>{const l=t.getDatasetMeta(0).controller.getStyle(r);return{text:e,fillStyle:l.backgroundColor,fontColor:n,hidden:!t.getDataVisibility(r),lineDash:l.borderDash,lineDashOffset:l.borderDashOffset,lineJoin:l.borderJoinStyle,lineWidth:l.borderWidth,strokeStyle:l.borderColor,textAlign:s,pointStyle:i,borderRadius:o&&(a||l.borderRadius),index:r}})):[]}},onClick(t,e,i){i.chart.toggleDataVisibility(e.index),i.chart.update()}}}};constructor(t,e){super(t,e),this.enableOptionSharing=!0,this.innerRadius=void 0,this.outerRadius=void 0,this.offsetX=void 0,this.offsetY=void 0}linkScales(){}parse(t,e){const i=this.getDataset().data,s=this._cachedMeta;if(!1===this._parsing)s._parsed=i;else{let n,a,r=t=>+i[t];if(o(i[t])){const{key:t="value"}=this._parsing;r=e=>+M(i[e],t)}for(n=t,a=t+e;nJ(t,r,l,!0)?1:Math.max(e,e*i,s,s*i),g=(t,e,s)=>J(t,r,l,!0)?-1:Math.min(e,e*i,s,s*i),p=f(0,h,d),m=f(E,c,u),x=g(C,h,d),b=g(C+E,c,u);s=(p-x)/2,n=(m-b)/2,o=-(p+x)/2,a=-(m+b)/2}return{ratioX:s,ratioY:n,offsetX:o,offsetY:a}}(u,d,r),x=(i.width-o)/f,b=(i.height-o)/g,_=Math.max(Math.min(x,b)/2,0),y=c(this.options.radius,_),v=(y-Math.max(y*r,0))/this._getVisibleDatasetWeightTotal();this.offsetX=p*y,this.offsetY=m*y,s.total=this.calculateTotal(),this.outerRadius=y-v*this._getRingWeightOffset(this.index),this.innerRadius=Math.max(this.outerRadius-v*l,0),this.updateElements(n,0,n.length,t)}_circumference(t,e){const i=this.options,s=this._cachedMeta,n=this._getCircumference();return e&&i.animation.animateRotate||!this.chart.getDataVisibility(t)||null===s._parsed[t]||s.data[t].hidden?0:this.calculateCircumference(s._parsed[t]*n/O)}updateElements(t,e,i,s){const n="reset"===s,o=this.chart,a=o.chartArea,r=o.options.animation,l=(a.left+a.right)/2,h=(a.top+a.bottom)/2,c=n&&r.animateScale,d=c?0:this.innerRadius,u=c?0:this.outerRadius,{sharedOptions:f,includeOptions:g}=this._getSharedOptions(e,s);let p,m=this._getRotation();for(p=0;p0&&!isNaN(t)?O*(Math.abs(t)/e):0}getLabelAndValue(t){const e=this._cachedMeta,i=this.chart,s=i.data.labels||[],n=ne(e._parsed[t],i.options.locale);return{label:s[t]||"",value:n}}getMaxBorderWidth(t){let e=0;const i=this.chart;let s,n,o,a,r;if(!t)for(s=0,n=i.data.datasets.length;s{const o=t.getDatasetMeta(0).controller.getStyle(n);return{text:e,fillStyle:o.backgroundColor,strokeStyle:o.borderColor,fontColor:s,lineWidth:o.borderWidth,pointStyle:i,hidden:!t.getDataVisibility(n),index:n}}))}return[]}},onClick(t,e,i){i.chart.toggleDataVisibility(e.index),i.chart.update()}}},scales:{r:{type:"radialLinear",angleLines:{display:!1},beginAtZero:!0,grid:{circular:!0},pointLabels:{display:!1},startAngle:0}}};constructor(t,e){super(t,e),this.innerRadius=void 0,this.outerRadius=void 0}getLabelAndValue(t){const e=this._cachedMeta,i=this.chart,s=i.data.labels||[],n=ne(e._parsed[t].r,i.options.locale);return{label:s[t]||"",value:n}}parseObjectData(t,e,i,s){return ii.bind(this)(t,e,i,s)}update(t){const e=this._cachedMeta.data;this._updateRadius(),this.updateElements(e,0,e.length,t)}getMinMax(){const t=this._cachedMeta,e={min:Number.POSITIVE_INFINITY,max:Number.NEGATIVE_INFINITY};return t.data.forEach(((t,i)=>{const s=this.getParsed(i).r;!isNaN(s)&&this.chart.getDataVisibility(i)&&(se.max&&(e.max=s))})),e}_updateRadius(){const t=this.chart,e=t.chartArea,i=t.options,s=Math.min(e.right-e.left,e.bottom-e.top),n=Math.max(s/2,0),o=(n-Math.max(i.cutoutPercentage?n/100*i.cutoutPercentage:1,0))/t.getVisibleDatasetCount();this.outerRadius=n-o*this.index,this.innerRadius=this.outerRadius-o}updateElements(t,e,i,s){const n="reset"===s,o=this.chart,a=o.options.animation,r=this._cachedMeta.rScale,l=r.xCenter,h=r.yCenter,c=r.getIndexAngle(0)-.5*C;let d,u=c;const f=360/this.countVisibleElements();for(d=0;d{!isNaN(this.getParsed(i).r)&&this.chart.getDataVisibility(i)&&e++})),e}_computeAngle(t,e,i){return this.chart.getDataVisibility(t)?$(this.resolveDataElementOptions(t,e).angle||i):0}}var Un=Object.freeze({__proto__:null,BarController:class extends js{static id="bar";static defaults={datasetElementType:!1,dataElementType:"bar",categoryPercentage:.8,barPercentage:.9,grouped:!0,animations:{numbers:{type:"number",properties:["x","y","base","width","height"]}}};static overrides={scales:{_index_:{type:"category",offset:!0,grid:{offset:!0}},_value_:{type:"linear",beginAtZero:!0}}};parsePrimitiveData(t,e,i,s){return Vn(t,e,i,s)}parseArrayData(t,e,i,s){return Vn(t,e,i,s)}parseObjectData(t,e,i,s){const{iScale:n,vScale:o}=t,{xAxisKey:a="x",yAxisKey:r="y"}=this._parsing,l="x"===n.axis?a:r,h="x"===o.axis?a:r,c=[];let d,u,f,g;for(d=i,u=i+s;dt.controller.options.grouped)),o=i.options.stacked,a=[],r=this._cachedMeta.controller.getParsed(e),l=r&&r[i.axis],h=t=>{const e=t._parsed.find((t=>t[i.axis]===l)),n=e&&e[t.vScale.axis];if(s(n)||isNaN(n))return!0};for(const i of n)if((void 0===e||!h(i))&&((!1===o||-1===a.indexOf(i.stack)||void 0===o&&void 0===i.stack)&&a.push(i.stack),i.index===t))break;return a.length||a.push(void 0),a}_getStackCount(t){return this._getStacks(void 0,t).length}_getAxisCount(){return this._getAxis().length}getFirstScaleIdForIndexAxis(){const t=this.chart.scales,e=this.chart.options.indexAxis;return Object.keys(t).filter((i=>t[i].axis===e)).shift()}_getAxis(){const t={},e=this.getFirstScaleIdForIndexAxis();for(const i of this.chart.data.datasets)t[l("x"===this.chart.options.indexAxis?i.xAxisID:i.yAxisID,e)]=!0;return Object.keys(t)}_getStackIndex(t,e,i){const s=this._getStacks(t,i),n=void 0!==e?s.indexOf(e):-1;return-1===n?s.length-1:n}_getRuler(){const t=this.options,e=this._cachedMeta,i=e.iScale,s=[];let n,o;for(n=0,o=e.data.length;n=i?1:-1)}(u,e,r)*a,f===r&&(x-=u/2);const t=e.getPixelForDecimal(0),s=e.getPixelForDecimal(1),o=Math.min(t,s),h=Math.max(t,s);x=Math.max(Math.min(x,h),o),d=x+u,i&&!c&&(l._stacks[e.axis]._visualValues[n]=e.getValueForPixel(d)-e.getValueForPixel(x))}if(x===e.getPixelForValue(r)){const t=F(u)*e.getLineWidthForValue(r)/2;x+=t,u-=t}return{size:u,base:x,head:d,center:d+u/2}}_calculateBarIndexPixels(t,e){const i=e.scale,n=this.options,o=n.skipNull,a=l(n.maxBarThickness,1/0);let r,h;const c=this._getAxisCount();if(e.grouped){const i=o?this._getStackCount(t):e.stackCount,d="flex"===n.barThickness?function(t,e,i,s){const n=e.pixels,o=n[t];let a=t>0?n[t-1]:null,r=t=0;--i)e=Math.max(e,t[i].size(this.resolveDataElementOptions(i))/2);return e>0&&e}getLabelAndValue(t){const e=this._cachedMeta,i=this.chart.data.labels||[],{xScale:s,yScale:n}=e,o=this.getParsed(t),a=s.getLabelForValue(o.x),r=n.getLabelForValue(o.y),l=o._custom;return{label:i[t]||"",value:"("+a+", "+r+(l?", "+l:"")+")"}}update(t){const e=this._cachedMeta.data;this.updateElements(e,0,e.length,t)}updateElements(t,e,i,s){const n="reset"===s,{iScale:o,vScale:a}=this._cachedMeta,{sharedOptions:r,includeOptions:l}=this._getSharedOptions(e,s),h=o.axis,c=a.axis;for(let d=e;d0&&this.getParsed(e-1);for(let i=0;i<_;++i){const g=t[i],_=x?g:{};if(i=b){_.skip=!0;continue}const v=this.getParsed(i),M=s(v[f]),w=_[u]=a.getPixelForValue(v[u],i),k=_[f]=o||M?r.getBasePixel():r.getPixelForValue(l?this.applyStack(r,v,l):v[f],i);_.skip=isNaN(w)||isNaN(k)||M,_.stop=i>0&&Math.abs(v[u]-y[u])>m,p&&(_.parsed=v,_.raw=h.data[i]),d&&(_.options=c||this.resolveDataElementOptions(i,g.active?"active":n)),x||this.updateElement(g,i,_,n),y=v}}getMaxOverflow(){const t=this._cachedMeta,e=t.dataset,i=e.options&&e.options.borderWidth||0,s=t.data||[];if(!s.length)return i;const n=s[0].size(this.resolveDataElementOptions(0)),o=s[s.length-1].size(this.resolveDataElementOptions(s.length-1));return Math.max(i,n,o)/2}draw(){const t=this._cachedMeta;t.dataset.updateControlPoints(this.chart.chartArea,t.iScale.axis),super.draw()}},PieController:class extends $n{static id="pie";static defaults={cutout:0,rotation:0,circumference:360,radius:"100%"}},PolarAreaController:Yn,RadarController:class extends js{static id="radar";static defaults={datasetElementType:"line",dataElementType:"point",indexAxis:"r",showLine:!0,elements:{line:{fill:"start"}}};static overrides={aspectRatio:1,scales:{r:{type:"radialLinear"}}};getLabelAndValue(t){const e=this._cachedMeta.vScale,i=this.getParsed(t);return{label:e.getLabels()[t],value:""+e.getLabelForValue(i[e.axis])}}parseObjectData(t,e,i,s){return ii.bind(this)(t,e,i,s)}update(t){const e=this._cachedMeta,i=e.dataset,s=e.data||[],n=e.iScale.getLabels();if(i.points=s,"resize"!==t){const e=this.resolveDatasetElementOptions(t);this.options.showLine||(e.borderWidth=0);const o={_loop:!0,_fullLoop:n.length===s.length,options:e};this.updateElement(i,void 0,o,t)}this.updateElements(s,0,s.length,t)}updateElements(t,e,i,s){const n=this._cachedMeta.rScale,o="reset"===s;for(let a=e;a0&&this.getParsed(e-1);for(let c=e;c0&&Math.abs(i[f]-_[f])>x,m&&(p.parsed=i,p.raw=h.data[c]),u&&(p.options=d||this.resolveDataElementOptions(c,e.active?"active":n)),b||this.updateElement(e,c,p,n),_=i}this.updateSharedOptions(d,n,c)}getMaxOverflow(){const t=this._cachedMeta,e=t.data||[];if(!this.options.showLine){let t=0;for(let i=e.length-1;i>=0;--i)t=Math.max(t,e[i].size(this.resolveDataElementOptions(i))/2);return t>0&&t}const i=t.dataset,s=i.options&&i.options.borderWidth||0;if(!e.length)return s;const n=e[0].size(this.resolveDataElementOptions(0)),o=e[e.length-1].size(this.resolveDataElementOptions(e.length-1));return Math.max(s,n,o)/2}}});function Xn(t,e,i,s){const n=vi(t.options.borderRadius,["outerStart","outerEnd","innerStart","innerEnd"]);const o=(i-e)/2,a=Math.min(o,s*e/2),r=t=>{const e=(i-Math.min(o,t))*s/2;return Z(t,0,Math.min(o,e))};return{outerStart:r(n.outerStart),outerEnd:r(n.outerEnd),innerStart:Z(n.innerStart,0,a),innerEnd:Z(n.innerEnd,0,a)}}function qn(t,e,i,s){return{x:i+t*Math.cos(e),y:s+t*Math.sin(e)}}function Kn(t,e,i,s,n,o){const{x:a,y:r,startAngle:l,pixelMargin:h,innerRadius:c}=e,d=Math.max(e.outerRadius+s+i-h,0),u=c>0?c+s+i+h:0;let f=0;const g=n-l;if(s){const t=((c>0?c-s:0)+(d>0?d-s:0))/2;f=(g-(0!==t?g*t/(t+s):g))/2}const p=(g-Math.max(.001,g*d-i/C)/d)/2,m=l+p+f,x=n-p-f,{outerStart:b,outerEnd:_,innerStart:y,innerEnd:v}=Xn(e,u,d,x-m),M=d-b,w=d-_,k=m+b/M,S=x-_/w,P=u+y,D=u+v,O=m+y/P,A=x-v/D;if(t.beginPath(),o){const e=(k+S)/2;if(t.arc(a,r,d,k,e),t.arc(a,r,d,e,S),_>0){const e=qn(w,S,a,r);t.arc(e.x,e.y,_,S,x+E)}const i=qn(D,x,a,r);if(t.lineTo(i.x,i.y),v>0){const e=qn(D,A,a,r);t.arc(e.x,e.y,v,x+E,A+Math.PI)}const s=(x-v/u+(m+y/u))/2;if(t.arc(a,r,u,x-v/u,s,!0),t.arc(a,r,u,s,m+y/u,!0),y>0){const e=qn(P,O,a,r);t.arc(e.x,e.y,y,O+Math.PI,m-E)}const n=qn(M,m,a,r);if(t.lineTo(n.x,n.y),b>0){const e=qn(M,k,a,r);t.arc(e.x,e.y,b,m-E,k)}}else{t.moveTo(a,r);const e=Math.cos(k)*d+a,i=Math.sin(k)*d+r;t.lineTo(e,i);const s=Math.cos(S)*d+a,n=Math.sin(S)*d+r;t.lineTo(s,n)}t.closePath()}function Gn(t,e,i,s,n){const{fullCircles:o,startAngle:a,circumference:r,options:l}=e,{borderWidth:h,borderJoinStyle:c,borderDash:d,borderDashOffset:u,borderRadius:f}=l,g="inner"===l.borderAlign;if(!h)return;t.setLineDash(d||[]),t.lineDashOffset=u,g?(t.lineWidth=2*h,t.lineJoin=c||"round"):(t.lineWidth=h,t.lineJoin=c||"bevel");let p=e.endAngle;if(o){Kn(t,e,i,s,p,n);for(let e=0;en?(h=n/l,t.arc(o,a,l,i+h,s-h,!0)):t.arc(o,a,n,i+E,s-E),t.closePath(),t.clip()}(t,e,p),l.selfJoin&&p-a>=C&&0===f&&"miter"!==c&&function(t,e,i){const{startAngle:s,x:n,y:o,outerRadius:a,innerRadius:r,options:l}=e,{borderWidth:h,borderJoinStyle:c}=l,d=Math.min(h/a,G(s-i));if(t.beginPath(),t.arc(n,o,a-h/2,s+d/2,i-d/2),r>0){const e=Math.min(h/r,G(s-i));t.arc(n,o,r+h/2,i-e/2,s+e/2,!0)}else{const e=Math.min(h/2,a*G(s-i));if("round"===c)t.arc(n,o,e,i-C/2,s+C/2,!0);else if("bevel"===c){const a=2*e*e,r=-a*Math.cos(i+C/2)+n,l=-a*Math.sin(i+C/2)+o,h=a*Math.cos(s+C/2)+n,c=a*Math.sin(s+C/2)+o;t.lineTo(r,l),t.lineTo(h,c)}}t.closePath(),t.moveTo(0,0),t.rect(0,0,t.canvas.width,t.canvas.height),t.clip("evenodd")}(t,e,p),o||(Kn(t,e,i,s,p,n),t.stroke())}function Jn(t,e,i=e){t.lineCap=l(i.borderCapStyle,e.borderCapStyle),t.setLineDash(l(i.borderDash,e.borderDash)),t.lineDashOffset=l(i.borderDashOffset,e.borderDashOffset),t.lineJoin=l(i.borderJoinStyle,e.borderJoinStyle),t.lineWidth=l(i.borderWidth,e.borderWidth),t.strokeStyle=l(i.borderColor,e.borderColor)}function Zn(t,e,i){t.lineTo(i.x,i.y)}function Qn(t,e,i={}){const s=t.length,{start:n=0,end:o=s-1}=i,{start:a,end:r}=e,l=Math.max(n,a),h=Math.min(o,r),c=nr&&o>r;return{count:s,start:l,loop:e.loop,ilen:h(a+(h?r-t:t))%o,_=()=>{f!==g&&(t.lineTo(m,g),t.lineTo(m,f),t.lineTo(m,p))};for(l&&(d=n[b(0)],t.moveTo(d.x,d.y)),c=0;c<=r;++c){if(d=n[b(c)],d.skip)continue;const e=d.x,i=d.y,s=0|e;s===u?(ig&&(g=i),m=(x*m+e)/++x):(_(),t.lineTo(e,i),u=s,x=0,f=g=i),p=i}_()}function io(t){const e=t.options,i=e.borderDash&&e.borderDash.length;return!(t._decimated||t._loop||e.tension||"monotone"===e.cubicInterpolationMode||e.stepped||i)?eo:to}const so="function"==typeof Path2D;function no(t,e,i,s){so&&!e.options.segment?function(t,e,i,s){let n=e._path;n||(n=e._path=new Path2D,e.path(n,i,s)&&n.closePath()),Jn(t,e.options),t.stroke(n)}(t,e,i,s):function(t,e,i,s){const{segments:n,options:o}=e,a=io(e);for(const r of n)Jn(t,o,r.style),t.beginPath(),a(t,e,r,{start:i,end:i+s-1})&&t.closePath(),t.stroke()}(t,e,i,s)}class oo extends $s{static id="line";static defaults={borderCapStyle:"butt",borderDash:[],borderDashOffset:0,borderJoinStyle:"miter",borderWidth:3,capBezierPoints:!0,cubicInterpolationMode:"default",fill:!1,spanGaps:!1,stepped:!1,tension:0};static defaultRoutes={backgroundColor:"backgroundColor",borderColor:"borderColor"};static descriptors={_scriptable:!0,_indexable:t=>"borderDash"!==t&&"fill"!==t};constructor(t){super(),this.animated=!0,this.options=void 0,this._chart=void 0,this._loop=void 0,this._fullLoop=void 0,this._path=void 0,this._points=void 0,this._segments=void 0,this._decimated=!1,this._pointsUpdated=!1,this._datasetIndex=void 0,t&&Object.assign(this,t)}updateControlPoints(t,e){const i=this.options;if((i.tension||"monotone"===i.cubicInterpolationMode)&&!i.stepped&&!this._pointsUpdated){const s=i.spanGaps?this._loop:this._fullLoop;hi(this._points,i,t,s,e),this._pointsUpdated=!0}}set points(t){this._points=t,delete this._segments,delete this._path,this._pointsUpdated=!1}get points(){return this._points}get segments(){return this._segments||(this._segments=zi(this,this.options.segment))}first(){const t=this.segments,e=this.points;return t.length&&e[t[0].start]}last(){const t=this.segments,e=this.points,i=t.length;return i&&e[t[i-1].end]}interpolate(t,e){const i=this.options,s=t[e],n=this.points,o=Ii(this,{property:e,start:s,end:s});if(!o.length)return;const a=[],r=function(t){return t.stepped?pi:t.tension||"monotone"===t.cubicInterpolationMode?mi:gi}(i);let l,h;for(l=0,h=o.length;l"borderDash"!==t};circumference;endAngle;fullCircles;innerRadius;outerRadius;pixelMargin;startAngle;constructor(t){super(),this.options=void 0,this.circumference=void 0,this.startAngle=void 0,this.endAngle=void 0,this.innerRadius=void 0,this.outerRadius=void 0,this.pixelMargin=0,this.fullCircles=0,t&&Object.assign(this,t)}inRange(t,e,i){const s=this.getProps(["x","y"],i),{angle:n,distance:o}=X(s,{x:t,y:e}),{startAngle:a,endAngle:r,innerRadius:h,outerRadius:c,circumference:d}=this.getProps(["startAngle","endAngle","innerRadius","outerRadius","circumference"],i),u=(this.options.spacing+this.options.borderWidth)/2,f=l(d,r-a),g=J(n,a,r)&&a!==r,p=f>=O||g,m=tt(o,h+u,c+u);return p&&m}getCenterPoint(t){const{x:e,y:i,startAngle:s,endAngle:n,innerRadius:o,outerRadius:a}=this.getProps(["x","y","startAngle","endAngle","innerRadius","outerRadius"],t),{offset:r,spacing:l}=this.options,h=(s+n)/2,c=(o+a+l+r)/2;return{x:e+Math.cos(h)*c,y:i+Math.sin(h)*c}}tooltipPosition(t){return this.getCenterPoint(t)}draw(t){const{options:e,circumference:i}=this,s=(e.offset||0)/4,n=(e.spacing||0)/2,o=e.circular;if(this.pixelMargin="inner"===e.borderAlign?.33:0,this.fullCircles=i>O?Math.floor(i/O):0,0===i||this.innerRadius<0||this.outerRadius<0)return;t.save();const a=(this.startAngle+this.endAngle)/2;t.translate(Math.cos(a)*s,Math.sin(a)*s);const r=s*(1-Math.sin(Math.min(C,i||0)));t.fillStyle=e.backgroundColor,t.strokeStyle=e.borderColor,function(t,e,i,s,n){const{fullCircles:o,startAngle:a,circumference:r}=e;let l=e.endAngle;if(o){Kn(t,e,i,s,l,n);for(let e=0;e("string"==typeof e?(i=t.push(e)-1,s.unshift({index:i,label:e})):isNaN(e)&&(i=null),i))(t,e,i,s);return n!==t.lastIndexOf(e)?i:n}function mo(t){const e=this.getLabels();return t>=0&&ts=e?s:t,a=t=>n=i?n:t;if(t){const t=F(s),e=F(n);t<0&&e<0?a(0):t>0&&e>0&&o(0)}if(s===n){let e=0===n?1:Math.abs(.05*n);a(n+e),t||o(s-e)}this.min=s,this.max=n}getTickLimit(){const t=this.options.ticks;let e,{maxTicksLimit:i,stepSize:s}=t;return s?(e=Math.ceil(this.max/s)-Math.floor(this.min/s)+1,e>1e3&&(console.warn(`scales.${this.id}.ticks.stepSize: ${s} would result generating up to ${e} ticks. Limiting to 1000.`),e=1e3)):(e=this.computeTickLimit(),i=i||11),i&&(e=Math.min(i,e)),e}computeTickLimit(){return Number.POSITIVE_INFINITY}buildTicks(){const t=this.options,e=t.ticks;let i=this.getTickLimit();i=Math.max(2,i);const n=function(t,e){const i=[],{bounds:n,step:o,min:a,max:r,precision:l,count:h,maxTicks:c,maxDigits:d,includeBounds:u}=t,f=o||1,g=c-1,{min:p,max:m}=e,x=!s(a),b=!s(r),_=!s(h),y=(m-p)/(d+1);let v,M,w,k,S=B((m-p)/g/f)*f;if(S<1e-14&&!x&&!b)return[{value:p},{value:m}];k=Math.ceil(m/S)-Math.floor(p/S),k>g&&(S=B(k*S/g/f)*f),s(l)||(v=Math.pow(10,l),S=Math.ceil(S*v)/v),"ticks"===n?(M=Math.floor(p/S)*S,w=Math.ceil(m/S)*S):(M=p,w=m),x&&b&&o&&H((r-a)/o,S/1e3)?(k=Math.round(Math.min((r-a)/S,c)),S=(r-a)/k,M=a,w=r):_?(M=x?a:M,w=b?r:w,k=h-1,S=(w-M)/k):(k=(w-M)/S,k=V(k,Math.round(k),S/1e3)?Math.round(k):Math.ceil(k));const P=Math.max(U(S),U(M));v=Math.pow(10,s(l)?P:l),M=Math.round(M*v)/v,w=Math.round(w*v)/v;let D=0;for(x&&(u&&M!==a?(i.push({value:a}),Mr)break;i.push({value:t})}return b&&u&&w!==r?i.length&&V(i[i.length-1].value,r,xo(r,y,t))?i[i.length-1].value=r:i.push({value:r}):b&&w!==r||i.push({value:w}),i}({maxTicks:i,bounds:t.bounds,min:t.min,max:t.max,precision:e.precision,step:e.stepSize,count:e.count,maxDigits:this._maxDigits(),horizontal:this.isHorizontal(),minRotation:e.minRotation||0,includeBounds:!1!==e.includeBounds},this._range||this);return"ticks"===t.bounds&&j(n,this,"value"),t.reverse?(n.reverse(),this.start=this.max,this.end=this.min):(this.start=this.min,this.end=this.max),n}configure(){const t=this.ticks;let e=this.min,i=this.max;if(super.configure(),this.options.offset&&t.length){const s=(i-e)/Math.max(t.length-1,1)/2;e-=s,i+=s}this._startValue=e,this._endValue=i,this._valueRange=i-e}getLabelForValue(t){return ne(t,this.chart.options.locale,this.options.ticks.format)}}class _o extends bo{static id="linear";static defaults={ticks:{callback:ae.formatters.numeric}};determineDataLimits(){const{min:t,max:e}=this.getMinMax(!0);this.min=a(t)?t:0,this.max=a(e)?e:1,this.handleTickRangeOptions()}computeTickLimit(){const t=this.isHorizontal(),e=t?this.width:this.height,i=$(this.options.ticks.minRotation),s=(t?Math.sin(i):Math.cos(i))||.001,n=this._resolveTickFontOptions(0);return Math.ceil(e/Math.min(40,n.lineHeight/s))}getPixelForValue(t){return null===t?NaN:this.getPixelForDecimal((t-this._startValue)/this._valueRange)}getValueForPixel(t){return this._startValue+this.getDecimalForPixel(t)*this._valueRange}}const yo=t=>Math.floor(z(t)),vo=(t,e)=>Math.pow(10,yo(t)+e);function Mo(t){return 1===t/Math.pow(10,yo(t))}function wo(t,e,i){const s=Math.pow(10,i),n=Math.floor(t/s);return Math.ceil(e/s)-n}function ko(t,{min:e,max:i}){e=r(t.min,e);const s=[],n=yo(e);let o=function(t,e){let i=yo(e-t);for(;wo(t,e,i)>10;)i++;for(;wo(t,e,i)<10;)i--;return Math.min(i,yo(t))}(e,i),a=o<0?Math.pow(10,Math.abs(o)):1;const l=Math.pow(10,o),h=n>o?Math.pow(10,n):0,c=Math.round((e-h)*a)/a,d=Math.floor((e-h)/l/10)*l*10;let u=Math.floor((c-d)/Math.pow(10,o)),f=r(t.min,Math.round((h+d+u*Math.pow(10,o))*a)/a);for(;f=10?u=u<15?15:20:u++,u>=20&&(o++,u=2,a=o>=0?1:a),f=Math.round((h+d+u*Math.pow(10,o))*a)/a;const g=r(t.max,f);return s.push({value:g,major:Mo(g),significand:u}),s}class So extends tn{static id="logarithmic";static defaults={ticks:{callback:ae.formatters.logarithmic,major:{enabled:!0}}};constructor(t){super(t),this.start=void 0,this.end=void 0,this._startValue=void 0,this._valueRange=0}parse(t,e){const i=bo.prototype.parse.apply(this,[t,e]);if(0!==i)return a(i)&&i>0?i:null;this._zero=!0}determineDataLimits(){const{min:t,max:e}=this.getMinMax(!0);this.min=a(t)?Math.max(0,t):null,this.max=a(e)?Math.max(0,e):null,this.options.beginAtZero&&(this._zero=!0),this._zero&&this.min!==this._suggestedMin&&!a(this._userMin)&&(this.min=t===vo(this.min,0)?vo(this.min,-1):vo(this.min,0)),this.handleTickRangeOptions()}handleTickRangeOptions(){const{minDefined:t,maxDefined:e}=this.getUserBounds();let i=this.min,s=this.max;const n=e=>i=t?i:e,o=t=>s=e?s:t;i===s&&(i<=0?(n(1),o(10)):(n(vo(i,-1)),o(vo(s,1)))),i<=0&&n(vo(s,-1)),s<=0&&o(vo(i,1)),this.min=i,this.max=s}buildTicks(){const t=this.options,e=ko({min:this._userMin,max:this._userMax},this);return"ticks"===t.bounds&&j(e,this,"value"),t.reverse?(e.reverse(),this.start=this.max,this.end=this.min):(this.start=this.min,this.end=this.max),e}getLabelForValue(t){return void 0===t?"0":ne(t,this.chart.options.locale,this.options.ticks.format)}configure(){const t=this.min;super.configure(),this._startValue=z(t),this._valueRange=z(this.max)-z(t)}getPixelForValue(t){return void 0!==t&&0!==t||(t=this.min),null===t||isNaN(t)?NaN:this.getPixelForDecimal(t===this.min?0:(z(t)-this._startValue)/this._valueRange)}getValueForPixel(t){const e=this.getDecimalForPixel(t);return Math.pow(10,this._startValue+e*this._valueRange)}}function Po(t){const e=t.ticks;if(e.display&&t.display){const t=ki(e.backdropPadding);return l(e.font&&e.font.size,ue.font.size)+t.height}return 0}function Do(t,e,i,s,n){return t===s||t===n?{start:e-i/2,end:e+i/2}:tn?{start:e-i,end:e}:{start:e,end:e+i}}function Co(t){const e={l:t.left+t._padding.left,r:t.right-t._padding.right,t:t.top+t._padding.top,b:t.bottom-t._padding.bottom},i=Object.assign({},e),s=[],o=[],a=t._pointLabels.length,r=t.options.pointLabels,l=r.centerPointLabels?C/a:0;for(let u=0;ue.r&&(r=(s.end-e.r)/o,t.r=Math.max(t.r,e.r+r)),n.starte.b&&(l=(n.end-e.b)/a,t.b=Math.max(t.b,e.b+l))}function Ao(t,e,i){const s=t.drawingArea,{extra:n,additionalAngle:o,padding:a,size:r}=i,l=t.getPointPosition(e,s+n+a,o),h=Math.round(Y(G(l.angle+E))),c=function(t,e,i){90===i||270===i?t-=e/2:(i>270||i<90)&&(t-=e);return t}(l.y,r.h,h),d=function(t){if(0===t||180===t)return"center";if(t<180)return"left";return"right"}(h),u=function(t,e,i){"right"===i?t-=e:"center"===i&&(t-=e/2);return t}(l.x,r.w,d);return{visible:!0,x:l.x,y:c,textAlign:d,left:u,top:c,right:u+r.w,bottom:c+r.h}}function To(t,e){if(!e)return!0;const{left:i,top:s,right:n,bottom:o}=t;return!(Re({x:i,y:s},e)||Re({x:i,y:o},e)||Re({x:n,y:s},e)||Re({x:n,y:o},e))}function Lo(t,e,i){const{left:n,top:o,right:a,bottom:r}=i,{backdropColor:l}=e;if(!s(l)){const i=wi(e.borderRadius),s=ki(e.backdropPadding);t.fillStyle=l;const h=n-s.left,c=o-s.top,d=a-n+s.width,u=r-o+s.height;Object.values(i).some((t=>0!==t))?(t.beginPath(),He(t,{x:h,y:c,w:d,h:u,radius:i}),t.fill()):t.fillRect(h,c,d,u)}}function Eo(t,e,i,s){const{ctx:n}=t;if(i)n.arc(t.xCenter,t.yCenter,e,0,O);else{let i=t.getPointPosition(0,e);n.moveTo(i.x,i.y);for(let o=1;ot,padding:5,centerPointLabels:!1}};static defaultRoutes={"angleLines.color":"borderColor","pointLabels.color":"color","ticks.color":"color"};static descriptors={angleLines:{_fallback:"grid"}};constructor(t){super(t),this.xCenter=void 0,this.yCenter=void 0,this.drawingArea=void 0,this._pointLabels=[],this._pointLabelItems=[]}setDimensions(){const t=this._padding=ki(Po(this.options)/2),e=this.width=this.maxWidth-t.width,i=this.height=this.maxHeight-t.height;this.xCenter=Math.floor(this.left+e/2+t.left),this.yCenter=Math.floor(this.top+i/2+t.top),this.drawingArea=Math.floor(Math.min(e,i)/2)}determineDataLimits(){const{min:t,max:e}=this.getMinMax(!1);this.min=a(t)&&!isNaN(t)?t:0,this.max=a(e)&&!isNaN(e)?e:0,this.handleTickRangeOptions()}computeTickLimit(){return Math.ceil(this.drawingArea/Po(this.options))}generateTickLabels(t){bo.prototype.generateTickLabels.call(this,t),this._pointLabels=this.getLabels().map(((t,e)=>{const i=d(this.options.pointLabels.callback,[t,e],this);return i||0===i?i:""})).filter(((t,e)=>this.chart.getDataVisibility(e)))}fit(){const t=this.options;t.display&&t.pointLabels.display?Co(this):this.setCenterPoint(0,0,0,0)}setCenterPoint(t,e,i,s){this.xCenter+=Math.floor((t-e)/2),this.yCenter+=Math.floor((i-s)/2),this.drawingArea-=Math.min(this.drawingArea/2,Math.max(t,e,i,s))}getIndexAngle(t){return G(t*(O/(this._pointLabels.length||1))+$(this.options.startAngle||0))}getDistanceFromCenterForValue(t){if(s(t))return NaN;const e=this.drawingArea/(this.max-this.min);return this.options.reverse?(this.max-t)*e:(t-this.min)*e}getValueForDistanceFromCenter(t){if(s(t))return NaN;const e=t/(this.drawingArea/(this.max-this.min));return this.options.reverse?this.max-e:this.min+e}getPointLabelContext(t){const e=this._pointLabels||[];if(t>=0&&t=0;n--){const e=t._pointLabelItems[n];if(!e.visible)continue;const o=s.setContext(t.getPointLabelContext(n));Lo(i,o,e);const a=Si(o.font),{x:r,y:l,textAlign:h}=e;Ne(i,t._pointLabels[n],r,l+a.lineHeight/2,a,{color:o.color,textAlign:h,textBaseline:"middle"})}}(this,o),s.display&&this.ticks.forEach(((t,e)=>{if(0!==e||0===e&&this.min<0){r=this.getDistanceFromCenterForValue(t.value);const i=this.getContext(e),a=s.setContext(i),l=n.setContext(i);!function(t,e,i,s,n){const o=t.ctx,a=e.circular,{color:r,lineWidth:l}=e;!a&&!s||!r||!l||i<0||(o.save(),o.strokeStyle=r,o.lineWidth=l,o.setLineDash(n.dash||[]),o.lineDashOffset=n.dashOffset,o.beginPath(),Eo(t,i,a,s),o.closePath(),o.stroke(),o.restore())}(this,a,r,o,l)}})),i.display){for(t.save(),a=o-1;a>=0;a--){const s=i.setContext(this.getPointLabelContext(a)),{color:n,lineWidth:o}=s;o&&n&&(t.lineWidth=o,t.strokeStyle=n,t.setLineDash(s.borderDash),t.lineDashOffset=s.borderDashOffset,r=this.getDistanceFromCenterForValue(e.reverse?this.min:this.max),l=this.getPointPosition(a,r),t.beginPath(),t.moveTo(this.xCenter,this.yCenter),t.lineTo(l.x,l.y),t.stroke())}t.restore()}}drawBorder(){}drawLabels(){const t=this.ctx,e=this.options,i=e.ticks;if(!i.display)return;const s=this.getIndexAngle(0);let n,o;t.save(),t.translate(this.xCenter,this.yCenter),t.rotate(s),t.textAlign="center",t.textBaseline="middle",this.ticks.forEach(((s,a)=>{if(0===a&&this.min>=0&&!e.reverse)return;const r=i.setContext(this.getContext(a)),l=Si(r.font);if(n=this.getDistanceFromCenterForValue(this.ticks[a].value),r.showLabelBackdrop){t.font=l.string,o=t.measureText(s.label).width,t.fillStyle=r.backdropColor;const e=ki(r.backdropPadding);t.fillRect(-o/2-e.left,-n-l.size/2-e.top,o+e.width,l.size+e.height)}Ne(t,s.label,0,-n,l,{color:r.color,strokeColor:r.textStrokeColor,strokeWidth:r.textStrokeWidth})})),t.restore()}drawTitle(){}}const Io={millisecond:{common:!0,size:1,steps:1e3},second:{common:!0,size:1e3,steps:60},minute:{common:!0,size:6e4,steps:60},hour:{common:!0,size:36e5,steps:24},day:{common:!0,size:864e5,steps:30},week:{common:!1,size:6048e5,steps:4},month:{common:!0,size:2628e6,steps:12},quarter:{common:!1,size:7884e6,steps:4},year:{common:!0,size:3154e7}},zo=Object.keys(Io);function Fo(t,e){return t-e}function Vo(t,e){if(s(e))return null;const i=t._adapter,{parser:n,round:o,isoWeekday:r}=t._parseOpts;let l=e;return"function"==typeof n&&(l=n(l)),a(l)||(l="string"==typeof n?i.parse(l,n):i.parse(l)),null===l?null:(o&&(l="week"!==o||!N(r)&&!0!==r?i.startOf(l,o):i.startOf(l,"isoWeek",r)),+l)}function Bo(t,e,i,s){const n=zo.length;for(let o=zo.indexOf(t);o=e?i[s]:i[n]]=!0}}else t[e]=!0}function No(t,e,i){const s=[],n={},o=e.length;let a,r;for(a=0;a=0&&(e[l].major=!0);return e}(t,s,n,i):s}class Ho extends tn{static id="time";static defaults={bounds:"data",adapters:{},time:{parser:!1,unit:!1,round:!1,isoWeekday:!1,minUnit:"millisecond",displayFormats:{}},ticks:{source:"auto",callback:!1,major:{enabled:!1}}};constructor(t){super(t),this._cache={data:[],labels:[],all:[]},this._unit="day",this._majorUnit=void 0,this._offsets={},this._normalized=!1,this._parseOpts=void 0}init(t,e={}){const i=t.time||(t.time={}),s=this._adapter=new In._date(t.adapters.date);s.init(e),b(i.displayFormats,s.formats()),this._parseOpts={parser:i.parser,round:i.round,isoWeekday:i.isoWeekday},super.init(t),this._normalized=e.normalized}parse(t,e){return void 0===t?null:Vo(this,t)}beforeLayout(){super.beforeLayout(),this._cache={data:[],labels:[],all:[]}}determineDataLimits(){const t=this.options,e=this._adapter,i=t.time.unit||"day";let{min:s,max:n,minDefined:o,maxDefined:r}=this.getUserBounds();function l(t){o||isNaN(t.min)||(s=Math.min(s,t.min)),r||isNaN(t.max)||(n=Math.max(n,t.max))}o&&r||(l(this._getLabelBounds()),"ticks"===t.bounds&&"labels"===t.ticks.source||l(this.getMinMax(!1))),s=a(s)&&!isNaN(s)?s:+e.startOf(Date.now(),i),n=a(n)&&!isNaN(n)?n:+e.endOf(Date.now(),i)+1,this.min=Math.min(s,n-1),this.max=Math.max(s+1,n)}_getLabelBounds(){const t=this.getLabelTimestamps();let e=Number.POSITIVE_INFINITY,i=Number.NEGATIVE_INFINITY;return t.length&&(e=t[0],i=t[t.length-1]),{min:e,max:i}}buildTicks(){const t=this.options,e=t.time,i=t.ticks,s="labels"===i.source?this.getLabelTimestamps():this._generate();"ticks"===t.bounds&&s.length&&(this.min=this._userMin||s[0],this.max=this._userMax||s[s.length-1]);const n=this.min,o=nt(s,n,this.max);return this._unit=e.unit||(i.autoSkip?Bo(e.minUnit,this.min,this.max,this._getLabelCapacity(n)):function(t,e,i,s,n){for(let o=zo.length-1;o>=zo.indexOf(i);o--){const i=zo[o];if(Io[i].common&&t._adapter.diff(n,s,i)>=e-1)return i}return zo[i?zo.indexOf(i):0]}(this,o.length,e.minUnit,this.min,this.max)),this._majorUnit=i.major.enabled&&"year"!==this._unit?function(t){for(let e=zo.indexOf(t)+1,i=zo.length;e+t.value)))}initOffsets(t=[]){let e,i,s=0,n=0;this.options.offset&&t.length&&(e=this.getDecimalForValue(t[0]),s=1===t.length?1-e:(this.getDecimalForValue(t[1])-e)/2,i=this.getDecimalForValue(t[t.length-1]),n=1===t.length?i:(i-this.getDecimalForValue(t[t.length-2]))/2);const o=t.length<3?.5:.25;s=Z(s,0,o),n=Z(n,0,o),this._offsets={start:s,end:n,factor:1/(s+1+n)}}_generate(){const t=this._adapter,e=this.min,i=this.max,s=this.options,n=s.time,o=n.unit||Bo(n.minUnit,e,i,this._getLabelCapacity(e)),a=l(s.ticks.stepSize,1),r="week"===o&&n.isoWeekday,h=N(r)||!0===r,c={};let d,u,f=e;if(h&&(f=+t.startOf(f,"isoWeek",r)),f=+t.startOf(f,h?"day":o),t.diff(i,e,o)>1e5*a)throw new Error(e+" and "+i+" are too far apart with stepSize of "+a+" "+o);const g="data"===s.ticks.source&&this.getDataTimestamps();for(d=f,u=0;d+t))}getLabelForValue(t){const e=this._adapter,i=this.options.time;return i.tooltipFormat?e.format(t,i.tooltipFormat):e.format(t,i.displayFormats.datetime)}format(t,e){const i=this.options.time.displayFormats,s=this._unit,n=e||i[s];return this._adapter.format(t,n)}_tickFormatFunction(t,e,i,s){const n=this.options,o=n.ticks.callback;if(o)return d(o,[t,e,i],this);const a=n.time.displayFormats,r=this._unit,l=this._majorUnit,h=r&&a[r],c=l&&a[l],u=i[e],f=l&&c&&u&&u.major;return this._adapter.format(t,s||(f?c:h))}generateTickLabels(t){let e,i,s;for(e=0,i=t.length;e0?a:1}getDataTimestamps(){let t,e,i=this._cache.data||[];if(i.length)return i;const s=this.getMatchingVisibleMetas();if(this._normalized&&s.length)return this._cache.data=s[0].controller.getAllParsedValues(this);for(t=0,e=s.length;t=t[r].pos&&e<=t[l].pos&&({lo:r,hi:l}=it(t,"pos",e)),({pos:s,time:o}=t[r]),({pos:n,time:a}=t[l])):(e>=t[r].time&&e<=t[l].time&&({lo:r,hi:l}=it(t,"time",e)),({time:s,pos:o}=t[r]),({time:n,pos:a}=t[l]));const h=n-s;return h?o+(a-o)*(e-s)/h:o}var $o=Object.freeze({__proto__:null,CategoryScale:class extends tn{static id="category";static defaults={ticks:{callback:mo}};constructor(t){super(t),this._startValue=void 0,this._valueRange=0,this._addedLabels=[]}init(t){const e=this._addedLabels;if(e.length){const t=this.getLabels();for(const{index:i,label:s}of e)t[i]===s&&t.splice(i,1);this._addedLabels=[]}super.init(t)}parse(t,e){if(s(t))return null;const i=this.getLabels();return((t,e)=>null===t?null:Z(Math.round(t),0,e))(e=isFinite(e)&&i[e]===t?e:po(i,t,l(e,t),this._addedLabels),i.length-1)}determineDataLimits(){const{minDefined:t,maxDefined:e}=this.getUserBounds();let{min:i,max:s}=this.getMinMax(!0);"ticks"===this.options.bounds&&(t||(i=0),e||(s=this.getLabels().length-1)),this.min=i,this.max=s}buildTicks(){const t=this.min,e=this.max,i=this.options.offset,s=[];let n=this.getLabels();n=0===t&&e===n.length-1?n:n.slice(t,e+1),this._valueRange=Math.max(n.length-(i?0:1),1),this._startValue=this.min-(i?.5:0);for(let i=t;i<=e;i++)s.push({value:i});return s}getLabelForValue(t){return mo.call(this,t)}configure(){super.configure(),this.isHorizontal()||(this._reversePixels=!this._reversePixels)}getPixelForValue(t){return"number"!=typeof t&&(t=this.parse(t)),null===t?NaN:this.getPixelForDecimal((t-this._startValue)/this._valueRange)}getPixelForTick(t){const e=this.ticks;return t<0||t>e.length-1?null:this.getPixelForValue(e[t].value)}getValueForPixel(t){return Math.round(this._startValue+this.getDecimalForPixel(t)*this._valueRange)}getBasePixel(){return this.bottom}},LinearScale:_o,LogarithmicScale:So,RadialLinearScale:Ro,TimeScale:Ho,TimeSeriesScale:class extends Ho{static id="timeseries";static defaults=Ho.defaults;constructor(t){super(t),this._table=[],this._minPos=void 0,this._tableRange=void 0}initOffsets(){const t=this._getTimestampsForTable(),e=this._table=this.buildLookupTable(t);this._minPos=jo(e,this.min),this._tableRange=jo(e,this.max)-this._minPos,super.initOffsets(t)}buildLookupTable(t){const{min:e,max:i}=this,s=[],n=[];let o,a,r,l,h;for(o=0,a=t.length;o=e&&l<=i&&s.push(l);if(s.length<2)return[{time:e,pos:0},{time:i,pos:1}];for(o=0,a=s.length;ot-e))}_getTimestampsForTable(){let t=this._cache.all||[];if(t.length)return t;const e=this.getDataTimestamps(),i=this.getLabelTimestamps();return t=e.length&&i.length?this.normalize(e.concat(i)):e.length?e:i,t=this._cache.all=t,t}getDecimalForValue(t){return(jo(this._table,t)-this._minPos)/this._tableRange}getValueForPixel(t){const e=this._offsets,i=this.getDecimalForPixel(t)/e.factor-e.end;return jo(this._table,i*this._tableRange+this._minPos,!0)}}});const Yo=["rgb(54, 162, 235)","rgb(255, 99, 132)","rgb(255, 159, 64)","rgb(255, 205, 86)","rgb(75, 192, 192)","rgb(153, 102, 255)","rgb(201, 203, 207)"],Uo=Yo.map((t=>t.replace("rgb(","rgba(").replace(")",", 0.5)")));function Xo(t){return Yo[t%Yo.length]}function qo(t){return Uo[t%Uo.length]}function Ko(t){let e=0;return(i,s)=>{const n=t.getDatasetMeta(s).controller;n instanceof $n?e=function(t,e){return t.backgroundColor=t.data.map((()=>Xo(e++))),e}(i,e):n instanceof Yn?e=function(t,e){return t.backgroundColor=t.data.map((()=>qo(e++))),e}(i,e):n&&(e=function(t,e){return t.borderColor=Xo(e),t.backgroundColor=qo(e),++e}(i,e))}}function Go(t){let e;for(e in t)if(t[e].borderColor||t[e].backgroundColor)return!0;return!1}var Jo={id:"colors",defaults:{enabled:!0,forceOverride:!1},beforeLayout(t,e,i){if(!i.enabled)return;const{data:{datasets:s},options:n}=t.config,{elements:o}=n,a=Go(s)||(r=n)&&(r.borderColor||r.backgroundColor)||o&&Go(o)||"rgba(0,0,0,0.1)"!==ue.borderColor||"rgba(0,0,0,0.1)"!==ue.backgroundColor;var r;if(!i.forceOverride&&a)return;const l=Ko(t);s.forEach(l)}};function Zo(t){if(t._decimated){const e=t._data;delete t._decimated,delete t._data,Object.defineProperty(t,"data",{configurable:!0,enumerable:!0,writable:!0,value:e})}}function Qo(t){t.data.datasets.forEach((t=>{Zo(t)}))}var ta={id:"decimation",defaults:{algorithm:"min-max",enabled:!1},beforeElementsUpdate:(t,e,i)=>{if(!i.enabled)return void Qo(t);const n=t.width;t.data.datasets.forEach(((e,o)=>{const{_data:a,indexAxis:r}=e,l=t.getDatasetMeta(o),h=a||e.data;if("y"===Pi([r,t.options.indexAxis]))return;if(!l.controller.supportsDecimation)return;const c=t.scales[l.xAxisID];if("linear"!==c.type&&"time"!==c.type)return;if(t.options.parsing)return;let{start:d,count:u}=function(t,e){const i=e.length;let s,n=0;const{iScale:o}=t,{min:a,max:r,minDefined:l,maxDefined:h}=o.getUserBounds();return l&&(n=Z(it(e,o.axis,a).lo,0,i-1)),s=h?Z(it(e,o.axis,r).hi+1,n,i)-n:i-n,{start:n,count:s}}(l,h);if(u<=(i.threshold||4*n))return void Zo(e);let f;switch(s(a)&&(e._data=h,delete e.data,Object.defineProperty(e,"data",{configurable:!0,enumerable:!0,get:function(){return this._decimated},set:function(t){this._data=t}})),i.algorithm){case"lttb":f=function(t,e,i,s,n){const o=n.samples||s;if(o>=i)return t.slice(e,e+i);const a=[],r=(i-2)/(o-2);let l=0;const h=e+i-1;let c,d,u,f,g,p=e;for(a[l++]=t[p],c=0;cu&&(u=f,d=t[s],g=s);a[l++]=d,p=g}return a[l++]=t[h],a}(h,d,u,n,i);break;case"min-max":f=function(t,e,i,n){let o,a,r,l,h,c,d,u,f,g,p=0,m=0;const x=[],b=e+i-1,_=t[e].x,y=t[b].x-_;for(o=e;og&&(g=l,d=o),p=(m*p+a.x)/++m;else{const i=o-1;if(!s(c)&&!s(d)){const e=Math.min(c,d),s=Math.max(c,d);e!==u&&e!==i&&x.push({...t[e],x:p}),s!==u&&s!==i&&x.push({...t[s],x:p})}o>0&&i!==u&&x.push(t[i]),x.push(a),h=e,m=0,f=g=l,c=d=u=o}}return x}(h,d,u,n);break;default:throw new Error(`Unsupported decimation algorithm '${i.algorithm}'`)}e._decimated=f}))},destroy(t){Qo(t)}};function ea(t,e,i,s){if(s)return;let n=e[t],o=i[t];return"angle"===t&&(n=G(n),o=G(o)),{property:t,start:n,end:o}}function ia(t,e,i){for(;e>t;e--){const t=i[e];if(!isNaN(t.x)&&!isNaN(t.y))break}return e}function sa(t,e,i,s){return t&&e?s(t[i],e[i]):t?t[i]:e?e[i]:0}function na(t,e){let i=[],s=!1;return n(t)?(s=!0,i=t):i=function(t,e){const{x:i=null,y:s=null}=t||{},n=e.points,o=[];return e.segments.forEach((({start:t,end:e})=>{e=ia(t,e,n);const a=n[t],r=n[e];null!==s?(o.push({x:a.x,y:s}),o.push({x:r.x,y:s})):null!==i&&(o.push({x:i,y:a.y}),o.push({x:i,y:r.y}))})),o}(t,e),i.length?new oo({points:i,options:{tension:0},_loop:s,_fullLoop:s}):null}function oa(t){return t&&!1!==t.fill}function aa(t,e,i){let s=t[e].fill;const n=[e];let o;if(!i)return s;for(;!1!==s&&-1===n.indexOf(s);){if(!a(s))return s;if(o=t[s],!o)return!1;if(o.visible)return s;n.push(s),s=o.fill}return!1}function ra(t,e,i){const s=function(t){const e=t.options,i=e.fill;let s=l(i&&i.target,i);void 0===s&&(s=!!e.backgroundColor);if(!1===s||null===s)return!1;if(!0===s)return"origin";return s}(t);if(o(s))return!isNaN(s.value)&&s;let n=parseFloat(s);return a(n)&&Math.floor(n)===n?function(t,e,i,s){"-"!==t&&"+"!==t||(i=e+i);if(i===e||i<0||i>=s)return!1;return i}(s[0],e,n,i):["origin","start","end","stack","shape"].indexOf(s)>=0&&s}function la(t,e,i){const s=[];for(let n=0;n=0;--e){const i=n[e].$filler;i&&(i.line.updateControlPoints(o,i.axis),s&&i.fill&&ua(t.ctx,i,o))}},beforeDatasetsDraw(t,e,i){if("beforeDatasetsDraw"!==i.drawTime)return;const s=t.getSortedVisibleDatasetMetas();for(let e=s.length-1;e>=0;--e){const i=s[e].$filler;oa(i)&&ua(t.ctx,i,t.chartArea)}},beforeDatasetDraw(t,e,i){const s=e.meta.$filler;oa(s)&&"beforeDatasetDraw"===i.drawTime&&ua(t.ctx,s,t.chartArea)},defaults:{propagate:!0,drawTime:"beforeDatasetDraw"}};const _a=(t,e)=>{let{boxHeight:i=e,boxWidth:s=e}=t;return t.usePointStyle&&(i=Math.min(i,e),s=t.pointStyleWidth||Math.min(s,e)),{boxWidth:s,boxHeight:i,itemHeight:Math.max(e,i)}};class ya extends $s{constructor(t){super(),this._added=!1,this.legendHitBoxes=[],this._hoveredItem=null,this.doughnutMode=!1,this.chart=t.chart,this.options=t.options,this.ctx=t.ctx,this.legendItems=void 0,this.columnSizes=void 0,this.lineWidths=void 0,this.maxHeight=void 0,this.maxWidth=void 0,this.top=void 0,this.bottom=void 0,this.left=void 0,this.right=void 0,this.height=void 0,this.width=void 0,this._margins=void 0,this.position=void 0,this.weight=void 0,this.fullSize=void 0}update(t,e,i){this.maxWidth=t,this.maxHeight=e,this._margins=i,this.setDimensions(),this.buildLabels(),this.fit()}setDimensions(){this.isHorizontal()?(this.width=this.maxWidth,this.left=this._margins.left,this.right=this.width):(this.height=this.maxHeight,this.top=this._margins.top,this.bottom=this.height)}buildLabels(){const t=this.options.labels||{};let e=d(t.generateLabels,[this.chart],this)||[];t.filter&&(e=e.filter((e=>t.filter(e,this.chart.data)))),t.sort&&(e=e.sort(((e,i)=>t.sort(e,i,this.chart.data)))),this.options.reverse&&e.reverse(),this.legendItems=e}fit(){const{options:t,ctx:e}=this;if(!t.display)return void(this.width=this.height=0);const i=t.labels,s=Si(i.font),n=s.size,o=this._computeTitleHeight(),{boxWidth:a,itemHeight:r}=_a(i,n);let l,h;e.font=s.string,this.isHorizontal()?(l=this.maxWidth,h=this._fitRows(o,n,a,r)+10):(h=this.maxHeight,l=this._fitCols(o,s,a,r)+10),this.width=Math.min(l,t.maxWidth||this.maxWidth),this.height=Math.min(h,t.maxHeight||this.maxHeight)}_fitRows(t,e,i,s){const{ctx:n,maxWidth:o,options:{labels:{padding:a}}}=this,r=this.legendHitBoxes=[],l=this.lineWidths=[0],h=s+a;let c=t;n.textAlign="left",n.textBaseline="middle";let d=-1,u=-h;return this.legendItems.forEach(((t,f)=>{const g=i+e/2+n.measureText(t.text).width;(0===f||l[l.length-1]+g+2*a>o)&&(c+=h,l[l.length-(f>0?0:1)]=0,u+=h,d++),r[f]={left:0,top:u,row:d,width:g,height:s},l[l.length-1]+=g+a})),c}_fitCols(t,e,i,s){const{ctx:n,maxHeight:o,options:{labels:{padding:a}}}=this,r=this.legendHitBoxes=[],l=this.columnSizes=[],h=o-t;let c=a,d=0,u=0,f=0,g=0;return this.legendItems.forEach(((t,o)=>{const{itemWidth:p,itemHeight:m}=function(t,e,i,s,n){const o=function(t,e,i,s){let n=t.text;n&&"string"!=typeof n&&(n=n.reduce(((t,e)=>t.length>e.length?t:e)));return e+i.size/2+s.measureText(n).width}(s,t,e,i),a=function(t,e,i){let s=t;"string"!=typeof e.text&&(s=va(e,i));return s}(n,s,e.lineHeight);return{itemWidth:o,itemHeight:a}}(i,e,n,t,s);o>0&&u+m+2*a>h&&(c+=d+a,l.push({width:d,height:u}),f+=d+a,g++,d=u=0),r[o]={left:f,top:u,col:g,width:p,height:m},d=Math.max(d,p),u+=m+a})),c+=d,l.push({width:d,height:u}),c}adjustHitBoxes(){if(!this.options.display)return;const t=this._computeTitleHeight(),{legendHitBoxes:e,options:{align:i,labels:{padding:s},rtl:n}}=this,o=Oi(n,this.left,this.width);if(this.isHorizontal()){let n=0,a=ft(i,this.left+s,this.right-this.lineWidths[n]);for(const r of e)n!==r.row&&(n=r.row,a=ft(i,this.left+s,this.right-this.lineWidths[n])),r.top+=this.top+t+s,r.left=o.leftForLtr(o.x(a),r.width),a+=r.width+s}else{let n=0,a=ft(i,this.top+t+s,this.bottom-this.columnSizes[n].height);for(const r of e)r.col!==n&&(n=r.col,a=ft(i,this.top+t+s,this.bottom-this.columnSizes[n].height)),r.top=a,r.left+=this.left+s,r.left=o.leftForLtr(o.x(r.left),r.width),a+=r.height+s}}isHorizontal(){return"top"===this.options.position||"bottom"===this.options.position}draw(){if(this.options.display){const t=this.ctx;Ie(t,this),this._draw(),ze(t)}}_draw(){const{options:t,columnSizes:e,lineWidths:i,ctx:s}=this,{align:n,labels:o}=t,a=ue.color,r=Oi(t.rtl,this.left,this.width),h=Si(o.font),{padding:c}=o,d=h.size,u=d/2;let f;this.drawTitle(),s.textAlign=r.textAlign("left"),s.textBaseline="middle",s.lineWidth=.5,s.font=h.string;const{boxWidth:g,boxHeight:p,itemHeight:m}=_a(o,d),x=this.isHorizontal(),b=this._computeTitleHeight();f=x?{x:ft(n,this.left+c,this.right-i[0]),y:this.top+c+b,line:0}:{x:this.left+c,y:ft(n,this.top+b+c,this.bottom-e[0].height),line:0},Ai(this.ctx,t.textDirection);const _=m+c;this.legendItems.forEach(((y,v)=>{s.strokeStyle=y.fontColor,s.fillStyle=y.fontColor;const M=s.measureText(y.text).width,w=r.textAlign(y.textAlign||(y.textAlign=o.textAlign)),k=g+u+M;let S=f.x,P=f.y;r.setWidth(this.width),x?v>0&&S+k+c>this.right&&(P=f.y+=_,f.line++,S=f.x=ft(n,this.left+c,this.right-i[f.line])):v>0&&P+_>this.bottom&&(S=f.x=S+e[f.line].width+c,f.line++,P=f.y=ft(n,this.top+b+c,this.bottom-e[f.line].height));if(function(t,e,i){if(isNaN(g)||g<=0||isNaN(p)||p<0)return;s.save();const n=l(i.lineWidth,1);if(s.fillStyle=l(i.fillStyle,a),s.lineCap=l(i.lineCap,"butt"),s.lineDashOffset=l(i.lineDashOffset,0),s.lineJoin=l(i.lineJoin,"miter"),s.lineWidth=n,s.strokeStyle=l(i.strokeStyle,a),s.setLineDash(l(i.lineDash,[])),o.usePointStyle){const a={radius:p*Math.SQRT2/2,pointStyle:i.pointStyle,rotation:i.rotation,borderWidth:n},l=r.xPlus(t,g/2);Ee(s,a,l,e+u,o.pointStyleWidth&&g)}else{const o=e+Math.max((d-p)/2,0),a=r.leftForLtr(t,g),l=wi(i.borderRadius);s.beginPath(),Object.values(l).some((t=>0!==t))?He(s,{x:a,y:o,w:g,h:p,radius:l}):s.rect(a,o,g,p),s.fill(),0!==n&&s.stroke()}s.restore()}(r.x(S),P,y),S=gt(w,S+g+u,x?S+k:this.right,t.rtl),function(t,e,i){Ne(s,i.text,t,e+m/2,h,{strikethrough:i.hidden,textAlign:r.textAlign(i.textAlign)})}(r.x(S),P,y),x)f.x+=k+c;else if("string"!=typeof y.text){const t=h.lineHeight;f.y+=va(y,t)+c}else f.y+=_})),Ti(this.ctx,t.textDirection)}drawTitle(){const t=this.options,e=t.title,i=Si(e.font),s=ki(e.padding);if(!e.display)return;const n=Oi(t.rtl,this.left,this.width),o=this.ctx,a=e.position,r=i.size/2,l=s.top+r;let h,c=this.left,d=this.width;if(this.isHorizontal())d=Math.max(...this.lineWidths),h=this.top+l,c=ft(t.align,c,this.right-d);else{const e=this.columnSizes.reduce(((t,e)=>Math.max(t,e.height)),0);h=l+ft(t.align,this.top,this.bottom-e-t.labels.padding-this._computeTitleHeight())}const u=ft(a,c,c+d);o.textAlign=n.textAlign(ut(a)),o.textBaseline="middle",o.strokeStyle=e.color,o.fillStyle=e.color,o.font=i.string,Ne(o,e.text,u,h,i)}_computeTitleHeight(){const t=this.options.title,e=Si(t.font),i=ki(t.padding);return t.display?e.lineHeight+i.height:0}_getLegendItemAt(t,e){let i,s,n;if(tt(t,this.left,this.right)&&tt(e,this.top,this.bottom))for(n=this.legendHitBoxes,i=0;it.chart.options.color,boxWidth:40,padding:10,generateLabels(t){const e=t.data.datasets,{labels:{usePointStyle:i,pointStyle:s,textAlign:n,color:o,useBorderRadius:a,borderRadius:r}}=t.legend.options;return t._getSortedDatasetMetas().map((t=>{const l=t.controller.getStyle(i?0:void 0),h=ki(l.borderWidth);return{text:e[t.index].label,fillStyle:l.backgroundColor,fontColor:o,hidden:!t.visible,lineCap:l.borderCapStyle,lineDash:l.borderDash,lineDashOffset:l.borderDashOffset,lineJoin:l.borderJoinStyle,lineWidth:(h.width+h.height)/4,strokeStyle:l.borderColor,pointStyle:s||l.pointStyle,rotation:l.rotation,textAlign:n||l.textAlign,borderRadius:a&&(r||l.borderRadius),datasetIndex:t.index}}),this)}},title:{color:t=>t.chart.options.color,display:!1,position:"center",text:""}},descriptors:{_scriptable:t=>!t.startsWith("on"),labels:{_scriptable:t=>!["generateLabels","filter","sort"].includes(t)}}};class wa extends $s{constructor(t){super(),this.chart=t.chart,this.options=t.options,this.ctx=t.ctx,this._padding=void 0,this.top=void 0,this.bottom=void 0,this.left=void 0,this.right=void 0,this.width=void 0,this.height=void 0,this.position=void 0,this.weight=void 0,this.fullSize=void 0}update(t,e){const i=this.options;if(this.left=0,this.top=0,!i.display)return void(this.width=this.height=this.right=this.bottom=0);this.width=this.right=t,this.height=this.bottom=e;const s=n(i.text)?i.text.length:1;this._padding=ki(i.padding);const o=s*Si(i.font).lineHeight+this._padding.height;this.isHorizontal()?this.height=o:this.width=o}isHorizontal(){const t=this.options.position;return"top"===t||"bottom"===t}_drawArgs(t){const{top:e,left:i,bottom:s,right:n,options:o}=this,a=o.align;let r,l,h,c=0;return this.isHorizontal()?(l=ft(a,i,n),h=e+t,r=n-i):("left"===o.position?(l=i+t,h=ft(a,s,e),c=-.5*C):(l=n-t,h=ft(a,e,s),c=.5*C),r=s-e),{titleX:l,titleY:h,maxWidth:r,rotation:c}}draw(){const t=this.ctx,e=this.options;if(!e.display)return;const i=Si(e.font),s=i.lineHeight/2+this._padding.top,{titleX:n,titleY:o,maxWidth:a,rotation:r}=this._drawArgs(s);Ne(t,e.text,0,0,i,{color:e.color,maxWidth:a,rotation:r,textAlign:ut(e.align),textBaseline:"middle",translation:[n,o]})}}var ka={id:"title",_element:wa,start(t,e,i){!function(t,e){const i=new wa({ctx:t.ctx,options:e,chart:t});ls.configure(t,i,e),ls.addBox(t,i),t.titleBlock=i}(t,i)},stop(t){const e=t.titleBlock;ls.removeBox(t,e),delete t.titleBlock},beforeUpdate(t,e,i){const s=t.titleBlock;ls.configure(t,s,i),s.options=i},defaults:{align:"center",display:!1,font:{weight:"bold"},fullSize:!0,padding:10,position:"top",text:"",weight:2e3},defaultRoutes:{color:"color"},descriptors:{_scriptable:!0,_indexable:!1}};const Sa=new WeakMap;var Pa={id:"subtitle",start(t,e,i){const s=new wa({ctx:t.ctx,options:i,chart:t});ls.configure(t,s,i),ls.addBox(t,s),Sa.set(t,s)},stop(t){ls.removeBox(t,Sa.get(t)),Sa.delete(t)},beforeUpdate(t,e,i){const s=Sa.get(t);ls.configure(t,s,i),s.options=i},defaults:{align:"center",display:!1,font:{weight:"normal"},fullSize:!0,padding:0,position:"top",text:"",weight:1500},defaultRoutes:{color:"color"},descriptors:{_scriptable:!0,_indexable:!1}};const Da={average(t){if(!t.length)return!1;let e,i,s=new Set,n=0,o=0;for(e=0,i=t.length;et+e))/s.size,y:n/o}},nearest(t,e){if(!t.length)return!1;let i,s,n,o=e.x,a=e.y,r=Number.POSITIVE_INFINITY;for(i=0,s=t.length;i-1?t.split("\n"):t}function Aa(t,e){const{element:i,datasetIndex:s,index:n}=e,o=t.getDatasetMeta(s).controller,{label:a,value:r}=o.getLabelAndValue(n);return{chart:t,label:a,parsed:o.getParsed(n),raw:t.data.datasets[s].data[n],formattedValue:r,dataset:o.getDataset(),dataIndex:n,datasetIndex:s,element:i}}function Ta(t,e){const i=t.chart.ctx,{body:s,footer:n,title:o}=t,{boxWidth:a,boxHeight:r}=e,l=Si(e.bodyFont),h=Si(e.titleFont),c=Si(e.footerFont),d=o.length,f=n.length,g=s.length,p=ki(e.padding);let m=p.height,x=0,b=s.reduce(((t,e)=>t+e.before.length+e.lines.length+e.after.length),0);if(b+=t.beforeBody.length+t.afterBody.length,d&&(m+=d*h.lineHeight+(d-1)*e.titleSpacing+e.titleMarginBottom),b){m+=g*(e.displayColors?Math.max(r,l.lineHeight):l.lineHeight)+(b-g)*l.lineHeight+(b-1)*e.bodySpacing}f&&(m+=e.footerMarginTop+f*c.lineHeight+(f-1)*e.footerSpacing);let _=0;const y=function(t){x=Math.max(x,i.measureText(t).width+_)};return i.save(),i.font=h.string,u(t.title,y),i.font=l.string,u(t.beforeBody.concat(t.afterBody),y),_=e.displayColors?a+2+e.boxPadding:0,u(s,(t=>{u(t.before,y),u(t.lines,y),u(t.after,y)})),_=0,i.font=c.string,u(t.footer,y),i.restore(),x+=p.width,{width:x,height:m}}function La(t,e,i,s){const{x:n,width:o}=i,{width:a,chartArea:{left:r,right:l}}=t;let h="center";return"center"===s?h=n<=(r+l)/2?"left":"right":n<=o/2?h="left":n>=a-o/2&&(h="right"),function(t,e,i,s){const{x:n,width:o}=s,a=i.caretSize+i.caretPadding;return"left"===t&&n+o+a>e.width||"right"===t&&n-o-a<0||void 0}(h,t,e,i)&&(h="center"),h}function Ea(t,e,i){const s=i.yAlign||e.yAlign||function(t,e){const{y:i,height:s}=e;return it.height-s/2?"bottom":"center"}(t,i);return{xAlign:i.xAlign||e.xAlign||La(t,e,i,s),yAlign:s}}function Ra(t,e,i,s){const{caretSize:n,caretPadding:o,cornerRadius:a}=t,{xAlign:r,yAlign:l}=i,h=n+o,{topLeft:c,topRight:d,bottomLeft:u,bottomRight:f}=wi(a);let g=function(t,e){let{x:i,width:s}=t;return"right"===e?i-=s:"center"===e&&(i-=s/2),i}(e,r);const p=function(t,e,i){let{y:s,height:n}=t;return"top"===e?s+=i:s-="bottom"===e?n+i:n/2,s}(e,l,h);return"center"===l?"left"===r?g+=h:"right"===r&&(g-=h):"left"===r?g-=Math.max(c,u)+n:"right"===r&&(g+=Math.max(d,f)+n),{x:Z(g,0,s.width-e.width),y:Z(p,0,s.height-e.height)}}function Ia(t,e,i){const s=ki(i.padding);return"center"===e?t.x+t.width/2:"right"===e?t.x+t.width-s.right:t.x+s.left}function za(t){return Ca([],Oa(t))}function Fa(t,e){const i=e&&e.dataset&&e.dataset.tooltip&&e.dataset.tooltip.callbacks;return i?t.override(i):t}const Va={beforeTitle:e,title(t){if(t.length>0){const e=t[0],i=e.chart.data.labels,s=i?i.length:0;if(this&&this.options&&"dataset"===this.options.mode)return e.dataset.label||"";if(e.label)return e.label;if(s>0&&e.dataIndex{const e={before:[],lines:[],after:[]},n=Fa(i,t);Ca(e.before,Oa(Ba(n,"beforeLabel",this,t))),Ca(e.lines,Ba(n,"label",this,t)),Ca(e.after,Oa(Ba(n,"afterLabel",this,t))),s.push(e)})),s}getAfterBody(t,e){return za(Ba(e.callbacks,"afterBody",this,t))}getFooter(t,e){const{callbacks:i}=e,s=Ba(i,"beforeFooter",this,t),n=Ba(i,"footer",this,t),o=Ba(i,"afterFooter",this,t);let a=[];return a=Ca(a,Oa(s)),a=Ca(a,Oa(n)),a=Ca(a,Oa(o)),a}_createItems(t){const e=this._active,i=this.chart.data,s=[],n=[],o=[];let a,r,l=[];for(a=0,r=e.length;at.filter(e,s,n,i)))),t.itemSort&&(l=l.sort(((e,s)=>t.itemSort(e,s,i)))),u(l,(e=>{const i=Fa(t.callbacks,e);s.push(Ba(i,"labelColor",this,e)),n.push(Ba(i,"labelPointStyle",this,e)),o.push(Ba(i,"labelTextColor",this,e))})),this.labelColors=s,this.labelPointStyles=n,this.labelTextColors=o,this.dataPoints=l,l}update(t,e){const i=this.options.setContext(this.getContext()),s=this._active;let n,o=[];if(s.length){const t=Da[i.position].call(this,s,this._eventPosition);o=this._createItems(i),this.title=this.getTitle(o,i),this.beforeBody=this.getBeforeBody(o,i),this.body=this.getBody(o,i),this.afterBody=this.getAfterBody(o,i),this.footer=this.getFooter(o,i);const e=this._size=Ta(this,i),a=Object.assign({},t,e),r=Ea(this.chart,i,a),l=Ra(i,a,r,this.chart);this.xAlign=r.xAlign,this.yAlign=r.yAlign,n={opacity:1,x:l.x,y:l.y,width:e.width,height:e.height,caretX:t.x,caretY:t.y}}else 0!==this.opacity&&(n={opacity:0});this._tooltipItems=o,this.$context=void 0,n&&this._resolveAnimations().update(this,n),t&&i.external&&i.external.call(this,{chart:this.chart,tooltip:this,replay:e})}drawCaret(t,e,i,s){const n=this.getCaretPosition(t,i,s);e.lineTo(n.x1,n.y1),e.lineTo(n.x2,n.y2),e.lineTo(n.x3,n.y3)}getCaretPosition(t,e,i){const{xAlign:s,yAlign:n}=this,{caretSize:o,cornerRadius:a}=i,{topLeft:r,topRight:l,bottomLeft:h,bottomRight:c}=wi(a),{x:d,y:u}=t,{width:f,height:g}=e;let p,m,x,b,_,y;return"center"===n?(_=u+g/2,"left"===s?(p=d,m=p-o,b=_+o,y=_-o):(p=d+f,m=p+o,b=_-o,y=_+o),x=p):(m="left"===s?d+Math.max(r,h)+o:"right"===s?d+f-Math.max(l,c)-o:this.caretX,"top"===n?(b=u,_=b-o,p=m-o,x=m+o):(b=u+g,_=b+o,p=m+o,x=m-o),y=b),{x1:p,x2:m,x3:x,y1:b,y2:_,y3:y}}drawTitle(t,e,i){const s=this.title,n=s.length;let o,a,r;if(n){const l=Oi(i.rtl,this.x,this.width);for(t.x=Ia(this,i.titleAlign,i),e.textAlign=l.textAlign(i.titleAlign),e.textBaseline="middle",o=Si(i.titleFont),a=i.titleSpacing,e.fillStyle=i.titleColor,e.font=o.string,r=0;r0!==t))?(t.beginPath(),t.fillStyle=n.multiKeyBackground,He(t,{x:e,y:g,w:h,h:l,radius:r}),t.fill(),t.stroke(),t.fillStyle=a.backgroundColor,t.beginPath(),He(t,{x:i,y:g+1,w:h-2,h:l-2,radius:r}),t.fill()):(t.fillStyle=n.multiKeyBackground,t.fillRect(e,g,h,l),t.strokeRect(e,g,h,l),t.fillStyle=a.backgroundColor,t.fillRect(i,g+1,h-2,l-2))}t.fillStyle=this.labelTextColors[i]}drawBody(t,e,i){const{body:s}=this,{bodySpacing:n,bodyAlign:o,displayColors:a,boxHeight:r,boxWidth:l,boxPadding:h}=i,c=Si(i.bodyFont);let d=c.lineHeight,f=0;const g=Oi(i.rtl,this.x,this.width),p=function(i){e.fillText(i,g.x(t.x+f),t.y+d/2),t.y+=d+n},m=g.textAlign(o);let x,b,_,y,v,M,w;for(e.textAlign=o,e.textBaseline="middle",e.font=c.string,t.x=Ia(this,m,i),e.fillStyle=i.bodyColor,u(this.beforeBody,p),f=a&&"right"!==m?"center"===o?l/2+h:l+2+h:0,y=0,M=s.length;y0&&e.stroke()}_updateAnimationTarget(t){const e=this.chart,i=this.$animations,s=i&&i.x,n=i&&i.y;if(s||n){const i=Da[t.position].call(this,this._active,this._eventPosition);if(!i)return;const o=this._size=Ta(this,t),a=Object.assign({},i,this._size),r=Ea(e,t,a),l=Ra(t,a,r,e);s._to===l.x&&n._to===l.y||(this.xAlign=r.xAlign,this.yAlign=r.yAlign,this.width=o.width,this.height=o.height,this.caretX=i.x,this.caretY=i.y,this._resolveAnimations().update(this,l))}}_willRender(){return!!this.opacity}draw(t){const e=this.options.setContext(this.getContext());let i=this.opacity;if(!i)return;this._updateAnimationTarget(e);const s={width:this.width,height:this.height},n={x:this.x,y:this.y};i=Math.abs(i)<.001?0:i;const o=ki(e.padding),a=this.title.length||this.beforeBody.length||this.body.length||this.afterBody.length||this.footer.length;e.enabled&&a&&(t.save(),t.globalAlpha=i,this.drawBackground(n,t,s,e),Ai(t,e.textDirection),n.y+=o.top,this.drawTitle(n,t,e),this.drawBody(n,t,e),this.drawFooter(n,t,e),Ti(t,e.textDirection),t.restore())}getActiveElements(){return this._active||[]}setActiveElements(t,e){const i=this._active,s=t.map((({datasetIndex:t,index:e})=>{const i=this.chart.getDatasetMeta(t);if(!i)throw new Error("Cannot find a dataset at index "+t);return{datasetIndex:t,element:i.data[e],index:e}})),n=!f(i,s),o=this._positionChanged(s,e);(n||o)&&(this._active=s,this._eventPosition=e,this._ignoreReplayEvents=!0,this.update(!0))}handleEvent(t,e,i=!0){if(e&&this._ignoreReplayEvents)return!1;this._ignoreReplayEvents=!1;const s=this.options,n=this._active||[],o=this._getActiveElements(t,n,e,i),a=this._positionChanged(o,t),r=e||!f(o,n)||a;return r&&(this._active=o,(s.enabled||s.external)&&(this._eventPosition={x:t.x,y:t.y},this.update(!0,e))),r}_getActiveElements(t,e,i,s){const n=this.options;if("mouseout"===t.type)return[];if(!s)return e.filter((t=>this.chart.data.datasets[t.datasetIndex]&&void 0!==this.chart.getDatasetMeta(t.datasetIndex).controller.getParsed(t.index)));const o=this.chart.getElementsAtEventForMode(t,n.mode,n,i);return n.reverse&&o.reverse(),o}_positionChanged(t,e){const{caretX:i,caretY:s,options:n}=this,o=Da[n.position].call(this,t,e);return!1!==o&&(i!==o.x||s!==o.y)}}var Na={id:"tooltip",_element:Wa,positioners:Da,afterInit(t,e,i){i&&(t.tooltip=new Wa({chart:t,options:i}))},beforeUpdate(t,e,i){t.tooltip&&t.tooltip.initialize(i)},reset(t,e,i){t.tooltip&&t.tooltip.initialize(i)},afterDraw(t){const e=t.tooltip;if(e&&e._willRender()){const i={tooltip:e};if(!1===t.notifyPlugins("beforeTooltipDraw",{...i,cancelable:!0}))return;e.draw(t.ctx),t.notifyPlugins("afterTooltipDraw",i)}},afterEvent(t,e){if(t.tooltip){const i=e.replay;t.tooltip.handleEvent(e.event,i,e.inChartArea)&&(e.changed=!0)}},defaults:{enabled:!0,external:null,position:"average",backgroundColor:"rgba(0,0,0,0.8)",titleColor:"#fff",titleFont:{weight:"bold"},titleSpacing:2,titleMarginBottom:6,titleAlign:"left",bodyColor:"#fff",bodySpacing:2,bodyFont:{},bodyAlign:"left",footerColor:"#fff",footerSpacing:2,footerMarginTop:6,footerFont:{weight:"bold"},footerAlign:"left",padding:6,caretPadding:2,caretSize:5,cornerRadius:6,boxHeight:(t,e)=>e.bodyFont.size,boxWidth:(t,e)=>e.bodyFont.size,multiKeyBackground:"#fff",displayColors:!0,boxPadding:0,borderColor:"rgba(0,0,0,0)",borderWidth:0,animation:{duration:400,easing:"easeOutQuart"},animations:{numbers:{type:"number",properties:["x","y","width","height","caretX","caretY"]},opacity:{easing:"linear",duration:200}},callbacks:Va},defaultRoutes:{bodyFont:"font",footerFont:"font",titleFont:"font"},descriptors:{_scriptable:t=>"filter"!==t&&"itemSort"!==t&&"external"!==t,_indexable:!1,callbacks:{_scriptable:!1,_indexable:!1},animation:{_fallback:!1},animations:{_fallback:"animation"}},additionalOptionScopes:["interaction"]};return Tn.register(Un,$o,go,t),Tn.helpers={...Hi},Tn._adapters=In,Tn.Animation=As,Tn.Animations=Ts,Tn.animator=bt,Tn.controllers=nn.controllers.items,Tn.DatasetController=js,Tn.Element=$s,Tn.elements=go,Tn.Interaction=Ki,Tn.layouts=ls,Tn.platforms=Ds,Tn.Scale=tn,Tn.Ticks=ae,Object.assign(Tn,Un,$o,go,t,Ds),Tn.Chart=Tn,"undefined"!=typeof window&&(window.Chart=Tn),Tn})); +//# sourceMappingURL=chart.umd.min.js.map From b8ddccd1ce04cf535c69c196f20a675ecb39993d Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Wed, 16 Sep 2026 17:02:19 +0200 Subject: [PATCH 06/18] Add reference-chains design and architecture docs Design doc for reference chains on surviving live-heap samples, architecture notes, explanation of the emitted signals, and a collection summary of what lands in JFR recordings. --- doc/architecture/LiveHeapReferenceChains.md | 610 ++++++++++++++++++ .../ReferenceChains-SignalsExplained.md | 366 +++++++++++ doc/reference-chains-collection-summary.md | 87 +++ doc/reference-chains-design.md | 259 ++++++++ 4 files changed, 1322 insertions(+) create mode 100644 doc/architecture/LiveHeapReferenceChains.md create mode 100644 doc/architecture/ReferenceChains-SignalsExplained.md create mode 100644 doc/reference-chains-collection-summary.md create mode 100644 doc/reference-chains-design.md diff --git a/doc/architecture/LiveHeapReferenceChains.md b/doc/architecture/LiveHeapReferenceChains.md new file mode 100644 index 0000000000..5276da3f6b --- /dev/null +++ b/doc/architecture/LiveHeapReferenceChains.md @@ -0,0 +1,610 @@ +# Reference Chains for Surviving Live Heap Samples + +**Status:** Implemented (see "Implementation status" below) +**Date:** 2026-07-07 +**Jira:** [PROF-15341](https://datadoghq.atlassian.net/browse/PROF-15341) + +## Implementation status + +The "Chosen design" section below has been implemented following +`LiveHeapReferenceChains-ImplementationPlan.md` (kept locally, not committed) +(Phases 0-7). It is off by default; the shipping switch is the `referencechains` argument +parsed by `Arguments` (`arguments.cpp`'s `CASE("referencechains")`), e.g. +`referencechains=true:hops=64:budget=2000:ttl=60000:framecap=65536`. + +Read this status note alongside the actual code before relying on it, not instead of it: + +- **The BFS engine (frontier table, tag lifecycle, incremental resumption, termination, + JFR event shapes) is implemented and unit-tested** (`ddprof-lib/src/main/cpp/referenceChains.h`/ + `.cpp`, `ddprof-lib/src/test/cpp/referenceChains_ut.cpp`). +- **The lifecycle gap is closed: it now runs inside a live profiling session.** + `Profiler::start()` (`profiler.cpp`) calls `ReferenceChainTracker::instance()->start(args)` + (gated on `args._reference_chains`, independent of the CPU/wall/alloc engine mask, the + same way `malloc_tracer`/`NativeSocketSampler` are gated on their own flags) followed by + the new `ReferenceChainTracker::startThread()`, which spawns the BFS thread + (`threadLoop()`) - safe there because the JVM/JVMTI environment is already fully up by + that point in the lifecycle, unlike inside `start()` itself, which must stay callable + with no live JVM for `referenceChains_ut.cpp`'s tests. `Profiler::stop()` calls the + matching `stopThread()`/`stop()` pair. Because `start()` runs, `SetEventNotificationMode` + for the GC callbacks is now actually invoked, so `onGCStart()`/`onGCFinish()` fire and the + BFS thread's `shouldRunPass()` scheduling loop (GC-epoch signal or the fixed cadence) is + live. A `datadog.ReferenceChainAbandoned` event now reaches a real `Recording`: when a + dump is requested (`Profiler::dump()`, the same call site that already flushes + `LivenessTracker`) and the search's state is `SearchState::ABANDONED`, + `buildAbandonedEvent()`'s output is written via the new + `Profiler::writeReferenceChainAbandoned()` / `FlightRecorder::recordReferenceChainAbandoned()` + wrappers (mirroring `writeHeapUsage()`'s exact shape). +- **`buildChainEvent()` now has a call site: the target-selection feed is closed.** + `ReferenceChainTracker::pollWatchedTargets()` (referenceChains.cpp), called from + `threadLoop()` once per scheduling cycle after `runPass()`, is that feed. It polls + `LivenessTracker::selectLeakCandidates()` (the positive population-slope ranking, Open + Question 3 below) and, for each ranked klass's live representative instance that an + ordinary `runPass()` walk has *already* tagged (`getTag() > 0` - a read, never a `SetTag` + seed; see Open Question 3 for why the design's original seeding proposal was replaced), + calls `buildChainEvent(tag, ...)`, which is cached (`cacheResolvedChain()`) rather than + written immediately. The actual JFR write happens later and on a different thread: enqueued + via `enqueueChainEvent()` and drained by `Profiler::dump()` -> `drainPendingChainEvents()` -> + `Profiler::writeReferenceChain()` / `FlightRecorder::recordReferenceChain()` (`profiler.cpp` + lines ~1960-1974), decoupling the walk from JFR I/O. Deduplication is `_resolved_chains` + (an `unordered_map` keyed by klass_id, `referenceChains.h`), not a + per-search tag set, so a klass flagged across consecutive polls emits its chain only once. + This is gated on `_gc_generations` *and* + liveness tracking both being enabled - `referencechains=...` alone still gets the + whole-graph-only behavior (no target seeding), resolving Open Question 3's "still + undecided" fallback. See + `ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainTrackingTest.java`: + `shouldReportAbandonedSearchOnTinyFrontierCap` exercises the abandonment path. + `shouldReconstructReferrerChainToGcRoot` (Phase E's own exit criterion) is no longer + `@Disabled`: it allocates a growing, real population of a fixture class, drives GCs and + `Profiler::dump()` calls until `LivenessTracker::selectLeakCandidates()` trusts the resulting + trend, then asserts on the `datadog.ReferenceChain` event `pollWatchedTargets()` produces - + a real end-to-end exercise of this whole mechanism against a live JVM, not a synthetic + frontier fixture. +- **Phase 5's tuning defaults are provisional, not empirically finalized.** The hop cap, + per-pass budget, TTL, and frontier-size cap (`arguments.h`'s `DEFAULT_REFERENCE_CHAINS_*` + constants) are explicitly-labeled placeholders; no benchmark against this codebase has + run yet (see Open Question 2 below and the implementation plan's Phase 5). + +## Goal + +For a subset of live-heap samples that survive past their allocation window, produce +a **reference chain** — a sequence of referrer *types* (not full field-level paths, not +necessarily to *all* GC roots) connecting the sampled instance back to *a* GC root. This +is diagnostic information ("what kind of object chain is keeping this alive"), not a +heap-dump-grade exact retainer analysis. + +## Constraints + +- Must run cheaply, with as short a safepoint / STW contribution as possible. +- Must work on stock vendor JDKs the agent attaches to — no forked/patched JVM builds. +- Exhaustive (all-roots, full-path) chains are explicitly **not** required; referrer-type-only, + bounded-depth, best-effort chains are acceptable. + +## Approaches considered + +Three approaches were evaluated; two are ruled out as launch requirements for concrete, +evidence-backed reasons. One sub-idea (Approach C's `ParallelObjectIterator` variant) is +explicitly kept open as a conditional future option; see its discussion below. + +| # | Approach | Completeness | Complexity | Feasibility | Status | +|---|---|---|---|---|---| +| A | Full JVMTI `FollowReferences` reverse-graph walk, piggybacked on an already-scheduled major GC | 4/5 | 4/5 | 2/5 | Rejected | +| B | Bounded BFS-from-roots with frontier pruning (JFR "leak profiler" technique, adapted) | 3/5 | 3/5\* | 4/5 | **Chosen** | +| C | Hook GC mark/copy closures (G1, ZGC) to record parent pointers inline during marking | 2/5 | 5/5 | 1/5 | Rejected | + +Scale (1-5 for each column): Completeness — higher is more complete (5 = closest to exhaustive all-roots/full-path); Complexity — higher is more complex to implement/maintain (5 = most complex, lower is better); Feasibility — higher is more feasible to ship on stock vendor JDKs (5 = most feasible). No single column dominates the decision; see the per-approach rationale below for why B was chosen despite not scoring highest on every column. + +\* This 3/5 reflects only the single-pass BFS sketched at selection time. The "Chosen +design" section below replaces that sketch with an incremental, resumable BFS +(JVMTI-tag-based frontier persistence across GC cycles, an agent-thread-driven pass that +gets its safepoint transparently from the JVMTI heap-walk call it makes, GC-callback +signaling, and explicit termination/tag-cleanup bookkeeping), which is materially more +complex than this score suggests — closer to 4/5 in implementation and maintenance +effort. The score is left unchanged above (it documents the state of the comparison at +decision time) rather than retroactively edited. + +### A — Full reverse-reachability walk (rejected) + +Safepoint length scales with live-set size regardless of how the walk is triggered. +Modern regionalized collectors (G1, Shenandoah) rarely perform a true full-heap walk +during ordinary major GCs, so "ride an already-paid pause" is not a reliable amortization +strategy. Cost is fundamentally at odds with the "short safepoint" constraint. + +### B — Bounded BFS-from-roots (chosen) + +Mirrors OpenJDK's own `jdk.OldObjectSample` leak-profiler implementation +(`src/hotspot/share/jfr/leakprofiler/chains/{edgeStore,bfsClosure,dfsClosure}.cpp`): +a `VM_Operation`-driven BFS from GC roots, retaining only edges on the frontier toward a +small, fixed sample set, with a hard hop cap (HotSpot itself caps chains at ~200 hops, +split 100/100 from leaf and from root). We can go cheaper than JFR because only the +**referrer class**, not object identity or field name, is needed — the `EdgeStore` +degenerates to `(referrer_klass, parent_tag, depth)` records, where `parent_tag` links +each record back to the record that discovered it, enabling chain reconstruction. + +Adopting this pattern is a re-scoping of proven, shipping HotSpot code, not a novel +algorithm design. + +### C — GC mark/copy closure piggyback (rejected) + +Investigated specifically for G1 and ZGC on the premise that per-edge referrer +information is already available inside the collector's own marking/evacuation closures +(`G1ParCopyClosure::do_oop_work`, ZGC's `ZMarkConcurrentRootsIteratorClosure` / +load-barrier closures), so recording it would cost nothing beyond what the GC already +pays. + +Rejected because there is no stable, externally reachable hook into these closures: + +- They are internal, template-instantiated C++ classes compiled into `libjvm.so` at + HotSpot build time — not a registrable/pluggable extension point. +- This differs categorically from `VMStructs`-style introspection already used in this + codebase (`ddprof-lib/src/main/cpp/hotspot/vmStructs.cpp`), which reads VM state + passively via an officially exported offset table. Intercepting a GC closure's + *behavior* would require either shipping a patched OpenJDK build (a fork/maintenance + commitment far beyond anything in this codebase) or binary-patching unversioned, + per-build-mangled function addresses — not shippable across JDK point releases. + +A related idea — using HotSpot's internal `ParallelObjectIterator` +(landed via [JDK-8322043](https://www.mail-archive.com/serviceability-dev@openjdk.org/msg12977.html), +used by `VM_HeapDumper` to partition heap regions across GC worker threads for parallel +heap dumping) to shrink Approach B's safepoint by parallelizing the walk — was also +investigated. Same verdict: it is an internal C++ class, not exposed via JVMTI, with no +stable ABI for an attached agent to call. Symbol-sniffing internal HotSpot functions *is* +an established pattern in this codebase (`VMStructs::findHeapUsageFunc`, +`vmStructs.cpp:489-509`), but that precedent covers a single leaf virtual method with a +value/POD-ish return; `ParallelObjectIterator` is a multi-class subsystem that coordinates +the VM's own GC worker threads under safepoint control — an order of magnitude larger +fragility surface, with a much higher blast radius if a layout assumption is wrong (GC +worker-thread coordination corruption vs. a bad JMX stat). Not pursued as a launch +requirement; revisit only if Approach B's single-threaded pause proves to be a measured +bottleneck, and treat it as an isolated, heavily version/flag-gated fast path with +automatic fallback — never a dependency. + +## Chosen design: incremental, resumable bounded BFS + +A single-pass bounded BFS still means one pause sized to whatever budget is configured. +The refinement below spreads that budget across multiple short passes instead of one +contiguous one, trading a possibly-higher *aggregate* STW total for a much better +*latency distribution* — no single long tail pause. + +### Why the frontier can survive across passes: JVMTI object tags + +The obstacle to pausing and resuming a BFS is that the frontier (the worklist of +not-yet-expanded objects) is normally a set of raw addresses, and a moving/compacting GC +between passes can relocate or collect any of them. + +JVMTI object tags solve this: + +- Tags are identity-based and GC-move-transparent — a tagged object can be re-resolved + after a GC regardless of where it moved. +- Tags are **non-retaining** — tagging does not keep an object alive. This is a *new* + subsystem dependency, not a reuse of one: the existing live-object sampler + (`LivenessTracker`, `livenessTracker.cpp`) does not use JVMTI tags at all — it + correlates sampled objects via JNI weak global references (`NewWeakGlobalRef`) held in + its own index table with its own locking and GC-triggered cleanup + (`livenessTracker.cpp:327-357`, `:53-70`). Non-retention is a property both mechanisms + happen to share, not evidence that this reuses proven infrastructure. +- **Investigated replacing tags outright with `LivenessTracker`'s weak-ref + index-table + pattern — resolved as "adopt the table pattern, keep the tags."** The pattern cannot + fully substitute for tags: `FollowReferences`/`IterateThroughHeap` (the calls that + actually discover a frontier object's referrers) can filter/report against a *tagged* + object set natively; a JNI weak-ref table has no hook into that machinery, so the + frontier-discovery step would still need tagged objects regardless of what stores the + metadata. What the investigation *does* carry over: `LivenessTracker`'s proven + `TrackingEntry`-style slot table — CAS-based index allocation, a signal-safe `SpinLock` + (`spinLock.h`), doubling-resize, and GC-epoch-triggered cleanup + (`livenessTracker.cpp:152-176`, `:213-278`, `:369-409`) — is a better-precedented design + for the frontier's *metadata* storage than inventing one from scratch, since a JVMTI tag + is a single `jlong` with no room for `(parent_tag, referrer_klass, depth)` on its own. + Recommendation: use the tag as an index into a `TrackingEntry`-style table (fields: + `parent_tag`, `referrer_klass`, `depth`) rather than encoding all three into the tag + value or a from-scratch hashmap. See "Frontier metadata storage" below and Open + Question 4. +- Non-retention gives incremental resumption a useful side effect for free: if a frontier + object dies between passes, it simply fails to re-resolve on the next pass. That branch + of the search is pruned automatically, with no extra liveness bookkeeping required. + +### Data structures + +- **Frontier**: a set of `(tag, parent_tag, referrer_klass, depth)` records. `tag` is the + JVMTI tag assigned to a not-yet-expanded object; `parent_tag` links back for chain + reconstruction; `depth` supports the hop cap. +- **EdgeStore**: accumulates `(referrer_klass, parent_tag, depth)` per discovered edge for + objects that are on a path toward a target sample. Keyed by tag, not address — + degenerate relative to JFR's `EdgeStore` since object identity/field names are not + required, but it retains the same `parent_tag` linkage field as the Frontier so a chain + can be walked back from a target sample to a root by following `parent_tag` across + EdgeStore records. + +### Frontier metadata storage: reusing `LivenessTracker`'s table pattern + +A JVMTI tag is one `jlong` — it can identify a frontier object and make it visible to +`FollowReferences`/`IterateThroughHeap`, but it cannot itself hold the three fields +(`parent_tag`, `referrer_klass`, `depth`) each Frontier/EdgeStore record needs. Two ways +to close that gap were considered: + +1. Build a bespoke hashmap keyed by tag value, from scratch. +2. Reuse `LivenessTracker`'s existing slot-table design (`livenessTracker.h:21-30` + `TrackingEntry`, `livenessTracker.cpp:152-176` sizing, `:213-278`/`:369-409` CAS slot + allocation and doubling resize, `spinLock.h`'s signal-safe `SpinLock`), using the tag + value as the slot index instead of a `jweak` as the identity handle. + +(2) is the better-precedented choice — it's shipping code, already exercises the exact +"per-slot payload, GC-cycle-driven cleanup, contention-safe locking" shape this needs — +provided the sizing formula is **not** copied as-is. `LivenessTracker` sizes its table +from `max_heap / sampling_interval` (a flat allocation-sample rate, +`livenessTracker.cpp:152-176`, capped at `MAX_TRACKING_TABLE_SIZE = 262144`, +`livenessTracker.h:39`); a BFS frontier's width is driven by per-hop fan-out in the object +graph, not by an allocation rate, and multiple concurrent searches (one per live-heap +sample being chased, see Open Question 3) each need their own capacity — the existing +formula does not transfer and a new one is an open question (folded into Open Question 2). + +### Algorithm + +1. Seed the frontier from GC roots (first pass) or from the persisted frontier + (resumed pass). +2. Resolve currently-live tagged frontier objects. Objects that fail to resolve are + dropped (dead — free pruning). +3. Expand the frontier up to a fixed per-pass budget (edge count or time slice). +4. Newly discovered objects are tagged and added to the frontier for the next pass. +5. Persist the frontier (native memory owned by the agent, not thread-local scratch) and + return control to the VM. +6. Repeat until: a target sample is reached, the hop cap is hit, or a per-search + abandonment limit (see Termination) is exceeded. + +### Triggering passes: resolved — the profiler never schedules its own safepoint + +Investigated whether pass-continuation work could ride the JVMTI +`GarbageCollectionStart`/`GarbageCollectionFinish` callbacks — the same callback this +codebase already uses to call `_heap_usage_func` (`vmStructs.cpp`) — instead of each pass +paying for its own safepoint. + +**Correction to an earlier framing in this doc**: a pass does not run "inside a dedicated +`VM_Operation::doit()`" that the profiler constructs — HotSpot's `VM_Operation`/ +`VMThread::execute()` machinery is internal, unexported C++ with no agent-facing entry +point; nothing outside HotSpot can submit one. What actually happens, confirmed against +`src/hotspot/share/prims/jvmtiTagMap.cpp` and `jvmtiEnv.cpp`: +`SetTag`/`GetTag` need no safepoint at all — they take only a `MutexLocker` over a +JVM-internal "hot lock" on the tag map. `FollowReferences`/`IterateThroughHeap` **do** +bring the VM to a safepoint, but the JVM does this internally and transparently +(`VM_HeapWalkOperation`/`VM_HeapIterateOperation`, dispatched via +`VMThread::execute()` *inside* HotSpot's own implementation of those calls) the moment an +ordinary attached agent thread calls them — the calling thread simply blocks until the +walk finishes. A pass is therefore: an agent-owned, already-attached thread (the same kind +`LivenessTracker` already runs on, see `livenessTracker.cpp:303-409`) calling +`FollowReferences`/`IterateThroughHeap` directly; the safepoint is a side effect of that +call, not something the profiler builds or schedules. + +**Confirmed the VM is genuinely at a safepoint (all mutators stopped) for the full +duration of both callbacks**, on every collector: + +- JVMTI spec: *"This event is sent while the VM is still stopped... the event handler + must not use JNI functions and must not use JVM TI functions except those which + specifically allow such use (see the raw monitor, memory management, and environment + local storage functions)."* +- openjdk/jdk source: delivery is synchronous on the VMThread + (`src/hotspot/share/prims/jvmtiExport.cpp:2752-2790`, comment *"this event is posted + from VM-Thread"*); every call site is inside a safepoint-executing `VM_Operation::doit()`, + backed by explicit asserts — e.g. Parallel GC's + `assert(SafepointSynchronize::is_at_safepoint())` (`gc/parallel/psScavenge.cpp:305-306`), + G1's `assert_at_safepoint_on_vm_thread()` (`gc/g1/g1VMOperations.cpp:141-157`), + Shenandoah and ZGC wrapping the same `SvcGCMarker` only inside their respective + `VM_Operation`/`VM_ZOperation::doit()` paths. Stable JDK 11 → mainline, across + Serial/Parallel/G1/Shenandoah/ZGC. + +**But this does not make the GC-triggered callback itself usable as the execution vehicle +for a pass.** The "functions which specifically allow such use" are exactly two: +`Allocate` and `Deallocate` (the entire **Memory Management** category). `SetTag`, +`GetTag`, `GetObjectsWithTags`, `FollowReferences`, and `IterateThroughHeap` are all in +the **Heap** category, which is *not* on that allowlist — calling any of them from inside +`GarbageCollectionStart`/`Finish` is exactly what the restriction forbids. The spec's own +prescribed escape hatch — notify a raw monitor from the callback, do the real work on a +separate agent thread — doesn't preserve "the pass rides the GC's own pause" property +either: the woken agent thread runs after the GC's collection pause has already ended +(mutators resumed), so calling `FollowReferences`/`IterateThroughHeap` there triggers a +**new**, separate safepoint of its own (per the corrected mechanism above) rather than +reusing the GC's. + +The only way to fold the tag/walk work into the GC's own STW window would be to bypass +the official JVMTI entry points and reach into HotSpot's internal `JvmtiTagMap` directly +via symbol-sniffing — reintroducing exactly the fragility class already rejected for +Approach C (unversioned internal C++ state, no stable ABI). Doing that here would undo +the reason C was rejected. + +**Conclusion: "no new marginal safepoints" is not achievable while staying within +official JVMTI usage.** Each pass still triggers its own safepoint — transparently, via +whichever agent thread calls `FollowReferences`/`IterateThroughHeap` for that pass, not +via anything the profiler schedules itself. The GC callbacks remain useful only as a +low-cost *signal* ("a GC just happened, a pass may be worth running soon") — not as the +execution vehicle for the pass itself. This does not change the core incremental design +(frontier persistence via JVMTI tags, self-pruning of dead branches, per-pass budget) — +it only removes the "zero marginal safepoints" claim from the cost/benefit case. The +design's actual value remains what it was framed as: trading one long pause for several +short, independently-triggered ones — a latency-distribution improvement, not a +total-STW reduction. + +### Termination and abandonment + +Because passes are spread across a mutating heap, a search that never reaches a root or +the hop cap could otherwise persist indefinitely, accumulating abandoned frontier state +across GC cycles. Required cutoffs: + +- Hop cap (as in Approach B's single-pass form). +- A hard cap on passes-per-search or wall-clock TTL from first observation (value TBD — + see Open Question 2). +- Explicit reporting of abandoned searches (no silent truncation) so this shows up as a + measurable "chain not found within budget" outcome rather than being indistinguishable + from "no chain exists." +- Tag release: on abandonment or completion, every JVMTI tag this search assigned to + frontier/`EdgeStore` objects (`SetTag(obj, 0)`) must be cleared before the search's + state is discarded. Without this, an abandoned search leaves its tags in place + indefinitely, directly aggravating the tag-table sizing/contention risk raised in + Open Question 4. +- A hard cap on frontier size (record count or native-memory footprint). The hop cap and + pass/TTL cap bound how *long* a search runs, but not how *wide* the frontier can grow + within that time — a wide fan-out graph could accumulate an unbounded number of + `(tag, parent_tag, referrer_klass, depth)` records before either cutoff is hit. When + the cap is reached, stop admitting new frontier entries for that search and report it + as an abandoned/truncated search (value TBD — see Open Question 2). + +### Correctness note: chains are historical, not a single consistent snapshot + +A chain built across multiple passes stitches together `"A referenced B"` facts observed +at different points in time, not one frozen graph. For the stated purpose — explaining, +by referrer type, what typically retains this class of surviving object — this is +sufficient, and is not meaningfully weaker than a single-pass walk: GC roots (e.g. thread +stack frames) are themselves a live-changing set across a single pause's boundary, so +"one true snapshot" is already an approximation in the single-pass case. Any +documentation or output surface built on this must describe results as an **observed** +retaining path, not a claim about the object's current exact retention state. + +### Cost/benefit summary + +- **Does not reduce total STW time.** Each safepoint/callback entry pays fixed + synchronization overhead; K short increments likely sum to equal or *more* aggregate + pause time than one contiguous walk covering the same work. +- **Improves latency distribution.** No single long tail pause — the thing most likely to + actually affect deployed application health (p99 latency, heartbeat timeouts), even + when total accumulated pause-ms is flat or slightly worse. + +## Non-goals + +- Exhaustive paths to all GC roots. +- Field-level or object-identity-level chains (referrer *type* only). +- Any GC-internal-closure hook (Approach C) or internal parallel-iteration API use as a + launch dependency. + +## Open questions before implementation + +1. ~~Confirm `GarbageCollectionStart`/`GarbageCollectionFinish` callback timing relative to + safepoint release.~~ **Resolved** (see Triggering section): the callback is genuinely + at a safepoint, but the JVMTI Heap-category functions needed to do frontier work + (`SetTag`/`GetTag`/`FollowReferences`/`IterateThroughHeap`) are not in the callback's + allowed function set. Each pass instead triggers its own safepoint transparently, via + whichever agent thread calls `FollowReferences`/`IterateThroughHeap` for that pass — + the profiler never constructs a `VM_Operation` itself (see correction in Triggering + section). The "no new marginal safepoints" framing is dropped; the design's value is + latency distribution, not total-STW reduction. +2. Choose per-pass budget defaults (edge count vs. time slice), hop cap, the + passes-per-search/wall-clock TTL abandonment cutoff, and the frontier-size/memory cap + (see Termination and "Frontier metadata storage") — needs measurement against + representative heap shapes and per-hop fan-out, not a guess; `LivenessTracker`'s + flat-sample-rate sizing formula does not transfer to a graph-search frontier. + **Not resolved — provisional defaults only, no measurement has occurred.** The + implementation currently ships explicitly-labeled "provisional default pending Phase 5 + empirical tuning" constants (`arguments.h`: `DEFAULT_REFERENCE_CHAINS_HOP_CAP = 200`, + citing this doc's own JFR ~200-hop/100-100 precedent; `DEFAULT_REFERENCE_CHAINS_BUDGET + = 1000`; `DEFAULT_REFERENCE_CHAINS_TTL_MS = 60000`; `DEFAULT_REFERENCE_CHAINS_FRONTIER_CAP + = 65536`, sized as a fraction of `LivenessTracker::MAX_TRACKING_TABLE_SIZE` rather than + derived from any BFS-specific measurement; plus `referenceChains.h`'s + `FrontierTable::INITIAL_TABLE_CAPACITY = 1024` and + `ReferenceChainTracker::PASS_CADENCE_NS` = 1 s). These let the subsystem run and be + tested end-to-end, but none are backed by a benchmark against this codebase — do not + describe them as measured. The real resolution path is + `LiveHeapReferenceChains-BenchmarkPlan.md` (kept locally, not committed), + which specifies the JMH/async-profiler matrix and decision rule Phase 5 still needs to + execute; this question stays open until that plan is actually run. + + **Pause-time-SLO feedback loop — SHIPPED, reusing the existing `PidController`.** + Implemented in + `LiveHeapReferenceChains-RemainingWorkPlan.md` (kept locally, not committed)'s + Phase D (`ReferenceChainTracker::updatePacing()`, `referenceChains.cpp`). This does not + replace the hop/TTL/frontier-cap constants raised in the first half of this question — only + the per-pass edge-count budget and the pass cadence, per the shipped mechanism below. + - New config sub-option `referencechains=...:pausetarget=` (`arguments.cpp`'s + `CASE("referencechains")` parser, field `Arguments::_reference_chains_pause_target_ms`), + defaulting to `arguments.h`'s `DEFAULT_REFERENCE_CHAINS_PAUSE_TARGET_MS = 5` — explicitly + labeled provisional/un-benchmarked, the same way the other + `DEFAULT_REFERENCE_CHAINS_*` constants are. Choosing its real value is still a Phase-5-style + empirical question, not resolved by this mechanism landing. + - `ReferenceChainTracker::start()` (re)constructs its own `PidController` instance + (`_pause_pid`) targeting `_pause_target_ms`, with its own gain triple — + **not** `ObjectSampler`/`MallocTracer`'s shared, uncited P=31/I=511/D=3/cutoff=15s + (`NativeSocketSampler` uses its own `RateLimiter`, not this `PidController` triple, and + was never part of the shared instance). The caveat this question raised about that triple + being copy-pasted, not independently derived, still stands for those two; it was not + "resolved," just not repeated here. The new instance uses P=10/I=1/D=2/window=1/cutoff=5s + — smaller proportional gain than the shared triple because a pass-duration-ms error is + single/low-double-digit in magnitude, unlike the shared triple's event-count scale + (`referenceChains.cpp`'s `start()`, inline comment on each gain). Gain *convergence* is + verified by gtest (three `ReferenceChainsTest` cases: steady-state at the ceiling, over- + ceiling, under-ceiling — see Phase D's exit criteria below), not by a live benchmark + against representative heap shapes; that remains a + `LiveHeapReferenceChains-BenchmarkPlan.md` (kept locally, not committed) item, + not fully closed by this mechanism landing. + - Measurement point: `runPass()` (`referenceChains.cpp`) times its own root + `IterateOverReachableObjects` call (first pass) or `expandFrontier()`'s + `GetObjectsWithTags`+`FollowReferences` pair (resumed pass) — already the thread blocked + inside the safepoint those calls trigger (Triggering section) — and converts to whole + milliseconds before feeding `_pause_pid.compute()` (matching every other `PidController` + caller in this codebase, which all feed integer counts). + - `updatePacing(u64 pass_wall_ns)` folds the budget-scaling and cadence-widening/relaxing + decisions into that one `compute()` call per pass: the signal is added to + `_effective_budget` (the *value* `runPass()` now passes to `expandFrontier()` instead of + the fixed `_budget`) and clamped to `[MIN_EFFECTIVE_BUDGET = 50, _budget + _borrowed_budget]` + — a later addition lets the ceiling temporarily borrow above `_budget` (up to + `BORROW_CEILING_MULTIPLIER`x) rather than capping hard at the config value; see the + rotation-budget-starvation fix in git history for why. Whatever the clamp could not + absorb (`overflow`) drives `_effective_cadence_ns` (see Open + Question 5 below for why cadence is folded into this same output rather than a second + controller): `CADENCE_NS_PER_EDGE_OVERFLOW = 1ms/edge` scales the unabsorbed overflow into + a nanosecond adjustment, widening `_effective_cadence_ns` toward + `MAX_EFFECTIVE_CADENCE_NS = 4s` when still over-ceiling even at the budget floor, relaxing + it toward `MIN_EFFECTIVE_CADENCE_NS = 10ms` when comfortably under-ceiling even at the + budget ceiling. + - The hop cap and the frontier-size hard cap (Termination section) are untouched by this + mechanism — they stay fixed correctness/memory-safety bounds, not controller-tuned, exactly + as this question originally specified. + - One known, deliberate scope limit carried over from the plan: `buildAbandonedEvent()`'s + `datadog.ReferenceChainAbandoned` event still reports the static config ceiling `_budget`, + not the adaptive `_effective_budget` — changing that event's semantics was out of Phase D's + stated scope. +3. Decide the sample-batching policy: one incremental search per live-heap sample, or + batched multi-target BFS sharing a single frontier walk (batching amortizes better but + couples unrelated samples' termination conditions together). + **Shipped, but not in either form this question anticipated.** The implemented + `ReferenceChainTracker::runPass()` (referenceChains.cpp) does not target any sample at + all: it runs a single, singleton-owned search that walks the whole root-reachable graph + (bounded by the hop/budget/frontier caps) with no per-sample seeding. Reconstructing a + chain for a specific tag is a separate, read-only step (`buildChainEvent(target_tag, ...)`) + applied after (or during) that one shared search - closer in spirit to "batched" (one + frontier walk can answer for many targets) than "one search per sample", but arrived at + by omission (the target-sample feed did not exist at that time, see the implementation + plan's Phase 7 report) rather than a deliberate batching design. + Whether this generalizes to true multi-target batching (explicit seeding from multiple + samples, coordinated termination) is still open and deferred, consistent with this + question's original framing. + + **Target-selection policy — SHIPPED (positive population-slope ranking).** Implemented in + `LiveHeapReferenceChains-RemainingWorkPlan.md` (kept locally, not committed)'s + Phases A-C. The missing piece above was *which* tag(s) `buildChainEvent()` should + reconstruct for. As shipped: per klass, `LivenessTracker` tracks a rolling window of its + live tracked-instance population count, sampled once per `LivenessTracker::cleanup_table()` + epoch advance (the same GC-epoch cadence that already recomputes survivor status, + `livenessTracker.cpp` — a *different*, slower cadence than `ReferenceChainTracker`'s own BFS pass cadence, Open + Question 5; "past N passes" below means GC epochs observed by `LivenessTracker`, not BFS + passes). A klass whose population trend over that window is positive — new instances + arriving faster than old ones are dying — is a leak candidate; a bounded cache/pool also + holds old objects but its population stabilizes or shrinks. Of all klasses with a + positive trend, seed only the top 3–5 by trend magnitude for the next BFS pass — this + doubles as the per-pass seeding cap this question's last bullet asks for, so no separate + budget constant is needed. + + Mechanics (as shipped): + - `LivenessTracker` resolves the class lazily at JFR-flush time (`flush_table()`) to keep + the allocation-sampling path free of a `GetObjectClass` call. Per-klass population is + computed by resolving the klass once per *surviving* entry inside the existing + `cleanup_table()` epoch-advance pass instead (`resolveKlassId()`, the same + `GetObjectClass` + `Class.getName()` + `Profiler::lookupClass()` sequence + `flush_table()` uses) — cost scales with live-table size × GC frequency, not with + allocation rate, so it does not touch the hot sampling path. This whole step is gated on + `_gc_generations` (`cleanup_table()`), so a plain liveness session pays none of it. + - A fixed-capacity table (`KlassPopulationEntry _klass_population[]`, + `MAX_KLASS_POPULATION_ENTRIES = 256`) keyed by klass `StringDictionary` id holds a + `KLASS_POPULATION_RING_SIZE = 30`-slot ring of per-epoch population counts plus one + `jweak` of a currently-live representative instance (minted fresh in + `foldKlassCountsLocked()`, deliberately *not* aliasing the source `TrackingEntry::ref`, + which `cleanup_table()` can reap out from under it). When full, the + least-recently-updated entry is evicted (`recordKlassPopulationSampleLocked()`), the + same fixed-capacity/LRU shape every other table in this design uses. Counts are + accumulated into a reused scratch array (`accumulateKlassCount()`) during the survivor + loop, then folded into the ring at the end of the pass. + - Trend/slope: computed by the shared `ringThirdsStats()` helper (`livenessTracker.cpp`) — + average (and minimum) of the earliest third of the filled window vs. the most recent third + (cheap, allocation-free, avoids full least-squares regression). Trend is trusted only once + the ring reaches a minimum fill (`KLASS_POPULATION_MIN_FILL_FOR_TREND = 10` samples) to + avoid noise right after a klass starts being tracked. A klass only qualifies as a leak + candidate once its growth (`hasQualifyingGrowth()`, which also checks a floor-rise bar to + reject oscillations whose peak passes a magnitude test but whose baseline never rises) has + held for `consecutive_positive` consecutive epochs at or above a hysteresis threshold — + `LEAK_TREND_HYSTERESIS_BASE = 5`, lowered to `LEAK_TREND_HYSTERESIS_CORROBORATED = 3` when + the aggregate post-GC heap floor is itself rising (`heapFloorRising()`, fed by a + lock-free, single-writer `_heap_floor_ring`). This closes the false-positive gap a + single-epoch positive-slope test had: a see-saw/oscillating population could trigger a + search without any real longer-term growth. The exact window size (up to 30) and the + "top 3–5" cutoff are starting points, not measured values — a separate tuning pass, not + folded into Open Question 2's pause-time work (this is a leak-detection sensitivity + tradeoff, not a safepoint-cost tradeoff). See + [reference-chains-collection-summary.md](../reference-chains-collection-summary.md) for + the full mechanism. + - `LivenessTracker::selectLeakCandidates(KlassCandidate *out, int max)` returns, on + demand under a shared lock, the positive-slope klasses ranked by magnitude descending, + capped at `min(max, MAX_LEAK_CANDIDATES = 5)` — this top-N cutoff *is* the per-pass + seeding cap, so no separate budget constant is needed. Each `KlassCandidate` carries the + klass id and its representative `jweak`. + - Known limitation, stated rather than solved: if population trends positive across + *many* klasses simultaneously, that's more likely heap-wide growth (warm-up, load + increase) than several independent leaks. The top-3–5 ranking limits how many candidates + get chased, but does not distinguish this case from true multi-leak; a future refinement. + - The leak *judgment* is retrospective (needs a full window of GC-epoch history to see the + trend) but the *reconstruction target* does not need to be the exact instance that built + up the trend — any currently-live tracked instance of the flagged klass is evidence of + the same leak. **The bridging step is a READ, not a `SetTag` write** (a correction to + this doc's original proposal, found while grounding + `LiveHeapReferenceChains-RemainingWorkPlan.md` (kept locally, not committed); + see its "Correction to the design doc's Open Question 3 mechanism"). Pre-`SetTag`ing a + candidate before the forward walk reached it would make `heapReferenceCallback()`'s + `*tag_ptr == 0` branch — the *only* branch that records `parent_tag`/`depth` — skip it, + yielding an empty/root chain. Instead `ReferenceChainTracker::pollWatchedTargets()` + resolves each candidate's `jweak`, and if `runPass()`'s whole-graph walk has already + tagged it (`getTag() > 0`, a pure read), calls `buildChainEvent(tag, ...)` and emits the + chain. A candidate still at tag 0 is retried on a later poll, since the whole-graph walk + eventually visits every root-reachable object (barring the hop/budget/frontier caps). No + new backward-walk primitive is needed; this reuses what the frontier table already + records. + - This couples two independently-flagged, independently-scheduled subsystems + (`referencechains=...` vs. `_record_liveness`/`_gc_generations`) that had no existing + relationship — `LivenessTracker` identifies objects via `jweak` and never calls JVMTI + `SetTag`/`GetTag`, while `ReferenceChainTracker` identifies objects purely via JVMTI tags + it assigns during its own traversal. `pollWatchedTargets()` bridging them is the one new + piece of machinery this adds; everything else reuses existing structures. + - **Resolved:** `referencechains=...` gets this target-seeding behavior only when liveness + tracking *and* `_gc_generations` are both enabled (`LivenessTracker::gcGenerationsEnabled()`, + checked in `pollWatchedTargets()`); otherwise it falls back to the whole-graph-only + behavior (no target seeding), which is the doc's originally-stated fallback. +4. ~~Decide whether the frontier should use JVMTI object tags at all, or adopt + `LivenessTracker`'s weak-ref + index-table pattern instead.~~ **Resolved** (see + "Frontier metadata storage"): tags stay, because `FollowReferences`/ + `IterateThroughHeap` need tagged objects to filter/report frontier membership and a + weak-ref table has no hook into that machinery — but the per-tag *metadata* + (`parent_tag`, `referrer_klass`, `depth`) should be stored in a + `LivenessTracker`-style slot table (tag value as index) rather than a bespoke + structure, reusing its proven `SpinLock`/CAS-allocation/resize code. Remaining open + item: the table-sizing formula, folded into Open Question 2. +5. Decide the actual pass-scheduling policy now that GC callbacks can only be a signal, + not a vehicle: e.g. a background agent thread woken by the GC-callback signal that + then calls `FollowReferences`/`IterateThroughHeap` for the next pass (paying its own + transparent safepoint), vs. a fixed-cadence timer independent of GC activity. Needs a + cost model for how many such safepoints per second are acceptable before this stops + being "more palatable" than one larger pause. + **A decision shipped, but not the cost-modeled one this question asks for.** + `ReferenceChainTracker::shouldRunPass()` (referenceChains.cpp) combines both candidates + rather than choosing between them: it triggers a pass when the GC-finish epoch has + advanced since the last pass, *or* a fixed `PASS_CADENCE_NS` (1 second, explicitly + labeled provisional in `referenceChains.h`) has elapsed, whichever comes first. No + safepoints-per-second/per-pause-duration measurement backs the 1-second cadence value - + it was chosen only so an idle search still makes progress without polling tightly. The + cost model this question actually asks for is still open, deferred to Phase 5's + benchmark plan (`LiveHeapReferenceChains-BenchmarkPlan.md` (kept locally, not committed)), + which has not been run. + + **SHIPPED — folded into Open Question 2's pause-time-SLO feedback loop, not solved + separately.** Implemented in the same `ReferenceChainTracker::updatePacing()` + (`referenceChains.cpp`, Phase D) described under Open Question 2: one `PidController` + `compute()` call per pass drives both that question's budget adjustment and this question's + cadence adjustment from the single measured per-pass safepoint duration, rather than two + independently-tuned mechanisms. `shouldRunPass()` and `threadLoop()` now compare against + `_effective_cadence_ns` in place of the fixed `PASS_CADENCE_NS` constant (which survives + only as `_effective_cadence_ns`'s starting value in `start()` and as the unit + `MAX_EFFECTIVE_CADENCE_NS` scales from); `threadLoop()`'s own sleep between iterations uses + `_effective_cadence_ns` too; so a controller-driven relaxed cadence actually shortens how + long an idle, no-GC-event search waits between passes, not just what the comparison in + `shouldRunPass()` reads. The GC-finish-epoch trigger in `shouldRunPass()` remains unconditional + on cadence, exactly as before — cadence only governs the fixed-interval fallback for an idle + search, per this question's original framing. See Open Question 2 above for the concrete + clamp/overflow mechanics (`MIN_EFFECTIVE_CADENCE_NS = 10ms`, `MAX_EFFECTIVE_CADENCE_NS = 4s`, + `CADENCE_NS_PER_EDGE_OVERFLOW`) and the gain-tuning caveat, which applies identically here + since it is the same controller instance. The cost-modeled "how many safepoints/sec is + acceptable" question this Open Question originally asked for is answered structurally (the + controller widens cadence exactly when passes are running long relative to the configured + ceiling) rather than by a specific measured number — that number is still a + `LiveHeapReferenceChains-BenchmarkPlan.md` (kept locally, not committed) item. diff --git a/doc/architecture/ReferenceChains-SignalsExplained.md b/doc/architecture/ReferenceChains-SignalsExplained.md new file mode 100644 index 0000000000..d45657e469 --- /dev/null +++ b/doc/architecture/ReferenceChains-SignalsExplained.md @@ -0,0 +1,366 @@ +# How reference-chain hunting decides when to run and when to back off + +*A guided tour of the signals in `ReferenceChainTracker` (`referenceChains.cpp`/`.h`), for +readers with no prior context on this subsystem.* + +## 1. What problem this is solving + +Java heaps leak. When they do, the useful question isn't "how big is the heap" — it's +*"what is holding onto these objects and refusing to let go?"* Answering that means +walking live references backwards from a suspect object to a GC root, i.e. reconstructing +a **reference chain**. + +The obstacle: walking the heap graph (JVMTI's `FollowReferences`/`IterateThroughHeap`) +requires the JVM to stop every thread at a safepoint — a Stop-The-World (STW) pause, the +same kind a GC pause is. A profiler that stops the world to investigate a leak is +trading one problem for another. So the whole design of this subsystem is really an +answer to one question: + +> **How do we get a useful heap walk without stopping the world for longer, or more +> often, than the leak justifies?** + +Everything below is the machinery that answers that question — split into two halves: +*when should the next slice of walking happen* (triggering), and *how big/frequent +should that slice be allowed to get* (pacing/pausing). + +## 2. The core trick: one long pause becomes many short ones + +A single "walk the whole reachable heap and find the chain" pass could take seconds on a +large heap — an unacceptable pause. Instead, this subsystem does **bounded, resumable +BFS**: each *pass* only visits a limited number of objects (a *budget*), remembers where +it left off using JVMTI object tags, and picks the walk back up on the next pass. So a +"search" for a leak's chain is really a sequence of many short passes, each its own small +safepoint, spread out over time. + +This reframes the whole design problem from "avoid the pause" (impossible — see §3) to +"decide, pass by pass, whether *now* is a good time to spend one of these small pauses, +and how big it should be." + +## 3. Why GC callbacks can only ever be a *signal*, never the *work* + +The natural instinct: "GC just ran, the heap just changed — hook the GC callback and do +the walk right there, for free, since the VM is already stopped." + +This doesn't work, and the reason is worth understanding because it shapes the rest of +the design. JVMTI's spec is explicit: inside `GarbageCollectionStart`/ +`GarbageCollectionFinish`, an agent may call only the **Memory Management** category +(`Allocate`/`Deallocate`). Everything a heap walk needs — `SetTag`, `GetTag`, +`GetObjectsWithTags`, `FollowReferences`, `IterateThroughHeap` — is in the **Heap** +category, which is explicitly *not* on that allow-list. Calling any of them from inside +the GC callback is exactly what the restriction forbids (confirmed against +`GCCallbackGuard` in `referenceChains.cpp`, which asserts this in debug builds). + +So the GC callback can do exactly one cheap, legal thing: bump an atomic counter. + +```cpp +void ReferenceChainTracker::onGCFinish() { + GCCallbackGuard guard; // "we are inside the forbidden window" + atomicIncRelaxed(_gc_finish_epoch, (u64)1); +} +``` + +That's it. No walk, no tag calls, nothing heap-related — just "a GC finished, epoch N". +This is the first, and most important, idea to internalize: + +> **A GC callback is a doorbell, not a worker.** It tells a separate thread "something +> happened, go check if it's worth acting on" — it never does the acting itself. + +The actual walk happens later, on a dedicated background thread, deliberately outside +the safepoint the GC callback fired inside. That thread calling `FollowReferences` +triggers its *own*, independent safepoint — the GC's pause and the walk's pause are two +separate STW events, not one shared one. (An earlier version of this design hoped to +"ride" the GC's own pause for free; that turned out to be architecturally impossible +without reaching into unversioned HotSpot internals — see +`doc/architecture/LiveHeapReferenceChains.md`'s Triggering section for the full +investigation.) + +## 4. The scheduling loop: cadence + epoch, not "run on every GC" + +The background thread (`threadLoop()`) wakes roughly once a second and asks +`shouldRunPass()`: "should I spend a pass right now?" Two independent signals feed that +decision: + +1. **The GC-finish epoch changed** since the last pass. A GC just happened — the heap + graph likely moved, so a fresh pass is probably worth its cost. +2. **A fixed cadence has elapsed** since the last pass, even with no new GC. Two distinct + roles, easy to conflate: + - For a **workload with no/rare GCs** it is the fallback the naive reader expects — + without it, the search would stall forever waiting for a signal that never comes. + - For an **in-progress search** it is the *crawl's pacing knob*, not a signal re-check: + passes between GCs are the only thing that drains the search's own frontier backlog + (its "found but not yet expanded" objects). GC time does not advance the crawl — + pass time does — so a RUNNING search legitimately runs cadence-driven passes with + zero new signal, and the pause-time controller (§7) widens/narrows exactly this + cadence. The only genuinely stale re-check is the terminal state's cheap + restart-gate re-evaluation (§5), two atomic loads, deliberately kept unconditional. + +```cpp +u64 gc_finish_epoch = gcFinishEpoch(); +if (gc_finish_epoch != _last_pass_gc_finish_epoch) { + return true; // "a GC just happened, a pass may be worth running soon" +} +... +return cadence_elapsed; +``` + +Notice what's deliberately *not* here: the loop does **not** wake up early on every GC. +`onGCFinish()` only bumps a counter — it never calls `pthread_kill()` to interrupt the +sleeping thread (the only early-wake signal is shutdown's abort, §10). Why swallow the +up-to-~1s latency instead of reacting instantly? Because under a GC-heavy workload, +waking on *every* GC would collapse the loop's cadence down to GC frequency — each wake +is a full iteration of scheduling logic, not free. The design accepts "at most ~1s of +extra latency" in exchange for not turning a GC storm into a scheduling storm. This is +a recurring theme in this subsystem: **every signal is deliberately made cheap to check +and expensive to over-react to.** + +One caveat the epoch trigger carries: it is an *unconditional* bypass — while it is +set, no cadence check applies. Minor young GCs bump the epoch as readily as majors, so +once the adaptive cadence (§7) has shrunk to its floor, a workload that GCs more often +than the wake interval effectively drives one pass per wake, at up to GC frequency. +That is bounded and acceptable for the crawl lanes (each pass is pause-budgeted by the +PID controller); the one lane that needs a rate bound of its own — the canary chase — +gets one explicitly (§8). + +## 5. Not every wake actually walks: the leak-signal gate + +Waking up and checking cheap counters is fine to do often. Actually walking the heap +costs real STW time, so before a **brand-new search** (or a **restart** of one that +finished) is allowed to begin, a second, independent gate applies: +`LivenessTracker`'s population-trend signal — "is there currently a class whose live +object count looks like it's growing without bound?" (`hasLeakSignal()`). + +```cpp +bool ReferenceChainTracker::hasLeakSignal() { + if (!LivenessTracker::instance()->gcGenerationsEnabled()) { + return true; // no trend signal available at all -> don't gate on it + } + ... + int n = LivenessTracker::instance()->selectLeakCandidates(probe, 1); + return n > 0; +} +``` + +This matters because a full-heap BFS is expensive relative to a targeted one, and there's +no point paying that cost speculatively, with nothing to justify it. If the leak-tracking +feature isn't even enabled, this reduces to "always true" — the walk runs unconditionally, +exactly as a simpler standalone version of this feature would. + +Once a *search is already running*, though, this leak-signal gate is deliberately **not** +re-applied per pass — an in-progress search's own frontier (its list of "found but not yet +expanded" objects) is allowed to keep converging pass after pass regardless of whether a +fresh leak candidate happens to be visible right now. Gating an already-running search on +"is there a candidate this instant" would stall its progress for no good reason; the gate's +job is to decide whether *starting* new expensive work is worth it, not to second-guess +work already committed to. + +The same "a potential leak is currently detected" signal also raises *liveness tracking +fidelity* itself: once candidate selection has fired, the qualifying tids' allocations are +admitted to the tracking table at 100% instead of the configured live-samples ratio +(default 10% — a 90% probabilistic drop that thins small per-(klass, tid) populations; +`LivenessTracker::admitForTracking()`). The raise is bounded by the candidate threads' +own allocation rate, not the process's whole allocation rate, so the tracking table's +cost scales with the leak's own threads; and it is refreshed — and cleared — poll by poll +with the candidate selection, so it never outlives the chase. The OOM urgency ramp +(§9) raises admission to 100% for *all* allocations for the same reason it drops every +pacing rule: the process is expected to die soon, and maximizing what the last chapter +captures outweighs the tracking table's transient volume. + +## 6. Paying for it: the "pain budget" leaky bucket + +Even with a real leak signal, restarting a full search back-to-back forever would be +its own kind of runaway cost. The safety valve here is a **leaky bucket over cost**, not +over time — `PainBudget` (`painBudget.h`): + +```cpp +// spend(): "that last search cost N milliseconds of wall-clock work" +// canStartNow(): "has that debt drained back to ~0 yet, at refill_rate?" +``` + +The intuition: if a search finished *cheaply*, it can restart again almost immediately. +If it was *expensive*, the next restart has to wait proportionally longer. This is a much +better model than a fixed cooldown timer, because "how expensive was the last search" +is exactly the thing worth reacting to — a fixed cooldown would either be too +conservative after a cheap search or too permissive after an expensive one. + +`canAffordNewSearch()` combines both gates from §5–6: the leak signal has to say "worth +it" *and* the pain budget has to say "affordable" before a new/restarted search is +allowed to begin. + +## 7. Self-tuning the pause itself: a PID controller on pause time + +So far: *when* to run a pass. Now: *how expensive should that pass be allowed to get?* + +Each pass has a target STW duration (`_pause_target_ms` — an operator-configured ceiling, +e.g. "no single pass should take more than N ms"). After every pass, the tracker measures +how long it actually took and feeds that into a small PID controller +(`_pause_pid.compute()`), which nudges the *per-pass budget* — how many objects the next +pass is allowed to visit — up or down: + +- Pass ran comfortably under target → budget can grow a bit (there's headroom). +- Pass ran over target → budget shrinks, so the *next* pass is smaller and faster. + +This is the same self-correcting idea used elsewhere in this profiler for sampling rates +(`ObjectSampler`, `MallocTracer`) — measure the actual cost, compare to a target, adjust +the knob that controls the next iteration's cost, repeat. It means the operator doesn't +have to hand-tune a "safe" fixed budget for every heap size and object graph shape; the +controller finds it empirically, pass by pass. + +There's a second knob the same controller feeds: if the budget is already at its floor +and *still* over target, that's a sign the real problem isn't "how much work per pass" — +it's "passes are happening too close together." In that case the controller widens the +*cadence* (the sleep interval between passes) instead. Conversely, if a pass finishes +comfortably under target even at the configured budget ceiling, the idle cadence is +shortened — there's slack to use it to converge faster. Either way, the same measured +signal (pass duration vs. target) drives both "how much work per pass" and "how often to +even try." + +There's also a small "savings account" on top of this (budget-borrowing): a *sustained* +run of comfortably-under-target passes slowly raises the ceiling itself, not just the +budget inside it — but a single pass that isn't comfortably under target revokes that +extra headroom immediately. The asymmetry is deliberate: earning slack should take +sustained good behavior; losing it should be instant, so a run of easy passes can never +turn into an excuse for one expensive one. + +## 8. When "wait for the next cadence tick" isn't good enough: canary mode + +The scheduling described in §4–7 assumes a slow, whole-heap background search. But +sometimes there's a much sharper signal available: `LivenessTracker` has already flagged +*specific* suspect objects (a small, pre-tagged "canary" set) worth confirming quickly. In +that mode: + +```cpp +bool canary_active = _candidate_count > 0 && + __builtin_popcountll(_candidate_found_bits) < (u64)_candidate_count; +if (canary_active) { + return true; // run the next pass immediately, no cadence wait +} +``` + +While a canary search is active, cadence is bypassed — passes run back-to-back *while the +chase is fresh or making candidate progress*, because the PID controller (§7) is already +keeping each individual pass's pause small, and a chase that is genuinely close to its +target should resolve in a handful of passes. + +Back-to-back-forever, however, is exactly the wrong promise for the failure mode this +feature actually meets in production: a candidate the crawl cannot reach soon (a leak +holder buried behind a deep frontier backlog — the coverage lottery inherent to an +external JVMTI agent with no reverse-edge primitive). Measured live on a production-like +analyzer pod: one such un-findable candidate held the chase open for 32 minutes at ~88 +passes/min — a full core of engine work — because the bypass applied unconditionally and +the loop skips its sleep whenever a pass will run. "Well under 20ms per 60s recording" +is only ever true for *findable* candidates. + +So the canary lane carries its own rate bound — a **progress-driven, work-scaled +exponential backoff**. The inter-pass spacing is a *multiple of the measured cost of a +pass itself* (an EMA of each pass's wall duration), not a fixed wall-clock constant: +- Every pass that makes candidate progress (a candidate found, or a new candidate + admitted) resets the spacing multiplier to 1 — back-to-back. At multiplier 1 the + next pass starts as soon as the last one ended, which is harmless by construction + for a cheap pass, and exactly the fast-resolution burst a chase that is genuinely + close should get. +- Every pass with *no* candidate progress doubles the multiplier, capped at 16. A + permanently stuck chase therefore settles at one pass per 16 × (its own cost) — + it keeps ticking indefinitely (abandonment is a separate, frontier-aware detector's + job, below), but its steady burn is structurally bounded to ~1/16 of a core on pass + work, *whatever that work is*. +- The work-scaling is the point: a fixed cap only binds when it exceeds the pass's own + duration — the production pod's passes ran 0.7–4s, so a 1s cap would have changed + nothing at all (the loop is work-bound, never sleep-bound, when the pass exceeds the + cap), while the same 1s cap starved a genuinely reachable deep chase whose passes + cost milliseconds. Scaling to the measured cost gives the expensive-pod chase a real + bound and the cheap-but-deep chase its density from the same law. +- The GC-epoch trigger deliberately does **not** bypass the backoff: a GC-heavy workload + bumps the epoch on virtually every wake, so letting GCs override the spacing would make + the backoff unreachable on exactly the deployments that burn the most. +- The OOM urgency ramp (§9) overrides it entirely — imminent OOM remains the one regime + where the chase burns budget back-to-back. + +The pain-budget refill rate is raised 100x while a chase is open, but note what that +means now: **not** a rate control (the backoff is the rate bound) — just a double-throttle +guard, so the conservative base refill rate tuned for the ordinary ~1 pass/s crawl doesn't +starve a chase the backoff has already paced. The earlier covering-vs-emergency refill +distinction existed to feed the unbounded back-to-back mode and is gone with it. + +## 9. The panic button: ramping up as OOM approaches + +All of the above optimizes for "acceptable background cost most of the time." But if +`LivenessTracker` projects the heap is genuinely on a collision course with +`OutOfMemoryError` within the next ~30 minutes (`secondsToOOM()`), "acceptable background +cost" is the wrong objective — the process might not survive long enough for a leisurely +search to finish. So there's a third mode: an **urgency ramp**. + +As projected time-to-OOM shrinks from 30 minutes toward zero, the pause-time target and +the pass cadence both ramp *exponentially* toward much more aggressive ceilings: + +```cpp +double x = 1.0 - seconds_to_oom / OOM_RAMP_START_S; // 0 at 30min out, 1 at OOM +target_ms = pause_target * pow(URGENT_PAUSE_TARGET_MS / pause_target, x); +cadence_ns = pow(URGENT_CADENCE_NS / PASS_CADENCE_NS, x) * PASS_CADENCE_NS; +``` + +The reasoning behind exponential (rather than linear) ramping: at 30 minutes out, the +situation still might resolve itself (a GC frees the suspect objects, the trend reverses) +— stay cheap. In the last seconds before OOM, the process is likely to die anyway, so it's +worth spending far more of the pause-time budget to collect a usable chain *before that +happens* than to protect a latency budget for a process that may not be there to benefit +from it. Held flat-out cheap the whole time, this urgency signal would arrive too late to +matter; held aggressive the whole time, it would waste budget on every one of the many +false alarms a rising trend that later reverses produces. + +Two more details make this practical rather than flappy: + +- **Hysteresis, not a bare threshold.** `isUrgent()` *latches* on when + time-to-OOM drops below a threshold, and only *releases* after several consecutive + observations comfortably clear of a separate (higher) release bar. A single noisy + reading crossing back and forth across one threshold would otherwise thrash the ramp + on and off every second. +- **One search per urgency episode.** Once an urgent episode has spent its one + authorized search, further ticks within the same episode don't keep tearing down and + restarting it from scratch — the per-candidate probe (§5) remains the only trigger + until the episode actually clears. + +## 10. Stopping mid-pass: the abort path + +Everything above is about *starting* passes thoughtfully. There's also a clean way to +*stop* one that's already in flight — needed when the profiler itself is shutting down +(or a test needs to reset state) while a `FollowReferences` call is still blocked inside +the JVM. + +```cpp +_abort_pass_requested.store(true, std::memory_order_relaxed); +pthread_kill(_thread, WAKEUP_SIGNAL); // interrupts a sleeping thread promptly +``` + +The callback JVMTI invokes for each visited object checks this flag and returns an abort +code the moment it sees it set — since nothing outside the JVM can interrupt a call +already inside `FollowReferences`, the flag has to be checked *from inside* the callback +JVMTI itself is driving. `pthread_kill` with a no-op-handler signal only helps the *other* +common case — a thread parked in `OS::sleep()` between passes — wake up promptly instead +of waiting out the rest of its interval. + +## 11. Putting it together + +| Question | Signal | Where | +|---|---|---| +| Did the heap graph just change? | GC-finish epoch bump | `onGCFinish()` → `shouldRunPass()` | +| No GC signal — is it time anyway? | Fixed/adaptive cadence elapsed | `shouldRunPass()` | +| Is a *new* search worth starting at all? | LivenessTracker population trend | `hasLeakSignal()` | +| Can we afford to spend that cost right now? | Pain-budget leaky bucket | `canAffordNewSearch()` | +| How big/frequent should passes be, steady-state? | PID controller on measured pause time | `updatePacing()` | +| Are we chasing specific known suspects? | Canary candidate set + progress-driven backoff | `canary_active` bypass + work-scaled `_canary_backoff_mult` | +| Is OOM close enough to abandon caution? | secondsToOOM() latch/release | urgency ramp in `threadLoop()` | +| Need to stop a pass already in flight? | Abort flag + wakeup signal | `_abort_pass_requested` | + +The unifying idea across all eight mechanisms: **every trigger is a cheap check, and +every response is proportional to real, measured cost** — never a fixed guess. GC +callbacks stay legal by doing nothing but incrementing a counter. Whether to search at +all is gated on an independent leak-trend signal, not "because a GC happened." Whether a +search can *restart* is gated on how expensive it actually was last time, not a flat +cooldown. How big a pass gets is tuned from its own measured pause time, not a static +config value. And the one scenario where none of that caution applies — imminent OOM — is +its own explicitly separate, hysteretic escalation path, not a tweak to the steady-state +knobs. + +That's what makes several short, adaptively-sized pauses a genuinely better trade than +one long one: the *decision* of when to pay each of those small costs is never blind — +it's always backed by a signal that says this particular pause is likely to be worth it. diff --git a/doc/reference-chains-collection-summary.md b/doc/reference-chains-collection-summary.md new file mode 100644 index 0000000000..645c9c8285 --- /dev/null +++ b/doc/reference-chains-collection-summary.md @@ -0,0 +1,87 @@ +# Reference Chain Collection: Design Summary + +## Problem + +Given a JVM heap with objects suspected of leaking (e.g., klasses whose live population grows monotonically across GC generations), reconstruct a **referrer chain** from a GC root down to a representative instance of the suspect klass — without pausing the JVM for longer than a small, bounded budget, and without assuming the entire heap graph can be walked in one pass. + +Three constraints drive the design: + +1. **Detecting *which* klasses are worth walking** must be near-free and based on survivorship trend, not raw allocation volume. +2. **The walk itself** (JVMTI `FollowReferences`) can be arbitrarily expensive on a large heap, so it must be interruptible and resumable. +3. **Total STW/JVMTI-callback time per pass** must stay under a small budget so the profiler doesn't visibly perturb the target application. + +--- + +## Component 1: Surviving-Generation Signal (`LivenessTracker`) + +Rather than triggering a heap walk on every allocation or every GC, the tracker maintains a **per-klass population history** and only nominates a klass as a "leak candidate" once it shows a **sustained positive trend across GC generations** — i.e., its live (surviving) instance count keeps growing generation over generation, not just spiking transiently. + +**Mechanics:** + +- Population sampling is driven off the existing allocation-sampling hot path (`track()`), but the actual **per-klass counts are only folded into history at `cleanup_table()`'s GC-epoch-advance pass** — i.e., once per GC, not once per allocation. This keeps the hot path allocation-free and cheap. +- Each klass gets a small **ring buffer of recent per-epoch surviving counts** (`KLASS_POPULATION_RING_SIZE = 30` samples). A ring, not an unbounded history, because we only care about recent trend, not lifetime totals. +- A klass's trend is only trusted once its ring has a **minimum fill (`KLASS_POPULATION_MIN_FILL_FOR_TREND = 10` samples)** — avoids false-positive trend detection on a klass that's simply new to being tracked (too few points to fit a slope to). +- `selectLeakCandidates()` computes a slope over each ring and returns the **top-N klasses by slope magnitude** (`MAX_LEAK_CANDIDATES = 5`), each paired with a live representative instance (a `jweak`) discovered during sampling — this weak reference is what seeds the walk in Component 2. +- A klass only qualifies once its growth (`hasQualifyingGrowth()`) has held for `consecutive_positive` epochs at or above a **hysteresis threshold** — `LEAK_TREND_HYSTERESIS_BASE = 5` by default, lowered to `LEAK_TREND_HYSTERESIS_CORROBORATED = 3` when the aggregate post-GC heap floor is itself rising (`heapFloorRising()`, fed by a lock-free, single-writer `_heap_floor_ring` populated from `onGC()`). Because the aggregate heap-floor signal can't attribute growth to any one klass, it only ever raises or lowers the bar uniformly for the whole scan — it never reorders or singles out individual candidates. +- The whole table (`_klass_population`, up to `MAX_KLASS_POPULATION_ENTRIES = 256` entries) is a flat array scanned linearly — deliberately no index structure, since 256 entries is cheap to scan and this stays off the allocation hot path. +- Everything under this table (population array, size counter) is guarded by a single `SpinLock` (`_table_lock`) — the *same* lock `cleanup_table()` already holds for its epoch-advance pass, rather than adding a second lock. **Any code path that mutates this table (including test-only reset seams) must take that lock — mutating `_klass_population_size` or the array unguarded is a data race against the epoch-advance pass**, discovered in practice while hardening test seams. + +**Why this design:** it decouples "is this klass suspicious" (cheap, GC-cadence, statistical) from "reconstruct why it's suspicious" (expensive, JVMTI, on-demand) — the expensive walk only ever runs against klasses that have already earned a positive trend signal, not against every allocation site. + +--- + +## Component 2: Resumable Frontier Walk (`ReferenceChainTracker`) + +Once a klass is nominated, a **persistent background BFS thread** reconstructs a path from a GC root to a tagged instance of that klass, using JVMTI's `FollowReferences`/heap-tag mechanism — but broken into many small, budgeted passes rather than one unbounded walk. + +**Mechanics:** + +- The tracker is a **process-wide singleton** with its own thread (`threadLoop()`), woken on a fixed cadence (`effectiveCadenceNs`) rather than synchronously from allocation or GC callbacks — decouples walk progress from the rate of GC/allocation events. +- `runPass()` dispatches on `_search_started`: + - **First pass for a search**: enumerates heap roots via `IterateOverReachableObjects()` (`heapRootCallback()`/`stackRefCallback()`), tagging root-referenced objects as it goes. + - **Every subsequent pass**: calls `expandFrontier()`, which resumes from a **persisted frontier** (the previous pass's boundary tags) instead of re-walking from roots. This is the resumability mechanism: each pass advances the frontier outward by one bounded increment and stops. +- Each pass is capped by an **edge-admission budget** (`effectiveBudget`, e.g. `edges_admitted` capped at a configured value like 4000/200000/500 depending on test config) — `expandFrontier()`'s nested loops (`while (!ctx.truncated && progress)` outer, `for (jlong tag : candidate_tags)` inner) both check a truncation flag and bail out the moment the budget is exhausted, so a single pass's JVMTI-callback time is bounded regardless of heap size. +- **Cooperative abort**: an `std::atomic _abort_pass_requested` flag, checked inside `heapReferenceCallback()` (the JVMTI callback invoked per edge), lets `stopThread()` interrupt an **in-flight** walk promptly — set before `pthread_kill(WAKEUP_SIGNAL)`/`pthread_join()`, cleared by `startThread()`. Without this, a `FollowReferences` call already in progress at JVM shutdown or profiler restart can't be interrupted, and `pthread_join()` blocks indefinitely (a real, previously-diagnosed shutdown hang). +- Search state is a small state machine: `RUNNING → {ABANDONED | COMPLETED}`. `RUNNING` can **restart itself** (fresh root walk) once a candidate's chain is found and its tags released, gated by `canAffordNewSearch()`'s **pacing budget** — self-throttling, not unconditional: a search won't restart back-to-back if it would blow the perturbation budget. Once a search reaches a terminal state (`ABANDONED`/`COMPLETED`) it stays there — restarts only happen from within `RUNNING`. +- `runPass()` only moves to `COMPLETED` once the frontier is fully drained **and** `_watched_leak_klass_count == 0` (no klass currently under active leak watch, Component 4). A fully-drained frontier while a klass is still watched leaves `_search_state` at `RUNNING` instead: the walk has visited every reachable object once, but a leak-shaped klass keeps growing by **mutating an already-visited container** (e.g. appending to a `static final` collection field long after the walk first admitted it), which a one-time visit can never observe again. Rotation (Component 4) is what re-observes those already-`EXPANDED` entries on later passes. +- A search is marked `ABANDONED` (with a reason code) if it runs out of frontier budget without completing — e.g. hitting a frontier-cap under a tiny configured budget. This is a deliberate, observable outcome, not a silent failure — surfaced so operators can distinguish "the walk gave up" from "the walk is still in progress." + +**Why this design:** treating the walk as a resumable state machine (persisted frontier + tags) rather than one atomic call means a heap graph of unbounded size never forces an unbounded pause — cost is amortized across many cheap passes, each individually bounded and individually abortable. + +--- + +## Component 3: Latency Budget Enforcement + +The system enforces its "don't perturb the app" guarantee at **three independent layers**, not just one: + +1. **Per-pass edge budget** (`effectiveBudget`) — caps JVMTI callback invocations per pass (Component 2). +2. **Pain budget** (`_pain_budget`/`_search_pain_ms`, spent via `_pain_budget.spend(...)`) — tracks cumulative walk cost against a wall-clock ceiling; used by `canAffordNewSearch()` to decide whether a new search/restart is affordable right now, not just whether the current pass fit its edge budget. This is what prevents "many cheap passes" from silently adding up to an expensive aggregate cost. +3. **Pass cadence** (`effectiveCadenceNs`) — the background thread only wakes and attempts a pass on a fixed cadence (plus GC-epoch-triggered wakeups), rather than continuously spinning, bounding CPU overhead between passes. + +Together these mean: a single pass is bounded (edge budget), a sequence of passes is bounded (pain budget), and idle overhead between passes is bounded (cadence) — the three layers target three different ways an unbounded-cost walk could otherwise leak into the target application's latency. + +--- + +## Component 4: Rediscovering Growth in an Already-Visited Container + +Once a klass is leak-flagged, `pollWatchedTargets()` refreshes `_watched_leak_klass_ids` (up to `MAX_WATCHED_LEAK_KLASSES = 5`, matching `LivenessTracker::MAX_LEAK_CANDIDATES`) from `LivenessTracker::topKlassesByGenerationCount()` — a faster, un-hysteresis-gated ranking than the `selectLeakCandidates()` canary set, but only consulted once `hasLeakSignal()` has already fired via that slower path. + +**Why a matching mechanism is needed at all:** the real leak shape this targets is a `static final` collection field that gets *appended to*, not reassigned — the container itself was already admitted and `EXPANDED` in an early pass, long before `selectLeakCandidates()`'s hysteresis authorized watching its element klass. New elements can only be rediscovered by re-expanding that already-visited container, not by discovering a brand-new root. + +**Mechanics:** + +- Matching a newly-admitted object against `_watched_leak_klass_ids` must use a class identity that stays valid for the object's whole lifetime, not `referrer_klass` — a classMap `StringDictionary` id that can differ for the same class at different times if that dictionary is compacted/regenerated. Both `ReferenceChainTracker` and `LivenessTracker` mint from a single shared, process-wide `ClassTagAllocator` (`classTagAllocator.h`) and store the resulting stable `class_tag` (`FrontierEntry::class_tag`, `KlassPopulationEntry::stable_class_tag`) instead. +- `trackLeakAccumulation()` runs on every successful admission (`admitObject()`'s `ADMITTED` result) and aggregates, per `(leaf_klass_id, parent_class_id)` signature, how many admitted children of a watched leaf klass were observed under a parent of that class (`_leak_signature_totals`, ranked by delta against the previous pass's snapshot — Tier 1), and per parent *object* tag, how many such children that specific parent holds (`_leak_parent_fanout` — Tier 2, ranked within the winning Tier-1 signature). +- `seedLeakAccumulationForNewlyWatchedKlass()` runs once, the moment a klass_id first enters `_watched_leak_klass_ids`: it scans the whole frontier table for already-`EXPANDED` entries whose `class_tag` matches, since a container that was fully admitted before its element klass started being watched would otherwise never get its first Tier-1/Tier-2 data point. +- `collectLeakAccumulationCandidatesForRotation()` re-queues the Tier-2 winner(s) for re-expansion (`LEAK_ACCUMULATION_ROTATION_BUDGET = 16` per pass) — this is what actually re-visits the growing container and picks up elements appended since its first expansion. + +**Two additional robustness fixes surfaced only under a real growing-collection repro, not by the unit suite alone:** + +- **Urgent-signal latch.** `isUrgent()` used to be a bare `secondsToOOM() < OOM_URGENT_THRESHOLD_S` comparison; that projection is derived from a short ring of heap deltas and can swing by orders of magnitude between consecutive observations of the same steadily-growing heap. A bare comparison flapped, and each flap back to "urgent" bypassed the per-klass hysteresis gate in `hasLeakSignal()` and restarted the search — which discards the frontier table and the Tier-1/Tier-2 accumulators above, so they never got the several passes they need to converge. `isUrgent()` now latches on first crossing and only releases after `URGENT_RELEASE_CONSECUTIVE` (5) consecutive observations at or above `OOM_URGENT_RELEASE_S` (2× the threshold); `_urgent_search_spent` limits each latched episode to authorizing one restart. +- **Classmap-generation sync at startup.** `LivenessTracker::initialize()` now seeds `_last_class_map_generation` from the real classMap generation instead of leaving it at the default `0`. Previously, the first `cleanup_table()` call after any profiler start saw `current_generation != 0`, treated it as a classMap reset, and wiped `_klass_population` — discarding any population history folded in between `initialize()` and that first `cleanup_table()` call. + +--- + +## Output Path + +Once a candidate's chain is fully reconstructed, `pollWatchedTargets()` builds a chain event (`buildChainEvent()`), which is enqueued (`enqueueChainEvent()`) and later drained (`drainPendingChainEvents()`, called from `Profiler::dump()`, not from the BFS scheduling thread) into `Profiler::writeReferenceChain()` — ultimately surfaced as a `datadog.ReferenceChain` JFR event on the next `Profiler::dump()`. This keeps the expensive walk and the (comparatively cheap, already-existing) JFR-write path decoupled — the walk never blocks on JFR I/O, and JFR writes never trigger a walk. diff --git a/doc/reference-chains-design.md b/doc/reference-chains-design.md new file mode 100644 index 0000000000..059a181189 --- /dev/null +++ b/doc/reference-chains-design.md @@ -0,0 +1,259 @@ +# Reference Chains for Surviving Live Heap Samples + +**Status:** Implemented (see `doc/reference-chains-collection-summary.md` for the as-built design) +**Date:** 2026-07-07 +**Jira:** TBD + +## Goal + +For a subset of live-heap samples that survive past their allocation window, produce +a **reference chain** — a sequence of referrer *types* (not full field-level paths, not +necessarily to *all* GC roots) connecting the sampled instance back to *a* GC root. This +is diagnostic information ("what kind of object chain is keeping this alive"), not a +heap-dump-grade exact retainer analysis. + +## Constraints + +- Must run cheaply, with as short a safepoint / STW contribution as possible. +- Must work on stock vendor JDKs the agent attaches to — no forked/patched JVM builds. +- Exhaustive (all-roots, full-path) chains are explicitly **not** required; referrer-type-only, + bounded-depth, best-effort chains are acceptable. + +## Approaches considered + +Three approaches were evaluated; two are ruled out as launch requirements for concrete, +evidence-backed reasons. One sub-idea (Approach C's `ParallelObjectIterator` variant) is +explicitly kept open as a conditional future option; see its discussion below. + +| # | Approach | Completeness | Complexity | Feasibility | Status | +|---|---|---|---|---|---| +| A | Full JVMTI `FollowReferences` reverse-graph walk, piggybacked on an already-scheduled major GC | 4/5 | 4/5 | 2/5 | Rejected | +| B | Bounded BFS-from-roots with frontier pruning (JFR "leak profiler" technique, adapted) | 3/5 | 3/5\* | 4/5 | **Chosen** | +| C | Hook GC mark/copy closures (G1, ZGC) to record parent pointers inline during marking | 2/5 | 5/5 | 1/5 | Rejected | + +\* This 3/5 reflects only the single-pass BFS sketched at selection time. The "Chosen +design" section below replaces that sketch with an incremental, resumable BFS +(JVMTI-tag-based frontier persistence across GC cycles, a dedicated `VM_Operation` per +pass, GC-callback signaling, and explicit termination/tag-cleanup bookkeeping), which is +materially more complex than this score suggests — closer to 4/5 in implementation and +maintenance effort. The score is left unchanged above (it documents the state of the +comparison at decision time) rather than retroactively edited. + +### A — Full reverse-reachability walk (rejected) + +Safepoint length scales with live-set size regardless of how the walk is triggered. +Modern regionalized collectors (G1, Shenandoah) rarely perform a true full-heap walk +during ordinary major GCs, so "ride an already-paid pause" is not a reliable amortization +strategy. Cost is fundamentally at odds with the "short safepoint" constraint. + +### B — Bounded BFS-from-roots (chosen) + +Mirrors OpenJDK's own `jdk.OldObjectSample` leak-profiler implementation +(`src/hotspot/share/jfr/leakprofiler/chains/{edgeStore,bfsClosure,dfsClosure}.cpp`): +a `VM_Operation`-driven BFS from GC roots, retaining only edges on the frontier toward a +small, fixed sample set, with a hard hop cap (HotSpot itself caps chains at ~200 hops, +split 100/100 from leaf and from root). We can go cheaper than JFR because only the +**referrer class**, not object identity or field name, is needed — the `EdgeStore` +degenerates to `(referrer_klass, parent_ref, depth)` records, where `parent_ref` links +each record back to the record that discovered it, enabling chain reconstruction. + +Adopting this pattern is a re-scoping of proven, shipping HotSpot code, not a novel +algorithm design. + +### C — GC mark/copy closure piggyback (rejected) + +Investigated specifically for G1 and ZGC on the premise that per-edge referrer +information is already available inside the collector's own marking/evacuation closures +(`G1ParCopyClosure::do_oop_work`, ZGC's `ZMarkConcurrentRootsIteratorClosure` / +load-barrier closures), so recording it would cost nothing beyond what the GC already +pays. + +Rejected because there is no stable, externally reachable hook into these closures: + +- They are internal, template-instantiated C++ classes compiled into `libjvm.so` at + HotSpot build time — not a registrable/pluggable extension point. +- This differs categorically from `VMStructs`-style introspection already used in this + codebase (`ddprof-lib/src/main/cpp/hotspot/vmStructs.cpp`), which reads VM state + passively via an officially exported offset table. Intercepting a GC closure's + *behavior* would require either shipping a patched OpenJDK build (a fork/maintenance + commitment far beyond anything in this codebase) or binary-patching unversioned, + per-build-mangled function addresses — not shippable across JDK point releases. + +A related idea — using HotSpot's internal `ParallelObjectIterator` +(landed via [JDK-8322043](https://www.mail-archive.com/serviceability-dev@openjdk.org/msg12977.html), +used by `VM_HeapDumper` to partition heap regions across GC worker threads for parallel +heap dumping) to shrink Approach B's safepoint by parallelizing the walk — was also +investigated. Same verdict: it is an internal C++ class, not exposed via JVMTI, with no +stable ABI for an attached agent to call. Symbol-sniffing internal HotSpot functions *is* +an established pattern in this codebase (`VMStructs::findHeapUsageFunc`, +`vmStructs.cpp:489-509`), but that precedent covers a single leaf virtual method with a +value/POD-ish return; `ParallelObjectIterator` is a multi-class subsystem that coordinates +the VM's own GC worker threads under safepoint control — an order of magnitude larger +fragility surface, with a much higher blast radius if a layout assumption is wrong (GC +worker-thread coordination corruption vs. a bad JMX stat). Not pursued as a launch +requirement; revisit only if Approach B's single-threaded pause proves to be a measured +bottleneck, and treat it as an isolated, heavily version/flag-gated fast path with +automatic fallback — never a dependency. + +## Chosen design: incremental, resumable bounded BFS + +A single-pass bounded BFS still means one pause sized to whatever budget is configured. +The refinement below spreads that budget across multiple short passes instead of one +contiguous one, trading a possibly-higher *aggregate* STW total for a much better +*latency distribution* — no single long tail pause. + +### Why the frontier can survive across passes: JVMTI object tags + +The obstacle to pausing and resuming a BFS is that the frontier (the worklist of +not-yet-expanded objects) is normally a set of raw addresses, and a moving/compacting GC +between passes can relocate or collect any of them. + +JVMTI object tags solve this: + +- Tags are identity-based and GC-move-transparent — a tagged object can be re-resolved + after a GC regardless of where it moved. +- Tags are **non-retaining** — tagging does not keep an object alive. This is the same + property the existing live-object sampler in this codebase already relies on, so this + is a new *use* of an existing mechanism, not new risk surface. +- Non-retention gives incremental resumption a useful side effect for free: if a frontier + object dies between passes, it simply fails to re-resolve on the next pass. That branch + of the search is pruned automatically, with no extra liveness bookkeeping required. + +### Data structures + +- **Frontier**: a set of `(tag, parent_tag, referrer_klass, depth)` records. `tag` is the + JVMTI tag assigned to a not-yet-expanded object; `parent_tag` links back for chain + reconstruction; `depth` supports the hop cap. +- **EdgeStore**: accumulates `(referrer_klass, parent_tag, depth)` per discovered edge for + objects that are on a path toward a target sample. Keyed by tag, not address — + degenerate relative to JFR's `EdgeStore` since object identity/field names are not + required, but it retains the same `parent_tag` linkage field as the Frontier so a chain + can be walked back from a target sample to a root by following `parent_tag` across + EdgeStore records. + +### Algorithm + +1. Seed the frontier from GC roots (first pass) or from the persisted frontier + (resumed pass). +2. Resolve currently-live tagged frontier objects. Objects that fail to resolve are + dropped (dead — free pruning). +3. Expand the frontier up to a fixed per-pass budget (edge count or time slice). +4. Newly discovered objects are tagged and added to the frontier for the next pass. +5. Persist the frontier (native memory owned by the agent, not thread-local scratch) and + return control to the VM. +6. Repeat until: a target sample is reached, the hop cap is hit, or a per-search + abandonment limit (see Termination) is exceeded. + +### Triggering passes: resolved — cannot avoid dedicated safepoints + +Investigated whether pass-continuation work could ride the JVMTI +`GarbageCollectionStart`/`GarbageCollectionFinish` callbacks — the same callback this +codebase already uses to call `_heap_usage_func` (`vmStructs.cpp`) — instead of +scheduling a dedicated `VM_Operation` per pass. + +**Confirmed the VM is genuinely at a safepoint (all mutators stopped) for the full +duration of both callbacks**, on every collector: + +- JVMTI spec: *"This event is sent while the VM is still stopped... the event handler + must not use JNI functions and must not use JVM TI functions except those which + specifically allow such use (see the raw monitor, memory management, and environment + local storage functions)."* +- openjdk/jdk source: delivery is synchronous on the VMThread + (`src/hotspot/share/prims/jvmtiExport.cpp:2752-2790`, comment *"this event is posted + from VM-Thread"*); every call site is inside a safepoint-executing `VM_Operation::doit()`, + backed by explicit asserts — e.g. Parallel GC's + `assert(SafepointSynchronize::is_at_safepoint())` (`gc/parallel/psScavenge.cpp:305-306`), + G1's `assert_at_safepoint_on_vm_thread()` (`gc/g1/g1VMOperations.cpp:141-157`), + Shenandoah and ZGC wrapping the same `SvcGCMarker` only inside their respective + `VM_Operation`/`VM_ZOperation::doit()` paths. Stable JDK 11 → mainline, across + Serial/Parallel/G1/Shenandoah/ZGC. + +**But this does not make the callback usable as the execution vehicle for a pass.** The +"functions which specifically allow such use" are exactly two: `Allocate` and +`Deallocate` (the entire **Memory Management** category). `SetTag`, `GetTag`, +`GetObjectsWithTags`, `FollowReferences`, and `IterateThroughHeap` are all in the +**Heap** category, which is *not* on that allowlist — calling any of them from inside +`GarbageCollectionStart`/`Finish` is exactly what the restriction forbids. The spec's own +prescribed escape hatch — notify a raw monitor from the callback, do the real work on a +separate agent thread — doesn't preserve the "rides the pause" property either: by the +time the woken agent thread runs, `VM_Operation::doit()` has already returned and the +safepoint has been released, so the tagging/walk work ends up running concurrently with +resumed mutators, not during the STW window. + +The only way to do the tag/walk work *while actually inside* the callback's STW window +would be to bypass the official JVMTI entry points and reach into HotSpot's internal +`JvmtiTagMap` directly via symbol-sniffing — reintroducing exactly the fragility class +already rejected for Approach C (unversioned internal C++ state, no stable ABI). Doing +that here would undo the reason C was rejected. + +**Conclusion: "no new marginal safepoints" is not achievable while staying within +official JVMTI usage.** Each pass needs its own dedicated, budget-capped `VM_Operation`. +The GC callbacks remain useful only as a low-cost *signal* ("a GC just happened, a pass +may be worth scheduling soon") — not as the execution vehicle for the pass itself. This +does not change the core incremental design (frontier persistence via JVMTI tags, +self-pruning of dead branches, per-pass budget) — it only removes the "zero marginal +safepoints" claim from the cost/benefit case. The design's actual value remains what it +was framed as: trading one long pause for several short, independently-scheduled ones — +a latency-distribution improvement, not a total-STW reduction. + +### Termination and abandonment + +Because passes are spread across a mutating heap, a search that never reaches a root or +the hop cap could otherwise persist indefinitely, accumulating abandoned frontier state +across GC cycles. Required cutoffs: + +- Hop cap (as in Approach B's single-pass form). +- A hard cap on passes-per-search or wall-clock TTL from first observation. +- Explicit reporting of abandoned searches (no silent truncation) so this shows up as a + measurable "chain not found within budget" outcome rather than being indistinguishable + from "no chain exists." + +### Correctness note: chains are historical, not a single consistent snapshot + +A chain built across multiple passes stitches together `"A referenced B"` facts observed +at different points in time, not one frozen graph. For the stated purpose — explaining, +by referrer type, what typically retains this class of surviving object — this is +sufficient, and is not meaningfully weaker than a single-pass walk: GC roots (e.g. thread +stack frames) are themselves a live-changing set across a single pause's boundary, so +"one true snapshot" is already an approximation in the single-pass case. Any +documentation or output surface built on this must describe results as an **observed** +retaining path, not a claim about the object's current exact retention state. + +### Cost/benefit summary + +- **Does not reduce total STW time.** Each safepoint/callback entry pays fixed + synchronization overhead; K short increments likely sum to equal or *more* aggregate + pause time than one contiguous walk covering the same work. +- **Improves latency distribution.** No single long tail pause — the thing most likely to + actually affect deployed application health (p99 latency, heartbeat timeouts), even + when total accumulated pause-ms is flat or slightly worse. + +## Non-goals + +- Exhaustive paths to all GC roots. +- Field-level or object-identity-level chains (referrer *type* only). +- Any GC-internal-closure hook (Approach C) or internal parallel-iteration API use as a + launch dependency. + +## Open questions before implementation + +1. ~~Confirm `GarbageCollectionStart`/`GarbageCollectionFinish` callback timing relative to + safepoint release.~~ **Resolved** (see Triggering section): the callback is genuinely + at a safepoint, but the JVMTI Heap-category functions needed to do frontier work + (`SetTag`/`GetTag`/`FollowReferences`/`IterateThroughHeap`) are not in the callback's + allowed function set, so each pass still needs its own dedicated `VM_Operation`. The + "no new marginal safepoints" framing is dropped; the design's value is latency + distribution, not total-STW reduction. +2. Choose per-pass budget defaults (edge count vs. time slice) and hop cap — needs + measurement against representative heap shapes, not a guess. +3. Decide the sample-batching policy: one incremental search per live-heap sample, or + batched multi-target BFS sharing a single frontier walk (batching amortizes better but + couples unrelated samples' termination conditions together). +4. Decide behavior when JVMTI tagging is already saturated by the existing live-object + sampler (tag-table sizing/contention) — this reuses infrastructure that has other + consumers in this codebase. +5. Decide the actual pass-scheduling policy now that GC callbacks can only be a signal, + not a vehicle: e.g. a background thread woken by the GC-callback signal that then + requests its own bounded `VM_Operation`, vs. a fixed-cadence timer independent of GC + activity. Needs a cost model for how many dedicated small safepoints per second are + acceptable before this stops being "more palatable" than one larger pause. From b1804fcfc92711380f5ac99daed899cc71718e0b Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Wed, 16 Sep 2026 17:02:37 +0200 Subject: [PATCH 07/18] Archive the missing-refchains-on-hotdog investigation notes Full record of the investigation that produced the reference-chain engine: STATE/INDEX memory, per-round pod evidence, root-cause analyses, and repro artifacts. --- .../missing-refchains-on-hotdog/INDEX.md | 122 ++++++ .../missing-refchains-on-hotdog/STATE.md | 375 ++++++++++++++++++ ...v-adaptive-batchsize-onpod-verification.md | 69 ++++ .../ev-candidate-count-latch-mismatch.md | 55 +++ .../ev-deadline-split-onpod-verification.md | 45 +++ .../ev-deployed-so-1481-no-symbols.md | 97 +++++ .../ev-fixes-compile-and-gtest-pass.md | 56 +++ .../evidence/ev-hotdog-trace-zero-runpass.md | 68 ++++ .../ev-hotspot-lifo-visitation-order.md | 99 +++++ .../evidence/ev-jafar-zero-refchain-events.md | 36 ++ .../ev-jfr-analysis-real-recording.md | 75 ++++ .../ev-kind-counts-constant-pool-dominates.md | 61 +++ .../ev-leaktag-correlation-local-repro.md | 74 ++++ .../evidence/ev-leaktag-onpod-round1.md | 54 +++ .../evidence/ev-leaktag-onpod-round10.md | 65 +++ .../evidence/ev-leaktag-onpod-round11.md | 87 ++++ .../evidence/ev-leaktag-onpod-round12.md | 89 +++++ .../ev-leaktag-onpod-round13-results.md | 98 +++++ .../evidence/ev-leaktag-onpod-round13.md | 81 ++++ .../ev-leaktag-onpod-round14-results.md | 89 +++++ .../evidence/ev-leaktag-onpod-round14.md | 89 +++++ .../ev-leaktag-onpod-round15-results.md | 93 +++++ .../evidence/ev-leaktag-onpod-round15.md | 88 ++++ .../ev-leaktag-onpod-round16-results.md | 76 ++++ .../evidence/ev-leaktag-onpod-round2.md | 57 +++ .../evidence/ev-leaktag-onpod-round3.md | 107 +++++ .../evidence/ev-leaktag-onpod-round4.md | 75 ++++ .../evidence/ev-leaktag-onpod-round5.md | 73 ++++ .../evidence/ev-leaktag-onpod-round6.md | 59 +++ .../evidence/ev-leaktag-onpod-round7.md | 79 ++++ .../evidence/ev-leaktag-onpod-round8.md | 82 ++++ .../evidence/ev-leaktag-onpod-round9.md | 76 ++++ .../evidence/ev-livelock-pod-logs.md | 65 +++ .../evidence/ev-marker-tag-arithmetic.md | 38 ++ .../ev-post-resync-deployment-verified.md | 64 +++ .../ev-postCB-onpod-live-verification.md | 72 ++++ .../ev-postfix-onpod-live-verification.md | 62 +++ ...ix-static-field-onpod-live-verification.md | 72 ++++ .../ev-postfixEF-onpod-live-verification.md | 104 +++++ .../evidence/ev-source-poll-vs-callback.md | 147 +++++++ .../ev-tid-clustering-onpod-verification.md | 55 +++ .../ev-tid-clustering-per-thread-working.md | 63 +++ .../ev-timing-split-callback-vs-jvmti.md | 65 +++ .../ev-toolkit-and-onpod-methodology.md | 65 +++ .../ev-uploaded-jfr-no-refchain-types.md | 77 ++++ .../missing-refchains-on-hotdog/meta.yaml | 8 + .../nodes/dead-hard-reference-kind-filter.md | 59 +++ .../dead-jcmd-jfr-dump-wrong-source-v2.md | 44 ++ .../nodes/dead-jcmd-jfr-dump-wrong-source.md | 49 +++ .../nodes/dead-toolkit-prod-datacenter.md | 55 +++ .../nodes/design-pod-in-a-jar-harness.md | 127 ++++++ ...bandon-event-lost-to-dump-sampling-race.md | 131 ++++++ .../nodes/find-abandon-event-queue-fix.md | 89 +++++ .../find-admission-boost-implementation.md | 84 ++++ .../nodes/find-age-heuristic-insufficient.md | 54 +++ .../nodes/find-ages-vector-not-cleared.md | 47 +++ ...nd-already-admitted-blocks-deeper-chain.md | 51 +++ ...ind-already-admitted-blocks-unreachable.md | 67 ++++ .../nodes/find-anchor-holder-eviction.md | 94 +++++ .../nodes/find-anchor-live-feed-design.md | 113 ++++++ .../nodes/find-anchor-tail-starvation.md | 81 ++++ .../find-attribution-standards-survey.md | 93 +++++ ...ary-continue-skips-discovered-instances.md | 52 +++ .../nodes/find-canary-fixes-e-f.md | 90 +++++ .../find-canary-found-criterion-unmigrated.md | 78 ++++ .../nodes/find-canary-lane-backoff-design.md | 70 ++++ .../find-canary-search-cannot-terminate.md | 71 ++++ .../find-canary-search-forces-max-cadence.md | 55 +++ .../find-canary-stuck-abandon-detector.md | 117 ++++++ ...ind-canary-stuck-restart-wipes-frontier.md | 113 ++++++ ...te-presence-suppresses-static-admission.md | 66 +++ .../nodes/find-candidate1-never-tagged.md | 370 +++++++++++++++++ ...nd-candidates-234-die-before-resolution.md | 55 +++ .../nodes/find-cpu-pain-budget-blocks-bfs.md | 47 +++ ...d-cpu-pain-budget-starves-canary-passes.md | 107 +++++ .../find-dangling-jclass-local-ref-cache.md | 35 ++ ...find-default-live-samples-ratio-lottery.md | 77 ++++ .../find-depth0-durable-root-upgrade-gap.md | 46 +++ .../nodes/find-edit-removed-critical-code.md | 69 ++++ .../nodes/find-ema-batch-collapse.md | 79 ++++ .../nodes/find-engine-seam-data-race.md | 43 ++ .../nodes/find-field-name-decoding.md | 109 +++++ .../nodes/find-fresh-lane-verification.md | 69 ++++ .../find-gate-bypass-representative-paths.md | 49 +++ ...getobjectswithtags-quadratic-bottleneck.md | 183 +++++++++ .../nodes/find-holistic-design-issues.md | 85 ++++ .../nodes/find-hotdog-deploy-last-mile.md | 50 +++ .../find-isqueuedforrotation-quad-scan.md | 36 ++ .../nodes/find-jvmti-heap-walk-stw-vmop.md | 51 +++ .../nodes/find-klass-id-notation-mismatch.md | 59 +++ .../find-lambda-fragments-calltrace-id.md | 62 +++ .../find-leak-tag-pool-implementation.md | 107 +++++ .../find-leaktag-jfr-field-misalignment.md | 48 +++ .../find-marker-tag-slot-index-mismatch.md | 113 ++++++ .../nodes/find-one-shot-pretag-gate.md | 97 +++++ .../nodes/find-onpod-evidence-methodology.md | 51 +++ .../find-option-c-descend-walk-design.md | 163 ++++++++ ...find-per-class-caching-blocks-instances.md | 53 +++ .../find-per-tid-qualification-design.md | 87 ++++ .../find-priority-queue-starves-bfs-crawl.md | 75 ++++ .../find-refchains-log-flood-configurable.md | 122 ++++++ .../nodes/find-refchains-not-deployed.md | 64 +++ ...find-representative-changes-lose-canary.md | 61 +++ .../find-rolling-resume-expandfrontier.md | 46 +++ .../nodes/find-rotation-resize-blindspot.md | 61 +++ .../find-round16-endgoal-verification.md | 72 ++++ .../find-shared-deadline-starves-expand.md | 52 +++ .../find-static-field-sweep-cursor-fix.md | 162 ++++++++ ...find-static-field-sweep-never-completes.md | 175 ++++++++ .../find-sweep-completes-but-bfs-starved.md | 81 ++++ .../nodes/find-test-seam-aliasing.md | 70 ++++ ...threadloop-presleep-blocks-back-to-back.md | 61 +++ .../nodes/find-tier1-tail-starvation.md | 82 ++++ .../find-togcroot-orphaned-slot-stranding.md | 91 +++++ .../nodes/find-urgentoom-null-fn-mislabel.md | 69 ++++ .../find-wrapper-demotion-self-parent.md | 103 +++++ .../nodes/find-wrapper-not-in-anchor-tier.md | 86 ++++ .../nodes/hyp-regression-of-five-fixes.md | 67 ++++ .../nodes/hyp-warmup-transience.md | 41 ++ .../nodes/meta-circle-review.md | 106 +++++ .../nodes/meta-whackamole-analysis.md | 116 ++++++ .../nodes/q-allocation-site-selection.md | 76 ++++ .../nodes/q-canary-stuck-fix-alternatives.md | 132 ++++++ .../q-coverage-tracking-per-combination.md | 48 +++ .../q-dominant-gens-still-one-with-tid.md | 58 +++ .../q-heapliveobject-absent-on-pod-chunks.md | 34 ++ .../nodes/q-implement-two-fixes.md | 60 +++ ...-resize-instrumentation-rescan-priority.md | 49 +++ .../nodes/q-safepoint-budget-model.md | 57 +++ .../nodes/q-togcroot-acceptance-paths.md | 77 ++++ 130 files changed, 10510 insertions(+) create mode 100644 .investigations/missing-refchains-on-hotdog/INDEX.md create mode 100644 .investigations/missing-refchains-on-hotdog/STATE.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-adaptive-batchsize-onpod-verification.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-candidate-count-latch-mismatch.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-deadline-split-onpod-verification.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-deployed-so-1481-no-symbols.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-fixes-compile-and-gtest-pass.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-hotdog-trace-zero-runpass.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-hotspot-lifo-visitation-order.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-jafar-zero-refchain-events.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-jfr-analysis-real-recording.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-kind-counts-constant-pool-dominates.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-correlation-local-repro.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round1.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round10.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round11.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round12.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round13-results.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round13.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round14-results.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round14.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round15-results.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round15.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round16-results.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round2.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round3.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round4.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round5.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round6.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round7.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round8.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round9.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-livelock-pod-logs.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-marker-tag-arithmetic.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-post-resync-deployment-verified.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-postCB-onpod-live-verification.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-postfix-onpod-live-verification.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-postfix-static-field-onpod-live-verification.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-postfixEF-onpod-live-verification.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-source-poll-vs-callback.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-tid-clustering-onpod-verification.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-tid-clustering-per-thread-working.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-timing-split-callback-vs-jvmti.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-toolkit-and-onpod-methodology.md create mode 100644 .investigations/missing-refchains-on-hotdog/evidence/ev-uploaded-jfr-no-refchain-types.md create mode 100644 .investigations/missing-refchains-on-hotdog/meta.yaml create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/dead-hard-reference-kind-filter.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/dead-jcmd-jfr-dump-wrong-source-v2.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/dead-jcmd-jfr-dump-wrong-source.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/dead-toolkit-prod-datacenter.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/design-pod-in-a-jar-harness.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-abandon-event-lost-to-dump-sampling-race.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-abandon-event-queue-fix.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-admission-boost-implementation.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-age-heuristic-insufficient.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-ages-vector-not-cleared.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-already-admitted-blocks-deeper-chain.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-already-admitted-blocks-unreachable.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-anchor-holder-eviction.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-anchor-live-feed-design.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-anchor-tail-starvation.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-attribution-standards-survey.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-canary-continue-skips-discovered-instances.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-canary-fixes-e-f.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-canary-found-criterion-unmigrated.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-canary-lane-backoff-design.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-canary-search-cannot-terminate.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-canary-search-forces-max-cadence.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-canary-stuck-abandon-detector.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-canary-stuck-restart-wipes-frontier.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-candidate-presence-suppresses-static-admission.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-candidate1-never-tagged.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-candidates-234-die-before-resolution.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-cpu-pain-budget-blocks-bfs.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-cpu-pain-budget-starves-canary-passes.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-dangling-jclass-local-ref-cache.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-default-live-samples-ratio-lottery.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-depth0-durable-root-upgrade-gap.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-edit-removed-critical-code.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-ema-batch-collapse.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-engine-seam-data-race.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-field-name-decoding.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-fresh-lane-verification.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-gate-bypass-representative-paths.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-getobjectswithtags-quadratic-bottleneck.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-holistic-design-issues.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-hotdog-deploy-last-mile.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-isqueuedforrotation-quad-scan.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-jvmti-heap-walk-stw-vmop.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-klass-id-notation-mismatch.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-lambda-fragments-calltrace-id.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-leak-tag-pool-implementation.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-leaktag-jfr-field-misalignment.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-marker-tag-slot-index-mismatch.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-one-shot-pretag-gate.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-onpod-evidence-methodology.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-option-c-descend-walk-design.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-per-class-caching-blocks-instances.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-per-tid-qualification-design.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-priority-queue-starves-bfs-crawl.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-refchains-log-flood-configurable.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-refchains-not-deployed.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-representative-changes-lose-canary.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-rolling-resume-expandfrontier.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-rotation-resize-blindspot.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-round16-endgoal-verification.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-shared-deadline-starves-expand.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-static-field-sweep-cursor-fix.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-static-field-sweep-never-completes.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-sweep-completes-but-bfs-starved.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-test-seam-aliasing.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-threadloop-presleep-blocks-back-to-back.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-tier1-tail-starvation.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-togcroot-orphaned-slot-stranding.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-urgentoom-null-fn-mislabel.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-wrapper-demotion-self-parent.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/find-wrapper-not-in-anchor-tier.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/hyp-regression-of-five-fixes.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/hyp-warmup-transience.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/meta-circle-review.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/meta-whackamole-analysis.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/q-allocation-site-selection.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/q-canary-stuck-fix-alternatives.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/q-coverage-tracking-per-combination.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/q-dominant-gens-still-one-with-tid.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/q-heapliveobject-absent-on-pod-chunks.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/q-implement-two-fixes.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/q-resize-instrumentation-rescan-priority.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/q-safepoint-budget-model.md create mode 100644 .investigations/missing-refchains-on-hotdog/nodes/q-togcroot-acceptance-paths.md diff --git a/.investigations/missing-refchains-on-hotdog/INDEX.md b/.investigations/missing-refchains-on-hotdog/INDEX.md new file mode 100644 index 0000000000..473bb66c5c --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/INDEX.md @@ -0,0 +1,122 @@ +# Investigation Index — missing-refchains-on-hotdog + +## Findings (confirmed) +- find-marker-tag-slot-index-mismatch | ROOT CAUSE (FIXED): pollWatchedTargets() indexes candidate arrays by loop position instead of the tag-decoded slot | [root-cause, fixed, referenceChains, canary, marker-tag, off-by-slot] +- find-one-shot-pretag-gate | Second defect (FIXED): candidate pre-tagging was one-shot, tag→slot map desynced from selectLeakCandidates() | [self-heal, pre-tagging, candidate-count, latch, fixed] +- find-canary-search-cannot-terminate | Third element (PARTIALLY ADDRESSED): at 0/N found the canary search could neither complete nor abandon | [livelock, search-state, completion-gating, urgent, partially-fixed] +- find-canary-stuck-abandon-detector | Fix C (implemented): canary-specific stuck/abandon detector not suppressed by isUrgent() | [fix, livelock, search-state, urgent-abandon, canary, new-this-session] +- find-abandon-event-lost-to-dump-sampling-race | NEW ROOT CAUSE: SearchState::ABANDONED is transient and almost always overwritten before Profiler::dump() ever samples it | [root-cause, jfr-emission, race-condition, search-state, dump, referenceChains, NEW-BUG] +- find-abandon-event-queue-fix | Fix D (implemented, compiled, gtest-pass): bounded pending-abandoned-events queue closes the dump() sampling race | [fix, jfr-emission, race-condition, referenceChains, queue, NEW-THIS-SESSION] +- find-cpu-pain-budget-starves-canary-passes | ROOT CAUSE (zero canary resolution): silent `_cpu_pain_budget` gate in shouldRunPass() drains at only 1%/wall-clock by default, blocking runPass() for tens of seconds at a time | [root-cause, referenceChains, pain-budget, canary, starvation, silent-gate, NEW-THIS-SESSION] +- find-threadloop-presleep-blocks-back-to-back | Contributing bug: threadLoop() sleeps cadence_ns unconditionally before shouldRunPass() runs, contradicting its own "run back-to-back" comment | [bug, referenceChains, threadLoop, cadence, canary, NEW-THIS-SESSION] +- find-canary-fixes-e-f | Fix E+F (implemented, compiled, gtest-pass, CONFIRMED LIVE ON-POD): deleted leftover threadLoop sleep, canary-aware _cpu_pain_budget escalation (4x while unresolved) | [fix, referenceChains, pain-budget, threadLoop, canary, NEW-THIS-SESSION] +- find-refchains-not-deployed | Phase 1 (RESOLVED): pod ran ddprof-lib 1.48.1, which has no reference-chain code | [phase-1, deployment, resolved, ddprof-1.48.1] +- find-onpod-evidence-methodology | Prove deployment state ON the pod, not from downloaded artifacts or image tags | [methodology, user-feedback, on-pod, evidence-quality] +- find-hotdog-deploy-last-mile | Getting a branch build of ddprof-lib onto a hotdog pod is not scripted in this repo | [deployment, tooling, hotdog, patch-dd-java-agent] +- find-canary-stuck-restart-wipes-frontier | ROOT CAUSE (CONFIRMED, FIX C+B IMPLEMENTED+gtest-verified+CONFIRMED LIVE ON-POD): CANARY_NO_PROGRESS_PASS_LIMIT + full frontier wipe on restart prevents ever reaching a distant-but-reachable candidate | [root-cause, referenceChains, canary, restart, frontier, no-progress-limit, NEW-THIS-SESSION] +- find-static-field-sweep-never-completes | ROOT CAUSE CONFIRMED, FIX IMPLEMENTED+gtest-verified: admitStaticFieldRoots() never completed a sweep (275/275 passes truncated) - shared per-pass wall-clock deadline (5-50ms) too small for one non-resumable FollowReferences() over ~34k loaded classes | [root-cause, referenceChains, static-field, admitStaticFieldRoots, truncation, pause-target-budget, fixed, NEW-THIS-SESSION] +- find-static-field-sweep-cursor-fix | Fix (implemented, gtest-verified, COMMITTED a86f0dd87, DEPLOYED, CONFIRMED LIVE ON-POD): resumable per-call cursor + app-classes-first reordering; found+fixed a singleton test-isolation regression during verification; later extended with reversed holder fill + per-class non-static quota | [fix, referenceChains, static-field, cursor, resumable, reordering, gtest-verified, NEW-THIS-SESSION] +- find-candidate1-never-tagged | ROOT CAUSE CONFIRMED (kind_counts diagnostic, live on-pod), FIX IMPLEMENTED+committed (0e93ab4f7): heapReferenceCallback() admitted non-STATIC_FIELD edges (dominated by CONSTANT_POOL, 5-15x volume) during static-field sweep, burning budget. Fix evolved from hard filter to per-class quota (32/class) + resumable cursor. Sweep now completes laps but candidates still 0/1 — bottleneck moved to BFS throughput | [root-cause, fix, referenceChains, candidate, canary, marker-tag, static-field, constant-pool, quota, NEW-THIS-SESSION] +- find-sweep-completes-but-bfs-starved | Sweep completes full lap (200ms deadline) but candidates still 0/1: expandFrontier() must reach sweep-admitted entries through 144k _pending_expand backlog at 65-87 edges/pass. ADDRESSED by adaptive batch_size (BFS throughput up 100-200x) | [root-cause, referenceChains, bfs, expandFrontier, backlog, throughput, sweep, NEW-THIS-SESSION] +- find-getobjectswithtags-quadratic-bottleneck | GetObjectsWithTags is O(tag_map × batch_size) quadratic: 226-243ms per call with 3400 tags on 163k tag map, starving BFS to 3-4 edges/pass. FIXED: self-calibrating adaptive batch_size (commit 337b4c21d), CONFIRMED LIVE ON-POD: 456-971 edges/pass, EMA converged to ~62k ns/tag | [root-cause, fix, referenceChains, bfs, expandFrontier, GetObjectsWithTags, quadratic, throughput, NEW-THIS-SESSION] +- find-shared-deadline-starves-expand | ROOT CAUSE (FIXED): shared _pass_deadline_ns across sweep+expand+rotation — sweep ate entire deadline, expand got 0-1 edges. FIXED: per-sub-op deadline reset (commit b2acdaee2), CONFIRMED LIVE ON-POD: 1708-2766 edges/pass | [root-cause, fix, referenceChains, deadline, expandFrontier, sweep, pause-target, NEW-THIS-SESSION] +- find-rolling-resume-expandfrontier | Fix (COMMITTED b2acdaee2): rolling resume for expandFrontier pops fully-processed entries on truncated batch, same pattern as sweep cursor | [fix, referenceChains, expandFrontier, rolling-resume, cursor, truncation, NEW-THIS-SESSION] +- find-representative-changes-lose-canary | ROOT CAUSE (FIXED, COMMITTED 7bd7255c3): canary representative changed (LRU-evicted) for high-churn classes like [B — new representative had tag=0, canary lost track. Fix: re-tag representative + auto-mark all discovered instances of watched leak classes | [root-cause, fix, referenceChains, canary, representative, lru-eviction, marker-tag, NEW-THIS-SESSION] +- find-canary-continue-skips-discovered-instances | Bug in fix 7bd7255c3 (FIXED, COMMITTED 5d06d7328): canary path `continue` skipped the discovered-instances check — [B candidate took canary path, never tried discovered instances. Removed `continue` so both paths fall through | [root-cause, fix, referenceChains, canary, discovered-instances, control-flow, NEW-THIS-SESSION] +- find-cpu-pain-budget-blocks-bfs | DIAGNOSED: cpu_pain_budget silently blocked shouldRunPass in RUNNING branch — balance ~1075ms, draining at 12.5ms/iteration. No log until diagnostic (fcc67179a). Canary multiplier (4×) may need to be higher | [root-cause, referenceChains, cpu-pain-budget, shouldRunPass, silent-gate, NEW-THIS-SESSION] +- find-per-class-caching-blocks-instances | ROOT CAUSE (FIXED, COMMITTED cd68be618): per-class _resolved_chains caching meant first chain found (noise [B, depth=1, jni_local) blocked all other [B chains. Fix: key by frontier tag (per-instance), emit all chains, backend aggregates by class | [root-cause, fix, referenceChains, resolved-chains, per-instance, caching, NEW-THIS-SESSION] +- find-age-heuristic-insufficient | DIAGNOSED: oldest [B (age=172) was noise (136B, s3-netty-2), not leak (78MB, simulated-memory-leak). Pure age ranking picks noise. Proposed: allocation-site clustering + growth×survival (Cork/Melt/Swat) | [root-cause, referenceChains, livenessTracker, age-heuristic, representative-selection, NEW-THIS-SESSION] +- find-lambda-fragments-calltrace-id | CONFIRMED: lambdas fragment call_trace_id — synthetic methods produce different stack hashes for one logical site → N sites × 1 gen → no clustering signal. FIXED: switch to tid-based clustering (commit 2c50bf0cd) | [root-cause, fix, referenceChains, livenessTracker, call-trace-id, lambda, tid, NEW-THIS-SESSION] +- find-ages-vector-not-cleared | BUG (FIXED, COMMITTED 36c7fc8c8): KlassCountScratch::ages vector never cleared between epochs — gen_count inflated cumulatively. Ring-buffer slope unaffected (measures rate of change). Fix: ages.clear() in new-slot branch | [bug, fix, referenceChains, livenessTracker, ages, vector, inflation, NEW-THIS-SESSION] +- find-already-admitted-blocks-deeper-chain | ROOT CAUSE (FIXED, COMMITTED d30538fe3): admitObject returns ALREADY_ADMITTED for tagged objects, blocking deeper chain. [B first admitted as JNI-local root (parent_tag=0, depth=0) → depth=1 chain with no holder. Fix: FrontierTable::improveChain replaces shallow entry with deeper chain-attached entry | [root-cause, fix, referenceChains, admitObject, already-admitted, depth-1, chain, jni-local, NEW-THIS-SESSION] +- q-dominant-gens-still-one-with-tid | RESOLVED: per-thread tracking IS working (klass_id=6/9 shows dominant tid with age_count=3). dominant_gens=1 was from other classes, not the leak class. ages vector inflation bug found and fixed | [referenceChains, livenessTracker, tid, dominant-gens, subsampling, RESOLVED, NEW-THIS-SESSION] | CONFIRMED +- find-holistic-design-issues | Four design problems from holistic JFR analysis: (1) 27 leaking [B (78MB) NOT in the 40 chains; (2) chains cached/re-emitted after objects die; (3) no ReferenceChain↔HeapLiveObject correlation key; (4) 100× multiplier eats 45% CPU continuously. User approved redesign A-D | [root-cause, design, referenceChains, jni-local, targetTag, correlation, cpu-budget, NEW-THIS-SESSION] +- find-leak-tag-pool-implementation | Redesign A-D IMPLEMENTED (round 2 = commit 0db70994d adds alignment/AIMD/deadline/tag-priority/emergency-counter fixes, gtest pass, NOT yet verified on-pod): reusable 256-tag pool tags ALL tracked instances of candidate classes by per-tid age-diversity priority; depth==0 filter in discovered loop; ReferenceChain.targetTag=leak_tag + HeapLiveObject.leakTag correlation; adaptive CPU 100×/15×/1× with per-object coverage. Includes lessons: jlong unusable in event.h, filter must not live in buildChainEvent (breaks canary/tests). TEMP: CANARY_NO_PROGRESS_PASS_LIMIT 30→3 MUST REVERT | [fix, design, referenceChains, livenessTracker, leak-tag, correlation, adaptive-cpu, NEW-THIS-SESSION] +- find-ema-batch-collapse | REGRESSION (FIXED 0db70994d): per-tag EMA batch calibration collapsed batch 400→2 on-pod (O(tag_map) floor inflates per-tag cost at small batch — positive feedback); expand loop had no deadline check for gotw calls (1400 × 20ms = 10.4s CPU/pass). Fixed with AIMD on per-call EMA + per-iteration deadline check | [root-cause, fix, referenceChains, expandFrontier, GetObjectsWithTags, ema, aimd, NEW-THIS-SESSION] +- find-priority-queue-starves-bfs-crawl | ROUND-3 ROOT CAUSE (fix pending): _priority_expand flood (rotation 256/pass in vs ~150-300 deadline-bounded drain) + strict priority-first drain starves _pending_expand entirely; compounding: tagLeakInstances SetTag overwrites frontier tags of admitted instances; MAX_DISCOVERED=8 slots permanently held by noise; growth-gated leak tier dead in steady state. Fix = fair-share alternation + cap 1024 + correlate-not-retag state machine + leak-preferential slot eviction + fanout-priority rotation + reparentToDurableRoot + depth gate (depth==1 transient suppressed, durable KEPT - blanket depth>1 would drop the real static depth-1 shape) | [root-cause, fix, referenceChains, expandFrontier, priority-queue, starvation, NEW-THIS-SESSION] +- find-gate-bypass-representative-paths | LOCAL REPRO catch #1: canary/dead-rep/normal-rep chain-build paths bypassed the noise gate (cached stack-rooted depth-1 chain re-emitted forever = the pod's noise-chain symptom); plus round-4 gate design flaw: unconditional depth==0 suppression dropped real direct-retention chains (static value itself, Thread roots). Fixed: shared suppressChainEvent at ALL cache sites, predicate = depth<2 && transient | [root-cause, fix, gate, noise, canary, NEW-THIS-SESSION] +- find-klass-id-notation-mismatch | PRODUCTION BUG (pod-exhibited all along): Class.getName() dot-form vs GetClassSignature+normalize slash-form are different StringDictionary keys -> LT candidates never matched RCT discovered resolution ("resolved but no candidate match" on EVERY pod auto-mark); only array classes ([B) matched accidentally. Fixed: resolveKlassId uses the signature+normalize sequence (3rd user) | [root-cause, fix, klass-id, PRODUCTION-BUG, NEW-THIS-SESSION] +- find-test-seam-aliasing | Redesign broke synthetic-klass-id test seams (tagged=0, no candidate match; none of the pre-existing scenarios had been run since). Fixed: setKlassPopulationRepresentativeForTest0 resolves the rep's REAL id + aliases the synthetic id (re-key entry, create-if-absent, rep-before-seed order); scenarios also need per-round trend maintenance (one-shot ramps age out of hysteresis - observed: candidate dropped right after first interceptions) | [fix, test-seams, klass-id, aliasing, NEW-THIS-SESSION] +- find-rotation-resize-blindspot | USER'S CHALLENGE CORRECT: growing containers' new internals are structurally invisible (frozen children + fanout full of dead old backing arrays + starved blind lap). Fix set: fair-share fanout/lap rotation (fanout capped at ceil-half - unbounded fanout-first starved the lap, observed 4/206 passes), fanout hygiene (erase gone/ABANDONED parents), ancestor fanout (insert holder chain to root), requeueChainRootForRotation (per-poll enqueue of each candidate's chain ROOT - the profiler-native version of the resize-instrumentation idea). Observed: 4 leak-tag interceptions under the live elementData | [root-cause, fix, rotation, growing-collections, NEW-THIS-SESSION] +- q-resize-instrumentation-rescan-priority | USER IDEA (parked): bytecode instrumentation on collection resizes (ArrayList.grow/HashMap.resize) as a precise rescan-priority signal; the requeueChainRootForRotation fix is its profiler-native counterpart - revisit whether the machinery is still justified once the rotation fixes are verified on the pod | [design, rotation, growing-collections, instrumentation, NEW-THIS-SESSION] +- find-leaktag-jfr-field-misalignment | ROOT CAUSE of “leakTag=0/garbage in recording” (FIXED 0db70994d): leakTag declared before || contextAttributes but written after the context-attribute values — parsers read the first attribute byte as leakTag. Invariant: a per-event field must sit on the same side of contextAttributes in metadata as its write does of writeContextSnapshot() | [root-cause, fix, jfr, jfrMetadata, flightRecorder, leak-tag, contextAttributes, NEW-THIS-SESSION] +- find-wrapper-not-in-anchor-tier | ROOT CAUSE: LEAK_BUFFER wrapper (SynchronizedRandomAccessList, NOT Unmodifiable) IS root-attached STATIC_FIELD in the frontier but the collector's wrapping cursor (4 anchors/pass over 228k frontier) never reaches it before the candidate flaps out — the round-11 "wrapper walked" was a decoy (NetUtil's UnmodifiableRandomAccessList). Fix: index root_kind=8 entries (O(29) vs O(228k) scan) + prioritize by leak-tagged children | [root-cause, referenceChains, anchor-tier, collector-lottery, wrapper, SynchronizedRandomAccessList, NEW-THIS-SESSION] +- find-anchor-tail-starvation | ROOT CAUSE (round 13, probe-verified): anchor-selection tail starvation — index ~28k anchors, ~4k coverage per search (21 walks/pass × ~190-pass lifetime), wrapper at position 12-21k deterministically unreachable; fixed round 14 (tiered selection, 6eb72f994) | [root-cause, referenceChains, anchor-tier, selection-order, starvation, coverage-invariant, NEW-THIS-SESSION] +- find-tier1-tail-starvation | ROUND-15 DIAGNOSIS (round-14 build verified live): tiering works, container cohort 1633-1680 — but search lifetimes collapsed to 44-75 passes (fill 2.5-7.5k/pass incl. ~650/min seeding noise) while the wrapper (holder class at sweep index 24627/33270) admits at the anchor-index TAIL = container-ordinal ~1634 of 1633 ⇒ walk owed at pass ~102+ ⇒ deterministic miss; every search is a canary chase (candidate [B, 0/1 found) that needs exactly that walk to exit while its backoff slows the passes — self-sustaining deadlock. Fix: fresh-admission priority (walk anchors admitted since last pass first, STW-free reorder) | [root-cause, referenceChains, anchor-tier, canary, search-lifetime, fresh-admission-priority, NEW-THIS-SESSION] +- find-fresh-lane-verification | ROUND-15 VERIFIED LIVE (build f7b1ea416, pod cgdtx - the deploy REPLACED rz992, re-find the pod name after deploys): fresh lane works (fresh_queue <= 615 vs 1024 cap); the wrapper admitted root-attached AND WALKED once (19 edges, whole subtree incl. 17 [B chunks); chunks auto-marked with deep chains (6-12 hops); datadog.ReferenceChain events emitted (re-emitted=4/6) - first on this pod. Gap exposed: later searches demote the wrapper (see find-wrapper-demotion-self-parent). Streams mandatory (~800k lines/min flood) | [fix, referenceChains, anchor-tier, fresh-admission-priority, pod-verification, NEW-THIS-SESSION] +- find-wrapper-demotion-self-parent | ROUND-16 VERIFIED (build 4afe870a2, pod cgdtx): fix A live at scale (self_edge_skips=30816+; wrapper admitted root-attached parent=0 root_kind=8, WALKED as anchor tag 2184 - the round-15 parent==self demotion is gone); fix B live (quota_drops=6351, fifo_size=132 << 1024 - floods contained); FIRST leak-tag interceptions ever (12+, leak_tag >= 2^30 admitted with parent chains); leak-correlated chains cached (auto-marked target_tag=1073742*) and re-emitted 6->14 sustained per dump - END GOAL: leak-correlated datadog.ReferenceChain events flowing. Canary 0/1 remains (its representative not yet intercepted; the round-15 chase deadlock is broken - walks + interceptions every pass) | [fix, referenceChains, improveChain, at-risk-fifo, leak-tag-interception, pod-verification, END-GOAL, NEW-THIS-SESSION] +- find-urgentoom-null-fn-mislabel | ROOT CAUSE (round 16, fixed in 4afe870a2): autoTuneDefaults (facdc70c0, 2026-08-17) calls GetAvailableProcessors from start() when a max heap is set; SearchRestartTest's fixture (5adb32831, 2026-08-12) never wired it -> UrgentOOM null-crashed at start() for 5 WEEKS with its real subject (urgency bypass of the candidate gate) never executing - valid the whole time. Fixed by wiring mock_GetAvailableProcessors; gtestRelease green with NO exclusions for the first time since Aug 17. LESSONS: 'reproduces on HEAD' only proves 'not from today's diff' - bisect to the introducing commit (git log -S, minutes) before calling anything pre-existing; a crashing test is LOST SIGNAL, never separate debt. Same round: the reset() seam gap (FIFO/fresh-queue never cleared - quota test's saturated residue crashed the next fixture's runPass on a null GetObjectsWithTags) | [root-cause, test-seams, autoTuneDefaults, GetAvailableProcessors, gtest, mislabel-lesson, NEW-THIS-SESSION] +- find-round16-endgoal-verification | END GOAL VERIFIED (build 4afe870a2, pod cgdtx): FIRST leak-tag interceptions ever (12+, leak_tag >= 2^30 admitted with parent chains); leak-correlated chains cached (target_tag=1073742*); re-emitted 6->14 sustained - datadog.ReferenceChain events tied to the leak flowing. Fix A wider than designed: many synchronized-list statics (ProcessTags$Lazy + more) were being demoted - self_edge_skips=30816+ protects the whole population. Fix B: quota_drops=6351, fifo 132 << 1024. ACCIDENTAL CONTROL: 3h non-leaking app image = ZERO tracker activity (gate keeps machinery dormant on healthy apps - verified live; also: quiet app after deploy != broken build, check the app version + traffic first). Canary 0/1 remains (representative not yet intercepted; chase deadlock broken, exit expected). See ev-leaktag-onpod-round16-results.md | [pod-verification, END-GOAL, leak-tag-interception, control-run, canary, NEW-THIS-SESSION] +- find-canary-found-criterion-unmigrated | ROOT CAUSE (pod 289f8, FIXED 788d7b2a7, pending pod verification): the marker->leak-tag design migration never migrated the found criterion - _candidate_found_bits set ONLY in the marker-tag block, markers never set under leak tags -> the chase structurally unresolvable (0/1 on every JVM, searches ending via frontier-cap/no-progress); the rep path doubly broken (cache checked by leak tag - never a key; buildChainEvent(leak tag) always misses - interceptions insert fresh frontier tags with entry.leak_tag inside). Fix: a leak-tag-target chain (target >= 2^30) for a candidate slot marks it found + records its canary link; rep path resolves by the slot's chain key; noise chains do not mark found. Criterion is slot-any-instance (the per-instance guarantee moved into the leak-tag correlation; exit-on-first-leak-chain vs sporadic subtree coverage - coverage itself is an upstreaming-round topic). Events were never blocked - this was chase lifecycle. Test LeakTagChainMarksCanaryFound, 122/122. Next deploy: watch for 'canary found: ...' + 1/1 + a clean search-restart cycle | [root-cause, canary, leak-tag, found-criterion, design-migration, fix, round-19, NEW-THIS-SESSION] +- find-refchains-log-flood-configurable | LOG FLOOD SOLVED (f70bcc454): round-12..16 TEST_LOG diagnostics now runtime-gated via rcDebugLevel.h (referenceChains.cpp + livenessTracker.cpp re-point TEST_LOG at a level check + TEST_LOG_SUMMARY). 0=SILENT default (pod-safe with the same debug .so), 1=lifecycle/summary (<300 lines/min: canary/candidates, rotation counters, re-emit, leak-tag correlation), 2=full firehose (per-object admits/auto-marks/fold/selectEntries - sized from the 3.1M-line round-16 stream). Env DD_PROFILING_REFERENCE_CHAINS_DEBUG=N; runtime override no restart: echo N > /tmp/ddprof_root/refchains_debug_level (~1s TTL from threadLoop; heap callbacks read only the cached atomic - open/read never there). LESSONS: gtest binary compiles main sources WITHOUT DEBUG (level machinery unconditional so the UT can test it); rejected multi-edit batches apply NOTHING atomically - re-apply every edit and grep each landed. Pod still floods until next deploy | [observability, debug-logs, runtime-knob, fix, NEW-THIS-SESSION] +- meta-whackamole-analysis | META (round 19, user challenge 'are we running in circles again'): rounds 13-16 convergent (funnel measured then fixed to goal); rounds 17-19 whackamole (2 of 3 about our own diagnostics + a lifecycle bug present since the marker->leak migration). TAXONOMY of all 30+ defects by producing mechanism: (1) scheduler starvation under diverse load ~12x - no liveness invariant asserted anywhere; (2) keyspace confusion ~6x - three tag keyspaces, every join a bug farm; (3) half-migrated designs (marker->leak, TEST_LOG->gate) - consumers never enumerated; (4) test-seam gaps - unit suite green while system broken; (5) restart state lifetime; (6) observability as defect source; (7) deploy drift (solved). STRUCTURAL STATEMENT: scheduler + 3 keyspaces + per-search state verified by 348 unit gtests + a staging pod at hours/iteration n=1 - classes 1/2/3/5 are invisible to unit tests BY CONSTRUCTION, so the pod surfaces them one deploy at a time in random order. Whackamole is the predictable output of this verification topology, not a discipline failure. EXIT: stop making the pod the oracle - P0 pod-in-a-jar harness (design-pod-in-a-jar-harness) + P1 keyspaces audit + P2 always-on health line + P3 reset-contract test + P4 deferred queue triaged to harness-flagged | [meta, root-pattern, defect-taxonomy, strategy, methodology, NEW-THIS-SESSION] +- find-candidate-presence-suppresses-static-admission | RETRACTED DEADEND (round 20): the harness's first 'catch' was its OWN builder bug - classes capture &node_tags[node] as jclass identity, the 300-filler addNode()s reallocated the vector, every pointer dangled, indexOfNode failed in the sweep, the gate closed on an EMPTY lap - all silently with a clean COMPLETED search. Capacity contract (reserve before capture) encoded + TopologyCapacityContractStaticAdmits regression. LESSONS: fixture capture-pointer contracts are a new test-seam-gap class (mock data structures have invariants tests must maintain, violations fail silently in production paths); when a harness catches something, bisect the HARNESS's own construction first (the full-vs-minimal topology difference pointed at the builder all along); test-side env-gated mock printfs are decisive in non-DEBUG binaries. Pod round-18 sporadic coverage NOT explained by this (open, upstreaming round). With the fix ALL invariants live: 128/127/1skip | [deadend, retracted, harness-fixture-bug, capacity-contract, methodology, NEW-THIS-SESSION] +- design-pod-in-a-jar-harness | DESIGN (pending user approval): system-level simulation harness - the REAL tracker loop over a mock JVMTI topology encoding every pod-discovered shape (synchronized wrapper with mutex==this self-edge, flood classes, accumulating leak chunks, noise, thread-locals, healthy-app mode), driven across multiple search lifetimes + restarts. 8 invariants (L1 coverage liveness, L2 canary resolution - fails on the pre-round-19 build, L3 natural completion, L4 LOG BUDGET - the flood and tier stragglers as assertions, L5 quotas/monotonicity, L6 restart hygiene as a test not discipline, L7 boundedness, L8 dormancy - the accidental control encoded). 7 of the last 10 pod rounds would have been red local tests before any deploy. Cost: core 1 session (reuses BfsTest's 65-test mock machinery), then P1-P4 small. Not proposed: more pod-driven fix rounds | [design, system-test, invariants, harness, NEW-THIS-SESSION] +- meta-circle-review | META: rounds 9-13 circular at the strategy level (5 rounds tuning the same funnel: coverage=rate×lifetime≥population, never measured whole until r13); structural exit = option C one-shot whole-heap pass (the JFR-LeakProfiler shape, per our own standards survey); process fixes: measure scale invariants first, scale gtest, end-to-end success gates | [meta, process, strategy, coverage-invariant, scale-testing, NEW-THIS-SESSION] +- find-edit-removed-critical-code | LESSON: removing a TEMP diagnostic interleaved inside a `for` loop accidentally deleted the entire loop (holder-fill in admitStaticFieldRoots); build compiled clean, caught only by diff review. Prevention: never replace a block mixing diagnostic + original code; edit only the diagnostic lines; always review full git diff before building | [lesson, edit-safety, methodology, NEW-THIS-SESSION] + +- ev-leaktag-onpod-round7 | Pod round 7 (build 13d3f87b3, JVM 4818, replacement pod srwnz - the ORIGINAL pod was evicted by an in-pod heap dump probe: DO NOT jcmd GC.heap_dump on these pods): both descend-walk prongs LIVE and per design (thread walk idempotent after 15-edge ThreadLocalMap admission, leak NOT ThreadLocal-held; 4 root-attached statics walked wholesale every pass, single walks up to 3119 edges; registerExistingThreads works - leak thread predates the recording and IS registered; tagged=22 stable; pendingExpand still net-growing), interception STILL ZERO - tagged chunks sit past the 6-hop cap or under an uncovered root; NOT frame-locals (thread parks idle between task runs). Round-8 fix (user-picked both, c6635fe0e): DESCENT_HOPS 16 + JNI_GLOBAL anchors in the same tier | [pod-verification, round-7, descend-walk, depth-cap, NEW-THIS-SESSION] +- ev-leaktag-onpod-round11 | Pod round 11 (04539b821): A WORKED — search lifetimes up, cycle_complete=1 REPEATEDLY (sweep laps complete for the first time; laps wrap in <2 min; the alternation initially misread as restarts is lap wraps) so the root-attached static cohort is now large (collector walks ~4 root_kind=8 anchors/pass). B is a NO-OP in production (GOTW returns unspecified order; round-10's starvation model was mock-order-based; truncation drops both cohorts uniformly). Wrapper still not reached by the collector's 4/pass cursor lottery (~40% of picks truncation-dropped) before the search went terminal — but the WATCH CONTINUATION REFUTED "not in the eligible set": at 21:10 UTC the wrapper WAS walked with ZERO interception, narrowing the question to decoy-instance / budget-starved-descend / untagged-at-walk-time (round-12 diagnostic 8ca24a524 discriminates). NEW: the candidate flapped out WHILE the leak grows (slope=-47.3, consecutive_positive=0, secondsToOOM=4351s, heap 4.7GB) — suspicion: self-reinforcing dropout (slot drops → watched-tid boost clears → sample thins → negative slope keeps it out). Next window: OOM urgency phase | [pod-verification, round-11, cycle-complete, candidate-flap, oom-endgame, NEW-THIS-SESSION] +- ev-leaktag-onpod-round10 | Pod round 10 (build 6f3c6cc2e = B' + nesting fix): B' mechanically LIVE (at-risk pushes 250-500/min, cap-pinned FIFO, 16 FIFO + 4 collector anchors/pass, chain-attached machinery anchors walked) and leak tagging healthy (21 tagged, 78MB) — interception STILL ZERO, new blocker chain OBSERVED: restarts every ~3-6 min (frontierSize non-monotonic, candidateFound flapping 0/1↔0/0 = the known epoch-noise residual) reset the sweep cursor (cycle_complete=0, ~25k/34397 after 7 min ⇒ 15-60 min/lap) so ProfileAnalyzer (high class index) is never swept ⇒ the LEAK wrapper never admitted ⇒ invisible to BOTH the collector cohort and the at-risk FIFO. Suspected restart driver: TEMP CANARY_NO_PROGRESS_PASS_LIMIT=3 (~3 min at 1 pass/min). Secondary: FIFO-first order starves the collector suffix (walked 6-16 of 20) in ~60% of passes; at-risk push rate measured 15-30x drain rate | [pod-verification, round-10, b-prime, fifo, sweep-lap, restarts, NEW-THIS-SESSION] +- ev-leaktag-onpod-round9 | Pod round 9 (c9a57f681, JVM 77972): anchor tier = exactly 76 machinery statics (jnr/charsets/reflect), ZERO app classes, NO LEAK_BUFFER holder — the holder never enters the tier; UPLOADED recording contains 12 ReferenceChain events with working per-hop edge names (machinery cohort fully explained: Mac/HmacCore k_opad/k_ipad ThreadLocals, charset constants) AND HeapLiveObject events with leakTags on the 78MB leak chunks; chains' targetTags disjoint from leak tags (leak still chainless). ORIGINAL zero-events question RESOLVED | [pod-verification, round-9, anchor-tier, edges, heapliveobject, events-flowing, NEW-THIS-SESSION] +- find-anchor-holder-eviction | ROUND-9 ROOT CAUSE (fix pending user pick): dual-reachable static holders are permanently excluded from the anchor tier — parent_tag==0 is unidirectional (improveChain replaces root-attached with chain-attached; re-root refused at referenceChains.cpp:2376 as documented limitation), so the tier only holds statics reachable by NO other path (machinery constants, exactly the 76 observed). Options RE-EVALUATED after standards survey: A refuted, B minimal-diff fallback, B' live-feed recommended | [root-cause, referenceChains, anchor-tier, improveChain, parent-tag, eviction, NEW-THIS-SESSION] +- find-attribution-standards-survey | Survey (JVM + non-JVM, web-sourced, primary sources read): JFR LeakProfiler keeps NO persistent attribution — root set enumerated fresh at every emit, one BFS with first-path-wins mark bits (openjdk leakprofiler source verified); MAT persists the full graph and computes shortest paths on demand; LeakCanary/shark compute dominators offline via Lengauer-Tarjan (workstation-only); dynamic-SSSP theory (Even-Shiloach, arXiv:2407.09651) proves exact incremental path labels are hopeless — lazy + bounded recompute is the sanctioned compromise. PRINCIPLE: a per-object single attribution is a heuristic stand-in for a dominator, unmaintainable incrementally, must NEVER gate selection — only reporting | [survey, methodology, attribution, dominator-tree, dynamic-graph-algorithms, anytime-search, jfr, mat, leakcanary, NEW-THIS-SESSION] +- find-already-admitted-blocks-unreachable | PRE-EXISTING PRODUCTION BUG (FOUND implementing B', FIXED same session): 57aec4895 placed improveChain INSIDE heapReferenceCallback's `if (*tag_ptr == 0)` first-admission block — where it is a guaranteed no-op (fresh entry == this edge's (parent,depth)) — so improveChain, reparentToDurableRoot, AND maybeUpgradeRootAttachedRootKind were ALL dead code for already-admitted objects from the sweep/BFS/descend paths since 2026-08-28 (heapRootCallback has its own working copy). REWRITES the eviction mechanism: improveChain demotion NEVER fired on any pod round — the holder was excluded purely by admission-order birth + the dead upgrade path. Gtests missed it: they drove the functions via the test accessor, not the wiring. Fix: real `else if (*tag_ptr > 0)` arm; 112 gtests + full gtestDebug green | [root-cause, PRODUCTION-BUG, referenceChains, heapReferenceCallback, improveChain, dead-code, nesting, NEW-THIS-SESSION] +- find-anchor-live-feed-design | Fix B' (IMPLEMENTED this session, gtest-verified, NOT yet committed/deployed): feed the anchor walk-set from live enumeration — the sweep's static-edge callback pushes frontier-present but not-root-attached-static holder tags into a bounded preallocated FIFO (deduped, PriorityExpandSet pattern); walkStaticFieldAnchors drains it. Anchor selection stops reading parent_tag/root_kind → eviction structurally impossible, both admission orders covered by construction; O(k) drain replaces the collector's 190k-entry full-table scan. Caveat (inferred, unmeasured): push-rate ≪ drain-rate depends on the at-risk filter actually shrinking the population — verify with the planned TEMP counter before sizing the FIFO. Chain correctness unaffected (thread-path chain of the holder is complete; interception only needs the walk to enumerate tagged chunks). JNI_GLOBAL stays on the existing table path | [fix, design, referenceChains, anchor-tier, live-feed, fifo, eviction-proof, NEW-THIS-SESSION] +- ev-leaktag-onpod-round8 | Pod round 8 (post-c6635fe0e, RENAMED pod jb1-668df5bcff-f75l8, JVM 4445): both prongs live, walks un-truncated at 16 hops (per-pass edges 10→3608, rotation cycling anchors), tagging healthy (klass_id=5 tid=4655 tagged=8 max_size=78MB), interception STILL ZERO — and the app's retention shape found in its bytecode: ProfileAnalyzer.LEAK_BUFFER, static final List in Collections.unmodifiableList, 3-4 hops from root static = textbook prong-2. Local scenario intercepts the identical shape → suspect is WHICH anchors enter the tier (wrapper never admitted root-attached / root_kind misclassified / root-attached entry replaced by chain-attached via improveChain). Per-anchor TEMP diagnostic committed c9a57f681 for round 9. Also: ZERO HeapLiveObject events in any chunk → q-heapliveobject-absent-on-pod-chunks | [pod-verification, round-8, descend-walk, interception-zero, leak-buffer, NEW-THIS-SESSION] +- find-field-name-decoding | Per-hop retention-edge field names in datadog.ReferenceChain (new "edges" T_STRING|F_ARRAY array, aligned leaf-to-root): ordinal captured at admission (jvmtiHeapReferenceInfoField.index is a jint SPEC ordinal over the referrer's flattened interface+superclass+own field space - source-verified HotSpot; NOT a jfieldID), name resolved at emission via GetObjectsWithTags->IsInterface/JNI GetSuperclass/GetClassFields/GetFieldName with a per-class decoded-ordinal cache; any decode failure degrades to the edge KIND label, never a fabricated name; buildChainEvent env-threaded (partial mock tables); OPEN: J9 walker ordinal compliance inferred, not source-verified (fail-safe degrades if wrong) | [field-names, hop-labels, jvmti-spec, ReferenceChain, jfr, NEW-THIS-SESSION] +- find-option-c-descend-walk-design | Option C (user-picked: both prongs, taxonomy-driven, NOT shaped by probing hotdog) COMMITTED (186468437 core + 01c591eea scenario) + PUSHED, all suites green, NOT yet deployed: unified bounded descend-walk FollowReferences(initial_object=anchor) reusing heapReferenceCallback; prong 1 = walkCandidateThreadLocals (tid->jthread registry: onThreadStart/End hooks + registerExistingThreads() sweep at Profiler::start - a leak thread is typically alive since before the recording; anchor gate descends the Thread's edges only into ThreadLocalMap - class-tag comparison, NOT jvmtiHeapReferenceInfoField.index, whose jint ordinal vs GetClassFields order the design avoids depending on), prong 2 = collectStaticFieldAnchorsForRotation + walkStaticFieldAnchors (parent_tag==0 root_kind==STATIC_FIELD, wrapping cursor, batched GOTW resolve). Interception = complete chain in one bounded STW. 553 gtests green incl. 2 new descend-walk gtests; reference-chain slow family 9/9 incl. NEW ThreadLocalLeakReferenceChainTest (thread walk engages live: walked=1 edges=18, [tl-correlation-found]); JFR-roundtrip crash taught: new JVMTI calls must not run in RCT::start() (partial mock tables) - run from Profiler::start() lifecycle | [option-C, implemented, descend-walk, thread-local, static-holder, referenceChains, NEW-THIS-SESSION] + +## Open questions (post-static-field-fix) +- find-candidates-234-die-before-resolution | REFUTED as bug: user confirmed only one real leak exists, other candidates are ordinary GC-able instances dying before search reaches them (expected) | [refuted, referenceChains, candidate, canary, sample-lifetime, NEW-THIS-SESSION] +- q-safepoint-budget-model | User clarified: per-call STW cap 50ms (per chunk) + cumulative ≤500ms/sec rate cap. Current code wrongly shares one deadline across all sub-operations | [safepoint, budget, deadline, per-call, cumulative, rate-cap, design, NEW-THIS-SESSION] +- q-allocation-site-selection | How to select representatives by allocation site? User approved combining directions 1+3 (Cork/Melt allocation-site clustering + Swat growth×survival). Implemented (0b492612b) with call_trace_id, then switched to tid (2c50bf0cd) after lambda fragmentation confirmed on-pod | [design, referenceChains, livenessTracker, allocation-site, representative-selection, tid, NEW-THIS-SESSION] | CONFIRMED +- q-coverage-tracking-per-combination | Adaptive-CPU coverage is currently per-OBJECT (assigned vs resolved leak tags) but user required per-(call_trace_id, tid) COMBINATION — 10 objects sharing one combination should need only 1 chain. Option 2 (dedupe at tagLeakInstances time, tag one representative per combination) is likely leaner; decide after on-pod verification | [adaptive-cpu, coverage, call_trace_id, tid, referenceChains, NEW-THIS-SESSION] +- q-heapliveobject-absent-on-pod-chunks | RESOLVED: local pod chunks never contained them — UPLOADED recordings do (78MB leak chunks with leakTags, simulated-memory-leak thread). Evidence rule: dump-time/liveness events verified only from uploaded recordings; jfr print crashes on ReferenceChain (use JMC API) | [heapliveobject, liveness, jfr, evidence-source, RESOLVED, NEW-THIS-SESSION] | CONFIRMED + +## Hypotheses +- hyp-warmup-transience | Zero events is just warm-up | [warm-up, transient, refuted] | REFUTED +- hyp-regression-of-five-fixes | Regression of one of the five `reference-chains` fixes | [regression-check, sibling-investigation, refuted] | REFUTED + +## Dead ends +- dead-jcmd-jfr-dump-wrong-source | `jcmd JFR.dump` is the wrong evidence source for ddprof events | [methodology, wrong-evidence-source, jcmd, jfr] | REFUTED +- dead-jcmd-jfr-dump-wrong-source-v2 | Repeated the jcmd mistake — jcmd dumps JDK JFR only, NOT ddprof native JFR. Must kubectl cp from /tmp/ddprof_root/pid_XXX/jfr/ | [methodology, wrong-evidence-source, jcmd, jfr, ddprof, NEW-THIS-SESSION] | REFUTED +- dead-toolkit-prod-datacenter | `--datacenter us1.prod.dog` for staging hotdog profiles | [profiling-toolkit, tooling-gotcha, datacenter, org-id] | REFUTED +- dead-hard-reference-kind-filter | Hard reference_kind filter (drop ALL non-STATIC_FIELD) superseded by per-class quota — excluded CP-based leaks entirely | [referenceChains, static-field, constant-pool, reference-kind, filter, refuted, NEW-THIS-SESSION] + +## Evidence +- ev-source-poll-vs-callback | Source excerpts: writer decodes the slot, reader uses the loop index | [source, referenceChains, citations, verified-at-head] +- ev-marker-tag-arithmetic | Logged marker tag decodes to slot 1 but code uses index 0 | [marker-tag, arithmetic, proof, referenceChains] +- ev-livelock-pod-logs | Pod logs show a stable livelock, not warm-up | [pod-logs, livelock, canary, buildCanaryChainEvent, TEST_LOG] +- ev-candidate-count-latch-mismatch | _candidate_count latched at 3 while selectLeakCandidates offers 5 | [pod-logs, candidate-count, latch, pre-tagging] +- ev-jafar-zero-refchain-events | Event types registered, event count still 0 | [jfr, jafar, zero-events, post-resync] +- ev-post-resync-deployment-verified | After the resync the feature IS deployed and enabled | [hotdog, deployment, post-resync, md5, on-pod] +- ev-deployed-so-1481-no-symbols | Deployed .so is ddprof-lib 1.48.1 with zero reference-chain symbols | [hotdog, deployment, native-lib, md5, ddprof-1.48.1, on-pod] +- ev-uploaded-jfr-no-refchain-types | Pre-resync uploaded profiles do not even declare datadog.ReferenceChain | [jfr, profiling-toolkit, event-types, pre-resync] +- ev-toolkit-and-onpod-methodology | Methodology corrections, user quotes, and Profiling Toolkit gotchas | [methodology, user-quote, profiling-toolkit, tooling-gotcha] +- ev-fixes-compile-and-gtest-pass | Fix A/B/C compile cleanly and pass the existing referenceChains gtest suites | [build, gtest, verification, referenceChains] +- ev-postfix-onpod-live-verification | Fix A + Fix C confirmed live on the resynced hotdog pod; Fix B implied | [pod-logs, on-pod, post-fix, live-verification, md5, fix-a, fix-c] +- ev-hotdog-trace-zero-runpass | 45s continuous trace: 0 runPass, 1472 pollWatchedTargets — pinpoints a long-lived silent gate, plus a kubectl --since log-retention gotcha | [pod-logs, on-pod, trace, runPass, canary, NEW-THIS-SESSION] +- ev-postfixEF-onpod-live-verification | Fix D/E/F confirmed live (0->50 runPass); new bottleneck surfaced: CANARY_STUCK abandon/restart every ~20s wipes frontier before reaching candidate | [pod-logs, on-pod, post-fix, live-verification, fix-e, fix-f, fix-d, canary-stuck, frontier-wipe] +- ev-postCB-onpod-live-verification | Fix C+B confirmed live: zero CANARY_STUCK abandons across 3 traces (430s), frontier grows unbroken 25k->80k; candidate-slot-churn concern investigated and retracted (slots verified stable); candidates still 0/5 | [pod-logs, on-pod, post-fix, live-verification, fix-c, fix-b, canary-stuck, frontier-growth] +- ev-kind-counts-constant-pool-dominates | 36 live samples of temp per-reference_kind tally: CONSTANT_POOL (k9) is largest/most variable kind every sample (2353-10894), 5-15x STATIC_FIELD (k8); ARRAY_ELEMENT (k3) flat at 512=chunk size | [pod-logs, on-pod, kind_counts, constant-pool, static-field, chunk-truncation, NEW-THIS-SESSION] +- ev-postfix-static-field-onpod-live-verification | Sweep cursor advances by 512/call, real edges admitted (up to 737/chunk), lap-truncation latch works; cycle_complete=1 not yet observed | [pod-logs, on-pod, post-fix, live-verification, sweep, cursor, NEW-THIS-SESSION] +- ev-hotspot-lifo-visitation-order | HotSpot FollowReferences uses LIFO visit_stack — drives reversed holder fill for ascending class visitation order | [hotspot, jvmti, FollowReferences, LIFO, visit-stack, source, holder-fill, reversed, NEW-THIS-SESSION] +- ev-timing-split-callback-vs-jvmti | Our callback 5-17% of wall-time, JVMTI heap-walking 83-92%; deferring admitObject won't fix truncation; sweep completes cleanly at 200ms | [timing, callback, jvmti, FollowReferences, safepoint, overhead-split, NEW-THIS-SESSION] +- ev-adaptive-batchsize-onpod-verification | Adaptive batch_size CONFIRMED LIVE ON-POD: EMA converged to ~62k ns/tag, batch_size ~390-420, gotw_ms 20-35, edges/pass 456-971 (was 3-4, ~100-200x improvement) | [pod-logs, on-pod, post-fix, live-verification, adaptive-batchsize, gotw, ema, NEW-THIS-SESSION] +- ev-deadline-split-onpod-verification | Deadline split CONFIRMED LIVE ON-POD: expand_phase 793-1606 edges (was 0-1), total 1708-2766 edges/pass (was 0-21). 5 candidates, frontier growing 92k→179k | [pod-logs, on-pod, post-fix, live-verification, deadline-split, rolling-resume, NEW-THIS-SESSION] +- ev-jfr-analysis-real-recording | JFR analysis of real ddprof recording: 2 ReferenceChain events (canary depth=14 rootKind=unknown, auto-mark depth=1 jni_local=noise). 13 [B in HeapLiveObject (12 leak 78MB + 1 noise 136B). Age heuristic picks noise (age=172) over leak (ages 33-168) | [jfr, jafar, on-pod, ddprof-jfr, analysis, NEW-THIS-SESSION] +- ev-tid-clustering-onpod-verification | SUPERSEDED by ev-tid-clustering-per-thread-working | [pod-logs, on-pod, post-fix, live-verification, tid-clustering, dominant-gens, NEW-THIS-SESSION, SUPERSEDED] +- ev-tid-clustering-per-thread-working | Per-thread generation tracking CONFIRMED working on-pod: klass_id=6/9 shows dominant tid with age_count=3. Re-mint fired. 8 ReferenceChain events emitted. But chains are depth=1 with no holder | [pod-logs, on-pod, post-fix, live-verification, tid-clustering, per-thread, dominant-gens, NEW-THIS-SESSION] +- ev-leaktag-onpod-round1 | On-pod round 1 of the leak-tag redesign (build 1ce2b4f03): pool tagging works (tagged=28, rep tag 0x4000000F), 15x multiplier fires, 8 chains cached+drained — BUT batch=2 collapse (3 passes/15min, 10.4s CPU/pass), chains are depth=1 noise with frontier-tag targetTags (BFS never reached leak instances), leakTag column misparsed (field-order bug) | [pod-logs, on-pod, post-fix, live-verification, leak-tag, batch-collapse, jfr, NEW-THIS-SESSION] +- ev-leaktag-onpod-round2 | Round 2 on-pod (0db70994d, PID 48355): AIMD verified working (oscillates 34→17→81 around 25ms budget), cadence 12 passes/min, tagged=129, drains batch=8 — BUT priority queue 39k→103k, pending never drained, zero interceptions, 8 noise chains re-emitted. Operational lessons: identify deployed build via per-iteration log fields (ema_call_ms), pod chunks lack ReferenceChain/HeapLiveObject types (need merged upload) | [pod-logs, on-pod, post-fix, live-verification, aimd, leak-tag, NEW-THIS-SESSION] +- ev-leaktag-correlation-local-repro | Local repro delivered: LeakTagCorrelationScenario + leak-correlation launcher mode + LeakTagCorrelationReferenceChainTest (slow suite; failure diagnostics = filtered child TEST_LOG tail in the assertion message). 8-step trail: gate bypass -> notation bug -> seam aliasing -> tagging determinism (1MB chunks + scan-every-3rd-round) -> resize blindspot fix set -> root requeue (4 interceptions) -> candidate hysteresis aging (per-round seeding). Replaces the pod deploy loop for the whole correlation subsystem. PASSED and committed (663784137): [correlation-found] 1073742077 end-to-end through the growing ArrayList; full testDebug then exposed 3 more real defects via the seams test (dangling jclass, engine race, depth-0 upgrade gap) - all fixed | [local-repro, E2E, post-fix, NEW-THIS-SESSION] +- ev-leaktag-onpod-round3 | Build 663784137 on hotdog (JVM 75258): proportional batch + static sweep laps + depth-0 root upgrades (76) + leak-tag pool (247 tagged) + discovered loop + noise gate + first-ever on-pod ReferenceChain emission (2 THREAD-rooted events) all verified live. NOT firing: intercepted/recordDiscoveredInstance/auto-marked/requeue = 0 - tagged instances and frontier-admitted instances are DISJOINT; both correlation bridges (interception, correlateAdmittedLeakTag) require crawl touch, and the crawl (~200-500 edges/min vs 199k frontier) has not reached the 18 tagged machinery byte[]s in 2.5h. Fix direction: intercept-only full-graph sweep (sweep laps DO complete on pod - walk throughput is not the wall, GetObjectsWithTags batch resolve is) | [pod-logs, on-pod, post-fix, live-verification, proportional-batch, leak-tag, NEW-THIS-SESSION]- find-dangling-jclass-local-ref-cache | `_cached_object_class` (java/lang/Object jclass) was cached as a LOCAL ref - safe only for the never-returns-to-Java BFS thread; JNI-entered test seams freed it on return, and pass 2 SIGSEGV'd in NewObjectArray. Fixed: global ref; JNIEnv*-keying + detach-time invalidation + startThread() stale-cache clearing all deleted as unnecessary; gtest mock got a NewGlobalRef slot | [root-cause, fix, jni, crash, test-seams, NEW-THIS-SESSION] +- find-jvmti-heap-walk-stw-vmop | FollowReferences = VM_HeapWalkOperation (jdk21 jvmtiTagMap.cpp:2379, doit :2934, VMThread::execute, no allow_nested_safepoints override): a full-heap walk is ONE stop-the-world pause (~10s on the 7GB hotdog heap). Option B (full-graph intercept sweep) RETRACTED. Static sweep laps are RESTRICTED walks (tiny subgraph) - not evidence about full-walk cost. GetObjectsWithTags is NOT a VM op - O(tag-map) scan under the tag-map mutex per call. No cheap reach-unknown-holders mechanism exists for an external JVMTI agent (JFR's leak profiler rides GC internally) | [root-cause, hotspot, jvmti, stw, design-constraint, NEW-THIS-SESSION] +- find-canary-search-forces-max-cadence | While any candidate canary is unfound (needRefresh=1), shouldRunPass returns true EVERY iteration - passesRun=2803/32min ~= 88 passes/min, no cadence throttle, no pain-budget gating; each pass ~5x18-21ms gotw scans + ~15ms STW walk ~= the user's 30%+ CPU burn. Ends either by retiring the candidate (C: (klass,tid) qualification, no walks - the [B tagged instances are 24-16KB machinery byte[]s with stable per-site retention) or by making the marker findable (the disjoint-set lottery). C is the CPU-burn fix, not just signal quality | [root-cause, cpu-burn, canary, shouldRunPass, pod-logs, NEW-THIS-SESSION]- find-engine-seam-data-race | Seam-driven runPass/poll on a test thread raced the BFS thread on the engine's unsynchronized maps (_class_tags, _candidate_*, _leak_parent_fanout) -> SIGSEGV in ClassTagTable rehash. Fixed: `_engine_lock` serializes both drivers (runPassSerialized/pollWatchedTargetsSerialized). Residual: threadLoop still interleaves BETWEEN seam steps (SetTag/discovery-order races) - deterministic fixtures must not depend on discovery order | [root-cause, fix, threading, crash, test-seams, NEW-THIS-SESSION] +- find-isqueuedforrotation-quad-scan | isQueuedForRotation() was O(PRIORITY_EXPAND_CAP) per frontier-SLOT visit (~199k slots) = ~200M comparisons per rotation pass - its old "sub-millisecond" comment assumed per-selection calls, not the collectors' per-slot calls; user saw it in profiles. Fixed with PriorityExpandSet: fixed 2048-slot open-addressing index, no allocation after construction, push=insert / drain=rebuild-from-deque (no tombstones), O(1) contains. All mutation under _engine_lock | [fix, root-cause, performance, profiles, NEW-THIS-SESSION] +- ev-leaktag-onpod-round4 | Pod round 4, build c5490156e (option C), 13 min: per-tid gate confirmed - 183 klass-trend candidates rejected for no qualifying tid, exactly one candidate (klass 2 [B, tid 120852, 75MiB byte[]s ages 4-18) gets ALL pool tags (round 3: 247 tags over machinery byte[]s). Interception still zero: gotw floor 14-41ms at a 242k-entry tag map exceeds the ~10ms expand window -> batch ratchets to 8, ~120-200 objects/min drain vs a 126,895-entry pendingExpand backlog = ~10-17 hours per lap; leak holder never expanded post-tagging so the tagged byte[]s' incoming edge is never enumerated. Code-confirmed: the targeted leak-accumulation rotation tier selects ZERO every pass because its Tier-2 filter requires EXPANDED parents while the growing holders are un-expanded FRONTIER-state backlog entries (51k known parent candidates idle). Pass rate dropped 88/min -> 14/min (decomposition needs a TEMP timing log) | [pod-verification, option-C, per-tid, crawl-throughput, NEW-THIS-SESSION]- find-per-tid-qualification-design | Option C implemented: selectLeakCandidates() requires a qualifying ALLOCATING thread (per-klass TidTrend rings folded from existing scratch, 8 tids x 16 slots, no JVMTI calls); a tid qualifies by sustained age-trend OR retained-count bar >= 8 tracked instances (the OR covers one-cohort-per-thread accumulation - LeakingCache shape - that the trend gate structurally misses); tagLeakInstances() tags only qualifying tids' instances. POD IMPLICATION: hotdog has a deliberate simulated-memory-leak thread (age_count 2->3 rising), so C SCOPES the pod tagging to the leak thread rather than retiring the candidate - burn relief must come from interception becoming possible. Seam: seedTidTrendSample0 with REAL allocating tids (getTid() on the leaking thread); synthetic trends exempt from fold decay. 543 gtests + slow suite 8/8 green, correlation-found through the per-tid-scoped path | [fix, design, livenessTracker, leak-tag, option-C, NEW-THIS-SESSION]- find-depth0-durable-root-upgrade-gap | PRODUCTION: maybeUpgradeRootAttachedRootKind was wired only into heapRootCallback; the static sweep's class->field edges are ROOT-LIKE (parent_tag==0, class referrer negative) and fell through improveChain/reparentToDurableRoot (both need parent!=0) - an object first admitted via a stack local kept its transient classification forever, and the real static-retained depth-0 chain was noise-gated. Fixed: root-kind upgrade for root-like edges in heapReferenceCallback + chain-cache invalidation | [root-cause, fix, root-kind, gate, durability, NEW-THIS-SESSION] +- find-canary-lane-backoff-design | Option A (user point 3): the canary lane's unbounded back-to-back chase (pod: 1 core for 32 min, ~88 passes/min; the doc's tiny-cost claim only held for findable candidates) replaced by a work-scaled backoff - spacing = multiplier x EMA(pass wall), mult 1 (gate off) doubling per no-candidate-progress pass to cap 16, reset on candidate progress, OOM ramp overrides, GC-epoch deliberately does not bypass; pain-budget refill flattened to 100x-while-open (double-throttle guard, covering/emergency split deleted). Work-scaled because a fixed cap binds only when > pass duration: pod passes ran 0.7-4s (1s cap = no-op) while a 1s cap starved the local deep chase (20-30ms passes, ~200 passes needed). Burn bound structural: <=1/16 core at cap. 546 gtests + deterministic seeded-EMA backoff gtest; ToGcRoot green at load 6; doc ReferenceChains-SignalsExplained.md sections 4/8/11 updated (points 1+2 clarifications too: no per-GC wake, epoch-bypass caveat, cadence dual role) | [fix, design, canary, cpu-burn, work-scaled, NEW-THIS-SESSION] +- find-default-live-samples-ratio-lottery | The intermittent "sampler-dead" suite failures (leak-correlation children with zero leak tagging, table near-empty, tag-out-of-pool) root-caused: memory=64:l without an explicit ratio leaves _live_samples_ratio at its DEFAULT 0.1 (arguments.h) and LivenessTracker::track() drops 90% of tracked instances probabilistically - the scenario's ~35-50-chunk leak cohort is a per-run Bernoulli lottery (passing runs' stable tagged=5 = exactly ~10% of ~50 chunks; the scenario's own "every allocation is tracked" assumption was never true). Instrumented every drop path (diagnostics since fully reverted), then fixed the TEST to pass memory=64:l:1.0. Post-fix: zero leak-correlation failures across suite runs. Pod unaffected (continuous leak volume; the ratio just subsamples a big population). Remaining flaky family is separate: ToGcRoot/UnboundedCache chase timing under machine load 25-55 | [fix, livenessTracker, tests, flakiness, sampler, NEW-THIS-SESSION]## Questions +- q-togcroot-acceptance-paths | OPEN: the remaining ToGcRoot/UnboundedCache slow-suite flakiness is NOT the ratio lottery (fixed; rep-died symptom gone at :l:1.0) and not pure machine load (latest fail at load 3). Confusing data on the backoff build: one green run PASSED with 13 passes and ZERO canary prunes (chain arrived via a non-canary path), while fail runs show 114-200 passes, hundreds of Tier-2 holder selections, and the marker NEVER pruned. Candidate factors: the work-scaled backoff's EMA polluted by ~300ms root-enum passes inflating chase spacing 10x; Tier-2 selections possibly not covering the marker rep's actual parent signature; and which non-canary path served the green run. Next: per-pass EMA split (root-enum vs chase) + a Tier-2-selection-to-marker-tag tie diagnostic in one failing run | [flakiness, tests, canary, pacing, open, NEW-THIS-SESSION]- q-implement-two-fixes | RESOLVED: yes — both fixes were implemented, plus a third (Fix C) | [decision-made, fix, implemented] | CONFIRMED +- find-togcroot-orphaned-slot-stranding | ANSWERS q-togcroot-acceptance-paths: the ToGcRoot 124-pass mystery was slot stranding, not pass starvation. The walk admitted only 8 of 69,001 ChainLinks (66/124 passes zero-edge; holder never re-walked - fanout ranking drowned by 97k noise edges, all 821 Tier-2 selections went to noise); the seeded candidate aged out of the poll list (ring_fill=22, slope~0.5, consecutive_positive=0) the pass BEFORE the 8 discoveries were recorded; the discovered-chain loop iterates only CURRENT poll candidates, so the persistent slot ("can still be found there") was never read again - chains never built. Fix 1 (load-bearing, user-picked): buildDiscoveredInstanceChains() slot-driven + orphan sweep over slots absent from the poll's candidates; gtest OrphanedSlotBuildsDiscoveredChainsAfterCandidateDropsOut; suite green at load 28.7 AND 53 (was failing at load 3). Fix 2: persistent allocator thread, tid seeds on its tid; candidate still ages out (54 zero-cand polls) - shared-JVM epoch noise resets consecutive_positive, fix 1 remains the guarantee; fix-2 improvement is the active thread | [fix, root-cause, canary, slots, flakiness, NEW-THIS-SESSION]- find-admission-boost-implementation | Option A (user-picked) implemented: LivenessTracker::admitForTracking(tid) is track()'s admission gate with two raises over the default 10% live-samples ratio - (1) watched tids: selectLeakCandidates() qualifying_tids published by noteSelectedCandidates() from RCT's full poll, admitted at 100%, cleared poll-by-poll (zero-candidate polls clear; not called from hasLeakSignal's max=1 probe), bounded by candidate threads' own allocation rate; (2) urgency: setUrgentTracking(urgent) every threadLoop iteration next to _oom_ramp_active - admits everything (OOM endgame). Fail-open by construction; two-phase count+array publish RELEASE/ACQUIRE (arm64 rule); boosted admissions skip the RNG draw; cleared on fresh start(). Rejected: global-only (10x table cost whole-process during chase) and klass-scoped via jclass (GC-move-unsafe identity; per-tid already covers tagLeakInstances' tagging scope). USER SCOPE CORRECTION (accepted): the default 10% does NOT compromise detection at any ratio - the subsample scales the signal, it does not gate it (a heap-filling leak leaves a proportional population + positive trend every epoch); only detection LATENCY is affected; cohorts small enough to zero out are below every machinery threshold anyway. 4 new gtests (deterministic via admissionResetForTest RNG reset), 550 green; slow suite: boost engaged in children, ToGcRoot unchanged (its own family); doc section 5 extended | [fix, design, livenessTracker, admission, work-scaled, NEW-THIS-SESSION]- q-canary-stuck-fix-alternatives | RESOLVED: user chose C+B; implemented, compiles clean, gtest-verified (new regression test fails pre-fix/passes post-fix), not yet committed/deployed | [decision-made, fix-implemented, referenceChains, canary, research, NEW-THIS-SESSION] | CONFIRMED diff --git a/.investigations/missing-refchains-on-hotdog/STATE.md b/.investigations/missing-refchains-on-hotdog/STATE.md new file mode 100644 index 0000000000..de81131b66 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/STATE.md @@ -0,0 +1,375 @@ +# Current State + +## Active investigation + +Why the hotdog pod emits zero `datadog.ReferenceChain` events. + +The session ran in two phases and produced two different answers. + +**Phase 1 — RESOLVED, not a code bug.** The pod was running stock +ddprof-lib **1.48.1**, which predates all reference-chain work. + +**Phase 2 — real defects, progressively fixed.** Multiple root causes +found and fixed across sessions. Current state: BFS throughput is up +~1000x (3-4 → 1700-2800 edges/pass), 5 candidates appear, canary +representatives are re-tagged and all discovered instances auto-marked, +chains are cached per-instance (not per-class). JFR analysis confirmed 2 +ReferenceChain events emitted — but one was for a noise [B instance. + +## Corrections + +- (2026-09-16, round 19) `open(path, "w")` TRUNCATES the file BEFORE the + write argument is evaluated — a failed `write(list)` left INDEX.md + empty and the chained `git add` committed the wreckage (118 lines + lost, restored from git). Build the complete string first, write + once; never open-for-write before the content exists. Same family as + the atomic-multi-edit lesson: a failed step plus an unconditional + follow-up destroys state. + +- "Pre-existing gtest SIGSEGV UrgentOOMProjectionBypassesCandidateGate" + was MISLABELED — it was this investigation's own regression, hidden + from its own round notes by lazy "clean HEAD" testing (HEAD = this + branch's tip, all investigation commits). True timeline: the test + landed 5adb32831 (2026-08-12); autoTuneDefaults facdc70c0 (2026-08-17) + began calling GetAvailableProcessors from start(); the fixture's table + was never wired → the ONLY max-heap-setting test in the file null- + crashed at start() for 5 weeks — its real subject (OOM-urgency bypass + of the candidate gate) never executed. FIXED this round: wired + mock_GetAvailableProcessors (returns 1) — test passes its subject, + gtestRelease green with no exclusions for the first time since + 2026-08-17. LESSON: "reproduces on HEAD" proves "not from today's + diff", nothing more; bisect to the introducing commit before calling + anything pre-existing — and a crashing test is never "separate debt", + it is lost signal (this one carried valid assertions the whole time). + +## Current focus: round 20 COMPLETE — P0 harness fully live (de15e521f) + +All system invariants run in CI (128 tests / 127 pass / 1 skip-with-reason): +L1 leak chains + L2 canary resolution + L3 natural completion + L6 restart +hygiene (chains persist by design) + L8 dormancy + L4 log budget (live in +DEBUG-built test binaries only). The round-20 "candidate suppression" catch +is RETRACTED — it was the harness builder's own dangling jclass-identity +pointer (vector reallocation; find-candidate-presence-suppresses-static- +admission, now a retracted-deadend node with the fixture capacity contract +encoded as a permanent regression test). NEXT: P1 keyspaces audit + dead +marker-path retirement, P2 always-on per-search health line, P4 deferred +queue (sporadic batch coverage stays open for upstreaming). The pod remains +verification-only (no more pod-driven discovery rounds). + +Prior focus: STRATEGY SHIFT (round 19) — stop pod-driven whackamole; build the missing oracle + +Meta-review 2 (meta-whackamole-analysis, user challenge): goal achieved +(round 16) but rounds 17-19 are whackamole — 30+ fixed defects fall into +7 producing classes, 5 of which are invisible to the 348-test unit +suite by construction, so the staging pod (hours/iteration, n=1) is the +only end-to-end oracle and surfaces them in random order. THE PLAN +(design-pod-in-a-jar-harness, pending user approval): P0 pod-in-a-jar +system harness (real loop + mock topology + 8 invariants incl. LOG +BUDGET and dormancy; 1 session), P1 keyspaces audit + dead marker-path +retirement, P2 always-on per-search health line (exits-by-reason, +canary-resolution histogram, coverage ratio — pod verification becomes +one line), P3 reset-contract invariant test, P4 deferred queue triaged +to harness-flagged-only. NO more pod-driven fix rounds; the pod becomes +verification, not discovery. + +Prior focus (round 19, superseded by the shift): canary found-criterion +root-caused + fixed (788d7b2a7, pending pod verification on next +deploy — watch 'canary found:' + 1/1 + natural completion).: ROUND 16 VERIFIED ON POD — leak-correlated ReferenceChain events FLOWING (end goal reached) + +Build 4afe870a2 on pod cgdtx (JVM 07:45:11Z, wrapper klass 28516, leak +[B klass 5): fix A live at scale (self_edge_skips=30816+, wrapper +admitted root-attached parent=0 root_kind=8 and WALKED as anchor tag +2184 — the round-15 parent==self demotion is gone); fix B live +(quota_drops=6351, fifo_size=132 ≪ 1024 — the floods contained, 16 +drained/pass); FIRST leak-tag interceptions ever (12+, leak_tag ≥ +2^30 admitted with parent chains); leak-correlated chains cached +(auto-marked target_tag=1073742071/72/77 — leak-tag targets) and +re-emitted 6→14 sustained per dump. Also an accidental CONTROL run: +3h of the non-leaking app image → ZERO tracker activity — the +candidate gate keeps the machinery dormant on a healthy app. +Remaining watch: canary 0/1 (its representative not yet intercepted; +the round-15 chase deadlock is broken — walks + interceptions every +pass; exit expected as coverage reaches it). LOG FLOOD SOLVED (f70bcc454): +the round-12..16 TEST_LOG diagnostics are now runtime-gated +(rcDebugLevel.h) — level 0 SILENT by default (same debug .so, pod-safe), +1 = lifecycle/summary (per-pass/per-poll: candidates, canary, rotation +counters, re-emit, leak-tag correlation), 2 = full per-object firehose. +Env DD_PROFILING_REFERENCE_CHAINS_DEBUG=N at start; runtime override +(no restart): kubectl exec … -- sh -c 'echo N > +/tmp/ddprof_root/refchains_debug_level' (single digit, re-checked ~1s +by the RC thread loop; remove the file to fall back to env). The TEMP +removal list below is superseded: diagnostics KEPT but gated; slimming +for upstream is a later round. Node: find-refchains-log-flood-configurable +(tier empirics, the callback-context constraint, the two build/workflow +lessons + the two pod-verified escapes: header-inline TEST_LOG sites +bypass the TU macro gate — fixed 7aa68744b by moving buildChainEvent/ +buildCanaryChainEvent out of line; tick-cadence lines demoted 524c35325 +— level 1 is transitions/bounded outcomes only, verified <500/min). +CANARY 0/1 ROOT-CAUSED + FIXED (788d7b2a7, +find-canary-found-criterion-unmigrated): the marker->leak-tag migration +never migrated the found criterion — found bits set only in the dead +marker path; the rep path checked/built by leak tag (never a frontier +key). Moving-target hypothesis REFUTED (stable tag across 203 ticks). +Fix: leak-tag-target chains mark slots found (any-instance criterion — +the per-instance guarantee lives in the leak-tag correlation); rep +path resolves by the slot key. Pending pod verification on the next +deploy: 'canary found:' + 1/1 + clean search-restart cycle. Separate +upstreaming topic observed: subtree coverage is sporadic at batch scale +(5-min window: wrapper not walked, zero new interceptions, 34 cached +chains). See ev-leaktag-onpod-round16-results.md, +find-wrapper-demotion-self-parent.md (FIXED+verified), +find-round16-endgoal-verification.md (the milestone + the control +run), find-urgentoom-null-fn-mislabel.md (the same round's test-infra +root cause: the 5-week null-fn crash + the 'pre-existing' mislabel +lesson). + +Prior focus: round 15 verified (first wrapper walk, first events); +round-16 gap measured+fixed (demotion guard, FIFO quota, UrgentOOM +root-cause — all in 4afe870a2, gtests 118/118 no exclusions). + +Prior focus (round 13 prepared): fix B+C verified working mechanically on +the pod (ccdb03b89), but the LEAK_BUFFER wrapper is STILL never walked; + +Design/review/implement/review loop run for B' (user-picked). Two +load-bearing discoveries during the loop (both in +find-already-admitted-blocks-unreachable + find-anchor-live-feed-design): +1. The "sweep re-laps every pass" feed assumption was FALSE — the sweep + gate re-laps only while the class count is in flux → B' needed TWO + push sites (demotion time + sweep time), not one. +2. BOTH push sites sat in a pre-existing DEAD NEST: 57aec4895 misplaced + improveChain inside heapReferenceCallback's first-admission block, + where improveChain/reparentToDurableRoot/maybeUpgradeRootAttachedRootKind + were all guaranteed no-ops for already-admitted entries (since + 2026-08-28). This rewrites the eviction mechanism: improveChain + demotion never fired on any pod round; the holder was born + chain-attached and the upgrade path was dead. Fixed with a real + `else if (*tag_ptr > 0)` arm — the enabling fix for B'. + +Verification: 112 gtests green (4 new deterministic B' tests + no +regressions from the now-live re-attribution), full gtestDebug green, +ddprof-test *ReferenceChain* family green. ONE pre-existing failure on +clean HEAD (proven via git stash, 3/3 runs): +AggressiveLeakReferenceChainTest.shouldOpenSearchGateOnAggressiveHeapWideGrowth- +WithNoLeakCandidate — urgent-OOM-gate test unrelated to B'; needs its own +investigation (user flagged). spotlessApply clean. NOT COMMITTED. + +Round-6 verdict made Option C a correctness requirement (breadth- +first FIFO over a rising heap can never drain; pendingExpand net-growing). +User decision: implement BOTH prongs, taxonomy-driven, explicitly NOT +shaped by probing hotdog's simulator ("we don't want to overfit this +particular scenario"). Design + implementation in +find-option-c-descend-walk-design (unified bounded descend-walk +mechanism, both prongs reuse heapReferenceCallback's whole admission +chain so interception = complete chain in one bounded STW): +- Prong 1 walkCandidateThreadLocals: qualifying tids' Thread objects, + anchor-gated to ThreadLocalMap, no-descend set (ClassLoader/ThreadGroup/ + ProtectionDomain), own deadline slice before the static sweep. +- Prong 2 collectStaticFieldAnchorsForRotation + walkStaticFieldAnchors: + root-attached STATIC_FIELD holders, wrapping cursor, batched GOTW + resolve, descend walks at the head of rotation's slice. +- tid->jthread registry: onThreadStart/onThreadEnd hooks PLUS a + one-time registerExistingThreads() sweep at Profiler::start() (a leak + thread is typically alive since before the recording - the first + ThreadLocalLeakScenario run caught this: walked=0 with the thread + unregistered; the sweep must live in profiler.cpp's lifecycle, NOT in + RCT::start(), or the JFR-roundtrip gtest's partial mock env crashes + on the null GetAllThreads slot). +Verification: 553 gtests green (incl. 2 new descend-walk gtests); +reference-chain slow family 9/9 green (incl. NEW +ThreadLocalLeakReferenceChainTest - the missing thread-local taxonomy +scenario, found correlation live with the thread walk engaging: +walked=1 edges=18); spotlessApply clean. + +## Next steps + +1. DONE this session: committed (186468437 descend-walk core + + gtests; 01c591eea ThreadLocal scenario; 93868362e memory sync) and + pushed to origin/jb/reference-chains-pi. +1b. DONE this session, committed+pushed: per-hop retention-edge field + names (4d473e727) + memory sync; pod round 7 verified both prongs + live with interception still zero (ev-leaktag-onpod-round7) and the + user-picked fixes implemented and pushed (c6635fe0e: DESCENT_HOPS + 6->16 + JNI_GLOBAL anchors in the rotation tier; 9d3d0afe6 memory + sync). One slow-suite failure en route was the known intermittent + hysteresis family (candidate never qualified; green on rerun). +2. DONE (previous session, checkpointed now): pod round 8 verified on + RENAMED pod prof-analyzer-hotdog-jb1-668df5bcff-f75l8 (JVM 4445, + agent 1.66.0-SNAPSHOT~e188d0ff7f): both prongs live, walks + un-truncated at 16 hops (per-pass edges 10→3608, rotation cycling + different anchors), tagging healthy (klass_id=5 tid=4655 tagged=8 + max_size=78MB), interception STILL ZERO — and the app's retention + shape was found in its bytecode: ProfileAnalyzer.LEAK_BUFFER, a + static final List wrapped in Collections.unmodifiableList, + 3-4 hops from the root static = textbook prong-2 (see + ev-leaktag-onpod-round8). Local scenario intercepts the identical + shape, so the machinery is sound in-process — the suspect is WHICH + anchors enter the anchor tier (wrapper never admitted root-attached / + root_kind misclassified / root-attached entry replaced by a + chain-attached one via improveChain). +3. DONE this session: round 9 verified (ev-leaktag-onpod-round9). + Anchor diagnostic answered: tier = 76 machinery statics, NO holder, + zero app classes. UPLOADED recordings contain 12 ReferenceChain + events with working edge names + HeapLiveObject events with leakTags + on the 78MB leak chunks — the ORIGINAL zero-events question is + RESOLVED; the machinery cohort's retention is fully explained + (Mac/HmacCore ThreadLocals, charset constants). Local pod chunks are + a bad source for dump-time events (q-heapliveobject resolved: use + uploads only; jfr print crashes on ReferenceChain — use JMC API). + ROOT CAUSE of the remaining gap isolated in code: + find-anchor-holder-eviction (parent_tag==0 is unidirectional; + improveChain evicts root-attached holders; re-root refused at + referenceChains.cpp:2376). +4. DONE this session: B' implemented, committed (6f3c6cc2e), deployed, + and round-10 VERIFIED live (see ev-leaktag-onpod-round10 + the round-10 + blocker chain). Round-10 response (user-picked A+B) committed and + pushed: 8888e6d42 (TEMP CANARY_NO_PROGRESS_PASS_LIMIT 3→30 revert — + the suspected restart driver) + 04539b821 (collector-first anchor + ordering — the collector's root-attached cohort no longer starved by + the cap-pinned at-risk flood; truncated passes fall on the FIFO suffix + which the requeue path protects). +5. DONE this session: round 11 verified (ev-leaktag-onpod-round11): A + WORKED (search lifetimes up; cycle_complete=1 REPEATEDLY — sweep laps + complete for the first time ever on this pod), B is a production no-op + (real GetObjectsWithTags returns unspecified order — the starvation + model was mock-order-based; requeue still works, order-independent). + The candidate flapped out once (slope=-47.3 while the heap marches to + OOM — suspected self-reinforcing dropout via boost-clearing) then + RE-QUALIFIED on its own (~20:45 UTC). DECISIVE: at 21:10 UTC the + wrapper WAS walked with ZERO interception → "not admitted/not + eligible" REFUTED. Round-12 diagnostic prepared+committed+pushed + (8ca24a524: holder_class naming via GOTW on referrer_class_tag + + admission-sequence trace (klass_id x fresh/leak/already, 48 entries) + + per-anchor walk outcome line). + NEXT: user deploys 8ca24a524 → round 12. READ THE DIAGNOSTIC — the + three outcomes and their fixes: + (a) holder_class != ProfileAnalyzer → every walked wrapper is a decoy; + the leak wrapper is never selected → collector-lottery problem → + fix = raise STATIC_ANCHOR_ROTATION_BUDGET (4→16, nearly free under + the GOTW floor) or prioritize by holder size. + (b) holder_class = ProfileAnalyzer + walk outcome edges=0 truncated=1 + → budget starved before the descend → fix = per-anchor budget + reservation or anchor-count reduction. + (c) holder_class = ProfileAnalyzer + edges>0 + [B entries seen_as=0 + (fresh) → chunks enumerated but UNTAGGED at walk time → tag-lifetime + problem (tagLeakInstances vs walk timing) → a wholly different fix. + Fallback still queued: option C (sweep-cursor persistence across + restarts) — only if restarts remain the blocker after A. + ALSO: investigate the pre-existing AggressiveLeak urgent-OOM-gate + failure (3/3 on clean HEAD). + SUPERSEDED: user deploys 04539b821 → round 11: watch for (a) restarts spacing + to ~30+ min (limit 30) and the sweep reaching cycle_complete=1 with + cursor past ProfileAnalyzer's class, (b) the LEAK_BUFFER + UnmodifiableRandomAccessList wrapper admitted root-attached STATIC + and walked by the collector picks, (c) `leak-tag intercepted` → the + first LEAK chunk's chain (static_field → ... → byte[]). Fallback if + restarts persist: option C (sweep-cursor persistence across + restarts) — NOT yet designed. ALSO: investigate the pre-existing + AggressiveLeak urgent-OOM-gate failure (3/3 on clean HEAD). +4b. SUPERSEDED: user picks the anchor-eviction fix. Original options A/B were + re-evaluated after a standards survey (this session, see + find-attribution-standards-survey + find-anchor-live-feed-design): + **A refuted** (freezes one attribution where every standard system + re-derives or queries — JFR enumerates its root set fresh at every emit, + source-verified; MAT computes paths on demand; dominator-tree + attribution is offline-only per LeakCanary docs; dynamic-SSSP theory + says exact incremental path labels are hopeless), **B minimal-diff + fallback**, **B' (live feed) recommended**: the sweep's static-edge + callback pushes frontier-present-but-chain-attached holder tags into a + bounded FIFO; walkStaticFieldAnchors drains it; anchor selection stops + reading parent_tag/root_kind → eviction structurally impossible. + Verify the at-risk filter shrinks the population (TEMP counter) before + sizing the FIFO. Then implement + gtest + deploy → round 10: watch + `leak-tag intercepted` and the first LEAK chunk's chain + (static_field → ... → byte[]). +5. TEMP reverts before finalizing (list below). + +## TEMP — MUST REVERT before finalizing + +- ~~`CANARY_NO_PROGRESS_PASS_LIMIT` 30 → 3~~ REVERTED this session (commit + `8888e6d42`) — round 10 showed the TEMP value 3 abandons an unfound + canary every ~3 min at 1 pass/min, and each restart resets the sweep + cursor so the lap never reached the leak holder's class. +- Temp diagnostics still in code (kind_counts, gotw logs, blocking logs, + discovered-loop logs) — remove before production. +- TEST_LOG in `maybeUpgradeRootAttachedRootKind` (upgrade attempts) — + added this session. +- TEST_LOG `static_sweep_gate` in `runPassManualWalk` (per-pass sweep gate + decision) — added this session. +- Per-tag TEST_LOG in `tagLeakInstances` RETIRED BY DESIGN (round-5 + follow-up): replaced by the per-poll per-(klass, tid) summary line + ("tagLeakInstances summary klass_id=... tid=... tagged=... need_set=... + min_age/max_age/max_size") - the per-instance flood rotated the pod's + 10MB container log inside a verification window. Still TEMP: + `fanout-insert` log in `trackLeakAccumulation`, + `requeueChainRootForRotation` log — from the repro round. +- Old marker-tag decode in heapReferenceCallback is now unused — confirm + and remove. +- TEMP per-anchor diagnostic in walkStaticFieldAnchors (c9a57f681): + class signature + chain shape per walked anchor — remove once round 9 + names the tier-membership answer. KEEP for round 10 (it names the + at-risk anchors B' now feeds into the walk). +- TEMP (round 12, commit 8ca24a524, deployed tomorrow): wrapper-class + anchor trace - holder_class via GOTW on referrer_class_tag + the descend + walk's admission sequence (klass_id x seen_as, 48 entries) + per-anchor + walk outcome (edges/truncated). Round-11 watch data that shaped it: at + 21:10-21:12 UTC the wrapper WAS walked (2 lines) with ZERO interception - + so "not in the eligible set" is REFUTED; the diagnostic must answer + whether the wrapper walk enumerates list -> elementData -> chunks at + all, and whether enumerated chunks carry leak tags at that moment. + Remove once the pod answers the wrapper question. +- TEMP (B', this session): static_anchor_fifo_size/drained/pushed_total + fields in runPassManualWalk's rotation_candidates TEST_LOG — remove + after round 10 sizes the at-risk population. + +## Confirmed findings (do NOT re-derive) + +1. **`find-marker-tag-slot-index-mismatch`** — slot decode bug, FIXED. +2. **`find-one-shot-pretag-gate`** — pre-tagging one-shot, FIXED. +3. **`find-canary-search-cannot-terminate`** — termination/livelock, PARTIALLY ADDRESSED. +4. **`find-abandon-event-lost-to-dump-sampling-race`** — transient state race, FIXED (queue). +5. **`find-cpu-pain-budget-starves-canary-passes`** — pain budget starvation, FIXED (4x escalation). +6. **`find-canary-stuck-restart-wipes-frontier`** — frontier wipe on restart, FIXED (C+B). +7. **`find-static-field-sweep-never-completes`** — sweep never completes, FIXED (resumable cursor). +8. **`find-candidate1-never-tagged`** — CP edges burning sweep budget, FIXED (per-class quota). +9. **`find-sweep-completes-but-bfs-starved`** — sweep works but BFS can't reach entries. ADDRESSED by adaptive batch_size + deadline split. +10. **`ev-timing-split-callback-vs-jvmti`** — our callback 5-17%, JVMTI 83-92%. Deferring won't help. +11. **`find-getobjectswithtags-quadratic-bottleneck`** — GetObjectsWithTags O(tag_map × batch). FIXED: adaptive batch_size. +12. **`find-shared-deadline-starves-expand`** — shared deadline ate expand's time. FIXED: per-sub-op deadline reset. +13. **`find-rolling-resume-expandfrontier`** — truncated batch re-walked. FIXED: rolling resume cursor. +14. **`find-representative-changes-lose-canary`** — representative LRU-evicted, canary lost track. FIXED: re-tag + auto-mark all instances. +15. **`find-canary-continue-skips-discovered-instances`** — canary `continue` skipped discovered check. FIXED: removed `continue`. +16. **`find-cpu-pain-budget-blocks-bfs`** — cpu_pain_budget silently blocked RUNNING state. DIAGNOSED: diagnostic added. +17. **`find-per-class-caching-blocks-instances`** — per-class caching blocked all but first instance. FIXED: per-instance caching. +18. **`find-age-heuristic-insufficient`** — age heuristic picks noise over leak. DIAGNOSED: allocation-site clustering proposed. +19. **`find-lambda-fragments-calltrace-id`** — lambdas fragment call_trace_id. FIXED: switched to tid-based clustering. +20. **`q-dominant-gens-still-one-with-tid`** — RESOLVED: per-thread tracking works, dominant_gens=1 was from non-leak classes. +21. **`find-ages-vector-not-cleared`** — ages vector inflated across epochs. FIXED: ages.clear(). +22. **`find-already-admitted-blocks-deeper-chain`** — ALREADY_ADMITTED blocks deeper chain. FIXED: improveChain(). +23. **`find-holistic-design-issues`** — 4 design problems (wrong objects, no correlation, dead chains, flat 100× CPU). Redesign approved. +24. **`find-leak-tag-pool-implementation`** — redesign A-D implemented (294f09ff3..1ce2b4f03), gtest pass, NOT yet on-pod verified. +25. **`q-coverage-tracking-per-combination`** — coverage is per-object; user wants per-(call_trace_id, tid). Refine after on-pod verification. +26. **`find-ema-batch-collapse`** — round-1 regression: batch 400→2, passes 10.4s CPU. FIXED with AIMD + deadline check (0db70994d). +27. **`find-leaktag-jfr-field-misalignment`** — leakTag parsed from attribute byte. FIXED (0db70994d). Field-order invariant recorded. +28. **`find-priority-queue-starves-bfs-crawl`** — round-3 root cause chain (priority flood + slot exhaustion + tag overwrite + dead growth tier). FIXED (f4c73ba0f). Key insight: a cap alone does NOT fix starvation — a capped-but-pinned priority queue still never empties; fair-share alternation is what restores the pending drain. + +## Ruled out (do NOT re-investigate) + +- **Warm-up / needs more time** (`hyp-warmup-transience`). +- **Regression of the five fixes** (`hyp-regression-of-five-fixes`). +- **`jcmd JFR.dump` as evidence** (`dead-jcmd-jfr-dump-wrong-source`, `dead-jcmd-jfr-dump-wrong-source-v2`). +- **Toolkit `us1.prod.dog`** (`dead-toolkit-prod-datacenter`). +- **Hard reference_kind filter** (`dead-hard-reference-kind-filter`). +- **Deferring `admitObject` out of safepoint** — would save only 5-17%. + +## Reproduction handle + +Pod `prof-analyzer-hotdog-jb1-668df5bcff-7h5n9` in `profiling-stg`, +container `prof-analyzer`, JVM 77972 (round 9 verified on c9a57f681, see +`ev-leaktag-onpod-round9`). Pod clock is UTC (2h behind local). +Verify the deployed build via a per-iteration log field (e.g. +`ema_call_ms=` in every gotw line) BEFORE interpreting event-driven +logs — round 3 saw one stale-build deploy (JVM 44624 ran the old +build; the redeploy as JVM 48355 had the right one). + +JFR: use `kubectl cp` from `/tmp/ddprof_root/pid_XXX/jfr/` (NOT jcmd). +Or use profiling toolkit `download.py` for uploaded profiles. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-adaptive-batchsize-onpod-verification.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-adaptive-batchsize-onpod-verification.md new file mode 100644 index 0000000000..491d7f192b --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-adaptive-batchsize-onpod-verification.md @@ -0,0 +1,69 @@ +--- +id: ev-adaptive-batchsize-onpod-verification +type: evidence +source: pod logs (kubectl logs) +collected: 2026-08-27 +tags: [pod-logs, on-pod, post-fix, live-verification, adaptive-batchsize, gotw, ema, NEW-THIS-SESSION] +--- + +# Adaptive batch_size confirmed live on-pod + +Pod `prof-analyzer-hotdog-jb-c944876b9-q8vd8`, PID 263646, build `8f69683f9`. + +## Warmup phase (EMA converging from default 64) + +``` +batch_size=58 resolved=57 edges=3 gotw_ms=7 ema=107486 +batch_size=1 resolved=1 edges=4 gotw_ms=2 ema=553786 +batch_size=1 resolved=1 edges=5 gotw_ms=2 ema=884800 +batch_size=1 resolved=1 edges=6 gotw_ms=2 ema=1127789 +batch_size=1 resolved=1 edges=7 gotw_ms=1 ema=1288262 +batch_size=19 resolved=19 edges=7 gotw_ms=2 ema=1061539 +batch_size=23 resolved=23 edges=9 gotw_ms=3 ema=876599 +batch_size=28 resolved=28 edges=61 gotw_ms=3 ema=724782 +batch_size=34 resolved=34 edges=65 gotw_ms=3 ema=600752 +batch_size=41 resolved=41 edges=0 gotw_ms=10 ema=531578 +batch_size=47 resolved=47 edges=8 gotw_ms=8 ema=459475 +batch_size=54 resolved=54 edges=35 gotw_ms=8 ema=400035 +batch_size=62 resolved=62 edges=49 gotw_ms=9 ema=349554 +batch_size=71 resolved=71 edges=116 gotw_ms=8 ema=303889 +batch_size=82 resolved=82 edges=352 gotw_ms=8 ema=262992 +batch_size=95 resolved=93 edges=0 gotw_ms=10 ema=231582 +batch_size=107 resolved=105 edges=2 gotw_ms=11 ema=206435 +``` + +EMA spikes from initial gotw_ms=2 on batch_size=1 (cost_per_tag = 2ms/1 = +2M ns), then recovers as larger batches show lower per-tag cost. + +## Steady state (EMA converged to ~62k ns/tag) + +``` +batch_size=417 resolved=417 edges=0 gotw_ms=25 ema=60193 +batch_size=415 resolved=415 edges=263 gotw_ms=22 ema=59192 +batch_size=422 resolved=422 edges=503 gotw_ms=35 ema=63988 +batch_size=390 resolved=390 edges=0 gotw_ms=25 ema=64147 +batch_size=389 resolved=389 edges=37 gotw_ms=20 ema=61881 +batch_size=404 resolved=404 edges=0 gotw_ms=26 ema=62518 +batch_size=399 resolved=399 edges=3 gotw_ms=24 ema=62317 +batch_size=401 resolved=401 edges=227 gotw_ms=27 ema=63628 +batch_size=392 resolved=391 edges=330 gotw_ms=26 ema=64219 +batch_size=389 resolved=388 edges=0 gotw_ms=28 ema=66038 +``` + +## Per-pass throughput + +``` +runPass done: edges_admitted=593 truncated=1 frontierSize=52724 +runPass done: edges_admitted=648 truncated=1 frontierSize=53372 +runPass done: edges_admitted=971 truncated=1 frontierSize=54343 +runPass done: edges_admitted=456 truncated=1 frontierSize=54799 +``` + +**Before fix**: 3-4 edges/pass. **After fix**: 456-971 edges/pass. +~100-200x throughput improvement. + +## Candidate status + +0 candidates at time of verification — JVM restarted (PID 263646), liveness +tracker needs warmup (`heapFloorRising=0`, `required_hysteresis=5`). +Leak still growing (`simulated-memory-leak: allocated 75 MB`). diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-candidate-count-latch-mismatch.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-candidate-count-latch-mismatch.md new file mode 100644 index 0000000000..a8430d3209 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-candidate-count-latch-mismatch.md @@ -0,0 +1,55 @@ +--- +id: ev-candidate-count-latch-mismatch +type: evidence +status: confirmed +depends_on: [ev-livelock-pod-logs] +supersedes: [] +related: [find-one-shot-pretag-gate, find-canary-search-cannot-terminate] +tags: [pod-logs, candidate-count, latch, pre-tagging] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Live proof that _candidate_count is latched while the selected set has moved on + +## Source +`kubectl logs -n profiling-stg prof-analyzer-hotdog-jb-c944876b9-f762h + -c prof-analyzer --since=40m` + +## Raw excerpts + +`selectLeakCandidates()` currently offers 5 candidates, every poll: + +``` +[TEST::INFO] ReferenceChainTracker::pollWatchedTargets candidate_count=5 +[TEST::INFO] ReferenceChainTracker::pollWatchedTargets candidate_count=5 +[TEST::INFO] ReferenceChainTracker::pollWatchedTargets candidate_count=5 + ... (10 shown, all identical) +``` + +but the latched `_candidate_count` (the one `shouldRunPass()` prints) is +still 3, from the single pre-tag event: + +``` +[TEST::INFO] ReferenceChainTracker::shouldRunPass -> true (canary search, 0/3 candidates found) +``` + +and only one representative in the whole 25-minute window still carries a +marker tag: + +``` +$ ... | grep -oE "canary candidate\[[0-9]+\] klass_id=[0-9]+ marker_tag=-?[0-9]+" | sort -u +canary candidate[0] klass_id=8 marker_tag=-4611686018427387905 +``` + +The corresponding run-pass line independently reports 5 watched leak +klasses: + +``` +[TEST::INFO] ReferenceChainTracker::runPassManualWalk rotation_candidates root_kind_tags=0 leak_accumulation_tags=0 stale_expanded_tags=0 watched_leak_klass_count=5 leak_signatures=1 leak_parents=1 +``` + +No `"candidates pre-tagged with marker tags"` line appeared in the +40-minute window — pre-tagging ran once, before the log window, and the +`if (_candidate_count == 0)` gate (`referenceChains.cpp:3375`) has kept it +from running again. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-deadline-split-onpod-verification.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-deadline-split-onpod-verification.md new file mode 100644 index 0000000000..48d87ac67d --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-deadline-split-onpod-verification.md @@ -0,0 +1,45 @@ +--- +id: ev-deadline-split-onpod-verification +type: evidence +source: pod logs (kubectl logs) +collected: 2026-08-27 +tags: [pod-logs, on-pod, post-fix, live-verification, deadline-split, rolling-resume, NEW-THIS-SESSION] +--- + +# Deadline split + rolling resume confirmed live on-pod + +Pod `prof-analyzer-hotdog-jb-c944876b9-q8vd8`, PID 286999, build `b2acdaee2`. + +## Per-phase breakdown (after fix) + +``` +static_field_phase edges_admitted=322 truncated=1 +expand_phase edges_admitted=946 truncated=1 remaining_budget=3102 +rotation_phase edges_admitted=1138 truncated=1 rotation_budget=2444 +runPass done: edges_admitted=2537 truncated=1 frontierSize=121565 + +static_field_phase edges_admitted=377 truncated=1 +expand_phase edges_admitted=1055 truncated=1 remaining_budget=3047 +rotation_phase edges_admitted=691 truncated=1 rotation_budget=2280 +runPass done: edges_admitted=2218 truncated=1 frontierSize=123783 + +static_field_phase edges_admitted=320 truncated=1 +expand_phase edges_admitted=1606 truncated=1 remaining_budget=3104 +rotation_phase edges_admitted=742 truncated=1 rotation_budget=1786 +runPass done: edges_admitted=2766 truncated=1 frontierSize=126549 +``` + +## Before/after comparison + +| Phase | Before deadline split | After deadline split | +|-------|----------------------|---------------------| +| static_field_phase | 0-11 edges (ate entire deadline) | 252-377 edges | +| expand_phase | 0-1 edges (no time left) | 793-1606 edges | +| rotation_phase | 0-7 edges | 578-1138 edges | +| **Total** | **0-21 edges/pass** | **1708-2766 edges/pass** | + +5 candidates with `heapFloorRising=1`. Frontier growing from 92k → 179k +at ~2000 edges/pass. Canary candidates (klass_id=145, 211, 2292) have +marker tags but `buildCanaryChainEvent -> 0` (not yet reached by BFS). +[B candidate has `tag=0` (representative changed — see +find-representative-changes-lose-canary). diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-deployed-so-1481-no-symbols.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-deployed-so-1481-no-symbols.md new file mode 100644 index 0000000000..719bb5f2ab --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-deployed-so-1481-no-symbols.md @@ -0,0 +1,97 @@ +--- +id: ev-deployed-so-1481-no-symbols +type: evidence +status: confirmed +depends_on: [] +supersedes: [] +related: [find-refchains-not-deployed, find-onpod-evidence-methodology] +tags: [hotdog, deployment, native-lib, md5, ddprof-1.48.1, on-pod] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Phase 1: the deployed libjavaProfiler.so is ddprof-lib 1.48.1 with zero reference-chain symbols + +## Source +Pod `prof-analyzer-hotdog-jb-c944876b9-f762h`, namespace `profiling-stg`, +container `prof-analyzer`, JVM PID 231 (pre-resync). All commands run via +`kubectl exec` on the pod itself. + +## Raw excerpts + +Loaded library, from `/proc/231/maps`: + +``` +/tmp/ddprof_root/pid_231/scratch/libjavaProfiler-dd-tmp927042133104699179.so +``` + +Symbol scan on the loaded scratch copy: + +``` +=== symbols mentioning ReferenceChain / LivenessTracker in .so === +0 # strings ... | grep -icE 'ReferenceChainTracker' +20 # strings ... | grep -icE 'LivenessTracker|liveheap' +=== jar: reference chain native methods === +0 # pollReferenceChainTargets0 +``` + +Extracting the jar-bundled `.so` on the pod with `jar xf` (the pod has +`/usr/bin/jar`, no `unzip`): + +``` +jar xf /usr/local/app/agent/dd-java-agent.jar shared/META-INF/native-libs/linux-x64/libjavaProfiler.so +-rw-r--r-- 1 root root 1367456 Feb 1 1980 libjavaProfiler.so +=== md5 === +794906a03568ff284c2cb557af693e22 shared/META-INF/native-libs/linux-x64/libjavaProfiler.so +=== referencechain strings === +0 +=== compare vs scratch-loaded copy === +794906a03568ff284c2cb557af693e22 /tmp/ddprof_root/pid_231/scratch/libjavaProfiler-dd-tmp927042133104699179.so +``` + +Version strings from the jar manifest and the bundled `.so`: + +``` +Manifest-Version: 1.0 +Implementation-Version: 1.65.0 # dd-trace-java +1.48.1 # embedded ddprof-lib version +``` + +Repo-side ancestry check: + +``` +$ git log --oneline v_1.48.1 -1 +c96ea85f7 [Automated] Release 1.48.1 +$ git merge-base --is-ancestor v_1.48.1 jb/reference-chains && echo yes || echo no +no +$ git log -1 --format=%cd v_1.48.1 +Tue Aug 4 10:30:27 2026 +0200 +``` + +Branch-only reference-chain commits (never in any release): + +``` +4993cb6c3 Add reference-chains architecture and design docs +6f1fd14e7 Add Java integration tests and chaos/repro harness for reference chains +1a3165d48 Add C++ unit tests for reference-chain tracking +dc8071dc9 Implement reference chains for surviving live-heap samples +5935c5cfa Add repro/sweep tooling and chaos-harness build wiring for reference chains +``` + +Deployment context: + +``` +image: 727006795293.dkr.ecr.us-east-1.amazonaws.com/prof-analyzer-hotdog:v130436965-4aea7d55-amd64 + @sha256:4e4db006b76f0c25b29bbd1b0165fcac73f9b48c5e479b54228e4f3810e52191 +DD_ENV=staging DD_SERVICE=prof-analyzer-hotdog DD_VERSION=v130436965-4aea7d55-amd64 +"version":"1.65.0~dd00372bdd" (DATADOG TRACER CONFIGURATION status log) +``` + +Pre-resync JVM command line had no reference-chain flag; only: + +``` +-javaagent:agent/dd-java-agent.jar +-Ddd.profiling.ddprof.liveheap.enabled=true +``` + +Pod log grep over 24h for reference-chain activity: `0` matches. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-fixes-compile-and-gtest-pass.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-fixes-compile-and-gtest-pass.md new file mode 100644 index 0000000000..bf89a8ac00 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-fixes-compile-and-gtest-pass.md @@ -0,0 +1,56 @@ +--- +id: ev-fixes-compile-and-gtest-pass +type: evidence +status: confirmed +depends_on: [] +supersedes: [] +related: [find-marker-tag-slot-index-mismatch, find-one-shot-pretag-gate, find-canary-stuck-abandon-detector] +tags: [build, gtest, verification, referenceChains] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Fix A/B/C compile cleanly and pass the existing referenceChains gtest suites + +## What was run + +``` +./gradlew :ddprof-lib:compileDebug -Pskip-gtest +``` +-> `BUILD SUCCESSFUL`, 69 C++ source files compiled. + +``` +./gradlew :ddprof-lib:gtestDebug_referenceChains_ut +``` +-> 89/89 tests passed, including `SearchRestartTest` cases exercising +`shouldRunPass`/candidate restart/urgent-bypass paths touched by Fix B/C. + +``` +./gradlew :ddprof-lib:gtestDebug_referenceChainJfrRoundtrip_ut +``` +-> 1/1 test passed (`ProducesValidStandaloneJfrWithChainEvent`), confirming +the `flightRecorder.cpp` `kReasons` table change didn't break JFR +serialization of `ReferenceChainAbandoned`. + +## Caveat — does not confirm the actual fix + +None of these 90 tests exercises **more than one** canary candidate, so +they cannot by themselves confirm Fix A's slot-decode correction or Fix +B's growing-admission logic against the specific multi-candidate scenario +observed on the hotdog pod (`hyp-regression-of-five-fixes`'s caveat +applies here too). This is exactly the sub-question flagged in +`q-implement-two-fixes` and still open: add a multi-candidate regression +test. All this evidence establishes is: the changes compile, and they do +not regress any existing single-candidate/urgent-bypass/pain-budget +behavior. + +## State at time of this evidence + +Uncommitted changes on `jb/reference-chains`, HEAD `8114019c2` (diverged, +not committed): +- `ddprof-lib/src/main/cpp/referenceChains.cpp` +- `ddprof-lib/src/main/cpp/referenceChains.h` +- `ddprof-lib/src/main/cpp/flightRecorder.cpp` + +No on-pod re-verification has been done — the hotdog pod's deployed `.so` +still predates these fixes. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-hotdog-trace-zero-runpass.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-hotdog-trace-zero-runpass.md new file mode 100644 index 0000000000..1a62d92942 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-hotdog-trace-zero-runpass.md @@ -0,0 +1,68 @@ +--- +id: ev-hotdog-trace-zero-runpass +type: evidence +status: confirmed +depends_on: [] +supersedes: [] +related: [find-cpu-pain-budget-starves-canary-passes] +tags: [pod-logs, on-pod, trace, runPass, pollWatchedTargets, canary] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# 45-second continuous trace: zero `runPass`, 1472 `pollWatchedTargets` + +## Command + +``` +kubectl logs -n profiling-stg prof-analyzer-hotdog-jb-c944876b9-f762h \ + -c prof-analyzer -f --since=1s > /tmp/hotdog_trace.log +``` +run for a fixed ~45-second wall-clock window, captured 25,865 lines. + +## Key counts + +``` +grep -c "runPass done" -> 0 +grep -c "pollWatchedTargets" -> 1472 +grep -c "LivenessTracker::selectLeakCandidates" -> 23925 +grep -c "shouldRunPass" -> 0 +grep -c "canAffordNewSearch" -> 0 +grep -c "searchState" -> 0 +grep -c "canary pruned" -> 0 +grep -c "canary: admitted" -> 0 +``` + +`pollWatchedTargets` sample output: +``` +[TEST::INFO] ReferenceChainTracker::pollWatchedTargets candidate_count=5 +[TEST::INFO] ReferenceChainTracker::pollWatchedTargets canary: klass_id=33319 qualifies but all 5 slots are occupied - not tracked this search +[TEST::INFO] ReferenceChainTracker::pollWatchedTargets candidate[0] klass_id=150 +``` + +## Interpretation + +`pollWatchedTargets()` is called unconditionally every `threadLoop()` +iteration (`referenceChains.cpp:801`, not gated on `should_run`), so its +1472 hits in 45s just confirm the thread is alive and looping at roughly +its expected cadence. The complete absence of any `runPass`-family log line +in the same window is the actual finding: `shouldRunPass()` is returning +false on every single iteration for the full 45 seconds. Since the loop's +own cadence sleeps cap out at ~1-2s per idle iteration (see +`find-threadloop-presleep-blocks-back-to-back`), a 45s silent stretch needs +a longer-lived gate — pointing at `_cpu_pain_budget.canStartNow()` +(`referenceChains.cpp:893`), the only branch in `shouldRunPass()` with no +`TEST_LOG`, as the mechanism (see `find-cpu-pain-budget-starves-canary-passes`). + +## Methodology note (do not re-waste time on this) + +`kubectl logs --since=` (tried `90m`) and `--since-time=` both silently returned only ~45-60 seconds of actual content on this +pod — the extremely high `TEST_LOG` volume (tens of thousands of lines per +minute) fills whatever retention buffer the log driver keeps, with no error +or warning when older lines are unavailable. Historical evidence from +earlier in this pod's lifetime (the 5 confirmed abandon cycles referenced in +`ev-postfix-onpod-live-verification`) is no longer retrievable from live pod +logs at all. For any future retrospective query on this pod, do not rely on +`--since`/`--since-time` for windows longer than roughly a minute — capture +forward with `-f` instead if you need a clean window. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-hotspot-lifo-visitation-order.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-hotspot-lifo-visitation-order.md new file mode 100644 index 0000000000..85ce721216 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-hotspot-lifo-visitation-order.md @@ -0,0 +1,99 @@ +--- +id: ev-hotspot-lifo-visitation-order +type: evidence +status: confirmed +depends_on: [] +supersedes: [] +related: [find-candidate1-never-tagged] +tags: [hotspot, jvmti, FollowReferences, LIFO, visit-stack, source, holder-fill, reversed] +created: 2026-08-26 +updated: 2026-08-26 +--- + +# HotSpot FollowReferences uses LIFO visit_stack — drives reversed holder fill + +## Source + +`openjdk-jdk21/src/hotspot/share/prims/jvmtiTagMap.cpp` + +## The visitation loop (lines 2943-2957) + +```cpp +// the heap walk starts with an initial object or the heap roots +if (initial_object().is_null()) { + // ... root collection ... +} else { + visit_stack()->push(initial_object()()); // holder array pushed +} + +if (is_following_references()) { + while (!visit_stack()->is_empty()) { + oop o = visit_stack()->pop(); // LIFO pop + if (!_bitset.is_marked(o)) { + if (!visit(o)) { + break; + } + } + } +} +``` + +`FollowReferences(initial_object=holder_array)` pushes the holder, then +the while-loop pops **LIFO** (last in, first out). + +## Array element iteration (lines 2498-2517) + +```cpp +inline bool VM_HeapWalkOperation::iterate_over_array(oop o) { + objArrayOop array = objArrayOop(o); + // array reference to its class + oop mirror = ObjArrayKlass::cast(array->klass())->java_mirror(); + if (!CallbackInvoker::report_class_reference(o, mirror)) return false; + // iterate over the array and report each reference to a non-null element + for (int index=0; indexlength(); index++) { + oop elem = array->obj_at(index); + if (elem == nullptr) continue; + if (!CallbackInvoker::report_array_element_reference(o, elem, index)) return false; + } + return true; +} +``` + +`visit(holder)` calls `iterate_over_array`, which reports elements +0..n-1 in order via `report_array_element_reference`. Each reported +element that the callback returns `JVMTI_VISIT_OBJECTS` for gets +`check_for_visit()` → `visit_stack()->push(elem)` (line 1462): + +```cpp +static inline bool check_for_visit(oop obj) { + if (!_bitset->is_marked(obj)) visit_stack()->push(obj); + return true; +} +``` + +So elements are **pushed in index order 0..n-1**, but the while-loop +**pops LIFO** — meaning the class at `holder[n-1]` is descended first, +`holder[0]` last. + +## Consequence for admitStaticFieldRoots() + +If the holder is filled as `holder[i] = classes[chunk_start + i]` +(ascending), LIFO descent visits classes in **reverse** order +(chunk_end-1 first, chunk_start last). An abort at class p means the +*done* set is {p+1..chunk_end-1} and the *pending* set is +{chunk_start..p-1} — so "resume at last seen" would skip the pending +low-index classes. + +**Fix:** fill the holder in **reversed** order: +`holder[i] = classes[chunk_end - 1 - i]`. Then LIFO pop visits +`holder[n-1] = classes[chunk_start]` first → **ascending original index +order**. An abort at class p now means done={chunk_start..p-1}, +partial={p}, pending={p+1..chunk_end-1}. Resume at p (redo the partial +class) — no completed classes re-walked, no pending classes skipped. + +## Verification + +Read directly from openjdk-jdk21 source at +`/System/Volumes/Data/Users/jaroslav.bachorik/opensource/openjdk/openjdk-jdk21/src/hotspot/share/prims/jvmtiTagMap.cpp`. +No runtime experiment needed — the visitation order is structural in +the C++ code, not configurable or JVM-version-dependent. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-jafar-zero-refchain-events.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-jafar-zero-refchain-events.md new file mode 100644 index 0000000000..77c5ce3f54 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-jafar-zero-refchain-events.md @@ -0,0 +1,36 @@ +--- +id: ev-jafar-zero-refchain-events +type: evidence +status: confirmed +depends_on: [ev-post-resync-deployment-verified] +supersedes: [] +related: [find-marker-tag-slot-index-mismatch] +tags: [jfr, jafar, zero-events, post-resync] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Phase 2: event types registered, event count still 0 + +## Source +jafar MCP `jfr_open` + `jfr_list_types` on the freshest uploaded profile +after the resync. + +## Raw excerpts + +```json +{"id":19,"alias":"jb-fresh", + "path":"/tmp/hotdog-jb-fresh3/prof-analyzer-hotdog-2026-08-24_14-46-42.156Z-ip-10-128-190-53.ec2.internal-stripe.jfr", + "availableTypes":229,"chunkCount":2,"message":"Recording opened successfully"} +``` + +```json +{"sessionId":19,"totalTypes":2,"totalEvents":0,"scanned":true, + "eventTypes":[{"name":"datadog.ReferenceChain","count":0}, + {"name":"datadog.ReferenceChainAbandoned","count":0}], + "filter":"chain"} +``` + +So: the types are registered by the new binary (proving the feature is +compiled in and the JFR writer is wired), but **zero** events of either +type were emitted in the recording window. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-jfr-analysis-real-recording.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-jfr-analysis-real-recording.md new file mode 100644 index 0000000000..a42d7f788e --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-jfr-analysis-real-recording.md @@ -0,0 +1,75 @@ +--- +id: ev-jfr-analysis-real-recording +type: evidence +source: ~/Downloads/20260828-094323_prof-analyzer-hotdog_AaBHwZLWAAAayqog74k1iwAA/main.jfr +collected: 2026-08-28 +tags: [jfr, jafar, on-pod, ddprof-jfr, analysis, NEW-THIS-SESSION] +--- + +# JFR analysis of real ddprof recording + +## Source + +Real ddprof JFR (NOT jcmd dump — jcmd only dumps JDK's built-in JFR, not +ddprof's native JFR writer). Obtained via profiling toolkit download.py +from uploaded profiles. + +## Event types found + +``` + 105 - datadog.HeapLiveObject + 129 - datadog.ReferenceChain + 130 - datadog.ReferenceChainAbandoned +``` + +## ReferenceChain events (2) + +1. `targetTag=-4611686018427387906` (marker tag, slot 0) — depth=14, + totalHops=15, rootKind=unknown + → Canary chain for [B candidate. rootKind="unknown" means chain + reconstruction didn't find a proper root. + +2. `targetTag=5940` — depth=1, totalHops=2, + rootKind=first_observed_via:jni_local + → Auto-mark chain for the **noise** [B (136B, s3-netty-2 thread, + JNI local root). BFS reached this shallow instance before the deep + leaking instances. + +## HeapLiveObject events (17 total) + +### [B instances (13) + +| Age | Size | Thread | Stack top | Description | +|-----|------|--------|-----------|-------------| +| 168 | 78MB | simulated-memory-leak | lambda$static$1 | LEAK (has stack) | +| 161 | 78MB | simulated-memory-leak | no-stack | LEAK | +| 133 | 78MB | simulated-memory-leak | no-stack | LEAK | +| 124 | 78MB | simulated-memory-leak | no-stack | LEAK | +| 118 | 78MB | simulated-memory-leak | no-stack | LEAK | +| 106 | 78MB | simulated-memory-leak | no-stack | LEAK | +| 97 | 78MB | simulated-memory-leak | no-stack | LEAK | +| 81 | 78MB | simulated-memory-leak | no-stack | LEAK | +| 77 | 78MB | simulated-memory-leak | no-stack | LEAK | +| 69 | 78MB | simulated-memory-leak | no-stack | LEAK | +| 61 | 78MB | simulated-memory-leak | no-stack | LEAK | +| 33 | 78MB | simulated-memory-leak | no-stack | LEAK | +| 172 | 136B | s3-netty-2 | initClassName | NOISE | + +### Other classes (4) + +- `[Ljava.lang.Object;` × 2 (ages 49, 91, 1040B each, netty threads) +- `PSWMS` × 1 (age 103, 48B, cpu-intensive-8) +- `ArrayList` × 1 (age 172, 24B, s3-netty-2 — the container holding the leak) +- `IntPriorityQueue` × 1 (age 168, 24B, grpc worker) + +## Key findings + +1. The noise [B (age=172, 136B) is the oldest — age heuristic picks it + first as representative. +2. The leaking [B instances are all tid=172 (simulated-memory-leak), + all 78MB, ages 33-168. 12 instances from the same allocation site. +3. The noise [B is tid=284 (s3-netty-2), 136B, different allocation + site entirely. +4. The auto-mark chain (depth=1, jni_local) was for the noise [B. + The leaking [B at depth=14 was never reached by BFS before the + chain was cached (per-class caching blocked it). diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-kind-counts-constant-pool-dominates.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-kind-counts-constant-pool-dominates.md new file mode 100644 index 0000000000..632db02868 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-kind-counts-constant-pool-dominates.md @@ -0,0 +1,61 @@ +# Evidence: per-kind callback tally confirms CONSTANT_POOL dominates admitStaticFieldRoots() volume + +Pod `prof-analyzer-hotdog-jb-c944876b9-8vtzw` / `profiling-stg`, deployed with +`3a2fc0d5e` (temp diagnostic on top of `a86f0dd87`). 36 samples over a 10-minute +window (`kubectl logs ... --since=10m | grep -i kind_counts`). + +``` +kind_counts k1=1 k2=0 k3=512 k4=450 k5=0 k6=450 k7=308 k8=290 k9=5144 k10=130 +kind_counts k1=1 k2=0 k3=512 k4=435 k5=0 k6=426 k7=299 k8=279 k9=4236 k10=113 +kind_counts k1=1 k2=0 k3=512 k4=83 k5=0 k6=83 k7=66 k8=480 k9=2862 k10=9 +kind_counts k1=1 k2=0 k3=512 k4=466 k5=0 k6=466 k7=209 k8=2339 k9=7553 k10=234 +kind_counts k1=1 k2=0 k3=512 k4=430 k5=0 k6=428 k7=308 k8=485 k9=4520 k10=123 +kind_counts k1=1 k2=0 k3=512 k4=286 k5=0 k6=286 k7=200 k8=230 k9=2557 k10=24 +kind_counts k1=1 k2=0 k3=512 k4=455 k5=0 k6=455 k7=330 k8=352 k9=4509 k10=64 +kind_counts k1=1 k2=0 k3=512 k4=53 k5=0 k6=53 k7=53 k8=470 k9=2954 k10=0 +kind_counts k1=1 k2=0 k3=512 k4=485 k5=0 k6=485 k7=400 k8=1617 k9=10894 k10=20 +kind_counts k1=1 k2=0 k3=512 k4=205 k5=0 k6=205 k7=68 k8=188 k9=2836 k10=81 +... (26 more, same shape) +``` + +Kind legend (jvmti.h `jvmtiHeapReferenceKind`): k1=CLASS, k3=ARRAY_ELEMENT, +k4=CLASS_LOADER, k6=PROTECTION_DOMAIN, k7=INTERFACE, k8=STATIC_FIELD, +k9=CONSTANT_POOL, k10=SUPERCLASS. + +## Reading + +- **k9 (CONSTANT_POOL) is the largest and most variable kind in every + sample**: range 2353-10894, always 5-15x larger than k8 (STATIC_FIELD: + 174-2339) in the same sample. Confirms the code-grounded hypothesis in + `find-candidate1-never-tagged` — per-chunk callback volume is dominated by + constant-pool-derived heap references, not by a class's own static-field + count. +- **k3 (ARRAY_ELEMENT) is pinned at exactly 512 in all 36 samples** — equal + to `STATIC_FIELD_SWEEP_CHUNK_CLASSES`. This is a flat one-per-class + overhead (one array-element-kind callback per class in the chunk, + independent of the class's actual content) and not a leak-volume signal. +- k4/k6 (CLASS_LOADER/PROTECTION_DOMAIN) track each other almost exactly in + every sample (e.g. 450/450, 435/426, 83/83) — expected, since both walk + the same classloader-chain edge from the class object. +- k1 (CLASS) constant at 1 — the seed edge itself, one per call. +- k2 (FIELD) and k5 (SIGNERS) are 0 in every sample — FIELD kind is not used + by this code path (STATIC_FIELD is the relevant kind for static fields, + regular FIELD applies to instance-field walks which this call doesn't do); + SIGNERS is legitimately almost always empty (unsigned classes). + +## Confirms + +Resolves the "not yet measured directly" gap in `find-candidate1-never-tagged` +that was blocking the mechanism from being called CONFIRMED. Per-chunk +truncation is most plausibly explained by one or a few CONSTANT_POOL-heavy +classes near the front of a 512-class chunk generating enough callback volume +alone to trip the 4096-callback deadline-check granularity before the walk +reaches deeper into the chunk. + +## Does not yet confirm + +- Whether this is deterministic (same chunk always truncates at the same + point every lap) vs. jitter-driven — would need per-class-index visibility + inside the deadline-check path, not just per-chunk totals. +- Whether candidate[1]'s (klass_id=283) holder class is itself CP-heavy or + simply sits behind one in classlist order. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-correlation-local-repro.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-correlation-local-repro.md new file mode 100644 index 0000000000..bb165ac558 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-correlation-local-repro.md @@ -0,0 +1,74 @@ +--- +id: ev-leaktag-correlation-local-repro +type: evidence +status: verified +related: [find-rotation-resize-blindspot, find-gate-bypass-representative-paths, find-klass-id-notation-mismatch, find-test-seam-aliasing, find-ema-batch-collapse, find-leak-tag-pool-implementation] +tags: [local-repro, E2E, post-fix, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# Local repro (LeakTagCorrelationScenario) - build + debugging trail + +The hotdog deployment loop replaced by a local E2E: +`ddprof-test/.../referencechains/LeakTagCorrelationScenario` + +`ExternalLauncher` mode `leak-correlation` + +`LeakTagCorrelationReferenceChainTest` (slow suite, ~2min). Child JVM: +static growing List leak (1MB byte[] chunks), ephemeral stack-local +NoisePayload churn on a noise thread, ~100k FillerNode scale graph. +Asserts: in-pool targetTag == a byte[] HeapLiveObject's leakTag, durable +root, depth>=1, no transient-rooted shallow chains, no NoisePayload +chains. Failing runs embed a filtered child TEST_LOG tail in the +assertion message (queue depths, gotw batches, interception lines). + +## What the harness peeled apart, in order + +1. Run 1: `NoisePayload@depth0:static_field` chain - the fixture held the + noise rep in a static field (truthful output; fixed the fixture). +2. Run 2: `NoisePayload@depth1:first_observed_via:stack_local` chain + EMITTED -> the canary/rep path gate bypass (real bug, fixed). +3. Depth-0 durable chains suppressed by the round-4 gate -> predicate + corrected (design flaw, fixed). +4. Slow-suite run: 3 scenarios broken since the redesign -> synthetic-id + seam breakage -> aliasing seam (fixed; LeakingCache + StaticField + + in-process gcRoot/cache tests green again). +5. `tagged=0` forever for synthetic ids; then real-id candidates but + `resolveClassMap` ChainLink id 2 vs LT candidate id 63 -> dot/slash + notation bug (production, fixed). +6. Tagging selection: 288B payloads flaked (sampling-dependent); + 300KB companion arrays sampled the ARRAY not the payload (companion + steals sampling - exactly CachedPayload's documented warning). Fix: + the leaked object IS the big allocation (1MB byte[] chunks) + scan + every 3rd round (JMC parse churn was evicting chunks from the tracking + table before they aged into tagging priority). Result: `tagged` lines + show size=1000016, ages accumulating - tagging deterministic. +7. Zero interceptions with everything above: fixed-slot array still 0 -> + the resize-blindspot series (fair-share, hygiene, ancestor fanout, + requeueChainRootForRotation). After all four: 4 interceptions at + depth 2 under the live elementData, correlated discoveries recorded + with noise eviction. +8. Last-mile found: the byte[] candidate dropped (hysteresis aging) right + after the interceptions -> per-round seeding added to the scenario + (find-test-seam-aliasing). FINAL RUN RESULT PENDING - the verification + run was interrupted by the checkpoint; re-run first thing next session. + +Also verified locally: proportional batch control behaves on a real JVM +(batch 270, next 512, gotw 16-18ms) and the fair-share lane drain + +`filtered depth=1 root_kind=24` gate lines fire in a real run. + +## FINAL RESULT (this session): PASSED, committed + +`[correlation-found] 1073742077` - the scenario passes end-to-end through +the GROWING ArrayList (per-round trend maintenance was the last mile: +a one-shot seeded ramp ages out of LivenessTracker's hysteresis and the +byte[] candidate dropped right after the first interceptions, stranding +the correlated discoveries). + +The post-green full testDebug run then exposed, via the seams test, +three more real defects (all fixed, all nodes): the dangling jclass +local-ref cache crash, the seam-vs-BFS-thread engine race, and the +depth-0 durable-root upgrade gap. Suites at commit 663784137: +539 gtests green, slow suite 16/16, seams test green; the 7 remaining +testDebug failures are pre-existing environment failures (proven at +base via stash for CollapsingSleepTest; wallclock/nativethread/vtable/ +lifecycle subsystems untouched by this branch). diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round1.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round1.md new file mode 100644 index 0000000000..58c2b7a791 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round1.md @@ -0,0 +1,54 @@ +--- +id: ev-leaktag-onpod-round1 +type: evidence +status: verified +related: [find-leak-tag-pool-implementation, find-ema-batch-collapse, find-leaktag-jfr-field-misalignment, q-coverage-tracking-per-combination] +tags: [pod-logs, on-pod, post-fix, live-verification, leak-tag, batch-collapse, jfr, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# On-pod round 1 of the leak-tag redesign (build 1ce2b4f03) + +Pod prof-analyzer-hotdog-jb-86d8bf5854-zng9s, PID 32849 (restart 10:22Z). +15-min log capture (`kubectl logs --since=15m`, analyzed as +/tmp/hotdog-15m.log) + uploaded recording 20260831-103955. + +## What worked + +- `tagLeakInstances tagged=28` — pool tagging live, re-tagging idempotent + (already-tagged entries counted, not re-acquired). +- Representative JVMTI tag read back 0x4000000F (real pool tag). +- Adaptive CPU: multiplier=15.0 firing (canary active, no emergency), + refill_rate=0.45. +- Chains cached for 8 discovered [B instances and drained to JFR: + `Profiler::dump reference-chain batch=8 write_dropped=0`. +- Depth==0 filter active (no depth-0 chains emitted). +- Search/rotation/sweep all running; frontier 220-225k. + +## What failed + +- **batch_size=2** (find-ema-batch-collapse): 3 runPass/15min, ~1000 gotw + calls per pass, 10.4s CPU spend per pass (debt 10436ms), edges/pass ~100. +- **emergency never fired** — wrong counter (frontier progress, see node). +- **The 8 chains are noise**: depth=1 jni_local/stack_local, targetTags are + frontier tags (7167/7267/7291/7554/7807/7810/7818/7821), NOT leak tags — + admitted via the ordinary path (leak_tag=0). BFS (crawling at batch=2) + never reached the leak-tagged 78MB [B instances. +- **leakTag column garbage** (find-leaktag-jfr-field-misalignment): + 75/50 on rows with context, 0 without; user's "only two entries tagged" + was the misparse, not the table state. +- **Marker re-tag leftover found in code**: pollWatchedTargets' tag==0 + branch still SetTag(MARKER_TAG_BASE - slot) — resurrects the dead + mechanism when a rep is temporarily untagged. Removed in 0db70994d. + +## Unresolved + +- Whether the leak-tag interception fires on-pod (no TEST_LOG in that + build; the 8 "discovered" came from the ordinary auto-mark). The gtest + (LeakTagInterceptionConvertsToFrontierTagAndCorrelates) proves the + mechanism; 0db70994d adds the interception TEST_LOG for on-pod proof. +- Coverage semantics still per-object (q-coverage-tracking-per-combination). + +Fix round 2 = commit 0db70994d (alignment + AIMD + deadline + priority +tagging + emergency counter + marker re-tag removal). Awaits redeploy. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round10.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round10.md new file mode 100644 index 0000000000..77e817f115 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round10.md @@ -0,0 +1,65 @@ +--- +id: ev-leaktag-onpod-round10 +type: evidence +status: confirmed +depends_on: [find-anchor-live-feed-design, find-already-admitted-blocks-unreachable] +related: [find-anchor-holder-eviction, find-canary-stuck-restart-wipes-frontier, find-togcroot-orphaned-slot-stranding] +tags: [pod-verification, round-10, b-prime, live-feed, fifo, sweep-lap, restarts, NEW-THIS-SESSION] +created: 20260914 +updated: 20260914 +--- + +# Pod round 10 (build 6f3c6cc2e = B' + nesting fix): B' machinery LIVE; interception zero — restarts reset the sweep lap before it reaches the leak holder's class + +Pod `prof-analyzer-hotdog-jb-d9d8cf-plwpx` (fresh from redeploy), JVM +26.0.2+10, container `prof-analyzer`, deployed build verified via the +`static_anchor_fifo_*` fields (unique to 6f3c6cc2e) in rotation_candidates. + +## B' mechanically verified live + +- Pushes: `static_anchor_fifo_pushed_total` +250-500/min; FIFO size pins at + ~1008-1021 (the 1024 cap — cap-drop throttling active); counter NOT reset + on restartSearch (by design), size IS (1008→477 observed = a restart). +- Drain/walk: `static_anchor_tags=20` = 16 FIFO + 4 collector per pass; + chain-attached anchors (`parent!=0 root_kind=0` — bytebuddy/micronaut + machinery, e.g. MethodSortMatch$Sort enums, ConstructorComparator) ARE + being walked — the population round 9's tier structurally excluded. +- Walk truncation: `walked=6..16 of selected=20 truncated=1` in ~60% of + passes — the FIFO-first order starves the collector's suffix those + passes (secondary issue; see below). +- Leak tagging healthy: klass_id=2 tid=12875 tagged=21 max_size=78643216. + +## Interception STILL zero — the new blocker chain (observed, not inferred) + +1. Restarts every few minutes: frontierSize non-monotonic across samples + (17k → 370k → 24k → 237k), FIFO cleared mid-run (1008→477), + candidateFound flapping 0/1 ↔ 0/0 (matches the known shared-JVM epoch + noise resetting consecutive_positive — find-togcroot-orphaned-slot- + stranding fix-2 residual). +2. Each restart resets the static sweep cursor: `cycle_complete=0` always, + cursor at 20022→24897 of 34397 after ~7 min of one search + (~500-2800 classes/pass, ~1 pass/min ⇒ 15-60 min/lap uninterrupted). +3. ProfileAnalyzer (app class, HIGH class index — machinery loads first) + is never swept before the next restart ⇒ LEAK_BUFFER's + UnmodifiableRandomAccessList wrapper is never admitted root-attached + static ⇒ neither the collector cohort NOR the at-risk FIFO can ever + select it (it has no frontier entry at all). +4. The 2 wrapper instances walked (tags 5999, 9519) are OTHER unmodifiable + lists (their walks found no tagged chunks — no intercept lines). + +## Suspected restart driver (candidate, not yet caught in a log window) + +TEMP `CANARY_NO_PROGRESS_PASS_LIMIT` 30→3 (deployed to the pod, on the +MUST-REVERT list): unfound canary ⇒ abandon+restart after only ~3 passes +≈ 3 min at the observed ~1 pass/min — matches the observed restart +cadence. Also unfound-candidate no-progress abandons during the 0/0 +flaps. The exact abandon REASON line has not been captured yet (1-min log +retention; the transition is a rare line). + +## Secondary findings + +- FIFO-first drain order starves the collector suffix on truncated + passes — the LEAK wrapper (single-referrer static ⇒ collector cohort, + never at-risk) would be in the starved suffix whenever it IS admitted. +- At-risk population ≈ 1024 (cap) — the design's "push-rate ≪ drain-rate + needs the filter" caveat is now MEASURED: pushes 15-30x the drain rate. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round11.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round11.md new file mode 100644 index 0000000000..31f7e3e546 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round11.md @@ -0,0 +1,87 @@ +--- +id: ev-leaktag-onpod-round11 +type: evidence +status: confirmed +depends_on: [ev-leaktag-onpod-round10, find-anchor-live-feed-design] +related: [find-anchor-holder-eviction, find-per-tid-qualification-design, find-togcroot-orphaned-slot-stranding] +tags: [pod-verification, round-11, cycle-complete, collector-cursor, candidate-flap, oom-endgame, NEW-THIS-SESSION] +created: 20260914 +updated: 20260914 +--- + +# Pod round 11 (04539b821 = canary-limit revert + collector-first order): sweep laps COMPLETE for the first time; interception still zero; candidate itself flapped out while the heap marches to OOM + +Same pod `prof-analyzer-hotdog-jb-d9d8cf-plwpx`, JVM 70341 restarted with the +new build (fresh tracker confirmed: pushed_total reset to ~0 and grew +1945→5035→7719+). + +## A (canary limit 30) — WORKED + +- Search lifetimes up from round 10's 3-6 min to tens of minutes; pass + pacing observed via `canary backoff mult=16 ema_ms=171` (~2s/pass) and + threadLoop cadence ~1s. +- **`cycle_complete=1` observed repeatedly — the static sweep completes + laps for the first time on this pod** (rounds 3-10 never completed one: + either pre-cursor-fix truncation or restart-driven resets). Laps now + wrap in <2 min at high pacing (cursor alternation 0↔30k across 30s + polls initially misread as restarts — they are LAP WRAPS). +- Consequence: the root-attached static cohort is no longer the + round-9-sized 76 machinery statics — every swept class's statics are + admitted (collector picks walk ~4/pass, 69 root_kind=8 anchors walked + in an 8-min sample). + +## B (collector-first order) — no-op in production + +Real `GetObjectsWithTags` returns results in UNSPECIFIED order (observed +interspersed parent=0/parent!=0 walk lines) — the walk order is the GOTW +iteration order, not the request-vector order. The round-10 "collector +suffix starvation" model was mock-order-based; in production truncation +drops both cohorts uniformly (~40% of the 20-anchor batch). The requeue +protection still functions (order-independent). + +## Interception still zero + +Wrapper (`Collections$UnmodifiableRandomAccessList`) never appeared in +walked anchors. It is admitted root-attached STATIC (laps reach +ProfileAnalyzer) and sits in the collector's wrapping-cursor lottery: +4 picks/pass over a now-large cohort, ~40% of picks dropped by truncation +— the wrapper was not reached before the search went terminal. + +## NEW: the candidate flapped out while the leak demonstrably grows + +At ~20:35: `selectLeakCandidates entry[0] klass_id=4 ring_fill=30 +has_trend=1 slope=-47.348387 consecutive_positive=0 required=3` → +0 candidates → search terminal (search_state=2, tags_released=1, +`hasLeakSignal -> false (secondsToOOM=4351.6, candidates=0, urgent=0)`) +— WHILE `secondsToOOM` shows +389MB/280s and the heap at ~4.7GB. +Suspicion (unproven): self-reinforcing dropout — candidate slot drops +(one epoch of negative delta) → admitForTracking's watched-tid boost +clears → the tracked sample thins → negative slope keeps the candidate +out. Tagging had been healthy at scale all along (59 x 78MB chunks, +klass_id=4 tid=70542). + +## Next window + +OOM endgame: secondsToOOM ~4300s at observation time and shrinking — +when it crosses the urgency threshold, the urgent search opens without a +candidate and urgency tracking tags everything: the interception +machinery gets a fresh window with laps completing. Watch for the +wrapper walk + `leak-tag intercepted` during the urgency phase. + +## Watch continuation (20:45-21:18 UTC) — DECISIVE NEW FACT + +- ~20:45: the candidate RE-QUALIFIED on its own (fresh chunk wave: ages + 1-97, tagged=13, klass_id=4 tid=70542) — the flap was not terminal; + chase resumed with laps wrapping. +- **~21:10-21:12: the wrapper WAS walked (2 walk lines) with ZERO + interception.** This REFUTES the "wrapper not admitted / not in the + eligible set" hypothesis — the wrapper is admitted, selected by the + collector, and walked; its walk just never intercepts. The open + question narrowed to exactly three branches (captured in STATE.md + next-steps): (a) the walked instance is a decoy wrapper (holder class + ≠ ProfileAnalyzer); (b) the walk is budget-starved before descending + (edges=0 truncated=1); (c) the walk enumerates [B chunks but they were + UNTAGGED at that moment (tag-lifetime problem). The round-12 + diagnostic (8ca24a524) discriminates all three. +- No urgency phase observed yet at 21:18 (secondsToOOM still large); + monitoring continues at the next session. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round12.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round12.md new file mode 100644 index 0000000000..3a29e9f13a --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round12.md @@ -0,0 +1,89 @@ +--- +id: ev-leaktag-onpod-round12 +type: evidence +status: confirmed +depends_on: [ev-leaktag-onpod-round11, find-anchor-live-feed-design] +related: [find-anchor-holder-eviction, find-option-c-descend-walk-design] +tags: [pod-verification, round-12, wrapper-class, synchronized, diagnostic-fix, NEW-THIS-SESSION] +created: 20260915 +updated: 20260915 +--- + +# Pod round 12 (build 5218bd1dd + diagnostic fix): wrapper class is +# SynchronizedRandomAccessList, NOT UnmodifiableRandomAccessList + +Deployed build 5218bd1dd (includes 8ca24a524 round-12 diagnostic) to pod +`prof-analyzer-hotdog-jb-d9d8cf-plwpx` (JVM 244587, started 07:26 UTC). +The diagnostic fired ZERO times — the wrapper never appeared in any anchor +walk. + +## ROOT CAUSE: wrong wrapper class name + +Bytecode inspection of `prof-analyzer-0.0.1.jar` on the pod reveals: + +``` +private static final java.util.List LEAK_BUFFER; +// static initializer: +86: new // class java/util/ArrayList +90: invokespecial // Method java/util/ArrayList."":()V +93: invokestatic // Method java/util/Collections.synchronizedList:(Ljava/util/List;)Ljava/util/List; +96: putstatic // Field LEAK_BUFFER:Ljava/util/List; +``` + +**LEAK_BUFFER = `Collections.synchronizedList(new ArrayList<>())`**, NOT +`Collections.unmodifiableList`. The wrapper class is +`Collections$SynchronizedRandomAccessList`, NOT +`Collections$UnmodifiableRandomAccessList`. + +The round-12 diagnostic (commit 8ca24a524) checked for +`strstr(sig, "UnmodifiableRandomAccessList")` — it will NEVER match the +actual wrapper. **Fixed this session**: broadened the check to also match +`SynchronizedRandomAccessList` and `SynchronizedList`. Rebuilt and +redeployed (JVM 244587). + +**Implication for round 11**: the "wrapper WAS walked" observation at +21:10 UTC (ev-leaktag-onpod-round11) identified the wrapper as +`UnmodifiableRandomAccessList` — that was a DIFFERENT unmodifiable list +(a decoy), not the LEAK_BUFFER wrapper. The LEAK_BUFFER wrapper +(`SynchronizedRandomAccessList`) was never walked in round 11 either. + +## Secondary finding: rotation phase always truncated + +All 11 `walkStaticFieldAnchors selected=20` lines in the container log +show `truncated=1` with only `walked=1-13` of 20 selected anchors. The +collector's cursor (`_static_anchor_rotation_cursor`) advances past ALL +20 selected entries, but collector-sourced unwalked anchors are +**deliberately not requeued** (only FIFO-sourced unwalked anchors are +requeued to the FIFO front). So an anchor selected at position 14-20 that +is truncated before being walked is skipped until the cursor wraps around +the entire frontier table (~50k entries / 20 per pass = ~2500 passes = +~2.3 hours at 3.3s/pass). + +The wrapper, if admitted as root-attached, is only walked when: +1. The cursor rotates to its position (~2.3 hours), AND +2. The walk is not truncated before reaching it + +This explains why the wrapper was "never walked" in this recording +(54 min of search before terminal) and was only intermittently walked +in round 11 (hours of recording). + +## Search state + +The candidate (klass_id=8 [B) flapped out due to self-reinforcing dropout +(same pattern as round 11): candidate slot drops → watched-tid boost +clears → tracked sample thins → negative slope keeps candidate out. +`selectLeakCandidates returning 0 candidates` → `search_state=2` +(terminal). secondsToOOM ~4400s (73 min to OOM); urgency path opens at +<1800s (~50 min away). + +## Next steps + +1. Wait for urgency path (secondsToOOM < 1800) → urgent search opens + without candidate, urgency tracking tags everything (100% admission) + → search restarts → sweep + rotation run → wrapper walked → diagnostic + fires with correct class name. +2. Consider fixing the rotation truncation issue: requeue collector-sourced + unwalked anchors (same as FIFO requeue), or prioritize anchors with + leak-tagged children. +3. The diagnostic fix (SynchronizedRandomAccessList) is deployed but + uncommitted — commit after verification. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round13-results.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round13-results.md new file mode 100644 index 0000000000..608c3ddb80 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round13-results.md @@ -0,0 +1,98 @@ +# Round 13 RESULTS: root cause of missing ReferenceChain events identified (ev) + +Date: 2026-09-15, pod prof-analyzer-hotdog-jb-d9d8cf-rz992 (redeployed 17:27Z +with b5dd09675), container prof-analyzer. + +## Probe results (decisive) + +LEAK_BUFFER probe (JNI read of ProfileAnalyzer.LEAK_BUFFER each sweep chunk +pass) fired as designed: + +``` +probe: sweeping holder class Lcom/.../ProfileAnalyzer; at index 25301 +probe wrapper_tag=0 (lap 1, pre-admission) +probe wrapper_tag=0 (lap 2, still pre-admission — chunk sweep + aborted or search restarted before the k8 edge) +probe wrapper_tag=198536 frontier_found=1 parent=0 root_kind=8 state=0 + leak_tag=0 depth=0 referrer_klass=29448 +``` + +**The wrapper IS admitted root-attached STATIC_FIELD (root_kind=8, parent=0, +FRONTIER state)** — the sweep's k8 edge works. Admission is NOT the bug. + +## The actual bug: deterministic anchor-selection tail starvation + +Measured in one log window (134k-line buffer, 16 passes, single search): + +- `root-attached STATIC_FIELD admit` lines: **4735** (and this window is + mid-search) — sweep cursor 16596→22323 of 34310 classes, ~0.83 admits per + class. Full-lap index size ≈ **28k anchors**. +- Anchor walks: **340** in the window = ~21/pass (budget 16+16), avg + **81 edges** per anchor walk (max 2784) against effectiveBudget=3741 — + the walk budget is already nearly saturated at 32/pass. +- Search lifetime: frontier 131k→250k cap at ~1.3-2k inserts/pass ≈ + **~190 passes** per search; collector coverage ≈ 190 × 21 ≈ **4k of 28k** + (14%) per search, always from index position 0 (cursor reset on restart). +- The wrapper is admitted at sweep cursor 25301 (≈75% through the class + array) → index position ≈ **12-21k** — **deterministically beyond the + ~4k coverage of every search**. Fix B+C made the walk rate 8x faster but + the selection ORDER (admission order = sweep class-array order) puts the + leak holder at the unreachable tail. The old budget-4 lottery was + actually a lottery; now it is deterministic never. + +Why fix C (leak_tag priority) doesn't help: the wrapper entry has leak_tag=0 +(the [B chunks carry leak tags, not their holder) — visible in the probe log. + +## Downstream corollaries (all observed, all explained) + +1. BFS expansion of the wrapper subtree is equally tail-starved: the wrapper + is admitted mid-search into a pendingExpand FIFO that is already 148k + deep; its subtree (ArrayList → elementData → chunks) is never expanded. +2. Zero `leak-tag intercepted` logs: no walk ever reaches a leak-tagged + chunk (they are only reachable through the wrapper's subtree). +3. `_leak_parent_fanout`/leak_parents=4717 is mostly NOISE: fed by + `seedLeakAccumulationForNewlyWatchedKlass`'s scan of [B entries ALREADY + in the frontier — ordinary byte[]s admitted via root-reachable paths + (kafka/netty buffers), each with fanout=1. The real chunks never get + admitted, so their parents (elementData fanout=13, the wrapper) are never + attributed. This also explains round-12's "all leak-accumulation + selections fanout=1". +4. `pollWatchedTargets buildChainEvent failed ... reconstructChain failed + for target_tag=8851/8853 klass_id=1`: **restartSearch() resets the + frontier table and _next_tag=1 but does NOT clear + _candidate_discovered_tags** (frontier tags of discovered watched-class + instances). Stale tags from a previous search resolve (lookup returns + found for any idx < table_size) into zeroed/reused slots → dangling + parent chains → reconstructChain fails. Worse: when such a slot holds a + live new-search entry, the poll can emit a chain event for the WRONG + OBJECT — the likely origin of the earlier session's "noise [B instance" + ReferenceChain event. Hygiene bug to fix regardless. + +## Fix options considered + +A. **Anchor-selection priority by class shape**: index only "interesting" + anchors — exclude leaf classes (String/Class/boxed/primitive-arrays: + identifiable from a fixed well-known class list) and/or prioritize + collection-shaped holders (the wrapper is a List). Shrinks the effective + cohort to (est.) 1-3k; coverage 4k/search then reaches the wrapper within + one search. Implementation: shape must be resolved OUTSIDE heap callbacks + (JNI in callback = illegal); lazy per-class cache (class_tag → shape), + filled by a bounded post-sweep resolver pass; collector sorts + (leak_tag, shape, cursor-fairness). +B. **Cross-restart coverage rotation + budget tuning**: persist a selection + start fraction across restarts; measured avg 81 edges caps the walk at + ~46/pass (3741/81), so full 28k coverage needs ~6-14 search lifetimes + (hours). Cheap but slow convergence. +C. **One-shot unrestricted whole-heap pass at search start**: a single + whole-heap FollowReferences would descend class→static→wrapper→… + →chunks and fire leak-tag interception for everything in one bounded + STW. Highest value per cost, but seconds-of-STW once per search and + frontier-cap interplay (millions of edges vs 250k cap) need design work. +D. A+B combined. + +## Also unchanged + +- Pre-existing SIGSEGV in SearchRestartTest.UrgentOOMProjectionBypassesCandidateGate + (clean-HEAD repro) — separate issue. +- Sweep-cycle `cycle_complete=1` never logs (_static_field_sweep_cycle_truncated + not reset by restartSearch) — latent, unfixed. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round13.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round13.md new file mode 100644 index 0000000000..1420044241 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round13.md @@ -0,0 +1,81 @@ +# Round 13: ground-truth probe for the LEAK_BUFFER wrapper (ev) + +Date: 2026-09-15 +Context: fix B+C (ccdb03b89) deployed and verified mechanically working on the +pod (prof-analyzer-hotdog-jb-d9d8cf-plwpx, JVM restarted 16:14:50Z), but the +LEAK_BUFFER wrapper still never walks. + +## Round 12 observations on the pod (fix B+C live) + +- Anchor walks went from budget 4/pass to 635 walks/15 min; 41–100+ distinct + anchor classes cycled (SynchronizedMap, UnmodifiableSet, ...), collector + static_anchor_tags=16 + FIFO 16 = 32/pass as designed. +- 5843 root-attached STATIC_FIELD first-admissions logged (JVM fresh → sweep + admits statics before BFS, unlike the old recording where all sweep k8 hits + were already-admitted). +- maybeUpgradeRootAttachedRootKind 9→8 upgrades firing. +- Leak rebuilding fast ("allocated 75 MB, retained chunks 9→13, heap 23-25%"). +- **NO `SynchronizedRandomAccessList` walk ever, zero interception, zero + ReferenceChain events, and zero com/dd/profiling app classes among walked + anchors.** + +## Deduction chain (why the probe was needed) + +`klass_id` in every log (admit / sweep-hit / push / upgrade) is the object's +**own** class id — `class_tag` in heapReferenceCallback params is the REFEREE's +class; the variable named `referrer_klass` is misnamed. All logs share one id +space; only the walk anchor log lacked the id (printed signature only). + +Every admission shape for the wrapper should end in a walk log: +- k8 root-attached admit → index → collector walk +- chain-attached → sweep k8 re-hit → pushAtRiskStaticAnchor → FIFO walk +- already-admitted re-hit → maybeUpgrade (parent==0) or push (parent!=0) + +None observed → contradiction; deduction exhausted. Also established: the +sweep re-partitions app classes (non-null classloader) to the FRONT every call, +so ProfileAnalyzer's class was swept within the first minute of JVM life, once +per ~70 passes (~30 s) since. + +## Round 13 instrumentation (this commit) + +1. **LEAK_BUFFER probe** in `admitStaticFieldRoots` (before FollowReferences, + before class-local cleanup): for each chunk class whose signature contains + `ProfileAnalyzer`, GetStaticFieldID("LEAK_BUFFER", "Ljava/util/List;") + (descriptor confirmed from the pod jar bytecode: `strings` on + ProfileAnalyzer.class shows `LEAK_BUFFER` / `Ljava/util/List;`) → + GetStaticObjectField → GetTag → frontier lookup → log + `wrapper_tag / frontier_found / parent / root_kind / state / leak_tag / + depth / referrer_klass`. Once per sweep chunk containing the holder class. + Interpretable outcomes: + - `wrapper_tag=0` persistently → wrapper never admitted at all (find the + filter that drops it) + - `frontier_found=0` with tag>0 → tag is not a frontier tag (leak-tag + range? tag-range bug?) + - `frontier_found=1 parent=0 root_kind=8` → admitted root-attached but + never walked → index/collector bug (cross-check with 2.) + - `frontier_found=1 parent!=0` → chain-attached → FIFO question +2. **klass_id added to walkStaticFieldAnchors anchor log** (entry.referrer_klass + = own class id): walked anchors now cross-referenceable with sweep/admit/ + push id space. If the wrapper's id (visible in the probe log via + referrer_klass) appears in sweep hits but never in walk logs → collector + bug; if never in sweep logs at all → admission bug. + +## Verification + +- gtestRelease: all ReferenceChainsBfsTest pass; 344 OK; only crash is the + pre-existing SearchRestartTest.UrgentOOMProjectionBypassesCandidateGate + SIGSEGV (known on clean HEAD 685f414b1, separate issue). +- spotlessApply clean; buildRelease -Pskip-tests succeeds. + +## Next + +- NOTE 16:50Z: the old pod (plwpx) was deleted; the replacement + `prof-analyzer-hotdog-jb-d9d8cf-rz992` (up 16:45Z) is STOCK - no + `-Ddd.profiling.experimental.ddprof.referencechains.enabled=true` flag, + stock image `prof-analyzer-hotdog:v137233187-2ad7b593`, zero engine logs. + The in-pod modifications died with the pod; the deployment spec was never + patched. User must re-apply their deploy procedure on rz992 (flag + .so + from b5dd09675 + restart). +- Grep after redeploy: `LEAK_BUFFER probe` (every ~30 s while the holder + class is in the current chunk), then interpret per the outcome table + above. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round14-results.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round14-results.md new file mode 100644 index 0000000000..22a3772999 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round14-results.md @@ -0,0 +1,89 @@ +# Round 14 on-pod verification + round-15 diagnosis (ev) + +Deploy: 6eb72f994 (tiered anchor selection + restart hygiene) on +prof-analyzer-hotdog-jb-d9d8cf-rz992, JVM restarted 20:05:48Z, verified +2026-09-15 20:06-20:53Z. User deployed; agent verified from logs only. + +## Round-14 mechanics VERIFIED WORKING + +1. **Tiering works.** `anchorTierHistogram index=20499-30711 leak_tier=0 + container_tier=1328-1680 other_tier=17686-24651 budget=16` — the container + cohort is **1633-1680**, far under the ~4k arithmetic gate. Container + anchors are visibly walked first-class: + ImmutableCollections$SetN/Set12/MapN/Map1/ListN, LinkedHashMap, HashSet, + Collections$SingletonSet, **Collections$SynchronizedMap** (walked via + at-risk FIFO), String (leftover budget flowing to tier 2, ~306 walks). +2. **Classification works** (containers tiered from pass ~27 onward). The + `reconcileAnchorClassShapes container class ` log appears ZERO times + in every held window — explained, not a bug: classification is one-time + per class per JVM (cache is JVM-lifetime, restartSearch keeps it), all + ~100-200 anchor classes were classified in the first search's first + passes (20:13-20:22, every captured buffer since has rotated past it). + Later container_tier growth = new ANCHORS of already-classified classes. +3. **Restart hygiene works.** No stale-tag `reconstructChain failed` flavor + observed; remaining `buildChainEvent false: target_tag=1073742079 not in + frontier` is the genuine standing gap (chunk not yet admitted). +4. LEAK_BUFFER probe caught the real holder class sweep live: + `sweeping holder class Lcom/datadog/profiling/analyzer/steps/ProfileAnalyzer; + at index 24627` → `wrapper_tag=0` (pre-admission, expected on first + crossing) → the containing chunk (24265→24777) completed cleanly + (1146 edges, truncated=0) → **the wrapper WAS admitted root-attached into + that search.** + +## Round-15 root cause: tail starvation reproduced AT TIER-1 SCALE + +Measured live (20:22-20:53, all in /tmp/pod_t*.log, pod_stream.log): + +- **Search lifetimes collapsed to 44-75 passes** (vs round-13's ~190): + frontier fills at 2.5-7.5k inserts/pass; cap-hit abandon observed at + size=245170 at pass ~75 and again ~242440 at pass 44+. Fill composition: + static sweep admits ~0.7-1.4k/pass, expand phase ~0-57, leak-accumulator + noise seeding `trackLeakAccumulation fanout-insert` ~650/min (round-13 + documented this seeding as fanout=1 noise), remainder expansion/roots. +- **The wrapper's holder class sits at sweep index 24627 of ~33270** (app + partition; class list re-partitioned app-first per call, index drifts + ±50 with lambda churn). The sweep therefore admits the wrapper MID/LATE + in a search (observed: pass ~43-44 of the current search). +- **Admission order = sweep order ⇒ the wrapper lands at the anchor-index + TAIL** (~27000+ of 27739), i.e. **container-ordinal ~1634 of 1633 — the + LAST container**. Walk owed at pass ~44 + (1634-cursor)/16 ≈ **pass 102+**. + Search lifetime 44-75 ⇒ **deterministic miss by ~30-50 passes.** +- **Compounding: the searches are canary chases** (`canary search, 0/1 + candidates found, backoff_mult=16 ema_ms=191`) — the candidate is the [B + leak class (klass_id=2, target tag 1073742079 = LEAK_TAG_BASE+255, + tagged=42-53). The canary's candidate can only be "found" via a resolved + chunk chain; the chain needs the wrapper walk; the wrapper walk needs + ~102 passes in one search; the search caps at 44-75 AND the canary + backoff (16 × ema 191ms ≈ 3s/pass) slows the very passes that would + reach it. **Self-sustaining chase deadlock.** +- Sweep gate behavior measured: runs only while the loaded-class count + changes (lambda churn keeps it open in waves); cursor wraps at count + (~33253), resets to 0 if count shrinks below cursor (lambda unload); + laps complete in ~2-10 min during churn waves; the real class is crossed + roughly once per wave/lap. + +## Round-15 fix direction (STW-free, surgical) + +**Fresh-admission priority in the container tier**: anchors admitted since +the last collector pass (index high-water mark) — at minimum the +freshly-admitted CONTAINER-shaped ones — are walked FIRST in the very next +pass (newest-first). The wrapper admitted at pass N is walked at pass N+1 +(classified by the same-pass reconcile, or fresh-priority independent of +shape). No new STW cost: same 16-walk budget, reordered. Breaks the chase +deadlock end-to-end: wrapper walk → subtree expansion admits the +leak-tagged chunk → buildChainEvent(tag=1073742079) resolves → candidate +found → canary exits → ReferenceChain event emitted. + +Rejected/deferred: raising STATIC_ANCHOR_ROTATION_BUDGET 16→32 (STW risk, +anchor walks already compete with expansion for the pass deadline); +killing the seeding noise (correctness issue from round 13, only ~10-20% +of frontier fill); whole-heap anything (user hard constraint). + +## Verification channels still open on pod + +- wrapper probe line each lap: `sweeping holder class ...ProfileAnalyzer; + at index 24627` → next laps should show wrapper_tag= parent=0 + root_kind=8 (admitted) — watch after any fix build too. +- `walkStaticFieldAnchors anchor ... class=Ljava/util/Collections$SynchronizedRandomAccessList;` +- `leak-tag intercepted`, `auto-marked chain`, `drainPendingChainEvents drained>0`, + datadog.ReferenceChain events. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round14.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round14.md new file mode 100644 index 0000000000..eb94599141 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round14.md @@ -0,0 +1,89 @@ +# Round 14: tiered anchor selection (leak → container → other) + restart hygiene (ev) + +Date: 2026-09-15, direction D (user-picked; C rejected on hard STW constraint), +implemented gated on the round-13 measurements and the cohort histogram ships +in the same build. + +## What was implemented + +1. **Tiered anchor selection** (collectStaticFieldAnchorsForRotation): + - Tier 0: leak-tagged anchors (entry.leak_tag != 0) — always selected + first, no cursor. + - Tier 1: container-shaped anchors (own class implements + java/util/Collection or java/util/Map) — cursor-fair within the tier. + - Tier 2: everything else (String/Class/boxed/enum/unknown-shape) — + cursor-fair, explicitly NOT guaranteed within one search lifetime. + - No within-call wrap: a tier stops at its lap end and leftover budget + flows to the next tier (re-walking covered anchors wastes budget). + - The deterministic coverage guarantee: a container anchor admitted at + any index position is walked within ceil(container_cohort / budget) + passes of admission — kills the admission-order lottery (round-13: + wrapper at position ~12-21k vs ~4k coverage). +2. **Class shape classification**: + - `_class_shape_cache` (class tag → CONTAINER/NON_CONTAINER), + process-lifetime (class tags are stable; NOT cleared on restart). + - `reconcileAnchorClassShapes` runs per pass before selection, on the + engine thread with JNI (outside heap callbacks/locks): up to + ANCHOR_SHAPE_RECONCILE_BUDGET=128 distinct unclassified anchor classes + per pass via ONE GetObjectsWithTags call, each classified by a + depth-bounded superclass+interface BFS against the cached + Collection/Map interface class tags. + - Index entries store the anchor's OWN class tag (parallel array), so + selection needs no JNI; classification lag demotes to tier 2 only + until classified. + - Class tags are NEGATIVE (namespace disjoint from positive frontier + tags) — all new code uses `!= 0` as the resolved/unresolved test. + - JNI local-ref hygiene: per-pop deletes + single-exit work-vector + cleanup + EnsureLocalCapacity(512). +3. **Index dedupe O(1)**: companion `_static_anchor_index_tags` hash set + (the old linear scan was O(n²) at the measured 28k population). +4. **Leak-tagged root-attached admits now indexed too** (interception + branch) — a static directly holding a tagged chunk is tier 0. +5. **Restart hygiene**: restartSearch() (and the test reset) now clears + `_candidate_discovered_tags/_count` — stale frontier tags previously + survived into the next search, resolving into zeroed slots + (reconstructChain failures, observed round 13) or live new-search + entries (WRONG-OBJECT chain events — likely the earlier "noise [B" + event). +6. **TEMP diagnostics (remove after pod verification)**: + - `anchorTierHistogram index=... leak_tier=... container_tier=... + other_tier=... budget=...` per pass — THE arithmetic gate (container + cohort must be ≲4k for within-one-search coverage). + - `reconcileAnchorClassShapes container class (class_tag=...)` — + names each newly classified container (once per class per JVM). + - Round-13 probe + klass_id walk log retained for verification. + +## Known prior art (documented in STALE_EXPANDED_ROTATION_BUDGET's comment) + +A class-shape priority tier was previously tried on the stale-expanded +collector and reverted: "no purely structural property can distinguish the +one specific container that is actually leaking from the thousands of +ordinary ones a real classpath contains." That argument does not apply to +anchor selection the same way — the tier is not trying to pick THE leaking +container, it is shrinking the coverage problem (all containers get covered +within ceil(cohort/budget) passes). BUT its premise (thousands of ordinary +containers) is exactly what the histogram measures: if container_tier ≫ 4k +on hotdog, this round's arithmetic fails too and the frontier-cap/lifetime +lever (config: _reference_chains_frontier_cap) is the next step. + +## Verification + +- 347 gtests OK (344 + 3 new: ContainerAnchorLeapsQueueAcrossLargeIndex — + 28k synthetic anchors, container at position 24k selected in the FIRST + call, leak-tagged anchor leads; AnchorOtherTierFairCoverageAcrossWraps — + 40/budget16 = 16+16+8 full coverage, no within-call duplicates; + RestartSearchClearsDiscoveredInstanceTags). Zero failures; only crash is + the pre-existing UrgentOOMProjectionBypassesCandidateGate SIGSEGV (clean + HEAD). +- spotlessApply clean; buildRelease -Pskip-tests clean. + +## Pod verification plan (next deploy) + +1. `anchorTierHistogram` — read container_tier vs ~4k. THE gate. +2. `reconcileAnchorClassShapes container class` — expect + Collections$SynchronizedRandomAccessList classified CONTAINER early. +3. `LEAK_BUFFER probe` — wrapper admitted root-attached (unchanged). +4. `walkStaticFieldAnchors anchor ... class=...SynchronizedRandomAccessList + klass_id=29448` — within ceil(cohort/16) passes of admission. +5. `leak-tag intercepted` + `datadog.ReferenceChain` events — the end goal. +6. Confirm `reconstructChain failed` lines GONE (stale discovered tags). diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round15-results.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round15-results.md new file mode 100644 index 0000000000..df90bb23d8 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round15-results.md @@ -0,0 +1,93 @@ +--- +id: ev-leaktag-onpod-round15-results +type: evidence +status: confirmed +depends_on: [find-tier1-tail-starvation] +related: [find-fresh-lane-verification, find-wrapper-demotion-self-parent] +tags: [pod-verification, round-15, fresh-lane, wrapper-walk, NEW-THIS-SESSION] +created: 20260916 +--- + +# Round 15 on-pod verification (ev) — first wrapper walk, first events; new gap measured + +Pod: NEW replacement `prof-analyzer-hotdog-jb-d9d8cf-cgdtx` (rz992 replaced by +the deploy; new JVM ~05:24Z 2026-09-16). Round-15 build f7b1ea416 verified in +the loaded .so (fresh_tier/fresh_queue strings). Streams (buffer rotates in +SECONDS at the current ~800k lines/min — the 10MB container cap cannot hold a +minute): /tmp/r15_stream.log (05:35-05:48), /tmp/r15_stream2.log (05:49+, +8M+ lines), tails in /tmp/r15_tail.log. + +## VERIFIED — the round-15 machinery works end-to-end up to the chains + +1. **Fresh lane**: `anchorTierHistogram … fresh_tier=1-11 fresh_queue=2-615 + container_tier=304-352 other_tier=8564-9205 budget=32` — the queue stays + ~one pass of admits (≪ 1024 cap), the lane keeps its picks. Arithmetic + gate passed. +2. **THE WRAPPER WAS ADMITTED ROOT-ATTACHED AND WALKED** (search #2, ~05:44Z): + `walkStaticFieldAnchors anchor tag=164392 + class=Ljava/util/Collections$SynchronizedRandomAccessList; klass_id=28366 + parent=0 root_kind=8 state=0 field_index=41` + the wrapper diagnostic line + (`wrapper anchor … holder_class=Lcom/…/steps/ProfileAnalyzer; field_index=41`) + + the round-12 walk diag: 21 entries — klass 64 (ArrayList), 28366 (wrapper, + seen_as=2 already-admitted), 8, then **17 [B chunks (klass_id=6) all + seen_as=0 (fresh plain admissions)** — `edges=19 truncated=0`. The walk + covered the whole subtree. (seen_as: 0=plain admission, 1=leak-tag + interception — none, the chunks were untagged at that instant.) +3. **The chunks were discovered, auto-marked, and got CHAINS**: discovered + slot filled (recordDiscoveredInstance klass 6 count=8), auto-marked chains + `tag=5748-5762` with `buildChainEvent … chain_size=6-12 depth=5-6 + root_kind=21` — deep multi-hop chains cached for the leak-class [B. + (root_kind 21 = the tracker's own root-kind enum, not the static 8 — the + chains root through a thread-root path; the STATIC shape is proven by the + walk diag instead.) +4. **datadog.ReferenceChain events are being emitted**: + `drainPendingChainEvents re-emitted=4` and `=6` on successive dumps — + cached chains are flowing into recordings for the first time on this pod. +5. Search #2 then hit the frontier cap and died; its tags/entries/chains were + wiped (tagLeakInstances tagged count collapsed 137→13 on + releaseSearchTags) and re-tagging restarted (tagged grew 12→36 — the + retained [B, which include the wrapper's chunks, now carry LEAK TAGS, + waiting for one interception). + +## NEW GAP (round-16 items, all measured live, in priority order) + +A. **Wrapper entry demotion — root-attached becomes chain-attached with + parent==its-own-tag**. Every holder-class crossing after search #2 + probes: `wrapper_tag=2469/2679/2853 frontier_found=1 parent= + root_kind=0 state=1 leak_tag=0 depth=1-3 referrer_klass=28366`. The + entry is no longer parent=0/root_kind=8, so the collector's eligibility + filter skips it — the wrapper is never re-walked (0 + SynchronizedRandomAccessList walk lines in stream2, across 3 holder-class + crossings). parent==self smells like a self-edge insert (improveChain? a + fanout/seed insert with parent==child?). FIND THE DEMOTER. +B. **B' at-risk FIFO cap-pinned by noise**: `pushAtRiskStaticAnchor … + fifo_size=988-1024` — klass 1 (1396 pushes), 1733, 215 (1063) flood the + 1024 cap; pushes for the wrapper's class 28366: ZERO (dropped at the cap). + The designed repair path for demoted static holders (the sweep re-encounter + push) cannot fire. Per-class quota or flood eviction needed (round-10's + known flood shape, now with measured per-klass counts). +C. **Leak correlation never fires**: zero `correlateAdmittedLeakTag` logs. + The emitted chunk chains carry target_tag = frontier tags (leak_correlated=0) + — real ReferenceChain events, but without the leakTag correlation the + backend cannot tie them to the leak (HeapLiveObject.leakTag). Correlation + fires via the state machine's frontier-tag branch — the wrapper's chunks + simply haven't been in the frontier while a poll ran since they were + tagged. ONE wrapper re-walk (fix A or B) while the chunks hold leak tags + converts everything: interception (seen_as=1) → entry.leak_tag set → + chain built with targetTag=leak tag → canary target resolves → events + correlate. + +The canary target (leak tag 1073741943, then later ones) still resolves +`not in frontier` — expected until the first interception. + +## Operational notes + +- Log flood ~800k lines/min (TEMP diagnostics × search churn × canary + polls) — rotate in seconds; verification MUST use `kubectl logs -f` + piped to files. Remove TEMP diagnostics after this round (the list is in + STATE.md). +- Search lifetimes observed: 44-100 passes (one search #3 lived 84+); sweep + laps ~10-20 min during class churn; holder class at index 24611-24874 + (drifts +60/lap with class churn). +- Class 1733's pushes at 988-1024: the flooders are klass 1/215/1733 — + one per-class quota would free the FIFO for the wrapper. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round15.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round15.md new file mode 100644 index 0000000000..a612125bc5 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round15.md @@ -0,0 +1,88 @@ +--- +id: ev-leaktag-onpod-round15 +type: evidence +status: confirmed +depends_on: [ev-leaktag-onpod-round14-results, find-tier1-tail-starvation] +related: [find-fresh-lane-verification] +tags: [implementation, fresh-lane, round-15, NEW-THIS-SESSION] +created: 20260916 +--- + +# Round 15: fresh-admission priority (queue lane) — implementation (ev) + +Direction from the round-15 diagnosis (see ev-leaktag-onpod-round14-results.md, +find-tier1-tail-starvation.md). Implemented same-turn, all gtests green +(348 OK: 347 + FreshContainerWalkedBeforeFairBacklog), spotless clean, +buildRelease -Pskip-tests clean. + +## Design (as implemented, after one discarded approach) + +**Fresh-admission queue lane** in collectStaticFieldAnchorsForRotation: + +- `std::deque _static_anchor_fresh_queue` (cap 1024, front-drop on + overflow): every addToStaticAnchorIndex() append pushes here; restart + and test-reset clear it. The queue is always a contiguous ordered SUFFIX + WINDOW of the anchor index (both append together; removals only ever + pop the front), which lets the drain recover each entry's index + position in O(1) by lockstep counting from idx_size - queue_len (with a + defensive std::find fallback if the invariant ever breaks). +- Selection order: leak-tagged (tier 0) → fresh (queue order = admission + order) → container fair cursor → other fair cursor. +- The drain pops EVERY entry (each anchor gets exactly ONE first look): + kept if eligible AND (container-shaped OR not-yet-classified) AND + budget room; dropped otherwise (dead/demoted/queued/leak-tagged are + dropped — the leak tier owns those via the index scan; classified + non-containers drop to the other tier; budget-exhausted drop to their + fair tier's ordinal). Nothing is ever lost — a spent first look only + removes urgency. +- Not-yet-classified rides the lane: the wrapper admits one pass before + reconcileAnchorClassShapes() can classify its class; unclassified is + bounded in practice (shape cache is JVM-lifetime; churn classes are + lambdas with no static fields). +- Fair-tier consumption skips fresh-kept tags (no double-select within a + call) but advances the cursor past them. +- `STATIC_ANCHOR_ROTATION_BUDGET` 16 → 32: the measured fresh-container + admit rate (~10-37/pass) exceeded 16, which would make the lane a + growing backlog — the same starvation one level down. STW safety does + NOT depend on the number: walkStaticFieldAnchors stops at the per-pass + deadline and requeues (selection size and walked size are decoupled — + observed on-pod round 10: "walked 6-16 of selected 20"). + +**Discarded first attempt — a positions-since-mark watermark:** simpler, +but a stale mark reorders selection when state leaks between units (the +StaticAnchorRotation* test caught it in test-order runs: a mark of 1 left +by the previous test demoted anchor 101 behind 104 — {101,104} became +{104,101}). The queue is per-anchor: each anchor's first look happens +exactly once, in admission order, regardless of stale state. Lesson +(reinforces find-test-seam-aliasing): singleton test-order sensitivity is +a design smell for production state that outlives a search. + +## TEMP diagnostics retained for the pod round + +- `anchorTierHistogram index=… leak_tier=… fresh_tier=… fresh_queue=… + container_tier=… other_tier=… budget=…` — fresh_queue is the PRE-drain + length (the admit-rate signal); fresh_tier is what the lane kept. THE + gate: if fresh_queue consistently ≫ budget, the lane runs as a backlog + and the wrapper waits fresh/budget passes (bounded, but belongs in the + next analysis); if fresh_tier is 0 while the wrapper admits, the lane + is not firing at all (shape/eligibility bug). +- Round-13 LEAK_BUFFER probe, round-12 walk klass_id logs, container + classification log: all retained. + +## Pod verification plan (next deploy, user builds + deploys) + +Use `kubectl logs -f` streams (buffer rotates ~4 min at the current 16k +lines/min; --since windows burned ~20 min in round 14): + +1. `anchorTierHistogram` — fresh_queue vs 32 budget (the gate above). +2. `LEAK_BUFFER probe … sweeping holder class …ProfileAnalyzer;` → + `wrapper_tag=` (admitted, this JVM's crossing) — next lap after + the crossing, wrapper_tag should be nonzero when a search is live. +3. `walkStaticFieldAnchors anchor … class=…Collections$SynchronizedRandomAccessList` + within 1-2 passes of the admission (the fresh lane working). +4. `leak-tag intercepted` / `auto-marked chain` / + `drainPendingChainEvents drained>0` / datadog.ReferenceChain events. +5. `buildChainEvent false: target_tag=1073742079` should STOP once the + wrapper's subtree expansion admits the tagged chunks. +6. If the wrapper still never admits: the sweep gate (class churn) has + not reopened around a live search — watch static_sweep_gate lines. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round16-results.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round16-results.md new file mode 100644 index 0000000000..eee6c45598 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round16-results.md @@ -0,0 +1,76 @@ +--- +id: ev-leaktag-onpod-round16-results +type: evidence +status: confirmed +depends_on: [find-wrapper-demotion-self-parent] +related: [find-round16-endgoal-verification, find-refchains-log-flood-configurable] +tags: [pod-verification, round-16, END-GOAL, leak-tag-interception, control-run, NEW-THIS-SESSION] +created: 20260916 +--- + +# Round 16 on-pod verification (ev) — LEAK-CORRELATED ReferenceChain events flowing + +Pod `prof-analyzer-hotdog-jb-d9d8cf-cgdtx` (ns `profiling-stg`), build +`4afe870a2` verified in the loaded .so (self_edge_skips/quota_drops +markers; JVM pid 46655 from 07:45:11Z 2026-09-16). Stream: +`/tmp/r16_stream2.log` (~24k lines/min at steady search activity; the +~800k/min flood of round 15 was the canary-chase TEST_LOG spam, now +quieter because the chase is healthy). + +## Accidental control run first (worth keeping): healthy app = dormant tracker + +The first redeploy of this round put the pod on the NON-LEAKING app +image (v137233187): 3h of a healthy app, round-16 .so loaded, and +ZERO reference-chain activity — no candidates, no searches, only +boot-level + liveness logs. The candidate/generations gate held the +whole machinery dormant on a healthy app exactly as designed. (User's +observation; recorded as a production-readiness data point.) + +## All round-16 gates verified live + +1. **Fix A (self-edge guard) live at scale**: `self_edge_skips=30816+` + and climbing — the guard refusing real `mutex == this` self-edges on + every Synchronized* collection the walks touch. The LEAK_BUFFER + wrapper itself: admitted ROOT-ATTACHED (`walkStaticFieldAnchors + anchor tag=2184 class=Ljava/util/Collections$SynchronizedRandomAccessList; + klass_id=28516 parent=0 root_kind=8`) and WALKED as a collector + anchor — the round-15 demotion shape (parent==self root_kind=0) is + gone. The round-12 wrapper diag also fires for OTHER + synchronized-list statics (e.g. `ProcessTags$Lazy` field 3) — the + app has many, which is why the skip count is 30k+. +2. **Fix B (per-class quota) live**: `static_anchor_fifo_quota_drops_ + total=6351` while `static_anchor_fifo_size=132` — the floods are + back (6.3k quota-dropped pushes) and CONTAINED (round 15: FIFO + pinned 1024/1024, wrapper's pushes dropped; now 16 drained/pass, + room always available). +3. **THE INTERCEPTION CASCADE — live for the first time in the whole + investigation**: `heapReferenceCallback leak-tag intercepted: + leak_tag=1073742064..77 -> frontier_tag=186955+ depth=3 + parent_tag=` — 12+ leak-tagged [B chunks (the real leak + instances, tags ≥ LEAK_TAG_BASE) admitted into the frontier WITH + their leak tags and parent chains. Never fired before round 16. +4. **Leak-correlated chains cached and emitted**: + `pollWatchedTargets auto-marked chain for klass_id=5 … + target_tag=1073742071/72/77` (leak-tag targets, not frontier-tag + noise) — 6 leak-tag chains in `_resolved_chains`; + `drainPendingChainEvents re-emitted=6 -> 14` sustained on every + dump. datadog.ReferenceChain events tied to the leak's HeapLiveObject + leakTags are flowing into recordings. +5. Fresh lane intact (`anchorTierHistogram … fresh_queue=12` on the + first search); leak candidate live (`candidate[0] klass_id=5` = + [B in this JVM; wrapper klass 28516). + +## Still open (watch, not blockers) + +- **Canary 0/1**: the chase still hunts its one pre-tagged + representative instance (backoff_mult=16, ema ~190ms). The + intercepted instances so far are other pool-assigned chunks; the + representative is in the same LEAK_BUFFER population, so per-pass + interceptions should reach it — round-15's chase DEADLOCK (chase + needing the never-happening walk) is broken (walks + interceptions + every pass). When it exits, the search completes and restarts clean. +- `buildChainEvent false: reconstructChain failed` lines exist for + small frontier-tag targets (noise instances whose entries died) — + the leak-tag chains build fine. +- TEMP diagnostics removal once this verification is accepted — the + remaining log flood is ours. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round2.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round2.md new file mode 100644 index 0000000000..95c94fcb6d --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round2.md @@ -0,0 +1,57 @@ +--- +id: ev-leaktag-onpod-round2 +type: evidence +status: verified +related: [find-ema-batch-collapse, find-priority-queue-starves-bfs-crawl, find-leak-tag-pool-implementation, find-leaktag-jfr-field-misalignment] +tags: [pod-logs, on-pod, post-fix, live-verification, aimd, leak-tag, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# On-pod round 2 of the leak-tag redesign (build 0db70994d, PID 48355) + +Two deploys happened: JVM 44624 (11:11Z) was still the round-1 build — its +logs showed the old `ema=` per-tag format; the user redeployed again and +JVM 48355 (11:24Z) had the round-2 format (`ema_call_ms=`/`next_batch=`). + +## What worked (round-2 fixes verified) + +- **AIMD**: no collapse. Oscillates correctly around the 25ms per-call + budget: `batch_size=34 gotw_ms=24 ema_call_ms=25 next_batch=17` → halve; + `batch 17 → next 81` → +64. Batch settles ~17-80 because the O(tag_map) + call floor (~20-25ms at ~200k entries) dominates; AIMD converges to the + largest affordable batch, as designed. +- **Cadence**: ~12 runPass/min (was 0.2/min), ~5 gotw calls/pass, + deadline-bounded; no more 10s-CPU passes. `pendingExpand=66k` though. +- **Pool tagging**: tagged up to 129 instances, by age-diversity priority. +- **Emergency**: `emergency=0` but multiplier 15× firing correctly on + candidate progress (no false emergency). +- **Drains**: `Profiler::dump reference-chain batch=8 write_dropped=0` x3. + +## What failed → find-priority-queue-starves-bfs-crawl + +- `_priority_expand` 39k→103k in 20 min; `_pending_expand` never drained + (priority-first drain); zero `intercepted:` logs; zero + `auto-marked chain ... leak_tag=`; only the 8 depth-1 noise chains + (cached, re-emitted); `leak_accumulation_tags=0` every pass; + `leak_parents=19058` unreachable; `isQueuedForRotation` linear scan over + the 103k queue under the frontier lock (hidden per-pass cost). + +## Operational lessons + +- **Build identification via distinctive TEST_LOG strings**: concatenate + split source string literals before grepping pod logs — the interception + log is "heapReferenceCallback leak-tag intercepted: ..." in output but + "leak-tag " + "intercepted:" in source; grepping "leak-tag intercepted" + for the OUTPUT is correct, but earlier confusion came from grepping for + a field name that changed (`ema=` vs `ema_call_ms=`). Verify the deployed + build by a log field that prints EVERY iteration (e.g. the gotw line), + not an event-driven one. +- Pod chunk JFRs still show no `datadog.ReferenceChain` / + `datadog.HeapLiveObject` types in jafar (both "Event type not found") — + the chains appear only in the uploaded/merged recording, same as round 1. +- Chunk dir path: `/tmp/ddprof_root/pid_/jfr/_/`, only + the last ~6 minutes of chunks retained (grab immediately). + +Fix round 3 = commit f4c73ba0f (see find-priority-queue-starves-bfs-crawl). +Awaits redeploy. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round3.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round3.md new file mode 100644 index 0000000000..e6e80b582e --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round3.md @@ -0,0 +1,107 @@ +--- +id: ev-leaktag-onpod-round3 +type: evidence +status: verified-partial +related: [find-ema-batch-collapse, find-rotation-resize-blindspot, find-depth0-durable-root-upgrade-gap, find-leak-tag-pool-implementation] +tags: [pod-logs, on-pod, post-fix, live-verification, proportional-batch, leak-tag, NEW-THIS-SESSION] +created: 20260901 +updated: 20260901 +--- + +# On-pod round 3 (build 663784137, JVM 75258, started 07:03:38Z) + +Deploy note: first "redeploy" (07:03:29Z) restarted the JVM but left the +Aug-23 ddprof binary in place - the extracted lib had ZERO round-2+ markers. +After the real binary copy, JVM 75258 loaded the 663784137 build (verified: +requeueChainRootForRotation=4, maybeUpgradeRootAttachedRootKind=4, +ema_call_ms, static_sweep_gate all present in the extracted .so). + +## Verified working on pod (2.5h window) + +- **Proportional batch control**: gotw batch 8-14, gotw_ms 18-21, + ema_call_ms 19-20, next_batch 8-14 - stable, no AIMD collapse, no + collapse to MIN. ~12 passes/min. +- **Static sweep**: full laps complete (resolved=33504 swept=33504 + cursor=0, repeatedly) - chunked sweep covers all 33.5k classes. +- **Depth-0 durable-root upgrade (NEW)**: 76 fires - 9(CONSTANT_POOL)->25 + (THREAD), 25->21(JNI_GLOBAL) - the root-like-edge upgrade path works in + production. +- **Leak-tag pool**: 247 tagLeakInstances tagged lines (klass_id=4 [B], + tids 7525x/7565x, ages 31-156s, sizes 24-16016, need_set=0 - stable + waiters, pool not churning). +- **Discovered loop**: finds 8 live candidate instances per poll, + discoveredCounts=[8,0,0,0,0]. +- **Noise gate + emission**: buildChainEvent built THREAD(25)-rooted + depth-1 chains for frontier-admitted byte[]s; transient (24) ones + suppressed; TWO ReferenceChain events drained to JFR + (`Profiler::dump reference-chain batch=1 write_dropped=0` x2, same + target 6215 re-emitted from cache). ReferenceChain emission verified + end-to-end ON POD for the first time. +- **Fanout tracking**: leak_parents=37619, fanout-inserts flowing + (child_class_tag=-2 = byte[]). + +## Not firing (the remaining last mile) + +- `intercepted:` = 0, `recordDiscoveredInstance` = 0, + `auto-marked chain` = 0, `requeueChainRootForRotation` = 0 (log line + count; the function early-returns before its TEST_LOG when the tag + isn't in the frontier). +- Recurring failure: `pollWatchedTargets buildChainEvent(tag=1073742076) + -> 0` + `buildChainEvent false: target_tag=... not in frontier`. + +## Root cause (evidence-backed) + +The tagged instances and the frontier-admitted instances are DISJOINT +sets. Both bridges from "leak tag on object" to "frontier entry" require +the crawl to touch the object: +- interception fires in heapReferenceCallback only when the walk + delivers an edge whose TARGET is leak-tagged; +- correlateAdmittedLeakTag fires only if the object already carries a + frontier tag at tagging time. + +On this workload the crawl admits ~15-40 edges/pass at ~12 passes/min +(~200-500 edges/min) against a 199k frontier; the 18 tagged instances +(age-priority-selected old machinery byte[]s - sizes 24-16016, the same +shape the local scenario's TEMP comment documented: "age-priority kept +re-selecting old machinery byte[]s while the young leak chunks churned +out") sit under holders the crawl has not reached in 2.5h. + +KEY OBSERVATION for the fix design: the static sweep completes FULL +33.5k-class laps - raw JVMTI heap-walk throughput is NOT the wall; the +crawl's GetObjectsWithTags batch resolution + frontier admission is. A +dedicated intercept-only sweep (full-graph walk that reacts ONLY to +leak-tagged targets: record + admit that one edge, no batch tag +resolution, no frontier growth) would cover the whole graph per lap at +sweep-like cost and bridge the disjoint sets deterministically. + +Also notable: the pod's own container was OOMKilled at 04:00:30Z (the +07:03 JVM start was the recovery) - there is real memory pressure on this +analyzer; [B as the leak candidate is plausible, but the tagged selection +landing on machinery byte[]s suggests the [B growth signal is dominated +by machinery churn, not one growing collection. + +## CORRECTIONS + option retraction (user challenges, all verified) + +- Elapsed was 32 MINUTES (JVM 75258 started 09:13:29Z), not 2.5h - the + 2.5h log window covered the stale JVM 44911 too. All counts (76 + upgrades, 247 tagged, 0 intercepted, 2 emitted chains) are from ~32 + min. +- Pass rate: passesRun=2803/32min ~= 88 passes/min (NOT ~12 - that was + round-2's pre-proportional number). At ~15 edges admitted/pass that is + ~1300 edges/min, ~40k edges in 32 min vs a 199k frontier. Disjoint-set + observation stands, margin ~5x less dire than first stated. +- The tracker is in BASIC/canary-search mode: `blocked` (cpu pain + budget) never fires; `shouldRunPass -> true (canary search, 0/1 + candidates found)` every iteration - the unfindable candidate[0] marker + forces max cadence (the user's 30%+ CPU burn; see + find-canary-search-forces-max-cadence). +- Option B (intercept-only full-graph sweep) RETRACTED: FollowReferences + is a VM operation - a full-heap lap is one ~10s stop-the-world pause + (find-jvmti-heap-walk-stw-vmop, jdk21 jvmtiTagMap.cpp:2379/:2934). The + static sweep argument was invalid: it is a RESTRICTED walk over a tiny + subgraph, not evidence about full-graph walk affordability. +- C does NOT reduce B's cost (irrelevant now) - but C attacks the current + burn: retiring a churn candidate ends the forced canary-search + cadence. Remaining lever for real candidates is crawl throughput: + the per-pass cost is dominated by O(tag-map) GetObjectsWithTags scans + (18-21ms x ~5/pass), not the walk itself. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round4.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round4.md new file mode 100644 index 0000000000..ffc5d9633f --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round4.md @@ -0,0 +1,75 @@ +--- +id: ev-leaktag-onpod-round4 +type: evidence +status: confirmed +depends_on: [find-per-tid-qualification-design, ev-leaktag-onpod-round3] +related: [find-canary-search-forces-max-cadence, find-getobjectswithtags-quadratic-bottleneck, q-coverage-tracking-per-combination] +tags: [pod-verification, option-C, per-tid, crawl-throughput, NEW-THIS-SESSION] +created: 20260901 +updated: 20260901 +--- + +# Pod round 4: per-tid gate works exactly as designed; interception still blocked by crawl throughput + +Build c5490156e (per-tid qualification + PriorityExpandSet), JVM 120642, +13 min into the run, logs in build/logs local copy /tmp/hotdog-round4.log +(58257 lines). Binary verified via marker strings in the extracted lib +("klass trend OK but no qualifying tid - skipped", "selectLeakCandidates +entry", "PriorityExpandSet"). + +## What works (option C confirmed on-pod) + +- selectLeakCandidates logged 183 "klass trend OK but no qualifying tid - + skipped" (klass_id=7 among them: klass slope 1.40, consecutive 9 - a + genuine klass-level rise rejected because no single thread qualifies). +- Exactly ONE candidate qualifies: klass_id=2 ([B), and all pool tags go + to ONE thread: tid=120852, 75 MiB byte[]s (size=78643216), ages 4-18 + rising. Compare round 3: 247 tags scattered over machinery byte[]s + (16-24KB, many tids, flat retention). Pool pollution eliminated. +- Poll cadence: candidate[0] klass_id=2 tag=1073742079 needRefresh=1 + each pass; tagLeakInstances tagged=5 per poll, need_set=0 (stable + waiters, never admitted). + +## What still fails (unchanged from round 3) + +- Zero interceptions ("leak-tag intercepted" absent), zero + recordDiscoveredInstance/requeueChainRootForRotation, + candidateFound=0/1 every pass -> shouldRunPass -> true (canary search) + on all 183 passes (forced cadence persists). +- buildChainEvent false: target_tag=1073742079 not in frontier - all + 183 polls. + +## New precise numbers (the throughput wall, quantified) + +- frontierSize=242106 (grew from 199k in round 3); pendingExpand=126,895 + CONSTANT (never meaningfully drained); priorityExpand=1016. +- expandFrontier: 877 gotw calls, batch 8-9 (GOTW_MIN regime), resolved + =batch, edges=0 on 701/877 calls; total edges admitted ~1389 in 13 min + (~7.5/pass, occasional 100+). +- gotw_ms: min 14 / median 24 / p90 34 / max 41 - the O(tag-map) floor at + a 242k-entry map, EXCEEDING the ~10ms pass expand window, so the + proportional control ratchets to MIN_BATCH and each pass fits only + ~2-4 calls. +- Backlog drain rate ~120-200 objects/min vs 127k backlog = ~10-17 + hours per lap; the leak holder (elementData-style Object[] of the + growing list) sits unexpanded in that backlog, so the tagged byte[]s' + incoming edge is never enumerated post-tagging -> interception never + fires. PROVEN not-retagged: need_set=0 with leak tags means SetTag ran + while the objects were unvisited (tag 0), and zero intercepts since. +- Pass rate DROPPED: 183 passes/13min = ~14/min vs 88/min in round 3 + (inferred ~4.3s/pass; per-pass duration decomposition needs a TEMP + timing log - static sweep full 33007-static lap happens every pass, + cursor=0 each time). +- rotation_candidates every pass: root_kind_tags=16 + leak_accumulation_tags=0 stale_expanded_tags=0-9, + leak_signatures=476->487, leak_parents=50248->51033. + +## Root cause of the dead targeted tier (code-confirmed) + +collectLeakAccumulationCandidatesForRotation (referenceChains.cpp:2482) +Tier 2 only selects parents whose frontier state is EXPANDED +(:2527-ish lookup + entry.state == EXPANDED check). The growing holders +of the watched leak klass are freshly admitted FRONTIER-state backlog +entries that never get expanded (starvation above) - so the tier that is +supposed to target exactly the accumulation point selects ZERO every +pass while holding 51k known parent candidates. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round5.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round5.md new file mode 100644 index 0000000000..6dbed81647 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round5.md @@ -0,0 +1,73 @@ +--- +id: ev-leaktag-onpod-round5 +type: evidence +status: confirmed +depends_on: [ev-leaktag-onpod-round4, find-canary-lane-backoff-design, find-admission-boost-implementation, find-togcroot-orphaned-slot-stranding] +related: [find-priority-queue-starves-bfs-crawl, q-resize-instrumentation-rescan-priority] +tags: [pod-verification, round-5, backoff, admission-boost, crawl-throughput, NEW-THIS-SESSION] +created: 20260902 +updated: 20260902 +--- + +# Pod round 5: all machinery fixes verified live; interception still blocked by crawl reach + +Accumulated build (A+B+backoff+boost+orphan sweep, post-7c37ee833), +deployed by the user (in-place, pkill). JVM 6600, ~20 min in, log tail +~13 min (container 10MB rotation - the per-tag tagLeakInstances flood +re-logs all 256 stable tags every poll, ~600k lines/20min). Extracted +lib marker-verified: backoff, Tier-2 state=%s selection, watched tids, +orphan sweep, per-tid gate, PriorityExpandSet - all present. + +## Verified live this round + +- Canary backoff: mult=16 x ema_ms=184-190 -> ~3s spacing, ~292 + passes/20 min, burn structurally bounded. (Round 3: back-to-back 1 + core for 32 min.) +- Fix B (gotw window widening): batch_size=196-512 under a 107k + pendingExpand backlog with ema_call_ms=63-65 - no MIN-collapse + (round 4 collapsed to batch=8). +- Fix B+backoff pass walls: pass ema_ms=190 (round 4: 0.7-4s); passes + admit real edges again (100-198/pass, truncated=1). +- Admission boost: noteSelectedCandidates watched tids=859 polls, + tid=6809 published throughout. +- Pool economy PERFECT: every tag line is klass_id=6 tid=6809 + size=78643216 (the real 78MB leak chunks, ages 64-75); 656 polls + tag the full 256-pool + 203 tag 16; zero machinery noise (round 3: + 247 scattered tags). +- Orphan sweep: live, 0 firings - candidate qualifies every poll + (correct; sweep is the fallback). + +## Still blocked (the one remaining gap) + +- Interception 0, need_set=0 (stable waiters never admitted): + the 256 tagged real leak chunks are never ENUMERATED - they sit + under holders the crawl has not reached. pendingExpand=107,526- + 107,547 (drains ~10-20/pass = hours); static sweep cursor advances + 512/pass (lap ~67 passes, multiple laps done) but the flood of + machinery admissions keeps the queue buried. +- Fix A (FRONTIER-state Tier-2): 48 selections all state=EXPANDED + (fanout 648-2048 - machinery arrays; [B is the watched class so any + byte[] admission builds fanout). 0 FRONTIER selections: the leak + holder has ZERO leak-accumulation edges because its children (the + tagged chunks) were never admitted - chicken-egg: edges exist only + after expansion; selection needs edges. The signature path needs at + least one admitted (leaf,parent) edge of the real leak to exist. +- discoveredCounts=[8,...]: 8 noise [B recorded as the candidate's + "discovered instances" - [B is too broad a candidate class on-pod; + first-8-fresh-[B-wins, none are the leak chunks. No chains built + from them (no auto-marked-chain lines in the tail). + +## Conclusions / next actions + +1. All landed machinery is verified live and behaving exactly as + designed; the remaining blocker is pure crawl REACH - the same + diagnosis as round 4, now with every other factor eliminated. + This is the strongest evidence yet for the deferred option C + (candidate-scoped frontier: prioritize expansion toward the + candidate's holders/tids rather than FIFO over the whole graph). +2. The tag flood: make tagLeakInstances log per-poll summary, not per + instance (or skip re-logging stable tags) - it rotates the + container log every ~10MB and cost us log-head this round. +3. Re-check the pod after the run accumulates (1-2h): if interception + lands as the backlog drains, option C is a throughput win, not a + correctness gap; if not, it is the correctness gap. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round6.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round6.md new file mode 100644 index 0000000000..5985dc69f6 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round6.md @@ -0,0 +1,59 @@ +--- +id: ev-leaktag-onpod-round6 +type: evidence +status: confirmed +depends_on: [ev-leaktag-onpod-round5] +related: [find-per-tid-qualification-design, q-resize-instrumentation-rescan-priority] +tags: [pod-verification, round-6, crawl-reach, option-C, NEW-THIS-SESSION] +created: 20260902 +updated: 20260902 +--- + +# Pod round 6: full sweep lap + 31 min, interception still zero - breadth-first reach is the correctness gap + +Build c7f207a51 (round 5 + per-poll tag summary), JVM 18354, ~31 min, +summary log fix verified live (one summary line per poll: "summary +klass_id=4 tid=18565 tagged=15 ... max_size=78643216" - no per-instance +flood; log-head survived the whole window). + +## Verified this round + +- Per-poll tag summary works on-pod; leak thread identified: tid 18565, + 78MB chunks, ages 7-129. +- Per-tid gate + warmup: candidate qualified after ~20 min (klass 4 + trend consecutive_positive=30 with many short-lived tids at + age_count=1 skipped; the leak tid crossed the age-trend cusp). +- Backoff: mult=16 x ema ~180-195ms, 407 held-offs - pacing as designed. +- Tier-2 re-walk: 1351 selections (all EXPANDED, mostly fanout=1 + machinery); static sweep completed ~1 full lap (cursor 30666/33275, + 512/pass); passes admit 146-645 edges each. + +## The verdict: breadth-first reach is a correctness gap, not a throughput gap + +- pendingExpand ~98.5k, NET GROWING (193k -> 194k frontier in 3 passes): + the sweep's per-lap re-admissions replenish the queue as fast as the + crawl drains it. The backlog will NEVER drain on a rising heap. +- Interception 0 after 31 min + a full sweep lap. The tagged chunks + (need_set=0, ages up to 129) sit under a holder that was admitted + (sweep lapped past the app's classes) but never EXPANDED - buried + behind ~98k machinery entries. +- Rounds 5+6 with all machinery verified live (backoff, boost, pool + targeting, Tier-2, gotw widening, summary) eliminate every other + factor. Option C (candidate-scoped frontier) is now a CORRECTNESS + requirement, exactly as predicted in ev-leaktag-onpod-round5. + +## Option C concrete sub-moves (for the design discussion) + +- C1 THREAD-SCOPED WALK: the candidate's allocating thread objects are + known (qualifying tids -> live Thread objects). A FollowReferences + starting at the Thread OBJECT (not the whole heap) is a BOUNDED walk + of that thread's reachable graph (thread-locals map, its Entry[] + arrays, the tagged chunks) - tiny STW cost, and heapReferenceCallback + intercepts the leak tags during it. Directly serves the user's leak + taxonomy "thread-local leaks via Thread objects". +- C2 STATIC-HOLDER PRIORITY: for static-held leaks, the sweep re-admits + the holder's CURRENT backing array every lap - those fresh + root-attached entries should take a priority lane (durable root_kind + + admitted-since-last-pass) instead of FIFO behind 98k machinery. +- Unknown which shape hotdog's simulated leak uses; the two prongs + cover both taxonomy categories. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round7.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round7.md new file mode 100644 index 0000000000..7887b0aa1c --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round7.md @@ -0,0 +1,79 @@ +--- +id: ev-leaktag-onpod-round7 +type: evidence +status: confirmed +depends_on: [ev-leaktag-onpod-round6, find-option-c-descend-walk-design, find-field-name-decoding] +related: [find-priority-queue-starves-bfs-crawl] +tags: [pod-verification, round-7, descend-walk, interception-zero, depth-cap, NEW-THIS-SESSION] +created: 20260902 +updated: 20260902 +--- + +# Pod round 7: both descend-walk prongs live, interception still zero - holder sits deeper than the 6-hop cap or outside the covered root kinds + +Deploy: 13d3f87b3 (descend walks + retention-edge labels), deployed in +place by the user on REPLACEMENT pod srwnz (the original zng9s was +EVICTED - my fault: an in-pod jcmd GC.heap_dump wrote a ~5GB hprof to the +node overlay and the container got OOM-killed (exit 137) then the pod +evicted; the in-place patch died with it. DO NOT dump in-pod on these +pods). JVM 4818, leak thread tid 5025, klass_id 7, 22 tagged chunks +(max_size 78643216), need_set=0 stable, ~1500+ passes over ~40 min. + +## Verified live this round + +- Prong 1 (walkCandidateThreadLocals): candidates=1 tids=1 walked=1 + every pass; first pass admitted the ThreadLocalMap structure (15 + edges), then idempotent 0s - and registerExistingThreads() worked: the + leak thread predates the recording and IS registered (the round-7 + pre-flight catch: walked=0 before the sweep existed). +- Prong 2 (walkStaticFieldAnchors): exactly 4 root-attached STATIC_FIELD + anchors exist (sweep lapped all 33622 classes; the low count is itself + a signal - see below), all 4 walked every pass, single walks admitting + up to 3119 edges (deep subgraphs - executor-task-shaped structures). +- Edge labels shipped (chains will carry field names once one is found). +- Tagging healthy: tagLeakInstances summary per poll, tagged=22 stable, + ages growing (retained), need_set=0. +- BFS still can't drain: pendingExpand 73.0k -> 73.6k net-growing, + frontier 113k->114k; gotw healthy (batch 512, ema_call_ms 29-33). + +## The verdict: interception 0 with EVERYTHING live + +The tagged chunks are neither (a) in the leak thread's ThreadLocalMap +(thread walk covers that subgraph fully and idempotently) nor (b) +within DESCENT_HOPS=6 below any root-attached static (all 4 anchors are +descend-walked every pass, wholesale, thousands of edges). They are also +not frame-locals (the leak thread parks idle between task runs - a +frame-local accumulator would be collectible while parked; tagged chunks +survive GCs, ages to 236). + +## Remaining shapes (hypotheses, not yet proven) + +1. DEEPER THAN 6 HOPS below a static anchor - the leading hypothesis: + the leak task graph shape is static ExecutorService(0) -> + DelayedWorkQueue(1) -> q[](2) -> ScheduledFutureTask(3) -> task(4) + -> accumulator(5) -> list(6) -> chunks(7) - the 3000+-edge static + walks admitting exactly that kind of task graph but stopping at the + cap. The holder interior (accumulator at depth 5) IS likely admitted + already. +2. A root kind outside both prongs (JNI_GLOBAL / MONITOR / another + thread's stack root) - no current evidence for or against. + +## Fix candidates (user picked BOTH - implemented in c6635fe0e, pending round 8) + +- Minimal, taxonomy-consistent: raise DESCENT_HOPS (6 -> 16); the + walk is already deadline-bounded per slice, so cost is unchanged + structurally - this is a one-constant rebuild+redeploy. +- Same-pattern extension: extend + collectStaticFieldAnchorsForRotation's filter to JNI_GLOBAL roots, + same wrapping-cursor bounded walks. + +En route to that deploy, one slow-suite run failed +shouldReconstructReferrerChainThroughUnboundedCacheLeak with +candidateFound=0/0 and ZERO tagLeakInstances calls - the candidate never +qualified at all, which the walk changes cannot influence (they only act +after a candidate exists). Green on rerun and in the full family pass: +the KNOWN intermittent hysteresis family (q-togcroot-acceptance-paths), +not a regression. + +Pod log retention is ~30s at current volume (3.2M lines/6min) - use +kubectl logs -f streaming for evidence windows, not --since. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round8.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round8.md new file mode 100644 index 0000000000..bcb28a6a4f --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round8.md @@ -0,0 +1,82 @@ +--- +id: ev-leaktag-onpod-round8 +type: evidence +status: confirmed +depends_on: [ev-leaktag-onpod-round7, find-option-c-descend-walk-design] +related: [find-field-name-decoding, find-depth0-durable-root-upgrade-gap] +tags: [pod-verification, round-8, descend-walk, interception-zero, leak-buffer, NEW-THIS-SESSION] +created: 20260902 +updated: 20260902 +--- + +# Pod round 8: 16-hop + JNI_GLOBAL walks live, interception STILL zero — retention shape is a textbook prong-2 static List, so the suspect is which anchors enter the tier + +Deploy: user redeployed the latest build (post-c6635fe0e: DESCENT_HOPS +16 + JNI_GLOBAL anchor tier) on RENAMED pod +`prof-analyzer-hotdog-jb1-668df5bcff-f75l8` (profiling-stg; pod name +changed from hotdog-jb to hotdog-jb1), JVM 4445, agent +`1.66.0-SNAPSHOT~e188d0ff7f` (datadog.ProfilerSetting in-chunk). + +## Verified live this round + +- Tagging healthy: `tagLeakInstances summary klass_id=5 tid=4655 + tagged=8 need_set=8→0 max_size=78643216`, ages up to 317 (new pod → + new klass/tid ids vs round 7's 7/5025). +- Prong 1: `walkCandidateThreadLocals candidates=1 tids=1 walked=1 + edges_admitted=0` every pass (idempotent after the ThreadLocalMap + admission). +- Prong 2: `walkStaticFieldAnchors selected=4 walked=4 truncated=0` — + per-pass edges_admitted spans 10 → 3608 (rotation cycling through + different anchors; large walks are first-encounter passes, small ones + re-walks of already-admitted subgraphs). +- `intercepted=0` across both streaming windows (23k and 271k log + lines). runPass line sample: `edges_admitted=1597 frontier=132307 + pendingExpand=121519 priorityExpand=1008 candidateFound=0/1 + discoveredCounts=[8,0,0,0,0]`. +- JFR chunks (`/tmp/ddprof_root/pid_4445/jfr/2026_09_02_16_33_35_4445/`): + ZERO `datadog.HeapLiveObject` events in ANY chunk — liveness samples + are not flowing to the recording (BCI_LIVENESS path in + flightRecorder.cpp:2471 exists; why nothing emits is UNRESOLVED — + secondary, see q-heapliveobject-absent-on-pod-chunks). No + ReferenceChain events either (expected: zero interceptions). + +## The app's actual retention shape (found in the pod's own bytecode, diagnosis-only) + +`com.datadog.profiling.analyzer.steps.ProfileAnalyzer`: + +```java +private static final List LEAK_BUFFER; // static final field +``` + +- Initialized via `Collections.unmodifiableList(...)` (javap: the + ``-region `invokestatic java/util/Collections.unmodifiableList`). +- Fed by a dedicated `new Thread(runnable, "simulated-memory-leak")`. +- Self-trims at a heap-usage threshold: `List.clear()` + `System.gc()` + + log "simulated-memory-leak: heap usage {}/{} ({}%) >= {}%, dropping + {} chunks". +- Taxonomy: 3-4 hops from the root-attached static (wrapper → backing + ArrayList → elementData array → byte[] chunks). This is EXACTLY + prong 2's shape — the 16-hop walks over every root-attached static + should reach the tagged chunks and intercept. + +## The verdict + +Local scenario intercepts the identical static-List shape (machinery +sound in-process). Both prongs live and un-truncated at 16 hops, all +4 root-attached static anchors walked every pass, anchor rotation +cycling — yet zero interceptions. The remaining suspect is WHICH +entries enter the anchor tier on the pod: +1. `LEAK_BUFFER`'s wrapper (java.util.Collections$UnmodifiableList) + never admitted root-attached (parent_tag==0, root_kind=STATIC_FIELD), +2. admitted but root_kind misclassified, or +3. admitted root-attached then REPLACED by a chain-attached entry + (improveChain/reparentToDurableRoot both produce parent_tag!=0 + entries — plausible: the wrapper is reachable via many chains). + +Per-anchor diagnostic (committed c9a57f681, TEMP): `walkStaticFieldAnchors +anchor tag= class= parent= root_kind= state= field_index=` per walked +anchor — will name the wrapper's absence/misclassification directly. + +Operational: commit signing (1Password SSH agent) failed at session end +("communication with agent failed" for every agent key); worked again +after the agent recovered. The commit sat staged overnight. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round9.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round9.md new file mode 100644 index 0000000000..cd99ba54e0 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-leaktag-onpod-round9.md @@ -0,0 +1,76 @@ +--- +id: ev-leaktag-onpod-round9 +type: evidence +status: confirmed +depends_on: [ev-leaktag-onpod-round8, find-option-c-descend-walk-design] +related: [find-field-name-decoding, q-heapliveobject-absent-on-pod-chunks, find-anchor-holder-eviction] +tags: [pod-verification, round-9, anchor-tier, edges, heapliveobject, events-flowing, NEW-THIS-SESSION] +created: 20260903 +updated: 20260903 +--- + +# Pod round 9 (c9a57f681 diagnostic): anchor tier = 76 machinery statics, NO holder; events ARE flowing end-to-end in uploaded recordings + +Deploy: c9a57f681 on pod `prof-analyzer-hotdog-jb1-668df5bcff-7h5n9` +(Renamed again; JVM 77972, agent .so extracted 09:43 UTC, contains the +diagnostic string — build verified BEFORE interpreting logs). + +## Anchor diagnostic answer (the round-9 question) + +`walkStaticFieldAnchors anchor tag=... class=...` lines: the tier cycles +through exactly **76 root-attached anchors** (selected=4/pass, wrapping +cursor), all `parent=0 root_kind=8 state=1(EXPANDED)`, and ALL are +machinery: jnr/jffi/kenai statics, charsets +(`Lsun/nio/cs/UTF_16LE;`...), `Ljava/lang/reflect/Method;`, `[I`. **Zero +`com/datadog` classes — not even ProfileAnalyzer's `log` Logger — and no +`Collections$UnmodifiableList`/`ArrayList` holder.** The LEAK_BUFFER +holder NEVER enters the anchor tier. + +Sweep itself is healthy: k8 (STATIC_FIELD) tally ~1000-1100 edges/chunk, +cursor advancing 22651→23675 of 33098 classes, per-chunk admissions 0-11 +(first lap admits; later laps ALREADY_ADMITTED). + +## Events ARE flowing (uploaded recordings — local pod chunks are a bad source) + +- Local pod chunks (kubectl cp) AND `jfr summary` of uploads show NO + ReferenceChain/HeapLiveObject types, but the UPLOADED .jfr (toolkit + download) contains both. Same artifact as round 2's "need merged upload" + lesson — extends to HeapLiveObject. RULE: uploaded recordings are the + only valid evidence source for dump-time events. +- 12 unique `datadog.ReferenceChain` events in one 60s upload + (re-emitted from cache per dump, per design): depth-0 static_field + chains (byte[] held by statics named `CRLF`, `buf`, `COLONSPACE`, + `DASHDASH` — charset/constant machinery) and 5-13-hop jni_global chains + through `Mac`/`HmacCore` (`k_opad`/`k_ipad`!) → ThreadLocalMap → + ForkJoinWorkerThread — the tagged 128KB machinery cohort (klass_id=2, + tid 78360) fully explained WITH per-hop edge names. The edges feature + works on-pod, first real-world validation. +- `datadog.HeapLiveObject` events present, including `objectClass=byte[] + size=78643216B eventThread=simulated-memory-leak age=99-108 + leakTag=1073741987/1073741990` — THE LEAK CHUNKS carry leak tags in the + recording. Liveness is flowing; the local-chunk absence was an + evidence-source artifact (q-heapliveobject RESOLVED). +- Correlation check: the 12 chains' targetTags (8479-9376, frontier tags) + are DISJOINT from the leak tags (0x40000063/66) — leak chunks still + have no chain (interception still zero, consistent with the holder + never being walked). + +## Interpretation + +The ORIGINAL question ("zero datadog.ReferenceChain events") is +functionally RESOLVED: events flow end-to-end on-pod with named edges, +for the machinery cohort. The remaining correctness gap is now sharply +isolated: the LEAK_BUFFER holder never enters the anchor tier (76/76 +anchors are machinery), so the tagged 78MB chunks are never enumerated by +any walk → no chain for the actual leak. Mechanism candidate confirmed +in code (see find-anchor-holder-eviction): parent_tag==0 is required by +the collector, improveChain replaces root-attached entries with +chain-attached ones, and re-rooting a chain-attached entry is explicitly +refused (referenceChains.cpp:2376, "known, documented limitation"). + +Operational: `jfr print --events datadog.ReferenceChain` CRASHES +(ClassCastException in PrettyWriter on the chain F_CPOOL|F_ARRAY field — +RecordedClass cannot cast to Object[]) — parse with JMC API instead +(RcDump.java pattern: accessors by identifier, ReferenceChainAssertions +has the canonical code). JMC classpath: flightrecorder-9.1.1.jar + +common-9.1.1.jar + lz4-java-1.4.0.jar from the Gradle cache. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-livelock-pod-logs.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-livelock-pod-logs.md new file mode 100644 index 0000000000..303c74e0a1 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-livelock-pod-logs.md @@ -0,0 +1,65 @@ +--- +id: ev-livelock-pod-logs +type: evidence +status: confirmed +depends_on: [ev-post-resync-deployment-verified] +supersedes: [] +related: [find-marker-tag-slot-index-mismatch, find-canary-search-cannot-terminate, hyp-warmup-transience] +tags: [pod-logs, livelock, canary, buildCanaryChainEvent, TEST_LOG] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Phase 2: pod logs show a stable livelock, not warm-up + +## Source +`kubectl logs -n profiling-stg prof-analyzer-hotdog-jb-c944876b9-f762h + -c prof-analyzer --since=25m`, filtered on reference-chain / liveness +TEST_LOG lines. JVM started 14:29; logs read ~14:55. + +## The repeating pair (dozens of identical occurrences) + +``` +[TEST::INFO] ReferenceChainTracker::pollWatchedTargets canary candidate[0] klass_id=8 marker_tag=-4611686018427387905 needRefresh=1 +[TEST::INFO] ReferenceChainTracker::pollWatchedTargets canary buildCanaryChainEvent(candidate=0) -> 0 +[TEST::INFO] ReferenceChainTracker::shouldRunPass -> true (canary search, 0/3 candidates found) +``` + +`grep -oE "canary candidate\[[0-9]+\] klass_id=[0-9]+ marker_tag=-?[0-9]+" | sort -u` +over the whole 25-minute window returned exactly ONE distinct line: + +``` +canary candidate[0] klass_id=8 marker_tag=-4611686018427387905 +``` + +The session summary records `143 identical iterations over 25 minutes` +with `0/3 candidates found` never advancing (derived from the +`grep -oE "canary search, [0-9]+/[0-9]+ candidates found" | sort | uniq -c` +count; the raw uniq -c output line was truncated in the captured transcript). + +There were **no** `chain-found`, `writeReferenceChain`, +`drainPendingChainEvents` or `ERROR.*chain` lines at all. + +## The search itself is healthy and running + +``` +[TEST::INFO] ReferenceChainTracker::threadLoop iteration=61 shouldRunPass=1 searchState=0 passesRun=60 effectiveCadenceNs=1000000000 effectiveBudget=3741 gcFinishEpoch=27 lastPassGcFinishEpoch=27 nowMinusLastPassNs=3000773439 +[TEST::INFO] ReferenceChainTracker::runPass starting JVMTI walk: search_started=1 frontierSize=10027 +[TEST::INFO] ReferenceChainTracker::runPassManualWalk static_field_phase edges_admitted=0 truncated=1 frontier_cap_hit=0 last_resolved_class_count=33501 last_static_field_class_count=-1 +[TEST::INFO] ReferenceChainTracker::runPassManualWalk expand_phase edges_admitted=1 truncated=1 frontier_cap_hit=0 remaining_budget=3453 +[TEST::INFO] ReferenceChainTracker::runPassManualWalk rotation_candidates root_kind_tags=0 leak_accumulation_tags=0 stale_expanded_tags=0 watched_leak_klass_count=5 leak_signatures=1 leak_parents=1 +[TEST::INFO] ReferenceChainTracker::runPass done: err=0 edges_admitted=184 truncated=1 frontier_cap_hit=0 searchState=0 abandonReason=0 frontierSize=10211 effectiveBudget=3741 effectiveCadenceNs=1000000000 +``` + +## Leak signal is real and past hysteresis (rules out warm-up) + +``` +[TEST::INFO] LivenessTracker::heapFloorRising FLOOR_RISING recent_mean=1138212368 earliest_mean=286975797 recent_min=773009712 earliest_min=223877728 floor_bar=2238777 floor_rising=1 +[TEST::INFO] LivenessTracker::selectLeakCandidates scanning 118 klass_population entries +[TEST::INFO] LivenessTracker::selectLeakCandidates entry[0] klass_id=160 ring_fill=20 has_trend=1 slope=19.000000 consecutive_positive=11 required=3 representative=0x7da38c0956e1 +[TEST::INFO] LivenessTracker::selectLeakCandidates entry[1] klass_id=8 ring_fill=20 has_trend=1 slope=20.000000 consecutive_positive=11 required=3 representative=0x7da38c0956f9 +[TEST::INFO] LivenessTracker::selectLeakCandidates entry[4] klass_id=44 ring_fill=12 has_trend=1 slope=17.384615 consecutive_positive=3 required=3 representative=0x7da2f00886c1 +``` + +`consecutive_positive=11 >= required=3` for the very klass (`klass_id=8`) +that the canary poll is stuck on. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-marker-tag-arithmetic.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-marker-tag-arithmetic.md new file mode 100644 index 0000000000..55c04513c7 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-marker-tag-arithmetic.md @@ -0,0 +1,38 @@ +--- +id: ev-marker-tag-arithmetic +type: evidence +status: confirmed +depends_on: [ev-livelock-pod-logs] +supersedes: [] +related: [find-marker-tag-slot-index-mismatch] +tags: [marker-tag, arithmetic, proof, referenceChains] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# The logged marker tag decodes to slot 1, but the code uses index 0 + +## Definition (verified in the repo at HEAD 8114019c2) + +`ddprof-lib/src/main/cpp/referenceChains.h:1924` + +```cpp +static constexpr jlong MARKER_TAG_BASE = -(1LL << 62); +``` + +Pre-tagging assigns `MARKER_TAG_BASE - i` per candidate slot `i` +(`referenceChains.cpp:3384`, inside the `_candidate_count == 0` block). + +## Arithmetic run in-session + +``` +$ python3 -c "base=-(1<<62); tag=-4611686018427387905; print(base); print(tag); print(base-tag)" +MARKER_TAG_BASE = -4611686018427387904 +observed tag = -4611686018427387905 +decoded slot = 1 +``` + +The live log line reports `candidate[0]` while carrying the marker tag of +slot **1**. That mismatch is the bug: `pollWatchedTargets()` indexes the +per-candidate arrays with the loop position, not with the slot encoded in +the tag it just read. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-post-resync-deployment-verified.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-post-resync-deployment-verified.md new file mode 100644 index 0000000000..11847c116b --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-post-resync-deployment-verified.md @@ -0,0 +1,64 @@ +--- +id: ev-post-resync-deployment-verified +type: evidence +status: confirmed +depends_on: [ev-deployed-so-1481-no-symbols] +supersedes: [] +related: [find-marker-tag-slot-index-mismatch, ev-jafar-zero-refchain-events] +tags: [hotdog, deployment, post-resync, md5, on-pod] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Phase 2: after the user's resync the feature IS deployed and enabled + +## Source +Same pod `prof-analyzer-hotdog-jb-c944876b9-f762h` (never rescheduled — +still 44h old, `RESTARTS 5`), but the JVM was restarted in place at 14:29. + +## Raw excerpts + +New agent jar on disk: + +``` + File: /usr/local/app/agent/dd-java-agent.jar + Size: 37388225 + Birth: 2026-08-24 14:27:19.763707765 +0000 +3b204607ab88b99e18e577d23368ca45 /usr/local/app/agent/dd-java-agent.jar +``` + +New JVM PID (231 is gone; `jcmd 231 JFR.check` -> "Could not find any +processes"): + +``` +root 20807 ... java -Xms7g -Xmx7g -XX:+UseG1GC ... -javaagent:agent/dd-java-agent.jar ... + -Ddd.profiling.ddprof.liveheap.enabled=true + -Ddd.profiling.experimental.ddprof.referencechains.enabled=true <-- NEW + -Ddatadog.slf4j.simpleLogger.log.com.datadog.profiling=debug + -cp classpath/*:libs/* com.datadog.profiling.analyzer.Main +``` + +Newly loaded native library and its symbol scan: + +``` +/tmp/ddprof_root/pid_20807/scratch/libjavaProfiler-dd-tmp1075478356263294699.so +=== referencechain strings === +1097 # strings ... | grep -icE 'ReferenceChainTracker|referencechains' +=== md5 === +4d1dc48eaccfb6d0c35831669b71a31a +=== JFR.check === +20807: +Recording 1: name=dd-profiling maxsize=64.0MB maxage=5m (running) +``` + +Freshly uploaded JFR +(`/tmp/hotdog-jb-fresh3/prof-analyzer-hotdog-2026-08-24_14-46-42.156Z-ip-10-128-190-53.ec2.internal-stripe.jfr`, +12.6 MB) now declares the types: + +``` +datadog.ReferenceChain +datadog.ReferenceChainAbandoned +``` + +(plus new-in-this-build `datadog.UnwindFailure`, +`datadog.WallClockSamplingEpoch`, `datadog.DatadogProfilerClassRefCache`, …) diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-postCB-onpod-live-verification.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-postCB-onpod-live-verification.md new file mode 100644 index 0000000000..eb4454b9bc --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-postCB-onpod-live-verification.md @@ -0,0 +1,72 @@ +--- +id: ev-postCB-onpod-live-verification +type: evidence +status: confirmed +depends_on: [q-canary-stuck-fix-alternatives] +supersedes: [] +related: [find-canary-stuck-restart-wipes-frontier, find-static-field-sweep-never-completes] +tags: [pod-logs, on-pod, post-fix, live-verification, fix-c, fix-b, canary-stuck, frontier-growth] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# Fix C+B confirmed live: zero CANARY_STUCK abandons, frontier grows unbroken past 80k; candidates still 0/5 + +## Deployment observed + +User resynced the agent on `prof-analyzer-hotdog-jb-c944876b9-f762h` (same +pod, PID 62384). Confirmed via `strings` on the loaded `.so` +(`/tmp/ddprof_root/pid_62384/scratch/libjavaProfiler-dd-tmp8721684202363682288.so`, +md5 `f92957d3e32ecfbce358f1d854d66182`) that this session's new symbols are +present: `CANARY_PAIN_BUDGET_REFILL_MULTIPLIER` (1 hit), +`canaryStuckPassLimit`/`CanaryStuckRestartCount`/`_canary_stuck_restart_count` +(4 hits combined). + +## Three live traces (`/tmp/hotdog_trace{3,4,5}.log`, 40s + 90s + 300s, 122+276 `runPass done` samples total) + +``` +CANARY_STUCK abandons (reason=3): 0 (was 2 per ~40s pre-fix) +any abandon at all: 0 +searchState: always 0 (RUNNING) +abandonReason: always 0 (NONE) +frontier_cap_hit: always 0 +frontierSize progression: 25,040 -> 26,626 -> 33,137 -> 38,075 -> 80,382 +candidates found: 0/5 throughout all three traces (~430s combined) +``` + +**C+B confirmed working as designed**: the search that previously died +every ~20s at 12k-16k frontier entries (`ev-postfixEF-onpod-live-verification`) +now runs uninterrupted, growing its frontier past 80k entries with zero +restarts. This is exactly the mechanism the fix targeted. + +## Candidate-slot identity: verified stable (initial concern retracted) + +Mid-session, misread two different "candidate[N]" log-line families as +evidence of candidate churn: +- `pollWatchedTargets candidate[i] klass_id=...` (referenceChains.cpp:3486) + - `i` is the position in `LivenessTracker::selectLeakCandidates()`'s + freshly re-ranked top-N list for that single poll. Reshuffles every + poll by design (trend/growth score re-ranking) - cosmetic, unrelated to + search bookkeeping. +- `canary candidate[i] klass_id=... marker_tag=... slot=N` - the `slot=N` + field is the actual persistent identity, decoded from the JVMTI marker + tag (`MARKER_TAG_BASE - slot`). + +Verified directly: extracting only `(klass_id, slot)` pairs from the 5-min +trace yields exactly 2 distinct pairs for the whole window - `klass_id=10 +([B]) -> slot=0` and `klass_id=232 ([Ljava/lang/Class;) -> slot=3` - zero +reassignment. Confirms the code's own guarantee +(referenceChains.cpp:3433-3440: "Slots are never retired or reassigned once +occupied") holds on a live pod. Candidate-identity churn is ruled out as a +contributor to the 0/5 result. + +## What this confirms / rules out + +- Confirms `find-canary-stuck-restart-wipes-frontier`'s fix (C+B) is + effective: destructive restart-on-stuck no longer fires, frontier + accumulates unbroken. +- Rules out candidate-slot reassignment/churn as an explanation for + continued 0/5 candidate resolution. +- Does NOT explain the continued 0/5 result by itself - see + `find-static-field-sweep-never-completes` for the new root cause + surfaced by digging into per-pass diagnostic counters after this trace. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-postfix-onpod-live-verification.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-postfix-onpod-live-verification.md new file mode 100644 index 0000000000..90b4ebd0c5 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-postfix-onpod-live-verification.md @@ -0,0 +1,62 @@ +--- +id: ev-postfix-onpod-live-verification +type: evidence +status: confirmed +depends_on: [] +supersedes: [] +related: [find-marker-tag-slot-index-mismatch, find-canary-stuck-abandon-detector, find-canary-search-cannot-terminate] +tags: [pod-logs, on-pod, post-fix, live-verification, md5, fix-a, fix-c] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# Fix A + Fix C confirmed live on the resynced hotdog pod; Fix B implied + +## Reasoning chain + +After `623d3712a` (Fix A/B/C) was committed and pushed to +`jb/reference-chains`, the pod was resynced. On-pod evidence (per +`find-onpod-evidence-methodology`: `kubectl exec` + `jar xf` + `md5sum` + +`strings`, not image tags): + +- jar md5 `f3c01e345c73484d021ab405b27cc3fd` +- `.so` md5 `7437bd8a137ac79126d3cfebe3887b3b`, contains `CANARY_STUCK` / + `canary_stuck` strings (Fix C's enum value and `kReasons` string) and + 1057 `ReferenceChainTracker`-related symbol hits +- JVM PID 3238, started 2026-08-25 08:50 UTC + +**Fix A (slot decode) confirmed live**: pod logs show +`candidate[N] ... marker_tag=... slot=M` where `M != N` in general, e.g. +`candidate[1] klass_id=122 marker_tag=-4611686018427387906 slot=2` and +`candidate[2] klass_id=11 marker_tag=-4611686018427387905 slot=1` — the +tag-decoded slot, not the loop index, is now used for lookup, exactly as +designed. + +**Fix C (`CANARY_STUCK` abandon detector) confirmed live**: `abandonReason=3` +appears in `runPass done: ...` log lines, always immediately followed by +`shouldRunPass -> true (restarting search)` and `passesRun` resetting to 0 +on the next iteration. 5 such abandon/restart cycles were observed over +~20 minutes at timestamps 09:00:18, 09:01:53, 09:03:23, 09:04:59, 09:06:37. +No infinite livelock — each search cleanly terminates after +`CANARY_NO_PROGRESS_PASS_LIMIT = 30` stuck passes and restarts, instead of +running forever as it did pre-fix (`ev-livelock-pod-logs`: 143 identical +iterations over 25 minutes with 0/N found). + +## What this rules out + +- Any doubt that Fix A/C reached this build or are behaving as designed — + both are directly observable in pod logs with concrete, code-matching + values (decoded slot != loop index; `abandonReason=3` with the exact + restart follow-through the design doc predicts). + +## New open finding surfaced by this verification + +Despite all three fixes working exactly as designed, **zero canary +candidates have ever been resolved** on this pod across all 5 observed +restart cycles (`buildCanaryChainEvent(slot) -> 1` never once observed). +This is a new, undiagnosed symptom — not assumed to be a regression from +Fix A/B/C, since the search no longer livelocks silently and this may be a +pre-existing reachability/budget limit that was simply invisible before. +See `find-abandon-event-lost-to-dump-sampling-race` for the *separate* +zero-JFR-event-count question this triggered — that one turned out to have +a definite root cause, unrelated to whether candidates ever resolve. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-postfix-static-field-onpod-live-verification.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-postfix-static-field-onpod-live-verification.md new file mode 100644 index 0000000000..5cc3dcca72 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-postfix-static-field-onpod-live-verification.md @@ -0,0 +1,72 @@ +--- +id: ev-postfix-static-field-onpod-live-verification +type: evidence +status: confirmed +depends_on: [find-static-field-sweep-cursor-fix] +supersedes: [] +related: [] +tags: [pod-logs, on-pod, post-fix, live-verification, static-field, cursor, lap-truncation, NEW-THIS-SESSION] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# Static-field sweep cursor/lap-truncation fix confirmed live on hotdog pod + +## Deployment proof + +Resynced by user. JVM PID 92618 (started 15:29). Scratch `.so` md5 +`aeab8726e90e5f21e33393c3bfea043e`. `strings` on that file confirms +presence of `_static_field_sweep_cursor`, `static_field_cycle_complete`, +and the exact new `TEST_LOG` format `cycle_complete=%d sweep_cursor=%d`. + +## Cursor advances correctly, real edges admitted + +Multiple `runPass` samples across two separate check windows show +`sweep_cursor` advancing by exactly 512 per call, e.g.: + +``` +6656 -> 7168 -> 7680 -> 8192 -> 8704 -> 9216 -> 9728 -> 10240 -> 10752 -> 11264 +32768 -> 33280 -> 33792 -> [wrap to 0] -> 512 -> 1024 -> 1536 -> 2048 -> 2560 -> 3072 +``` + +`edges_admitted` per chunk is now real and varying (contrast with the +pre-fix pattern of 0-1 forever, see `find-static-field-sweep-never-completes`): +`3, 0, 1, 27, 0, 15, 1, 98, 7, 7` in one window; `331, 414, 737, 59` in +another; a later 2-minute check window showed `84, 320, 198` on +successive `runPass done` lines with `frontierSize` growing from 86817 +to 87335 in step. + +## Lap-truncation-latch behaves exactly as designed + +Direct confirmation of the fix's core correctness property: a lap is +only "done" (`cycle_complete=1`) if **zero** chunks within it truncated. +Observed concrete example: chunk at `sweep_cursor=33792` reported +`truncated=1`; the very next chunk closed the lap +(`edges_admitted=59 truncated=0`, `sweep_cursor` wrapped to `0`) — and +`cycle_complete` was still correctly `0` for that closing chunk, because +an earlier chunk in the same lap had truncated. This is the exact +semantics the fix was designed to produce (see +`find-static-field-sweep-cursor-fix`'s "gated on cycle_complete" section), +observed under real production load, not just in the gtest mock. + +`cycle_complete=1` has not been observed in any sample checked so far +(multiple windows, ~10+ minutes combined) — plausible given how large the +JDK bootstrap classlist tail is relative to the chunk size +(`STATIC_FIELD_SWEEP_CHUNK_CLASSES=512`); not itself evidence of a bug +since forward progress (cursor advancing, edges accumulating) is directly +observed every call. + +## Health of the rest of the pipeline, unchanged/still good + +`CANARY_STUCK` count: 0 across all windows checked this round. Frontier +grows cleanly and monotonically (e.g. 86817 -> 87137 -> 87335, and +separately observed growing past 64k earlier in the same deployment) with +no restarts/wipes observed. + +## What this does NOT show + +End-to-end candidate resolution is still 0/5 in every sample checked — +see `find-candidate1-never-tagged` and +`find-candidates-234-die-before-resolution`. The sweep mechanism working +correctly is necessary but evidently not sufficient; the bottleneck has +moved downstream of admission into BFS reach / candidate lifetime. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-postfixEF-onpod-live-verification.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-postfixEF-onpod-live-verification.md new file mode 100644 index 0000000000..6bebbd9a25 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-postfixEF-onpod-live-verification.md @@ -0,0 +1,104 @@ +--- +id: ev-postfixEF-onpod-live-verification +type: evidence +status: confirmed +depends_on: [find-canary-fixes-e-f, find-abandon-event-queue-fix] +supersedes: [] +related: [find-cpu-pain-budget-starves-canary-passes, find-threadloop-presleep-blocks-back-to-back] +tags: [pod-logs, on-pod, post-fix, live-verification, fix-e, fix-f, fix-d, canary-stuck, frontier-wipe] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# Fix D/E/F confirmed live; new bottleneck surfaced: CANARY_STUCK abandon/restart every ~20s wipes the frontier before reaching the candidate + +## Deployment observed + +Pod redeployed again since the last check (without an explicit ask from the +user this turn - discovered by routine "watch the pod" polling): new JVM +PID `39108` (started `2026-08-25T11:38Z`), jar md5 +`c5140ef0e4ee45bc7365f337480bed43`, loaded `.so` at +`/tmp/ddprof_root/pid_39108/scratch/libjavaProfiler-dd-tmp18062115394516649584.so`, +md5 `22497e41e381bec86cd549972fd557e0`. Confirmed this build contains this +session's fixes via `strings`: `CANARY_PAIN_BUDGET_REFILL_MULTIPLIER` (1 +hit) and `pending_abandoned_events` (2 hits) both present. + +## 40-second live trace (`kubectl logs -f --since=1s > /tmp/hotdog_trace2.log`, 382,388 lines) + +``` +grep -c "runPass done" -> 50 (was 0 pre-fix) +grep -c "pollWatchedTargets" -> 47025 +grep -c "shouldRunPass -> true (canary search" -> 45 +grep -c "enqueuePendingAbandonedEvent" -> 2 +grep -c "REFERENCE_CHAIN_EVENTS_DROPPED" -> 0 +``` + +**Fix E + F confirmed working**: `runPass` now fires continuously (50 times +in 40s, `effectiveCadenceNs=10000000` = 10ms) instead of the pre-fix zero +over a 45s window (`ev-hotdog-trace-zero-runpass`). The pain-budget +starvation and the threadLoop double-sleep are both resolved. + +**Fix D confirmed working**: both abandons in this window were correctly +queued - +``` +enqueuePendingAbandonedEvent reason=3 queue_size=2 +enqueuePendingAbandonedEvent reason=3 queue_size=3 +``` +(`reason=3` = `CANARY_STUCK`). No drops, well under +`MAX_PENDING_ABANDONED_EVENTS=16`. + +## New bottleneck surfaced (not yet fixed, not yet fully diagnosed) + +Despite passes now running continuously, **still 0/5 candidates found** +throughout the entire window (`grep -oE "canary search, [0-9]/[0-9] +candidates found"` -> 45× `0/5`). `candidate[0]`'s `klass_id` stayed stable +at `2` (`[B`, byte array) across the whole window (2545 hits, no churn) - +class-tag stability is not the problem here. + +Two full search cycles were captured, both ending in the same way: +``` +runPass done: ... searchState=2 abandonReason=3 frontierSize=12776 ... +runPass done: ... searchState=2 abandonReason=3 frontierSize=15699 ... +``` +i.e. `CANARY_STUCK` abandon fires once the frontier reaches roughly +12-16k entries, then `restartSearch()` wipes it back to a fresh walk (per +`restartSearch()`'s own documented behavior, `referenceChains.cpp:1054-1056` +- `_frontier->resetForRestart()` + `_next_tag = 1`). Two cycles happened in +this 40s window alone - each search gets roughly ~20s and ~13-16k frontier +entries before being killed and restarted from scratch, with zero carryover. + +This matches, and is now live confirmation for, the previously-flagged +(not yet fixed) concern in `find-canary-search-cannot-terminate`/the +`CANARY_NO_PROGRESS_PASS_LIMIT` design comment +(`referenceChains.h:1988-1999`): the canary-specific stuck detector fires +on zero *candidate* progress specifically, independent of how much the +*general* frontier is still healthily growing elsewhere. If `candidate[0]` +(a `byte[]`) sits behind a long, indirect reference chain that takes more +than `CANARY_NO_PROGRESS_PASS_LIMIT=30` passes worth of BFS expansion to +reach, this search design can never reach it: every restart re-walks from +scratch, so cumulative BFS "reach" never exceeds what one ~20s cycle can +cover. + +## What this confirms / rules out + +- Confirms `find-cpu-pain-budget-starves-canary-passes` and + `find-threadloop-presleep-blocks-back-to-back` are genuinely fixed - not + just gtest-clean, but observably changing pod behavior (0 -> 50 runPass + in a comparable window). +- Confirms `find-abandon-event-queue-fix` (Fix D) works end-to-end on a + live pod, not just in gtest. +- Rules out candidate churn (LivenessTracker offering a different + candidate list each restart) as a contributor here - `klass_id=2` was + stable across both cycles. +- Does NOT yet confirm the frontier-wipe-on-restart / `CANARY_NO_PROGRESS_PASS_LIMIT` + hypothesis as proven root cause of the *new* zero-resolution pattern - + it is the most consistent explanation given this evidence, but has not + been isolated from alternatives (e.g. the candidate genuinely being + unreachable from any sampled root in this heap at all, independent of + pass count). + +## Not yet done + +- No fix proposed or requested for this new bottleneck - user only asked + to "continue watching"; this is a fresh observation to report, not an + ask to act on. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-source-poll-vs-callback.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-source-poll-vs-callback.md new file mode 100644 index 0000000000..0c18c83e6c --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-source-poll-vs-callback.md @@ -0,0 +1,147 @@ +--- +id: ev-source-poll-vs-callback +type: evidence +status: confirmed +depends_on: [] +supersedes: [] +related: [find-marker-tag-slot-index-mismatch, find-one-shot-pretag-gate, find-canary-search-cannot-terminate] +tags: [source, referenceChains, citations, verified-at-head] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Source excerpts: writer decodes the slot, reader uses the loop index + +All line numbers re-verified against the working tree at HEAD +`8114019c2e97ca7641e151e4f7de080e3ec6fc5b`. + +## Writer — `heapReferenceCallback()`, `referenceChains.cpp:1508-1529` (CORRECT) + +```cpp + if (ctx->tracker->_candidate_count > 0 && + *tag_ptr <= ReferenceChainTracker::MARKER_TAG_BASE) { + int candidate_idx = (int)(ReferenceChainTracker::MARKER_TAG_BASE - *tag_ptr); // :1510 + if (candidate_idx >= 0 && candidate_idx < ctx->tracker->_candidate_count) { + jlong rtag = (referrer_tag_ptr != nullptr) ? *referrer_tag_ptr : 0; + u32 candidate_klass = ctx->tracker->classTags()->resolve(class_tag); + if (rtag > 0) { + FrontierEntry parent{}; + if (ctx->frontier->lookup(rtag, &parent)) { + jlong frontier_tag = *tag_ptr; + ctx->frontier->insert(frontier_tag, rtag, parent.referrer_klass, + parent.depth + 1, FrontierEntryState::FRONTIER, + parent.root_kind); + ctx->tracker->_candidate_parent_tags[candidate_idx] = rtag; // :1525 + ctx->tracker->_candidate_frontier_tags[candidate_idx] = frontier_tag;// :1526 + ctx->tracker->_candidate_referrer_klasses[candidate_idx] = candidate_klass; + ctx->tracker->_candidate_depths[candidate_idx] = parent.depth + 1; + ctx->tracker->_candidate_found_bits |= (1ULL << candidate_idx); // :1529 +``` + +## Reader — `pollWatchedTargets()`, `referenceChains.cpp:3456-3490` (BUG) + +```cpp + jlong tag = getTag(jvmti, obj); // :3456 + if (tag <= MARKER_TAG_BASE) { // :3465 + bool need_refresh = false; + _resolved_chains_lock.lock(); + auto it = _resolved_chains.find(klass_id); + need_refresh = (it == _resolved_chains.end() || + it->second.source_tag != tag || + it->second.source_search_ns != current_search_ns); // :3469-3471 + _resolved_chains_lock.unlock(); + TEST_LOG("...canary candidate[%d] klass_id=%u marker_tag=%lld needRefresh=%d", + i, klass_id, (long long)tag, need_refresh); // :3473-3475 + if (need_refresh) { + ReferenceChainEvent event; + bool built = buildCanaryChainEvent(i, &event); // :3478 <-- i, not slot + TEST_LOG("...buildCanaryChainEvent(candidate=%d) -> %d", i, built); + if (built) { + event._start_time = TSC::ticks(); + cacheResolvedChain(klass_id, std::move(event), + _candidate_frontier_tags[i], // :3485 <-- same bug + current_search_ns); + } + } + jni->DeleteLocalRef(obj); + continue; + } +``` + +## Failure path actually taken — `referenceChains.h:2093-2141` + +```cpp +bool buildCanaryChainEvent(int candidate_idx, ReferenceChainEvent *out) { // :2093 + if (_frontier == nullptr || out == nullptr || + candidate_idx < 0 || candidate_idx >= _candidate_count) return false; // :2094-2097 + jlong parent_tag = _candidate_parent_tags[candidate_idx]; + u32 candidate_klass= _candidate_referrer_klasses[candidate_idx]; + jlong frontier_tag = _candidate_frontier_tags[candidate_idx]; + ... + if (parent_tag > 0) { + if (!_frontier->lookup(parent_tag, &entry)) return false; // :2107 + for (jlong tag = parent_tag; tag > 0;) { + if (!_frontier->lookup(tag, &entry)) return false; // :2112 + ... + } + } else if (parent_tag == 0 && frontier_tag > 0) { + if (!_frontier->lookup(frontier_tag, &entry)) return false; // :2122-2124 + ... + } else { + return false; // never pruned (candidate not reached) // :2127 <-- HIT + } +``` + +With slot 0 never written, `parent_tag == 0 && frontier_tag == 0`, so the +`:2127` branch fires every time. + +## Self-heal blockers + +`referenceChains.cpp:3375-3397` — one-shot pre-tagging: + +```cpp + if (_candidate_count == 0) { // :3375 + _candidate_count = candidate_count; + _candidate_found_bits = 0; + for (int i = 0; i < candidate_count; i++) { + jlong tag = MARKER_TAG_BASE - i; + _candidate_tags[i] = tag; + jobject obj = LivenessTracker::instance()->resolveCandidateRepresentative( + jni, candidates[i].klass_id); + if (obj != nullptr) { jvmti->SetTag(obj, tag); jni->DeleteLocalRef(obj); } + } + TEST_LOG("...canary: %d candidates pre-tagged with marker tags", _candidate_count); + Counters::increment(REFERENCE_CHAIN_CANDIDATE_COUNT, _candidate_count); + } +``` + +`referenceChains.cpp:3090-3108` — abandon suppressed while urgent, and +completion needs all bits set: + +```cpp + } else if (_passes_since_last_progress >= NO_PROGRESS_PASS_LIMIT && + !isUrgent()) { // :3090-3091 + store(_abandon_reason, (u8)SearchAbandonReason::TTL); + storeRelease(_search_state, (u8)SearchState::ABANDONED); + } else if (_candidate_count > 0 && + __builtin_popcountll(_candidate_found_bits) == + (u64)_candidate_count) { // :3101-3103 + storeRelease(_search_state, (u8)SearchState::COMPLETED); + Counters::increment(REFERENCE_CHAIN_CANDIDATES_FOUND, + __builtin_popcountll(_candidate_found_bits)); + } +``` + +`referenceChains.cpp:914-922` — `shouldRunPass()` keeps returning true for +as long as any candidate bit is unset (so the search never idles out): + +```cpp + if (_candidate_count > 0 && + __builtin_popcountll(_candidate_found_bits) < (u64)_candidate_count) { + TEST_LOG("ReferenceChainTracker::shouldRunPass -> true (canary search, " + "%d/%d candidates found)", + (int)__builtin_popcountll(_candidate_found_bits), + (int)_candidate_count); + return true; + } +``` diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-tid-clustering-onpod-verification.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-tid-clustering-onpod-verification.md new file mode 100644 index 0000000000..ea4a94a33a --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-tid-clustering-onpod-verification.md @@ -0,0 +1,55 @@ +--- +id: ev-tid-clustering-onpod-verification +type: evidence +status: pending-verification +depends_on: [find-lambda-fragments-calltrace-id, q-allocation-site-selection] +supersedes: [] +related: [find-lambda-fragments-calltrace-id, q-allocation-site-selection] +tags: [pod-logs, on-pod, post-fix, live-verification, tid-clustering, dominant-gens, NEW-THIS-SESSION] +created: 2026-08-28 +updated: 2026-08-28 +--- + +# tid-based clustering on-pod: still dominant_gens=1 + +## Context + +After switching the clustering key from `call_trace_id` to `tid` +(commit `2c50bf0cd`), deployed to pod PID 58100, .so md5 +`abe3c2dc7e...`. + +## Observation + +All `foldKlassCountsLocked minted=` lines show `dominant_gens=1`: + +``` +minted=2 for klass_id=205 dominant_tid=58393 dominant_gens=1 +minted=3 for klass_id=20 dominant_tid=58471 dominant_gens=1 +minted=1 for klass_id=2213 dominant_tid=58470 dominant_gens=1 +minted=1 for klass_id=279 dominant_tid=58470 dominant_gens=1 +minted=2 for klass_id=12 dominant_tid=58393 dominant_gens=1 +``` + +klass_id=4 has the strongest leak signal (gen_count=33, slope=30.4, +consecutive_positive=13) but no `minted=` line — representatives +already live, `need_mint=false`. + +## Problem + +User pushback: the liveness tracking table IS cross-epoch, surviving +objects accumulate ages across GC cycles. The 12 leaking [B from +tid=172 (`simulated-memory-leak`) should all be in the table with +different ages. So `insertThreadGen` should be called 12 times for +tid=172 with 12 different ages → `dominant_gens=12`, not 1. + +Possible explanations (not yet verified): +1. Liveness tracker subsampling: only 1 of 12 leaking [B is tracked +2. `tid` field is 0 or wrong for some entries +3. Only 1 surviving object per thread per epoch in the scratch + +## Next step + +Diagnostic commit `5e4493dbf` logs per-thread `age_count` breakdown +in `foldKlassCountsLocked`. Needs redeploy to see whether tid=172 +(or equivalent) has 1 age (subsampling) or multiple ages (bug in +insertThreadGen). diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-tid-clustering-per-thread-working.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-tid-clustering-per-thread-working.md new file mode 100644 index 0000000000..74213dd80e --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-tid-clustering-per-thread-working.md @@ -0,0 +1,63 @@ +--- +id: ev-tid-clustering-per-thread-working +type: evidence +status: confirmed +depends_on: [find-lambda-fragments-calltrace-id, q-dominant-gens-still-one-with-tid] +supersedes: [ev-tid-clustering-onpod-verification] +related: [find-lambda-fragments-calltrace-id, q-dominant-gens-still-one-with-tid] +tags: [pod-logs, on-pod, post-fix, live-verification, tid-clustering, per-thread, dominant-gens, NEW-THIS-SESSION] +created: 20260828 +updated: 20260828 +--- + +# Per-thread generation tracking confirmed working on-pod + +## Context + +After switching from call_trace_id to tid (2c50bf0cd) and adding +per-thread diagnostic (5e4493dbf), deployed to pod PID 91958. + +## Observation + +klass_id=6 (later klass_id=9 on new JVM) shows dominant thread with +multiple ages: + +``` +foldKlassCountsLocked scratch[1] klass_id=6 gen_count=5 thread_count=10 oldest_count=3 + thread[0] tid=92169 age_count=3 ← dominant leak thread + thread[1] tid=93010 age_count=2 + thread[2] tid=92380 age_count=1 + ... +``` + +`tid=92169` has `age_count=3` — the leak thread. The per-thread +tracking IS working. Earlier `dominant_gens=1` lines were from other +classes, not the leak class. + +## Re-mint confirmed + +``` +foldKlassCountsLocked re-minting klass_id=9: dominant_tid=92383 dominant_gens=2 but no rep matches +``` + +Re-mint fired when dominant thread changed and no existing rep matched. + +## Leak candidate confirmed + +klass_id=9 ([B]) is the leak candidate: +``` +selectLeakCandidates entry[1] klass_id=9 ring_fill=30 has_trend=1 + slope=4.025806 consecutive_positive=8 required=3 rep_count=1 +``` + +## Chains emitted + +8 ReferenceChain events emitted: +``` +drainPendingChainEvents re-emitted=8 +Profiler::writeReferenceChain ×8 +Profiler::dump reference-chain batch=8 write_dropped=0 +``` + +But chains are depth=1 with no holder — see +`find-already-admitted-blocks-deeper-chain`. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-timing-split-callback-vs-jvmti.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-timing-split-callback-vs-jvmti.md new file mode 100644 index 0000000000..4042755e1c --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-timing-split-callback-vs-jvmti.md @@ -0,0 +1,65 @@ +--- +id: ev-timing-split-callback-vs-jvmti +type: evidence +status: confirmed +depends_on: [] +supersedes: [] +related: [find-candidate1-never-tagged, dead-hard-reference-kind-filter] +tags: [timing, callback, jvmti, FollowReferences, safepoint, overhead-split, NEW-THIS-SESSION] +created: 2026-08-26 +updated: 2026-08-26 +--- + +# Timing split: our callback 5-17%, JVMTI heap-walking 83-92% + +## Setup + +Temporarily bumped `_pass_deadline_ns` to 200ms (override at +`referenceChains.cpp:2337`, commit `fd18425c6`) and added an RAII +`ScopedCallbackTimer` that accumulates TSC ticks inside +`heapReferenceCallback()` only when `static_field_seed` is true. After each +`FollowReferences` call, logs: +- `follow_ms` — total FollowReferences wall-time +- `callback_ms` — cumulative time in our callback code +- `jvmti_ms` — follow - callback (JVMTI's own object-iteration/metadata-walk cost) +- `callback_pct` — our share of total +- `callback_count` — total callbacks delivered + +## Data (31 samples, sorted) + +| follow_ms | callback_ms | jvmti_ms | callback_pct | callbacks | +|-----------|-------------|----------|--------------|-----------| +| 3-4 | 0 | 3-4 | 5-9% | 4.5-6.6k | +| 5-6 | 0 | 4-6 | 5-7% | 6.2-8.2k | +| 7-8 | 0-1 | 6-8 | 5-10% | 7.1-16k | +| 9-11 | 0-1 | 8-10 | 5-12% | 8.3-13.7k | + +One outlier at 35% (10ms follow, 3ms callback). + +## Conclusion + +Our callback code (`admitObject` + quota check + kind_counts tally) is +**5-17% of total wall-time** (0-1ms per chunk). JVMTI's own heap-walking +(reading object metadata, iterating class fields, resolving oops, +delivering callbacks) is **83-92%** (3-10ms per chunk). + +**Deferring `admitObject` out of the safepoint would save at most 5-17%** +— not enough to fix truncation under a 50ms deadline. The bottleneck is +JVMTI delivering 5k-16k CP callbacks per chunk, not our callback code +processing them. + +The only way to avoid the CP callback volume is to not let +`FollowReferences` descend into the class's metadata graph at all — i.e., +use `GetClassFields` + JNI `GetStaticObjectField` to read static fields +directly, without triggering CP/interface/superclass callbacks. This +would also eliminate the STW entirely (both APIs are non-Heap-category, +no `VMThread::execute()`). + +## Side effect: sweep completes cleanly at 200ms + +With the 200ms deadline, **all chunks complete without truncation** +(`truncated=0` on every chunk). `edges_admitted` jumped from 0-5 (under +50ms shared budget) to 369-2155 per chunk. A full lap completed +(`cycle_complete=1` observed, `last_static_field_class_count` advanced +from 33677 → 33752). This confirms the sweep mechanism is healthy — +the truncation was a budget-sharing problem, not a sweep-cost problem. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-toolkit-and-onpod-methodology.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-toolkit-and-onpod-methodology.md new file mode 100644 index 0000000000..facb1fb603 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-toolkit-and-onpod-methodology.md @@ -0,0 +1,65 @@ +--- +id: ev-toolkit-and-onpod-methodology +type: evidence +status: confirmed +depends_on: [] +supersedes: [] +related: [find-onpod-evidence-methodology, dead-jcmd-jfr-dump-wrong-source, dead-toolkit-prod-datacenter] +tags: [methodology, user-quote, profiling-toolkit, tooling-gotcha] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Methodology corrections and tooling gotchas from this session + +## User quote (verbatim) + +> you should be able to read the pods from the pod + +Context: the first conclusion had been assembled from Profiling Toolkit +downloads plus image-tag inference. The correction was to gather the +evidence on the pod itself — `kubectl exec` + `/usr/bin/jar xf` + +`md5sum` + `strings` — which is what produced +`ev-deployed-so-1481-no-symbols`. + +Other verbatim user messages this session: + +> use the jfr-analyzer and dd-triage skills to download hotdog profiles from +> prof-analyzer jb hotdog and check whether we are getting any reference +> chain events. if not, investigate the logs from the pod to see what is +> happening + +> I resynced and reuploaded the agent. It SHOULD be there now + +## Profiling Toolkit gotchas + +Working invocation: + +``` +python3 ~/.claude/skills/profiling-toolkit-analyze/download.py \ + --org-id 2 --service prof-analyzer-hotdog --app prof-analyzer-hotdog-jb \ + --datacenter us1.staging.dog --from "20 minutes ago" --to "now" \ + --limit 5 --output-dir /tmp/hotdog-jb-fresh3 +``` + +Failure modes hit before landing on that: + +- `--datacenter us1.prod.dog` returns query hits but the blob download 404s: + ``` + ERROR: Download failed: HTTP 404 - {"errors":[{"code":"ObjectNotFoundException"}]} + Query: * + Found 1 matching profiles + ``` +- Adding `--env staging` alongside `--service`/`--app` yielded + `Found 0 matching profiles` even though profiles existed. +- Many results are `.tar` blobs containing only `cpu.pprof` + ("no JFR found in tar"); only some entries are real `.jfr`. +- Pod env for reference: `DD_ENV=staging`, `DD_SERVICE=prof-analyzer-hotdog`, + `DD_SERVICE_MAPPING=kafka:prof-analyzer-hotdog-jb`, `POD_NAME=` (empty). + +## Other tooling notes + +- jafar `jfr_open` on a 9.5 MB uploaded JFR exceeded the 120 s MCP timeout + and had to be `TaskStop`ped (task `k1scf1c6f`); `strings | grep` on the + raw file was the fast substitute for checking which event types exist. +- The pod has `/usr/bin/jar` but **no** `unzip` and no `python3`. diff --git a/.investigations/missing-refchains-on-hotdog/evidence/ev-uploaded-jfr-no-refchain-types.md b/.investigations/missing-refchains-on-hotdog/evidence/ev-uploaded-jfr-no-refchain-types.md new file mode 100644 index 0000000000..1b722d7a66 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/evidence/ev-uploaded-jfr-no-refchain-types.md @@ -0,0 +1,77 @@ +--- +id: ev-uploaded-jfr-no-refchain-types +type: evidence +status: confirmed +depends_on: [] +supersedes: [] +related: [find-refchains-not-deployed, dead-jcmd-jfr-dump-wrong-source, dead-toolkit-prod-datacenter] +tags: [jfr, profiling-toolkit, event-types, pre-resync] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Phase 1: uploaded profiles do not even declare datadog.ReferenceChain + +## Source +Profiling Toolkit `download.py`, org 2, `--datacenter us1.staging.dog`, +`--service prof-analyzer-hotdog --app prof-analyzer-hotdog-jb`, +last 60 minutes, downloaded to `/tmp/hotdog-jb-only/`. + +## Raw excerpts + +``` +Query: service:prof-analyzer-hotdog app:prof-analyzer-hotdog-jb +Found 5 matching profiles + -> prof-analyzer-hotdog-2026-08-24_13-41-35.010Z-...-stripe.jfr (9.5M) + -> prof-analyzer-hotdog-2026-08-24_13-40-35.011Z-...-stripe.jfr (8.6M) + (three .tar entries contained cpu.pprof only, "no JFR found in tar") +``` + +`strings | grep -oE "datadog\.[A-Za-z]+" | sort -u` for both real JFRs: + +``` +datadog.AggregatedSmapEntry datadog.HeapLiveObject +datadog.AvailableProcessorCores datadog.HeapUsage +datadog.BackpressureSample datadog.MethodSample +datadog.DatadogProfilerClassRefCache +datadog.DatadogProfilerConfig datadog.NativeMemoryAllocation +datadog.Deadlock datadog.NativeSocketEvent +datadog.DeadlockedThread datadog.ObjectSample +datadog.DirectAllocationSample datadog.ProfilerCounter +datadog.DirectAllocationTotal datadog.ProfilerSetting +datadog.Endpoint datadog.QueueTime +datadog.ExceptionCount datadog.SmapEntry +datadog.ExceptionSample +datadog.ExecutionSample +``` + +``` +=== referencechain hit count === +0 +``` + +`datadog.ReferenceChain` / `datadog.ReferenceChainAbandoned` are absent +entirely — not zero-count, not declared. `datadog.HeapLiveObject` IS +present, so liveheap itself was working. + +Agent version confirmed from the same JFRs: + +``` +1.65.0~dd00372bdd +``` + +## Contrast: jcmd JFR.dump (wrong source) + +`jcmd 231 JFR.dump filename=/tmp/hotdog-check-jb.jfr` produced a 1.7 MB +recording; jafar `jfr_list_types filter=datadog scan=true` on it returned +only 11 types / 220 events: + +``` +datadog.ProfilerSetting 110, datadog.ExceptionSample 83, +datadog.ExceptionCount 19, datadog.AvailableProcessorCores 6, +datadog.DirectAllocationTotal 1, datadog.DirectAllocationSample 1, +datadog.AggregatedSmapEntry 0, datadog.DeadlockedThread 0, +datadog.SmapEntry 0, datadog.BackpressureSample 0, datadog.Deadlock 0 +``` + +i.e. dd-trace-java's own JFR only — see `dead-jcmd-jfr-dump-wrong-source`. diff --git a/.investigations/missing-refchains-on-hotdog/meta.yaml b/.investigations/missing-refchains-on-hotdog/meta.yaml new file mode 100644 index 0000000000..09a62cf2a9 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/meta.yaml @@ -0,0 +1,8 @@ +name: missing-refchains-on-hotdog +description: Why prof-analyzer-hotdog-jb emits zero datadog.ReferenceChain events +status: active +created: 2026-08-24 +updated: 2026-09-16 +completed: null +last_commit: 50edce5bb8c1d8fe471efb07dd850543215cb55c +last_session: e6d2c60c-e959-420c-9b65-5e0400073a24 diff --git a/.investigations/missing-refchains-on-hotdog/nodes/dead-hard-reference-kind-filter.md b/.investigations/missing-refchains-on-hotdog/nodes/dead-hard-reference-kind-filter.md new file mode 100644 index 0000000000..f37acbf770 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/dead-hard-reference-kind-filter.md @@ -0,0 +1,59 @@ +--- +id: dead-hard-reference-kind-filter +type: deadend +status: refuted +depends_on: [ev-kind-counts-constant-pool-dominates] +supersedes: [] +related: [find-candidate1-never-tagged] +tags: [referenceChains, static-field, constant-pool, reference-kind, filter, refuted, NEW-THIS-SESSION] +created: 2026-08-26 +updated: 2026-08-26 +--- + +# Hard reference_kind filter (drop ALL non-STATIC_FIELD) — superseded by per-class quota + +## What was tried + +Implemented at `referenceChains.cpp:1672-1686` (uncommitted, this +session's first iteration): when `ctx->static_field_seed` is true and +the referrer is the opened class object (`*referrer_tag_ptr < 0`), skip +admission (`return 0`) for any `reference_kind != +JVMTI_HEAP_REFERENCE_STATIC_FIELD`. This dropped ALL non-STATIC_FIELD +edges (CONSTANT_POOL, INTERFACE, SUPERCLASS, CLASS_LOADER, etc.) during +the static-field seed sweep. + +## Why it was rejected + +User's real-world leak taxonomy (from direct field experience): +- **Static fields** — most common leak source +- **Unmaintained collections in singletons, reachable via static fields** — + common; the singleton is reached via a static field, the collection via + the singleton's instance fields +- **Thread-local leaks** — possible; `Thread.threadLocals` is an instance + field of `Thread` objects, reached via `JVMTI_HEAP_ROOT_THREAD` in the + general root enumeration path (`heapRootCallback`, line 2169), NOT via + static fields. So thread-local leaks are NOT in scope of the + static-field sweep at all — unaffected by any filter on this path. + +The hard filter completely excluded CP-based leaks. While CP leaks are +rarer than static-field leaks (CONSTANT_POOL volume is 5-15x STATIC_FIELD +per class, systemically — see `ev-kind-counts-constant-pool-dominates`), +they are still a real leak category. The user's explicit requirement: +"we need to design a system working with this priority and not pushing +completely out one or the other." + +## What replaced it + +Per-class non-static quota (`find-candidate1-never-tagged`'s updated fix +section): STATIC_FIELD edges always admitted; non-STATIC_FIELD edges +admitted up to `STATIC_FIELD_SWEEP_NON_STATIC_CAP_PER_CLASS = 32` per +class per lap, then dropped for the rest of that class. Cap resets on +class boundary. This keeps CP-based leaks in scope (bounded) while +prioritizing static fields — exactly the "priority, not exclusion" the +user asked for. + +## What this rules out + +Any design that completely drops a real leak category from the +static-field sweep. The sweep must admit non-static edges in at least a +bounded quantity to keep CP-based leaks discoverable. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/dead-jcmd-jfr-dump-wrong-source-v2.md b/.investigations/missing-refchains-on-hotdog/nodes/dead-jcmd-jfr-dump-wrong-source-v2.md new file mode 100644 index 0000000000..8cdcfbc1f8 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/dead-jcmd-jfr-dump-wrong-source-v2.md @@ -0,0 +1,44 @@ +--- +id: dead-jcmd-jfr-dump-wrong-source-v2 +type: deadend +status: refuted +depends_on: [] +supersedes: [dead-jcmd-jfr-dump-wrong-source] +related: [] +tags: [methodology, wrong-evidence-source, jcmd, jfr, ddprof, NEW-THIS-SESSION] +created: 2026-08-28 +updated: 2026-08-28 +--- + +# jcmd JFR.dump does NOT capture ddprof native JFR events + +## What happened + +Used `jcmd JFR.dump name=dd-profiling filename=/tmp/dump.jfr` to +try to read `datadog.ReferenceChain` events. The dump showed only JDK +built-in event types — no `datadog.ReferenceChain`, no +`datadog.HeapLiveObject`, none of ddprof's custom events. + +## Root cause + +`jcmd JFR.dump` dumps the JVM's built-in JFR recording (managed by +`jdk.jfr` API). ddprof writes its own JFR chunks directly to +`/tmp/ddprof_root/pid_XXX/jfr/` via its native `FlightRecorder` class +(`flightRecorder.cpp`). These are two completely separate JFR writers. +`jcmd` has no visibility into ddprof's native writer. + +## Correct method + +To get ddprof's JFR events: +1. `kubectl cp` the chunk files from + `/tmp/ddprof_root/pid_XXX/jfr/` on the pod, OR +2. Use the profiling toolkit `download.py` script to download uploaded + profiles from the Profiling Toolkit API (if the service uploads + profiles — hotdog staging does not). + +## Lesson + +This was already documented as `dead-jcmd-jfr-dump-wrong-source` but +was repeated again in this session. The mistake is easy to make because +`jcmd JFR.dump` appears to work (produces a .jfr file with events) — +it just produces the wrong JFR stream. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/dead-jcmd-jfr-dump-wrong-source.md b/.investigations/missing-refchains-on-hotdog/nodes/dead-jcmd-jfr-dump-wrong-source.md new file mode 100644 index 0000000000..05680f9e22 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/dead-jcmd-jfr-dump-wrong-source.md @@ -0,0 +1,49 @@ +--- +id: dead-jcmd-jfr-dump-wrong-source +type: deadend +status: refuted +depends_on: [ev-uploaded-jfr-no-refchain-types] +supersedes: [] +related: [find-onpod-evidence-methodology, dead-toolkit-prod-datacenter] +tags: [methodology, wrong-evidence-source, jcmd, jfr] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Dead end: `jcmd JFR.dump` is the wrong evidence source for ddprof events + +## Reasoning chain + +The first attempt to answer "are there reference-chain events?" was to +dump the live recording on the pod: + +``` +jcmd 231 JFR.dump filename=/tmp/hotdog-check-jb.jfr # 1.7 MB +kubectl cp … /tmp/hotdog-jfr/hotdog-check-jb.jfr +``` + +jafar's `jfr_list_types filter=datadog scan=true` on that dump returned 11 +types / 220 events: `ProfilerSetting` 110, `ExceptionSample` 83, +`ExceptionCount` 19, `AvailableProcessorCores` 6, two `DirectAllocation*` +at 1, and five zero-count types. `filter=chain` and `filter=liveheap` +returned **0 types** — which looks like damning evidence but proves +nothing. + +Reason: `JFR.dump` captures only the JVM-side `dd-profiling` recording +that dd-trace-java itself drives. ddprof writes its own recording and +uploads it separately. The absence of ddprof event types in a `JFR.dump` +is expected regardless of whether the feature works. Note the same dump +also lacked `datadog.ExecutionSample`, `MethodSample`, `ObjectSample` and +`HeapLiveObject` — all features known to be working. + +The correct source is the uploaded profile fetched via the Profiling +Toolkit (`ev-uploaded-jfr-no-refchain-types`), whose type list is ~2x +longer and does include the ddprof events. + +## Evidence +- `evidence/ev-uploaded-jfr-no-refchain-types.md` (contains both type lists + side by side) + +## What this rules out +- Using `jcmd JFR.dump` / the `hotdog-check --jfr` path to reason about any + ddprof-emitted event type. Always download the uploaded profile instead. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/dead-toolkit-prod-datacenter.md b/.investigations/missing-refchains-on-hotdog/nodes/dead-toolkit-prod-datacenter.md new file mode 100644 index 0000000000..f5c15196ec --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/dead-toolkit-prod-datacenter.md @@ -0,0 +1,55 @@ +--- +id: dead-toolkit-prod-datacenter +type: deadend +status: refuted +depends_on: [ev-toolkit-and-onpod-methodology] +supersedes: [] +related: [dead-jcmd-jfr-dump-wrong-source, find-onpod-evidence-methodology] +tags: [profiling-toolkit, tooling-gotcha, datacenter, org-id] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Dead end: `--datacenter us1.prod.dog` for staging hotdog profiles + +## Reasoning chain + +`download.py` has no datacenter default and refuses to guess. The first +several attempts used `--org-id 2 --datacenter us1.prod.dog` and reported +`Found 0 matching profiles` for every service/app/query combination — +which briefly looked like "the pod isn't uploading at all". + +Probing with no service filter disambiguated it: + +``` +=== no service filter, org 2, prod dc === + ERROR: Download failed: HTTP 404 - {"errors":[{"code":"ObjectNotFoundException"}]} + Query: * + Found 1 matching profiles +=== no service filter, org 2, staging dc === + Query: * + Found 1 matching profiles +``` + +So `us1.prod.dog` can *find* profiles but 404s on blob download. The +working combination is org 2 + `--datacenter us1.staging.dog`, which then +returned 5 profiles for +`service:prof-analyzer-hotdog app:prof-analyzer-hotdog-jb`. + +Secondary gotchas found along the way: adding `--env staging` alongside +`--service`/`--app` produced 0 hits even though matching profiles existed; +`--service` etc. must precede other args or argparse rejects them as +unrecognized; and a large share of results are `.tar` blobs containing +only `cpu.pprof` ("no JFR found in tar"). + +## Evidence +- `evidence/ev-toolkit-and-onpod-methodology.md` + +## What this rules out +- Concluding "the pod isn't uploading profiles" from a toolkit + `Found 0 matching profiles` result before verifying org-id and + datacenter. +- Note this contradicts the previously recorded + "profiling-toolkit needs the prod URL" rule for *staging pods* in this + particular case: here the staging datacenter was required for the blob + download to succeed. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/design-pod-in-a-jar-harness.md b/.investigations/missing-refchains-on-hotdog/nodes/design-pod-in-a-jar-harness.md new file mode 100644 index 0000000000..27dd3ec496 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/design-pod-in-a-jar-harness.md @@ -0,0 +1,127 @@ +--- +id: design-pod-in-a-jar-harness +type: design +status: implemented-live +depends_on: [meta-whackamole-analysis] +related: [find-round16-endgoal-verification, find-fresh-lane-verification, find-canary-found-criterion-unmigrated, find-refchains-log-flood-configurable, find-urgentoom-null-fn-mislabel] +tags: [design, system-test, invariants, harness, NEW-THIS-SESSION] +created: 20260916 +updated: 20260916 +--- + +# Design: "pod in a jar" — system-level simulation harness (the whackamole exit) + +STATUS (round 20, de15e521f): BUILT AND FULLY LIVE — all six invariants +run in CI (128 tests / 127 pass / 1 skip-with-reason): L1 leak chains, +L2 canary resolution, L3 natural completion, L6 restart hygiene +(asserting the designed chains-persist contract), L8 dormancy, L4 log +budget (compile-aware — live in DEBUG-built test binaries). The +construction addendum below records what it took; the first-catch +retraction (find-candidate-presence-suppresses-static-admission) +records the builder's own capacity-contract lesson. + +One new gtest suite driving the REAL ReferenceChainTracker loop +(threadLoop cadence or directly-driven passes with a fake clock) over a +mock JVMTI heap topology. No kubectl, no deploy, seconds per run. + +## Topology builder (encodes every pod-discovered shape) + +- 1 static holder (the LEAK_BUFFER shape) holding a synchronized-list + wrapper — including its `mutex == this` self-edge (round 16's fix-A + shape) and a real ArrayList c → elementData → chunks subgraph +- K leak-tagged [B chunks that ACCUMULATE per simulated poll (the leak) +- Flood classes: ≥16 statics each of 2-3 classes at high volume (the + quota/flood shape, round 15-16) +- Direct-static [B noise + transient-root [B noise (the suppression- + filter shapes) + thread-locals with candidate tids +- Optional healthy-app mode: no candidates at all (the accidental + control run — dormancy invariant) +- Configurable scale knobs: anchors count, flood volume, pass budget, + frontier cap, search lifetime — the matrix that took 10 pod rounds to + explore one point at a time + +The BfsTest suite's mock JVMTI machinery (65 tests) already implements +the FollowReferences/poll primitives — this harness composes them into a +full-loop driver rather than inventing mocks. + +## Invariants (each maps to real rounds it would have caught) + +- **L1 liveness/coverage**: every leak-tagged chunk reachable from a + walked static has a cached chain with target_tag = its leak tag + within K passes. [rounds 9-16: the entire funnel saga] +- **L2 canary resolution**: the chase reports 1/1 within K passes of the + first leak-tag chain (round 19 — fails on the pre-788d7b2a7 build; the + found-criterion bug becomes a CI failure, not a 3-JVM pod mystery) +- **L3 natural completion**: a healthy-topology search exits "all + candidates found" — exits-by-frontier-cap/no-progress are harness + failures (rounds 10-19; also covers the sweep-truncation flag + cycle_complete oddity) +- **L4 log budget**: at any rcDebugLevel, no distinct log line shape + exceeds N lines/pass (rounds 17-18: the 800k flood AND the two tier + stragglers become assertion failures — capture stdout per pass and + group by line prefix) +- **L5 quotas/monotonicity**: per-class FIFO ≤ 64, self_edge_skips and + quota counters monotonic, FIFO+queue sizes within caps (round 16) +- **L6 restart hygiene**: after restartSearch, enumerate and assert + ZERO surviving state — tags, frontier entries, discovered slots, + resolved chains, marker/leak tags, rotation queues (rounds 12-14's + restart-survivor bugs + the reset() contract as a TEST, not discipline) +- **L7 boundedness**: no per-search structure grows across M consecutive + searches (the rotation-resize-blindspot class) +- **L8 dormancy**: healthy topology → zero searches, zero candidates, + zero log volume (the 3h accidental control, encoded) + +Plus a soak mode: M simulated search generations with a leak that grows +and occasionally mutates shape (holder changes klass) — the +representative/canary lifecycle under churn. + +## What this changes operationally + +- Every class-1/2/3/5/6 defect becomes locally reproducible in seconds: + the pod stops being the oracle. Deploys become verification, not + discovery. +- Pod verification collapses to the always-on health line (P2) + one + level-1 window — checking numbers, not grepping for surprises. +- Upstreaming review runs the harness as the regression net. + +## Construction addendum (what it took, round 20 — for the next extension) + +- The driver replicates threadLoop's body EXACTLY (shouldRunPass -> + runPass -> poll, poll unconditional) with a fake clock SEEDED FROM + OS::nanotime() — the pain budgets' _last_update_ns is real-clock + based, so a fake epoch at ~0 blocks every pass (only the + pass-running tests catch this; dormancy passes either way). +- Assertions on tags MUST use tags_ever_assigned, never node_tags: a + completed search releases node_tags (the fixture's own comment + documents this; re-learned in bisect). +- The first pass's budget auto-scales 10x (arguments.cpp), so budget=N + does not mean N edges on pass one; a "multi-pass" search needs a + filler graph, not just a small budget. +- Two-phase bring-up resolves the klass-id circularity (candidates need + the id; the id comes from the system's own resolve() over a real + admission): throwaway candidate arms the signal -> pass 1 admits -> + classTags-based id resolution (class tags are process-lifetime, + survive completion+release) -> restart -> re-seed leak tags (the + poll's re-tagging) -> poll BEFORE the first pass (slots register in + the poll; admission auto-mark needs the slot to exist). +- The Bfs fixture lacks the NewLocalRef JNI slot + (PollWatchedTargetsTest wires its own) — resolveCandidateRepresentative + crashes without it. +- The gtest binary is non-DEBUG: TEST_LOG is compiled out — probe with + ASSERT/EXPECT-based diagnostics, not logs. + +## Cost and sequencing (proposal) + +- P0 harness core (topology builder + L1-L3, L6) — 1 focused session; + reuses BfsTest mocks + the tier/quota fixtures from round-16 tests +- P0b L4 log-budget + L8 dormancy + soak — small follow-up +- P1 keyspaces audit + dead marker-path retirement — half session + (grep-driven consumer enumeration; one gtest per join) +- P2 always-on per-search health line (Counters: exits-by-reason, + canary-resolved-pass histogram, coverage ratio) — small +- P3 = L6 (already in P0), P4 deferred queue triaged to harness-flagged + +Not proposed: any new pod-driven fix rounds. The deferred queue (sporadic +batch coverage, cycle_complete, reconstructChain noise, wrapper tier +position, TEMP diag slimming) is handled by the harness or deliberately +accepted for upstreaming. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-abandon-event-lost-to-dump-sampling-race.md b/.investigations/missing-refchains-on-hotdog/nodes/find-abandon-event-lost-to-dump-sampling-race.md new file mode 100644 index 0000000000..4da2fb51f1 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-abandon-event-lost-to-dump-sampling-race.md @@ -0,0 +1,131 @@ +--- +id: find-abandon-event-lost-to-dump-sampling-race +type: finding +status: confirmed +depends_on: [ev-postfix-onpod-live-verification] +supersedes: [] +related: [find-canary-stuck-abandon-detector, find-canary-search-cannot-terminate, ev-jafar-zero-refchain-events] +tags: [root-cause, jfr-emission, race-condition, search-state, dump, referenceChains, NEW-BUG] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# NEW ROOT CAUSE: SearchState::ABANDONED is transient and almost always overwritten before Profiler::dump() ever samples it + +## Reasoning chain + +With Fix A/B/C live and working (`ev-postfix-onpod-live-verification`), the +canary search now cleanly reaches `SearchState::ABANDONED` every time it +gets stuck (`abandonReason=3`, confirmed 5x in pod logs). But +`datadog.ReferenceChainAbandoned` and `datadog.ReferenceChain` both show +count **0** in every downloaded JFR chunk checked so far, including one +whose empirically-derived time range (`09:01:37.95` -> `09:02:37.56`, +established via `datadog.SafepointBegin`'s `startTime` min/max, see below) +squarely brackets a confirmed pod-log abandon event at `09:01:53`. + +The user's working hypothesis mid-session was that abandon *counters* might +not carry over across ~60s profile-chunk boundaries. **That hypothesis is +refuted by this evidence**: the abandon event at 09:01:53 falls fully +inside the chunk's own 09:01:37-09:02:37 window, not near either edge, so a +pure boundary-clipping explanation cannot account for the 0 count. + +The actual root cause is a **read/write race on `_search_state`, on two +different clocks**: + +1. `Profiler::dump()` (`profiler.cpp:2055-2062`) is the *only* place that + reads `ReferenceChainTracker::instance()->searchState() == + SearchState::ABANDONED` to build and emit a + `datadog.ReferenceChainAbandoned` event. It only runs on JFR chunk + rotation - roughly a 60s cadence in this deployment, entirely decoupled + from the BFS thread's own timing. +2. `ReferenceChainTracker::threadLoop()`'s own BFS thread calls + `shouldRunPass()` on a fixed ~1s cadence + (`referenceChains.cpp:840-935`). When `_search_state != RUNNING` (i.e. + terminal - COMPLETED or ABANDONED) and `_tags_released` is true, + `shouldRunPass()` calls `canAffordNewSearch()` and, if it returns true, + calls `restartSearch()` **synchronously, inline** (`:876-881`), which + sets `_search_state` back to `RUNNING` + (`referenceChains.cpp:1145` inside `restartSearch()`, confirmed at + `:1036`). + +Pod logs already show this restart happening on the very next iteration +after every observed abandon (`abandonReason=3` -> next line +`shouldRunPass -> true (restarting search)` -> `passesRun` reset to 0) - +i.e. `SearchState::ABANDONED` is only actually readable for **about one BFS +loop iteration (~1s)** out of every ~60s JFR chunk. `dump()`'s snapshot has +roughly a 1-in-60 chance of landing inside that window on any given chunk +rotation. Across 5 confirmed abandon events and the several chunks sampled +so far, seeing 0 hits is the expected outcome of this race, not evidence of +a missing/broken emission path. + +This is architecturally the same shape of bug as +`find-canary-search-cannot-terminate`'s livelock, but on the *reporting* +side rather than the *search* side: a transient state that self-heals +faster than the only consumer that reads it gets scheduled. + +## Evidence + +- `flightRecorder.cpp:2140` `Recording::recordReferenceChain()`, + `:2192` `Recording::recordReferenceChainAbandoned()` - the JFR write + functions themselves, confirmed to exist and be well-formed (not the + bug). +- `profiler.cpp:863-876` `Profiler::writeReferenceChainAbandoned()` and + `:898-943` `Profiler::writeReferenceChain()` - both correctly wired to + `_jfr.recordReferenceChain(Abandoned)`; confirmed the only two call sites + of the record functions in the whole codebase (grepped + `ddprof-lib/src/main/cpp/`). +- `profiler.cpp:2049-2086` `Profiler::dump()` - the **only** call site of + `writeReferenceChainAbandoned()`/`writeReferenceChain()`, gated on + `searchState() == SearchState::ABANDONED` read at dump-time. +- `referenceChains.cpp:857-881` `shouldRunPass()`'s terminal-state branch - + calls `restartSearch()` inline the very next time it's invoked once + `canAffordNewSearch()` allows it, flipping state back to `RUNNING`. +- `referenceChains.cpp:1036` `restartSearch()`, + `referenceChains.cpp:1145` sets `_search_state = RUNNING`. +- Empirical chunk-boundary check (this session): + `prof-analyzer-hotdog-2026-08-25_09-02-37.564Z-...jfr` opened via jafar + (sessionId 21). `events/datadog.ReferenceChainAbandoned | count()` = 0, + `events/datadog.ReferenceChain | count()` = 0. + `events/datadog.SafepointBegin | select(startTime)` (limit 2000, 820 + results) gives min `1787648497953939200` ns = `2026-08-25 09:01:37.953939` + UTC and max `1787648557564531000` ns = `2026-08-25 09:02:37.564531` UTC - + a ~59.6s chunk squarely containing the confirmed pod-log abandon at + `09:01:53`. +- (Earlier, less precise) `events/datadog.ProfilerCounter | + select(startTime,name)` returned 163 results all with the identical + `startTime=1787648497743735000` (~09:01:37.74) - a registration burst at + chunk start, useful only for establishing the chunk's start, not its end; + superseded by the SafepointBegin min/max above for the full range. + +## What this rules out + +- **The user's chunk-boundary/counter-carryover hypothesis**, at least as + a full explanation - refuted directly: the 09:01:53 abandon is not near + either edge of the 09:01:37-09:02:37 chunk, yet the event count is still + 0. +- **A broken/missing JFR write path** - the record/write functions are + present, correctly wired, and the only bug is in *when* the ABANDONED + state is observable, not in how the event is serialized once built. +- **Fix C not actually abandoning** - it clearly is (5 confirmed + `abandonReason=3` cycles); the gap is entirely in `dump()`'s sampling of + that transient state. + +## Not yet done + +- No fix proposed or implemented for this - purely diagnostic so far. A fix + would need `dump()` to observe the abandon event through something other + than a live re-read of `_search_state` at an arbitrary later time - e.g. + a one-shot "pending abandon" flag/queue set by `runPass()`'s abandon + branch and cleared by `dump()`'s own read, mirroring how + `_resolved_chains` already handles the analogous case for successful + chains (`drainPendingChainEvents()` snapshots without clearing - though + note that pattern re-reports on every dump rather than being one-shot, + so it is not a direct template; a `ReferenceChain` success is presumably + affected by the exact same underlying race, since it is also only + read/emitted from `dump()`). +- Whether `datadog.ReferenceChain` (the success event, not the abandon + event) is affected by the *same* race, or a different one, is not yet + separately confirmed - `find-onpod-verification` above notes zero + candidates have ever resolved on this pod, so that path has not yet even + been exercised to test independently. Worth revisiting once a resolution + is observed. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-abandon-event-queue-fix.md b/.investigations/missing-refchains-on-hotdog/nodes/find-abandon-event-queue-fix.md new file mode 100644 index 0000000000..4c6856eb89 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-abandon-event-queue-fix.md @@ -0,0 +1,89 @@ +--- +id: find-abandon-event-queue-fix +type: finding +status: confirmed +depends_on: [find-abandon-event-lost-to-dump-sampling-race] +supersedes: [] +related: [find-canary-stuck-abandon-detector, ev-fixes-compile-and-gtest-pass] +tags: [fix, jfr-emission, race-condition, referenceChains, queue, NEW-THIS-SESSION] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# Fix D: bounded pending-abandoned-events queue closes the dump() sampling race + +## Reasoning chain + +User's own framing of the two remaining directions (verbatim): "keep counter +of abandoned searches and emit the event using that counter; resetting it on +dump" vs. "figure out why we are not getting closer to the canaries". User +chose both, in that order. + +A literal single integer counter cannot work: `restartSearch()` +(`referenceChains.cpp:1036`) clears `_abandon_reason`, `_search_start_ns`, +`_last_pass_ns`, `_passes_run` — the exact fields `buildAbandonedEvent()` +needs to populate `ReferenceChainAbandonedEvent`'s payload +(`_reason`/`_passes_run`/`_frontier_size`/`_hop_cap`/`_budget`/`_ttl_ms`/ +`_elapsed_ns`, see `event.h:126-141`). A bare count would lose all of that +diagnostic detail. The user was told this reasoning directly and separately +questioned whether per-event fidelity was worth it at all ("why not just +emit the number of abandoned searches") — answered with the same point plus +the observed abandon rate (~5 per 20 min, under one per 60s dump interval), +making per-event cost negligible. User accepted ("ok. fair points"). + +Implemented instead: a bounded queue (`_pending_abandoned_events`, +`MAX_PENDING_ABANDONED_EVENTS = 16`) that snapshots a fully-built +`ReferenceChainAbandonedEvent` at the exact moment of abandon — synchronously, +on the BFS thread, before `restartSearch()` can run and clear the source +fields. `Profiler::dump()` now does a **true drain** (not the +snapshot-and-keep re-emit pattern `_resolved_chains`/`drainPendingChainEvents()` +uses), since an abandon is a one-off past occurrence, not an ongoing live +sample. + +## Changes made + +- `referenceChains.h`: added `_pending_abandoned_events` (vector), + `_pending_abandoned_events_lock` (SpinLock), `MAX_PENDING_ABANDONED_EVENTS` + constant; declared `enqueuePendingAbandonedEvent()` (private) and + `drainPendingAbandonedEvents(std::vector*)` + (public). +- `referenceChains.cpp`: + - `enqueuePendingAbandonedEvent()` builds the event via the existing + `buildAbandonedEvent()`, pushes it under the lock, drops with + `Counters::increment(REFERENCE_CHAIN_EVENTS_DROPPED)` + `TEST_LOG` if + the queue is at cap (mirrors `cacheResolvedChain()`'s no-silent-drop + pattern). + - `drainPendingAbandonedEvents()` moves the whole vector out under the + lock and clears it. + - Called `enqueuePendingAbandonedEvent()` immediately after + `storeRelease(_search_state, (u8)SearchState::ABANDONED)` at all three + abandon sites in `runPass()` (FRONTIER_CAP, TTL, CANARY_STUCK branches). + - `resetSearchStateForTest()` now also clears `_pending_abandoned_events` + under its lock, next to the existing `_resolved_chains.clear()` — same + test-isolation rationale (one test's abandons must not leak into the + next). +- `profiler.cpp` `Profiler::dump()` (was lines 2049-2062): replaced the old + `searchState() == SearchState::ABANDONED` + single `buildAbandonedEvent()` + read with a call to `drainPendingAbandonedEvents()` and a loop calling + `writeReferenceChainAbandoned()` per drained event. + +## Verification + +- `./gradlew :ddprof-lib:compileDebug -Pskip-gtest` — compiles cleanly, no + new warnings. +- `./gradlew :ddprof-lib:gtestDebug` (full suite, no filter available on this + gradle task) — BUILD SUCCESSFUL, no test failures. +- **Not yet done**: no new gtest added that specifically exercises the queue + (e.g. abandon -> drain -> assert event contents survive a subsequent + `restartSearch()`) — not requested, existing suite already passes so no + regression, but the new code path itself has no direct test coverage yet. +- **Committed and pushed** as `01047a6aa` on `jb/reference-chains` (on top + of `623d3712a`), together with Fix E + Fix F + (`find-canary-fixes-e-f`). **Not yet done**: not deployed/re-verified on + the hotdog pod. + +## What this rules out + +- Plain-integer-counter design — insufficient, would lose per-event + diagnostic payload; confirmed by design necessity, then separately + challenged and re-confirmed by the user. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-admission-boost-implementation.md b/.investigations/missing-refchains-on-hotdog/nodes/find-admission-boost-implementation.md new file mode 100644 index 0000000000..9e698a9835 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-admission-boost-implementation.md @@ -0,0 +1,84 @@ +--- +id: find-admission-boost-implementation +type: finding +status: implemented +depends_on: [find-default-live-samples-ratio-lottery, find-canary-lane-backoff-design] +related: [q-togcroot-acceptance-paths, find-per-tid-qualification-design] +tags: [fix, design, livenessTracker, admission, work-scaled, NEW-THIS-SESSION] +created: 20260901 +updated: 20260901 +--- + +# Chase-phase admission boost (user option A): raise tracking probability once a leak is detected, or under urgency + +## The design (user-picked from A/B/C) + +`LivenessTracker::admitForTracking(tid)` is track()'s new admission +gate. Two raises over the configured _live_samples_ratio (default 10%): + +1. Watched tids - selectLeakCandidates()'s qualifying_tids (at most 8) + published by noteSelectedCandidates() from RCT's FULL poll + (pollWatchedTargets()); admitted at 100%. Bounded by the candidate + threads' own allocation rate, so tracking-table volume scales with + the leak's threads, not the process. Refreshed and CLEARED poll by + poll (zero-candidate polls clear it - a stale watched tid would + 100%-admit an unrelated thread across OS tid reuse). NOT called from + hasLeakSignal()'s max=1 probe (partial view could drop other active + candidates' tids). +2. Urgency - setUrgentTracking(urgent) called every threadLoop() + iteration next to _oom_ramp_active, so the boost tracks the ramp + exactly. Admits everything: under the OOM ramp maximizing last- + chapter capture outweighs table volume. Transition-only TEST_LOG. + +Fail-open by construction: boosts only add admissions. Two-phase +count+array publish with RELEASE/ACQUIRE per the project's arm64 +ordering rule. Boosted admissions skip the RNG draw. Cleared on +fresh start(). Rejected alternatives: global-only raise (10x table +cost for the whole process during the entire chase) and klass-scoped +via jclass (GC-move-unsafe pointer identity; IsSameObject on the +reject path is not free; per-tid already covers tagLeakInstances()' +exact tagging scope). + +## The user's scope correction (accepted, important) + +Detection is NOT compromised by the default 10% at any ratio: the +subsample scales the signal, it does not gate it. A leak that fills a +large part of the heap leaves a proportional tracked population and a +positive trend every epoch - detection fires as surely as at 100%. +The 10% residue on detection is LATENCY only (spotty per-epoch counts +on slow small-rate leaks delay hysteresis clearing). A cohort small +enough for the lottery to zero it out is too small for ANY machinery +to act on early (per-tid bar 8, trend hysteresis would rank it +nowhere) - the local test's artificial small cohort, not a production +shape. Division of labor: default ratio bounds steady-state cost; +detection stays asymptotically certain; the boost restores 100% +fidelity exactly for the chase phase where fidelity matters. + +## Verification + +- 4 gtests (AdmissionBoostTest, livenessTracker_ut.cpp): watched-tid + admitted despite ratio 0 (RNG reset via admissionResetForTest makes + the fall-through reject deterministic - fresh mt19937 default seed's + first draw is strictly in (0,1)); urgency admits all + releases; + dedupe + cap at MAX_QUALIFYING_TIDS=8 in candidate order; + zero-candidate poll clears. 550 gtests green. +- Slow suite: boost engaged in children (noteSelectedCandidates watched + tid lines; scenario tid published), leak-correlation + UnboundedCache + green; ToGcRoot failed with the IDENTICAL pre-boost signature - the + known q-togcroot-acceptance-paths family, not a regression (suite + tests run :l:1.0 so admission was already 100%). +- Doc: ReferenceChains-SignalsExplained.md section 5 extended (the + leak-signal also raises tracking fidelity; urgency raises admission + to 100%). + +## Code landmarks + +admitForTracking()/noteSelectedCandidates()/setUrgentTracking() decls ++ _watched_tids/_watched_tid_count/_urgent_tracking fields in +livenessTracker.h (fields next to _subsample_ratio); implementations +in livenessTracker.cpp near releaseThreadLocalState; track() gate +replaces the inline RNG block; RCT call sites: threadLoop (~_oom_ramp_active +store, setUrgentTracking next to it) and pollWatchedTargets (right +after selectLeakCandidates). ForTest seams: admitForTrackingForTest, +setSubsampleRatioForTest, admissionResetForTest (cpp - clears TLS RNG), +watchedTidCountForTest, watchedTidForTest. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-age-heuristic-insufficient.md b/.investigations/missing-refchains-on-hotdog/nodes/find-age-heuristic-insufficient.md new file mode 100644 index 0000000000..665cd25a93 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-age-heuristic-insufficient.md @@ -0,0 +1,54 @@ +--- +id: find-age-heuristic-insufficient +type: finding +status: diagnosed +depends_on: [find-representative-changes-lose-canary] +supersedes: [] +related: [find-representative-changes-lose-canary, q-allocation-site-selection] +tags: [root-cause, referenceChains, livenessTracker, age-heuristic, representative-selection, NEW-THIS-SESSION] +created: 2026-08-28 +updated: 2026-08-28 +--- + +# Age heuristic for representative selection is insufficient + +## Observation + +JFR analysis showed 13 [B instances in `datadog.HeapLiveObject`: + +- 12 leaking: 78MB each, tid=172 (`simulated-memory-leak`), ages 33-168 + (only age=168 has allocation stack: `lambda$static$1` in `ProfileAnalyzer`) +- 1 noise: 136B, tid=284 (`s3-netty-2`), age=172, stack=`initClassName` + +The oldest [B (age=172) is the **noise** — a class-name initialization +buffer. The leaking instances have ages 33-168. Pure age ranking (Lindy +heuristic) picks the noise first. + +## Root cause + +LivenessTracker's representative was selected as the oldest surviving +instance per klass (GC age descending). For [B, the oldest surviving +instance is a 136B class-init buffer, not a 78MB leaking array. The +Lindy heuristic assumes older = more likely to be a leak, but for +common classes with diverse allocation sites, the oldest instance may +be a long-lived framework object, not a leak. + +## Proposed directions (user-approved: 1+3 combined) + +1. **Allocation-site clustering** (Cork, SOSP'23; Melt, OSDI'15): group + surviving objects by `(klass_id, call_trace_id)` instead of just + `klass_id`. Track growth trend per allocation site. Select + representatives from the highest-growth site, not just the oldest + instance. `TrackingEntry` already has `call_trace_id`. + +3. **Growth-rate × survival-count** (Swat, SOSP'19): per allocation + site, compute `growth_rate × survival_count`. A site with 12 + surviving instances across 12 GC ages (leak) outscores a site with + 1 surviving instance (noise) by 12×, regardless of individual age + or size. Addresses the "frequent but small" concern: frequent + small allocations that die quickly score low (low survival); frequent + small allocations that survive score high (real leak). + +User rejected: size-weighted selection (shadows frequent small leaks). +User noted: retained-size weighting (direction 2) lacks data in current +implementation. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-ages-vector-not-cleared.md b/.investigations/missing-refchains-on-hotdog/nodes/find-ages-vector-not-cleared.md new file mode 100644 index 0000000000..0494e32386 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-ages-vector-not-cleared.md @@ -0,0 +1,47 @@ +--- +id: find-ages-vector-not-cleared +type: finding +status: confirmed +depends_on: [q-dominant-gens-still-one-with-tid] +supersedes: [] +related: [q-dominant-gens-still-one-with-tid] +tags: [bug, fix, referenceChains, livenessTracker, ages, vector, inflation, NEW-THIS-SESSION] +created: 20260828 +updated: 20260828 +--- + +# ages vector not cleared between epochs + +## Context + +Per-thread diagnostic (5e4493dbf) showed `gen_count=33` for klass_id=4 +but all threads had `age_count=1`. User challenged: 11 distinct ages +across 5 threads means at least one thread has ≥3 ages — mathematically +impossible to have all threads at 1. + +## Root cause + +`KlassCountScratch::ages` is a `std::vector`. When the scratch +slot is reused across epochs (after `_klass_count_scratch_size = 0`), +the vector is **never cleared**. `push_back` adds to stale ages from +previous epochs. So `gen_count=33` was accumulated across multiple +epochs, not within one. + +Meanwhile `oldest_count` and `thread_count` ARE reset to 0, so +per-thread tracking is per-epoch only. + +## Impact + +- Logged `gen_count` was inflated (cumulative across epochs) +- Ring-buffer slope was **unaffected** — it measures rate of change + of `ages.size()`, which equals the per-epoch increment regardless + of the absolute value +- So the leak signal was correct, but the diagnostic logs were + misleading + +## Fix (commit 36c7fc8c8) + +Added `slot.ages.clear()` before `slot.ages.push_back((u32)age)` in +the new-slot branch of `accumulateKlassCount()`. After fix, on-pod +logs show `gen_count=1` for a class with 1 surviving object per epoch +— correct. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-already-admitted-blocks-deeper-chain.md b/.investigations/missing-refchains-on-hotdog/nodes/find-already-admitted-blocks-deeper-chain.md new file mode 100644 index 0000000000..70c4b022fc --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-already-admitted-blocks-deeper-chain.md @@ -0,0 +1,51 @@ +--- +id: find-already-admitted-blocks-deeper-chain +type: finding +status: confirmed +depends_on: [find-per-class-caching-blocks-instances] +supersedes: [] +related: [find-per-class-caching-blocks-instances] +tags: [root-cause, fix, referenceChains, admitObject, already-admitted, depth-1, chain, jni-local, NEW-THIS-SESSION] +created: 20260828 +updated: 20260828 +--- + +# Already-admitted objects block deeper chain reconstruction + +## Context + +After per-instance chain caching (cd68be618) and allocation-site +clustering (2c50bf0cd, 36c7fc8c8), 8 ReferenceChain events were +emitted for [B on-pod. But the user reported chains are depth=1 with +only [B and no holder — not useful for diagnosing the leak. + +## Root cause + +`admitObject()` checks `if (*tag_ptr != 0) return ALREADY_ADMITTED;` +— once an object is tagged and admitted, it's never re-admitted. + +When [B is first reached as a JNI-local root (referrer_tag_ptr == +nullptr, parent_tag=0, depth=0), it gets tagged. When the +static-field → ArrayList → [B path reaches it later with a non-zero +parent_tag and deeper depth, `ALREADY_ADMITTED` is returned and the +deeper chain is lost. + +`reconstructChain()` then walks parent_tag back to root, but +parent_tag == 0 → chain is just [referrer_klass of the [B itself] → +depth=1, no holder. + +## Fix (commit d30538fe3) + +`FrontierTable::improveChain(tag, parent_tag, referrer_klass, depth, +root_kind)` — when `admitObject` encounters an already-tagged object +and the new path has a non-zero parent_tag, call `improveChain` to +replace the shallow root-attached entry with the deeper chain-attached +entry. Only improves if the new depth is greater than the existing +depth (never degrades a deeper chain to a shallower one). + +## Evidence + +On-pod logs show `buildCanaryChainEvent false: never pruned +(candidate=0 parent_tag=0 frontier_tag=0)` — the canary representative +was never reached by BFS. The 8 discovered instances were admitted as +roots (parent_tag=0), giving depth=1 chains. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-already-admitted-blocks-unreachable.md b/.investigations/missing-refchains-on-hotdog/nodes/find-already-admitted-blocks-unreachable.md new file mode 100644 index 0000000000..8a43c74c4a --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-already-admitted-blocks-unreachable.md @@ -0,0 +1,67 @@ +--- +id: find-already-admitted-blocks-unreachable +type: finding +status: confirmed +depends_on: [find-anchor-holder-eviction, find-already-admitted-blocks-deeper-chain, find-depth0-durable-root-upgrade-gap] +related: [find-anchor-live-feed-design, find-leak-tag-pool-implementation] +tags: [root-cause, PRODUCTION-BUG, referenceChains, heapReferenceCallback, improveChain, maybeUpgradeRootAttachedRootKind, dead-code, nesting, NEW-THIS-SESSION] +created: 20260914 +updated: 20260914 +--- + +# PRE-EXISTING PRODUCTION BUG (found while implementing B'): the already-admitted re-attribution blocks in heapReferenceCallback were unreachable + +## The bug + +Commit 57aec4895 ("Fix improveChain: call from heapReferenceCallback, +not admitObject", 2026-08-28) intended to move improveChain out of +admitObject's first-admission-only path into the already-tagged path of +heapReferenceCallback. The diff placed the call right after the +`switch (result)` — which is INSIDE the `if (*tag_ptr == 0)` first- +admission block. Every subsequently-added already-admitted behavior was +placed in the same nest: + +- improveChain (57aec4895) — dead for its stated purpose: a freshly + admitted entry carries exactly this edge's (parent_tag, depth), so the + `depth > entry.depth` check is a guaranteed no-op where it sat. +- reparentToDurableRoot (f4c73ba0f round) — same nest, same deadness. +- maybeUpgradeRootAttachedRootKind + the B' sweep push + (find-depth0-durable-root-upgrade-gap's fix) — same nest. + +For an object with *tag_ptr != 0 at callback entry, the whole block was +skipped: the callback ran the descend gates and returned. So from the +sweep/BFS/descend paths, NO already-admitted entry was ever improved, +re-parented, or root-kind-upgraded. (heapRootCallback has its own working +ALREADY_ADMITTED handling — only the root-walk path could upgrade +root kinds.) + +## Why the gtests never caught it + +The verifying tests drove the functions DIRECTLY via +ReferenceChainsTestAccessor (maybeUpgradeRootAttachedRootKindForTest, +insertFrontierEntry + improveChain), not the callback wiring — classic +"tested the function, not the call site". The B' end-to-end test failed +deterministically (pushed_total=0, entries never improving), which +surfaced it. + +## Impact on prior findings + +- find-anchor-holder-eviction's mechanism 2 (improveChain demotion) NEVER + fired on any pod round — the pod's holder was excluded purely by + mechanism 1 (admission order: born chain-attached via the BFS before + the sweep's lap reached its class) + the then-dead upgrade path + (mechanism 3). The eviction OBSERVATION stands; the demotion + CONTRIBUTOR was never live. +- The round-3-observed "depth-0 root upgrades (76)" came from + heapRootCallback's own path, not this callback. + +## The fix (implemented with B') + +Restructured into a real `else if (*tag_ptr > 0)` arm after the +first-admission block: parent_tag != 0 → improveChain / reparent (+ B' +demotion push); parent_tag == 0 → maybeUpgradeRootAttachedRootKind (+ B' +sweep push). Verified: 112 gtests green (no existing test depended on +the dead behavior), full gtestDebug suite green, ddprof-test +*ReferenceChain* family green (1 pre-existing failure on HEAD unrelated: +AggressiveLeakReferenceChainTest.shouldOpenSearchGate... fails 3/3 on +clean ff060cfac too — flagged separately, not caused by B'). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-anchor-holder-eviction.md b/.investigations/missing-refchains-on-hotdog/nodes/find-anchor-holder-eviction.md new file mode 100644 index 0000000000..cbfc1a4e14 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-anchor-holder-eviction.md @@ -0,0 +1,94 @@ +--- +id: find-anchor-holder-eviction +type: finding +status: open +depends_on: [ev-leaktag-onpod-round9, find-option-c-descend-walk-design, find-already-admitted-blocks-deeper-chain, find-depth0-durable-root-upgrade-gap] +related: [find-priority-queue-starves-bfs-crawl] +tags: [root-cause, referenceChains, anchor-tier, improveChain, parent-tag, eviction, NEW-THIS-SESSION] +created: 20260903 +updated: 20260914 +--- + +# ROOT CAUSE (round 9): static holders get evicted from the anchor tier — parent_tag==0 is unidirectional and improveChain actively destroys it + +## The observation chain + +Pod round 9 (ev-leaktag-onpod-round9): all 76 walked anchors are +machinery statics; the LEAK_BUFFER holder (a static `List` 3-4 +hops from the root static, textbook prong-2) NEVER enters the anchor +tier, while the sweep demonstrably enumerates ~1000 STATIC_FIELD edges +per chunk and laps all 33k classes. + +## The mechanism (code-confirmed; CORRECTED 20260914 — see find-already-admitted-blocks-unreachable) + +The anchor collector (collectStaticFieldAnchorsForRotation, +referenceChains.cpp:3161) requires `entry.parent_tag == 0 && root_kind in +{STATIC_FIELD, JNI_GLOBAL}`. Three code paths interact to make a +dual-reachable holder lose that shape PERMANENTLY. MECHANISM CORRECTION: +mechanism 2 (improveChain demotion) NEVER FIRED from heapReferenceCallback +on any pod round — the call sat inside the `*tag_ptr == 0` first-admission +block since 57aec4895, dead for already-admitted entries. The pod's holder +was excluded purely by mechanism 1 (admission order) + the then-dead +upgrade path. The eviction OBSERVATION and outcome stand unchanged. + +1. **Admission order**: if the holder's FIRST admission comes via a + non-root edge (e.g. Thread → target → lambda → captured list — the + classic self-registering-collection shape; Thread objects ARE + admitted+expanded in the frontier, proven by the on-pod chains ending + in ForkJoinWorkerThread), the entry is chain-attached (parent!=0, + root_kind=0) from birth. The sweep's later static edge hits the + ALREADY_ADMITTED path. +2. **improveChain (referenceChains.cpp:2083)**: a root-attached (depth-0) + entry reached via any NEW deeper path is REPLACED with the deeper + chain-attached entry — find-already-admitted-blocks-deeper-chain's + fix, which had no anchor-tier awareness. A static holder that is also + a task-closure capture gets demoted the first time a walk reaches it + through the thread path. +3. **Re-rooting is refused (referenceChains.cpp:2376)**: + maybeUpgradeRootAttachedRootKind returns false for parent_tag != 0 by + design ("known, documented limitation" — reconstructChain() can't + walk a re-rooted chain). So once chain-attached, the entry can NEVER + regain anchor eligibility, no matter how many laps re-prove the + static edge. The eviction is permanent for the search's lifetime. + +Net: the anchor tier only ever holds statics whose values are reachable +by NO other path — i.e., leaf-like machinery constants (exactly the 76 +observed: charsets, jnr enums, Method singletons). Any holder richly +referenced from the running graph (the interesting leaks by definition) +is systematically EXCLUDED. This also explains round 7's "exactly 4 +anchors" and round 8's "walk magnitudes vary but interception zero". + +## Why interception depends on it + +The tagged 78MB chunks are only enumerated by walking DOWN from the +holder (anchor walk enumerates holder → elementData → chunk → leak-tag +interception in one bounded STW). The BFS backlog (pendingExpand +190k+) can't reach them in practice, and thread walks are +ThreadLocalMap-gated. + +## Fix options (re-evaluated after standards survey, B' recommended) + +- **A. Sticky durable roots**: improveChain must not replace an entry + whose current attribution is root-attached STATIC_FIELD. **REFUTED + 20260914** by find-attribution-standards-survey: freezes one guess where + every standard system re-derives or queries; still does not cover the + admission-order case. Do not implement. +- **B. Static-attachment registry**: when the sweep's root-like + STATIC_FIELD edge hits an already-admitted chain-attached entry + (parent!=0), record the tag in a small bounded set; the anchor collector + selects from root-attached entries PLUS this registry. Covers both + orders. Defensible (lazy per-lap re-derivation) but keeps selection + coupled to the mutated table + preserves the full-frontier scan. + Minimal-diff fallback only. +- **B'. Live feed (RECOMMENDED)** — see find-anchor-live-feed-design: + feed the anchor FIFO directly from the sweep's static-edge enumeration + (at-risk filter: frontier-present, not root-attached-static), drain in + walkStaticFieldAnchors. Anchor selection stops reading + parent_tag/root_kind entirely → eviction structurally impossible. + +## Open verification + +Which order actually evicted the pod's holder is NOT yet proven by logs +(needs a TEMP counter/log on the sweep's static-edge-onto-chain-attached +path — the eviction symptom regardless of order). The fix (B especially) +works for both, so the distinction is not load-bearing for the fix. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-anchor-live-feed-design.md b/.investigations/missing-refchains-on-hotdog/nodes/find-anchor-live-feed-design.md new file mode 100644 index 0000000000..9abaea01b0 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-anchor-live-feed-design.md @@ -0,0 +1,113 @@ +--- +id: find-anchor-live-feed-design +type: finding +status: implemented +depends_on: [find-anchor-holder-eviction, find-attribution-standards-survey, find-option-c-descend-walk-design] +related: [find-isqueuedforrotation-quad-scan, find-priority-queue-starves-bfs-crawl] +tags: [fix, design, referenceChains, anchor-tier, live-feed, fifo, eviction-proof, NEW-THIS-SESSION] +created: 20260914 +updated: 20260914 +--- + +# Fix B' (recommended, NOT yet picked/implemented): feed the anchor walk-set from live enumeration, not from the mutated attribution table + +Born from the standards survey (find-attribution-standards-survey): JFR +enumerates its root set fresh at operation time, never cached; theory says +per-object attribution is a heuristic that must not gate selection. + +## Design + +- **Push site**: heapReferenceCallback's root-like branch (the static + sweep's class→field edge arrives with negative class referrer tag, + referenceChains.cpp ~:2089 comment). When the edge's TARGET tag is in the + frontier but its entry is NOT root-attached STATIC (the at-risk + population — dual-reachable holders like LEAK_BUFFER; machinery statics + stay on the existing collector path), push the target tag into a bounded + preallocated FIFO (dedup via PriorityExpandSet pattern: fixed + open-addressing, no allocation after construction). +- **Drain site**: walkStaticFieldAnchors drains the FIFO instead of + collectStaticFieldAnchorsForRotation's wrapping-cursor full-table scan + (190k+ entries per selection on pod). +- The anchor tier no longer reads parent_tag/root_kind AT ALL → the + find-anchor-holder-eviction mechanism (admission-order birth + + improveChain demotion + refused re-root) becomes structurally impossible. + Both eviction orders covered by construction. + +## Properties (argued, partially unverified) + +- Fair by construction: FIFO order = sweep-cursor order; no priority + starvation class. Bounded memory, O(1) push in the callback (no malloc, + no lock — engine discipline already serializes drivers). +- Push rate ≪ drain rate REQUIRES the at-risk-only filter to actually + shrink the population: INFERRED, not measured — the planned TEMP counter + on the static-edge-onto-chain-attached path (already in + find-anchor-holder-eviction's open verification) must confirm before + sizing the FIFO. Without the filter, pushes (~1000+/chunk enumerated + statics) would overwrite entries before the walk drains them. +- Perf win vs current: O(k) FIFO drain replaces the collector's + full-frontier scan per pass (aligns with the project's O(N) linear-scan + cutoff rule). +- **Chain correctness unaffected**: the holder's own frontier entry may + remain chain-attached (its thread-path chain is a complete chain to a GC + root; reparentToDurableRoot handles equal-depth durable re-parenting). + Interception only needs the walk to ENUMERATE the tagged chunks — + independent of the holder's own entry shape. +- **Scope**: JNI_GLOBAL anchor eligibility stays on the existing table + path (that cohort is machinery anyway); a root-callback feed can be added + later if dual-reachable JNI-global holders ever appear. + +## Status + +IMPLEMENTED 20260914 (design/review/implement/review loop, uncommitted). +Final shape: +- Members: _static_anchor_fifo (deque) + second PriorityExpandSet + instance (cap 1024) + STATIC_ANCHOR_FIFO_DRAIN=16 + cumulative + _static_anchor_fifo_pushed counter. pushAtRiskStaticAnchor() shared by + both push sites (cap-drop + dedupe). +- TWO push sites (both required — the "next lap re-pushes" assumption + was FALSIFIED: the sweep gate re-laps only while the class count is in + flux, so a post-lap demotion in a stable-class JVM never sees another + static edge): (1) DEMOTION TIME — in the improveChain success path when + the replaced entry was root-attached durable (rootKindDurability >= 2); + (2) SWEEP TIME — the STATIC_FIELD edge onto a chain-attached entry on + the failed maybeUpgradeRootAttachedRootKind. +- Both sites were UNREACHABLE until find-already-admitted-blocks- + unreachable's nesting fix (same session) — the enabling fix. +- Drain/requeue: runPass drains FIFO FIRST (16/pass) ahead of the + collector, one GOTW batch for all; walkStaticFieldAnchors gained an + `unwalked` out-vector (resolved-but-unwalked tags at the truncation + break — a consumed index is useless because GOTW drops dead tags in + its own order); runPass requeues the unwalked ∩ drained-fifo tags to + the deque front (lookup-filtered, <=16x20 scan). No re-push on + successful drain (next lap re-pushes; the sweep-push site covers it + when laps recur). +- Review-corrected: the collector STAYS (root-attached statics incl. + LEAK_BUFFER's unmodifiableList wrapper + JNI_GLOBALs are its disjoint + cohort — the node's earlier "O(k) drain replaces the 190k scan" claim + was WRONG, scan cost unchanged). +- Verification plumbing: per-pass rotation_candidates TEST_LOG extended + with static_anchor_fifo_size/drained/pushed_total (TEMP — remove after + round 10). +- Tests (referenceChains_ut, all green): DemotionPushFiresWhenImproveChain- + EvictsRootAttachedStatic (real expandFrontier batch walk, includes + no-double-push on re-walk), SweepPushFiresOnStaticEdgeOntoChainAttached- + Holder (full runPass: sweep delivers, rotation drains same pass, entry + never re-attributed), AtRiskAnchorFifoDrainAndWalkIntercept (drain+walk + intercepts leak 3 hops below a chain-attached holder the collector + demonstrably cannot select — negative control), TruncatedAnchorWalk- + RequeuesUnwalkedFifoTags (budget-truncated walk reports exactly the + un-walked tag; requeue restores FIFO order + set consistency). + A full-E2E runPass test was ABANDONED: the mock's root-phase walk order + + stale-expanded rotation queueing (isQueuedForRotation floods on tiny + graphs) make the eviction shape non-deterministic there — the + deterministic drives + the slow-suite scenario family are the coverage. +- Verified: 112 gtests green, full :ddprof-lib:gtestDebug green, ddprof-test + *ReferenceChain* family green EXCEPT a pre-existing failure on clean + HEAD (AggressiveLeakReferenceChainTest.shouldOpenSearchGateOnAggressive- + HeapWideGrowthWithNoLeakCandidate, fails 3/3 with changes stashed — + unrelated to B', flagged to user). spotlessApply clean. + +NEXT: user review → commit → deploy → round 10: watch +`static_anchor_fifo_pushed_total` (sizes the at-risk population — the +push-rate caveat was inferred, now measurable) and `leak-tag intercepted` +→ the first leak chunk's chain (static_field → ... → byte[]). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-anchor-tail-starvation.md b/.investigations/missing-refchains-on-hotdog/nodes/find-anchor-tail-starvation.md new file mode 100644 index 0000000000..f95dc0a74a --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-anchor-tail-starvation.md @@ -0,0 +1,81 @@ +--- +id: find-anchor-tail-starvation +type: root-cause +status: confirmed +depends_on: [ev-leaktag-onpod-round13-results] +related: [find-tier1-tail-starvation] +tags: [root-cause, referenceChains, anchor-tier, static-field, round-13, NEW-THIS-SESSION] +created: 20260915 +updated: 20260916 +--- + +# Find: anchor-selection tail starvation (root cause of zero ReferenceChain events) + +## What was found (round 13, probe-verified) + +The LEAK_BUFFER wrapper is admitted root-attached STATIC_FIELD correctly, +but the static-anchor collector can never reach it: + +- Anchor index ≈ 28k entries (the sweep admits ~0.83 statics per loaded + class; 34.3k classes on this JVM, nearly all non-bootstrap — app-priority + or per-holder-class cohorting do NOT shrink the set meaningfully here). +- Selection iterates in admission order (= sweep class-array order), 16 + collector + 16 FIFO per pass, avg 81 edges per anchor walk vs the 3741 + edge pass budget → the walk budget is saturated at ~32-46/pass. +- Search lifetime ≈ 190 passes (frontier 0→250k cap at 1.3-2k inserts/pass). + Coverage per search ≈ 4k of 28k (14%), always from index position 0 + (cursor reset by restartSearch — the index itself is cleared, so there is + nothing to resume from). +- The wrapper is admitted at sweep cursor 25301/34310 (75%) → index position + 12-21k → deterministically out of reach, every search, forever. + +Fix B+C (budget 4→16/32, leak_tag priority, O(index) iteration) fixed the +walk RATE but could not fix the ORDER: the wrapper's entry has leak_tag=0 +(the [B chunks carry the leak tags), so the leak-priority sort does not +promote it. + +## Why the leak-side machinery cannot rescue it + +The leak chunks are only reachable THROUGH the wrapper's subtree (static → +wrapper → ArrayList → elementData → chunks). No walk reaches them, so: +- zero `leak-tag intercepted` conversions, +- `_leak_parent_fanout` is seeded only from ordinary root-reachable [B + entries (seedLeakAccumulationForNewlyWatchedKlass scans existing frontier + entries) → fanout=1 noise; the wrapper/elementData never attributed, +- BFS expansion of the wrapper is equally tail-starved (pendingExpand FIFO + is 148k deep when the wrapper is admitted mid-search). + +Corollary bug found: restartSearch() does not clear +_candidate_discovered_tags (frontier tags) → stale tags resolve into +zeroed/new-search slots → pollWatchedTargets reconstructChain failures; +when the slot holds a live new-search entry this can emit a chain event for +the WRONG object (the likely origin of the earlier "noise [B instance" +event). Clear discovered tags on restart. + +## Fix options (presented to user, decision pending) + +A. Class-shape priority: exclude leaf-class anchors from the index + (String/Class/boxed/primitive arrays — fixed well-known list, resolvable + without interface walks) and/or prioritize collection-shaped holders. + Wrinkle: shape resolution cannot run inside heap callbacks (no JNI) — + needs a lazy per-class shape cache filled outside callbacks. +B. Cross-restart coverage rotation (persist a selection start fraction) + + budget tuning (measured: ~46/pass ceiling from 81-edge avg walk cost). + Slow convergence: 6-14 search lifetimes (hours) for full coverage. +C. One-shot unrestricted whole-heap FollowReferences at search start — + admits the whole graph including static subtrees, firing leak-tag + interception everywhere in one bounded STW; costs seconds of STW once + per search and must contend with the 250k frontier cap. +D. A+B combined. + +## Lessons + +- The "budget-4 lottery" framing (round 10-11) was wrong in an important + way: the selection ORDER is admission order, and admission order follows + the sweep's class-array order. Any fix that only changes RATE leaves the + holder's position unchanged. Rate × lifetime ≥ index size is the + invariant that must hold for coverage; measure all three before assuming + a rate fix is enough. +- Ground-truth probes (JNI reads of the actual app object) resolve + contradictory deduction chains in one redeploy; prefer them over + enumerating every possible admission shape. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-attribution-standards-survey.md b/.investigations/missing-refchains-on-hotdog/nodes/find-attribution-standards-survey.md new file mode 100644 index 0000000000..21a5c424b3 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-attribution-standards-survey.md @@ -0,0 +1,93 @@ +--- +id: find-attribution-standards-survey +type: finding +status: confirmed +depends_on: [find-anchor-holder-eviction] +related: [find-holistic-design-issues, find-jvmti-heap-walk-stw-vmop, find-already-admitted-blocks-deeper-chain] +tags: [survey, methodology, attribution, dominator-tree, dynamic-graph-algorithms, anytime-search, jfr, mat, leakcanary, NEW-THIS-SESSION] +created: 20260914 +updated: 20260914 +--- + +# Survey: how standard tools and theory answer "which reference chain explains retention" + +Motivated by find-anchor-holder-eviction's fix options (A sticky / B +registry): user asked to check standard solutions (JVM and non-JVM, +including bleeding-edge research) before picking. + +## Standard tools (source-verified) + +- **JFR OldObjectSample / LeakProfiler** (openjdk/jdk + `src/hotspot/share/jfr/leakprofiler/`, read at master 2026-09): keeps NO + persistent path attribution. At every emit, + `PathToGcRootsOperation::doit()` runs at a safepoint, enumerates the live + root set FRESH (`RootSetClosure` feeding `BFSClosure`), ONE BFS over the + live graph — heap-proportional EdgeQueue (5% of heap, min 32MB), + first-path-wins mark bits (`closure_impl` marks on first visit → one + path/object, BFS = shortest), DFS fallback when the queue fills, + `GranularTimer` deadline. The attribution lives only for the operation; + re-derived wholesale next emit. Nothing can go stale because nothing + survives. +- **Eclipse MAT** (help.eclipse.org, `IPathsFromGCRootsComputer`): opposite + tradeoff — stores the FULL reference graph of the snapshot, computes + shortest paths ON DEMAND at query time (user can exclude weak/soft at + query time, not baked in at walk time). Multiple paths remain available. +- **LeakCanary/shark** (square.github.io API docs): dominator tree + "computed with Lengauer-Tarjan, which needs the whole graph up front… + meant for tools running on a workstation, not for on-device analysis". + MAT/Chrome DevTools/JProfiler all use dominator tree + retained size for + attribution; MAT's "accumulation point" (object whose retained size dwarfs + its largest child) is the principled version of our holder/growth-gating + heuristics. + +Common thread: **no standard tool keeps a long-lived, mutable +single-attribution per object.** JFR: recompute-often, persist-nothing. +MAT/shark: persist-everything, compute-on-demand. Our FrontierTable is a +third design — persist one guess per object and keep overwriting it — and +the anchor-tier eviction is the direct consequence. + +## Theory (dynamic graph algorithms) + +- Even–Shiloach decremental SSSP (JACM 1981): O(mn) total update time, best + for three decades; near-linear improvements are heavy machinery. +- Fine-grained optimality of partially dynamic SSSP (arXiv:2407.09651, + 2024): conditional lower bounds — incremental/decremental exact SSSP + can't asymptotically beat recompute in general. Fully dynamic APSP + (arXiv:2306.02662; SODA 2023 hop-restricted) is further from practical. +- Dynamic dominator tree maintenance (Georgiadis–Italiano et al. + arXiv:1604.02711; ESA 2019 low-high orders): exists, but with O(mn)-flavored + conditional hardness even on DAGs — research-grade, deliberately avoided + by every practical tool. +- Anytime/contract search (AIJ 2008 "Anytime search in dynamic graphs"; + Anytime Contract Search; Deadline-Aware Search): budget-sliced graph + search under deadlines where the search frontier is DECOUPLED from answer + bookkeeping and answers are recomputed at query time, not accumulated as + persistent mutable guesses. + +## Derived design principles (used to re-evaluate fix options) + +1. Re-enumerate what is cheap to re-observe; maintain nothing you can + derive fresh (JFR root set; our sweep lap re-proves every static edge). +2. A per-object single attribution is at best a heuristic stand-in for a + dominator; it is known-unmaintainable incrementally, so it must NEVER + gate selection (who to walk) — only reporting (which chain to print). +3. Lazy labels + bounded recompute is the theory-sanctioned compromise + under mutation (matches our sweep-lap re-registration pattern). +4. Hop-bounded walks (DESCENT_HOPS=16) are their own cheaper problem + class (SODA 2023 hop-restricted dynamic paths). + +## Verdicts on the anchor-eviction fix options + +- **A (sticky durable roots): REFUTED** from all three directions — freezes + one guess harder, where every standard system re-derives or queries; + still does not cover admission-order eviction. Do not implement. +- **B (static-attachment registry): defensible** — lazy per-lap + re-derivation of the static-attachment fact, the sanctioned pattern; but + keeps walk-set selection coupled to the mutated table and preserves the + collector's full-frontier wrapping scan. +- **B' (live feed): selected design** — see + find-anchor-live-feed-design; the confluence of all surveyed precedents. + +Note (parked): the MAT accumulation-point pattern suggests a long-term +dominator-flavored retained-size ranking as the principled holder-selection +signal; out of scope for this fix. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-canary-continue-skips-discovered-instances.md b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-continue-skips-discovered-instances.md new file mode 100644 index 0000000000..953f6744d2 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-continue-skips-discovered-instances.md @@ -0,0 +1,52 @@ +--- +id: find-canary-continue-skips-discovered-instances +type: finding +status: fixed +depends_on: [find-representative-changes-lose-canary] +supersedes: [] +related: [find-representative-changes-lose-canary] +tags: [root-cause, fix, referenceChains, canary, discovered-instances, control-flow, NEW-THIS-SESSION] +created: 2026-08-27 +updated: 2026-08-27 +--- + +# Canary path `continue` skipped discovered-instances check + +## Observation + +After deploying the auto-mark fix (`7bd7255c3`), the live pod showed all +5 candidates with `buildCanaryChainEvent -> 0` and no chain events being +emitted. The [B candidate (klass_id=2) had a marker tag, so it entered +the canary path in `pollWatchedTargets()`. The canary path ended with +`continue`, skipping the discovered-instances check that was added +below it. + +## Root cause + +The discovered-instances chain-building code was placed *after* the +canary `if (tag <= MARKER_TAG_BASE) { ... continue; }` block. Since the +[B candidate has a marker tag, it took the canary path and hit +`continue`, never reaching the discovered-instances check. The +auto-marking in the callback was correctly recording discovered +instances, but `pollWatchedTargets` never checked them for canary +candidates. + +This was a bug in the fix itself (`7bd7255c3`), not a pre-existing +defect. + +## Fix (COMMITTED 5d06d7328) + +Removed the `continue` from the canary path so it falls through to the +discovered-instances check. Both canary and non-canary paths now try +discovered instances when no chain is cached for the class. The +discovered-instances check is guarded by `no_chain_cached` so it only +runs when the canary path didn't already produce a chain. + +## Lesson + +When adding a new code path that must run for *all* candidates, verify +it is not placed after an existing `continue`/`break` that would skip +it. The canary path's `continue` was correct *before* the +discovered-instances feature existed — adding the new check below +without removing the `continue` silently disabled it for all canary +candidates. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-canary-fixes-e-f.md b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-fixes-e-f.md new file mode 100644 index 0000000000..a50606c8c5 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-fixes-e-f.md @@ -0,0 +1,90 @@ +--- +id: find-canary-fixes-e-f +type: finding +status: confirmed +depends_on: [find-cpu-pain-budget-starves-canary-passes, find-threadloop-presleep-blocks-back-to-back] +supersedes: [] +related: [] +tags: [fix, referenceChains, pain-budget, threadLoop, canary, NEW-THIS-SESSION] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# Fix E + Fix F: threadLoop() sleep dedup, canary-aware pain-budget escalation + +## Reasoning chain + +User was given 2-3 alternatives for each of the two findings from this +session (`find-cpu-pain-budget-starves-canary-passes`, +`find-threadloop-presleep-blocks-back-to-back`) and chose: threadLoop +alternative A (delete the leftover unconditional sleep), pain-budget +alternative B (dynamic escalation while a canary is active, mirroring the +existing urgency-driven `_budget *= 4` ramp already in `threadLoop()`). + +**Fix E (threadLoop dedup).** Deleted the unconditional +`if (cadence_ns > 0) { OS::sleep(cadence_ns); }` block at the old +`referenceChains.cpp:736-738`, confirmed by `git log -L` to be a leftover +from before commit `1f86edf3f` ("move sleep after shouldRunPass") added the +correctly-guarded sleep later in the same loop iteration +(`if (!should_run && cadence_ns > 0) { OS::sleep(cadence_ns); ... }`). That +commit never removed the original, leaving every iteration paying an extra, +unconditional `cadence_ns` sleep regardless of `should_run` - including +canary-bypass-eligible iterations the surrounding comment explicitly says +should skip it. No design tradeoff; the deleted block's own `!_running` +early-exit is already duplicated by the surviving guarded sleep's, so +nothing is lost. + +**Fix F (canary-aware pain-budget escalation).** `shouldRunPass()` +(RUNNING branch) now computes `canary_active` *before* the +`_cpu_pain_budget.canStartNow()` check (previously computed only later, for +the bypass branch) and calls the new `PainBudget::setRefillRate()` to +temporarily scale `_cpu_pain_budget`'s refill rate by +`CANARY_PAIN_BUDGET_REFILL_MULTIPLIER = 4.0` (same factor already trusted +for `_budget`'s own urgency ramp), capped at `1.0` (100%/wall-clock), while +`canary_active` holds. Reverts to the configured base rate +(`_pain_budget_refill_rate`) automatically the instant `canary_active` goes +false (recomputed every call - no separate revert path needed). Both call +sites (escalation and the later bypass-return branch) now share the same +`canary_active` snapshot instead of recomputing `popcount` twice. + +New helper: `PainBudget::setRefillRate(double, u64 now_ns)` +(`painBudget.h`) - drains at the *old* rate up to `now_ns` first (so the +rate change only affects time elapsed after the call), then swaps +`_refill_rate`. Necessary because assigning a freshly-constructed +`PainBudget(rate)` (the pattern `start()`/`resetSearchStateForTest()` use) +would reset `_balance_ms` to 0, silently forgiving any already-accumulated +debt every time canary state flips - not the intended semantics for a +live, per-call toggle. + +## Changes made + +- `ddprof-lib/src/main/cpp/painBudget.h`: added `setRefillRate()`. +- `ddprof-lib/src/main/cpp/referenceChains.h`: added + `CANARY_PAIN_BUDGET_REFILL_MULTIPLIER = 4.0` constant next to the other + reference-chains tuning constants. +- `ddprof-lib/src/main/cpp/referenceChains.cpp`: + - `threadLoop()`: deleted the leftover unconditional sleep block (old + `:728-741`, comment + sleep + running-check). + - `shouldRunPass()`: moved `canary_active` computation ahead of the + `_cpu_pain_budget` check, added the `setRefillRate()` call, reused + `canary_active` in the bypass-return branch instead of recomputing. + +## Verification + +- `./gradlew :ddprof-lib:compileDebug -Pskip-gtest` - compiles cleanly. +- `./gradlew :ddprof-lib:gtestDebug` (full suite) - BUILD SUCCESSFUL, 188 + actionable tasks, no failures reported. + +## What this rules out + +Nothing new ruled out - this closes the two findings above rather than +refuting an alternative hypothesis. + +## Not yet done + +- **Committed and pushed** as `01047a6aa` on `jb/reference-chains`, together + with Fix D (`find-abandon-event-queue-fix`). Not yet deployed/re-verified + on the hotdog pod. +- No dedicated regression test added for either fix (e.g. a test asserting + `_cpu_pain_budget`'s effective rate is 4x while `canary_active` and back + to base once resolved) - not requested. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-canary-found-criterion-unmigrated.md b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-found-criterion-unmigrated.md new file mode 100644 index 0000000000..fa2b15a9f6 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-found-criterion-unmigrated.md @@ -0,0 +1,78 @@ +--- +id: find-canary-found-criterion-unmigrated +type: root-cause +status: confirmed-fixed-pending-pod-verification +depends_on: [find-round16-endgoal-verification, ev-leaktag-onpod-round16-results] +related: [find-representative-changes-lose-canary, find-canary-continue-skips-discovered-instances, find-refchains-log-flood-configurable] +tags: [root-cause, canary, leak-tag, found-criterion, design-migration, fix, round-19, NEW-THIS-SESSION] +created: 20260916 +updated: 20260916 +--- + +# ROOT CAUSE: canary chase structurally unresolvable — the marker→leak-tag migration never migrated the found criterion (round 19, FIXED 788d7b2a7) + +Evidence chain (all pod 289f8, JVM pid_28570, level-1/2 runtime windows +via the rcDebugLevel knob — no deploys needed): "canary search, 0/1 +candidates found" on EVERY JVM since the leak-tag design change, while +`drainPendingChainEvents re-emitted` grew (2→34) — events flowing, chase +never resolving. The representative `candidate[0] tag= +needRefresh=1` on every tick, with `buildChainEvent false: target_tag +not in frontier` for its leak tag. + +## The mechanism (three unmigrated pieces) + +1. **Found bits**: `_candidate_found_bits` (the chase's ONLY exit + criterion, popcount in shouldRunPass) is set ONLY in + heapReferenceCallback()'s MARKER-tag block — and markers are never + set since the slot registration changed to `_candidate_tags[slot]=0; + // no marker tags — using leak tags now`. Under leak tags NOTHING + sets found bits → the canary can never report 1/1 → every search ends + via frontier-cap/no-progress instead of "all candidates found". +2. **Rep path check**: `_resolved_chains.find(rep_leak_tag)` — the cache + is keyed by FRONTIER tags (disc_tag); the rep's leak tag is never a + key → needRefresh=1 forever. +3. **Rep path build**: `buildChainEvent(rep_leak_tag)` → + `frontier->lookup(leak_tag)` always misses (interceptions insert with + a FRESH frontier tag; entry.leak_tag rides inside the entry) → a + doomed rebuild every poll (~4/s frontier lookups + level-2 noise). + +Why this survived rounds 15-18: events flow via the DISCOVERED-instance +path (recordDiscoveredInstance → buildDiscoveredInstanceChains → +auto-mark with target_tag = the leak tag), which is orthogonal to the +chase exit. The 0/1 state was chase lifecycle noise (search churn, +backoff CPU), not an event blocker — which is why the end goal verified +while the canary stayed stuck. + +## The fix (788d7b2a7) + +In `buildDiscoveredInstanceChains`, where leak-tag chains are built: a +chain with `target_tag >= LEAK_TAG_BASE` for a candidate slot marks the +slot found and records its canary link (frontier tag = disc_tag, parent +0, depth, referrer klass) — the leak-tag-world "canary found" = a walk +reached the leaked population AND the correlation carried. Noise chains +(targets below LEAK_TAG_BASE) do not mark found. The rep path resolves +by the slot's chain key and stops the doomed rebuilds. Regression test +`LeakTagChainMarksCanaryFound` (leak slot → found + link; noise slot → +not found); 122/122. + +## Design decision recorded + +The criterion is per-SLOT-any-instance (any leak-tag-target chain of the +candidate klass), not the representative's own tag. Rationale: the +per-instance guarantee moved INTO the leak-tag correlation (events +carry target_tag = leak tags); the rep's own chain still emits via the +discovered path whenever its instance is intercepted (its leak tag is +the event target). Exit-on-first-leak-chain ends the search promptly +instead of chasing one specific chunk through sporadic subtree coverage +(5-min window on 289f8: zero NEW interceptions, wrapper not walked, +34 cached chains from earlier in the search — coverage is sporadic at +batch scale, a separate upstreaming-round topic). + +## What to verify on the next deploy + +`canary search, 1/1 candidates found` (level 1) → the chase exits → the +search completes → `releaseSearchTags` → a clean restart cycle +(`canary found: klass_id=%u slot=%d leak chain target_tag=...` at +level 1 is the new marker). The rep path's needRefresh should clear +(no more `buildChainEvent false: target_tag= not in frontier` +retry spam at level 2). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-canary-lane-backoff-design.md b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-lane-backoff-design.md new file mode 100644 index 0000000000..5620620430 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-lane-backoff-design.md @@ -0,0 +1,70 @@ +--- +id: find-canary-lane-backoff-design +type: finding +status: implemented +depends_on: [find-canary-search-forces-max-cadence, ev-leaktag-onpod-round4] +related: [find-getobjectswithtags-quadratic-bottleneck, q-safepoint-budget-model] +tags: [fix, design, canary, cpu-burn, work-scaled, NEW-THIS-SESSION] +created: 20260901 +updated: 20260901 +--- + +# Canary-lane work-scaled backoff (user point 3, option A) + +## The defect (user-observed: ~1 core on hotdog) + +canary_active bypassed every cadence check and threadLoop() skips its +sleep whenever a pass will run: an un-findable candidate held the +chase open back-to-back for 32 min at ~88 passes/min (round 3) - a +full core. The doc's "measured cost is tiny (<20ms per 60s)" claim +only ever held for findable candidates. The 15x-covering/100x-emergency +pain-budget refill multipliers existed to feed exactly this unbounded +mode. + +## The law (first fixed-cap, reworked work-scaled after local evidence) + +Inter-pass spacing = _canary_backoff_mult x EMA(pass wall duration, +0.8/0.2, ms). Multiplier: 1 (gate OFF - natural pass rate) -> doubles +per pass with no candidate progress -> caps at CANARY_BACKOFF_MULT_MAX +(16); candidate progress (found bit or new candidate admitted) resets +to 1; search start/restart resets to 1; the OOM urgency ramp +(_oom_ramp_active, set by threadLoop) overrides the gate; the +gc-finish-epoch trigger deliberately does NOT bypass it (GC-heavy +workloads bump the epoch every wake). + +Why work-scaled: a fixed cap only binds when it exceeds the pass's +own duration - the pod's passes ran 0.7-4s, so a 1s cap would have +changed nothing (work-bound loop), while the same 1s cap starved +ReferenceChainTrackingTest's deep ~200-pass chase outright (held-off +wakes outpaced the test window; pass wall there ~20-30ms). Scaled +against measured pass cost the burn bound is structural: <= ~1/16 of +a core on pass work at the cap, whatever the work is; a deep-but-cheap +chase keeps density (200 passes at mult 16 x ~25ms = ~80s - passes). + +The pain-budget refill is now a flat 100x while a chase is open - a +double-throttle guard only (the backoff owns the rate); the covering +(15x)/emergency (100x) split is deleted. + +## Verification + +546 gtests green (CanaryLaneBacksOffWithoutProgressAndResetsOnProgress: +deterministic seeded-EMA arithmetic - doubling, cap-hold, progress +reset, urgency bypass, hold-off/re-admit). Slow suite: ToGcRoot green +at load ~6 (backoff engaged mult 2..16, ema 20-30ms, chase completed); +leak-correlation unaffected (its chase resolves early). ToGcRoot/ +UnboundedCache remain load-sensitive (separate family - the executor +JVM's in-process passes hit ~4.4s under the 5s pausetarget at load +25+, so the work-scaled spacing inflates with them; observed failing +only at load 25-55). + +Doc: ReferenceChains-SignalsExplained.md sections 4/8/11 rewritten +(no-GC-wake + epoch-bypass caveat + cadence dual role; work-scaled +backoff with pod evidence; corrected tiny-cost claim). + +## Pod expectation (round 5+) + +`held off by canary backoff mult=N ema_ms=M` TEST_LOG lines; stuck-chase +burn decays to <= ~6% of a core on pass work; a chase that makes +progress keeps its burst. If the pod's pass-wall decomposition (static +sweep lap share - still unmeasured) shows ema dominated by the sweep, +consider sweep-lap pacing separately. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-canary-search-cannot-terminate.md b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-search-cannot-terminate.md new file mode 100644 index 0000000000..bd8755342f --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-search-cannot-terminate.md @@ -0,0 +1,71 @@ +--- +id: find-canary-search-cannot-terminate +type: finding +status: confirmed +depends_on: [ev-source-poll-vs-callback, ev-livelock-pod-logs] +supersedes: [] +related: [find-marker-tag-slot-index-mismatch, find-one-shot-pretag-gate, find-canary-stuck-abandon-detector, ev-fixes-compile-and-gtest-pass] +tags: [livelock, search-state, completion-gating, urgent, partially-fixed] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Third element (PARTIALLY ADDRESSED): with 0/N candidates found the canary search could neither complete nor abandon + +## Reasoning chain + +Once `find-marker-tag-slot-index-mismatch` pins `_candidate_found_bits` at +0, all three exits from the canary search are closed: + +1. **Completion** requires + `popcount(_candidate_found_bits) == _candidate_count` + (`referenceChains.cpp:3101-3103`). At 0/3 that is unreachable. +2. **TTL / no-progress abandon** is gated on `!isUrgent()` + (`referenceChains.cpp:3090-3091`), and the search on this pod IS urgent + (heap floor rising, `FLOOR_RISING … floor_rising=1`). So the abandon + branch is suppressed by design — the very safety valve that would have + reset state is disabled exactly when the leak signal is strongest. +3. **Cadence back-off** does not apply: `shouldRunPass()` short-circuits to + `true` for as long as any candidate bit is unset + (`referenceChains.cpp:914-922`), so a pass runs on every wake. + +Net effect is a livelock: passes keep running (`passesRun=60`, +`iteration=61` at read time), the frontier keeps growing +(`frontierSize 10027 → 10211`), STW pause budget keeps being spent, and +nothing is ever emitted. 143 identical iterations were observed over 25 +minutes. + +This is also why `datadog.ReferenceChainAbandoned` is 0 alongside +`datadog.ReferenceChain` — the search never abandons either. + +## Evidence +- `evidence/ev-source-poll-vs-callback.md` (`:3090-3108`, `:914-922`) +- `evidence/ev-livelock-pod-logs.md` + +## What this rules out +- Any expectation that leaving the pod running longer would produce + events. It is a stable fixed point, not a slow convergence. +- The theory that a bad candidate would be aged out by the TTL/abandon + path — that path is explicitly suppressed while urgent. + +## Status update (this session) + +Fixing `find-marker-tag-slot-index-mismatch` (Fix A) and +`find-one-shot-pretag-gate` (Fix B) removes the *original* trigger (a +candidate that can never be marked found because of the index bug, or +because it was never admitted). But Fix B's never-retire design means a +different, narrower version of this same livelock is still reachable: a +candidate that legitimately drops out of `selectLeakCandidates()`'s later +polls (e.g. it stopped growing) stays latched in its slot and can never be +found, so `popcount(found_bits) == candidate_count` still can't be +satisfied for that search. Point 2 (TTL/abandon suppressed while urgent) +is unchanged and is exactly the case that matters most, since urgency is +what makes the search run in the first place. + +`find-canary-stuck-abandon-detector` (Fix C, new node) closes this +remaining gap with a completion-agnostic stuck detector that is +deliberately NOT suppressed by `isUrgent()`. This finding is not fully +superseded — its point 3 (`shouldRunPass()` always true while any bit is +unset) and the general "no exit while urgent" shape of the problem are +still the reason Fix C exists — but the specific livelock instance +observed on the hotdog pod is expected to be fixed by A+B+C together. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-canary-search-forces-max-cadence.md b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-search-forces-max-cadence.md new file mode 100644 index 0000000000..1fe8bf171b --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-search-forces-max-cadence.md @@ -0,0 +1,55 @@ +--- +id: find-canary-search-forces-max-cadence +type: finding +status: confirmed +depends_on: [find-cpu-pain-budget-blocks-bfs] +related: [find-representative-changes-lose-canary, find-leak-tag-pool-implementation, q-allocation-site-selection] +tags: [root-cause, cpu-burn, canary, shouldRunPass, pod-logs, NEW-THIS-SESSION] +created: 20260901 +updated: 20260901 +--- + +# An unfindable candidate canary forces a pass at EVERY iteration - the 30%+ CPU burn + +Observed on hotdog (JVM 75258, build 663784137): the user flagged 30%+ +CPU burn. The threadLoop logs show `shouldRunPass -> true (canary +search, 0/1 candidates found)` on EVERY logged decision, `blocked` (cpu +pain budget) never firing, `passesRun=2803` in 32 min = ~88 passes/min +(the ~12/min figure in round-2 evidence was pre-proportional-build). + +Mechanism: candidate[0]'s marker/canary is never found by the walks +(`needRefresh=1` persistent - the same never-found-marker state as the +seams test, but here in production). While any candidate's canary is +unfound, `shouldRunPass` returns true every iteration (10ms cadence), +bypassing the cadence throttle AND the pain budget. Each pass costs +~5 GetObjectsWithTags scans (18-21ms each, O(tag-map) under the tag-map +lock) + a ~15ms STW walk VM op -> roughly 100-130ms of tagged-CPU+STW +work per pass, ~90 passes/min ~= 15%+ of one core in scanning alone, +before GC/pause interference - matching the user's 30%+ observation +shape. + +Two ways it ends: +- The candidate is retired (e.g. by (klass,tid)-qualified selection - + the [B candidate looks like machinery churn: tagged instances are + 24-16KB kafka-ish byte[]s with stable, not rising, per-site retention) + -> no candidate -> no canary search -> cadence returns to throttle. +- The marker becomes findable (needs the holders of tagged instances to + be walked - the disjoint-set lottery; without a bridge this never + resolves on a large heap). + +Implication for planning: option C (per-(klass,tid) candidate +qualification, no walks needed) is not just signal quality - it is the +CPU-burn fix. It attacks the forced-cadence burn at zero walk cost. +B (full-graph intercept sweep) would have ADDED ~10s STW per lap and is +retracted (find-jvmti-heap-walk-stw-vmop). + +## Correction (post option-C implementation) + +On hotdog specifically, C does NOT end the canary search by retiring the +candidate: the pod runs a DELIBERATE simulated-memory-leak thread whose +per-tid signal is real and rising (ev-tid-clustering evidence), so +per-(klass,tid) qualification SCOPES the tagging to the leak thread +instead. Burn relief there must come from interception becoming possible +(tagged objects = leak-site instances under holders the crawl rotates +to), closing the canary the normal way. The "retire the false candidate" +path remains C's effect for genuine machinery-churn false positives. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-canary-stuck-abandon-detector.md b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-stuck-abandon-detector.md new file mode 100644 index 0000000000..f732a36b0e --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-stuck-abandon-detector.md @@ -0,0 +1,117 @@ +--- +id: find-canary-stuck-abandon-detector +type: finding +status: confirmed +depends_on: [find-canary-search-cannot-terminate, find-one-shot-pretag-gate] +supersedes: [] +related: [find-marker-tag-slot-index-mismatch, q-implement-two-fixes, ev-fixes-compile-and-gtest-pass] +tags: [fix, livelock, search-state, urgent-abandon, canary, new-this-session] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Fix C (implemented, this session): canary-specific stuck/abandon detector not suppressed by isUrgent() + +## Reasoning chain + +`find-canary-search-cannot-terminate` identified that the ordinary +TTL/no-progress abandon path is gated on `!isUrgent()` +(`referenceChains.cpp:3090-3091`), so a search that is both urgent +(heap floor rising, near projected OOM) and livelocked has **no exit at +all** — it keeps running every scheduling cycle at urgency-boosted +(higher-budget, higher-cadence) settings, because nothing in the urgent +path ever asks "am I actually converging." + +Fix B (`find-one-shot-pretag-gate`) makes `_candidate_count` growing-only +and never retires a slot once a candidate is admitted. That is deliberate +(never-retire was the user-approved design), but it means a single +candidate that stops appearing in `selectLeakCandidates()`'s later results +(and so is never walked-to and never marked found) can permanently block +completion (`popcount(_candidate_found_bits) == _candidate_count`) for the +rest of that search's life — even after Fix A/B, not just before them. + +This finding's fix is a completion-agnostic stuck detector, independent of +`isUrgent()`, so a livelocked urgent search can still terminate and free +its STW/frontier budget instead of running forever. + +## Design + +Two counters added to `ReferenceChainTracker` (`referenceChains.h`): + +- `int _passes_since_last_candidate_progress` — passes since candidate + discovery last advanced. +- `int _last_candidate_progress_mark` — high-water mark of a monotonic + "progress" quantity. +- `static constexpr int CANARY_NO_PROGRESS_PASS_LIMIT = 30` (same value as + the existing `NO_PROGRESS_PASS_LIMIT`, but tracked separately and + independently gated). +- `SearchAbandonReason::CANARY_STUCK = 3` added to the existing + `NONE/FRONTIER_CAP/TTL` enum (`referenceChains.h`), kept in lockstep with + `flightRecorder.cpp`'s `kReasons` string table + (`{"none","frontier_cap","ttl","canary_stuck"}`) and its bounds check + (`< 3` -> `< 4`) — the existing code comment explicitly warns these two + must match index-for-index for JFR serialization. + +Progress mark, computed once per pass in `runPass()`: + +```cpp +int candidate_progress_mark = + _candidate_count + (int)__builtin_popcountll(_candidate_found_bits); +if (candidate_progress_mark > _last_candidate_progress_mark) { + _last_candidate_progress_mark = candidate_progress_mark; + _passes_since_last_candidate_progress = 0; +} else { + _passes_since_last_candidate_progress++; +} +``` + +This mark is provably monotonic **because** of Fix B: `_candidate_count` +only grows (never-retire) and `_candidate_found_bits` bits are only ever +set, never cleared, while `RUNNING`. That guarantee did not hold before +Fix B (a retire/reuse design would have made the mark oscillate), which is +why this detector was designed and landed together with Fix B rather than +independently. + +Abandon branch added to `runPass()`'s completion/abandon `if/else if` +chain, after the existing branches: + +```cpp +} else if (_candidate_count > 0 && + _passes_since_last_candidate_progress >= + CANARY_NO_PROGRESS_PASS_LIMIT) { + store(_abandon_reason, (u8)SearchAbandonReason::CANARY_STUCK); + storeRelease(_search_state, (u8)SearchState::ABANDONED); +} +``` + +Deliberately has no `!isUrgent()` guard — this is the whole point: it must +fire *especially* when urgent, since that's the only condition under which +the ordinary TTL path is disabled. + +Reset alongside `_candidate_count`/`_candidate_found_bits` at all three +existing lifecycle reset sites: `start()`, `resetSearchStateForTest()`, and +`runPass()`'s terminal-state tag-release cleanup block. + +## Why this was scoped in, not deferred + +The user's own framing, quoted back mid-session: "a search that's both +urgent and livelocked has no exit at all... since nothing in the urgent +path checks 'am I actually converging.'" I proposed this as a fix; the +user's instruction was explicit: "implement alongside; seems like a +'minor' change." Scope was user-approved, not self-initiated scope creep. + +## Evidence +- `ddprof-lib/src/main/cpp/referenceChains.cpp` — `runPass()` diff (Fix C) +- `ddprof-lib/src/main/cpp/referenceChains.h` — new fields/constants/enum + value +- `ddprof-lib/src/main/cpp/flightRecorder.cpp` — `kReasons` table update +- `ev-fixes-compile-and-gtest-pass` — build + gtest verification + +## What this does not cover +- Does not fix `find-canary-search-cannot-terminate` point 3 + (`shouldRunPass()` always true while any candidate bit is unset) as a + cadence/backoff concern — it only guarantees the search eventually + reaches `ABANDONED` and gets a fresh restart, not that intermediate + passes back off before that point. +- No on-pod re-verification yet; no multi-candidate regression test added + (open sub-question carried over from `q-implement-two-fixes`). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-canary-stuck-restart-wipes-frontier.md b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-stuck-restart-wipes-frontier.md new file mode 100644 index 0000000000..c45a7a0089 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-canary-stuck-restart-wipes-frontier.md @@ -0,0 +1,113 @@ +--- +id: find-canary-stuck-restart-wipes-frontier +type: finding +status: confirmed +depends_on: [find-canary-fixes-e-f, find-abandon-event-queue-fix] +supersedes: [] +related: [find-canary-stuck-abandon-detector, find-canary-search-cannot-terminate, ev-postfixEF-onpod-live-verification, q-canary-stuck-fix-alternatives] +tags: [root-cause, referenceChains, canary, restart, frontier, no-progress-limit, NEW-THIS-SESSION] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# CONFIRMED: CANARY_NO_PROGRESS_PASS_LIMIT + full frontier wipe on restart prevents ever reaching a distant-but-reachable candidate + +## Reasoning chain + +With Fix E/F confirmed live (starvation resolved — `runPass` went from 0 to +50 in a comparable ~40s window, see `ev-postfixEF-onpod-live-verification`), +candidates still resolve at 0/5 in every single sample. This rules out the +pain-budget/cadence gates as the (sole) explanation for zero resolution — +passes are now running continuously — and shifts the question to why a +continuously-running search still never reaches any candidate. + +Live trace shows the search hits `CANARY_STUCK` abandon roughly every ~20s, +with the frontier having grown to 12,776 and then 15,699 entries at the two +observed abandon points before `restartSearch()` resets it +(`_frontier->resetForRestart()` + `_next_tag = 1`, +`referenceChains.cpp:1054-1056`). `candidate[0]`'s `klass_id` stayed at `2` +(`[B`) across the entire window (2545 matching log lines) — ruling out +candidate-list churn as a contributor. + +The canary-specific stuck detector (`CANARY_NO_PROGRESS_PASS_LIMIT=30`, +hardcoded in `referenceChains.h`, see the design comment around +`:1988-1999`) fires on zero *candidate-bit* progress specifically, +independent of whether the general BFS frontier is still healthily +expanding elsewhere. Combined with `restartSearch()`'s unconditional full +wipe, this means: every search gets one shot at reaching the candidate +within roughly one cycle's worth of BFS depth (bounded by ~13-16k frontier +entries here), and if the candidate sits behind a reference chain longer +than that, no amount of wall-clock time helps — the search can never +accumulate progress across restarts. + +## Evidence + +- `ev-postfixEF-onpod-live-verification` — full 40s trace, `runPass done` + frontier-size progression, two `CANARY_STUCK` abandon events at trace + lines 143402 and 380564 in `/tmp/hotdog_trace2.log`. +- Code confirms the mechanism precisely: + `_passes_since_last_candidate_progress` (`referenceChains.cpp:3148-3155`) + increments on every pass where + `_candidate_count + popcount(_candidate_found_bits)` fails to increase — + once all 5 slots are admitted and none found, this ticks up every single + pass regardless of `_passes_since_last_progress` (whole-graph frontier + growth), by the design comment's own explicit intent + (`referenceChains.h:709-725`). At `CANARY_NO_PROGRESS_PASS_LIMIT=30` + (`referenceChains.h:1999`) this fires `CANARY_STUCK`, and + `restartSearch()` performs a destructive full reset + (`_frontier->resetForRestart()` + `_next_tag=1`, + `referenceChains.cpp:1054-1056`) with zero carryover. +- **User-confirmed ground truth**: the target is a synthetic, permanently + reachable leak (deliberately never released) — this eliminates "candidate + is genuinely unreachable from any sampled root" as a competing + explanation. The only remaining explanation for persistent 0/5 is that + the algorithm cannot accumulate enough continuous BFS depth/reach in one + ~20s/30-pass cycle to get to it, and loses all progress every cycle. + +## What this rules out + +- Candidate-list churn (a different `candidate[0]` offered on each restart) + — ruled out, `klass_id=2` stable across both cycles. +- The starvation/cadence gates (`find-cpu-pain-budget-starves-canary-passes`, + `find-threadloop-presleep-blocks-back-to-back`) as sole explanation for + zero resolution — ruled out, they are fixed and confirmed live, yet 0/5 + persists. +- **Candidate unreachable from any sampled root** — ruled out by user + confirmation (synthetic leak, deliberately retained, permanently + reachable). + +## Alternatives analysis + +See `q-canary-stuck-fix-alternatives` for the 4 proposed fixes (C/A/B/D), +research grounding (incremental BFS, G1 SATB marking, Luby/adaptive restart +theory), and the recommendation (lean: C, the merged-stuck-detector +condition). User chose **C+B together**; implemented this session — see +`q-canary-stuck-fix-alternatives` for the code changes. + +## Live-pod verification (post-deploy) + +Committed+pushed as `82fec4210`. User resynced the agent on +`prof-analyzer-hotdog-jb-c944876b9-f762h` (PID 62384); confirmed via +`strings` on the loaded `.so` that `CANARY_PAIN_BUDGET_REFILL_MULTIPLIER` +and the new `_canary_stuck_restart_count`/`canaryStuckPassLimit` symbols +are present. + +Two live traces (40s + 90s, 122 total `runPass done` samples, +`/tmp/hotdog_trace{3,4}.log`): **zero `CANARY_STUCK` abandons**, `searchState=0` +(RUNNING) and `abandonReason=0` (NONE) on every single pass. Frontier grew +continuously and monotonically across the whole combined window: 25,040 -> +26,626 -> 33,137 entries — no wipes, no restarts. This directly confirms +C+B fixed the destructive-restart bug: the search that previously died at +~12-16k entries every ~20s now keeps accumulating BFS reach uninterrupted +past 33k. + +Candidates are still `0/5 found` in both traces — expected at this stage, +since the search was never previously allowed to run this long +uninterrupted. Whether/when it resolves is now purely a function of actual +candidate depth vs. observation time, not the restart bug. Next step: a +longer-duration trace (few minutes) to check for eventual resolution. + +## Not yet done + +- Longer-duration live trace to confirm eventual candidate resolution now + that the restart-wipe bug is fixed. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-candidate-presence-suppresses-static-admission.md b/.investigations/missing-refchains-on-hotdog/nodes/find-candidate-presence-suppresses-static-admission.md new file mode 100644 index 0000000000..ab9a181bc3 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-candidate-presence-suppresses-static-admission.md @@ -0,0 +1,66 @@ +--- +id: find-candidate-presence-suppresses-static-admission +type: deadend +status: retracted +depends_on: [design-pod-in-a-jar-harness, meta-whackamole-analysis] +related: [find-round16-endgoal-verification, find-urgentoom-null-fn-mislabel] +tags: [deadend, retracted, harness-fixture-bug, capacity-contract, methodology, round-20, NEW-THIS-SESSION] +created: 20260916 +updated: 20260916 +--- + +# RETRACTED: "candidate presence suppresses static admission" — it was the harness builder's own dangling pointer + +The round-20 harness's first "catch" (with a candidate armed, the next +pass does not admit the wrapper static) is RETRACTED. Root cause, traced +to ground with test-side mock instrumentation (holder array built and +filled, holder-elem seeding never firing): + +The Bfs fixture's class-registration pattern captures +`&node_tags[node]` as the jclass identity +(`classes.push_back({(void*)&node_tags[classNode], ...})`). The +harness's topology builder then added 300 filler nodes, growing +`node_tags` past capacity — the vector reallocated and every captured +identity pointer DANGLED. `indexOfNode()` then returned -1 for those +classes in the static sweep: the holder array seeded no expandable +class, the STATIC_FIELD edges never replayed, the sweep lapped +admitting NOTHING, and the gate closed on the empty lap — all +silently, with the search state machine still reporting a clean +COMPLETED search. + +## Why this mattered more than a test bug + +The failure shape was maximally deceptive: no crash, no error, a +COMPLETED search, zero admissions. It survived ~6 bisect rounds +(looked like: candidate presence, budget starvation, restart-gate +state, check-placement artifacts — each partially true because the +dangling pointer made results order-dependent and non-monotone). The +capacity contract (reserve the full node count BEFORE any +`&node_tags[i]` capture) is now encoded as a comment at the builder + +the permanent TopologyCapacityContractStaticAdmits regression test. + +## Methodology notes (the meta-value of this deadend) + +1. The pod's round-18 "sporadic coverage" is NOT explained by this — + that remains an open upstreaming-round question (batch selection + determinism). The earlier pod-correlation hypothesis in this node's + first version is withdrawn. +2. Fixture capture-pointer contracts are a NEW class of test-seam gap + (same family as find-test-seam-aliasing / the reset() seam gap): + the mock's data structures have invariants the tests must maintain, + and violating them fails silently inside production code paths. + When a harness "catches" something, FIRST bisect against the + harness's own construction (the repro that failed with the full + topology but passed with the minimal one pointed AT the builder + all along). +3. The mock instrumentation approach (test-side printf in the mock + JVMTI/JNI slots, gated by an env var) was decisive — printfs in + TEST code are always available even in non-DEBUG builds. + +## Outcome + +With the capacity fix, ALL the harness invariants went live (see the +de15e521f commit): L1 leak chains, L2 canary resolution, L6 restart +hygiene (with the chains-persist-by-design contract), L4 log budget +(compile-aware), joining L3 natural completion and L8 dormancy. +128 tests, 127 pass, 1 skip-with-reason. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-candidate1-never-tagged.md b/.investigations/missing-refchains-on-hotdog/nodes/find-candidate1-never-tagged.md new file mode 100644 index 0000000000..aace8e08a8 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-candidate1-never-tagged.md @@ -0,0 +1,370 @@ +--- +id: find-candidate1-never-tagged +type: finding +status: confirmed +depends_on: [ev-postfix-static-field-onpod-live-verification, ev-kind-counts-constant-pool-dominates] +supersedes: [] +related: [find-static-field-sweep-cursor-fix] +tags: [referenceChains, candidate, canary, marker-tag, static-field, constant-pool, fix-implemented] +created: 2026-08-25 +updated: 2026-08-26 +--- + +# Open: candidate[1] (klass_id=283) never tagged/resolved despite sweep confirmed making real progress + +## Observation + +`candidate[1]` (klass_id=283, class `[Ljava/lang/Object;`, +`marker_tag=-4611686018427387907`, `slot=3`, `needRefresh=1`) shows +`buildCanaryChainEvent(slot=3) -> 0` (unresolved) on **every single +sample** checked, across two separate windows: a ~40s live tail and a +targeted 5-minute historical grep +(`kubectl logs ... --since=5m ... | grep -B1 -A3 "candidate\[1\] +klass_id=283$"`). + +Critically: `candidate[0]` (klass_id=10, `[B`) prints a `tag=` line on +its samples; a matching grep for a `tag=` line following +`candidate[1] klass_id=283` returns **zero results** across the same +5-minute window. This means candidate[1]'s target object has never even +been found/tagged by the search at all — not merely "found but chain not +yet built." + +This is despite `ev-postfix-static-field-onpod-live-verification` +confirming the static-field sweep is genuinely admitting real edges +every call (up to 737/chunk) and advancing its cursor correctly. + +## Not yet confirmed (working hypothesis only) + +One hypothesis worth checking against logs before asserting: the +app-classes-first partition in `admitStaticFieldRoots()` sorts by the +classloader of the **class being swept for static fields** (the holder +class), not by the referenced object's own class. `candidate[1]`'s own +class (`[Ljava/lang/Object;`) is very likely bootstrap-loaded (null +loader) itself, but that says nothing about which class *holds the +static field pointing to it* — if that holder is a JDK/bootstrap class, +it sweeps in the large tail-end of the partition and may simply not have +been reached yet in the current (or any completed) lap, since +`cycle_complete=1` has not yet been observed at all (see +`ev-postfix-static-field-onpod-live-verification`). + +Alternative hypotheses not yet ruled out: +- candidate[1]'s referencing object is reachable via the general + frontier/expandFrontier path, not the static-field-only path, and the + bottleneck is unrelated to this fix entirely. +- some other admission gate/slot bug specific to slot=3 (distinct from + the previously-fixed `find-marker-tag-slot-index-mismatch`). + +## Timing check supports the "not-yet-swept" hypothesis (not proof) + +JVM PID 92618 uptime at time of check: 23m42s (`ps -o etime`). Over an 8- +minute log window, `sweep_cursor` advanced from 13312 to 18944 (5632 +classes / 8 min ≈ 700 classes/min) out of `last_resolved_class_count=34053` +total loaded classes — i.e. the sweep is still inside its **first lap** +(~40-55% through), extrapolating to ~45-50 min for one full lap. This is +consistent with `cycle_complete=1` never having been observed +(`ev-postfix-static-field-onpod-live-verification`) and supports (but +does not prove) the hypothesis that candidate[1]'s holder class — if it's +a JDK/bootstrap class ordered in the tail of the app-classes-first +partition — simply hasn't been reached yet by any lap. Not proof: did +not confirm which specific class holds the static field pointing to +candidate[1]'s object, nor whether app-classes-first laps that already +completed for early parts of the classlist would have covered it. + +## Update: a full lap has since wrapped, "just needs time" is now weaker + +Re-checked at JVM uptime 1:27:00. `sweep_cursor` observed wrapping +(`33792 -> 0 -> 512 -> 1024 -> ... -> 3072` in one continuous 40s live +capture) — i.e. the classlist has now been walked end-to-end at least +once (`cycle_complete` still never observed as `1`, consistent with +`truncated=1` recurring in most chunks). `buildCanaryChainEvent(slot=3) +-> 0` is still the outcome on every one of 870 samples in that same +window; no `tag=` line for klass_id=283 anywhere in ~250k lines of live +log. The "hasn't been reached by any lap yet" explanation is therefore +weaker than initially thought, since at least one full pass over the +classlist has now happened. + +Also noted: candidate slot **assignment** (the `candidate[N]` loop-index +label in `pollWatchedTargets`) has shifted — klass_id=283 was previously +logged as `candidate[1]`, now as `candidate[0]`. The persistent identity +(`slot=3`, decoded from the marker tag) is unchanged, so this is NOT a +new occurrence of `find-marker-tag-slot-index-mismatch` — it matches the +already-investigated-and-retracted slot-churn concern in +`ev-postCB-onpod-live-verification` (loop-index label churns, `slot=N` +does not). + +## New hypothesis (unconfirmed): per-chunk truncation may recur at the same point every lap + +Chunks near the start of a lap (`sweep_cursor` 512-3072, right after +wrap) consistently show `truncated=1` with `edges_admitted≈0`, while +chunks further along (e.g. `sweep_cursor≈33280`) admitted hundreds of +edges in the same capture. Per the fix's design +(`find-static-field-sweep-cursor-fix`), a truncated chunk's cursor still +advances to the chunk boundary — the chunk is only revisited on the +*next* lap, called again with `FollowReferences` over the same 512-class +window from its start. If JVMTI's class enumeration order is stable +within a JVM's lifetime (typical, not guaranteed) and per-class +static-field-walk cost is roughly deterministic, the *same* subset of +classes near the tail of that chunk could be skipped by the deadline on +every single lap — a structural starvation gap at chunk granularity, +echoing the original whole-classlist version of this bug +(`find-static-field-sweep-never-completes`) but scoped smaller. **Not +confirmed** — would need per-class truncation-point visibility the +current `TEST_LOG` doesn't provide (it only reports chunk-level +`edges_admitted`/`truncated`, not which class index inside the chunk the +`FollowReferences` callback actually reached). If candidate[1]'s holder +class sits in such a starved region, that would fully explain the +persistent non-resolution independent of how many laps run. + +## Code-confirmed mechanism strengthening the hypothesis + +Read `heapReferenceCallback()` (`referenceChains.cpp:1480-1511`) and +`runPassManualWalk()` (`:2202-2345`) directly against the log data: + +1. On deadline exceeded, `heapReferenceCallback()` returns + `JVMTI_VISIT_ABORT` (`:1509-1510`), and the method's own comment at + `:1516` states this "aborts the entire FollowReferences walk (JVMTI + spec)" — i.e. a truncated chunk's call stops the whole 512-class + holder-array walk at whatever point it had reached, not just the + current class. +2. `_pass_deadline_ns` is set fresh (≈full `_effective_pause_target_ms`, + ≤50ms) at the top of `runPassManualWalk()` (`:2222-2224`), and + `admitStaticFieldRoots()` is the first thing that spends it + (`:2323`) — `resolveLoadedClasses()`'s own classlist-tagging cost + runs earlier, outside this budget (`:3000` vs. `:3067`). So each + chunk call gets a near-full fresh deadline, not a leftover sliver. +3. Despite that, `edges_admitted` is 0-1 in the large majority of + samples, including 5 consecutive post-wrap chunks + (`sweep_cursor=1024,1536,2048,2560,3072`, all `truncated=1`, + `edges_admitted∈{0,1,3}` - + `ev-postfix-static-field-onpod-live-verification`) — i.e. the walk + is tripping the 4096-callback-granularity deadline check almost + immediately, visiting only a small prefix of the chunk before + aborting, on call after call. + +Combined: if `GetLoadedClasses()` ordering is stable within the JVM's +lifetime (no spec guarantee, but no reason to expect churn for classes +already loaded) and the app-classes-first partition is deterministic per +call, each chunk boundary lands on roughly the same classes every lap. If +an early class within a chunk reliably generates enough callback traffic +(e.g. a class with a large/richly-connected static field graph) to trip +the deadline before the walk reaches deeper into that chunk, **the tail +of that chunk would be starved on every lap, not merely delayed** — a +smaller-scoped recurrence of the same failure mode +`find-static-field-sweep-never-completes` originally described for the +whole classlist, now at chunk granularity instead. If the class holding +the static field to candidate[1]'s object sits in such a tail, this +would fully explain zero resolution even after a full lap has wrapped. + +**Not proven, still a hypothesis**: cannot distinguish "deterministic +per-chunk cutoff" from "just unlucky most of the time under GC/scheduler +jitter" from current logs — `TEST_LOG` reports only chunk-level +`edges_admitted`/`truncated`, not which class index inside the chunk the +walk actually reached before aborting. Confirming would need either +per-class-index instrumentation inside `heapReferenceCallback()`'s +deadline-check path, or observing whether the SAME chunk's +`edges_admitted` value is stable/repeats across multiple laps (supports +determinism) vs. varies a lot (supports jitter-driven, eventually-covers- +everything). + +## Correction: volume driver is NOT a class's own field count + +User directly challenged the "a class's own static-field graph is large +enough to blow the full deadline" framing above: JVM class-file format +caps `fields_count` at `u2` (65535), and ordinary classes have far fewer +static fields than that — implausible as the sole explanation for +near-universal near-immediate per-chunk truncation. + +Re-read `heapReferenceCallback()` (`referenceChains.cpp:1576-1600, +1690-1703`) to test this directly: + +1. `:1576-1588` — when the static-field sweep's seed edge (holder[i] -> + class, tag<0) is walked, the continuation explicitly returns + `JVMTI_VISIT_OBJECTS`, and the surrounding comment (`:1594-1599`) + states this opens up "this class's own outgoing references - static + fields, superclass, interfaces, constant pool, class loader, ...". + I.e. NOT limited to static fields — JVMTI reports the class's *entire* + metadata reference graph as separate callbacks, one per edge. +2. `:1690-1703` — for the `admitStaticFieldRoots()` call, + `ctx->batch_tags` points at an *empty* set (`:2859-2860`), so every + referent reached one hop past the class fails + `batch_tags->count(my_tag) != 0` and returns `0` — descent stops dead + after exactly one hop. Confirms the "one hop past class" design intent + IS correctly enforced: a referent's own array elements/collection + internals are never visited via this path. This is not runaway + recursion. + +Conclusion: the user is correct that static-field count specifically +cannot explain the volume. The revised, code-grounded driver is the +class's *full* JVMTI-reported reference set — most plausibly dominated by +`JVMTI_HEAP_REFERENCE_CONSTANT_POOL` callbacks (one per resolved +constant-pool entry referencing a heap object: interned String literals, +Class literals, MethodHandle/MethodType constants). Constant-pool size is +capped at 65535 by the class file format (same order as `fields_count`), +but realistic population is very different from static-field count: +ordinary classes routinely carry hundreds to low-thousands of resolved CP +entries, vs. a handful of static fields — a fanout source with no +comparably small practical bound. Combined with the flat one-hop-wide +(not deep) expansion confirmed above, a small number of CP-heavy classes +near the front of a chunk could plausibly trip the 4096-callback +deadline-check granularity almost immediately, consistent with observed +`edges_admitted∈{0,1,3}` on successive chunks. + +**Not yet measured directly** — this is inference from code + +JVMTI-spec semantics (`JVMTI_HEAP_REFERENCE_CONSTANT_POOL`'s +per-resolved-entry firing), not a logged per-`reference_kind` callback +count on-pod. To confirm, would need temporary instrumentation adding a +per-`reference_kind` counter inside the deadline-check path to verify +CONSTANT_POOL (vs. static fields/interfaces/superclass) actually +dominates volume in a truncated chunk. + +## Confirmed via live per-kind instrumentation + +Deployed the temp `kind_counts` diagnostic (commit `3a2fc0d5e`, pod +`prof-analyzer-hotdog-jb-c944876b9-8vtzw`) and observed 36 samples over a +10-minute window: `CONSTANT_POOL` (k9) is the largest and most variable +kind in every single sample (2353-10894), always 5-15x larger than +`STATIC_FIELD` (k8: 174-2339) in the same sample. Full data and reading in +`ev-kind-counts-constant-pool-dominates`. This closes the "not yet +measured directly" gap — the driver is confirmed, not inferred. + +## Root cause, precisely + +`heapReferenceCallback()`'s admission path +(`referenceChains.cpp:1672-1705` pre-fix) has no `reference_kind` filter: +once the `static_field_seed` branch (`:1593-1604`) opens descent into a +class's own metadata graph via `JVMTI_VISIT_OBJECTS`, JVMTI delivers +*every* kind of edge from that class object as a separate callback — not +just `STATIC_FIELD`, but also `CONSTANT_POOL`, `INTERFACE`, `SUPERCLASS`, +`CLASS_LOADER`, etc. All of these were being fully processed through +`admitObject()` (hash-insert, tag, `edges_admitted++`, +`trackLeakAccumulation()`), each counted as a JVMTI callback against the +4096-callback deadline-check granularity and the pass's overall budget — +despite `admitStaticFieldRoots()` only wanting the `STATIC_FIELD` edges. +`CONSTANT_POOL` callbacks specifically only fire for already-*resolved* +constant-pool slots (interned Strings, resolved Class mirrors, +MethodHandle/MethodType/CallSite constants) — they are real, live GC +reachability edges, not JVMTI bookkeeping artifacts or symbolic/unresolved +references — so admitting them isn't "wrong" in the sense of being fake, +it's wrong in *category*: this sweep is specifically hunting for +objects reachable only via a static field, and constant-pool-interned +objects are a distinct, effectively-permanent-root category that doesn't +belong labeled under this walk's `root_kind`. + +## Fix — evolved through two iterations this session + +### Iteration 1 (superseded): hard reference_kind filter + +Initially implemented a hard filter at `referenceChains.cpp:1672-1686`: +when `ctx->static_field_seed` is true and the referrer is the opened +class object (`*referrer_tag_ptr < 0`), skip admission for any +`reference_kind != JVMTI_HEAP_REFERENCE_STATIC_FIELD`. This dropped ALL +non-STATIC_FIELD edges during the seed sweep. See +`dead-hard-reference-kind-filter` for why this was rejected. + +### Iteration 2 (current, implemented, built, gtest-pass, NOT yet committed/deployed): per-class non-static quota + resumable cursor + +User's real-world leak taxonomy drove the redesign: static fields are +the most common leak source, but CP-based leaks are still possible (if +rarer). The user's explicit requirement: "we need to design a system +working with this priority and not pushing completely out one or the +other." Two-part design: + +**Part 1 — Per-class non-static quota** (`referenceChains.cpp:1699-1731`): +STATIC_FIELD edges always admitted (high-priority leak root). +Non-STATIC_FIELD edges (CONSTANT_POOL, INTERFACE, SUPERCLASS, +CLASS_LOADER, etc.) admitted up to `STATIC_FIELD_SWEEP_NON_STATIC_CAP_PER_CLASS += 32` per class per lap, then dropped for the rest of that class. Cap +resets on class boundary (detected by `*referrer_tag_ptr != +ctx->_seed_class_tag`). One fat class cannot exhaust the quota for any +other. The cap (32, per user request) covers almost all classes' full +constant-pool/interface sets while bounding outliers. + +**Part 2 — Resumable cursor** (`referenceChains.cpp:2896-2908, +2993-3013`): the holder array is filled in **reversed** chunk order +(`holder[i] = classes[chunk_end - 1 - i]`) so HotSpot's LIFO +`FollowReferences` descent visits classes in **ascending** original +index order (see `ev-hotspot-lifo-visitation-order` for the source +proof). On truncation, the cursor resumes at the class being processed +(`chunk_start + _classes_in_chunk_visited - 1`) instead of skipping to +`chunk_end` — so classes after the interruption point are reached on +the next pass rather than lost for the rest of the lap. The partial +class is redone (already-admitted edges hit `ALREADY_ADMITTED` cheaply; +non-static edges complete within the cap). Chunk size stays at 512 — +the resumable cursor handles the timeout problem, so no chunk-size +decrease is needed. + +**PassContext additions** (`referenceChains.cpp:1463-1491`): +`_seed_class_tag` (current class being descended), +`_class_other_admitted` (non-static count for current class), +`_class_other_cap` (per-class cap, 0 disables), +`_classes_in_chunk_visited` (class-boundary counter for resumable +cursor). All zero/false outside the seed sweep — no overhead elsewhere. + +**Header** (`referenceChains.h:666-675`): new constexpr +`STATIC_FIELD_SWEEP_NON_STATIC_CAP_PER_CLASS = 32`. + +### Scoping verified + +`static_field_seed` has exactly one true-assignment call site +(`admitStaticFieldRoots()`), confirmed via grep — the quota logic +cannot fire during the general frontier walk. No risk of masking a +genuine static-field-reachable leak: JVMTI reports each edge separately +by kind, so an object reachable via a static field still gets its own +distinct `STATIC_FIELD`-kind callback regardless of whether +non-static callbacks for the same class were quota-dropped. + +### Build + test status + +`buildDebug` clean. `gtestDebug` full suite: 525 tests pass, 0 fail, 8 +skipped (sigaction-interception + TSan-only stress test). Includes the +90-test `referenceChains_ut` suite. `spotlessApply` clean (no changes). + +### Not yet committed / not yet deployed + +The fix is uncommitted in the working tree (`referenceChains.cpp` + +`.h` modified). The temp diagnostic `kind_counts` tally (commit +`3a2fc0d5e`) is still in place. Next step: user reviews, commits, +deploys to the hotdog pod, and live-verifies whether candidate[1] +(klass_id=283, slot=3) finally resolves and a real `ReferenceChain` +event gets emitted. + +## Update: sweep completed a full lap, candidates still 0/1 + +With the 200ms deadline diagnostic build (commit `fd18425c6`), the +sweep completed a full untruncated lap (`cycle_complete=1` observed, +`last_static_field_class_count` advanced from 33677 → 33752). All +chunks completed without truncation. See +`ev-timing-split-callback-vs-jvmti`. + +Despite the sweep being healthy, candidates are **still 0/1**. The +bottleneck moved downstream: the sweep admits the static field value +(e.g., an `ArrayList`) to the frontier, but `expandFrontier()` must +then expand that `ArrayList` to reach the leaking object. The +`ArrayList` is pushed to the **back** of `_pending_expand` (144k +entries deep), and at 65-87 edges/pass the BFS can't reach it. See +`find-sweep-completes-but-bfs-starved`. + +## Two-hop chain architecture (confirmed from code) + +The sweep admits only **one hop** past the class: `batch_tags` is empty +for the sweep call (`:2859`), so `heapReferenceCallback()` returns `0` +(no descent) for every referent (`:1813`). A static field value gets +admitted to the frontier but its elements are NOT visited by the sweep. + +`expandFrontier()` picks entries from the **front** of `_pending_expand` +(FIFO deque, `:2654`), calls `FollowReferences` on them — that's when +the collection's elements get visited and the canary match can fire. + +``` +Sweep: class → STATIC_FIELD → ArrayList [admitted, pushed to BACK of _pending_expand] +expandFrontier: front of _pending_expand → ... → ArrayList → elements → leakingObject +``` + +## Status + +Root cause confirmed and fixed (per-class quota + resumable cursor, +committed `0e93ab4f7`). Sweep now completes laps cleanly. But the +end-to-end problem persists: BFS throughput is starved by a 144k +backlog. See `find-sweep-completes-but-bfs-starved`. Next: report to +user and propose a fix direction (prioritize sweep-admitted entries or +increase BFS throughput). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-candidates-234-die-before-resolution.md b/.investigations/missing-refchains-on-hotdog/nodes/find-candidates-234-die-before-resolution.md new file mode 100644 index 0000000000..10b51ec48a --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-candidates-234-die-before-resolution.md @@ -0,0 +1,55 @@ +--- +id: find-candidates-234-die-before-resolution +type: question +status: refuted +depends_on: [ev-postfix-static-field-onpod-live-verification] +supersedes: [] +related: [] +tags: [open, referenceChains, candidate, canary, sample-lifetime, NEW-THIS-SESSION] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# Open: candidates 2-4 repeatedly report representative died/evicted before resolution + +## Observation + +Across multiple post-fix check windows, candidate slots 2-4 (klass_id +varying: 50, 236, 237, 232, 279 observed at different times/slots — the +slot<->klass_id mapping appears to shift between checks, not yet +investigated) repeatedly log: + +``` +candidate[N] klass_id= representative could not be resolved (died/evicted) +``` + +Counts from one 2-minute window: klass_id=236 x3, klass_id=50 x10 (slot +1); klass_id=237 x38, klass_id=50 x36 (slot 2); klass_id=236 x36, +klass_id=50 x43 (slot 3); klass_id=232 x36, klass_id=236 x40, +klass_id=237 x53 (slot 4). This is a dominant, high-frequency pattern — +not a rare edge case. + +## Hypothesis (not yet investigated) + +Likely a sample-lifetime vs. candidate-selection/search-cadence mismatch: +the sampled representative object for these candidates is short-lived +enough to be GC'd before the reference-chain search gets around to acting +on it, independent of the static-field-sweep fix. This is a plausible +**pre-existing** issue (the static-field-sweep cursor fix only changes +static-field root admission, not sample/candidate selection or object +lifetime), but has NOT been confirmed as pre-existing — no comparison +against pre-fix logs has been done, and no code-read of the +representative-resolution path has happened yet. + +## Status: REFUTED as a bug — expected app behavior + +User confirmed directly: "the 2-4 dying is because we have only one +reliable leak in the app. everything else are +more-than-ephemeral-instances but they keep on getting garbage +collected." I.e. only one candidate (the actual synthetic leak target, +resolved as candidate[1]/klass_id=283 per `find-candidate1-never-tagged`) +is a genuine permanently-retained object; the other selected candidates +are just longer-lived-than-typical instances that are still ordinarily +collectible, so their sampled representatives dying before the search +reaches them is exactly the expected outcome, not a search/tooling +defect. No further investigation needed here. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-cpu-pain-budget-blocks-bfs.md b/.investigations/missing-refchains-on-hotdog/nodes/find-cpu-pain-budget-blocks-bfs.md new file mode 100644 index 0000000000..715e267aae --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-cpu-pain-budget-blocks-bfs.md @@ -0,0 +1,47 @@ +--- +id: find-cpu-pain-budget-blocks-bfs +type: finding +status: diagnosed +depends_on: [] +supersedes: [] +related: [find-shared-deadline-starves-expand, q-safepoint-budget-model] +tags: [root-cause, referenceChains, cpu-pain-budget, shouldRunPass, silent-gate, NEW-THIS-SESSION] +created: 2026-08-28 +updated: 2026-08-28 +--- + +# cpu_pain_budget silently blocks BFS in RUNNING state + +## Observation + +After the leak signal fired (5 candidates, `heapFloorRising=1`), `shouldRunPass` +returned `false` on every iteration with **no log**. Only `pollWatchedTargets` +ran (unconditional). The BFS never started — zero `runPass` logs. + +## Root cause + +In `shouldRunPass()`, the RUNNING branch (line ~895) checks +`_cpu_pain_budget.canStartNow(now_ns)` and returns `false` with no log when +it fails. A previous search had spent ~1075ms of CPU pain into the budget. +At 12% refill rate (canary escalation: 3% × 4), the budget drained at +~12.5ms per ~100ms iteration — taking ~90 seconds to drain enough for the +next pass. + +This was **completely silent** — no log, no counter, no diagnostic. The +only symptom was `pollWatchedTargets` running with zero `runPass` logs. + +## Fix (diagnostic, committed fcc67179a) + +Added `TEST_LOG` to the cpu_pain_budget block path, reporting balance, +refill_rate, and canary_active. Confirmed on-pod that the budget was the +block. + +## Note + +This is not a bug per se — the pain budget is working as designed (rate- +limiting CPU cost). But the 12% refill rate for canary searches means +~90 seconds between passes after a search that spent 1000ms of CPU. For +a 2-3 hop leak that needs ~5 passes, that's ~7.5 minutes of BFS time +after the initial debt drains. The canary multiplier (4×) may need to +be higher, or the initial debt should be reset when a canary search is +active. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-cpu-pain-budget-starves-canary-passes.md b/.investigations/missing-refchains-on-hotdog/nodes/find-cpu-pain-budget-starves-canary-passes.md new file mode 100644 index 0000000000..1c7e3d1031 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-cpu-pain-budget-starves-canary-passes.md @@ -0,0 +1,107 @@ +--- +id: find-cpu-pain-budget-starves-canary-passes +type: finding +status: confirmed +depends_on: [find-canary-stuck-abandon-detector] +supersedes: [] +related: [find-threadloop-presleep-blocks-back-to-back, ev-hotdog-trace-zero-runpass] +tags: [root-cause, referenceChains, pain-budget, canary, starvation, shouldRunPass, silent-gate] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# `_cpu_pain_budget` silently starves `runPass()`, explaining zero canary resolution + +## Reasoning chain + +Second direction the user asked to "proceed" on: why has zero canary +candidate ever resolved, even with Fix A/B/C all confirmed working? + +`shouldRunPass()` (`referenceChains.cpp:840-937`) checks, in order: first-pass +branch, terminal-state branch, then at `:893` +`if (!_cpu_pain_budget.canStartNow(now_ns)) { return false; }` — **the only +branch in the whole function with no `TEST_LOG` call** — before it ever +reaches the canary-active bypass (`_candidate_count > 0 && popcount(found) < +candidate_count` → "run immediately", which does have its own `TEST_LOG`). +A blocked pain budget is therefore invisible in logs: you see +`pollWatchedTargets` firing every cycle (it's unconditional, `:801`) and +never see a single `runPass` line, with no diagnostic clue why. + +`PainBudget` (`painBudget.h`) is a leaky bucket over *cost*, not *rate*: +`spend(pain_ms)` adds debt, `canStartNow()` drains `elapsed_ms * refill_rate` +off the balance and returns true only once balance ≤ 0. `_cpu_pain_budget` is +spent every single pass (`referenceChains.cpp:3033`, +`_cpu_pain_budget.spend(TSC::ticks_to_millis(non_safepoint_ticks))`) with the +pass's non-safepoint CPU cost (root/stack-ref enumeration dispatch, frontier +admission, rotation-candidate collection — i.e. real cost proportional to +heap/frontier size, not a fixed small constant). + +Traced `_refill_rate`'s value (not done in the prior session, completed this +turn): NOT hardcoded to 0.0 in production. `referenceChains.cpp:505-511` +constructs it from `_pain_budget_refill_rate = std::max(args. +_reference_chains_pain_budget_percent, 0) / 100.0`, and +`_reference_chains_pain_budget_percent` defaults to +`DEFAULT_REFERENCE_CHAINS_PAIN_BUDGET_PERCENT = 1` (`arguments.h:104,359`) — +i.e. **1% by default**, tunable via `painbudget=<0-100>` in the reference-chains +config string (`arguments.cpp:545-548`). So the earlier "refill_rate == 0.0 +forever" landmine from `painBudget.h`'s own comment does NOT apply verbatim — +this rules out a literal permanent block. + +What a 1% refill rate means in practice: draining N ms of debt takes N/0.01 = +**100×N milliseconds** of wall-clock time. A pass costing just 500ms of +non-safepoint work (plausible on a large heap with a wide frontier — this is +exactly the "genuinely expensive, bounded operation" `painBudget.h`'s own +comment describes) creates a **50-second** drain requirement before the next +pass is even considered, regardless of how urgently a canary candidate is +waiting. This is consistent with, and sufficient to explain, the live +evidence below. + +## Evidence + +Continuous 45-second trace captured directly from the pod (`kubectl logs -f +--since=1s` to `/tmp/hotdog_trace.log`, 25,865 lines — see +`ev-hotdog-trace-zero-runpass` for the full evidence writeup): + +- `grep -c "runPass done"` → **0** matches in 45s. +- `grep -c "pollWatchedTargets"` → **1472** matches in the same window + (confirms this is unconditional and unaffected — `referenceChains.cpp:801` + runs it every loop iteration regardless of `should_run`). +- `pollWatchedTargets` output shows `candidate_count=5`, all 5 slots + occupied, candidates repeating unchanged across many polls — consistent + with a search that is not advancing because `runPass()` itself is not + running, not because candidates aren't being found/tracked. + +This directly rules out "cadence sleep alone" as sufficient explanation: +`threadLoop()`'s cadence sleeps top out at roughly 1-2s per idle iteration +(see `find-threadloop-presleep-blocks-back-to-back`), which cannot account +for a full 45-second window with literally zero passes. A silent gate that +can block for tens of seconds at a time is required, and `_cpu_pain_budget` +is the only such gate in the code (confirmed: the only branch in +`shouldRunPass()` without a `TEST_LOG`). + +## What this rules out + +- **Literal permanent block from `_refill_rate == 0.0`** — refuted. + `_pain_budget_refill_rate` is constructed from a nonzero default (1%, + `DEFAULT_REFERENCE_CHAINS_PAIN_BUDGET_PERCENT`), not left at the + `PainBudget` default-constructor's `0.0`. The budget *does* eventually + drain — just far too slowly relative to realistic per-pass cost for the + canary-bypass "run back-to-back" design intent to hold. +- **Cadence sleeps as the sole cause of the 45s silent window** — refuted by + the arithmetic (max ~1-2s per idle iteration) not matching the observed + duration. + +## Not yet done + +- Have not measured the actual non-safepoint cost of a single `runPass()` on + this pod's heap directly (no `TEST_LOG` exists at the spend site either — + same silent-gate problem, one level removed). The 100×-drain-time argument + is derived from the documented refill-rate semantics and the config + default, not from a directly observed `_balance_ms` value. + `PainBudget::balanceMs()` exists for exactly this kind of introspection but + is not currently logged anywhere. +- No fix proposed or implemented yet — this is a design-level tradeoff (how + aggressively the canary bypass should be allowed to override the pain + budget, or whether the budget needs a canary-aware carve-out) that should + be proposed to the user before touching `shouldRunPass()`'s gating order or + `PainBudget`'s semantics. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-dangling-jclass-local-ref-cache.md b/.investigations/missing-refchains-on-hotdog/nodes/find-dangling-jclass-local-ref-cache.md new file mode 100644 index 0000000000..7c7af81e2d --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-dangling-jclass-local-ref-cache.md @@ -0,0 +1,35 @@ +--- +id: find-dangling-jclass-local-ref-cache +type: finding +status: confirmed-and-fixed +depends_on: [find-test-seam-aliasing] +related: [find-test-seam-aliasing, find-engine-seam-data-race] +tags: [root-cause, fix, jni, crash, test-seams, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# java/lang/Object jclass cached as a LOCAL ref dangles across JNI invocations + +Observed: SIGSEGV (hs_err) inside `jni_NewObjectArray` on the SECOND +`runReferenceChainPass0()` call from a Java test thread - +`ReferenceChainTracker::_cached_object_class` held a jclass obtained via +`FindClass()` and cached with a `_cached_object_class_jni` JNIEnv* key. + +## Root cause (JNI semantics) + +A local ref is freed when its creating native method invocation returns to +Java. The cache was safe for the BFS thread (a native attach that never +returns to Java), which was its only caller - the JNI-entered test seams +broke that assumption: pass 1 caches the local ref, returns to Java, the +JVM frees it, pass 2 passes the dangling jclass into `NewObjectArray()`. + +## Fix + +`_cached_object_class` is now a GLOBAL ref (valid across invocations, +threads, and detach/attach cycles), which also made the JNIEnv*-identity +keying, the detach-time invalidation in threadLoop's Cleanup, and the +startThread() stale-cache clearing all unnecessary - deleted. Never freed: +java/lang/Object is a bootstrap class and the tracker is a +process-lifetime singleton. The gtest mock JNIEnv needed a `NewGlobalRef` +slot added (mock = identity - the fixture's fake refs are raw pointers). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-default-live-samples-ratio-lottery.md b/.investigations/missing-refchains-on-hotdog/nodes/find-default-live-samples-ratio-lottery.md new file mode 100644 index 0000000000..48cc713f56 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-default-live-samples-ratio-lottery.md @@ -0,0 +1,77 @@ +--- +id: find-default-live-samples-ratio-lottery +type: finding +status: fixed +depends_on: [ev-leaktag-correlation-local-repro, find-per-tid-qualification-design] +related: [ev-leaktag-onpod-round4, find-test-seam-aliasing] +tags: [fix, livenessTracker, tests, flakiness, sampler, NEW-THIS-SESSION] +created: 20260901 +updated: 20260901 +--- + +# The intermittent "sampler-dead" suite failures were the default 10% live-sample ratio lottery + +## The failure (mis-named "sampler dead") + +Intermittently (suite-only at first glance, ~2 of 7 full-suite runs, +both LeakTagCorrelation children failing together some runs), the +leak-correlation child JVM produced zero leak tagging: zero folds +with leak-klass scratch, tagLeakInstances matched nothing, result +[correlation-tag-out-of-pool]/NOT_FOUND. Isolated reruns always +passed. Looked environmental (machine load). + +## Root cause (instrumented, then code-proven) + +`memory=64:l` without an explicit ratio leaves +`Arguments::_live_samples_ratio` at its default **0.1** +(arguments.h: "default to liveness-tracking 10% of the allocation +samples"). LivenessTracker::track() drops each tracked instance with +p=0.9 (the `skipped` accumulator compensates in JFR weights - fine +for volume workloads, silent for tiny cohorts). + +The leak-correlation scenario's leak cohort is ~35-50 1MB chunks - +P(zero survivors) lands in the observed few-percent tail, and the +per-run pass/fail was literally this lottery. The scenario's own +comment ("every allocation is sampled, tracked, and taggable") was +never true under the default. The passing runs' stable "tagged=5" +is exactly the ~10% of ~50 chunks. + +Instrumentation (now fully reverted) pinned it: track() entries +showed the chunks arriving, zero subsample-DROP logs with an +explicit ratio, and the class-id red herrings (cached_klass_id=0, +id namespace 1-vs-2) dissolved once the cohort actually survived +to be resolved. + +## Fix + +LeakTagCorrelationReferenceChainTest now passes +`memory=64:l:1.0` (arguments.cpp parses the third segment as the +live-sample ratio, clamped [0.01,1.0]) with a comment explaining +the lottery. Verified: post-fix suite runs show zero +leak-correlation failures, correlation-found every time. + +Follow-up: the same explicit `:l:1.0` was applied to the other three +`:l`-without-ratio reference-chain sites (ReferenceChainTrackingTest's +shared config; both ExternalProcessReferenceChainTest children) - +removes the silent probabilistic default from every reference-chain +test and matches the scenarios' documented "every allocation is +tracked" intent. Effect on the TrackingTest scenarios: the +"representative could not be resolved (died/evicted)" symptom is GONE +(10x thicker rep pool), but the ToGcRoot chase-pacing flakiness is NOT +this bug - see q-togcroot-acceptance-paths. + +## Why NOT the pod + +The pod's `--preset cpu_live_heap` also uses the default ratio, but +the simulated-leak thread allocates continuously - a 90% drop just +subsamples a large population (round 4's stable "tagged=5 need_set=0" +tags). The lottery only bites cohorts of tens. No pod change needed. + +## Remaining known flaky family (separate, NOT this bug) + +ReferenceChainTrackingTest ToGcRoot/UnboundedCache still fail under +heavy machine load (observed at load 25-55; the in-process executor +JVM's passes run ~4.4s under the 5s pausetarget, and the canary +backoff spaces the chase by design - see +find-canary-lane-backoff-design). Load-sensitive timing, not the +ratio lottery. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-depth0-durable-root-upgrade-gap.md b/.investigations/missing-refchains-on-hotdog/nodes/find-depth0-durable-root-upgrade-gap.md new file mode 100644 index 0000000000..27f67cb1d5 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-depth0-durable-root-upgrade-gap.md @@ -0,0 +1,46 @@ +--- +id: find-depth0-durable-root-upgrade-gap +type: finding +status: confirmed-and-fixed +depends_on: [find-gate-bypass-representative-paths] +related: [find-gate-bypass-representative-paths, find-rotation-resize-blindspot] +tags: [root-cause, fix, root-kind, gate, durability, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# Depth-0 root-kind upgrade never fired for root-like heap edges + +Found while fixing the seams test (its target stayed +`root_kind=24 (STACK_LOCAL)` through every poll even though it was also +the direct value of a static field). + +## Root cause + +`maybeUpgradeRootAttachedRootKind()` (the durability tie-break: a later, +more durable root discovery upgrades an entry's root_kind) was wired ONLY +into `heapRootCallback`'s ALREADY_ADMITTED case. But the static-field sweep +delivers its class -> field edges through `heapReferenceCallback`, where a +class referrer (negative tag) is deliberately treated as ROOT-LIKE +(`parent_tag == 0`, depth 0 - class objects are never frontier entries). +Root-like edges on already-admitted entries fell through every handler: +`improveChain` and `reparentToDurableRoot` both require `parent_tag != 0`. +So an object first admitted through a stack local kept its transient +classification FOREVER, even after the sweep proved static retention. + +## Production impact + +Real: a leak instance is typically a local variable while being stored +into the static/field-held collection - root enumeration admits it as +STACK_LOCAL first, the static discovery arrives later, and the depth-0 +chain was then noise-gated (`depth < 2 && isTransientRootKind`) exactly +like the noise it was meant to distinguish. + +## Fix + +`heapReferenceCallback` now calls +`maybeUpgradeRootAttachedRootKind(*tag_ptr, reference_kind)` for +root-like edges to already-admitted entries (same tie-break as the root +callback), with `invalidateResolvedChain()` so a cached chain rebuilds +with the upgraded root. Durability ranking (STATIC_FIELD/SYSTEM_CLASS=3 > +JNI_GLOBAL=2 > transient=1) unchanged. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-edit-removed-critical-code.md b/.investigations/missing-refchains-on-hotdog/nodes/find-edit-removed-critical-code.md new file mode 100644 index 0000000000..accde318ea --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-edit-removed-critical-code.md @@ -0,0 +1,69 @@ +--- +id: find-edit-removed-critical-code +type: lesson +status: confirmed +tags: [methodology, edit-safety, regression, lesson, NEW-THIS-SESSION] +created: 20260915 +updated: 20260915 +--- + +# Edit accidentally removed a critical algorithm loop + +## What happened + +While removing a TEMP diagnostic from `admitStaticFieldRoots()` in +`referenceChains.cpp`, the `edit` tool's `oldText` matched the entire +block containing BOTH the diagnostic code AND the original holder-fill +`for` loop: + +```cpp + for (jint i = 0; i < chunk_count; i++) { + // TEMP DIAGNOSTIC ... + { ... GetClassSignature ... } + jni->SetObjectArrayElement(holder, i, classes[chunk_end - 1 - i]); + if (jniExceptionCheck(jni)) { holder = nullptr; break; } + } +``` + +The replacement `newText` was just `}` — the closing brace of the +surrounding `if (holder != nullptr)` block. This silently deleted the +entire holder-fill loop, leaving the holder array empty. The sweep +would have admitted zero static fields — a silent total regression. + +The build compiled clean (syntactically valid: an empty `if` block). +The bug was caught only by reviewing the full `git diff` afterward. + +## Root cause + +The diagnostic was interleaved INSIDE the existing `for` loop body. +Removing the diagnostic by replacing the whole `for` block removed the +original code along with it. The `edit` tool has no concept of +"original code vs. diagnostic code" — it matches text, and the +`oldText` spanned both. + +## Prevention rules + +1. **Never replace a block that mixes diagnostic code with original + code.** When removing a diagnostic that was interleaved into existing + logic, edit to restore the original lines, not to delete the whole + block. The `oldText` should be the diagnostic lines only; the + `newText` should be empty or the original lines without the + diagnostic. + +2. **Always review the full `git diff` after edits, before building.** + A compiling build does NOT prove correctness — a missing loop is + syntactically valid. The diff is the only safety net. + +3. **When a diagnostic is inside a loop, prefer adding a separate + `TEST_LOG` line BEFORE or AFTER the loop, not interleaved inside it.** + A diagnostic that sits in its own statement is trivial to remove + without touching surrounding code. + +4. **If the diagnostic MUST inspect per-iteration state, add it as a + minimal inline check** (e.g., a single `if` + `TEST_LOG`), never a + multi-line block that visually merges with the loop body. Remove by + deleting only those lines. + +5. **Diff against the version BEFORE the diagnostic was added, not the + version with the diagnostic.** `git diff -- + ` shows whether the only changes are the diagnostic removal. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-ema-batch-collapse.md b/.investigations/missing-refchains-on-hotdog/nodes/find-ema-batch-collapse.md new file mode 100644 index 0000000000..a173812264 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-ema-batch-collapse.md @@ -0,0 +1,79 @@ +--- +id: find-ema-batch-collapse +type: finding +status: confirmed +depends_on: [find-getobjectswithtags-quadratic-bottleneck] +supersedes: [] +related: [find-getobjectswithtags-quadratic-bottleneck, find-leak-tag-pool-implementation, find-cpu-pain-budget-blocks-bfs] +tags: [root-cause, fix, referenceChains, expandFrontier, GetObjectsWithTags, ema, aimd, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# Per-tag EMA batch calibration collapses to batch=2 (regression of 337b4c21d) + +## Observed on-pod (build 1ce2b4f03, /tmp/hotdog-15m.log analysis) + +- `expandFrontier gotw batch_size=2 resolved=2 ... gotw_ms=15-25 ema=~10.5e6` + — ema (per-tag) at 10.5 **milliseconds**/tag vs ~62k ns/tag at the previous + build. Edges/pass collapsed from 1700-2800 to ~100; only 3 runPass in 15min. +- Each pass: ~1000+ GetObjectsWithTags calls × ~20ms = 10.4s CPU spend into + the cpu_pain_budget (debt 10436ms observed), ~28s wall per pass. +- Result: BFS never reached the leak-tagged objects → zero useful chains. + +## Mechanism (proven by the numbers) + +GetObjectsWithTags call time is dominated by a batch-independent +O(tag_map) floor (~20ms at a 225k-entry frontier tag map — observed at +both batch=2 (15-25ms) and batch=400 (20-35ms)). The calibration +`ema = elapsed / batch_size` therefore INFLATES as batch shrinks: +small batch → per-tag cost up → `batch = budget/ema` down → positive +feedback. The equilibrium "call time = deadline" is degenerate once the +floor alone ≈ deadline, so the batch falls to the minimum and stays. + +## Second hole found at the same time + +The expand loop had NO wall-clock deadline check: heapReferenceCallback's +amortized (0xFFF) deadline check only runs inside FollowReferences, so the +gotw calls between walks were never bounded — hence the 1400-call passes. + +## Fix (commit 0db70994d) + +AIMD directly on batch size against a per-CALL EMA vs GOTW_CPU_BUDGET_NS: +- over budget → batch /= 2 (floor 8) +- under budget → batch += 64 (cap 512; JNI local-ref bound) +Per-call cost is batch-insensitive in the floor regime, so AIMD converges +to the largest affordable batch instead of collapsing. Plus a deadline +check at the top of every expand while-iteration (gotw is non-safepoint, +invisible to the in-callback check). + +Also fixed in the same commit: the emergency multiplier used +`_passes_since_last_progress` (frontier growth — resets every pass, +emergency=0 for the entire run) instead of +`_passes_since_last_candidate_progress`. + +## Test + +AdaptiveBatchSizeAimdDecreaseAndIncrease (referenceChains_ut.cpp) drives +one expandFrontier call per phase via test seams (a full runPass drains a +small graph AND adds rotation-phase gotw calls, making per-call AIMD +arithmetic unverifiable). AIMD state is NOT covered by +ReferenceChainsTestAccessor::reset() — tests must zero it explicitly. + +## SUPERSeded by proportional batch control (this session, verified locally) + +The AIMD fix (per-call EMA vs fixed 25ms budget) survived only until the +tag map's batch-independent floor (~27ms at 243k entries) exceeded the +fixed budget - measured live on the pod round 4: batch=8 forever while +batch=72 cost only 36ms for 9x objects. The EWMA itself was never the +problem (it survives, alpha=0.8); the FIXED THRESHOLD below the floor was. +Current law: same per-call EMA, `next = batch x remaining_deadline / ema`, +clamped [MIN,MAX] - sizes one call to fill the pause window; converges up +while calls are cheap, shrinks as the window drains, tracks the floor as +the map grows/shrinks. Verified on a real JVM locally: batch 270 -> +next 512, gotw 16-18ms. Unit: AdaptiveBatchSizeProportionalToWindow +(replaces the AIMD test). Also fixed in the same round: the fair-share +lane toggle must PERSIST across expandFrontier invocations (a local reset +made priority win every invocation - observed live: pending 109k->113k +never drained); FairShareLaneAlternationPersistsAcrossInvocations covers +it (mock gotw busy-wait knob bounds one batch per invocation). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-engine-seam-data-race.md b/.investigations/missing-refchains-on-hotdog/nodes/find-engine-seam-data-race.md new file mode 100644 index 0000000000..89610714bd --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-engine-seam-data-race.md @@ -0,0 +1,43 @@ +--- +id: find-engine-seam-data-race +type: finding +status: confirmed-and-fixed +depends_on: [find-test-seam-aliasing] +related: [find-dangling-jclass-local-ref-cache, find-priority-queue-starves-bfs-crawl] +tags: [root-cause, fix, threading, crash, test-seams, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# Seam-driven runPass/poll races the BFS thread on unsynchronized engine maps + +Observed: SIGSEGV in `ClassTagTable::insert`'s unordered_map rehash, from a +test thread inside `resolveLoadedClasses()` (runPass) while the BFS thread +was mid-pass of its own (the seams test creates a live leak candidate, so +threadLoop runs passes concurrently with the test's synchronous +runReferenceChainPass0/pollReferenceChainTargets0). + +## Root cause + +runPass()/pollWatchedTargets() mutate plain containers +(`_class_tags`, `_candidate_*`, `_leak_parent_fanout`, ...) with no +cross-thread locking - safe only because threadLoop was the single caller. +The JNI-entered seams added a second caller. + +## Fix + +`Mutex _engine_lock` serializes the two engine drivers at entry: +`runPassSerialized()` / `pollWatchedTargetsSerialized()` (lock-scoped +wrappers) used by BOTH threadLoop and javaApi.cpp's seam functions. Full +pthread mutex, not a spin lock - the critical section is a whole BFS pass +(tens of ms). Lock order is engine_lock -> inner locks (frontier spinlock, +LivenessTracker's table lock); no reverse path exists. +`resetSearchStateForTest` was already safe (stopThread() first). + +## Semantic residual (not a memory bug) + +threadLoop still interleaves BETWEEN the test's steps: it can re-admit the +seam-tagged target via root enumeration, SetTag races tagAsRootForTest, +and the chunked static-field sweep can lap before the fixture populates +the holder. The seams test's fixture must therefore not depend on a +particular discovery order (see find-test-seam-aliasing's scenario rules). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-field-name-decoding.md b/.investigations/missing-refchains-on-hotdog/nodes/find-field-name-decoding.md new file mode 100644 index 0000000000..0114c3c6ef --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-field-name-decoding.md @@ -0,0 +1,109 @@ +--- +id: find-field-name-decoding +type: finding +status: implemented +depends_on: [find-option-c-descend-walk-design, find-leaktag-jfr-field-misalignment] +related: [ev-hotspot-lifo-visitation-order] +tags: [field-names, hop-labels, jvmti-spec, ReferenceChain, jfr, NEW-THIS-SESSION] +created: 20260902 +updated: 20260902 +--- + +# Per-hop retention-edge field names in datadog.ReferenceChain (edges array) + +User ask: "record the field name through which the referrer reaches +referee in the refchains events". Answer: yes, via capture-at-admission + +spec-decode-at-emission. IMPLEMENTED and green (554 gtests incl. new decode +test, 9/9 reference-chain slow family, Java JMC parser test asserts the new +T_STRING|F_ARRAY field end to end). + +## The three load-bearing facts (source-verified, not inferred) + +1. JVMTI heap callbacks expose the field identity as + reference_info->field.index - a jint ORDINAL over the referrer's + FLATTENED field space, NOT a jfieldID and NOT a position in one + GetClassFields result (found via a compile error against the real + jvmti.h). The numbering is SPEC-DEFINED (jvmti.xml + jvmtiHeapReferenceInfoField, lines ~3635-3690): count of all + interfaces-implemented fields (transitive, each once) + superclass + chain root-first (java.lang.Object's fields first) + own fields, each + class's fields in GetClassFields order, all modifiers included. The + interface-referrer branch (static interface constants) uses only the + interface's own fields over its superinterfaces' count. +2. HotSpot's implementation matches the spec exactly: + jvmtiTagMap.cpp ClassFieldMap::create_map_of_static_fields / + create_map_of_instance_fields build exactly that ordinal + (interfaces_field_count + superclass chain + JavaFieldStream), and + GetClassFields returns the own-declared JavaFieldStream order + (jvmtiEnv.cpp:2805-2847) - the two orders compose into the spec space. +3. During heap callbacks only tag/alloc JVMTI functions are callable - + GetFieldName/GetClassFields are NOT. So the ordinal MUST be captured at + admission (same cannot-replay-the-callback rationale as + FrontierEntry::class_tag) and resolved to a name later, outside walks. + +## Implementation + +- FrontierEntry += referrer_field_index (jint, -1 = not a field edge) + + edge_kind (u8, interior-hop edge kind) + referrer_class_tag (jlong, + declaring class for root-attached static edges where the referrer is a + class object with no parent entry). ~16B/entry, zero hot-path + allocation; improveChain()/reparentToDurableRoot() refresh the edge + identity on re-parent so a replaced chain's label describes the WINNING + edge. +- reconstructChain() collects ChainHopEdge per hop; buildChainEvent() + (signature now env-threaded: partial mock tables would crash on + unmocked slots - the JFR-roundtrip crash lesson applied) calls + fillHopEdgeLabels(). +- hopLabelClassFor(): per-referrer-class decoded ordinal->name list, + cached (cap 1024, cleared on restart; chains re-emit every dump, the + decode walks the class's interface closure + superclass chain so caching + is not optional). Decode = GetObjectsWithTags(class_tag) -> jclass; + IsInterface picks the spec branch; JNI GetSuperclass walks the chain. +- FAIL-SAFE CONTRACT: any decode failure (class gone, ordinal out of the + computed range, partial function table, null env) degrades to the edge + KIND label ("element", "constant_pool", ...) - never a fabricated name. + A wrong numbering on an unverified JVM degrades; it does not lie. +- Event: ReferenceChainEvent._edges (std::vector, aligned + leaf-to-root with _chain: edges[i] = the retention edge INTO chain[i]); + new JFR field "edges" T_STRING|F_ARRAY written AFTER the chain array in + both metadata and writer (the leakTag field-order invariant + generalized); MAX_REFERENCE_CHAIN_EDGE_LABEL=96 shared between the + collector and the writer's buffer reservation. + +## JVMTI gotchas found the hard way (each caught by compile against the +real JDK 26 jvmti.h - the project's own headers are the ground truth) + +- jvmtiHeapReferenceInfoField.index is a jint ordinal, not a jfieldID. +- GetMethodName is for METHODS; fields use GetFieldName. +- Modern JVMTI has NO GetSuperclass - it is a JNI function + (jvmti.xml delivers superclass references via heap callbacks only). + +## Verification + +- gtest HopEdgeLabelsDecodeSpecFieldOrdinals: interface offset, + superclass order, interface-referrer branch, and the fail-safe degrade + all asserted against a fake hierarchy driven through the mock slots. +- referenceChainJfrRoundtrip_ut now seeds edge kinds and asserts the + kind-label edges; ReferenceChainJfrParserTest (Java/JMC) asserts the + "edges" field parses as a String[] with exact labels - the T_STRING + F_ARRAY encoding is validated by a real JMC read, not assumed. +- 554 gtests green; reference-chain slow family 9/9 green (every emitted + chain now carries edges). + +## OPEN: J9 numbering compliance is INFERRED, not verified + +The spec defines the numbering and J9 passes the walker's referrerIndex +through for FIELD refs (runtime/jvmti/jvmtiHeap10.c +J9GC_REFERENCE_TYPE_FIELD -> JVMTI_REFERENCE_FIELD), but the OMR +reference-chain-walker arithmetic that produces referrerIndex was not +source-located (unauthenticated GitHub search hit a wall; the walker lives +under eclipse-omr/openj9's gc side). The fail-safe design means a J9 +mismatch degrades labels to kinds, never misnames - but verifying the J9 +walker against the spec ordinal (a local openj9 checkout would make it a +one-file read) is the remaining follow-up. NOT a pod blocker. + +## Value + +Chains become readable retention paths ("LeakHolder.SINK -> +HashMap.table -> Entry.value") - the diagnostic point of the feature; also +directly names the holder field for the pod round-7 verification. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-fresh-lane-verification.md b/.investigations/missing-refchains-on-hotdog/nodes/find-fresh-lane-verification.md new file mode 100644 index 0000000000..60f4faaa14 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-fresh-lane-verification.md @@ -0,0 +1,69 @@ +--- +id: find-fresh-lane-verification +type: fix-verification +status: confirmed +depends_on: [find-tier1-tail-starvation, ev-leaktag-onpod-round15, ev-leaktag-onpod-round15-results] +related: [find-wrapper-demotion-self-parent] +tags: [fix-verification, fresh-lane, pod-verification, round-15, NEW-THIS-SESSION] +created: 20260916 +updated: 20260916 +--- + +# Find: fresh-admission lane verified live on pod (round 15) — wrapper walked once, events emitted + +Parent: find-tier1-tail-starvation (the round-15 diagnosis this fix +answers). Evidence: ev-leaktag-onpod-round15.md (design + discarded +mark-vs-queue lesson), ev-leaktag-onpod-round15-results.md (live +verification). Build f7b1ea416, new pod prof-analyzer-hotdog-jb-d9d8cf-cgdtx +(the deploy REPLACED rz992 — always re-find the pod name after a deploy), +JVM from ~05:24Z 2026-09-16. + +## What the pod verified (all from kubectl logs -f streams; the log flood +is ~800k lines/min — the 10MB container cap rotates in SECONDS, --since +windows are useless for early-lap evidence) + +1. **The fresh lane works mechanically**: anchorTierHistogram shows + fresh_tier=1-11 kept per pass, fresh_queue=2-615 (one pass of sweep + admits, far under the 1024 cap), budget=32. The arithmetic gate + (fresh demand vs budget) passed. +2. **The wrapper admitted root-attached AND was walked** — the first + successful walk in the whole investigation: + `walkStaticFieldAnchors anchor tag=164392 + class=Ljava/util/Collections$SynchronizedRandomAccessList; klass_id=28366 + parent=0 root_kind=8 state=0 field_index=41`, holder-class diagnostic + `ProfileAnalyzer; field_index=41`, round-12 diag: 21 entries (ArrayList, + wrapper seen_as=2 already-admitted, Object[], then 17 [B chunks all + seen_as=0 = fresh plain admissions), edges=19 truncated=0 — the whole + subtree covered in one bounded call. +3. **The chunks were discovered, auto-marked, and got deep chains** + (chain_size=6-12, depth=5-6) — the leak-class [B instances have + multi-hop holder chains in the frontier. +4. **datadog.ReferenceChain events were emitted**: + drainPendingChainEvents re-emitted=4, then =6 — cached chains flowed + into JFR dump chunks. First events on this pod. + +## Why the end goal still isn't closed (chain continues in +find-wrapper-demotion-self-parent) + +Search #2 (the one that walked the wrapper) hit the frontier cap and +died; releaseSearchTags wiped the chunk entries; the chunks (retained, +tracked, tid-matched) got re-tagged with leak tags by the poll loop +(tagged grew 12→36→48 — they now INCLUDE the wrapper's chunks, waiting +for one interception). In later searches the wrapper's frontier entry is +DEMOTED (parent==its-own-tag, root_kind=0 — collector-invisible) and the +B' at-risk FIFO is cap-pinned by noise floods, so the wrapper never +re-walks: zero wrapper walk lines across 3 holder-class crossings. One +re-walk while the chunks hold leak tags converts everything +(interception → entry.leak_tag → correlated chains → canary resolves). + +## Reasoning chain (design level) + +Fresh-priority is the correct invariant for late-loaded holder classes: +admission order = sweep order = class order, so any holder whose class +loads late admits at the index tail and fair-only coverage needs +ceil(cohort/budget) passes — which must fit inside the MEASURED search +lifetime, not the designed one. The queue form (one first-look per +anchor) is order-immune to stale cross-search/cross-test state where the +mark form (positions-since-watermark) reorders selection — caught +in-test by StaticAnchorRotation* in order runs; see +find-test-seam-aliasing for the singleton-leakage lesson pattern. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-gate-bypass-representative-paths.md b/.investigations/missing-refchains-on-hotdog/nodes/find-gate-bypass-representative-paths.md new file mode 100644 index 0000000000..07ccf30cc9 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-gate-bypass-representative-paths.md @@ -0,0 +1,49 @@ +--- +id: find-gate-bypass-representative-paths +type: finding +status: confirmed-and-fixed +depends_on: [find-priority-queue-starves-bfs-crawl] +related: [find-priority-queue-starves-bfs-crawl, find-leak-tag-pool-implementation, q-resize-instrumentation-rescan-priority] +tags: [root-cause, fix, gate, noise, canary, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# Noise gate bypassed by canary/representative chain paths + depth-0 design flaw + +Caught by the local repro (LeakTagCorrelationReferenceChainTest) on its +first runs - the exact wrinkles the harness was built to find. + +## Defect 1: gate coverage + +The discovered-loop gate (suppress transient-rooted shallow chains) existed +ONLY in pollWatchedTargets' discovered-instances loop. Three other +cacheResolvedChain() sites in pollWatchedTargets bypassed it: +- buildCanaryChainEvent (dead-representative path) +- buildCanaryChainEvent (live canary marker path) +- buildChainEvent (normal representative path) + +A seeded noise-class representative got a stack-local-rooted depth-1 chain +cached via the canary path, and snapshot-and-keep re-emitted it in every +dump forever - the same "noise chains re-emitted forever" symptom observed +on the pod (the pod's 8 depth-1 noise chains were plausibly exactly this). + +Fix: file-local `suppressChainEvent()` predicate shared by ALL cache sites +(each also calls invalidateResolvedChain so cached noise stops re-emitting). + +## Defect 2: depth==0 unconditional suppression was wrong + +The round-4 gate suppressed depth==0 UNCONDITIONALLY. That drops real +direct-retention chains: a static field's value ITSELF is a depth-0 root +(the singleton collection), a `thread` root is the thread-local-leak +taxonomy. Corrected predicate: `depth < 2 && isTransientRootKind(root)` +only. Durable-rooted depth-0/1 chains are real; deeper chains pass +regardless. + +## Scenario lesson (fixture, not profiler) + +Holding the noise representative in a STATIC field makes that one instance +genuinely durably retained - the profiler is RIGHT to emit a +`NoisePayload@depth0:static_field` chain for it (observed). Transient +retention requires a non-static handoff (run()-local exchanger-style +holder, reference dropped after take). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-getobjectswithtags-quadratic-bottleneck.md b/.investigations/missing-refchains-on-hotdog/nodes/find-getobjectswithtags-quadratic-bottleneck.md new file mode 100644 index 0000000000..a3f48d96b3 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-getobjectswithtags-quadratic-bottleneck.md @@ -0,0 +1,183 @@ +--- +id: find-getobjectswithtags-quadratic-bottleneck +type: finding +status: fixed +depends_on: [ev-timing-split-callback-vs-jvmti, find-sweep-completes-but-bfs-starved] +supersedes: [] +related: [find-sweep-completes-but-bfs-starved] +tags: [root-cause, referenceChains, bfs, expandFrontier, GetObjectsWithTags, quadratic, throughput, NEW-THIS-SESSION] +created: 2026-08-26 +updated: 2026-08-27 +--- + +# GetObjectsWithTags is a quadratic O(tag_map × batch_size) bottleneck starving BFS + +## Observation + +`expandFrontier` admits only 3-4 edges/pass despite `remaining_budget=3440`. +Per-iteration timing diagnostics (commit `2d310e0fa`) show: + +| batch_size | gotw_ms | follow_ms | edges | truncated | +|------------|---------|-----------|-------|----------| +| 3440 | 226 | 6 | 1 | 1 | +| 272 | 40 | 1 | 0 | 0 | +| 3727 | 230 | 5 | 0 | 1 | +| 3452 | 226 | 5 | 1 | 1 | +| 272 | 35 | 1 | 0 | 0 | +| 3739 | 243 | 5 | 0 | 1 | +| 3439 | 242 | 6 | 0 | 1 | +| 272 | 40 | 2 | 1 | 0 | +| 1 | 14 | 0 | 2 | 0 | +| 1 | 11 | 0 | 2 | 0 | +| 3727 | 236 | 6 | 2 | 1 | + +`GetObjectsWithTags` takes 226-243ms when batch_size is ~3400-3700, but only +35-40ms when batch_size is 272, and 11-14ms when batch_size is 1. The cost +scales linearly with batch_size. + +## Root cause: O(tag_map_size × batch_size) in HotSpot + +`GetObjectsWithTags` (`jvmtiTagMap.cpp:1305`) calls `entry_iterate` which +visits every tagged object in the JVMTI tag map (163k entries on this pod). +For each entry, `TagObjectCollector::do_entry` (`:1249`) does an O(N) +linear scan of the `_tags` array: + +```cpp +bool do_entry(JvmtiTagMapKey& key, jlong& value) { + for (int i = 0; i < _tag_count; i++) { + if (_tags[i] == value) { ... collect ... } + } + return true; + } +``` + +So the total cost is O(tag_map_size × batch_size) = 163k × 3400 ≈ 554M +comparisons. That's the 226ms. This is a **quadratic blowup**: the tag +map is O(tag_map_size) and the search array is O(batch_size), nested loop. + +`FollowReferences` only gets 5-6ms (the leftover after GetObjectsWithTags +consumes the entire 200ms deadline), admits 0-1 edges, and truncates. +The BFS is starved by `GetObjectsWithTags`'s quadratic cost, not by +`FollowReferences` or our callback code. + +## Why batch_size is ~3400 + +`expandFrontier` caps batch_size at `min(source.size(), min(budget, _budget))` +(`:2680`). On this pod, `budget=3741` and `_budget=3741`, so batch_size +collapses to the entire `_pending_expand` backlog (163k entries) divided +into chunks of ~3741. Each chunk triggers one `GetObjectsWithTags` call +with ~3400 tags. + +## Fix direction: eliminate GetObjectsWithTags from the hot path + +### Constraint: no O(1) tag→object lookup in JVMTI + +`GetObjectsWithTags` is the **only** JVMTI API that reverse-looks-up +tags to objects. There is no `GetObjectByTag`. `GetTag(env, object, &tag)` +takes an object and returns its tag — the wrong direction. So any +fix must either (a) avoid the reverse lookup entirely, or (b) make +the lookup cheaper. + +### Cannot create JNI ref at admission time + +Inside `heapReferenceCallback`, we get `tag_ptr` (pointer to JVMTI's +tag slot) but NOT a `jobject`. The actual `oop` is inside +HotSpot's `CallbackWrapper` (`jvmtiTagMap.cpp:193`) and not +exposed to the callback. We cannot call `JNIHandles::make_local` or +`NewWeakGlobalRef` from inside the callback — we don't have the +object reference, only the tag pointer. So we cannot populate a +tag→jobject cache at admission time. + +### Can cache jweak after first GetObjectsWithTags resolution + +`expandFrontier` gets jobjects back from `GetObjectsWithTags` as JNI +local refs. We could promote them to weak global refs (`NewWeakGlobalRef`) +and cache them in a parallel array indexed by tag-1 (matching +FrontierTable's flat-array layout). On subsequent expansions (rotation +re-expansion of already-EXPANDED entries), check the jweak first — if +non-null, the object is alive, skip `GetObjectsWithTags` for that entry. + +This eliminates `GetObjectsWithTags` for rotation (256 entries/pass → +~0ms instead of ~17ms). But it does NOT help first-time `_pending_expand` +expansion — those entries have never been resolved before, so there's no +cached jweak. + +### First-time expansion still needs GetObjectsWithTags + +For `_pending_expand` entries (first-time expansion), we have no way to +get the jobject without `GetObjectsWithTags`. The only mitigation is +reducing batch_size to keep the O(tag_map × batch) cost bounded. + +Adaptive batch_size: if we want `GetObjectsWithTags` to take ≤5ms, and +the tag map is 163k entries, then batch_size ≤ 5ms / (163k × ~1.4ns) ≈ 22. +That's tiny — each `FollowReferences` would only expand 22 entries, and +draining 163k entries would need ~7400 iterations (safepoints). + +### Self-calibrating adaptive batch_size (approved, implementing) + +User rejected a fixed per-comparison cost estimate (varies by CPU, +cache, memory bandwidth). Instead: self-calibrate from measured +`GetObjectsWithTags` elapsed time. + +``` +// After each GetObjectsWithTags call: +measured_cost_per_tag = gotw_elapsed_ns / batch_size +// Before the next call: +batch_size = max(1, cpu_budget_ns / ema_cost_per_tag) +// Exponential moving average to smooth jitter: +ema_cost_per_tag = ema * 0.8 + measured * 0.2 +``` + +Properties: +- No machine-specific constants — adapts to whatever CPU it's on +- Adapts as tag map grows (batch_size shrinks automatically) +- Only knob is `cpu_budget_ns` (policy choice, not machine estimate) +- First call: conservative small default (e.g. 64), then adapt +- `cpu_budget_ns` target: ~25ms per pass (CPU overhead, not safepoint) + +Not yet implemented — implementing now. + +## Fix: self-calibrating adaptive batch_size (COMMITTED, DEPLOYED, CONFIRMED LIVE ON-POD) + +Commit `337b4c21d` + verification log `8f69683f9`. Deployed on pod +`prof-analyzer-hotdog-jb-c944876b9-q8vd8`, JVM PID 263646. + +### Implementation + +- `_gotw_ema_cost_per_tag_ns` (EMA of cost-per-tag) on the tracker +- After each `GetObjectsWithTags` call: `measured = gotw_elapsed_ns / batch_size`, + `ema = ema * 4/5 + measured / 5` +- Before each call: `batch_size = max(1, GOTW_CPU_BUDGET_NS / ema)` +- `GOTW_CPU_BUDGET_NS = 25ms` (CPU overhead target, NOT safepoint) +- `GOTW_INITIAL_BATCH_SIZE = 64` (conservative first call) +- Still capped at `min(budget, _budget)` for JNI safety +- Also reverted temp 200ms deadline override + removed temp timing diagnostics + +### On-pod results (2026-08-27) + +| Phase | batch_size | gotw_ms | edges/call | edges/pass | +|-------|-----------|---------|------------|------------| +| Before fix | ~3400 | 226-243 | 0-1 | 3-4 | +| After fix (warmup) | 1-64 | 2-7 | 3-7 | — | +| After fix (converged) | ~390-420 | 20-35 | 0-503 | 456-971 | + +EMA converged to ~62k ns/tag. BFS throughput up **~100-200x** (456-971 +edges/pass vs 3-4 before). `GetObjectsWithTags` bounded at 20-35ms +(target was 25ms). + +### Remaining + +0 candidates at time of verification because JVM restarted and liveness +tracker needs warmup (`heapFloorRising=0`, `required_hysteresis=5`). +Monitoring for candidates to appear once enough GC generations pass. + +## Semantics clarification (this session, jdk21 source + pod) + +GetObjectsWithTags is NOT a VM op (jvmtiTagMap.cpp:1305): it scans the +whole tag map under the tag-map mutex, O(tag-map size) per call +regardless of tags requested - it blocks other tag users but does not +stop the world. Measured on hotdog JVM 75258: 18-21ms/call at ~200k +entries, ~5 calls/pass at ~88 passes/min = ~100ms of tag-map-locked +scanning per pass - the dominant per-pass cost, vs the walk VM op's +bounded ~15ms STW. See find-jvmti-heap-walk-stw-vmop for the full +pause-vs-lock picture. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-holistic-design-issues.md b/.investigations/missing-refchains-on-hotdog/nodes/find-holistic-design-issues.md new file mode 100644 index 0000000000..1731432692 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-holistic-design-issues.md @@ -0,0 +1,85 @@ +--- +id: find-holistic-design-issues +type: finding +status: confirmed +depends_on: [find-already-admitted-blocks-deeper-chain, find-per-class-caching-blocks-instances] +supersedes: [] +related: [find-already-admitted-blocks-deeper-chain, find-per-class-caching-blocks-instances, find-cpu-pain-budget-blocks-bfs] +tags: [root-cause, design, referenceChains, jni-local, targetTag, correlation, cpu-budget, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# Holistic design issues in reference chain pipeline + +## Four problems identified from JFR recording +~/Downloads/20260831-092433_prof-analyzer-hotdog_AaBXI2uGAACSSh31cW40WAAA/main.jfr + +### 1. Leaking [B not in the chains + +The 27 leaking [B instances (78MB each, allocated by +`lambda$static$1` in `ProfileAnalyzer`) are NOT among the 40 +ReferenceChain events. The chains are for OTHER [B instances: +- 6 depth=13 jni_global chains (some other [B reached via deep path) +- 2 depth=0 static_field chains (direct static field refs) +- 32 depth=1 jni_local/stack_local chains (noise) + +The leaking [B instances are held via static field → ArrayList → [B, +but the BFS admits them as JNI-local roots (depth=0) before the +static field sweep reaches them. The `improveChain` fix should help +but never fires (0 invalidate logs) — likely because the BFS doesn't +revisit already-tagged objects' incoming edges in the same pass. + +### 2. More chains than surviving live heap objects + +40 ReferenceChain events vs 36 HeapLiveObject events. Chains are +cached and re-emitted across dump cycles even after the original +object has been GC'd. The cache (`_resolved_chains`) is never +invalidated when objects die — it's only cleared on search restart. + +### 3. No correlation between HeapLiveObject and ReferenceChain + +HeapLiveObject events have no tag/ID that matches ReferenceChain's +`targetTag`. The `targetTag` is a JVMTI tag (frontier table index), +while HeapLiveObject events are identified by (klass, tid, age, size). +There is no shared identifier to join them. + +### 4. 45% CPU consumption from continuous heap walking + +The 100× canary pain budget multiplier makes the BFS run +continuously, eating 45% CPU. The multiplier was set to 100 to +overcome cpu_pain_budget blocking, but it's way too high for +production. Need to find the right balance — enough to make progress +but not eat the CPU. + +## Root cause + +The fundamental issue is that the BFS and the liveness tracker operate +on different object sets with no correlation: + +- **Liveness tracker** subsamples allocations (10%), tracks survivors + by (klass_id, tid, age), and selects leak candidates by ring-buffer + growth trend. It tags representatives with marker tags. +- **BFS** walks the entire heap from roots, admits objects into the + frontier, and builds chains from frontier entries. It discovers + [B instances via auto-mark (class match) but those may be different + instances than the liveness tracker's representatives. + +The leaking [B (78MB, from `lambda$static$1`) is in the liveness +tracker's table (we see it in HeapLiveObject events), but the BFS +finds OTHER [B instances first (JNI locals, small buffers from other +threads) and caches their chains. The leaking [B is never reached +because: +1. It's admitted as a JNI-local root (depth=0) before the static + field sweep reaches it +2. `improveChain` doesn't fire because the BFS doesn't revisit + already-tagged objects' incoming edges +3. Even if it did, the cached chain (depth=1) blocks rebuilding + +## Proposed approach + +Need a design-level rethink, not whack-a-mole fixes. Key questions: +1. How to correlate ReferenceChain targetTag with HeapLiveObject events +2. How to invalidate cached chains when objects die +3. How to prioritize the BFS to reach leaking instances before noise +4. How to bound CPU consumption while still making progress diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-hotdog-deploy-last-mile.md b/.investigations/missing-refchains-on-hotdog/nodes/find-hotdog-deploy-last-mile.md new file mode 100644 index 0000000000..366effe580 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-hotdog-deploy-last-mile.md @@ -0,0 +1,50 @@ +--- +id: find-hotdog-deploy-last-mile +type: finding +status: confirmed +depends_on: [find-refchains-not-deployed] +supersedes: [] +related: [ev-post-resync-deployment-verified, find-onpod-evidence-methodology] +tags: [deployment, tooling, hotdog, patch-dd-java-agent] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Getting a branch build of ddprof-lib onto a hotdog pod is not scripted in this repo + +## Reasoning chain + +Having established the feature simply wasn't deployed, the next question +was how it gets deployed. The pieces exist but the last mile does not +live in java-profiler: + +1. `./gradlew :ddprof-lib:jar` +2. `utils/patch-dd-java-agent.sh` (`DD_AGENT_JAR=… DDPROF_JAR=…`) produces + `dd-java-agent-patched.jar`. dd-trace-java does **not** need rebuilding; + this is exactly what `.gitlab/dd-trace-integration/.gitlab-ci.yml:57-61` + does for integration tests. +3. Getting the patched jar into the pod (overwrite + `/usr/local/app/agent/dd-java-agent.jar` and restart the JVM, or rebake + the hotdog image) — this step belongs to the prof-analyzer/hotdog + service repo, not here. +4. Enabling it. `referencechains=true` is a native ddprof option string + (`ddprof-lib/src/main/cpp/arguments.cpp:85`, parsed at `:468-490`); the + dd-trace-java property that plumbs it is + `-Ddd.profiling.experimental.ddprof.referencechains.enabled=true`, + which was absent pre-resync and present post-resync. + +The user performed steps 3-4 manually ("I resynced and reuploaded the +agent"): the jar was replaced in place at 14:27 (md5 +`3b204607ab88b99e18e577d23368ca45`) and the JVM restarted at 14:29 as PID +20807 — the pod itself was never rescheduled (still 44 h old). + +## Evidence +- `evidence/ev-post-resync-deployment-verified.md` +- `utils/patch-dd-java-agent.sh`, `.gitlab/dd-trace-integration/.gitlab-ci.yml:57-61` +- `ddprof-lib/src/main/cpp/arguments.cpp:85,468-490` + +## What this rules out +- Expecting a `hotdog-deploy` script in this repo. A background search of + doc/, scripts/, Makefiles and CI yaml was run for one; the conclusion + reflected in the session was that no such end-to-end path is scripted + here. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-isqueuedforrotation-quad-scan.md b/.investigations/missing-refchains-on-hotdog/nodes/find-isqueuedforrotation-quad-scan.md new file mode 100644 index 0000000000..5b85d5f4f3 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-isqueuedforrotation-quad-scan.md @@ -0,0 +1,36 @@ +--- +id: find-isqueuedforrotation-quad-scan +type: finding +status: fixed +depends_on: [find-priority-queue-starves-bfs-crawl, find-rotation-resize-blindspot] +related: [find-cpu-pain-budget-blocks-bfs] +tags: [fix, root-cause, performance, profiles, NEW-THIS-SESSION] +created: 20260901 +updated: 20260901 +--- + +# isQueuedForRotation() was O(N) per frontier-slot visit - ~200M comparisons per rotation pass + +User flagged ReferenceChainTracker::isQueuedForRotation() showing +prominently in profiles. Root cause: the deque-linear-scan implementation +was documented as "sub-millisecond" with arithmetic that assumed ~256 +per-SELECTION calls per pass - but the rotation collectors +(collectStaleRootKindEntriesForRotation / +collectStaleExpandedEntriesForRotation, referenceChains.cpp) call it for +EVERY FrontierTable slot they visit (~199k EXPANDED entries on a large +heap), each scanning up to PRIORITY_EXPAND_CAP (1024) deque entries: +~200M comparisons per rotation pass. The comment's assumption and the +call pattern disagreed; the call pattern wins. + +Fix: PriorityExpandSet - fixed 2048-slot open-addressing membership index +(Fibonacci-hashed near-sequential frontier tags, linear probing, <=0.5 +load), zero allocation after construction, maintained at every +_priority_expand mutation site (pushes insert, expandFrontier() drains +rebuild-from-deque at batch end - a rebuild is <=1024 inserts against the +~20ms GetObjectsWithTags call the same batch already paid; no tombstones +needed). isQueuedForRotation() is now O(1). All mutation sites are on the +engine thread under _engine_lock, so plain non-atomic access is safe; +gtest helpers that manipulate the deque directly keep the index in sync. + +Verified: 543 gtests green (referenceChains_ut 101/101), slow suite 8/8, +testDebug failures unchanged (5, all pre-existing env-flaky subsystems). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-jvmti-heap-walk-stw-vmop.md b/.investigations/missing-refchains-on-hotdog/nodes/find-jvmti-heap-walk-stw-vmop.md new file mode 100644 index 0000000000..1aa0dce9a4 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-jvmti-heap-walk-stw-vmop.md @@ -0,0 +1,51 @@ +--- +id: find-jvmti-heap-walk-stw-vmop +type: finding +status: confirmed +depends_on: [] +related: [find-getobjectswithtags-quadratic-bottleneck, find-cpu-pain-budget-blocks-bfs, find-sweep-completes-but-bfs-starved] +tags: [root-cause, hotspot, jvmti, stw, design-constraint, NEW-THIS-SESSION] +created: 20260901 +updated: 20260901 +--- + +# JVMTI FollowReferences is a VM operation - a full-heap walk is ONE stop-the-world pause + +From the openjdk-jdk21 source (user challenged the option-B sweep-lap +cost question; this killed option B): + +- `FollowReferences`/`IterateOverReachableObjects` = + `VM_HeapWalkOperation : public VM_Operation` + (src/hotspot/share/prims/jvmtiTagMap.cpp:2379), submitted via + `VMThread::execute`; the whole graph walk runs inside `doit()` + (`:2934`, `while (!visit_stack()->is_empty()) visit(o)`). + VM operations execute at a safepoint: the ENTIRE JVM is frozen for the + walk's duration. Neither heap-walk op overrides + `allow_nested_safepoints()`. There is no preemption, no yielding. + +- Consequence: a full-graph intercept sweep on the hotdog analyzer's 7GB + heap = a single ~10s stop-the-world pause per lap. (The 10s figure is + the pod's own pre-deadline-bound pass cost, round-2 evidence.) On a + traffic-serving analyzer this is categorically unacceptable - option B + (intercept-only full-graph sweep) is DEAD as designed. + +- What the existing design does instead is correct discipline: every + walk is a VM op, so every pass caps its walk at ~15ms = ~15ms STW per + pass, and the static sweep is a RESTRICTED walk (class-holder arrays -> + classes -> static field values, a tiny subgraph) - ms-scale STW per + chunk. The static sweep "completing laps" is therefore NOT evidence that + full-graph walks are affordable; it only proves restricted subgraph + walks are. + +- Corollary for the coverage lottery: there is NO cheap "reach unknown + holders of tagged objects" mechanism available to an external JVMTI + agent. JFR's leak profiler avoids this by living inside HotSpot and + riding the GC's own object traversal (leakprofiler closures during GC), + which ddprof cannot do from outside. + +- Distinct case: `GetObjectsWithTags` is NOT a VM op + (jvmtiTagMap.cpp:1305, `get_objects_with_tags`): it scans the WHOLE tag + map under the tag-map mutex per call - O(tag-map size) regardless of + the number of tags requested. Measured on hotdog: 18-21ms/call at ~200k + tag-map entries, ~5 calls/pass. It blocks other tag users rather than + stopping the world, but it is the DOMINANT per-pass CPU/lock cost. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-klass-id-notation-mismatch.md b/.investigations/missing-refchains-on-hotdog/nodes/find-klass-id-notation-mismatch.md new file mode 100644 index 0000000000..58eb5d135d --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-klass-id-notation-mismatch.md @@ -0,0 +1,59 @@ +--- +id: find-klass-id-notation-mismatch +type: finding +status: confirmed-and-fixed +depends_on: [find-leak-tag-pool-implementation] +related: [find-leak-tag-pool-implementation, find-priority-queue-starves-bfs-crawl] +tags: [root-cause, fix, klass-id, PRODUCTION-BUG, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# LT vs RCT klass-id notation mismatch (production bug, broke ALL discovered matching) + +## Root cause (proven from code) + +Two resolution sequences for the same class produce TWO different +StringDictionary ids: +- LivenessTracker::resolveKlassId(): `Class.getName()` -> "com.foo.Bar" + (DOT notation) -> Profiler::lookupClass() +- ReferenceChainTracker::resolveClassMap(): `GetClassSignature` -> + normalizeClassSignature -> "com/foo/Bar" (SLASH notation, L/; stripped, + slashes kept) -> same Profiler::lookupClass() + +StringDictionary keys by the exact string, so the same class has two ids: +locally observed LivenessTracker id 63 vs ReferenceChainTracker id 2 for +the same class. Only ARRAY classes matched accidentally ("[B" is +notation-identical in both) - which is why the correlation child's byte[] +candidate matching appeared to work while ChainLink/CachedPayload never +matched. + +## Production impact (the pod was exhibiting this all along) + +Every auto-mark on the pod logged +`resolved but no candidate match (candidates=[4,139,0,0,0])` - the pod's +discovered-instance chain path NEVER matched anything. The chains that did +appear (8 noise + canary rep chains) came via the representative/marker +path, which needs no klass-id matching. Every round of pod debugging that +hypothesized "candidates not matching discovered instances" was seeing +exactly this bug. + +## Fix + +LivenessTracker::resolveKlassId() now uses the same +GetClassSignature + normalizeClassSignature + lookupClass sequence as +ObjectSampler::recordAllocation() and RCT's resolveClassMap() - the third +user of it. Also strictly cheaper (plain JVMTI call vs Class.getName() +JNI upcall that could allocate). flush_table()'s getName()-based +resolution for JFR event class names is deliberately unchanged (dot-form +rendering preserved). + +## Third dot-form user found this session + +`ReferenceChainTracker::tagAsRootForTest()` resolved klass ids via +`Class.getName()` too - the seam's entries were in dot space while the +aliased population entries (candidates) were in signature space, so no +candidate could ever match the seam's frontier entries. Same fix applied +(GetClassSignature + normalizeClassSignature + lookupClass). The mock +gtest fixture needed no change (its GetClassSignature mock already feeds +signature-form names). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-lambda-fragments-calltrace-id.md b/.investigations/missing-refchains-on-hotdog/nodes/find-lambda-fragments-calltrace-id.md new file mode 100644 index 0000000000..e051ebcafc --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-lambda-fragments-calltrace-id.md @@ -0,0 +1,62 @@ +--- +id: find-lambda-fragments-calltrace-id +type: finding +status: confirmed +depends_on: [q-allocation-site-selection, find-age-heuristic-insufficient] +supersedes: [] +related: [q-allocation-site-selection] +tags: [root-cause, referenceChains, livenessTracker, allocation-site, call-trace-id, lambda, tid, NEW-THIS-SESSION] +created: 2026-08-28 +updated: 2026-08-28 +--- + +# Lambda allocation fragments call_trace_id, breaking site clustering + +## Context + +Allocation-site clustering (commit `0b492612b`) keyed per-site tracking +on `call_trace_id` — the full stack trace hash at allocation time. + +## Problem + +On-pod verification showed `dominant_gens=1` for every class every +epoch, even klass_id=5 with `gen_count=25`. Root cause: lambdas. + +Lambda allocation sites produce synthetic methods whose call stacks +vary slightly across instances — different callers above the lambda, +JIT inlining changes, etc. The full-stack hash (`call_trace_id`) +fragments what is logically one allocation site into many distinct +IDs. 12 leaking [B instances from the same logical site get 12 +different `call_trace_id`s → 12 sites × 1 generation each → no +clustering signal. + +## Evidence + +Pod logs (PID 51584, build `0b492612b`): +``` +foldKlassCountsLocked minted=1 for klass_id=371 dominant_site=30064777208 dominant_gens=1 +foldKlassCountsLocked minted=1 for klass_id=278 dominant_site=30064778118 dominant_gens=1 +foldKlassCountsLocked minted=1 for klass_id=302 dominant_site=30064778443 dominant_gens=1 +``` +All `dominant_gens=1` despite high per-class `gen_count` (15-25). + +## Fix + +Switch clustering key from `call_trace_id` to `tid` (thread ID). + +Rationale: +- **Stable**: thread ID doesn't vary with stack trace jitter +- **Naturally separates leak from noise**: leaking [B all from + tid=172 (`simulated-memory-leak`), noise [B from tid=284 (`s3-netty-2`) +- **Matches leak taxonomy**: static-field leaks typically allocated by + one thread; thread-local leaks ARE the thread +- **Already in TrackingEntry**, no new data plumbing +- **Multi-thread shared leak**: all contributing tids have similar + generation counts → pick any → still a leaking instance → correct +- **Noise thread with many short-lived instances**: few surviving + generations → loses to leak tids + +The only failure mode: one tid has many long-lived noise instances and +another has fewer long-lived leak instances with the same generation +count. But per-instance chain caching (cd68be618) emits chains for all +of them — the backend aggregates by class. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-leak-tag-pool-implementation.md b/.investigations/missing-refchains-on-hotdog/nodes/find-leak-tag-pool-implementation.md new file mode 100644 index 0000000000..3d4b6bd514 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-leak-tag-pool-implementation.md @@ -0,0 +1,107 @@ +--- +id: find-leak-tag-pool-implementation +type: finding +status: implemented-round2-unverified +depends_on: [find-holistic-design-issues] +supersedes: [] +related: [find-holistic-design-issues, find-marker-tag-slot-index-mismatch, find-candidate1-never-tagged, q-coverage-tracking-per-combination] +tags: [fix, design, referenceChains, livenessTracker, leak-tag, correlation, adaptive-cpu, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# Leak tag pool implementation (design A-D) + +Implements the approved design addressing all four problems from +find-holistic-design-issues. Commits: `294f09ff3`, `0b05d5f85` (TEMP), +`8e3f7584a`, `1ce2b4f03`. + +## A) Direct leak tagging (replaces marker tags) + +- `LivenessTracker` owns a pool of 256 tags in range + `[LEAK_TAG_BASE=0x40000000, LEAK_TAG_BASE+256)`, free-list + + `LeakTagInfo {call_trace_id, tid}` side table per tag. +- `tagLeakInstances(jvmti, klass_ids, count)` scans `_table`, matches + `cached_klass_id` against candidate klass_ids, acquires a tag, + `SetTag`s every tracked instance of candidate classes. +- Tags returned to pool in `cleanup_table` when jweak is dead + (reusable pool — user requirement, running out unlikely). +- `pollWatchedTargets` no longer mints marker tags + (`MARKER_TAG_BASE - slot`) on representatives. Candidate slots + still track klass_ids (`_candidate_tags[slot] = 0`), then + `tagLeakInstances` tags ALL tracked instances. +- `heapReferenceCallback` intercepts leak tags (`isLeakTag(*tag_ptr)`) + BEFORE the `*tag_ptr == 0` check: converts to frontier tag via + `nextTag()`, inserts into frontier storing `leak_tag` in + `FrontierEntry`, replaces `*tag_ptr`, auto-marks for candidate + bookkeeping. +- Old marker-tag decode in heapReferenceCallback still exists + (`tag <= MARKER_TAG_BASE`) but is now unused — may be dead code. + +## B) Chain filtering + +- Depth==0 discovered-instance chains (no holder path) are filtered + in the pollWatchedTargets discovered loop (`event._depth == 0`). +- IMPORTANT placement lesson: the filter must NOT live in + `buildChainEvent` itself — gtests (`EmitsEventForAlreadyDiscoveredCandidate`, + `NoDuplicateOnRepeatPoll`, `ChainPersistsAfterRepresentativeDies`) + and the canary path legitimately build depth==0 chains. Moving the + filter into buildChainEvent broke 3 tests; fixed by filtering only + in the discovered-instance path. + +## C) Correlation via leak tag + +- `ReferenceChainEvent.targetTag` = `FrontierEntry::leak_tag` when + nonzero, else frontier tag (buildChainEvent). +- `ObjectLivenessEvent` gained `int64_t leak_tag` (NOTE: `jlong` does + NOT work in event.h — no jni.h in TUs that include it; commit + `8e3f7584a`), `flush_table` copies it, `recordHeapLiveObject` + writes it, `datadog.HeapLiveObject` JFR metadata has `leakTag` + (T_LONG, F_UNSIGNED). +- Backend joins ReferenceChain.targetTag == HeapLiveObject.leakTag. + +## D) Adaptive CPU budget + +- Multipliers: 100x emergency (canary_active AND + `_passes_since_last_progress >= CANARY_NO_PROGRESS_PASS_LIMIT`), + 15x uncovered-but-progressing, 1x all covered. +- Coverage: `_leak_tags_assigned` (set from tagLeakInstances return), + `_leak_tags_resolved` (incremented when chain cached with + target_tag >= LEAK_TAG_BASE). Both reset in restartSearch and the + second reset path. +- TEMP (commit `0b05d5f85`): `CANARY_NO_PROGRESS_PASS_LIMIT` 30 → 3 + for faster testing. **MUST REVERT before finalizing.** + +## Lessons + +- event.h is included from TUs without jni.h — JNI types unusable + there (`jlong` → `int64_t`). +- `INT_MAX` needs `` in referenceChains.h. +- Edit-tool brace accounting: inserting a block inside an `if` produced + an extraneous `}` that closed the function early — caught only by + the remote Linux build (macOS-built fine earlier because the bad + block was inside a `#if 0`-style path? No — actually the compile + errors were in the pushed commit that had NOT been built locally; + lesson: always build locally before pushing, even "trivial" edits). + +## Round 2 fixes (commit 0db70994d) — from on-pod round 1 failures + +- HeapLiveObject.leakTag field-order alignment + (find-leaktag-jfr-field-misalignment). +- AIMD batch sizing + per-iteration deadline check in expandFrontier + (find-ema-batch-collapse). +- tagLeakInstances tags by per-tid age-diversity priority (user-directed: + "the leaking ones have the clearest surviving age diversity, they + should be tagged first"). +- Emergency keys on candidate progress, marker re-tag removed, + interception TEST_LOG added. +- Tests: 3 pool tests (livenessTracker_ut), AIMD decrease/increase, + leak-tag interception + correlation (referenceChains_ut). All pass. + +## Unverified on-pod + +Everything awaits hotdog redeploy verification: +- chains emitted only for actual leaking 78MB [B objects +- targetTag == HeapLiveObject.leakTag correlation works +- no depth==0 noise +- adaptive multiplier drops to 1x once covered diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-leaktag-jfr-field-misalignment.md b/.investigations/missing-refchains-on-hotdog/nodes/find-leaktag-jfr-field-misalignment.md new file mode 100644 index 0000000000..916938437b --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-leaktag-jfr-field-misalignment.md @@ -0,0 +1,48 @@ +--- +id: find-leaktag-jfr-field-misalignment +type: finding +status: confirmed +depends_on: [find-leak-tag-pool-implementation] +supersedes: [] +related: [find-leak-tag-pool-implementation] +tags: [root-cause, fix, jfr, jfrMetadata, flightRecorder, leak-tag, contextAttributes, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# HeapLiveObject.leakTag parsed from the wrong bytes (context-attribute shift) + +## Symptom (recording 20260831-103955) + +User: "none of the actually leaking tracked instance has a tag" — all +78MB [B rows showed leakTag=0, two 32-byte rows showed leakTag=75/50. + +## Root cause + +Metadata declared `leakTag` BEFORE `|| contextAttributes` +(jfrMetadata.cpp), but the binary write put `leak_tag` AFTER +`writeContextSnapshot()` — which writes spanId, rootSpanId, THEN the N +context-attribute values (flightRecorder.cpp:1918). Parsers therefore +read leakTag from the first context-attribute byte: rows with a tracing +context show an attribute encoding (75/50), rows without show 0. The +recording says NOTHING about the actual table state — the pod logs do +(`tagLeakInstances tagged=28`, rep JVMTI tag read back 0x4000000F). + +## Invariant (for any future per-event field) + +writeContextSnapshot() emits spanId, rootSpanId, then numContextAttributes() +attribute values. A per-event field must be declared on the SAME side of +`|| contextAttributes` in the metadata as its write is on the opposite +side of writeContextSnapshot() in the binary — i.e. both before, or both +after. + +## Fix (commit 0db70994d) + +leakTag declared between `weight` and `spanId`; written after `weight`, +before writeContextSnapshot(). Binary and metadata now both read: +`... weight, leakTag, spanId, localRootSpanId, [attrs]`. + +## Verification pending + +Redeploy → 28 tagged rows (incl. all 78MB [B) should show leakTag in +[0x40000000, 0x40000100). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-marker-tag-slot-index-mismatch.md b/.investigations/missing-refchains-on-hotdog/nodes/find-marker-tag-slot-index-mismatch.md new file mode 100644 index 0000000000..a9fcace37f --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-marker-tag-slot-index-mismatch.md @@ -0,0 +1,113 @@ +--- +id: find-marker-tag-slot-index-mismatch +type: finding +status: confirmed +depends_on: [ev-post-resync-deployment-verified, ev-jafar-zero-refchain-events, ev-livelock-pod-logs, ev-marker-tag-arithmetic, ev-source-poll-vs-callback] +supersedes: [] +related: [find-one-shot-pretag-gate, find-canary-search-cannot-terminate, find-canary-stuck-abandon-detector, hyp-warmup-transience, hyp-regression-of-five-fixes, q-implement-two-fixes, ev-fixes-compile-and-gtest-pass] +tags: [root-cause, fixed, referenceChains, canary, marker-tag, off-by-slot] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# ROOT CAUSE (FIXED): pollWatchedTargets() indexes candidate arrays by loop position instead of the tag-decoded slot + +## Reasoning chain + +After the resync the feature is provably deployed and enabled +(1097 symbol hits, `-Ddd.profiling.experimental.ddprof.referencechains.enabled=true`, +both event types registered in the JFR) yet jafar reports **count 0** for +`datadog.ReferenceChain` and `datadog.ReferenceChainAbandoned`. + +The pod logs show why, and the two lines are *internally inconsistent*: + +``` +canary candidate[0] klass_id=8 marker_tag=-4611686018427387905 needRefresh=1 +buildCanaryChainEvent(candidate=0) -> 0 +``` + +- `MARKER_TAG_BASE = -(1LL << 62) = -4611686018427387904` + (`ddprof-lib/src/main/cpp/referenceChains.h:1924`). +- The observed tag is `MARKER_TAG_BASE - 1`, i.e. candidate slot **1**. +- But the log — and the call — use `i`, the position in the *current* + `candidates[]` array returned by `selectLeakCandidates()`, which is 0. + +The write side is correct: `heapReferenceCallback()` decodes the slot out +of the tag (`referenceChains.cpp:1510`, +`int candidate_idx = (int)(MARKER_TAG_BASE - *tag_ptr);`) and writes +`_candidate_parent_tags` / `_candidate_frontier_tags` / +`_candidate_referrer_klasses` / `_candidate_depths` / `_candidate_found_bits` +at that decoded index (`:1525-1529`). + +The read side is not: `pollWatchedTargets()` calls +`buildCanaryChainEvent(i, &event)` (`referenceChains.cpp:3478`) and later +`cacheResolvedChain(..., _candidate_frontier_tags[i], ...)` +(`referenceChains.cpp:3485`). + +Consequence chain: + +1. Slot 0 was never written by the pruner, so both `parent_tag` and + `frontier_tag` read as 0. +2. `buildCanaryChainEvent()` falls through to + `return false; // never pruned (candidate not reached)` + (`referenceChains.h:2127`). +3. `built == false`, so `cacheResolvedChain()` is never invoked + (`referenceChains.cpp:3482-3487`). +4. Hence no `_resolved_chains` entry for `klass_id`, hence + `need_refresh` (`referenceChains.cpp:3469-3471`) is **1 forever** — + which is a symptom, not an independent fault. +5. No chain is ever cached, so nothing is ever queued for the JFR writer: + zero events. + +Corroborating: `shouldRunPass -> true (canary search, 0/3 candidates found)` +means `_candidate_found_bits == 0`, i.e. **no** slot — not even slot 1 — +was ever written by the pruning code. So the walk hadn't reached the +representative either; but even if it had, the reader would still have +looked at the wrong slot. + +## Evidence +- `evidence/ev-marker-tag-arithmetic.md` — the arithmetic, run in-session +- `evidence/ev-source-poll-vs-callback.md` — verified source excerpts +- `evidence/ev-livelock-pod-logs.md` — 25 min of identical iterations +- `evidence/ev-jafar-zero-refchain-events.md` — the observable outcome +- `ddprof-lib/src/main/cpp/referenceChains.h:1924`, `:2093-2141` (`:2127`) +- `ddprof-lib/src/main/cpp/referenceChains.cpp:1508-1529`, `:3456-3490` + (`:3478`, `:3485`) + +## What this rules out +- Warm-up / not-enough-time — see `hyp-warmup-transience` (REFUTED: + `consecutive_positive=11 >= required=3`, 143 identical iterations over + 25 min, `0/3` never moving). +- Regression of the five previously-fixed causes — see + `hyp-regression-of-five-fixes` (REFUTED: none of them touch + slot↔index correspondence). +- Frontier/budget starvation: `runPass done: err=0 edges_admitted=184 … + frontierSize=10211 effectiveBudget=3741` — the walk is running fine. +- A JFR-writer or upload problem: the types ARE registered in the + uploaded recording; the events simply are never produced. + +## Fix implemented (this session, uncommitted on `jb/reference-chains`) + +Applied in `pollWatchedTargets()` (`referenceChains.cpp`, canary branch +under `if (tag <= MARKER_TAG_BASE)`): decode `candidate_slot = (int)(MARKER_TAG_BASE - tag)`, +bounds-check it against `_candidate_count` (out-of-range slot -> `TEST_LOG` ++ `continue`, matches the existing OOB-safe behavior of +`buildCanaryChainEvent()`), and use `candidate_slot` (not the loop index +`i`) for both `buildCanaryChainEvent(candidate_slot, &event)` and +`_candidate_frontier_tags[candidate_slot]`. This is the "Fix A" design from +`q-implement-two-fixes`, implemented as proposed with an added explicit +out-of-range guard. + +Verified: `./gradlew :ddprof-lib:compileDebug -Pskip-gtest` succeeds; +`gtestDebug_referenceChains_ut` (89 tests) and +`gtestDebug_referenceChainJfrRoundtrip_ut` (1 test) both pass — see +`ev-fixes-compile-and-gtest-pass`. + +Implemented together with `find-one-shot-pretag-gate`'s fix (Fix B) and a +new `find-canary-stuck-abandon-detector` (Fix C), because Fix A alone does +not stop the search livelocking once the candidate set churns (see that +finding for why they had to land together). + +Not yet done: on-pod re-verification (the pod's `.so` still has the +pre-fix binary); a multi-candidate regression test, per the sub-question in +`q-implement-two-fixes`, has not been added. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-one-shot-pretag-gate.md b/.investigations/missing-refchains-on-hotdog/nodes/find-one-shot-pretag-gate.md new file mode 100644 index 0000000000..2f67204803 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-one-shot-pretag-gate.md @@ -0,0 +1,97 @@ +--- +id: find-one-shot-pretag-gate +type: finding +status: confirmed +depends_on: [ev-candidate-count-latch-mismatch, ev-source-poll-vs-callback] +supersedes: [] +related: [find-marker-tag-slot-index-mismatch, find-canary-search-cannot-terminate, find-canary-stuck-abandon-detector, q-implement-two-fixes, ev-fixes-compile-and-gtest-pass] +tags: [self-heal, pre-tagging, candidate-count, latch, fixed] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Second defect (FIXED): candidate pre-tagging was one-shot, so the tag→slot map desynchronised from selectLeakCandidates() + +## Reasoning chain + +Even if the slot/index mismatch were a transient, the design cannot +recover from it, because marker tags are assigned exactly once: + +```cpp +if (_candidate_count == 0) { // referenceChains.cpp:3375 + _candidate_count = candidate_count; + _candidate_found_bits = 0; + for (int i = 0; i < candidate_count; i++) { + jlong tag = MARKER_TAG_BASE - i; + _candidate_tags[i] = tag; + … jvmti->SetTag(obj, tag); … + } +} +``` + +`_candidate_count` is only reset when `runPass()` leaves `RUNNING`. So +once latched, later `selectLeakCandidates()` orderings and counts are +never reflected in the tags — the loop index `i` in `pollWatchedTargets()` +refers to the *current* selection while the tags encode the *original* +one. The two only agree by accident. + +Live proof of the divergence on the hotdog pod: + +- `pollWatchedTargets candidate_count=5` every poll (current selection), +- `shouldRunPass … 0/3 candidates found` (latched `_candidate_count == 3`), +- exactly one representative still carrying a marker tag, +- `runPassManualWalk … watched_leak_klass_count=5`, +- no `"candidates pre-tagged with marker tags"` line in a 40-minute window. + +So the candidate set legitimately churned (grew 3 → 5 as more klasses +passed the hysteresis threshold) and the tagging state never followed. + +## Evidence +- `evidence/ev-candidate-count-latch-mismatch.md` +- `evidence/ev-source-poll-vs-callback.md` +- `ddprof-lib/src/main/cpp/referenceChains.cpp:3375-3397` + +## What this rules out +- "It will fix itself on the next search generation" — it will not, because + the search can never leave `RUNNING` + (`find-canary-search-cannot-terminate`), so `_candidate_count` never + resets and the gate never reopens. +- Fixing only the slot decode being sufficient: with a shrinking/churning + candidate set, dead slots stay latched and new candidates never get + tagged at all, so they can never be found. + +## Fix implemented (this session, uncommitted on `jb/reference-chains`) + +Replaced the one-shot `if (_candidate_count == 0) { ... }` block with a +growing-admission loop that runs on **every** poll where `candidate_count > 0`: +for each candidate in the current `selectLeakCandidates()` result, skip it +if a slot already tracks its `klass_id` (new `_candidate_klass_ids[]` +array added to `referenceChains.h`, one entry per slot); otherwise, if a +free slot exists (`_candidate_count < MAX_LEAK_CANDIDATES_FROM_LT`), admit +it into the next slot, tag its representative object, and grow +`_candidate_count`. Slots that fill up when `MAX_LEAK_CANDIDATES_FROM_LT` +is reached are logged and left untracked for that search (no eviction). + +Design decision, made explicitly with the user via a clarifying question: +**slots are never retired/reused for the lifetime of a search** — a +candidate that later drops out of `selectLeakCandidates()`'s result stays +tagged and occupies its slot until the search resets. Rejected alternative: +retiring/reusing slots was ruled out as added complexity/risk (re-deriving +`_candidate_found_bits` semantics and untagging live objects mid-walk) that +the user judged not worth it for a bug of this shape. This is why +`find-canary-stuck-abandon-detector` (Fix C) had to be added alongside — +never-retire means completion (`popcount(found_bits) == candidate_count`) +can be permanently blocked by one never-found candidate, and Fix B alone +does not close that gap. + +`Counters::increment(REFERENCE_CHAIN_CANDIDATE_COUNT, 1)` is called once +per newly-admitted candidate (delta-based counter, confirmed via +`counters.h`), replacing the old one-shot absolute-count call. + +All three lifecycle reset sites (`start()`, `resetSearchStateForTest()`, +and `runPass()`'s terminal-state cleanup) reset `_candidate_count`, +`_candidate_found_bits`, and (per Fix C) the new progress-tracking fields +together, so a fresh search always starts from an empty admission table. + +Verified: compiles cleanly and both `referenceChains` gtest suites pass +(89 + 1 tests) — see `ev-fixes-compile-and-gtest-pass`. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-onpod-evidence-methodology.md b/.investigations/missing-refchains-on-hotdog/nodes/find-onpod-evidence-methodology.md new file mode 100644 index 0000000000..4efd9a4208 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-onpod-evidence-methodology.md @@ -0,0 +1,51 @@ +--- +id: find-onpod-evidence-methodology +type: finding +status: confirmed +depends_on: [ev-toolkit-and-onpod-methodology, ev-deployed-so-1481-no-symbols] +supersedes: [] +related: [dead-jcmd-jfr-dump-wrong-source, dead-toolkit-prod-datacenter, find-refchains-not-deployed] +tags: [methodology, user-feedback, on-pod, evidence-quality] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Methodology: prove deployment state ON the pod, not from downloaded artifacts or image tags + +## Reasoning chain + +The first version of the phase-1 answer was assembled from Profiling +Toolkit downloads plus inference from the container image tag +(`v130436965-4aea7d55-amd64`, "built from branch=prod"). The user pushed +back: + +> you should be able to read the pods from the pod + +That is the stronger method, and it changed the evidence from inferential +to direct: + +- `kubectl exec … jar xf /usr/local/app/agent/dd-java-agent.jar + shared/META-INF/native-libs/linux-x64/libjavaProfiler.so` — read the + artifact that is actually installed, out of the jar the JVM is actually + loading via `-javaagent`. +- `md5sum` that against `/tmp/ddprof_root/pid_/scratch/libjavaProfiler-dd-tmp*.so` + (the copy `DdprofLibraryLoader` extracted and the JVM mapped, found via + `/proc//maps`) — identical md5 closes the staleness question that + otherwise always lingers in this codebase. +- `strings` for the feature's own symbols, and for the embedded version + string, gives an artifact-identity answer that no metadata can dispute. + +Practical constraints discovered: the pod has `/usr/bin/jar` but no +`unzip` and no `python3`; PIDs change when the agent is re-synced (231 → +20807), so always re-derive the PID with `ps aux | grep java` before +`jcmd`. + +## Evidence +- `evidence/ev-toolkit-and-onpod-methodology.md` +- `evidence/ev-deployed-so-1481-no-symbols.md` + +## What this rules out +- Trusting image tags / build metadata as evidence of which ddprof-lib is + running. The jar is mutable in place on this pod (the user overwrote it + live at 14:27 without rescheduling the pod), so the image tag stayed + `v130436965-4aea7d55-amd64` while the contents changed completely. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-option-c-descend-walk-design.md b/.investigations/missing-refchains-on-hotdog/nodes/find-option-c-descend-walk-design.md new file mode 100644 index 0000000000..392426cb35 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-option-c-descend-walk-design.md @@ -0,0 +1,163 @@ +--- +id: find-option-c-descend-walk-design +type: design +status: committed +depends_on: [ev-leaktag-onpod-round5, ev-leaktag-onpod-round6] +related: [find-priority-queue-starves-bfs-crawl, find-per-tid-qualification-design, find-rotation-resize-blindspot] +tags: [option-C, descend-walk, thread-local, static-holder, referenceChains, NEW-THIS-SESSION] +created: 20260902 +updated: 20260902 +--- + +# Option C: candidate-scoped reach via bounded descend walks (both prongs, one mechanism) + +User decision: implement BOTH prongs (taxonomy-driven, deliberately NOT +shaped by probing hotdog's simulator - "we don't want to overfit this +particular scenario"). + +## The unified mechanism + +Both prongs are the same operation: a `FollowReferences(initial_object=anchor)` +with `batch_tags == nullptr` (so descent is gated only by hop_cap + +deadline), reusing heapReferenceCallback unchanged - leak-tag interception, +canary pruning, auto-mark, improveChain/reparent all work as-is. ctx.hop_cap +is lowered to `anchor_depth + DESCENT_HOPS` for the call so the walk is +bounded to a few hops BELOW the anchor. Children admitted by the walk get +parent links to the anchor's chain, so interception = complete chain in one +bounded STW. No queue-cascade design: admitted children joining the tail of +_pending_expand does not matter, the chain entries exist at admission time. + +- **Prong 1 (thread-scoped, thread-retained taxonomy)**: new phase in + runPassManualWalk; for each candidate's qualifying tids, look up the live + Thread object, admit it as a root-attached frontier entry (root_kind= + JVMTI_HEAP_REFERENCE_THREAD, depth 0), then descend-walk with: + - DESCENT_HOPS below the Thread anchor, + - an anchor gate on the Thread's own edges: descend only into + ThreadLocal$ThreadLocalMap (the value type of BOTH threadLocals and + inheritableThreadLocals) by exact class-tag match. Deliberately NOT + jvmtiHeapReferenceInfoField.index matching: that index is a jint field + ORDINAL, not a jfieldID (found via a compile error against the real + jvmti.h), and its correspondence to GetClassFields order is a spec + subtlety the design avoids depending on. Gate falls open (generic + descent, still no-descend+hop-capped) if the class is not resolvable. + - an exact-class no-descend set below that (ClassLoader, ProtectionDomain, + ThreadGroup - contextClassLoader's graph would otherwise reach every + loaded class and its statics; exact-tag match only, documented as such). +- **Prong 2 (static-holder taxonomy)**: rotation-tier collector with a + wrapping cursor over the frontier table selecting parent_tag==0 && + root_kind==STATIC_FIELD && (FRONTIER or EXPANDED) && !isQueuedForRotation; + each selected anchor is descend-walked (hop_cap = anchor.depth + + DESCENT_HOPS). Covers collection-shaped static holders (static Map -> + table -> Entry -> chunk is 3-4 hops) in ONE walk - no cascade needed. + +## Key code facts the design relies on (all verified) + +- heapReferenceCallback descent control: non-null batch_tags = one-hop gate; + nullptr = hop_cap-gated generic descent. Depth computed from frontier + parent lookup (parent.depth + 1), so hop_cap lowering bounds walk depth. + Caveat (documented in code): a pre-existing frontier entry reachable in + the subgraph carries its global depth, so its subtree can be walked deeper + than DESCENT_HOPS - still bounded by global hop_cap + deadline. +- jvmtiHeapReferenceInfoField.index IS the jfieldID for FIELD references - + the anchor field gate can match field IDs resolved before the walk (only + JVMTI GetClassFields/GetFieldName, both legal outside heap callbacks). +- nextTag() restarts at 1 per search; anchor idempotency = GetTag -> + frontier->lookup -> reuse-or-retag (stale thread tag after restart would + otherwise alias a reissued tag). +- FrontierEntry carries parent_tag/depth/state/root_kind/class_tag - the + prong-2 filter is directly expressible. +- THREAD root_kind ranks durability 1 (referenceChains.h rootKindDurability) + - thread anchor chains get correct (transient-tier) labeling. +- runPass + pollWatchedTargets both run on threadLoop serialized - retained + qualifying tids are safe to read in the walk phase. + +## Round-7 follow-up (user-picked option 2, commit c6635fe0e) + +Pod round 7 (ev-leaktag-onpod-round7): both prongs live, interception +still zero - the tagged chunks sit past the 6-hop descend cap (static +anchor walks admitted whole 3000+-edge task subgraphs but stopped +short) or under an uncovered root kind. Fixes in ONE rebuild: +- DESCENT_HOPS 6 -> 16: the walk is deadline-bounded per slice either + way; already-admitted entries are skipped so repeated passes march + deeper each pass. +- collectStaticFieldAnchorsForRotation now also selects root-attached + JNI_GLOBAL entries (the remaining durable root kind - same collector, + same wrapping cursor, same bounded walks). +- Round 8 = deploy 9d3d0afe6 and watch the same three lines + (walk engagement, leak-tag intercepted, first chain's edges names). + +## Round-8 verdict (ev-leaktag-onpod-round8): BOTH fixes live, interception STILL zero + +Both hypotheses from round 7 are REFUTED as sufficient explanations: +16-hop walks complete un-truncated over all 4 root-attached static +anchors (per-pass edges 10→3608, rotation cycling different anchors), +JNI_GLOBAL anchors selected in the same tier — zero interceptions. The +app's actual retention shape (found in its bytecode, diagnosis-only): +ProfileAnalyzer.LEAK_BUFFER, a static final List wrapped in +Collections.unmodifiableList, 3-4 hops from the root static — textbook +prong-2 taxonomy, well inside the cap. The local scenario intercepts +the identical shape, so the walk mechanism is sound in-process. The +remaining suspect is tier MEMBERSHIP on the pod: the +Collections$UnmodifiableList wrapper (a) never admitted root-attached, +(b) root_kind misclassified, or (c) admitted root-attached then +replaced by a chain-attached entry (improveChain / reparentToDurableRoot +both produce parent_tag!=0 entries — plausible: the wrapper is reachable +via many chains and improveChain favors deeper chain-attached entries). +TEMP per-anchor diagnostic (c9a57f681) logs every walked anchor's class +signature + parent/root_kind/state — round 9 = deploy it and read the +wrapper's tier state directly. + +## New machinery + +- tid -> jthread global-ref registry in ReferenceChainTracker, fed from + Profiler::onThreadStart/onThreadEnd (profiler.cpp already calls + ReferenceChainTracker::instance(); register gated on _enabled, + unregister deliberately NOT - a thread registered while enabled must + release its global ref even after the recording stopped). PLUS a + one-time registerExistingThreads() GetAllThreads sweep at + Profiler::start(): a leaking thread is typically alive since before the + recording began (the first ThreadLocalLeakScenario run caught this: + walked=0 with the leak thread unregistered), and onThreadStart never + fires for it. The sweep lives in the Profiler::start() lifecycle, NOT in + RCT::start(): the JFR-roundtrip gtest calls RCT::start() directly against + a partial mock JVMTI table (no GetAllThreads slot) and crashed on the + null function pointer - a real-lifecycle-only operation must run from + the real lifecycle. +- RCT retains per-candidate qualifying tids from each poll (bounded copy of + KlassCandidate::qualifying_tids), zeroed per poll so a dropped candidate + stops being walked. +- DESCENT_HOPS = 6 (ThreadLocalMap -> table[] -> Entry -> value -> holder -> + chunk is 5-6 hops; static Map shape is 3-4). +- Cost: each anchor walk is one bounded FollowReferences (deadline + hop + cap + per-pass anchor caps: THREAD_WALK_MAX_ANCHORS=4, + STATIC_ANCHOR_ROTATION_BUDGET=4); prong 1 runs before the static sweep on + its own deadline slice, prong 2 at the head of rotation's slice. + +## Verification (shape-agnostic, all green) + +- gtests (mock scripted heap, no live JVM): ThreadWalkDescendsOnly- + ThreadLocalMapAndInterceptsLeak (both anchor-admission paths, both gates + distinct - anchor gate vs no-descend - plus interception + THREAD-rooted + chain shape) and StaticAnchorRotationWalksRootAttachedStaticHolders + (collector filter + wrapping cursor + depth-3 interception through a + static Map -> table -> Entry -> chunk walk). 553 gtests green. +- Java slow suite: NEW ThreadLocalLeakReferenceChainTest + + ThreadLocalLeakScenario (ExternalLauncher mode threadlocal-leak) - the + missing thread-local taxonomy scenario: ThreadLocal-held growing byte[] + collection, correlated [tl-correlation-found] end-to-end with the thread + walk engaging live (walked=1 edges=18). Full reference-chain slow family + 9/9 green (ToGcRoot, UnboundedCache, correlation loose+tight, external + static-field, aggressive, seams). +- The pod only CONFIRMS the taxonomy coverage; it does not steer it (no + probing of hotdog's simulator shape - user's explicit constraint). + Pod round 7 = deploy and watch interception count. + +## Commits (pushed to origin/jb/reference-chains-pi) + +- 186468437: descend-walk core (both prongs, gates, phase wiring, + tid->Thread registry + start sweep) + the 2 gtests. +- 01c591eea: ThreadLocalLeakScenario + ThreadLocalLeakReferenceChainTest + + threadlocal-leak launcher mode. +- 93868362e: memory sync (this node). + +Deploy handle: build 93868362e. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-per-class-caching-blocks-instances.md b/.investigations/missing-refchains-on-hotdog/nodes/find-per-class-caching-blocks-instances.md new file mode 100644 index 0000000000..db43312705 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-per-class-caching-blocks-instances.md @@ -0,0 +1,53 @@ +--- +id: find-per-class-caching-blocks-instances +type: finding +status: fixed +depends_on: [find-representative-changes-lose-canary] +supersedes: [] +related: [find-representative-changes-lose-canary] +tags: [root-cause, fix, referenceChains, resolved-chains, per-instance, caching, NEW-THIS-SESSION] +created: 2026-08-28 +updated: 2026-08-28 +--- + +# Per-class resolved-chain caching blocks all but the first instance + +## Observation + +JFR analysis of the real ddprof recording (not jcmd) showed 2 +`datadog.ReferenceChain` events: + +1. `targetTag=-4611686018427387906` (marker tag) — depth=14, rootKind=unknown + (canary chain for [B, but rootKind was "unknown") +2. `targetTag=5940` — depth=1, rootKind=first_observed_via:jni_local + (auto-mark chain for the **noise** [B — 136B, s3-netty-2 thread) + +The leaking [B instances (12 × 78MB, `simulated-memory-leak` thread, +ages 33-168, held via static-field ArrayList at depth=14) were never +emitted. The noise [B at depth=1 (136B, JNI local root) was admitted +first by BFS, auto-marked, its chain built and cached. Per-class +caching (`_resolved_chains` keyed by `klass_id`) meant the class was +marked as "has a chain" — `no_chain_cached` was false — and all +further [B instances (including the actual leak) were skipped. + +## Root cause + +`_resolved_chains` was keyed by `klass_id` (u32). For common classes +like `[B` with thousands of live instances, the first chain found +(often noise — a shallow JNI-local instance 1-2 hops from a root) +permanently blocked all other chains for the same class. + +## Fix (COMMITTED cd68be618) + +Key `_resolved_chains` by frontier tag (jlong) instead of klass_id. +Each discovered instance gets its own chain entry. All chains are +emitted to JFR; the profiling backend aggregates by class. + +- `cacheResolvedChain`: first arg is `jlong source_tag`, not `u32 klass_id` +- Discovered path: build chains for ALL discovered instances (not just + first that succeeds), skip if already cached for that tag +- Remove `no_chain_cached` early-out +- Remove per-klass prune: chains expire on search restart (frontier wipe) +- `MAX_RESOLVED_CHAINS`: 128 (5×8 discovered + 5 canary + headroom) +- Tests updated: key by frontier tag, `PruneStopsReemitting` → + `ChainPersistsAfterRepresentativeDies` diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-per-tid-qualification-design.md b/.investigations/missing-refchains-on-hotdog/nodes/find-per-tid-qualification-design.md new file mode 100644 index 0000000000..624107a96a --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-per-tid-qualification-design.md @@ -0,0 +1,87 @@ +--- +id: find-per-tid-qualification-design +type: finding +status: implemented +depends_on: [ev-leaktag-onpod-round3, find-canary-search-forces-max-cadence, find-jvmti-heap-walk-stw-vmop] +related: [find-leak-tag-pool-implementation, q-allocation-site-selection, find-lambda-fragments-calltrace-id] +tags: [fix, design, livenessTracker, leak-tag, option-C, NEW-THIS-SESSION] +created: 20260901 +updated: 20260901 +--- + +# Option C implemented: (klass, tid) qualification - trend OR retained-count bar + +Option C (chosen after B was retracted - full-graph walks are one STW +pause, find-jvmti-heap-walk-stw-vmop). selectLeakCandidates() now requires, +on top of the klass-level trend gate, that at least one ALLOCATING thread +qualify; KlassCandidate carries the qualifying tids and +tagLeakInstances() tags ONLY tracked instances those tids allocated +(production tagging scope = the leak site, not the whole klass). + +## The gate (LivenessTracker::recordTidTrendSamplesLocked / tidPushQualifies) + +Per-klass TidTrend rings (8 tids x 16 slots) folded from the EXISTING +KlassCountScratch per-thread data (insertThreadGen) once per GC epoch - +no JVMTI call, pure table work. A tid qualifies for an epoch push if: + +1. sustained age-trend: mean-of-thirds rise on its distinct-surviving-age + ring (min fill 6, same LEAK_GROWTH bars), consecutive for the same + hysteresis as the klass gate; OR +2. retained-count bar: surviving tracked instances of this one klass + from this tid >= TID_RETAINED_COUNT_BAR (8) that epoch. + +The OR is load-bearing and NOT arbitrary: pure age-trend structurally +cannot see one-cohort-per-thread accumulation (each one-shot worker's +instances share one age -> distinct-age count stays 1 forever, exactly +LeakingCacheScenario's per-round allocator-thread shape - a genuine +leak that would have been rejected). Machinery threads fail BOTH +discriminators (stable low count, flat age span). Decay: absent tids +push 0 (resets hysteresis - a thread whose instances died must not keep +a stale rising ring); eviction beyond 8 tids picks the weakest +non-synthetic trend. Test-seeded trends are synthetic=1 and exempt from +real-fold updates/decay/eviction (scenarios interleave System.gc() folds +with seeded ramps). + +## Pod implication (corrects the earlier "C retires the pod candidate" claim) + +Hotdog runs a DELIBERATE simulated-memory-leak thread (tid 172 / 92169 +on different JVMs, age_count rising 2->3; ev-tid-clustering evidence) - +so per-tid qualification does NOT retire the pod's [B candidate: it +SCOPES the tagging to the leak thread's instances. The 247 tagged +machinery byte[]s (flat per-site retention, never intercepted) are out +of scope by construction; the pool tags go to leak-site instances whose +holders are the crawl's actual rotation targets. CPU-burn relief on the +pod comes from interception becoming possible, not from the candidate +disappearing. + +## Test seam + +seedTidTrendSample0 (JavaProfiler/javaApi.cpp -> tidTrendRecordForTest): +one seeded value lands in BOTH rings (age + retained-count), so tests +can qualify either way. Scenarios seed the REAL allocating tid +(JavaProfiler.getTid(), captured on the leaking thread - the noise +thread publishes its tid via the handoff) because production tagging +matches tracked instances by tid. gtests use fixed synthetic tids +(gate-only, no live heap). + +## Verified + +543 gtests (4 new: no-tid skip, flat-tid reject, retained-bar qualify, +rising-below-bar qualify); slow suite 8/8 including +LeakTagCorrelationReferenceChainTest's [correlation-found] through the +per-tid-scoped production tagging path; testDebug unchanged (5 +pre-existing env failures). + +## Inference (not verified) + commits + +LeakingCacheScenario's CachedPayload allocations are far below the +allocation-sampling floor (300 small objects per round vs a ~512KiB +interval), so its chain most likely comes from the representative +machinery (seed-0 via setKlassPopulationRepresentativeForTest0), not +leak tagging - its main-tid seeding is therefore gate-only in practice. +Inferred from the sampling floor + the scenario passing with zero +interception mechanics observed; not directly instrumented. + +Commits: 5d498811f (PriorityExpandSet), c5490156e (per-tid +qualification), 14add8140 (memory sync), pushed as +1a5055548..14add8140 on jb/reference-chains-pi. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-priority-queue-starves-bfs-crawl.md b/.investigations/missing-refchains-on-hotdog/nodes/find-priority-queue-starves-bfs-crawl.md new file mode 100644 index 0000000000..61f73649b8 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-priority-queue-starves-bfs-crawl.md @@ -0,0 +1,75 @@ +--- +id: find-priority-queue-starves-bfs-crawl +type: finding +status: confirmed +depends_on: [find-ema-batch-collapse, find-leak-tag-pool-implementation] +supersedes: [] +related: [find-leak-tag-pool-implementation, find-ema-batch-collapse, find-holistic-design-issues, q-coverage-tracking-per-combination] +tags: [root-cause, fix, referenceChains, expandFrontier, priority-queue, starvation, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# _priority_expand flood starves the BFS crawl (three compounding blockers) + +## Observed on-pod (round 3, build 0db70994d, PID 48355) + +- Pass throughput recovered (AIMD works: batch oscillates 34→17→81 around the + 25ms per-call budget, ~12 passes/min, deadline-bounded) BUT: +- `_priority_expand` grew 39k→103k in 20 min; `_pending_expand` (66k→77k) was + NEVER drained once. +- `leak-tag intercepted`: 0 fires in 20+ min despite 129 tagged instances. +- Only the same 8 noise chains, cached and re-emitted (batch=8 × 3 drains). +- `leak_accumulation_tags=0` every pass (growth-gated tier dead in steady + state); `leak_parents=19058` in the fanout, unreachable. + +## Causal chain (all links proven from code + logs) + +1. Rotation collectors enqueue 256+16 stale entries/pass into + `_priority_expand`; each phase's 50ms deadline allows only ~2 gotw calls + (~150-300 entries) of drain. Inflow > outflow every pass. +2. expandFrontier drains priority STRICTLY FIRST + (`from_priority = !_priority_expand.empty()`), so a never-empty priority + queue means pending is never touched: the BFS's own frontier never + expands new territory. The singleton holders sit in pending. +3. Without holder expansion, parents of the 129 leak-tagged arrays are never + walked → interception can never fire. +4. `tagLeakInstances` SetTag overwrites frontier tags of already-admitted + instances (admission-order hole) — orphaning entries and making + correlation depend on the very re-walks that are starved. +5. MAX_DISCOVERED_INSTANCES_PER_CLASS=8 slots are first-come-first-served — + the 8 noise instances permanently occupy them; even admitted leak + instances would be silently discarded at discovery recording. +6. `isQueuedForRotation` linear scan over the 103k queue (26M comparisons + per pass, under the frontier shared lock) adds hidden pass cost. + +## Fix (commit pending) + +1. FAIR-SHARE DRAIN: expandFrontier alternates batches between priority and + pending when both non-empty (priority still first each call) + hard cap + PRIORITY_EXPAND_CAP=1024 with skip-when-full at every push site + (collectors + admitObject fallback to pending). Caps bound memory AND + the isQueuedForRotation scan; alternation guarantees pending progress + regardless of inflow/outflow arithmetic (the cap alone does NOT fix + starvation - a capped-but-pinned queue still never empties). +2. tagLeakInstances state machine on the CURRENT JVMTI tag: frontier tag → + correlateAdmittedLeakTag (setLeakTag on the entry + recordDiscovered, + NEVER retag); no tag → SetTag (first tagging or re-establishment after a + search restart wiped tags via releaseSearchTags); leak tag already + present → record only. +3. recordDiscoveredInstance with leak-preferential eviction: noise fills + empty slots only; a leak-correlated discovery evicts an uncorrelated + slot (and invalidates its cached chain) when full. +4. Fanout-priority in collectStaleExpandedEntriesForRotation: leak parents + selected ahead of the blind table lap (rotating cursor), breaking the + steady-state deadlock of the growth-gated tier. +5. reparentToDurableRoot: equal-depth (depth-1) re-parent from a transient + root to a durable one — improveChain cannot express it (needs strictly + deeper), and the hotdog static shape is depth-0 root → depth-1 elements, + so without it noise-admitted leak instances would stay transient-rooted + forever. +6. Discovered-loop gate: depth==0 or (depth==1 && transient root) suppressed + (+ invalidate cached noise so drain stops re-emitting); depth==1 durable + root KEPT (it is the real direct-static-retention shape — a blanket + depth>1 filter would drop it; static sweep admits static-field VALUES as + depth-0 roots, referrer rtag<0 → root-like). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-refchains-log-flood-configurable.md b/.investigations/missing-refchains-on-hotdog/nodes/find-refchains-log-flood-configurable.md new file mode 100644 index 0000000000..8be15c6d4b --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-refchains-log-flood-configurable.md @@ -0,0 +1,122 @@ +--- +id: find-refchains-log-flood-configurable +type: fix-verification +status: confirmed-pending-pod-verification +depends_on: [find-wrapper-demotion-self-parent, ev-leaktag-onpod-round16-results] +related: [find-round16-endgoal-verification, find-jvmti-heap-walk-stw-vmop] +tags: [observability, debug-logs, runtime-knob, rcDebugLevel, pod-safety, fix, NEW-THIS-SESSION] +created: 20260916 +updated: 20260916 +--- + +# Find: reference-chains log flood made runtime-configurable, silent by default (f70bcc454) + +User need: keep hotdog running the DEBUG profiler build (TEST_LOG is +`#ifdef DEBUG`-only via common.h — that is why the pod floods at all) +but stop the reference-chains diagnostic logs from swamping the pod +("it will get killed sooner or later"). Deleting them would discard +what cracked rounds 12-16, so they are gated at a runtime level instead +(new header `rcDebugLevel.h`, included AFTER common.h by +referenceChains.cpp + livenessTracker.cpp; re-points TEST_LOG at a +level check and adds TEST_LOG_SUMMARY). + +## Levels and tier rule + +0 = SILENT (the new default — pod-safe with the same .so). 1 = +lifecycle/summary. 2 = full firehose. Tier rule, sized EMPIRICALLY from +the round-16 stream (3.1M lines/55min): **per-pass/per-poll outcomes +and state-machine transitions → 1; per-object/per-klass/per-entry inner +loops → 2.** The measured flood: auto-marks 1.2M (auto-marked chain +per instance per poll), selectLeakCandidates entries 490k, +trackLeakAccumulation 190k, heap admits 148k, sweep STATIC_FIELD 92k, +maybeUpgradeRootAttached 24k, resolveClassMap 23k, pushAtRisk 10k — +all level 2. Level 1 keeps: shouldRunPass/canary decisions, candidate +state machine, rotation_candidates (with self_edge_skips/quota_drops), +drain re-emitted, correlateAdmittedLeakTag, candidates-armed — est. +<300 lines/min. 49 RC + 11 liveness sites re-tiered to SUMMARY. + +## Sources and the callback-context constraint + +Env `DD_PROFILING_REFERENCE_CHAINS_DEBUG=N` at start; runtime override +WITHOUT restart: `echo N > /tmp/ddprof_root/refchains_debug_level` +(single digit; remove the file → fall back to env), re-checked ~1s TTL +from the RC threadLoop. JVMTI heap callbacks only READ the cached +atomic — the open/read never runs there (callbacks must not allocate; +the fprintf these lines already do in DEBUG builds is the tolerated +debug-build exception, but the knob adds nothing new there). Level +reported on the threadLoop-start line for confirmation. Flip from +outside the pod: `kubectl exec -c prof-analyzer -- sh -c 'echo 2 +> /tmp/ddprof_root/refchains_debug_level'`. + +## Two build/workflow lessons + +1. **The gtest binary compiles the main sources WITHOUT DEBUG** — + discovered when `#ifdef DEBUG`-only declarations failed the UT + compile (the library TU built fine; only the test TU saw no + declarations). So the level machinery is compiled UNCONDITIONALLY: + in non-DEBUG builds nothing calls it (the macros are no-ops) and the + gtest binary (non-DEBUG) can test it directly — RcDebugLevelTest + (parse/read/refresh-env), suite 121/121. +2. **A multi-edit tool call rejected on one ambiguous anchor applies + NOTHING (atomic)** — and re-applying only the remembered edit left + the include + impl block silently missing, surfacing only as + "undeclared identifier" compile errors far from the cause. After a + rejected batch: re-apply EVERY edit, then grep that each landed + (one `grep -n rcDebugLevel` would have shown the include missing). + +## Pod verification (build f70bcc454, pod 289f8, JVM 96 min) — TWO escapes found and fixed + +1. **Header escape (fixed 7aa68744b)**: 10 TEST_LOG call sites lived + INLINE in referenceChains.h (buildChainEvent + buildCanaryChainEvent + bodies) — compiled with common.h's ungated TEST_LOG because TUs + including the header never see rcDebugLevel.h's macro redefinition + (~130-1700/min of 'buildChainEvent false:' leaked at level 0). + Fixed by moving both member functions out of line into the .cpp + (buildChainEvent outcomes level 2; buildCanaryChainEvent outcomes + level 1 — the chase's lifecycle, bounded by candidate count). Could + NOT fix by including rcDebugLevel.h from referenceChains.h: that + would re-point TEST_LOG TU-wide for javaApi.cpp/profiler.cpp/ + vmEntry.cpp (24 unrelated sites) — macro state is per-TU from the + include point; header-inline bodies from headers included EARLIER + (e.g. profiler.h) stay ungated regardless. +2. **Tick-cadence mis-tier (fixed 524c35325)**: live level-1 window + measured ~2600/min — lines tiered "per poll" actually fire every + ~250ms threadLoop tick (poll candidate state, selectLeakCandidates + returning, noteSelected, tagLeakInstances summary, + collectLeakAccumulation selected). 10 sites demoted to level 2; + level 1 now = true transitions/bounded outcomes only (<500/min). + LESSON: tier by measured firing cadence, not by the call site's + intent — verify the tier empirically with a level-1 window before + declaring it leave-on safe. + +## Pod verification round 2 (build 524c35325, in-place JVM restart to pid_28570 at 12:09Z, same pod 289f8) + +DEFAULT SILENCE VERIFIED: 0 reference-chains/LivenessTracker lines at +level 0 (3-min window; header escape gone). Remaining TEST_LOG ~300/min +is the .so unpacker's pre-existing 'Unpacking' lines - other subsystem, +out of scope. Two level-1 stragglers found + demoted (48c1e1c22): +'shouldRunPass held off by canary backoff mult' fires per ~100ms tick +EVALUATION (835/85s), and the collectLeakAccumulation... selected line +(4/s during rotation) - missed by the 524c35325 demotion because its +format string splits across two lines and the concatenation probe has a +quote-quote '''''' between the halves; when demoting, probe the line +AFTER the TEST_LOG_SUMMARY line, not just the concatenation. + +The runtime flip ITSELF verified live: `echo 1 > file` → summary lines +within ~1s (no restart), rm → silent. Round-16 pipeline live on this +JVM: re-emitted=8, candidate[0] klass_id=10 ([B — ids differ per JVM), +self_edge_skips=26150, quota_drops=1644, fifo_size=161. CANARY STILL +0/1 (now 3 JVMs / hours; backoff 16, ~1 pass/3s). MOVING-TARGET +HYPOTHESIS REFUTED: level-2 window shows candidate[0] klass_id=6 +tag=221704 needRefresh=0 — the SAME tag on 203 consecutive ticks (the +representative is stable, a plain object tag 221704, not a leak tag). +The representative is simply NEVER ADMITTED by any walk while pool +instances are — next-round diagnosis: catch a full manual-walk sweep +(level 2, the walked=48 truncated=0 pass) and check whether the +wrapper's subtree admits chunks, and which holder holds tag 221704 +(cf. find-representative-changes-lose-canary for the refresh variant — +not this one). Walk machinery healthy on this JVM: full 48-anchor +sweeps completing untruncated, partial walks edges 198-3887 truncated=1, +self_edge_skips=5174, fresh lane active (fresh_tier=4 fresh_queue=324), +drain re-emitted=2. Remaining non-ours debug noise on the pod: ~180/min +'Unpacking' lines (the .so unpacker's pre-existing DEBUG logging). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-refchains-not-deployed.md b/.investigations/missing-refchains-on-hotdog/nodes/find-refchains-not-deployed.md new file mode 100644 index 0000000000..59d7b539be --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-refchains-not-deployed.md @@ -0,0 +1,64 @@ +--- +id: find-refchains-not-deployed +type: finding +status: confirmed +depends_on: [ev-deployed-so-1481-no-symbols, ev-uploaded-jfr-no-refchain-types] +supersedes: [] +related: [find-onpod-evidence-methodology, find-hotdog-deploy-last-mile, ev-post-resync-deployment-verified] +tags: [phase-1, deployment, resolved, ddprof-1.48.1] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Phase 1 (RESOLVED): the pod was running ddprof-lib 1.48.1, which has no reference-chain code + +## Reasoning chain + +The original question was "why zero `datadog.ReferenceChain` events". The +first answer turned out to be trivial and not a code bug at all: the +feature was never on the pod. + +1. Uploaded profiles for `prof-analyzer-hotdog-jb` did not even *declare* + `datadog.ReferenceChain` / `datadog.ReferenceChainAbandoned`. Absence + of the type declaration (as opposed to a zero count) already implies + the emitting code isn't in the binary. `datadog.HeapLiveObject` WAS + present, so liveheap sampling itself worked — narrowing it to the + reference-chain feature specifically. +2. The `.so` the JVM had actually mapped + (`/tmp/ddprof_root/pid_231/scratch/libjavaProfiler-dd-tmp927042133104699179.so`) + had **0** `ReferenceChainTracker` strings but **20** `LivenessTracker|liveheap` + strings — same conclusion from the binary side. +3. To exclude a stale-scratch-extraction artefact, the bundled + `shared/META-INF/native-libs/linux-x64/libjavaProfiler.so` was extracted + from `dd-java-agent.jar` on the pod and md5'd: + `794906a03568ff284c2cb557af693e22` — **byte-identical** to the loaded + scratch copy. So image content == running content; not a staleness bug. +4. The embedded version string in that `.so` is `1.48.1`. Tag `v_1.48.1` + (`c96ea85f7`, Tue Aug 4 2026) is **not an ancestor** of + `jb/reference-chains`, and all reference-chain commits + (`dc8071dc9` onwards) live only on that branch. So by construction the + deployed artifact cannot contain the feature. +5. dd-trace-java was stock `1.65.0~dd00372bdd` from image + `prof-analyzer-hotdog:v130436965-4aea7d55-amd64`; the deployment spec + had no init container, volume or env override injecting a custom agent + jar, and the JVM command line carried no reference-chain property. + +## Evidence +- `evidence/ev-deployed-so-1481-no-symbols.md` +- `evidence/ev-uploaded-jfr-no-refchain-types.md` + +## What this rules out +- Configuration/flag problem — there was no flag to set; the option + `referencechains=…` (`ddprof-lib/src/main/cpp/arguments.cpp:85,468-490`) + is parsed only by code that isn't in 1.48.1. +- Stale extracted native library (a known java-profiler failure mode): + refuted by the byte-identical md5 between jar-bundled and scratch copy. +- Any hypothesis about the reference-chain algorithm itself, for the + pre-resync data. That data is worthless for algorithm questions. + +## Status note +Superseded as the *current* explanation by +`ev-post-resync-deployment-verified` — after the user resynced and +reuploaded the agent the feature IS present, and a genuine code bug +(`find-marker-tag-slot-index-mismatch`) took over as the cause of zero +events. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-representative-changes-lose-canary.md b/.investigations/missing-refchains-on-hotdog/nodes/find-representative-changes-lose-canary.md new file mode 100644 index 0000000000..c50b20342f --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-representative-changes-lose-canary.md @@ -0,0 +1,61 @@ +--- +id: find-representative-changes-lose-canary +type: finding +status: fixed +depends_on: [find-candidate1-never-tagged] +supersedes: [] +related: [find-candidate1-never-tagged] +tags: [root-cause, fix, referenceChains, canary, representative, lru-eviction, marker-tag, NEW-THIS-SESSION] +created: 2026-08-27 +updated: 2026-08-27 +--- + +# Canary representative changes lose marker tag for high-churn classes + +## Observation + +On the live pod, the [B (byte array) leak candidate consistently showed +`tag=0` in `pollWatchedTargets()`, while other candidates (klass_id=145, +211, 2292) had stable marker tags. The [B representative kept changing +because there are many leaking [B instances — LivenessTracker's ring +buffer evicted the old representative and selected a new one. The new +representative had no marker tag; the old one's marker tag was on a +dead/evicted object. + +The canary mechanism pre-tagged one representative per class with a +marker tag. When the representative changed, the canary lost track — +no chain was ever built for [B. + +## Root cause + +`pollWatchedTargets()` admitted a candidate by `SetTag(obj, marker_tag)` +on the representative returned by `resolveCandidateRepresentative()`. +On subsequent polls, it called `getTag(obj)` on the NEW representative +(which returned 0) and treated `tag == 0` as "not yet discovered by the +walk" — but actually the marker tag was on the OLD (now dead) +representative. + +## Fix (COMMITTED 7bd7255c3) + +Two changes: + +1. **Re-tag the representative** when `tag == 0` for an already-tracked + candidate. `pollWatchedTargets()` finds the slot for this klass_id + and `SetTag(obj, marker_tag)` on the new representative, so the canary + mechanism can find it on the next walk pass. + +2. **Auto-mark all discovered instances**: when the BFS walk admits an + object whose class matches a watched leak class, record its frontier + tag in `_candidate_discovered_tags[slot]` (fixed-size array, no heap + allocation in the callback). `pollWatchedTargets()` builds chain + events for all discovered instances, not just the pre-tagged + representative. A leaking class typically has many live instances, each + with an independently useful reference chain. + +### Data structure + +`_candidate_discovered_tags[MAX_LEAK_CANDIDATES][MAX_DISCOVERED_INSTANCES_PER_CLASS]` +where `MAX_DISCOVERED_INSTANCES_PER_CLASS = 8`. When the per-slot array +fills, further instances are silently dropped (the representative + up +to 8 others is still far more coverage than the single-representative +design it replaces). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-rolling-resume-expandfrontier.md b/.investigations/missing-refchains-on-hotdog/nodes/find-rolling-resume-expandfrontier.md new file mode 100644 index 0000000000..e4f49e1c39 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-rolling-resume-expandfrontier.md @@ -0,0 +1,46 @@ +--- +id: find-rolling-resume-expandfrontier +type: finding +status: fixed +depends_on: [find-shared-deadline-starves-expand] +supersedes: [] +related: [find-static-field-sweep-cursor-fix] +tags: [fix, referenceChains, expandFrontier, rolling-resume, cursor, truncation, NEW-THIS-SESSION] +created: 2026-08-27 +updated: 2026-08-27 +--- + +# Rolling resume for expandFrontier: pop processed entries on truncated batch + +## Observation + +When `FollowReferences` truncates mid-batch (budget exhausted), the +entire batch was left at the front of `_pending_expand` for retry. On +the next pass, `GetObjectsWithTags` + `FollowReferences` re-walked +already-expanded entries — idempotent but wasteful (re-paying the full +O(tag_map × batch) GOTW cost and the FollowReferences STW for entries +that need no work). + +## Fix (COMMITTED b2acdaee2) + +Added `_last_visited_batch_tag` to `PassContext`, updated by the +callback's `batch_tags` descent gate (the point where the callback +decides to descend into a batch entry). After `FollowReferences` returns +truncated: + +- Entries fully processed BEFORE the truncation point are popped + (mark EXPANDED if live, clear if dead) +- The partially-visited entry (at `_last_visited_batch_tag`) stays: + some of its children may have been admitted before the abort, and + the rest are discovered on retry (admitObject is idempotent) +- Entries after the partial one stay (never visited) + +Same resumable-cursor pattern as `admitStaticFieldRoots()`'s sweep +cursor. + +## Test + +`RollingResumePopsProcessedEntriesOnTruncatedBatch` — static-field root +→ listNode → 20 chain children, budget=4, 20 distractors. Verifies +listNode gets EXPANDED after truncated expand, all children eventually +admitted across passes. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-rotation-resize-blindspot.md b/.investigations/missing-refchains-on-hotdog/nodes/find-rotation-resize-blindspot.md new file mode 100644 index 0000000000..ab04163fbc --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-rotation-resize-blindspot.md @@ -0,0 +1,61 @@ +--- +id: find-rotation-resize-blindspot +type: finding +status: confirmed-and-fixed +depends_on: [find-priority-queue-starves-bfs-crawl] +related: [find-priority-queue-starves-bfs-crawl, q-resize-instrumentation-rescan-priority, find-leak-tag-pool-implementation] +tags: [root-cause, fix, rotation, growing-collections, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# Rotation is blind to container RESIZES (the user's challenge was correct) + +A fixed-slot fixture array was proposed to make the scenario pass; the +user challenged it ("what if an ArrayList grows in the real app?") - the +challenge was correct on both counts: the workaround was hiding the +requirement, AND the production path was genuinely broken. + +## The structural blindspot (all links observed live in the child JVM) + +For a growing collection (ArrayList swaps elementData, HashMap resizes +table): +1. The live holder (the list) is EXPANDED with a FROZEN children set - the + NEW backing array is never admitted as its child. +2. The fanout only ever contains parents of watched instances ALREADY + admitted - i.e. the OLD, now-dead backing arrays (a growing fanout of + corpses: 11k->17k->23k entries observed, mostly machinery + dead). +3. The blind lap (the only thing that re-walks arbitrary EXPANDED holders) + is ~table_size/budget passes deep - never reached in any window. + +Net effect: chunks added after a resize are NEVER admitted; tagged +instances below them never intercepted; zero leak-correlated chains. +This plausibly blocked the POD too (its leak buffer is a growing +synchronizedList). + +## Fix set (all in referenceChains.cpp) + +1. FAIR-SHARE ROTATION: collectStaleExpandedEntriesForRotation() gives the + fanout at most ceil(max_count/2) and the blind lap the rest - an + unbounded fanout-first policy starves the lap, reproducing the + starvation bug one level down (observed: rotation admitted edges in + only 4 of 206 passes with an 11k fanout). +2. FANOUT HYGIENE: fanout entries whose frontier slot is gone or ABANDONED + (clear() marks ABANDONED, does not remove - lookup still succeeds) are + erased during selection. +3. ANCESTOR FANOUT: trackLeakAccumulation() now inserts every ancestor up + to the root-attached entry (bounded by hop_cap, watched-admissions + only), because the direct parent is not the part of the holder chain + that stays live. +4. requeueChainRootForRotation(): the deterministic piece - on every poll, + each candidate's representative chain ROOT is enqueued into + _priority_expand (bounded by MAX_LEAK_CANDIDATES, de-duplicated). The + holder's re-walk then admits each new backing array as it appears. + This is the cheap, profiler-native version of the user's + resize-instrumentation idea (q-resize-instrumentation-rescan-priority): + the candidate chain already knows the holder - no bytecode needed. + +Observed with all four in place: 4 leak-tag interceptions at depth 2 +under the CURRENT elementData, correlated discoveries recorded with +noise-slot eviction. Remaining last-mile (candidate hysteresis aging) +tracked in find-test-seam-aliasing. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-round16-endgoal-verification.md b/.investigations/missing-refchains-on-hotdog/nodes/find-round16-endgoal-verification.md new file mode 100644 index 0000000000..35be75383d --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-round16-endgoal-verification.md @@ -0,0 +1,72 @@ +--- +id: find-round16-endgoal-verification +type: fix-verification +status: confirmed +depends_on: [find-wrapper-demotion-self-parent, ev-leaktag-onpod-round16-results] +related: [find-fresh-lane-verification] +tags: [fix-verification, pod-verification, END-GOAL, leak-tag-interception, control-run, canary, round-16, NEW-THIS-SESSION] +created: 20260916 +updated: 20260916 +--- + +# Find: round-16 end-goal verification — leak-correlated ReferenceChain events flowing on the pod + +Parent: find-wrapper-demotion-self-parent (the round-16 fixes this +verifies; FIXED + verified). Evidence: +ev-leaktag-onpod-round16-results.md (pod cgdtx, build 4afe870a2, JVM +07:45:11Z 2026-09-16, stream /tmp/r16_stream2.log). + +## The milestone + +For the first time in the entire investigation, the full chain works +end-to-end on the real leak: + +1. Leak-tag INTERCEPTION fires (12+ and climbing): + `leak-tag intercepted: leak_tag=1073742xx -> frontier_tag=... + depth=3 parent_tag=` — leak-tagged [B chunks (the real leak + instances, tags ≥ LEAK_TAG_BASE = 2^30) admitted into the frontier + WITH their leak tags and parent chains. Never fired before round 16 + (round 15: 0 interceptions, all chains carried frontier-tag + targets). +2. Leak-correlated chains cached: `auto-marked chain for klass_id=5 … + target_tag=1073742071/72/77` — leak-tag targets, not noise. +3. Events flowing: `drainPendingChainEvents re-emitted=6 → 14`, + sustained on every dump. + +Fix A verified at scale — and wider than designed: the app has MANY +synchronized-list statics (LEAK_BUFFER's wrapper, but also +`ProcessTags$Lazy` field 3 and more — the round-12 wrapper diag fires +for all of them), and self_edge_skips=30816+ means the round-15 +parent==self demotion was corrupting a whole POPULATION of anchors, not +just the wrapper. The guard keeps them all collector-selectable +(wrapper admitted root-attached parent=0 root_kind=8, walked as anchor +tag 2184). Fix B verified: quota_drops=6351 with fifo_size=132 — the +floods (same shapes as round 15) are contained; 16 drained/pass. + +## The accidental control run (unplanned, valuable) + +The first redeploy of this round put the pod on the NON-LEAKING app +image: 3h with the round-16 .so loaded and ZERO reference-chain +activity — no candidates, no searches, only boot/liveness logs. The +candidate/generations gate keeps the entire machinery dormant on a +healthy app — a production-readiness property verified live, not just +by unit tests. (Also operationally: a quiet app after deploy does NOT +mean the build is broken — check whether the app is the leaking +version and whether traffic is flowing before debugging the tracker.) + +## Remaining at verification time (watch, not blockers) + +- Canary 0/1: the chase hunts its ONE pre-tagged representative while + interceptions correctly hit the rest of the pool population; the + round-15 chase DEADLOCK (chase needing the never-happening walk) is + broken — walks + interceptions every pass, so exit is expected as + coverage reaches the representative (which also validates the + chase-exit path end-to-end). When it exits, the search completes and + restarts clean. +- `buildChainEvent false: reconstructChain failed` for small + frontier-tag targets (noise instances whose entries died mid-pass) — + benign; look during the cleanup round. +- Log rate at healthy search activity ≈ 24k lines/min (the round-15 + 800k/min flood was the deadlocked chase's TEST_LOG spam) — the TEMP + diagnostics removal list is in STATE.md; keep self_edge_skips and the + quota counters as permanent low-rate observability candidates. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-shared-deadline-starves-expand.md b/.investigations/missing-refchains-on-hotdog/nodes/find-shared-deadline-starves-expand.md new file mode 100644 index 0000000000..2647f2bb1f --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-shared-deadline-starves-expand.md @@ -0,0 +1,52 @@ +--- +id: find-shared-deadline-starves-expand +type: finding +status: fixed +depends_on: [find-getobjectswithtags-quadratic-bottleneck] +supersedes: [] +related: [find-sweep-completes-but-bfs-starved, q-safepoint-budget-model] +tags: [root-cause, fix, referenceChains, deadline, expandFrontier, sweep, pause-target, NEW-THIS-SESSION] +created: 2026-08-27 +updated: 2026-08-27 +--- + +# Shared _pass_deadline_ns starves expandFrontier to 0-1 edges/pass + +## Observation + +After the adaptive batch_size fix, `GetObjectsWithTags` was bounded at +~25ms and expand admitted 456-971 edges/pass when it got time. But on +the live pod, expand_phase showed `edges_admitted=0 truncated=1` with +`remaining_budget=27500-28527`. The static_field_phase ate the entire +shared `_pass_deadline_ns` (5-50ms), leaving expand with zero wall-clock. + +## Root cause + +`runPassManualWalk()` sets `_pass_deadline_ns` once at the top, then +calls `admitStaticFieldRoots()`, `expandFrontier()`, and rotation in +sequence — all sharing that one deadline. The sweep's `FollowReferences` +over 512 loaded classes takes ~50ms (the full budget), so expand gets +0ms and truncates immediately. + +This is the bug identified in `q-safepoint-budget-model`: one shared +deadline across all sub-operations, instead of each getting its own +per-call cap. + +## Fix (COMMITTED b2acdaee2, DEPLOYED, CONFIRMED LIVE ON-POD) + +Reset `_pass_deadline_ns` before each sub-operation (expand, rotation). +Each gets its own fresh `_effective_pause_target_ms` budget. The +cumulative rate is still capped by the pass cadence +(effectiveCadenceNs). + +### On-pod results + +| Phase | Before fix | After fix | +|-------|-----------|-----------| +| static_field_phase | 0-11 edges | 252-377 edges | +| expand_phase | 0-1 edges | 793-1606 edges | +| rotation_phase | 0-7 edges | 578-1138 edges | +| **Total** | **0-21 edges** | **1708-2766 edges** | + +~100x improvement over the adaptive batch_size alone, ~1000x over the +original 3-4 edges/pass. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-static-field-sweep-cursor-fix.md b/.investigations/missing-refchains-on-hotdog/nodes/find-static-field-sweep-cursor-fix.md new file mode 100644 index 0000000000..1664523bd4 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-static-field-sweep-cursor-fix.md @@ -0,0 +1,162 @@ +--- +id: find-static-field-sweep-cursor-fix +type: finding +status: confirmed +depends_on: [find-static-field-sweep-never-completes] +supersedes: [] +related: [] +tags: [fix, referenceChains, static-field, admitStaticFieldRoots, cursor, resumable, reordering, gtest-verified, NEW-THIS-SESSION] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# Fix: resumable per-call cursor + app-classes-first reordering for admitStaticFieldRoots() + +## Reasoning chain + +`find-static-field-sweep-never-completes` established that +`admitStaticFieldRoots()` was architecturally incompatible with the +per-pass deadline: a single non-resumable `FollowReferences()` over +~34k loaded classes cannot complete within ≤50ms. This fix addresses +that by making the sweep incremental and resumable: + +1. **Chunking**: process only `STATIC_FIELD_SWEEP_CHUNK_CLASSES=512` + classes per call, advancing a persistent cursor. One expensive class + can no longer block forward progress — the cursor advances + regardless of truncation. +2. **App-classes-first**: in-place partition of `GetLoadedClasses()` + result so app/library classes (non-null loader) sweep before the + much larger JDK bootstrap tail — directly targeting the pod's + confirmed "static field growing, just a few hops" (application-code) + leak shape. +3. **Lap-level completion gating**: "sweep done" is now gated on + `cycle_complete` (true only when a full lap completes with zero + truncated chunks) instead of the old single-call "this one call + didn't truncate" check. A lap with any truncation starts a new lap + immediately. +4. **Reversed holder fill** (added in the per-class quota iteration, + commit `0e93ab4f7`): holder array filled in reversed order so + HotSpot's LIFO `FollowReferences` descent visits classes in + ascending original index order, enabling the resumable cursor to + resume at the partial class on truncation. See + `ev-hotspot-lifo-visitation-order`. +5. **Per-class non-static quota** (added in the same iteration): + STATIC_FIELD edges always admitted; non-STATIC_FIELD edges admitted + up to 32 per class, then dropped. See `find-candidate1-never-tagged`'s + fix section and `dead-hard-reference-kind-filter`. + +## What changed + +`ddprof-lib/src/main/cpp/referenceChains.h` / `.cpp`, +`admitStaticFieldRoots()`: + +- New persistent state: `_static_field_sweep_cursor` (int, index into the + per-call class list to resume from), `_static_field_sweep_cycle_truncated` + (bool, set if any chunk within the current lap truncated), + `STATIC_FIELD_SWEEP_CHUNK_CLASSES = 512` (chunk size per call, + unbenchmarked like this subsystem's other per-pass caps). +- Every call: fetch `GetLoadedClasses()` fresh (as before), then in-place + partition the array via `GetClassLoader()` so app/library classes + (non-null loader) come before bootstrap/JDK classes (null loader) - a + per-call heuristic since JVMTI gives no cross-call ordering guarantee. + Two-way in-place swap partition, no extra allocation. +- Build the `FollowReferences` holder array from only + `classes[cursor .. cursor+512)` instead of the whole list. Advance the + cursor to the chunk end regardless of whether that chunk's call + truncated (so one expensive class can't block forward progress + forever). When the cursor reaches the end of the list (a "lap"), wrap to + 0. +- "Sweep done" (`_last_static_field_class_count = _last_resolved_class_count`, + which suppresses future sweeps until the loaded-class count changes) is + now gated on `cycle_complete` - true only when a full lap completes with + **zero** truncated chunks - instead of the old single-call "this one + call didn't truncate" check. A lap with any truncation starts a new lap + immediately (cursor already back at 0) and keeps retrying. +- New out-param `bool *cycle_complete` on `admitStaticFieldRoots()`; + caller (`runPassManualWalk()`) updated accordingly, `TEST_LOG` extended + with `cycle_complete=%d sweep_cursor=%d`. + +Chosen over: a separate/larger deadline budget for this sub-phase +(rejected - works within the existing shared per-pass budget model +instead of requesting more of it) and reordering alone (rejected as +insufficient by itself - without a cursor, a longer classlist still +permanently drops whatever is past one deadline's reach, see +`find-static-field-sweep-never-completes`'s STW-confirmation section for +why chunking was accepted as safe). + +## Test-isolation regression found during verification, fixed + +`./gradlew :ddprof-lib:gtestDebug_referenceChains_ut` initially reported +`ReferenceChainsBfsTest.RotationDiscoversLateElementOfExpandedStaticFieldCollectionWithoutSearchCompleting` +failing (search reached `SearchState::COMPLETED` unexpectedly instead of +staying `RUNNING`). Bisected via `git stash` (full suite green on +pre-change code, confirming a real regression, not pre-existing flake). + +Root cause: `ReferenceChainTracker::instance()` is a process-wide +singleton shared across every gtest case in the binary. The fixture's +`ReferenceChainsTestAccessor::reset()` (`referenceChains_ut.cpp:67`, run +from every `ReferenceChainsBfsTest::SetUp()`) resets +`_last_resolved_class_count` but had **never** reset +`_last_static_field_class_count` - a pre-existing test-isolation gap that +the new cursor/lap fields also fell into. A single-class test running +earlier in the suite could leave `_last_static_field_class_count == 1`; +the next single-class test (the failing one) would then find +`_last_resolved_class_count == _last_static_field_class_count` already +true at its very first pass and skip `admitStaticFieldRoots()` entirely - +so the static-field-rooted node was never admitted, and the test's +"distractor chain" (500 nodes, designed to keep the search perpetually +`RUNNING` while the real target loiters in the frontier) drained to +completion instead, flipping `SearchState::COMPLETED`. + +Fix: added `_last_static_field_class_count = -1`, +`_static_field_sweep_cursor = 0`, `_static_field_sweep_cycle_truncated = +false` to `ReferenceChainsTestAccessor::reset()` +(`referenceChains_ut.cpp`), mirroring the same reset +`resetForRestart()` already performs for the production restart path +(`referenceChains.cpp:1085-1086`). Also applied the identical reset to +`resetSearchStateForTest()` (`referenceChains.cpp` - the separate, +production-code test-seam called from `javaApi.cpp:1220`, not from any +gtest fixture) for the same contract, since it had the same latent gap. + +Also added `mock_GetClassLoader` to `ReferenceChainsBfsTest`'s JVMTI +function-table mock (every fixture class reports as bootstrap/null-loader, +making the partition a no-op so existing tests' scripted class +order/indices are unaffected). + +## Verification + +- `./gradlew :ddprof-lib:gtestDebug_referenceChains_ut` - 90/90 tests pass. +- `./gradlew :ddprof-lib:gtestDebug` (full suite, not just this file) - + 188 actionable tasks, BUILD SUCCESSFUL. +- **Committed and pushed** as `a86f0dd87` on `jb/reference-chains` + (`82fec4210..a86f0dd87`), three files only + (`referenceChains.cpp`/`.h`/`referenceChains_ut.cpp`). +- **Deployed and confirmed live on the hotdog pod.** JVM PID 92618 + (started 15:29), scratch `.so` md5 `aeab8726e90e5f21e33393c3bfea043e`, + `strings` confirms `_static_field_sweep_cursor`, + `static_field_cycle_complete`, and the new `cycle_complete=%d + sweep_cursor=%d` TEST_LOG format. +- **Live mechanism confirmed correct** — see + `ev-postfix-static-field-onpod-live-verification`: cursor advances by + 512/call across laps, real edges admitted (up to 737/chunk observed, + vs. 0-1 forever pre-fix), lap-level truncation-latch behaves exactly as + designed (a lap with any truncated chunk never reports + `cycle_complete=1`, confirmed with a concrete final-chunk example). + `cycle_complete=1` has not yet been observed in any sample checked + across ~10+ minutes combined — the JDK bootstrap classlist tail is + large enough that a fully clean lap apparently hasn't happened yet; + not treated as a problem since `edges_admitted` and `sweep_cursor` + prove real forward progress every call. +- **End-to-end candidate resolution is still 0/5** despite the sweep + mechanism being confirmed healthy — see new open findings + `find-candidate1-never-tagged` and `find-candidates-234-die-before-resolution`. + +## Open follow-up (not yet actioned) + +- `STATIC_FIELD_SWEEP_CHUNK_CLASSES = 512` is a first guess, not + benchmarked against a real ~34k-class JVM's actual per-class static-field + fan-out cost. +- The app-classes-first partition is a per-call heuristic (JVMTI gives no + guaranteed stable ordering across separate `GetLoadedClasses()` calls) - + works well enough for the common case but isn't a persisted global + ordering. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-static-field-sweep-never-completes.md b/.investigations/missing-refchains-on-hotdog/nodes/find-static-field-sweep-never-completes.md new file mode 100644 index 0000000000..7586df27c7 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-static-field-sweep-never-completes.md @@ -0,0 +1,175 @@ +--- +id: find-static-field-sweep-never-completes +type: finding +status: confirmed +depends_on: [ev-postCB-onpod-live-verification, find-canary-stuck-restart-wipes-frontier] +supersedes: [] +related: [q-canary-stuck-fix-alternatives, find-static-field-sweep-cursor-fix] +tags: [root-cause, referenceChains, static-field, admitStaticFieldRoots, truncation, fixed, NEW-THIS-SESSION] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# New candidate root cause: admitStaticFieldRoots() never completes a single sweep, has no resume cursor, restarts from class 0 every attempt + +## Reasoning chain + +With C+B confirmed live (frontier growing unbroken past 80k entries, zero +`CANARY_STUCK` abandons - `ev-postCB-onpod-live-verification`), candidates +still resolve 0/5 across ~430s of combined observation. User confirmed the +live target is "static field growing, just a few hops" - i.e. an object +that should be trivially reachable, which made the continued 0/5 result +suspicious rather than "just needs more time." + +Per the code's own design comment (`referenceChains.cpp:2287-2292`): +> Static-field roots (SomeClass.staticField -> obj) are not reachable via +> IterateOverReachableObjects' root/stack-ref callbacks above ... so this +> pass would otherwise never discover an object retained only that way. + +This means an object reachable *only* through a static field is invisible +to the ordinary root-enumeration/`expandFrontier()`/rotation machinery +entirely - the general frontier growing to 80k entries is irrelevant to +finding it. The *only* discovery path is `admitStaticFieldRoots()` +(`referenceChains.cpp:2702-2822`), gated to run only when +`_last_resolved_class_count != _last_static_field_class_count` +(`:2305`), i.e. only re-attempted when the loaded-class count has changed +since the last time the sweep *completed*. + +Live trace shows this gate never closing: + +``` +static_field_phase edges_admitted=0-1 truncated=1 frontier_cap_hit=0 \ + last_resolved_class_count=33868 last_static_field_class_count=-1 +``` + +on all 275/275 `runPass` samples across the 5-minute trace +(`ev-postCB-onpod-live-verification`'s trace 5). `truncated=1` every time +means the sweep never finishes; `last_static_field_class_count` staying at +its initial `-1` sentinel forever confirms it has *never once* completed +successfully (the "sweep completed" branch at `:2337-2344` that would set +`_last_static_field_class_count = _last_resolved_class_count` has never +executed). `edges_admitted` staying at 0-1 per attempt (out of ~34k loaded +classes worth of static fields to sweep) shows each attempt dies almost +immediately. + +Code-read of `admitStaticFieldRoots()` confirms there is no resume/cursor +state: it calls `GetLoadedClasses()` fresh every invocation, builds a +holder array of ALL loaded classes, and does exactly one +`jvmti->FollowReferences(0, nullptr, holder, &callbacks, &ctx)` call over +the whole set at once (`:2812`). If that single call is truncated +(deadline or budget exhaustion inside `heapReferenceCallback`, same +mechanism as the general pass - see `:1490-1499`), the entire sweep is +discarded with `_last_static_field_class_count` left unset (`:2341-2342`, +"Left unset on a truncated sweep ... so the next pass retries instead of +wrongly treating a still-incomplete sweep as done"). The *next* attempt +starts the whole `FollowReferences` walk over from class index 0 again - +with JVMTI's class enumeration order presumably stable across calls, this +means whichever classes come after wherever the deadline/budget hits +first are structurally unreachable by this mechanism, no matter how many +times or how long it retries. + +## What this would explain + +- Why a "just a few hops" static-field-rooted candidate is never found + even after the CANARY_STUCK/frontier-wipe bug (C+B) is fixed and the + general frontier grows unboundedly - the general frontier's growth has + no bearing on this specific candidate shape at all. +- Why fixing C+B alone did not change the 0/5 outcome, despite directly + addressing the previously-confirmed restart-wipe mechanism. + +## WHY it truncates so early - CONFIRMED + +The per-pass wall-clock deadline (`_pass_deadline_ns`, +`referenceChains.cpp:2210-2212`) is derived from +`_effective_pause_target_ms`, which without urgency ramping equals +`_pause_target_ms` - itself auto-tuned at startup +(`referenceChains.cpp:367-375`): + +```cpp +// --- Pause target --- +// More available processors = the JVM can afford a slightly +// longer per-pass safepoint without impacting application +// throughput. Scale linearly: 1 core = 5ms, 4 cores = 10ms, +// 8 cores = 15ms, capped at 50ms. +long scaled_pause = DEFAULT_REFERENCE_CHAINS_PAUSE_TARGET_MS * (1 + (nprocs - 1) / 3); +args._reference_chains_pause_target_ms = std::min(scaled_pause, (long)50); +``` + +`DEFAULT_REFERENCE_CHAINS_PAUSE_TARGET_MS = 5` (`arguments.h:94`). Even at +the maximum core-scaled ceiling this is **50ms**, and this single deadline +is shared across the static-field sweep + ordinary `expandFrontier()` + +rotation for the *entire* pass (`:2205-2209`, "Shared wall-clock ceiling +for this whole call's static-field sweep, expandFrontier(), and rotation +sub-calls below"). + +`admitStaticFieldRoots()` is architecturally incompatible with this +budget: it is a single, non-resumable `jvmti->FollowReferences()` call +over a holder array of **all ~34,000 loaded classes at once** +(`:2775-2812`). The only escape hatch is the deadline check inside +`heapReferenceCallback()`, which samples wall-clock only every 4096 +callback invocations (`:1490-1491`, +`(++ctx->deadline_check_counter & 0xFFF) == 0`) - so the abort fires +somewhere inside whatever the first slice of ~4096-ish callback +invocations covers, which is evidently a small fraction of 34k classes. + +This matches the observed data precisely: `edges_admitted=0-1` on every +attempt, while the pass's *edge-count* budget was 3741 +(`effectiveBudget=3741` in the same trace's `runPass done` lines) - ruling +out the numeric edge budget as the limiter. The wall-clock deadline +(≤50ms, shared with two other sub-phases) is what kills it, and it dies +this early on **every single attempt** because there is no cursor: each +retry restarts `GetLoadedClasses()` + the whole `FollowReferences` walk +from class index 0, hitting the identical wall. Unless the leaking +static field's owning class happens to be among the very first classes in +JVMTI's enumeration order, it is structurally unreachable by this +mechanism - independent of C+B, independent of how many passes run, ever. + +This is an architecture mismatch, not a tuning knob: a per-pass STW-safety +budget designed for *incremental*, resumable graph expansion was reused +for a full-classlist sweep that has no incremental/resumable structure at +all. + +## STW confirmation (required before choosing a fix) + +User required proof the sweep is genuinely stop-the-world before accepting +a chunked/resumable rewrite as safe. Confirmed via code-read (Explore +agent): `FollowReferences()` triggers HotSpot's internal +`VMThread::execute()` on a `HeapWalkOperation` - a real JVM-internal +safepoint, not something the profiler schedules or controls +(`referenceChains.cpp:2300`, `referenceChains.h:627,1378-1381`, "the +safepoint is a side effect of that call, not something the profiler builds +or schedules"). Confirmed further by an existing +`assert(!t_inGCCallback ...)` at the top of `admitStaticFieldRoots()` +(JVMTI Heap-category calls must not run from GC callbacks, which +themselves already run at a safepoint). + +This made chunking safe by precedent: `expandFrontier()` already does +incremental, resumable batching across separate `FollowReferences` calls, +re-resolving JVMTI tags to current live objects via `GetObjectsWithTags` +(non-safepoint) immediately before each bounded call - safe against +GC/object-moving between calls because JVMTI tags persist across GC. +Chunking `admitStaticFieldRoots()`'s class list the same way is +correctness-safe, not a novel risk. + +## Fix implemented and gtest-verified — see `find-static-field-sweep-cursor-fix` + +Combined per user's explicit request: (1) a persistent resume cursor so +the sweep processes a bounded chunk of classes +(`STATIC_FIELD_SWEEP_CHUNK_CLASSES = 512`) per call instead of all ~34k at +once, advancing every call so one pathologically expensive class can never +block progress forever, with "done" now gated on a full untruncated lap +rather than a single non-truncated call; (2) an app-classes-first +in-place partition of the per-call `GetLoadedClasses()` result (via +`GetClassLoader() != nullptr`) so application/library code is swept +before the much larger JDK bootstrap classlist tail - directly targeting +the pod's confirmed "static field growing, just a few hops" (application- +code) leak shape. Full detail, code locations, and the test-isolation +regression found + fixed during verification are in +`find-static-field-sweep-cursor-fix`. `./gradlew +:ddprof-lib:gtestDebug` passes clean (188 tasks, full suite, not just +`referenceChains_ut`). **Committed (`a86f0dd87`), deployed, and confirmed +live on-pod** — see `find-static-field-sweep-cursor-fix`'s Verification +section and `ev-postfix-static-field-onpod-live-verification`. End-to-end +candidate resolution is still 0/5; two new distinct open questions +(`find-candidate1-never-tagged`, `find-candidates-234-die-before-resolution`) +surfaced downstream of this fix. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-sweep-completes-but-bfs-starved.md b/.investigations/missing-refchains-on-hotdog/nodes/find-sweep-completes-but-bfs-starved.md new file mode 100644 index 0000000000..88128a5a7b --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-sweep-completes-but-bfs-starved.md @@ -0,0 +1,81 @@ +--- +id: find-sweep-completes-but-bfs-starved +type: finding +status: confirmed +depends_on: [ev-timing-split-callback-vs-jvmti, ev-postfix-static-field-onpod-live-verification] +supersedes: [] +related: [find-candidate1-never-tagged, find-static-field-sweep-cursor-fix] +tags: [root-cause, referenceChains, bfs, expandFrontier, backlog, throughput, sweep, NEW-THIS-SESSION] +created: 2026-08-26 +updated: 2026-08-26 +--- + +# Sweep completes full lap, but BFS can't reach sweep-admitted entries through 144k backlog + +## Observation + +With the 200ms deadline (commit `fd18425c6`), the static-field sweep +completes cleanly: `cycle_complete=1` observed, `last_static_field_class_count` +advanced from 33677 → 33752, `truncated=0` on all chunks, real edges +admitted (369-2155/chunk). See `ev-timing-split-callback-vs-jvmti`. + +Despite this, candidates are **still 0/1** (only 1 candidate, klass_id=12, +unresolved). The sweep is working; the problem has moved downstream. + +## Root cause: two-hop chain architecture + +The static-field sweep admits only **one hop** past the class object. +At `referenceChains.cpp:1813`, `batch_tags` is an empty set for the sweep +call, so `heapReferenceCallback()` returns `0` (no descent) for every +referent. A static field value (e.g., an `ArrayList`) gets admitted to +the frontier and pushed to the **back** of `_pending_expand` +(`:1847`), but its elements are NOT visited by the sweep. + +`expandFrontier()` is what reaches the leaking object. It picks entries +from the **front** of `_pending_expand` (FIFO deque, `:2654`), puts +their tags into `batch_tags`, calls `FollowReferences` on them — that's +when the `ArrayList`'s elements get visited and the canary match can +fire. + +The chain is: +``` +Sweep: class → STATIC_FIELD → ArrayList [admitted to frontier, pushed to BACK of _pending_expand] +expandFrontier: processes FRONT of _pending_expand (65-87/pass) → ... → eventually ArrayList → elements → leakingObject +``` + +## The throughput bottleneck + +Live data from the pod (200ms deadline build): +- `frontierSize=144k`, `edges_admitted=65-87/pass` +- The sweep admitted the ArrayList to the **back** of `_pending_expand` — + behind ~144k entries already queued +- At 65-87 edges/pass, it would take ~1700-2200 passes to drain the + backlog and reach the ArrayList +- `_priority_expand` is drained first (`:2654`), but rotation selects + entries for priority — the sweep-admitted ArrayList isn't automatically + prioritized + +## Why only 65-87 edges/pass when budget is 3741? + +The `runPass done` lines show `effectiveBudget=3741` but only 65-87 edges +admitted per pass. This suggests the per-pass deadline (even at 200ms) is +being consumed by other sub-operations (root enum, rotation) before +`expandFrontier` gets its full share, OR `expandFrontier`'s batch sizing +(`:2680`, capped at `min(budget, _budget)`) limits how many entries are +processed per `FollowReferences` call. Not yet investigated which. + +## Possible fixes (not yet proposed to user) + +1. **Prioritize sweep-admitted entries** — push them to + `_priority_expand` instead of `_pending_expand`, so they get expanded + before the 144k backlog. The `_priority_expand` lane already exists + for rotation-selected entries; sweep-admitted entries are a natural + fit (they're the actual leak candidates the sweep found). +2. **Increase BFS throughput** — the 65-87 edges/pass is the real + bottleneck; increasing the per-pass budget or cadence would help. + +## Status + +Confirmed: the sweep works, the BFS is starved. Not yet reported to user +in full (conversation was interrupted by checkpoint). Next: report and +propose a fix direction. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-test-seam-aliasing.md b/.investigations/missing-refchains-on-hotdog/nodes/find-test-seam-aliasing.md new file mode 100644 index 0000000000..cc651569a4 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-test-seam-aliasing.md @@ -0,0 +1,70 @@ +--- +id: find-test-seam-aliasing +type: finding +status: confirmed-and-fixed +depends_on: [find-leak-tag-pool-implementation, find-klass-id-notation-mismatch] +related: [find-leak-tag-pool-implementation, find-klass-id-notation-mismatch] +tags: [fix, test-seams, klass-id, aliasing, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# Leak-tag redesign broke synthetic-klass-id test seams; fixed by real-id aliasing + +## What broke + +Since the redesign, every consumer keys candidates by REAL klass ids: +tagLeakInstances() scans the live-heap tracking table's cached_klass_id, +discovered recording resolves admitted objects' classes to the same +space. Synthetic seeded ids (987301/987302/987201/987202...) match +nothing: observed `tagged=0` every poll and +`resolved but no candidate match` for every auto-mark. None of the +pre-existing scenarios (LeakingCacheScenario, +StaticFieldGrowingCollectionScenario, ReferenceChainTrackingTest) had +been run since the redesign - the first full slow-suite run after the +redesign found all of them broken. + +## Fix: alias map in the debug seams (livenessTracker.h) + +`klassPopulationSetRepresentativeForTest()` is the only seam holding an +actual instance; it now resolves the representative's REAL klass id +(resolveKlassId) and registers a synthetic->real alias +(_test_klass_aliases, max 8, cleared by klassPopulationResetForTest()). +`klassPopulationRecordForTest()` re-routes seeding through the alias. The +seam also RE-KEYS an already-seeded synthetic entry to the real id (ring +history preserved) and creates the entry if absent (set-representative- +first is now the load-bearing order - previously a no-op when absent). + +## Scenario-side rules + +- Representative BEFORE seeding (alias must exist before seeds land). +- Per-round trend maintenance: a one-shot seeded ramp ages out of + hysteresis once real fold samples interleave (observed: the byte[] + candidate dropped right after the first interceptions, stranding the + correlated discoveries) - StaticFieldGrowingCollectionScenario's + per-round single-epoch seeding is the pattern. +- Durable fixtures: a stack-held "leak" is gate-suppressed BY DESIGN; + fixtures needing chains must retain via static fields (gcRootHolder + became static; the in-process cache test's HashMap chains are depth>=3 + so gate-legal as a local). + +## Seams-test fixture rules (ReferenceChainTestSeamsTest, this session) + +The seams test itself tripped three fixture-vs-mechanism traps, worth +recording because any future seam-driven test will hit them: +- A one-shot `runReferenceChainPass0` is not enough: admitStaticFieldRoots + sweeps loaded classes in budget-bounded chunks (~376/pass here); drive a + pass+poll loop until the event lands (bounded, e.g. 40 laps' worth). +- The target must not live in a local of the test frame: a live local IS + a STACK_LOCAL root, and the noise gate correctly suppresses the + resulting depth-0 transient chain. Reference the target only through + the static holder (all access via the static field) - the representative + jweak's own JNI-global root then makes the chain gate-legal even before + the static sweep reaches it. +- A unique nested class as target: java.lang.Object (or any common class) + collides with a real population entry, which outranks the seeded ramp. +- `new Object()`/stack-held targets make the chain event depend on + discovery ORDER (root enum's stack-local win vs the sweep's later + static win) - the upgraded root-kind now closes that gap + (find-depth0-durable-root-upgrade-gap), but deterministic fixtures + should still avoid the order dependence. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-threadloop-presleep-blocks-back-to-back.md b/.investigations/missing-refchains-on-hotdog/nodes/find-threadloop-presleep-blocks-back-to-back.md new file mode 100644 index 0000000000..2e81331793 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-threadloop-presleep-blocks-back-to-back.md @@ -0,0 +1,61 @@ +--- +id: find-threadloop-presleep-blocks-back-to-back +type: finding +status: confirmed +depends_on: [] +supersedes: [] +related: [find-cpu-pain-budget-starves-canary-passes] +tags: [bug, referenceChains, threadLoop, cadence, canary, design-contradiction] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# `threadLoop()` sleeps `cadence_ns` unconditionally BEFORE `shouldRunPass()` runs + +## Reasoning chain + +`threadLoop()` (`referenceChains.cpp:637-802`) contains two `OS::sleep` +call sites per iteration: + +1. `:736-738` — `if (cadence_ns > 0) { OS::sleep(cadence_ns); }`, with + **no other guard**, sitting right after the comment block at `:728-735` + that claims "Run passes back-to-back... the cadence sleep is only kept + as a fallback for an idle search... to avoid busy-waiting." This sleep + runs on *every* iteration, unconditionally — including ones where a + canary search is actively converging and `shouldRunPass()` (called + later, at `:771`) would return true for the "run immediately" canary + bypass. +2. `:775-776` — `if (!should_run && cadence_ns > 0) { OS::sleep(cadence_ns); + ...}`, which correctly matches the comment at `:772-774` ("Only sleep + when idle... skip the sleep to run passes back-to-back"). + +Only the second sleep is actually conditioned on `should_run`. The first +one contradicts its own neighboring comment: it fires before `should_run` +is even computed, so a canary-bypass-eligible pass still incurs one full +`cadence_ns` sleep per iteration regardless. On an idle iteration +(`should_run == false`), **both** sleeps fire back to back, doubling the +idle period to ~2×`cadence_ns` instead of the intended ~1×. + +## Evidence + +Read directly, `referenceChains.cpp:690-801` (see exact line numbers above). +No live-log evidence needed beyond the source itself — this is a structural +control-flow reading, not a runtime behavior that varies by heap state. + +## What this rules out + +Nothing on its own — this is an independent, always-reproducible ~1-2s/ +iteration ceiling on pass frequency, not something that varies with heap +size or debt like `find-cpu-pain-budget-starves-canary-passes`. It cannot, +by itself, explain a 45-second silent window (max ~2s per idle iteration), +but it does mean that even once the pain-budget gate clears, the canary +bypass is still not truly "back-to-back" as designed — every pass, canary +or not, pays at least one `cadence_ns` sleep first. + +## Not yet done + +Not fixed. This is a design-level question (should the first sleep be +removed, or moved after the `should_run` check, or merged with the second +sleep block into one guarded sleep) that should be proposed to the user +before editing `threadLoop()`'s control flow, since it's core BFS-thread +scheduling shared by every reference-chain search, not just canary ones. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-tier1-tail-starvation.md b/.investigations/missing-refchains-on-hotdog/nodes/find-tier1-tail-starvation.md new file mode 100644 index 0000000000..30d9687677 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-tier1-tail-starvation.md @@ -0,0 +1,82 @@ +--- +id: find-tier1-tail-starvation +type: root-cause +status: confirmed +depends_on: [find-anchor-tail-starvation, ev-leaktag-onpod-round14-results] +related: [find-fresh-lane-verification] +tags: [root-cause, referenceChains, tier-1, tail-starvation, containers, round-14, NEW-THIS-SESSION] +created: 20260915 +updated: 20260916 +--- + +# Find: tier-1 tail starvation (round-15 diagnosis — admitted-at-tail container never walked) + +Parent: find-anchor-tail-starvation (round 13, same invariant at index +scale). Round 14 shrank the cohort (containers 1633) but the invariant +`coverage = rate × lifetime ≥ cohort` STILL fails at tier scale, now with +cleaner measured numbers. + +## The mechanism (all measured on pod rz992, round-14 build, 20:22-20:53Z — +raw numbers in ev-leaktag-onpod-round14-results.md, deploy + design in +ev-leaktag-onpod-round14.md) + +1. Anchor index order = admission order; statics admit in SWEEP order = + loaded-class order (app-partitioned, ProfileAnalyzer at 24627/33270). +2. The wrapper's holder class is late in the class list ⇒ the wrapper is + admitted mid/late in whatever search is live when the sweep crosses it + (observed: pass 43-44), landing at the anchor-index TAIL ⇒ container + ordinal ~LAST (1634 of 1633). +3. Tier-1 fair cursor walks 16/pass from ordinal 0 ⇒ wrapper walked at + admission_pass + ~102. +4. Search lifetimes are 44-75 passes (frontier fills 2.5-7.5k/pass — noise + seeding ~650 inserts/min + sweep admits + expansion; cap 250k abandon). +5. ⇒ wrapper walk owed at pass ~102+ > lifetime 44-75 ⇒ **deterministic + miss by 30-50 passes**, every search, every lap. +6. Compounder: every search today is a CANARY CHASE (candidate = the [B + leak class, 0/1 found). The candidate is only "found" via a resolved + chain, which needs the wrapper walk (see 5), so the chase can never + succeed — while its exponential backoff (16×ema≈3s/pass) slows the + passes that would reach the wrapper. Self-sustaining deadlock. + +## Why round-14's own guarantee was insufficient + +Round-14 guaranteed "walked within ceil(cohort/budget) passes OF ADMISSION" +(ceil(1633/16)=102) — the guarantee is correct as stated; the missing +assumption was that a search LIVES ≥ 102 passes after the wrapper's +admission. Round-13 measured ~190-pass lifetimes; round-14 observed 44-75 +(the fill rate rose 2-5x: leak-accumulator seeding noise + faster sweep +admits). The scale invariant (lifetime × rate ≥ cohort) must be asserted +against the MEASURED lifetime, not the round-13 one — same lesson as the +circle review: measure the invariant whole, per deployment. + +## Fix (round 15): fresh-admission priority + +Walk anchors admitted since the last collector pass FIRST (newest-first), +at least the container-shaped ones. Admission at pass N ⇒ walk at N+1. +STW-neutral (same 16-walk budget, reordered). Breaks the deadlock +end-to-end: wrapper walk ⇒ subtree expansion admits leak-tagged chunks ⇒ +chain resolves ⇒ canary candidate found ⇒ chase exits ⇒ event emitted. + +Note: fresh-priority must not starve the fair tiers permanently — fresh +admits are bounded by the sweep's admit rate (~16-64/pass of containers), +so the fresh lane drains within a few passes and the fair cursors resume. +Shape-independent fresh-priority would flood the budget with fresh +String/other statics; prefer fresh+container (or fresh with shape lag: +classify-then-walk next pass). + +## Secondary observations (record, don't fix now) + +- Search lifetime regression driver candidates: seeding noise (fanout=1, + round-13 node) ~10-20% of fill; the rest is sweep admits + expansion. + If fresh-priority alone doesn't close the loop, the seeding noise is + the next lever (also a round-13-documented correctness bug). +- reconcile classification log fires once per class per JVM; early-lap + windows rotate out of the log buffer fast (67k lines ≈ 4 min) — capture + with `kubectl logs -f` streams, not --since windows, for early-lap + evidence. Buffer rotation burned ~20 min of round-14 verification time. +- Sweep gate: opens only while the loaded-class count changes (app loads + lambdas in waves); cursor resets to 0 when count shrinks below cursor + (lambda unloading) — laps complete in ~2-10 min during churn. +- Old-JVM debt-drain arithmetic sanity: PainBudget drains at exactly its + spec rate (30ms/s at refill 0.03); the "slow drain" illusion was buffer + window misestimation. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-togcroot-orphaned-slot-stranding.md b/.investigations/missing-refchains-on-hotdog/nodes/find-togcroot-orphaned-slot-stranding.md new file mode 100644 index 0000000000..1495a32f8f --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-togcroot-orphaned-slot-stranding.md @@ -0,0 +1,91 @@ +--- +id: find-togcroot-orphaned-slot-stranding +type: finding +status: fixed +depends_on: [q-togcroot-acceptance-paths, find-canary-lane-backoff-design] +related: [find-per-tid-qualification-design, find-leak-tag-pool-implementation] +tags: [fix, root-cause, canary, slots, flakiness, NEW-THIS-SESSION] +created: 20260901 +updated: 20260901 +--- + +# ToGcRoot failure root cause: discovered instances stranded when the candidate ages out of the poll list + +Answers q-togcroot-acceptance-paths. The 124-pass mystery was NOT pass +starvation - it was slot stranding. + +## The failure mechanism (log-proven, togcroot4 = failing run) + +- Candidate ChainLink (klass_id=4, both id spaces agree) admitted into + slot 0 at the first poll; qualifies via the 15-epoch seeded ramp only. +- The walk is slow: 66 of 124 passes admit ZERO edges; only 8 of 69,001 + ChainLinks are ever admitted (the elementData expansion admits its + first 8 children, budget truncates mid-array, and the holder is never + re-walked - the leak-accumulation fanout ranking is drowned by noise + parents: 97k edges under noise klasses vs the real holder's fanout 8; + all 821 Tier-2 selections went to noise, fanouts up to 56,329). +- The seeded ramp ages out of the ring at ring_fill=22 (slope decays to + ~0.5, consecutive_positive resets to 0, line 32028) - klass 4 never + returns in selectLeakCandidates() again (199 zero-candidate polls). +- The 8 discoveries are recorded the pass AFTER the drop-out (line + 50654, discoveredCounts=[8,...] for the remaining 116 passes) - but + the discovered-chain loop iterates only the CURRENT poll's + candidates. The slot persists by design ("can still be found there") + but NOTHING ever read it again. No chain is built -> test fails after + grace. Green runs = crawl lucky enough to discover while the seeds + still held. + +## Fix 1 (user-picked "1"): slot-driven chain building (load-bearing) + +buildDiscoveredInstanceChains(klass_id, search_gen) extracted from +pollWatchedTargets' discovered loop; two call sites: the per-poll- +candidate loop (fresh chains for still-qualifying klasses) + an orphan +sweep over persistent slots absent from the current poll's candidates +(idempotent via the source_search_ns cache check; skip still-qualifying +slots to keep log volume unchanged). gtest +OrphanedSlotBuildsDiscoveredChainsAfterCandidateDropsOut; 551 green. +VERIFIED: full slow suite green twice consecutively at load 28.7 and +53 (was failing at load 3). + +## Fix 2 (user-picked "then 2"): persistent allocator thread + +One "togcroot-leak-allocator" thread reused across all 16 rounds; +tid seeds use its tid so seeds and real folds agree. HONEST CAVEAT: the +candidate STILL ages out mid-run in the verified green run (54 +zero-candidate polls, 52 sweep firings) - shared-JVM GC-epoch noise +resets consecutive_positive, so sustained organic qualification is +inherently fragile; fix 1 remains the guarantee. Improvement of fix 2 +is the active thread (see STATE.md). + +## Fix 2 improved: seed-magnitude cliff (verified) + +The residual aging-out was a SEED artifact, not epoch noise: the seeds +planted epoch*10 (10..150) in the same 30-slot ring the organic +generation-count ramp (+1/epoch, values 1..15 by test end) later fills; +when real samples displaced the seeded tail, hasQualifyingGrowth()'s +third-window slope went NEGATIVE (-20..-40) and consecutive_positive +never recovered - a "transition cliff". Rescaled the seeds to the +organic ramp's own magnitude (count = epoch, tid counts = epoch): the +ring series becomes [1..15, 1, 2, 3, ...] and every third-window pairing +across the transition stays positive - consecutive_positive grew +organically to 19 through ring_fill 28, zero candidate drop-outs +(zeroCands 54 -> 2), zero negative slopes, chains built by the ordinary +per-poll path, the orphan sweep UNUSED (pure safety net, as designed). +Two incidental finds: (a) the fix-2 test code had a compile-scope bug +(allocator referenced from finally while declared inside try) and the +two earlier "fix 2 verified" suite runs had executed STALE classes +(Gradle testSlowDebug UP-TO-DATE quirk) - fix 2 only actually executed +after the hoist + --rerun; fix 1's verification stands (native code was +in those runs). (b) the correlation scenario's seeds grow unboundedly +per round (hysteresisEpoch++ with epoch*10) so they never cliff - no +change needed there. Suite green post-fix at load 9-15 and 28-53; the +single correlation-test failure en route was the known 4-crawl-pass +knife-edge at load 31 (green on rerun with identical code). + +## Code landmarks + +buildDiscoveredInstanceChains in referenceChains.cpp (~4690, before +recordDiscoveredInstance); orphan sweep in pollWatchedTargets (after +the per-candidate loop, "Orphan fix" comment); declaration in +referenceChains.h next to recordDiscoveredInstance; gtest at the +PollWatchedTargetsTest fixture's tail. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-urgentoom-null-fn-mislabel.md b/.investigations/missing-refchains-on-hotdog/nodes/find-urgentoom-null-fn-mislabel.md new file mode 100644 index 0000000000..9909d15eba --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-urgentoom-null-fn-mislabel.md @@ -0,0 +1,69 @@ +--- +id: find-urgentoom-null-fn-mislabel +type: root-cause +status: confirmed +depends_on: [] +related: [find-test-seam-aliasing, find-wrapper-demotion-self-parent] +tags: [root-cause, test-seams, autoTuneDefaults, GetAvailableProcessors, gtest, mislabel-lesson, NEW-THIS-SESSION] +created: 20260916 +updated: 20260916 +--- + +# Find: UrgentOOM SIGSEGV — a null-fn test-mock gap mislabeled "pre-existing" for 5 weeks (round 16) + +Parent: none (test-infrastructure root cause; discovered during round-16 +verification). Evidence: the stash-run logs (HEAD repro) + the fix commit +4afe870a2; the correction is also recorded in STATE.md's Corrections +section. + +## The defect + +`ReferenceChainTracker::autoTuneDefaults()` (added facdc70c0, +2026-08-17) calls `jvmti->GetAvailableProcessors()` from `start()` +whenever LivenessTracker reports a max heap > 0. The +`SearchRestartTest` fixture (5adb32831, 2026-08-12 — 5 days EARLIER) +wires only SetEventNotificationMode/GetLoadedClasses/FollowReferences/ +IterateOverReachableObjects — GetAvailableProcessors stayed a null table +entry. `UrgentOOMProjectionBypassesCandidateGate` is the only test in +the file that sets a max-heap (that is its subject), so it null-crashed +at `tracker->start(args)` — **before its subject ever executed** — on +every full-suite run for 5 weeks. Its actual subject (OOM-urgency +bypass of the candidate gate) was correct the whole time: fixed by +wiring `mock_GetAvailableProcessors` (fixed 1, deterministic auto-tune), +the test passes and `:ddprof-lib:gtestRelease` is green with NO +exclusions for the first time since 2026-08-17. + +## The mislabel, and the lesson it teaches + +Round notes recorded this crash as "pre-existing gtest SIGSEGV on clean +HEAD — separate issue" and deferred it for multiple rounds. "Clean +HEAD" was this investigation branch's own tip — every commit on it is +investigation work. The only thing the repro ever proved was "not from +today's diff". Bisect (git log -S) found both introducing commits in +minutes: the test 2026-08-12, the null call 2026-08-17 — OUR regression, +exactly the crash class I had just root-caused an hour earlier in the +same file (a minimal-mock fixture + production code adding a new JVMTI +call = null function pointer; the same class struck twice in one day: +the at-risk-FIFO residue crashed `WithoutGenerationsSignalRestart\ +StaysUnconditional` through the reset() seam gap below). + +LESSONS (both recorded in STATE.md Corrections): +1. "Reproduces on HEAD" proves "not from today's diff" — nothing more. + Bisect to the introducing commit before calling anything + pre-existing. +2. A crashing test is never "separate debt" — it is LOST SIGNAL. This + one carried a valid assertion (the urgency bypass) that went + unverified for 5 weeks while the suite crashed past it. + +## The sibling seam gap (fixed the same round) + +The new quota test's saturated-FIFO residue crashed the NEXT fixture's +`runPass` (null GetObjectsWithTags in SearchRestartTest's minimal +mock): the `reset()` test seam predates the B' at-risk FIFO and was +never given its clears (nor the round-15 fresh queue) — prior tests +survived only because later tests' own runPasses drained the residue +with the full Bfs mock. reset() now clears FIFO + set + klass counts + +counters + fresh queue; the quota test also drains its saturation. +Same family as find-test-seam-aliasing (singleton state leaking across +tests) — but inverted: there a stale mark reordered selection, here a +missing clear let state OUTLIVE the test that created it. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-wrapper-demotion-self-parent.md b/.investigations/missing-refchains-on-hotdog/nodes/find-wrapper-demotion-self-parent.md new file mode 100644 index 0000000000..53d92854a7 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-wrapper-demotion-self-parent.md @@ -0,0 +1,103 @@ +--- +id: find-wrapper-demotion-self-parent +type: root-cause +status: confirmed +depends_on: [find-fresh-lane-verification, ev-leaktag-onpod-round15-results] +related: [find-urgentoom-null-fn-mislabel, find-round16-endgoal-verification, find-refchains-log-flood-configurable] +tags: [root-cause, referenceChains, improveChain, self-edge, at-risk-fifo, round-16, FIXED, NEW-THIS-SESSION] +created: 20260916 +updated: 20260916 +--- + +# Find: wrapper frontier entry demoted to a self-parented chain-attached entry (round 16 — FIXED + VERIFIED ON POD: wrapper root-attached, walked, leak-tag interceptions, leak-correlated events) + +Parent: find-fresh-lane-verification (the round-15 verification that +exposed this). Evidence: ev-leaktag-onpod-round15-results.md (measured +live on pod cgdtx, 05:44-06:23Z 2026-09-16, streams in +/tmp/r15_stream.log, r15_stream2.log, r15_tail.log). Implementation: +commit 4afe870a2, gtests 118/118. + +## STATUS: implemented (awaiting user deploy + pod verification) + +- Fix A: improveChain()/reparentToDurableRoot() refuse parent==tag + (self-edge — mutex == this), counted in + FrontierTable::selfEdgeGuardSkips (2 refusals per delivered callback + edge; logged per pass as self_edge_skips in rotation_candidates). + With the guard, the wrapper's entry stays root-attached → the + collector's parent_tag==0 filter keeps selecting it. +- Fix B: B' at-risk FIFO entries carry their klass (AtRiskAnchor); + per-class quota 64 (a full 1024 FIFO holds ≥ 16 distinct classes); + drain/requeue maintain occupancy exactly; quota drops counted + logged + (static_anchor_fifo_quota_drops_total). The wrapper's pushes land + while the floods self-throttle. +- Pod verification channels: self_edge_skips climbs (2 per wrapper + subtree walk per pass), pushAtRiskStaticAnchor klass_id=28366 lines + with fifo_size ≪ 1024, wrapper probe returns to parent=0 root_kind=8, + walkStaticFieldAnchors … SynchronizedRandomAccessList lines resume, + then the interception cascade (seen_as=1, entry.leak_tag, chains with +targetTag=leak tag, canary resolves). + +## The defect shape (all measured, not deduced) [historical record] + +In search #2 the wrapper admitted ROOT-ATTACHED (parent=0 root_kind=8) +and was walked once. In every LATER search, the LEAK_BUFFER probe at each +holder-class crossing returns: + + wrapper_tag=2469/2679/2853 frontier_found=1 parent= + root_kind=0 state=1 leak_tag=0 depth=1-3 referrer_klass=28366 + +- The entry is CHAIN-ATTACHED (parent != 0) with **parent equal to the + wrapper's own tag** — a self-edge. +- The collector's eligibility filter (parent_tag == 0, root_kind + STATIC_FIELD/JNI_GLOBAL) therefore skips it forever within the search + → zero wrapper re-walks across 3 holder-class crossings (sweep laps + re-cross index 24611-24874 every ~10-20 min). +- maybeUpgradeRootAttachedRootKind refuses parent_tag != 0 by design + (B' comment) — so even the sweep's own re-encounter cannot restore the + root-attached shape. + +## The designed repair is dead too: B' FIFO cap-pinned by floods + +The at-risk push (sweep's static edge onto an already-admitted +chain-attached entry → pushAtRiskStaticAnchor → FIFO drain walks it) +is exactly the repair for this demotion. Measured: the FIFO sits at its +1024 cap, flooded by klass 1 (1396 pushes), klass 215 (1063), klass 1733 +(fifo_size 988-1024) — the wrapper's class 28366 got ZERO pushes through +(dropped at the cap check). Round-10 already observed "a cap-pinned +at-risk flood"; the per-klass counts are now measured. + +## Why this blocks the end goal + +The wrapper's chunks now carry leak tags (tagged=36-48, re-established +after each restart by the poll loop's correlate-not-retag state machine — +that machine itself verified working: no tag → SetTag branch observed +re-growing after releaseSearchTags collapses the count 137→13). ONE +wrapper walk while the chunks hold leak tags fires the heapReferenceCallback +leak-tag interception (seen_as=1) → frontier entry with leak_tag set → +chains built with targetTag=leak tag → correlateAdmittedLeakTag → the +canary's leak-tag target resolves → ReferenceChain events correlated with +HeapLiveObject.leakTag. Until then, the emitted events carry frontier-tag +targets (leak_correlated=0) — real chains, no leak correlation. + +## Root-cause candidates for the demotion (round-16 work, in order) + +1. **A self-edge insert: some path inserts/overwrites the wrapper's + entry with parent_tag == its own tag.** Suspects: the leak-accumulator + fanout/seed insert (trackLeakAccumulation — the fanout-insert path + round-13 flagged as fanout=1 noise; a parent==child pair would be its + degenerate case), improveChain's shallow→deep replacement (the + DemotionPush test shape), or reparentToDurableRoot. Repro: a gtest + admitting a root-attached static holder, then driving the suspects, + asserting parent stays 0/root_kind stays 8 while the entry is live. +2. **B' FIFO per-class quota**: cap each klass's occupancy (or evict the + oldest same-klass entry on push) so klass-1/1733/215 floods cannot pin + the 1024 cap. Independent of (1) — either fix alone unblocks the + re-walk; both cover each other's failure mode. + +## Note on the probe's parent==tag reading + +frontier_found=1 with parent==tag is what the probe READS — whether the +table literally contains a self-parented entry for the wrapper or the +lookup is landing on a related entry (the wrapper's child keyed +colliding) is part of the round-16 repro: dump the entry for the exact +tag in the gtest before assuming the self-edge is literal. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/find-wrapper-not-in-anchor-tier.md b/.investigations/missing-refchains-on-hotdog/nodes/find-wrapper-not-in-anchor-tier.md new file mode 100644 index 0000000000..db37c32308 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/find-wrapper-not-in-anchor-tier.md @@ -0,0 +1,86 @@ +--- +id: find-wrapper-not-in-anchor-tier +type: root-cause +status: confirmed +depends_on: [ev-leaktag-onpod-round12, find-anchor-live-feed-design] +related: [find-anchor-holder-eviction, find-option-c-descend-walk-design, find-rotation-resize-blindspot] +tags: [root-cause, referenceChains, anchor-tier, collector-lottery, wrapper, SynchronizedRandomAccessList, NEW-THIS-SESSION] +created: 20260915 +updated: 20260915 +--- + +# ROOT CAUSE: wrapper is root-attached STATIC_FIELD but the collector cursor never reaches it + +## The wrapper class + +Bytecode inspection confirmed: `LEAK_BUFFER = Collections.synchronizedList(new ArrayList<>())`. +The wrapper is `Collections$SynchronizedRandomAccessList`, NOT +`UnmodifiableRandomAccessList` (the round-12 diagnostic checked for the +wrong class — fixed). + +## What the deployed diagnostics show + +1. **The sweep DOES process the wrapper.** The sweep's STATIC_FIELD edge + hits the wrapper as an already-admitted entry (`*tag_ptr > 0`). The + wrapper was admitted by the BFS before the sweep reached it. + +2. **The wrapper IS root-attached STATIC_FIELD (root_kind=8).** The sweep + log shows entries with `root_kind=8 state=0 found=1` — already + root-attached. The wrapper is among the 29 unique klass_ids with + root_kind=8 in the sweep's already-admitted encounters. + +3. **The collector cursor never reaches the wrapper.** + `collectStaticFieldAnchorsForRotation` scans the frontier table in + tag order with a wrapping cursor, selecting 4 entries per pass + (`STATIC_ANCHOR_ROTATION_BUDGET=4`) + 16 FIFO-drained = 20 total. + The frontier has 228k entries. The walked root_kind=8 anchors are all + in tag range 5890-6030 — the cursor is stuck in a small range. The + root_kind=8 entries span tags 23-89608. At 20/pass, ~3s/pass, covering + 228k entries takes ~9 hours. The candidate flaps out (terminal) long + before the cursor reaches the wrapper. + +4. **The `SynchronizedRandomAccessList` never appears in anchor walks.** + Zero matches across the entire log. The wrapper IS in the anchor tier + (root_kind=8) but the collector's wrapping cursor lottery never + selects it. + +5. **The round-11 "wrapper was walked" observation was a decoy.** The + `UnmodifiableRandomAccessList` walked at 21:10 UTC (round 11) was + held by `io/netty/util/NetUtil`, NOT by `ProfileAnalyzer`. The actual + LEAK_BUFFER wrapper was never walked. + +## The bottleneck + +`collectStaticFieldAnchorsForRotation` does a full frontier table scan +(228k slots) with a wrapping cursor, selecting only 4 root-attached +STATIC_FIELD/JNI_GLOBAL entries per pass (`STATIC_ANCHOR_ROTATION_BUDGET=4`). +The scan is O(frontier_size) per selection — most slots are NOT +root-attached, so the cursor advances past thousands of non-matching +slots to find each anchor. With 228k entries and ~29 root_kind=8 +anchors scattered across tags 23-89608, the cursor takes hours to +sweep the whole table. + +The candidate is STABLE in this recording (klass_id=9, tid=349994, +33 tagged at 78MB, no flap-out). The collector has unlimited time. +But at 4 anchors/pass, ~3s/pass, covering 228k entries takes ~6+ +hours. The cursor IS advancing — just far too slowly. + +## Fix options + +1. **Raise `STATIC_ANCHOR_ROTATION_BUDGET`** (4→16 or higher): nearly + free under the GOTW floor (the GOTW call resolves all 20 in one shot), + but the cursor still scans the whole table. + +2. **Index root_kind=8 entries**: maintain a separate list/vector of + root-attached STATIC_FIELD entries, so the collector iterates only + the anchor set (O(anchors) instead of O(frontier_size)). The scan + becomes O(29) instead of O(228k). + +3. **Prioritize anchors by leak-tagged children**: walk anchors that + have leak-tagged descendants first. The wrapper holds the [B] chunks + (klass_id=4, 33 tagged at 78MB) — it should be walked before any + machinery static. + +Option 2 is the structural fix. Option 3 is the targeted fix. Both are +needed: option 2 makes the collector fast enough, option 3 makes it +walk the right anchor first. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/hyp-regression-of-five-fixes.md b/.investigations/missing-refchains-on-hotdog/nodes/hyp-regression-of-five-fixes.md new file mode 100644 index 0000000000..9bb01ff3b9 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/hyp-regression-of-five-fixes.md @@ -0,0 +1,67 @@ +--- +id: hyp-regression-of-five-fixes +type: hypothesis +status: refuted +depends_on: [ev-source-poll-vs-callback, ev-marker-tag-arithmetic] +supersedes: [] +related: [find-marker-tag-slot-index-mismatch] +tags: [regression-check, sibling-investigation, refuted] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Hypothesis (REFUTED): this is a regression of one of the five fixes from the `reference-chains` investigation + +## Reasoning chain + +The outer symptom — `buildCanaryChainEvent(candidate=0) -> 0` repeating +forever with `needRefresh=1` — is *literally the same log line* recorded in +the sibling investigation `.investigations/reference-chains/` +(`ev-integration-test-still-zero`), which was declared resolved by five +fixes. So the obvious first guess was a regression. + +It is not. The five fixes, all confirmed and committed on this branch +(`209336ea1`, `8114019c2`), are: + +- `find-classtag-stability-fix` — shared `ClassTagAllocator` / stable `class_tag` +- `find-test-seam-classtag-mint` — `mintStableClassTagIfNeeded()` in the + `klassPopulationSetRepresentativeForTest()` seam +- `find-urgent-signal-latch` — hysteresis on `isUrgent()`/`hasLeakSignal()` +- `find-runpass-completion-gating` — gate `COMPLETED` on + `_watched_leak_klass_count == 0` +- `find-livenesstracker-generation-sync` — sync + `LivenessTracker::_last_class_map_generation` in `initialize()` + +None of them touches the correspondence between a marker tag's encoded +slot and the array index used to read it back, nor the one-shot +`if (_candidate_count == 0)` pre-tagging gate. Those two are untouched +original code. + +Shape difference from the sibling evidence: the local integration test +that drove those five fixes exercised a **single** candidate, so index 0 +was trivially self-consistent with slot 0 and the defect was invisible. +Production selected multiple candidates (3 latched, 5 currently offered), +which is the first time slot != index could occur. That multi-candidate +path was never exercised by a test. + +## Evidence +- `evidence/ev-source-poll-vs-callback.md` +- `evidence/ev-marker-tag-arithmetic.md` +- `.investigations/reference-chains/STATE.md`, `INDEX.md`, + `evidence/ev-integration-test-still-zero.md` + +## What this rules out +- Reverting or re-auditing the five committed fixes as a remedy. +- Re-deriving any of the five root causes — see the sibling investigation, + which is `status: done`. + +## Caveat on one sub-claim +The "single candidate in the local test" reading is the session's +inference from `ev-integration-test-still-zero` (which logs +`candidate[0]`, `klass_id=987302`, `class_name=[B` and a consistently +assigned `marker_tag`); that evidence file does not record the +`candidate_count` value or the numeric marker tag, so slot-0/index-0 +self-consistency there is inferred, not directly measured. Worth +re-checking against the raw test log +(`build/logs/20260824-100906-_ddprof-test_testSlowDebug.log`) before +relying on it. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/hyp-warmup-transience.md b/.investigations/missing-refchains-on-hotdog/nodes/hyp-warmup-transience.md new file mode 100644 index 0000000000..5df5ad69f9 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/hyp-warmup-transience.md @@ -0,0 +1,41 @@ +--- +id: hyp-warmup-transience +type: hypothesis +status: refuted +depends_on: [ev-livelock-pod-logs, ev-jafar-zero-refchain-events] +supersedes: [] +related: [find-marker-tag-slot-index-mismatch, find-canary-search-cannot-terminate] +tags: [warm-up, transient, refuted] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# Hypothesis (REFUTED): zero events is just warm-up — the paced BFS hasn't had time yet + +## Reasoning chain + +The first reaction on seeing `datadog.ReferenceChain: count 0` in a +recording from a JVM that had only been up ~20 minutes was that this is +expected: the search needs a leak signal to accumulate over several GC +epochs, then runs a deliberately slow, pause-budgeted BFS. + +Refuted by three independent readings of the pod logs: + +1. The leak-detection hysteresis had long since fired for the very klass + under investigation: `klass_id=8 … consecutive_positive=11 required=3` + (and `klass_id=160` likewise at 11). `heapFloorRising … floor_rising=1`. +2. The search was not "still working" — it was repeating a bit-identical + iteration. 143 identical iterations over 25 minutes, with + `0/3 candidates found` never moving, and exactly one distinct + `canary candidate[…]` log line in the whole window. +3. The state machine has no exit at 0/N + (`find-canary-search-cannot-terminate`), so more time provably cannot + help. + +## Evidence +- `evidence/ev-livelock-pod-logs.md` +- `evidence/ev-jafar-zero-refchain-events.md` + +## What this rules out +- "Wait longer and re-check" as a next step. Do not spend another + observation window on this pod expecting events without a code fix. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/meta-circle-review.md b/.investigations/missing-refchains-on-hotdog/nodes/meta-circle-review.md new file mode 100644 index 0000000000..1aeb907f07 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/meta-circle-review.md @@ -0,0 +1,106 @@ +# Meta-review: are we running in circles? (weeks burned, leak still unreachable) + +Date: 2026-09-15, requested by user after round-13 root cause + fix-D pick. + +## Verdict + +**Convergent in facts, circular in strategy.** + +Not literal circles: every round found a real, distinct, verified defect +(30+ fixed bugs in the INDEX; engine 1000x faster; every mechanism now +probe-proven working). The state monotonically advances. + +But rounds 9→13 — five pod rounds, several weeks — all attacked the SAME +invariant failure inside the SAME architecture: + + coverage = walk_rate x search_lifetime >= anchor_population + +- Round 9: tier composition wrong (holder evicted → fix B' live feed) +- Round 10: sweep never completes (restarts) → fix A (search lifetimes) +- Round 11: collector 4/pass cursor lottery → fix budget+index+priority +- Round 12 (B+C): rate fixed 8x, order unchanged +- Round 13: measured the invariant whole: 28k population vs ~4k coverage + (21 walks/pass x ~190 passes, admission-order selection) — the holder at + position ~12-21k is deterministically unreachable. + +Each fix tuned a parameter of the funnel (tier composition, FIFO, budget, +iteration cost, priority). None questioned the funnel itself. The round-13 +measurement (rate x lifetime vs population) should have been the FIRST +analysis, before any of rounds 9-12 — it was computable from pod logs at +any point. + +## Why the funnel cannot win (the structural statement) + +The architecture explores a ~12M-object heap through a ~3741-edge/pass +budget. A static-held leak requires covering the static-holder population +(~28k anchors, ~6.6k distinct holder classes — MEASURED round 13: 1092 +distinct classes per 5.7k-class sweep window) within a ~190-pass search +lifetime, at ~46 anchors/pass ceiling (81 edges/anchor, measured). Even the +strongest tuning (D: cohort shrink + rotation) needs multiple search +lifetimes (hours), and its collection-shaped cohort size is UNMEASURED — +the leaf filter accounts for only ~17% of admits (String/[I/[B/INST_CODE +top admit classes). D's arithmetic is not established. The pattern of +"plausibly sufficient, then a new scale surprise" has now repeated 4 times. + +## The structural exit (option C, previously deprioritized) + +A one-shot whole-heap pass at search start eliminates the funnel for +discovery: class → static → holder → ... → leak-tagged chunk admitted in +one bounded STW, interception fires immediately, chains complete. Our own +standards survey (find-attribution-standards-survey) found that JFR's +production LeakProfiler does exactly this shape — fresh full BFS at every +emit, first-path-wins, bounded. The design rejected it for pause-time, but: +searches are leak-gated and rare; the pod tolerates seconds of STW; and the +frontier-cap interplay is a designable problem, not a wall (e.g. one-shot +admission policy: root-attached statics + leak-tag interceptions + the +parent chain of each intercepted object only — not the full graph). + +## Process failures that cost the weeks (fix these regardless) + +1. **Scale invariants were never measured as a whole.** rate x lifetime vs + population was computable from pod logs at any round; it was measured + only in round 13, after four tuning rounds. +2. **Local tests are toy-scale.** gtests verify mechanics at dozens of + entries; every scale defect (frontier caps, FIFO depth, index size, + sweep order, restart cadence) could only manifest on the pod — one + discovery per deploy cycle, deploys done only by the user, log windows + rotating in minutes. Need a scale gtest: synthetic 30k classes / + 28k-anchor index / 250k frontier cap / restart loop, asserting the + coverage invariant itself. +3. **Fixes were validated as "mechanically working"** (walk rate up, index + fast, budget 16) rather than against the end-to-end success criterion + (a leak-tagged chunk intercepted from a static-held holder). Every + round's success gate should be the end-to-end outcome, or a measured + invariant, never an intermediate rate. + +## Decision needed + +D (cohort shrink + rotation) was picked before this analysis. Given the +measured ~6.6k distinct holder classes vs 4k coverage/search and the +unmeasured collection cohort, D risks being the 5th tuning round. C exits +the failure class entirely. Recommend: design C now (with the minimal +admission policy to respect the frontier cap), keep D's cheap parts +(leaf-filter index exclusion) only if they fall out of C's design; add the +scale gtest before any next deploy. + +## DECISION (user, 2026-09-15): C is REJECTED — hard constraint + +"I can not do whole heap pass as that will do STW for god knows how many +seconds!!!" — the per-call ~50ms / cumulative 500ms-per-sec pause budget is +non-negotiable; no whole-heap walk in any form. The remaining bounded-STW +levers, all inside the existing funnel: +1. Cohort filtering (MEASURE first — the collection-shaped cohort is + unmeasured; leaf classes ≈17% of admits, measured). +2. Selection order (rotation/recency — note: the wrapper's index position + is ALREADY a slow lottery: it is admitted at sweep cursor 25301, so its + position depends on the cursor value when each search starts; only ~14% + of search-start cursor positions put it inside the ~4k coverage window. + That explains weeks of zero events and gives a cheap deterministic fix: + make coverage/order deterministic instead of cursor-lottery.) +3. Frontier cap / search lifetime (config: _reference_chains_frontier_cap; + ×4 cap ≈ ×4 coverage per search ≈ 16k of 28k, at 48-64MB native table + cost and larger GOTW maps — tunable with tradeoffs). +4. Scale gtest asserting the coverage invariant — BEFORE the next deploy. + +Standing direction: D gated on measurement + scale test (per user's earlier +pick; C withdrawn). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/meta-whackamole-analysis.md b/.investigations/missing-refchains-on-hotdog/nodes/meta-whackamole-analysis.md new file mode 100644 index 0000000000..8e73b5b343 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/meta-whackamole-analysis.md @@ -0,0 +1,116 @@ +--- +id: meta-whackamole-analysis +type: meta +status: confirmed +depends_on: [meta-circle-review, find-canary-found-criterion-unmigrated, find-refchains-log-flood-configurable, find-urgentoom-null-fn-mislabel, find-wrapper-demotion-self-parent, find-tier1-tail-starvation] +related: [find-holistic-design-issues, find-test-seam-aliasing, find-klass-id-notation-mismatch, find-canary-stuck-restart-wipes-frontier] +tags: [meta, root-pattern, defect-taxonomy, strategy, methodology, NEW-THIS-SESSION] +created: 20260916 +updated: 20260916 +--- + +# Meta-review 2: are we running in circles? (post-goal, requested at round 19) + +## Verdict + +**Rounds 13-16: no — convergent.** Meta-review 1's prescription (measure +the coverage invariant first) was followed: the funnel was measured, +the fresh lane + demotion guard + quota were the measured consequences, +and the END GOAL (leak-correlated events) verified at round 16. Facts +monotone, each fix distinct. + +**Rounds 17-19: yes — the shape is whackamole.** Three rounds since the +goal: two were about our OWN diagnostics (log flood, knob escapes, tier +miscalibration), one about a lifecycle bug (canary found criterion) that +existed since the marker→leak-tag migration — i.e., since ~round 12, on +every search, invisible because nothing asserted it. The whackamole +signature: **deploy → observe one broken link → fix → deploy → a different +broken link appears**. The per-iteration cost is a deploy+hours, n=1, and +each verification only checks the symptom that motivated the fix. + +## The defect taxonomy (all 30+ fixed defects, classified by PRODUCING mechanism) + +1. **Scheduler starvation under diverse load** (~12 defects — the largest + cluster): anchor-holder eviction, tail starvation x2, priority-queue + starvation, shared-deadline starvation, sweep-never-completes, + EMA-batch collapse, GetObjectsWithTags quadratic, CPU-pain-budget + blocks, the at-risk FIFO flood pin. Shape: one shared + budget/queue/cursor + a heterogeneous population deterministically + starves whatever matters. Each fix re-tuned ONE scheduler parameter; + no global liveness/fairness invariant was ever asserted anywhere. +2. **Keyspace/identity confusion** (~6): marker-tag slot mismatch, + klass-id notation mismatch, leaktag-JFR field misalignment, + representative-changes-lose-canary, candidate1-never-tagged, and + round-19's found criterion. Root: THREE tag keyspaces (marker/leak/ + frontier) + one JVMTI tag slot per object + per-JVM id instability. + Every cross-space join is a bug farm; round 19 was a join that broke + when one space was deprecated. +3. **Half-migrated designs** (2 proven, the class with the highest + future risk): marker→leak-tag migration (found criterion + rep path + left on the old contract); the TEST_LOG migration to the level gate + (header-inline sites + tick-tiering left behind). Shape: a design + change ships in pieces; consumers of the old contract are never + enumerated. +4. **Test-seam gaps — the suite is green while the system is broken** + (~8): test-seam aliasing, reset() missing clears, UrgentOOM null-fn + (5 weeks of lost signal), every round where 348 gtests passed and the + pod still failed. Root: unit tests verify components; NOTHING + verifies the integrated pipeline. **The pod is the only end-to-end + oracle — that IS the whackamole generator.** +5. **State lifetime across search restarts** (~4): restart-wipes-frontier, + ages-vector, discovered tags surviving, representative refresh. Every + new process-wide state adds a hand-written clear-site obligation; + the reset contract is enforced by discipline, not by test. +6. **Observability as a defect source** (rounds 17-18): diagnostics + added under time pressure, ungated, unmeasured — then needing their + own fixing rounds. +7. **Deploy environment drift** (~3, mostly solved by per-build markers). + +## The structural statement + +The system has grown: a scheduler (tiers/queues/budgets) + three tag +keyspaces + per-search state + a poll/chase state machine. Its +verification apparatus is: 348 unit gtests (components) + a staging pod +(integration, ~hours/iteration, n=1). **Every class-1/2/3/5 defect is +invisible to the unit suite by construction.** So the pod finds them one +deploy at a time — in whatever order they happen to fire. Whackamole is +not a discipline failure; it is the predictable output of this +verification topology. + +## The non-whackamole exit + +Stop making the pod the oracle. Two moves: + +- **P0 — build the missing oracle once**: a system-level simulation + harness ("pod in a jar") — the REAL tracker loop over a mock JVMTI + topology encoding every pod-discovered shape (synchronized wrapper with + its mutex self-edge, flood classes, leak chunks, noise, thread-locals), + running multiple search lifetimes + restarts, asserting INVARIANTS + (liveness, canary resolution, natural completion, log budget, quotas, + restart hygiene, boundedness, healthy-app dormancy). Every invariant + maps 1:1 to a real round's bug — 7 of the last 10 pod rounds would + have been a red local test BEFORE any deploy. See + design-pod-in-a-jar-harness. +- **P1-P4 — close the specific generator classes**: keyspaces audit + + dead marker-path retirement (class 2/3), always-on per-search health + counters so pod verification is one line not grep archaeology (class + 6 + observability), reset-contract invariant test (class 5), deferred + queue triaged to harness-flagged-only (no more pod-driven fixes). + +The alternative — continue pod-driven — predicts rounds 20+ of the same: +the scheduler, keyspaces, and restart state each still hold unfixed +members of their class (see the deferred queue in STATE), and they will +surface one deploy at a time. + +## Execution status + +P0 EXECUTED (round 20): the harness is built and all invariants live in +CI — the pod is no longer the only end-to-end oracle. P1 (keyspaces +audit + marker-path retirement) and P2 (health line) are next; P4's +deferred queue is harness-flagged-only. + +## Answer to the user's question + +No on the goal (achieved, round 16, reproduced on a second JVM). Yes on +the process since. The fix is not another patch — it is P0 + P1-P4 above, +then upstreaming review becomes mechanical instead of exploratory. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/q-allocation-site-selection.md b/.investigations/missing-refchains-on-hotdog/nodes/q-allocation-site-selection.md new file mode 100644 index 0000000000..10eba61905 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/q-allocation-site-selection.md @@ -0,0 +1,76 @@ +--- +id: q-allocation-site-selection +type: question +status: implemented +depends_on: [find-age-heuristic-insufficient] +supersedes: [] +related: [find-age-heuristic-insufficient] +tags: [design, referenceChains, livenessTracker, allocation-site, representative-selection, NEW-THIS-SESSION] +created: 2026-08-28 +updated: 2026-08-28 +--- + +# How to select representatives by allocation site? + +## Context + +User approved combining directions 1+3 from the research analysis: + +- **Direction 1** (Cork/Melt): group surviving objects by + `(klass_id, call_trace_id)` instead of just `klass_id`. Track growth + trend per allocation site. Select representatives from the + highest-growth site. +- **Direction 3** (Swat): per site, compute + `growth_rate × survival_count`. Sites with many surviving instances + outrank sites with few, regardless of individual object age or size. + +## Implementation questions + +1. `KlassCountScratch` currently keys by `klass_id` only. Keying by + `(klass_id, call_trace_id)` means more distinct entries per epoch. + The scratch table is capped at `MAX_KLASS_POPULATION_ENTRIES = 256`. + With per-site granularity, there could be >256 distinct sites. Need + to either increase the cap or prioritize sites with higher survival + counts when the table is full (drop low-survival sites = noise). + +2. `KlassPopulationEntry` tracks per-klass ring buffer of generation + counts. With per-site tracking, need a ring buffer per + `(klass_id, call_trace_id)` pair. Either: + - (a) Expand `_klass_population` to key by `(klass_id, call_trace_id)` + — more entries, more memory. + - (b) Keep per-klass tracking for the leak signal (selectLeakCandidates) + but add per-site tracking for representative selection only — + the leak signal fires per-class, but the representative is picked + from the highest-growth site within that class. + +3. `TrackingEntry` already has `call_trace_id` and `tid`. The + `accumulateKlassCount()` method receives `jlong age, jweak + sample_source` but not `call_trace_id`. Need to pass it through. + +4. `KlassCountScratch::oldest[]` tracks top-3 oldest instances. With + per-site selection, we'd want top-3 oldest from the **highest-growth + site**, not from all sites of the class. + +## Implementation (commit 0b492612b) + +**Implemented**: per-allocation-site generation cardinality tracking. + +- `KlassCountScratch` now has `SiteGens[]` (max 16 sites × 32 ages + each) tracking distinct surviving GC ages per `call_trace_id`. +- `accumulateKlassCount()` accepts `call_trace_id` and calls + `insertSiteGen()` for EVERY surviving object (not just first per + age — critical: per-site generation tracking must see all objects). +- `insertSiteGen()` maintains sorted distinct-age arrays per site. +- `foldKlassCountsLocked()` finds the dominant site by generation + cardinality (most distinct surviving ages), then mints + representatives preferentially from that site. +- `insertOldestSample()` now also stores `call_trace_id` per sample, + so the minting logic can filter by site. + +User refined the approach: use generation cardinality (distinct +surviving ages per site), not raw survival count — this reuses the +same signal that `selectLeakCandidates()` uses per-class, applied at +per-site granularity. A site with 12 distinct surviving ages +(continuous leak) outscores a site with 1 age (one-time burst). + +**Not yet deployed**: pod still running `21ff0a928` + `cd68be618`. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/q-canary-stuck-fix-alternatives.md b/.investigations/missing-refchains-on-hotdog/nodes/q-canary-stuck-fix-alternatives.md new file mode 100644 index 0000000000..199b3a97c5 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/q-canary-stuck-fix-alternatives.md @@ -0,0 +1,132 @@ +--- +id: q-canary-stuck-fix-alternatives +type: question +status: confirmed +depends_on: [find-canary-stuck-restart-wipes-frontier] +supersedes: [] +related: [find-canary-fixes-e-f, find-canary-stuck-abandon-detector] +tags: [decision-made, fix-implemented, referenceChains, canary, research, NEW-THIS-SESSION] +created: 2026-08-25 +updated: 2026-08-25 +--- + +# Which fix for the CANARY_STUCK/frontier-wipe convergence bug? RESOLVED: user chose C+B, implemented and gtest-verified + +## Context + +`find-canary-stuck-restart-wipes-frontier` is now a CONFIRMED root cause +(not just a hypothesis) for why candidates resolve 0/5 even after Fix E/F +resolved the earlier starvation. User confirmed the target is a synthetic, +permanently-reachable leak, ruling out "candidate unreachable" as a +competing explanation. User asked to analyze the leading suspect and +research well-established/bleeding-edge solutions before proposing fixes. + +## Research grounding + +- **Incremental BFS (IBFS)**: standard technique for "restart destroys + frontier/visited state" — reuse the queue/frontier across cycles instead + of resetting to empty. +- **G1 GC's SATB (snapshot-at-the-beginning) marking**: concurrent/ + incremental collectors never re-mark from roots per increment; they + snapshot the live-set logically once and drain a write-barrier log + instead of discarding accumulated marking work — same shape of problem + in GC's domain. +- **Luby/adaptive restart theory (SAT solving)**: a *fixed* cutoff before + restart is provably suboptimal when the true cost-to-converge is unknown + or variable — directly indicts `CANARY_NO_PROGRESS_PASS_LIMIT=30` as a + heap-size-agnostic constant. Established alternative: adaptive/growing + cutoffs, or restarting only when *multiple independent* progress signals + have all stalled. + +## Four alternatives proposed + +**C — Merge the two stuck detectors (recommended lean).** Only fire +`CANARY_STUCK` when `_passes_since_last_progress` (whole-graph) is ALSO +over its own limit, not just `_passes_since_last_candidate_progress` +alone. Live trace directly falsifies the current design's premise: the +frontier was still growing (12k→16k) throughout — the graph walk was not +stuck, only the (narrower) "found this specific candidate" signal was. +Smallest diff (one condition), lowest risk. + +**A — Stop wiping the frontier on `CANARY_STUCK` (IBFS-style).** Keep +`_frontier`/tag state intact across the restart; only clear the stuck +counter and keep walking. More invasive (touches `restartSearch()`'s +contract and candidate-tag release logic) but correct if `CANARY_STUCK` +should mean "pause and reassess" rather than "declare defeat". + +**B — Adaptive/Luby-style growing limit instead of fixed 30.** Scale +`CANARY_NO_PROGRESS_PASS_LIMIT` per retry (30, 60, 120, ...). Keeps the +existing destructive-restart architecture but is a bandage — still +forgets everything each cycle, just tries longer before forgetting. + +**D — Persistent tag/visited-set across restarts (full SATB-style +resumability).** Decouple "restart" from "re-walk from empty" by +preserving `_next_tag`/the tag-identity table so a restarted search treats +previously-explored regions as already-visited. Most architecturally +faithful to established incremental-GC design, but touches JVMTI tag +lifecycle and frontier-table capacity assumptions — highest implementation +risk for likely the same practical outcome as A here. + +## Recommendation given to user + +Lean: **C**, optionally combined with **B** as a belt-and-suspenders hard +ceiling in case the whole-graph frontier itself ever genuinely stalls on a +very large heap. A/D flagged as worth keeping in mind if canary searches +need true pause/resume semantics later, but more invasive than this +specific bug requires. + +## Decision and implementation (this session) + +User: "Try C/B first. Then I will deploy and you will check if it helped." +Implemented both, in `ddprof-lib/src/main/cpp/referenceChains.{h,cpp}`: + +- **C (merged stuck detectors)**: `runPass()`'s `CANARY_STUCK` branch + (`referenceChains.cpp:3120-3149`) now additionally requires + `_passes_since_last_progress >= NO_PROGRESS_PASS_LIMIT` — the whole-graph + frontier must ALSO have stalled — before firing, on top of the existing + `_passes_since_last_candidate_progress >= canaryStuckPassLimit()` check. + Directly counters the live-trace evidence (frontier still growing + 12k→16k while chasing the candidate). +- **B (adaptive/Luby-style escalating limit)**: new field + `_canary_stuck_restart_count` (referenceChains.h) tracks consecutive + `CANARY_STUCK` restarts of the same candidate-chase sequence — NOT reset + by `restartSearch()` itself, only reset in the terminal-state block when + a search ends for a reason OTHER than `CANARY_STUCK` + (`referenceChains.cpp:3200-3206`). New method `canaryStuckPassLimit()` + doubles `CANARY_NO_PROGRESS_PASS_LIMIT` (base 30) per consecutive + restart, capped at `MAX_CANARY_STUCK_BACKOFF_SHIFT=8` doublings (7680 + passes ceiling) so a genuinely-stuck-forever search still terminates in + finite time. +- Test seams added: `setCandidateCountForTest()`, + `passesSinceLastCandidateProgressForTest()`, + `canaryStuckRestartCountForTest()` (referenceChains.h, wrapped by + `ReferenceChainsTestAccessor` in the test file). Also fixed a + test-isolation gap in `ReferenceChainsTestAccessor::reset()` — it did not + clear `_passes_since_last_candidate_progress`, `_last_candidate_progress_mark`, + or the new `_canary_stuck_restart_count` between `TEST_F` cases. +- **New regression test**: + `ReferenceChainsBfsTest.CanaryStuckRequiresWholeGraphFrontierAlsoStalled` + in `referenceChains_ut.cpp` — builds a 50-node linear mock heap chain with + `budget=1` (frontier grows by exactly one node every pass, so + `_passes_since_last_progress` never reaches `NO_PROGRESS_PASS_LIMIT`) and + a canary candidate that is never found. Asserts the search stays + `RUNNING` for `CANARY_NO_PROGRESS_PASS_LIMIT + 2` passes — well past the + point the old single-condition check would have abandoned it. + **Verified rigorously**: temporarily reverted just the merged condition + back to the pre-fix single-condition form and reran — the new test + failed (canary abandoned prematurely, `Value of: SearchState::RUNNING` + assertion tripped). Restored the fix — test passes, and the full + `gtestDebug_referenceChains_ut` suite (90 tests) passes clean. +- Compiles cleanly: `./gradlew :ddprof-lib:compileDebug -Pskip-gtest` → + BUILD SUCCESSFUL, 69 files, no warnings. +- **Not yet committed/pushed** (not requested this session). **Not yet + deployed** — user will deploy and this investigation will need another + live-pod verification pass afterward. + +## Not yet done + +- Alternatives A and D remain unimplemented/deferred by explicit user + choice ("Try C/B first"). +- Commit, push, deploy, and the follow-up live-pod re-check are all + pending — see `find-canary-stuck-restart-wipes-frontier`'s "Not yet + done" section. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/q-coverage-tracking-per-combination.md b/.investigations/missing-refchains-on-hotdog/nodes/q-coverage-tracking-per-combination.md new file mode 100644 index 0000000000..3ca5a99f0d --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/q-coverage-tracking-per-combination.md @@ -0,0 +1,48 @@ +--- +id: q-coverage-tracking-per-combination +type: question +status: open +depends_on: [find-leak-tag-pool-implementation] +supersedes: [] +related: [find-leak-tag-pool-implementation, find-lambda-fragments-calltrace-id] +tags: [adaptive-cpu, coverage, call_trace_id, tid, referenceChains, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# Coverage tracking: per-object vs per-(call_trace_id, tid) combination + +## The gap + +The user's requirement (design D): CPU should drop to ~1x "the moment +we have at least one reference chain for each of the +call_trace_id/tid combination for classes with growing surviving gens +metric." + +Current implementation (commit `294f09ff3`) is per-OBJECT: +`_leak_tags_resolved` counts leak tags with chains vs +`_leak_tags_assigned`. If 10 objects share one (call_trace_id, tid) +combination, we need 10 chains before coverage completes — wasteful; +one chain per combination should suffice. + +## Why per-object was chosen first + +Simplicity: avoids a set data structure (allocation discipline). +`LivenessTracker::getLeakTagInfo(tag)` already returns +`{call_trace_id, tid}` per tag, so the data to do it properly exists. + +## Options to refine + +1. Fixed-size array of distinct (call_trace_id, tid) pairs in + ReferenceChainTracker (universe computed at tagLeakInstances time + from `_leak_tag_info[]`, covered incremented on first chain per + combination). Distinct combinations expected small (1-5 on + hotdog). +2. Keep per-object but only tag ONE representative per combination + in tagLeakInstances (dedupe at tag time) — then per-object coverage + IS per-combination coverage. Cleaner: no set needed downstream. + Risk: loses the multi-instance redundancy for BFS discovery. + +Option 2 is likely the leaner design — decide after on-pod +verification shows whether per-object counting actually stalls at +1x or wastes CPU. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/q-dominant-gens-still-one-with-tid.md b/.investigations/missing-refchains-on-hotdog/nodes/q-dominant-gens-still-one-with-tid.md new file mode 100644 index 0000000000..e575e80098 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/q-dominant-gens-still-one-with-tid.md @@ -0,0 +1,58 @@ +--- +id: q-dominant-gens-still-one-with-tid +type: question +status: open +depends_on: [find-lambda-fragments-calltrace-id, ev-tid-clustering-onpod-verification] +supersedes: [] +related: [find-lambda-fragments-calltrace-id, q-allocation-site-selection] +tags: [referenceChains, livenessTracker, tid, dominant-gens, subsampling, NEW-THIS-SESSION] +created: 2026-08-28 +updated: 2026-08-28 +--- + +# Why is dominant_gens=1 even with tid-based clustering? + +## Context + +Switched clustering from `call_trace_id` to `tid` (commit `2c50bf0cd`). +On-pod: still `dominant_gens=1` for every class every epoch. + +User says: the liveness table IS cross-epoch, and the leaking [B +instances ARE all from the same thread (tid=172). So the per-thread +age set should have 12 distinct ages, not 1. + +## Hypotheses + +1. **Subsampling**: The liveness tracker subsamples allocations + (`_subsample_ratio`). If the ratio is low, only 1 of 12 leaking [B + is tracked. That 1 instance has 1 age → `dominant_gens=1`. The + per-class `gen_count=33` comes from 33 different threads each + contributing 1 tracked instance. + +2. **tid field not populated**: `_table[target].tid` might be 0 for + some entries, causing all objects to cluster under tid=0. + +3. **insertThreadGen bug**: The sorted-insert or dedup logic might + be wrong — but the code looks correct (linear scan for existing + tid, sorted insert of distinct ages). + +## Diagnostic + +Commit `5e4493dbf` logs per-thread `age_count` breakdown: +``` +thread[0] tid=58393 age_count=1 +thread[1] tid=58471 age_count=1 +... +``` + +If all threads show `age_count=1`, hypothesis 1 (subsampling) is +likely. If one thread shows `age_count=12` but `dominant_gens` is +still 1, there's a bug in the dominant-thread selection. + +## Implication + +If subsampling is the cause, per-epoch per-thread tracking in the +scratch table can never produce `dominant_gens > 1` — the signal is +too sparse. Would need cross-epoch per-thread tracking in +`KlassPopulationEntry` (a per-thread ring buffer), only for leak +candidates. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/q-heapliveobject-absent-on-pod-chunks.md b/.investigations/missing-refchains-on-hotdog/nodes/q-heapliveobject-absent-on-pod-chunks.md new file mode 100644 index 0000000000..e550579da7 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/q-heapliveobject-absent-on-pod-chunks.md @@ -0,0 +1,34 @@ +--- +id: q-heapliveobject-absent-on-pod-chunks +type: question +status: confirmed +depends_on: [ev-leaktag-onpod-round8, ev-leaktag-onpod-round9] +related: [find-leak-tag-pool-implementation] +tags: [heapliveobject, liveness, jfr, evidence-source, RESOLVED, NEW-THIS-SESSION] +created: 20260902 +updated: 20260903 +--- + +# RESOLVED: HeapLiveObject events are absent from LOCAL pod chunks but present in UPLOADED recordings + +Round 9 settled it: the uploaded .jfr (toolkit download) contains +`datadog.HeapLiveObject` events — including the 78MB leak chunks +(`eventThread=simulated-memory-leak`, leakTags 1073741987/1073741990, +ages 99-108) — while the local chunks under +`/tmp/ddprof_root/pid_XXX/jfr/` contain neither HeapLiveObject nor +ReferenceChain (same artifact as round 2's "need merged upload" lesson). + +## Resolution + +Not an emission bug: liveness flows and leak tags reach the recording. +The round-8 "zero HeapLiveObject in any chunk" observation was an +evidence-source artifact. + +## Standing evidence rule + +Dump-time / liveness events must be verified from UPLOADED recordings +(profiling-toolkit download.py, us1.staging.dog), never from local pod +chunk files. Also: `jfr print` crashes on datadog.ReferenceChain +(PrettyWriter ClassCastException on the chain F_CPOOL|F_ARRAY field) — +use the JMC API (RcDump pattern / ReferenceChainAssertions +findAccessor). diff --git a/.investigations/missing-refchains-on-hotdog/nodes/q-implement-two-fixes.md b/.investigations/missing-refchains-on-hotdog/nodes/q-implement-two-fixes.md new file mode 100644 index 0000000000..f161f7c708 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/q-implement-two-fixes.md @@ -0,0 +1,60 @@ +--- +id: q-implement-two-fixes +type: question +status: confirmed +depends_on: [find-marker-tag-slot-index-mismatch, find-one-shot-pretag-gate, find-canary-search-cannot-terminate] +supersedes: [] +related: [find-canary-stuck-abandon-detector, ev-fixes-compile-and-gtest-pass] +tags: [decision-made, fix, implemented] +created: 2026-08-24 +updated: 2026-08-24 +--- + +# RESOLVED: yes — both fixes were implemented, plus a third (Fix C) + +## Reasoning chain + +Both fixes were offered at the end of the session ("Want me to implement +both?"). The user answered with `/investigation seed …` instead, so +**neither is approved and nothing has been changed** — no files modified, +nothing committed, HEAD still `8114019c2`. + +### Fix A — decode the slot from the tag in `pollWatchedTargets()` + +Mirrors what `heapReferenceCallback()` already does at +`referenceChains.cpp:1510`. Touches `referenceChains.cpp:3478` and `:3485`: + +```cpp +int slot = (int)(MARKER_TAG_BASE - tag); +bool built = buildCanaryChainEvent(slot, &event); +… cacheResolvedChain(klass_id, std::move(event), _candidate_frontier_tags[slot], …); +``` + +Note `buildCanaryChainEvent()` already range-checks its argument against +`_candidate_count` (`referenceChains.h:2094-2097`), so a stale slot >= +`_candidate_count` degrades to a clean `false` rather than an OOB read. +The `TEST_LOG` at `:3473-3475` and `:3479-3481` should log the slot too, +otherwise the same inconsistency remains invisible in logs. + +### Fix B — rework the `_candidate_count == 0` gate + +`referenceChains.cpp:3375-3397`. Must admit newly-flagged candidates and +retire dead ones, otherwise a churning candidate set (5 selected vs 3 +latched, as seen live) will pin the search in `RUNNING` again even with +Fix A. Interacts with the completion condition at `:3101-3103` and with +`_candidate_found_bits` bit positions, so the two have to be designed +together. + +## Sub-questions not yet answered +- Is a regression test with **more than one** canary candidate needed? + The existing integration test exercised a single candidate, which is + precisely why this escaped (see `hyp-regression-of-five-fixes`). +- Should the `!isUrgent()` suppression of the TTL/abandon path + (`referenceChains.cpp:3090-3091`) gain an absolute backstop, so a + livelocked urgent search still eventually resets rather than burning + STW budget forever? + +## Evidence +- `nodes/find-marker-tag-slot-index-mismatch.md` +- `nodes/find-one-shot-pretag-gate.md` +- `nodes/find-canary-search-cannot-terminate.md` diff --git a/.investigations/missing-refchains-on-hotdog/nodes/q-resize-instrumentation-rescan-priority.md b/.investigations/missing-refchains-on-hotdog/nodes/q-resize-instrumentation-rescan-priority.md new file mode 100644 index 0000000000..b9815c81a5 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/q-resize-instrumentation-rescan-priority.md @@ -0,0 +1,49 @@ +--- +id: q-resize-instrumentation-rescan-priority +type: question +status: open +depends_on: [find-priority-queue-starves-bfs-crawl] +related: [find-priority-queue-starves-bfs-crawl, find-holistic-design-issues] +tags: [design, rotation, growing-collections, instrumentation, NEW-THIS-SESSION] +created: 20260831 +updated: 20260831 +--- + +# User idea: bytecode instrumentation on collection resizes → rescan prioritization + +The resize blindspot fix (fair-share rotation) is probabilistic: a growing +collection's new backing array is only found when the blind lap happens to +re-walk the live holder. User's idea (2026-08-31, to revisit after the +current fix lands): if we could observe collection resizes directly - e.g. +via bytecode instrumentation of the resize/mutation points +(`ArrayList.grow`, `HashMap.resize`, or more generally any "collection +mutated" event for watched classes) - we could use that as a precise signal +to prioritize re-walks of the affected holder instead of relying on +rotation luck. + +## Why it fits + +- The profiler already ships bytecode instrumentation machinery + (dd-iast / dynamic instrumentation agent in the same -javaagent). +- The signal is exactly the mutation rotation exists to detect, with none + of the false-negative geometry (dead-old-parent fanout entries, holder + never in fanout, lap starvation). +- Cost profile: instrumentation fires only on resize of watched containers, + not per-element mutation (a per-put signal on a hot map would be a flood); + resizes are amortized O(1) per element, so the event rate is manageable. + +## Open questions + +- Which instrumentation point: `ArrayList.grow`/`HashMap.resize` only, or an + interface-level "collection structure changed" marker on the watched + classes' holders? +- Delivery to native: a JNI upcall into the tracker (allocation-adjacent, + not signal context) - or a flag on the frontier entry checked by the next + rotation pass. +- Cross-JVM: J9/Zing instrumentation parity. +- Whether the existing fair-share rotation (once verified) is good enough + that the added machinery isn't justified - the pod evidence (leak_parents + 19k, mostly dead) suggested fanout was heavily polluted, so a precise + signal could replace the fanout tier entirely rather than complement it. + +Parked until the rotation fair-share fix + correlation scenario are verified. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/q-safepoint-budget-model.md b/.investigations/missing-refchains-on-hotdog/nodes/q-safepoint-budget-model.md new file mode 100644 index 0000000000..3ea57223e8 --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/q-safepoint-budget-model.md @@ -0,0 +1,57 @@ +--- +id: q-safepoint-budget-model +type: question +status: open +depends_on: [ev-timing-split-callback-vs-jvmti] +supersedes: [] +related: [find-static-field-sweep-never-completes, find-static-field-sweep-cursor-fix] +tags: [safepoint, budget, deadline, per-call, cumulative, rate-cap, design, NEW-THIS-SESSION] +created: 2026-08-26 +updated: 2026-08-26 +--- + +# Safepoint budget model: per-call 50ms cap + cumulative 500ms/sec rate cap? + +## User's clarification + +The user clarified the intended safepoint budget model (two distinct +concerns, currently conflated in the code): + +1. **Single STW operation**: each chunk's `FollowReferences` call should + be capped at 50ms. If it exceeds, truncate and resume from the + resumable cursor next pass. This is what the per-chunk deadline + already does — but only if the deadline isn't already spent by other + sub-operations. + +2. **Cumulative per real-time second**: total STW time across all + operations (root enum + sweep + expandFrontier + rotation) should be + ≤500ms/sec. This is a *rate* cap, not a per-call cap. + +## Current code is wrong for this model + +`runPassManualWalk()` sets **one** `_pass_deadline_ns` at the top +(`:2334-2337`) and shares it across all sub-operations. So root +enumeration + expandFrontier can consume the budget before the sweep +runs. This is why the sweep was truncating at 50ms even though each +chunk only needs 3-11ms — the budget was already spent by other +sub-operations. + +## Proposed fix (not yet implemented) + +Give each sub-operation its own 50ms deadline (per-call cap), and add a +separate cumulative rate cap (500ms/sec) via the existing +`PainBudget`/cadence mechanism. The per-call cap ensures no single +`FollowReferences` exceeds 50ms; the rate cap ensures total STW time +stays ≤500ms/sec across all passes in a rolling 1-second window. + +This is the leaky-bucket follow-up discussed earlier in the session, +but now clearly motivated by the timing data: the sweep itself is +3-11ms/chunk (well under 50ms per-call), but the shared budget means +it gets truncated because other sub-operations spent the deadline first. + +## Status + +Open design question. User stated the model but hasn't asked for +implementation yet. The 200ms temporary override (commit `fd18425c6`) +needs to be reverted before any deployment — it was a measurement +diagnostic, not a production change. diff --git a/.investigations/missing-refchains-on-hotdog/nodes/q-togcroot-acceptance-paths.md b/.investigations/missing-refchains-on-hotdog/nodes/q-togcroot-acceptance-paths.md new file mode 100644 index 0000000000..a6a2bc6a2d --- /dev/null +++ b/.investigations/missing-refchains-on-hotdog/nodes/q-togcroot-acceptance-paths.md @@ -0,0 +1,77 @@ +--- +id: q-togcroot-acceptance-paths +type: question +status: answered +depends_on: [find-canary-lane-backoff-design, find-default-live-samples-ratio-lottery] +related: [find-canary-continue-skips-discovered-instances, find-representative-changes-lose-canary, find-shared-deadline-starves-expand] +tags: [flakiness, tests, canary, pacing, open, NEW-THIS-SESSION] +created: 20260901 +updated: 20260901 +--- + +# ToGcRoot/UnboundedCache flakiness: multiple acceptance paths x canary pacing - what actually closes it? + +ANSWERED by find-togcroot-orphaned-slot-stranding: slot stranding, not +pass starvation or path multiplicity. Fixes 1+2 landed; suite green at +load 28.7 and 53. Fix 2's organic-qualification improvement is the +remaining thread. + +The remaining intermittent slow-suite failures (ReferenceChainTrackingTest +shouldReconstructReferrerChainToGcRoot, sometimes UnboundedCacheLeak) are +NOT the live-samples-ratio lottery (fixed; the "representative +died/evicted" symptom is gone at :l:1.0) and are not purely +machine-load (latest failure at load ~3). + +## The confusing data (all on the work-scaled backoff build) + +- Green run (18:09, 10% ratio): PASS with only 13 passes, ZERO canary + prunes - the ReferenceChain event arrived via a NON-canary path + (discovered-instance/leak-tag machinery) without the marker chase + ever resolving. +- Fail run (19:1x, :l:1.0, load 3): 124 passes, 101+ Tier-2 rotation + selections, canary NEVER pruned, no chain within the window. +- Earlier fail runs (17:47-18:04, 10% ratio): ~114-200 passes, mostly + zero-edge, same no-prune outcome. + +## Candidate factors (untested) + +1. The chase needs ~200 mostly-cheap rotation passes to lap to the + marker's holder; work-scaled spacing = mult x EMA(pass wall), and + the EMA is polluted by occasional ~300ms root-enum passes + (ROOT_ENUM_MIN_INTERVAL gates them back in; pass wall includes the + full root-enum phase), inflating spacing 10x exactly when the chase + is otherwise cheap. Hypothesis: exclude root-enum passes from the + EMA (or measure chase-pass cost separately). +2. Tier-2 selected the right-looking holders (fanout up to 1700, + hundreds of selections) yet the marker was never pruned - is the + marker rep's actual parent among the winning (leaf, parent-class) + signatures? Needs a one-shot diagnostic tying a Tier-2 selection to + the marker tag. +3. The 13-pass green run shows the acceptance can be met without the + canary at all - which path served it, and why does it not serve it + in the failing runs? (Discovered-instance gate? leak-tag + interception ordering?) + +## New data points (post admission-boost, still failing identically) + +- The chase-phase admission boost (admitForTracking watched tids) is + neutral on this failure, as expected (suite tests already run + memory=64:l:1.0 so admission was 100% all along) - but it confirmed the + chase machinery engages: noteSelectedCandidates published the + scenario's tid in the failing child, 0 "representative died" lines. +- BOTH failing runs hit exactly 124 passes (pre- and post-boost) - + the pass budget looks like a structural ceiling (16 rounds x wakes), + not timing randomness: the chase needs ~200 rotation passes to lap + to the holder and the test only ever funds ~124. Under this model + the fix candidates are: fewer passes to resolution (Tier-2 holder + expansion actually pruning the marker - 535-821 selections never did, + which is itself suspicious: is the marker rep's parent EVER among the + selected parents?), or more funded passes (rounds/grace), or a + shorter resolution path (the 13-pass green run's non-canary path). + +## Next instrumentation (when this becomes the active thread) + +- TEST_LOG in tagLeakInstances/canary-prune with per-pass edges + + which pass pruned; per-pass EMA contributions split + root-enum vs chase. +- One failing run with those lines answers (1) and (2). From 78c4fd2caf2f81c3f3a03da81a474dd719276358 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Wed, 16 Sep 2026 17:02:38 +0200 Subject: [PATCH 08/18] Update .gitignore and build-summarize helper Ignore local build/repro artifacts; extend the build-and-summarize command with reference-chains build steps. --- .claude/commands/build-and-summarize | 169 +++++++++++++++++++++++++-- .gitignore | 12 +- 2 files changed, 171 insertions(+), 10 deletions(-) diff --git a/.claude/commands/build-and-summarize b/.claude/commands/build-and-summarize index b6f74dbaa8..8cc0af5fcb 100755 --- a/.claude/commands/build-and-summarize +++ b/.claude/commands/build-and-summarize @@ -67,13 +67,164 @@ else echo "✖ Gradle failed with status $status. Full log at: $LOG" fi -# Hand over to your logs analyst agent — keep the main session output tiny. +# Generate the summary artifacts deterministically — no LLM, no claude spawn. +# (The old tail spawned a headless `claude` per build; managed settings pin that +# to sonnet-5 via the AI gateway, billing real money for every log parse.) echo -echo "Delegating to gradle-logs-analyst agent…" -# If your CLI supports non-streaming, set it here to avoid verbose output. -# Example (uncomment if supported): export CLAUDE_NO_STREAM=1 - -# Sub-agents would inherit the full parent env, including CLAUDECODE, which -# apparently causes problems with some versions of Claude (e.g. 4.6). -unset CLAUDECODE -claude "Act as the gradle-logs-analyst agent to parse the build log at: $LOG. Generate the required Gradle summary artifacts as specified in the gradle-logs-analyst agent definition." +echo "Generating gradle summary…" +python3 - "$LOG" <<'PYEOF' || echo "⚠ gradle summary generation failed — full log at: $LOG" +import json, os, re, sys + +log_path = sys.argv[1] if len(sys.argv) > 1 else "" +OUT_DIR = os.path.join("build", "reports", "claude") +MD = os.path.join(OUT_DIR, "gradle-summary.md") +JS = os.path.join(OUT_DIR, "gradle-summary.json") + +status, total_time, cur_task = "UNKNOWN", None, None +failed_tasks, headlines, dep_issues, test_failures = [], [], [], [] +warnings, seen_warnings = [], set() +tests = {"total": 0, "failed": 0, "skipped": 0, "modules": {}} + +RE_STATUS = re.compile(r"^BUILD (SUCCESSFUL|FAILED)(?: in (.+?))?\s*$") +RE_TASK = re.compile(r"^> Task (\S+)") +RE_GSUM = re.compile(r"(\d+) tests? (successful|failed|skipped)") +RE_COMP = re.compile(r"(\d+) tests? completed(?:, (\d+) failed)?(?:, (\d+) skipped)?") +RE_GTEST_P = re.compile(r"\[ PASSED \] (\d+) tests?") +RE_GTEST_F = re.compile(r"\[ FAILED \] (\d+) tests?") +RE_DEP = re.compile(r"Could not (?:resolve|find|get)|timed out|status code: 40[13]|artifact .* not found", re.I) +RE_WARN = re.compile(r"\bw: |warning:|deprecat", re.I) + +def mod(name): + return tests["modules"].setdefault(name, {"total": 0, "failed": 0, "skipped": 0}) + +def add_counts(name, ok, failed, skipped): + m = mod(name) + m["total"] += ok + failed + skipped + m["failed"] += failed + m["skipped"] += skipped + tests["total"] += ok + failed + skipped + tests["failed"] += failed + tests["skipped"] += skipped + +if not log_path or not os.path.isfile(log_path) or os.path.getsize(log_path) == 0: + why = f"log not found: {log_path!r}" if not log_path or not os.path.isfile(log_path) else "log is empty" + os.makedirs(OUT_DIR, exist_ok=True) + open(MD, "w").write(f"# Gradle Summary\n\n- Status: UNKNOWN ({why})\n\nFull log unavailable or empty.\n") + json.dump({"status": "UNKNOWN", "totalTime": None, "failedTasks": [], "warnings": [], + "tests": tests, "slowTasks": [], "depIssues": [why], "actions": []}, open(JS, "w"), indent=1) + print(f"Gradle log unusable ({why}); wrote {MD} and {JS}") + sys.exit(0) + +grab_headline = False +with open(log_path, encoding="utf-8", errors="replace") as f: + for line in f: + line = line.rstrip("\n") + m = RE_STATUS.match(line) + if m: + status = m.group(1) + total_time = m.group(2) + continue + m = RE_TASK.match(line) + if m: + cur_task = m.group(1) + if line.endswith("FAILED") and m.group(1) not in failed_tasks: + failed_tasks.append(m.group(1)) + continue + if "FAILURE: Build failed with an exception" in line: + grab_headline = True + continue + if grab_headline and line.strip().startswith("* What went wrong"): + headlines.append("") + continue + if grab_headline: + s = line.strip() + if s.startswith("*") or not s: + if headlines and headlines[-1]: + grab_headline = False + continue + if len(headlines[-1]) < 200: + headlines[-1] += (" " if headlines[-1] else "") + s + continue + m = RE_COMP.search(line) + if m: + total = int(m.group(1)) + failed = int(m.group(2) or 0) + skipped = int(m.group(3) or 0) + add_counts(cur_task or "(unknown)", total - failed - skipped, failed, skipped) + continue + m = RE_GSUM.search(line) + if m: + n, verb = int(m.group(1)), m.group(2) + mm = mod(cur_task or "(unknown)") + mm["total"] += n + tests["total"] += n + if verb == "failed": + mm["failed"] += n + tests["failed"] += n + elif verb == "skipped": + mm["skipped"] += n + tests["skipped"] += n + continue + m = RE_GTEST_P.search(line) + if m: + add_counts(cur_task or "(gtest)", int(m.group(1)), 0, 0) + continue + m = RE_GTEST_F.search(line) + if m: + add_counts(cur_task or "(gtest)", 0, int(m.group(1)), 0) + continue + if re.search(r" > \S+(?:\.\S+)+.* FAILED$", line) or re.search(r"\[ FAILED \] (\S+)", line): + test_failures.append(line.strip()) + continue + if RE_DEP.search(line) and len(dep_issues) < 20: + if line not in dep_issues: + dep_issues.append(line.strip()) + continue + wm = RE_WARN.search(line) + if wm and len(warnings) < 40: + key = line.strip()[:160] + if key not in seen_warnings: + seen_warnings.add(key) + warnings.append(key) + +failed = [t + (" — " + h if h else "") for t, h in zip(failed_tasks, headlines + [""] * len(failed_tasks))] +data = {"status": status, "totalTime": total_time, "failedTasks": failed, "warnings": warnings, + "tests": tests, "slowTasks": [], "depIssues": dep_issues, "actions": []} + +os.makedirs(OUT_DIR, exist_ok=True) +with open(JS, "w") as f: + json.dump(data, f, indent=1) + +L = ["# Gradle Summary\n", f"- Status: {status}", f"- Total time: {total_time or 'UNKNOWN'}", + f"- Log: {log_path}\n"] +if failed: + L.append("## Failing Tasks") + L += [f"- {t}" for t in failed] + L.append("") +if headlines: + L.append("## Primary Failure") + L += [f"```text\n{h}\n```" for h in headlines[:3] if h] + L.append("") +if tests["total"]: + L.append("## Tests") + L.append(f"- Total: {tests['total']}, failed: {tests['failed']}, skipped: {tests['skipped']}") + for name, m in tests["modules"].items(): + L.append(f" - {name}: {m['total']} total, {m['failed']} failed, {m['skipped']} skipped") + if test_failures: + L.append("- Top failing tests:") + L += [f" - {t}" for t in test_failures[:10]] + L.append("") +if warnings: + L.append("## Warnings") + L += [f"- {w}" for w in warnings[:20]] + L.append("") +if dep_issues: + L.append("## Dependency / Network Issues") + L += [f"- {d}" for d in dep_issues[:10]] + L.append("") +L.append("- Per-task durations are not present in plain `-i` console logs; slowTasks omitted.") +open(MD, "w").write("\n".join(L) + "\n") + +print(f"{status} (time: {total_time or 'n/a'}); {len(failed)} failing task(s), {tests['failed']} test failure(s)") +print(f"Wrote {MD} and {JS}") +PYEOF diff --git a/.gitignore b/.gitignore index 1c9dd43f2d..80a392deea 100644 --- a/.gitignore +++ b/.gitignore @@ -44,4 +44,14 @@ doc/temp/ # CLAUDE.md is auto-generated from AGENTS.md bootstrap instructions CLAUDE.md -.sphinx +# OS/editor cruft +.DS_Store + +# AI review/agent tooling scratch state +.sphinx/ +.skill-builder-temp/ +.claude/scheduled_tasks.lock + +# Python bytecode cache +__pycache__/ +*.pyc From e507b4f52e904e6afcb66dc8bafbddd8dc9c0e1f Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Wed, 16 Sep 2026 17:41:34 +0200 Subject: [PATCH 09/18] Mark the missing-refchains-on-hotdog investigation done Close out the investigation record: status active -> done, completed 2026-09-16. last_commit stamps the content the investigation actually concluded at (9282d5f25, the pre-history-optimization branch head, preserved locally by backup/pre-history-optimize-rc-pi). --- .investigations/missing-refchains-on-hotdog/meta.yaml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/.investigations/missing-refchains-on-hotdog/meta.yaml b/.investigations/missing-refchains-on-hotdog/meta.yaml index 09a62cf2a9..201749c7a5 100644 --- a/.investigations/missing-refchains-on-hotdog/meta.yaml +++ b/.investigations/missing-refchains-on-hotdog/meta.yaml @@ -1,8 +1,8 @@ name: missing-refchains-on-hotdog description: Why prof-analyzer-hotdog-jb emits zero datadog.ReferenceChain events -status: active +status: done created: 2026-08-24 updated: 2026-09-16 -completed: null -last_commit: 50edce5bb8c1d8fe471efb07dd850543215cb55c +completed: 2026-09-16 +last_commit: 9282d5f25195f0330f12c108ba6493138d68bb88 last_session: e6d2c60c-e959-420c-9b65-5e0400073a24 From cce988a1ba945c9bb5868287ec8769633f894bf8 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Thu, 17 Sep 2026 14:12:46 +0200 Subject: [PATCH 10/18] Add as-built implementation reference for reference chains Detailed record of the walk engine as implemented: pass structure, root discovery (including the static-field sweep), candidate-scoped reach, canary tagging mechanics, rotation tiers, pacing/budget arithmetic, and the LivenessTracker admission boost. Cross-references the design, component-summary, and signals documents instead of repeating them. --- .../LiveHeapReferenceChains-Implementation.md | 388 ++++++++++++++++++ 1 file changed, 388 insertions(+) create mode 100644 doc/architecture/LiveHeapReferenceChains-Implementation.md diff --git a/doc/architecture/LiveHeapReferenceChains-Implementation.md b/doc/architecture/LiveHeapReferenceChains-Implementation.md new file mode 100644 index 0000000000..4d5bb2f754 --- /dev/null +++ b/doc/architecture/LiveHeapReferenceChains-Implementation.md @@ -0,0 +1,388 @@ +# Live Heap Reference Chains — As-Built Implementation Reference + +**Status:** matches `jb/reference-chains` as of 2026-09 +**Jira:** [PROF-15341](https://datadoghq.atlassian.net/browse/PROF-15341) + +This document is the detailed, as-built reference for the reference-chain walk +engine. It records the mechanisms exactly as implemented, with the rationale +that lives in the code comments distilled into one place. + +Companion documents, each with its own scope: + +| Document | Scope | +|---|---| +| `doc/reference-chains-design.md` | The original design decision: why bounded BFS-from-roots was chosen over the alternatives | +| `doc/reference-chains-collection-summary.md` | The four-component summary: detection signal, resumable walk, latency budgets, rotation | +| `doc/architecture/LiveHeapReferenceChains.md` | The full architecture: frontier/tag lifecycle, triggering, termination, data structures | +| `doc/architecture/ReferenceChains-SignalsExplained.md` | A guided tour of the *scheduling* side: cadence, leak-signal gate, pain budget, PID controller, canary backoff, OOM ramp, abort path | + +This document covers what none of the above covers in one place: the pass +structure, root discovery (including static-field roots), candidate-scoped +reach, the canary tagging mechanics, the rotation tiers, the as-built +pacing/budget arithmetic, and the `LivenessTracker` admission boost. Where a +topic belongs to the signals tour (scheduling behavior), this document +cross-references it instead of repeating it. + +--- + +## 1. Data flow + +```mermaid +flowchart TD + LT["LivenessTracker::selectLeakCandidates (population-slope ranking)"] -->|"ranked klass candidates"| PWT["ReferenceChainTracker::pollWatchedTargets"] + PWT -->|"pre-tag each candidate's representative with a distinct marker tag"| CHASE["canary candidate chase"] + BFS["BFS thread: threadLoop"] -->|"shouldRunPass gate (see SignalsExplained §4-§8)"| RP["runPass"] + RP -->|"every pass"| MW["runPassManualWalk"] + MW -->|"cadence-gated (>= 2s apart)"| RE["IterateOverReachableObjects seeds roots"] + MW -->|"when the loaded-class set changed since the last completed sweep"| SF["admitStaticFieldRoots: app-classes-first, chunked, resumable sweep"] + MW -->|"candidates open: before any breadth-first work"| PRONGS["candidate-scoped reach: walkCandidateThreadLocals + walkStaticFieldAnchors descend walks"] + MW --> EF["expandFrontier: batched array-holder FollowReferences"] + EF --> FT["FrontierTable: FRONTIER / EXPANDED / EDGE / ABANDONED"] + MW -->|"reserved rotation slice"| ROT["rotation: leak-accumulation (Tier 1/2) + stale-EXPANDED + root-kind + static-anchor FIFO"] + ROT --> EF + CHASE -->|"heapReferenceCallback prunes a marker-tagged candidate: chain link recorded"| PWT + PWT -->|"buildChainEvent (representative, pruned marker, and up to 8 auto-discovered instances/class)"| RC["cacheResolvedChain: one entry per klass id, cap 128"] + RC -->|"Profiler::dump, snapshot without clearing"| DR["drainPendingChainEvents"] + DR --> JFR["datadog.ReferenceChain / datadog.ReferenceChainAbandoned"] +``` + +Every pass is driven by the manual walk (`runPassManualWalk()`): pure JVMTI +heap calls, which run inside the `VM_HeapWalkOperation` safepoint and honor +ZGC's load barriers — the walk reads no raw oop, so concurrent relocation +cannot corrupt it. The phases below run in this order, each with its own +deadline slice (the per-sub-operation deadline is reset, so an earlier phase +cannot eat a later phase's slice): + +1. root/stack-ref enumeration (cadence-gated), +2. candidate thread-local descend walks (when a chase is open), +3. static-field sweep (when the loaded-class set changed), +4. `expandFrontier()` (ordinary breadth-first progress), +5. static-anchor descend walks, +6. rotation re-expansion. + +A rotation slice is reserved up front, before any of the above can spend the +whole pass budget. The reservation is capped at half the expand budget: a +pacing-throttled pass degrades both sides proportionally instead of starving +ordinary expansion to zero (or rotation to zero). + +## 2. Root discovery + +### 2.1 Root/stack-ref enumeration + +`IterateOverReachableObjects` walks the roots and dispatches +`heapRootCallback()`/`stackRefCallback()`. Root enumeration pays its fixed +root-walk-and-dispatch cost in full on every run, regardless of budget, so it +is gated by `ROOT_ENUM_MIN_INTERVAL_NS` (2s): it does not re-fire at the +per-second pass cadence. A budget-exhausted truncation here retries on the +next pass; a frontier-cap hit abandons the search (below). Note that root +enumeration alone never discovers a root's transitive children — the +callbacks are given no oop, only a tag pointer — so `expandFrontier()` is +always needed for further progress, first pass or resumed. + +### 2.2 Static-field roots (`admitStaticFieldRoots()`) + +An object held only by `SomeClass.staticField` is not reachable through +`IterateOverReachableObjects`' root/stack-ref callbacks at all. This is +precisely the leak shape production pods showed: a growing `static final` +collection. The sweep: + +- Repartitions the per-call `GetLoadedClasses()` array app-classes-first (any + non-bootstrap classloader), in place, every call — `GetLoadedClasses()` + gives no ordering guarantee across calls, so the sweep reprioritizes each + time. A likely leak source is reached within the first chunks instead of + after every JDK platform class. +- Advances `STATIC_FIELD_SWEEP_CHUNK_CLASSES` (512) classes per pass through + a resumable cursor (`_static_field_sweep_cursor`). A single un-chunked + `FollowReferences` over every loaded class could not finish inside one + pass's safepoint deadline on a JVM with tens of thousands of classes. +- Retries a truncated chunk on the next pass; a lap that truncated even once + is not marked done (`_static_field_sweep_cycle_truncated`), per the + subsystem's "no silent truncation" contract. +- Only runs at all when the loaded-class set has *actually changed* since the + last completed lap (`_last_static_field_class_count` differs from the count + `resolveLoadedClasses()` refreshed this same pass). Otherwise it would + re-pay a stop-the-world walk over every class on every pass, forever. +- Admits `STATIC_FIELD` edges always; caps non-static class edges + (constant-pool, interface, superclass, classloader) at + `STATIC_FIELD_SWEEP_NON_STATIC_CAP_PER_CLASS` (32) per class, so one outlier + class cannot blow the chunk's deadline. + +### 2.3 Candidate-scoped reach (descend walks) + +Once a candidate chase is open, breadth-first progress alone can leave the +tagged instances queued behind the ordinary backlog for many passes. Each +open chase therefore gets bounded *descend walks* — `FollowReferences` from a +specific anchor, up to `DESCENT_HOPS` (16) hops below it (raised from 6 after +pod round 7: the static-`ExecutorService` → queue → task → accumulator → +list → chunk shape is ~7 deep; the walk is deadline-bounded per slice either +way, so a deeper cap just lets each bounded walk cover the whole holder +interior — already-admitted entries are skipped, so repeated passes march +deeper each time): + +- **Prong 1, thread-locals** (`walkCandidateThreadLocals()`): descend-walks + the qualifying threads' `Thread` objects — registered via + `Profiler::onThreadStart/onThreadEnd` + (`registerThreadObject()`/`unregisterThreadObject()`, a mutex-guarded + tid → global-ref map; a running thread's `Thread` object is reachable via + the VM anyway, so the strong ref does not distort reachability) — through + their `ThreadLocalMap` subgraphs. A `Thread → ThreadLocalMap → table[] → + Entry → value → holder → chunk` chain is 5-6 hops that ordinary BFS may + reach very late, if at all. Capped at `THREAD_WALK_MAX_ANCHORS` (4) per + pass, rotated fairly across the (klass, tid) pairs by a cursor. +- **Prong 2, static anchors** (`walkStaticFieldAnchors()`): descend-walks + root-attached static holders. Anchors are selected by + `collectStaticFieldAnchorsForRotation()` with *tiered* per-tier cursors + (fresh container-interface holders first); classes are classified once by + `reconcileAnchorClassShapes()` (capped at + `ANCHOR_SHAPE_RECONCILE_BUDGET` = 128 distinct classes per pass — + `resolveContainerInterfaceTags()`/`classImplementsContainerOrMap()` decide + whether a class implements a `Collection`/`Map`-like interface); at-risk + holders (frontier entries whose parent died) flow through a FIFO + (`pushAtRiskStaticAnchor()`/`drainStaticAnchorFifo()`, popped + `STATIC_ANCHOR_FIFO_DRAIN` = 16 per pass). Capped at + `STATIC_ANCHOR_ROTATION_BUDGET` (32) anchors per pass. Selection size and + walked size are decoupled: a truncation requeues un-walked anchors + (`requeueStaticAnchorFifoFront()`), so a larger selection never extends the + pause — it can only spend selection-scan time, which holds no safepoint. + +Both prongs resolve their anchor batch with a single `GetObjectsWithTags` +call (the call's O(tag_map) cost is the dominant term on a large tag map — +the same batching rationale as `expandFrontier()`). + +## 3. Frontier expansion and batching + +`expandFrontier()` resolves up to `budget` pending tags in *one* +`GetObjectsWithTags` call, packs the live objects into a single JNI array +(`holder`), and expands the whole batch via exactly one +`FollowReferences(initial_object = holder)` — one BFS level per +VM-safepoint operation rather than one safepoint per entry: + +```mermaid +sequenceDiagram + participant BFS as BFS thread + participant JVMTI as JVMTI + participant JNI as JNI holder array + participant CB as heapReferenceCallback + + BFS->>BFS: pull up to budget tags from front of _pending_expand + BFS->>JVMTI: GetObjectsWithTags, resolve which tags are still live + JVMTI-->>BFS: live jobject references + BFS->>JNI: EnsureLocalCapacity, then NewObjectArray to build holder + alt exception, EnsureLocalCapacity failure, or null holder + BFS->>BFS: ctx.truncated = true, retry this batch next pass + else holder built successfully + BFS->>JNI: SetObjectArrayElement per resolved object + BFS->>JVMTI: FollowReferences, initial_object = holder + JVMTI->>CB: heapReferenceCallback per outgoing edge + CB-->>JVMTI: descend only for batch_tags boundary objects + JVMTI-->>BFS: one BFS hop expanded for the whole batch + BFS->>BFS: markExpanded, admitObject appends children to _pending_expand + end +``` + +The JNI/JVMTI error handling is defensive by construction: a null `holder`, +a pending JNI exception after `NewObjectArray`/`SetObjectArrayElement`, or an +`EnsureLocalCapacity` failure all set `ctx.truncated = true` (retry next +pass) instead of marking the batch permanently `EXPANDED`. A failed +`java/lang/Object` class resolution with pending work also forces +`truncated = true`, so `runPass()` cannot mistake it for +`SearchState::COMPLETED`. + +## 4. The canary candidate chase + +`pollWatchedTargets()` pre-tags each candidate's *specific representative +object* (identity match — matching by class alone could record a chain for an +unrelated, possibly short-lived, instance of the same class) with a distinct +negative marker tag (`MARKER_TAG_BASE - i`; the negative range is disjoint +from the positive frontier-tag counter and the negative class-tag range). +When the walk's `heapReferenceCallback()` later encounters one, it *prunes* +it: records the candidate's referrer link (parent tag, referrer klass, +depth) for `buildCanaryChainEvent()`, and does not expand through it. + +The walk also auto-marks *any* instance of a watched class it discovers (up +to `MAX_DISCOVERED_INSTANCES_PER_CLASS` = 8 per slot): a leaking class's +many live instances each have independently useful chains — different +parents, different retention paths — and the pre-tagged representative is +just one sample. + +The chase's *scheduling* (back-to-back on progress, work-scaled exponential +backoff on no progress, pain-budget refill) is covered in +`ReferenceChains-SignalsExplained.md` §8, including the measured pod incident +(32 minutes at ~88 passes/min) that motivated the backoff. The *termination* +half is here: a chase that has made zero candidate-discovery progress for +`CANARY_NO_PROGRESS_PASS_LIMIT` (30) consecutive passes is abandoned with +`SearchAbandonReason::CANARY_STUCK`. Unlike the TTL check, this fires even +while `isUrgent()` holds: a zero-progress chase at urgency-boosted +budget/cadence only burns STW pause budget the process needs during the same +OOM approach the chase was launched to diagnose. The abandon limit *widens* +with repeated stuck restarts (`_canary_stuck_restart_count` survives +`restartSearch()`), so a permanently un-findable candidate cannot keep +cycling at the base limit. + +## 5. Rotation: rediscovering growth in an already-visited container + +The frontier walk visits each object once. That is insufficient for the leak +shape this feature targets: a `static final` collection field that is +*appended to*, not reassigned. The container is already `EXPANDED` long +before its element klass earns a leak signal. `runPass()`'s completion branch +therefore requires `_watched_leak_klass_count == 0` (no klass under active +leak watch) before moving to `SearchState::COMPLETED`; while a klass is +watched, the search stays `RUNNING` and rotation re-queues the growing +container: + +```mermaid +flowchart TD + LT2["LivenessTracker::topKlassesByGenerationCount"] -->|"refreshed once per tick,
    only after hasLeakSignal() fires"| WK["_watched_leak_klass_ids
    (max 5)"] + WK -->|"klass_id newly watched"| SEED["seedLeakAccumulationForNewlyWatchedKlass:
    one-time scan of already-EXPANDED
    frontier entries by class_tag"] + ADM["admitObject: ADMITTED"] -->|"class_tag of new object"| TLA["trackLeakAccumulation"] + SEED --> TLA + WK -->|"class_tag match?"| TLA + TLA --> T1["Tier 1: _leak_signature_totals
    (leaf_klass_id, parent_class_id) -> count"] + TLA --> T2["Tier 2: _leak_parent_fanout
    parent_tag -> count, within winning signature"] + T1 -->|"delta vs previous pass's snapshot"| RANK["collectLeakAccumulationCandidatesForRotation:
    pick winning signature, then its top parent_tag(s)"] + T2 --> RANK + RANK -->|"re-queue for re-expansion,
    budget 16/pass"| EF2["expandFrontier"] + EF2 -->|"new elements admitted"| ADM +``` + +Matching a newly-admitted object against a watched klass id uses a stable +JVMTI class tag (`classTagAllocator.h`, shared between +`ReferenceChainTracker` and `LivenessTracker`) rather than the classMap +`StringDictionary` id — that id is not guaranteed stable if the dictionary +is compacted/regenerated mid-search (observed live: the same class resolved +to two different classMap ids from two subsystems). + +The rotation tiers, with their per-pass budgets: + +| Tier | Budget/pass | What it selects | +|---|---|---| +| Leak-accumulation (Tier 1/2 above) | 16 | The containers of currently-flagged klasses — the targeted tier | +| Stale-EXPANDED (own cursor) | 256 | Any long-`EXPANDED` entry — low-priority eventual coverage of the whole table; deliberately *not* scaled with table size (two earlier versions tried; see git history) | +| Root-kind (own cursor) | 16 | Transient-root entries, so attribution converges to a durable root kind | +| Static-anchor FIFO | 16 drained | At-risk holders whose parent died, into the same `walkStaticFieldAnchors()` batch | + +The stale-EXPANDED tier's own cursor exists so a frontier table full of +long-lived infrastructure objects cannot permanently starve higher-tag +entries of ever being re-queued. + +## 6. Termination + +```mermaid +stateDiagram-v2 + direction LR + [*] --> RUNNING + RUNNING --> COMPLETED: frontier drained,
    no truncation this pass,
    no leak klass still watched + RUNNING --> ABANDONED_TTL: wall-clock TTL exceeded
    with work still pending
    (suppressed while urgent) + RUNNING --> ABANDONED_CAP: frontier-size cap hit + RUNNING --> ABANDONED_CANARY: zero candidate-discovery progress for
    CANARY_NO_PROGRESS_PASS_LIMIT (30) passes + ABANDONED_TTL --> RUNNING: restartSearch + ABANDONED_CAP --> RUNNING: restartSearch + ABANDONED_CANARY --> RUNNING: restartSearch + COMPLETED --> RUNNING: restartSearch +``` + +The abandon reason is recorded (`SearchAbandonReason`) and surfaced as the +`datadog.ReferenceChainAbandoned` JFR event — no silent truncation. +`releaseSearchTags()` clears every live tag the search still owns once it +ends, without discarding the frontier table's own records: chain +reconstruction keeps working from memory after the search ends. + +Restarts are gated on an actual leak indication (a +`selectLeakCandidates()` candidate, or an urgent-latched seconds-to-OOM +projection — one search per latched episode via `_urgent_search_spent`) plus +the safepoint pain budget. This closes a structural gap: a one-shot walk +could finish before population-trend detection accumulated enough GC epochs +to flag a candidate, leaving anything allocated afterward permanently +undiscoverable. See `ReferenceChains-SignalsExplained.md` §5-§6 for the +gate's full behavior. + +## 7. Pacing and budgets (as-built arithmetic) + +- **PID controller on genuine in-safepoint time.** `updatePacing()` is fed + only each pass's in-safepoint ticks — measured across every + `IterateOverReachableObjects`/`FollowReferences` call the pass makes, + explicitly excluding `GetObjectsWithTags` (not a safepoint call) and every + bookkeeping line in between. Non-safepoint bookkeeping must not be + mistaken for pause-time-SLO pressure. The controller's overflow term also + widens/narrows the fallback cadence. +- **Separate CPU pain budget.** Non-safepoint pass cost is charged to + `_cpu_pain_budget`, a second `PainBudget` sharing the same + `painbudget=N` percent knob as the safepoint pain budget — one + operator-facing "acceptable background cost" percentage covers both leaky + buckets. +- **Budget borrowing.** A sustained run of comfortably-under-target passes + (after a warmup) earns extra headroom above `_budget` + (`_borrowed_budget`); any pass that is not comfortably under target + revokes it immediately (`maybeRevokeBorrowForRootEnumPass()` also revokes + on a root-enum pass), so `_budget` itself stays the ceiling the instant + the search stops proving it has room. +- **First-pass budget.** The search's one-shot root-seeded first pass draws + its own much larger edge budget (`firstpassbudget`, auto-scaled 10× from + `budget`, capped) exactly once — a steady-state budget sized for cheap + incremental expansion would truncate a cold full-graph walk long before + it reaches anything interesting — and its duration is excluded from the + pacing signal so it cannot throttle every cheap pass that follows. +- **Urgency ramp.** See `ReferenceChains-SignalsExplained.md` §9: pause + target and cadence ramp exponentially toward their ceilings as + `secondsToOOM()` falls inside `OOM_RAMP_START_S` (30 min); the budget + ceiling is held at 4× for the ramp's entire duration; the ramp owns + `_effective_cadence_ns` outright while active. +- **Auto-tuned defaults.** `autoTuneDefaults()` scales the defaults for + budget (√heap-proportional), first-pass budget, TTL, frontier cap, and + pause target (capped at 50ms) with the resolved max heap / container + limit, but only for sub-options the operator did not set explicitly. + +The urgency *latch* is covered in `ReferenceChains-SignalsExplained.md` §9. +On the `LivenessTracker` side, `secondsToOOM()` itself: accounts for the +container memory limit (not just `-Xmx`) when projecting the exhaustion +point; requires a confirmed rising heap-floor trend; and corroborates the +projection with a recent-half slope check, so a single stale sample cannot +spike or collapse it. + +## 8. `LivenessTracker` admission boost (chase phase) + +While a candidate chase is open, allocations by the candidates' *qualifying +threads* are admitted to liveness tracking at 100% (`admitForTracking()`, +published two-phase — slots first, then the count with RELEASE — by +`noteSelectedCandidates()`), and urgency admits everything. Without this, +the default 10% liveness subsampling thins small per-(klass, tid) +populations enough to make leak-tag correlation intermittently fail on real +pods (observed live as the intermittent zero-tag runs in +`LeakTagCorrelationReferenceChainTest`'s own lottery analysis). The boost +only ever adds admissions on top of the configured ratio — fail-open by +construction: a stale or missed boost cannot drop an allocation the ratio +would have admitted. The draw itself is the process-wide xorshift64 stream +from main's sampling refactor (#794): per-thread TLS state, an integer +threshold compare, no `` machinery. + +## 9. Output path and JFR persistence + +A resolved chain is cached per klass id (`_resolved_chains`, capped at +`MAX_RESOLVED_CHAINS` = 128 entries, drop-not-evict once full — surfaced as +`REFERENCE_CHAIN_EVENTS_DROPPED`) and re-stamped into *every* subsequent +dump the sample survives into (`drainPendingChainEvents()` snapshots without +clearing, mirroring how `LivenessTracker` re-emits live-object samples). A +long-lived leak's chain is therefore present in each JFR chunk, not only in +the chunk active when it was first reconstructed. The write happens on the +dump thread (`Profiler::dump()`), never on the BFS thread — the walk never +blocks on JFR I/O, and JFR writes never trigger a walk. + +## 10. Configuration + +`referencechains=true:hops=N:budget=N:ttl=N:framecap=N:pausetarget=N:painbudget=N:firstpassbudget=N` + +- Negative `hops`/`budget`/`framecap` values are floored (an unfloored + negative `hops` would wrap to ~4e9 as a `u32`, silently disabling the cap + it is meant to enforce); all three are also ceiling-clamped. +- `ttl <= 0` disables the wall-clock TTL cutoff. +- Unset values are auto-tuned (see §7). + +## 11. Temporary diagnostics + +Some `TEST_LOG_SUMMARY` output and tallies in the static-field sweep and the +anchor walk (chunk-consumption splits, per-`reference_kind` callback tallies, +walked-anchor naming) are *temporary diagnostic* instrumentation from the +live pod debugging rounds (`.investigations/missing-refchains-on-hotdog/`, +16 rounds) that shaped this design. They are pending removal once the +on-pod investigation notes are fully distilled. Production behavior does not +depend on them. From cf6ba2d6920399ebd1356fa1e1bbd9a459ca9520 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Thu, 17 Sep 2026 14:16:31 +0200 Subject: [PATCH 11/18] Word the collector-safety rationale collector-neutrally The walk is correct on every collector, not only ZGC: JVMTI's iterators apply the active collector's own barriers, and the walk reads no raw oop. ZGC's concurrent relocation is the worst case a raw-oop reader would face, not the only collector the walk supports. --- ddprof-lib/src/main/cpp/referenceChains.cpp | 10 ++++++---- .../LiveHeapReferenceChains-Implementation.md | 7 ++++--- 2 files changed, 10 insertions(+), 7 deletions(-) diff --git a/ddprof-lib/src/main/cpp/referenceChains.cpp b/ddprof-lib/src/main/cpp/referenceChains.cpp index 7e57c121ec..c0f330667c 100644 --- a/ddprof-lib/src/main/cpp/referenceChains.cpp +++ b/ddprof-lib/src/main/cpp/referenceChains.cpp @@ -5661,10 +5661,12 @@ bool ReferenceChainTracker::runPass(jvmtiEnv *jvmti, JNIEnv *jni, // Every pass is driven by the manual walk (runPassManualWalk() - // IterateOverReachableObjects for roots, then a batched array-holder - // FollowReferences per BFS level in expandFrontier()), on every collector - // including ZGC. The walk issues only JVMTI heap calls, which run inside - // the VM_HeapWalkOperation safepoint and honor ZGC's load barriers, so - // concurrent relocation cannot corrupt it - it reads no raw oop. Batching + // FollowReferences per BFS level in expandFrontier()), on every + // collector. The walk issues only JVMTI heap calls, which run inside the + // VM_HeapWalkOperation safepoint; JVMTI's iterators apply the active + // collector's own barriers, so the walk is correct on every collector - + // including ZGC, where concurrent relocation would corrupt a raw-oop + // reader. It reads no raw oop. Batching // one hop per level keeps each FollowReferences bounded, avoiding the // multi-hundred-ms-to-second STW pauses a whole-graph FollowReferences // would impose. diff --git a/doc/architecture/LiveHeapReferenceChains-Implementation.md b/doc/architecture/LiveHeapReferenceChains-Implementation.md index 4d5bb2f754..f0d256217f 100644 --- a/doc/architecture/LiveHeapReferenceChains-Implementation.md +++ b/doc/architecture/LiveHeapReferenceChains-Implementation.md @@ -47,9 +47,10 @@ flowchart TD ``` Every pass is driven by the manual walk (`runPassManualWalk()`): pure JVMTI -heap calls, which run inside the `VM_HeapWalkOperation` safepoint and honor -ZGC's load barriers — the walk reads no raw oop, so concurrent relocation -cannot corrupt it. The phases below run in this order, each with its own +heap calls, which run inside the `VM_HeapWalkOperation` safepoint. The walk +reads no raw oop — JVMTI's iterators apply the active collector's own +barriers, so the walk is correct on every collector, including ZGC, where +concurrent relocation would corrupt a raw-oop reader. The phases below run in this order, each with its own deadline slice (the per-sub-operation deadline is reset, so an earlier phase cannot eat a later phase's slice): From 812ace9bebc3b913beea00475ab5ea5ffbd16317 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Thu, 17 Sep 2026 14:40:52 +0200 Subject: [PATCH 12/18] Remove temporary diagnostic instrumentation from reference chains --- ddprof-lib/src/main/cpp/referenceChains.cpp | 370 +----------------- ddprof-lib/src/main/cpp/referenceChains.h | 31 +- .../src/test/cpp/referenceChains_ut.cpp | 116 +++--- 3 files changed, 51 insertions(+), 466 deletions(-) diff --git a/ddprof-lib/src/main/cpp/referenceChains.cpp b/ddprof-lib/src/main/cpp/referenceChains.cpp index c0f330667c..9f408952c8 100644 --- a/ddprof-lib/src/main/cpp/referenceChains.cpp +++ b/ddprof-lib/src/main/cpp/referenceChains.cpp @@ -347,11 +347,8 @@ bool FrontierTable::improveChain(jlong tag, jlong parent_tag, // root_kind=0 referrer_klass= - collector-invisible // (the parent_tag==0 eligibility filter) and never re-walked, because // improveChain(depth=holder.depth+1) "improved" the entry with itself. - // A self-parent would also dead-loop reconstructChain(). Refuse, and - // count the refusal (selfEdgeGuardSkips()): the counter climbing on the - // pod is the verification that the guard fires on the real wrapper. + // A self-parent would also dead-loop reconstructChain(). Refuse. if (parent_tag == tag) { - _self_edge_guard_skips.fetch_add(1, std::memory_order_relaxed); return false; } int idx = (int)(tag - 1); @@ -388,9 +385,6 @@ bool FrontierTable::reparentToDurableRoot(jlong tag, jlong new_parent_tag, // self-parent a this-field (mutex == this) would deliver. if (tag <= 0 || tag - 1 > (jlong)INT_MAX || new_parent_tag <= 0 || new_parent_tag - 1 > (jlong)INT_MAX || new_parent_tag == tag) { - if (new_parent_tag == tag) { - _self_edge_guard_skips.fetch_add(1, std::memory_order_relaxed); - } return false; } int idx = (int)(tag - 1); @@ -1438,8 +1432,6 @@ void ReferenceChainTracker::resetSearchStateForTest(jvmtiEnv *jvmti, _static_anchor_fifo.clear(); _static_anchor_fifo_set.clear(); _static_anchor_fifo_klass_counts.clear(); - _static_anchor_fifo_quota_drops = 0; - _static_anchor_fifo_pushed = 0; _static_anchor_index.clear(); _static_anchor_own_class_tags.clear(); _static_anchor_index_tags.clear(); @@ -1840,14 +1832,6 @@ struct PassContext { // re-discovered subtree, not just the immediate child, skips the backlog. bool admit_priority = false; - // TEMP DIAGNOSTIC (see doc/temp/ investigation notes): per-jvmtiHeapReference - // Kind callback tally for admitStaticFieldRoots()'s current chunk, to find - // which reference kind actually drives per-chunk callback volume (e.g. - // static fields vs. constant-pool entries vs. interfaces). Null everywhere - // else - only admitStaticFieldRoots() sets this to a non-null, zeroed, - // stack-local array sized for the full jvmtiHeapReferenceKind range. - int *kind_counts = nullptr; - // DESCEND-WALK controls (descendFromAnchor()'s calls only; null/0 // everywhere else, so every gate below is a no-op for the ordinary // walk phases): @@ -1877,24 +1861,6 @@ struct PassContext { int _no_descend_class_tag_count = 0; jlong _descent_anchor_tag = 0; jlong _anchor_descend_class_tag = 0; - - // TEMP DIAGNOSTIC (pod round 12 - wrapper walked but never intercepted): - // when _diag_trace is set (only descendFromAnchor() sets it, and only for - // an anchor whose class is Collections$UnmodifiableRandomAccessList - - // the LEAK_BUFFER wrapper shape), heapReferenceCallback()'s admission and - // already-tagged-encounter sites record (klass_id, how-seen) pairs for the - // first DIAG_MAX_ENTRIES entries, so one walk's actual traversal is - // observable: does the wrapper -> list -> elementData -> [B chunk chain - // get enumerated AT ALL, and did the enumerated chunks carry leak tags at - // that moment. _diag_leak_flags: 0 = fresh admission, 1 = admission via - // leak-tag conversion (an interception - never observed for the wrapper - // so far), 2 = already-frontier-tagged re-encounter. TEMP: overfit to the - // round-12 wrapper question - remove once the pod answers it. - static constexpr int DIAG_MAX_ENTRIES = 48; - bool _diag_trace = false; - int _diag_count = 0; - u32 _diag_klass_ids[DIAG_MAX_ENTRIES]; - u8 _diag_leak_flags[DIAG_MAX_ENTRIES]; }; } // namespace @@ -1905,15 +1871,6 @@ jint JNICALL ReferenceChainTracker::heapReferenceCallback( jlong *referrer_tag_ptr, jint length, void *user_data) { PassContext *ctx = (PassContext *)user_data; - // TEMP DIAGNOSTIC (see doc/temp/ investigation notes): tally every callback - // by kind before any early-return below, so an aborted/truncated chunk - // still reports what it actually saw. kind_counts is only non-null for - // admitStaticFieldRoots()'s call - zero overhead elsewhere. - if (ctx->kind_counts != nullptr && (int)reference_kind >= 0 && - (int)reference_kind < 32) { - ctx->kind_counts[(int)reference_kind]++; - } - if (ctx->tracker->_abort_pass_requested.load(std::memory_order_relaxed)) { // stopThread() has set this right before pthread_kill()/pthread_join() - // see that method's own comment. pthread_kill(WAKEUP_SIGNAL) only @@ -2170,13 +2127,6 @@ jint JNICALL ReferenceChainTracker::heapReferenceCallback( "parent_tag=%lld", (long long)leak_tag, (long long)frontier_tag, depth, (long long)parent_tag); - if (ctx->_diag_trace && - ctx->_diag_count < PassContext::DIAG_MAX_ENTRIES) { - ctx->_diag_klass_ids[ctx->_diag_count] = - ctx->tracker->classTags()->resolve(class_tag); - ctx->_diag_leak_flags[ctx->_diag_count] = 1; - ctx->_diag_count++; - } ctx->tracker->trackLeakAccumulation(ctx->frontier, class_tag, parent_tag, frontier_tag); // Index maintenance: a leak-tagged object admitted root-attached by @@ -2265,27 +2215,6 @@ jint JNICALL ReferenceChainTracker::heapReferenceCallback( // *tag_ptr == 0 rules out ALREADY_ADMITTED) - nothing further to do. break; } - if (ctx->_diag_trace && - ctx->_diag_count < PassContext::DIAG_MAX_ENTRIES && - result == ReferenceChainTracker::AdmitResult::ADMITTED) { - ctx->_diag_klass_ids[ctx->_diag_count] = referrer_klass; - ctx->_diag_leak_flags[ctx->_diag_count] = 0; - ctx->_diag_count++; - } - // TEMP DIAGNOSTIC (pod round 12): log every root-attached - // STATIC_FIELD admission so we can see which classes are admitted - // as static-field holders by the sweep. The LEAK_BUFFER wrapper - // (SynchronizedRandomAccessList) must appear here if the sweep - // processes ProfileAnalyzer's class. Remove once the wrapper - // question is answered. - if (result == ReferenceChainTracker::AdmitResult::ADMITTED && - parent_tag == 0 && - root_kind == (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD) { - TEST_LOG("ReferenceChainTracker::heapReferenceCallback " - "root-attached STATIC_FIELD admit klass_id=%u " - "frontier_tag=%lld depth=%u", - referrer_klass, (long long)*tag_ptr, depth); - } // Index maintenance: track root-attached durable anchors for O(anchors) // collector iteration instead of O(frontier_size) table scan. class_tag // is the anchor object's OWN class tag (the callback's class_tag param @@ -2341,31 +2270,6 @@ jint JNICALL ReferenceChainTracker::heapReferenceCallback( } } } else if (*tag_ptr > 0) { - if (ctx->_diag_trace && - ctx->_diag_count < PassContext::DIAG_MAX_ENTRIES) { - ctx->_diag_klass_ids[ctx->_diag_count] = - ctx->tracker->classTags()->resolve(class_tag); - ctx->_diag_leak_flags[ctx->_diag_count] = 2; - ctx->_diag_count++; - } - // TEMP DIAGNOSTIC (pod round 12): log every STATIC_FIELD edge - // from the sweep that hits an already-admitted entry, with the - // entry's current shape (parent_tag, root_kind, state). This shows - // whether the LEAK_BUFFER wrapper (SynchronizedRandomAccessList) - // is ever reached by the sweep and what its frontier entry looks - // like. Remove once the wrapper question is answered. - if (ctx->static_field_seed && - reference_kind == JVMTI_HEAP_REFERENCE_STATIC_FIELD) { - FrontierEntry e{}; - bool found = ctx->frontier->lookup(*tag_ptr, &e); - TEST_LOG("ReferenceChainTracker::sweep STATIC_FIELD " - "already-admitted klass_id=%u tag=%lld " - "parent=%lld root_kind=%u state=%u found=%d", - ctx->tracker->classTags()->resolve(class_tag), - (long long)*tag_ptr, (long long)(found ? e.parent_tag : 0), - (unsigned)(found ? e.root_kind : 0), - (unsigned)(found ? e.state : 0), (int)found); - } // Already-tagged object reached via a new edge. This arm - NOT the // first-admission block above - is where an already-admitted entry's // shape can be corrected: improveChain/reparentToDurableRoot for a @@ -2694,13 +2598,6 @@ bool ReferenceChainTracker::maybeUpgradeRootAttachedRootKind( return false; } frontier->updateRootKind(tag, new_root_kind); - // TEMP DIAGNOSTIC (pod round 12): include klass_id so the LEAK_BUFFER - // wrapper (SynchronizedRandomAccessList) can be identified among the - // upgraded entries. Remove once the wrapper question is answered. - TEST_LOG("ReferenceChainTracker::maybeUpgradeRootAttachedRootKind tag=%lld " - "old_root_kind=%d -> new_root_kind=%d klass_id=%u", - (long long)tag, (int)entry.root_kind, (int)new_root_kind, - entry.referrer_klass); addToStaticAnchorIndex(tag, entry.class_tag, new_root_kind); return true; } @@ -3414,11 +3311,10 @@ void ReferenceChainTracker::descendFromAnchor( jvmtiEnv *jvmti, JNIEnv *jni, jobject anchor, jlong anchor_tag, u32 anchor_depth, jlong anchor_descend_class_tag, int budget, int *edges_admitted, bool *truncated, bool *frontier_cap_hit, - u64 *safepoint_ticks, bool diag_trace) { + u64 *safepoint_ticks) { PassContext ctx; ctx.tracker = this; ctx.frontier = _frontier; - ctx._diag_trace = diag_trace; // Bound admission to DESCENT_HOPS below the anchor, still subject to the // global hop cap. Caveat (accepted, bounded): a pre-existing frontier // entry reachable inside the subgraph carries its GLOBAL depth (from @@ -3447,22 +3343,6 @@ void ReferenceChainTracker::descendFromAnchor( u64 follow_start_ticks = TSC::ticks(); jvmti->FollowReferences(0, nullptr, anchor, &callbacks, &ctx); *safepoint_ticks += TSC::ticks() - follow_start_ticks; - if (diag_trace) { - // TEMP DIAGNOSTIC (pod round 12, PassContext::_diag_trace's own - // comment): dump this walk's admission/re-encounter sequence - the - // wrapper-walk question is exactly "was list/elementData enumerated, - // and did any leak-class ([B) entry carry a leak tag". - for (int i = 0; i < ctx._diag_count; i++) { - TEST_LOG("ReferenceChainTracker::descendFromAnchor diag anchor=%lld " - "entry=%d klass_id=%u seen_as=%u", - (long long)anchor_tag, i, ctx._diag_klass_ids[i], - (unsigned)ctx._diag_leak_flags[i]); - } - TEST_LOG("ReferenceChainTracker::descendFromAnchor diag anchor=%lld " - "recorded=%d edges=%d truncated=%d", - (long long)anchor_tag, ctx._diag_count, ctx.edges_admitted, - (int)ctx.truncated); - } *edges_admitted += ctx.edges_admitted; *truncated = *truncated || ctx.truncated; *frontier_cap_hit = *frontier_cap_hit || ctx.frontier_cap_hit; @@ -3635,24 +3515,6 @@ ReferenceChainTracker::collectStaticFieldAnchorsForRotation(int max_count) { fresh_room--; } }); - // TEMP DIAGNOSTIC (pod round 14/15): the cohort histogram - the - // measured sizes of the tiers against the per-pass budget. This is - // the arithmetic gate for the container tier (a container cohort much - // larger than ~4k cannot be covered within a ~44-75-pass search - // lifetime and the frontier-cap conversation becomes the next lever) - // and for the fresh lane (fresh > budget means the lane runs as a - // backlog - the leak holder then waits fresh/budget passes, still - // bounded, but the number belongs in the next analysis). Remove once - // the pod verifies the arithmetic. - if (!leak_picks.empty() || !fresh_picks.empty() || - !container_picks.empty() || !other_picks.empty()) { - TEST_LOG_SUMMARY("ReferenceChainTracker::anchorTierHistogram " - "index=%zu leak_tier=%zu fresh_tier=%zu fresh_queue=%zu " - "container_tier=%zu other_tier=%zu budget=%d", - idx_size, leak_picks.size(), fresh_picks.size(), - fresh_queue_len, container_picks.size(), other_picks.size(), - max_count); - } // Cursor-fair consumption of one tier: scan picks (sorted by pos by // construction) starting at entries with pos >= cursor, stop at `want` // OR at the lap end (NO within-call wrap: re-walking anchors this same @@ -3733,7 +3595,6 @@ void ReferenceChainTracker::pushAtRiskStaticAnchor(jlong tag, u32 klass_id) { // the feed's next event (the next static edge / next demotion), so // nothing is lost - the entry just cannot crowd out every other // class's repair. - _static_anchor_fifo_quota_drops++; return; } if (count_it == _static_anchor_fifo_klass_counts.end()) { @@ -3742,16 +3603,6 @@ void ReferenceChainTracker::pushAtRiskStaticAnchor(jlong tag, u32 klass_id) { count_it->second++; _static_anchor_fifo.push_back(AtRiskAnchor{tag, klass_id}); _static_anchor_fifo_set.insert(tag); - _static_anchor_fifo_pushed++; - // TEMP DIAGNOSTIC (pod round 12/16): track which classes enter the - // at-risk FIFO and the cumulative quota drops (round 16: the drops - // should be the flood classes' pushes, while fifo_size stays well - // under the 1024 cap so the LEAK_BUFFER wrapper's pushes land). Remove - // once the wrapper question is answered. - TEST_LOG("ReferenceChainTracker::pushAtRiskStaticAnchor tag=%lld " - "klass_id=%u fifo_size=%zu quota_drops=%llu", - (long long)tag, klass_id, _static_anchor_fifo.size(), - (unsigned long long)_static_anchor_fifo_quota_drops); } void ReferenceChainTracker::addToStaticAnchorIndex(jlong tag, @@ -3963,20 +3814,6 @@ void ReferenceChainTracker::reconcileAnchorClassShapes(jvmtiEnv *jvmti, ? AnchorClassShape::CONTAINER : AnchorClassShape::NON_CONTAINER; _class_shape_cache[class_tag] = (u8)shape; - // TEMP DIAGNOSTIC (pod round 14): name newly classified container - // classes so the cohort is identifiable in logs. Remove once the - // pod verifies the arithmetic. - if (shape == AnchorClassShape::CONTAINER) { - char *sig = nullptr; - if (jvmti->GetClassSignature(klass, &sig, nullptr) == - JVMTI_ERROR_NONE && - sig != nullptr) { - TEST_LOG("ReferenceChainTracker::reconcileAnchorClassShapes " - "container class %s (class_tag=%lld)", - sig, (long long)class_tag); - jvmti->Deallocate((unsigned char *)sig); - } - } } jvmti->Deallocate((unsigned char *)objs); jvmti->Deallocate((unsigned char *)obj_tags); @@ -4060,71 +3897,6 @@ void ReferenceChainTracker::walkStaticFieldAnchors( jni->DeleteLocalRef(objects[i]); continue; } - // TEMP DIAGNOSTIC (pod round 8: interception still zero over 16-hop - // walks of every selected anchor): name WHICH - // anchor objects are actually being walked, with their recorded chain - // shape, so a holder that never makes it into this tier (wrong - // root_kind / chain-attached / never-admitted) is distinguishable from - // one that gets walked without reaching the tagged chunks. - bool wrapper_trace = false; - { - char *sig = nullptr; - if (jni->GetObjectClass(objects[i]) != nullptr) { - jvmti->GetClassSignature(jni->GetObjectClass(objects[i]), &sig, - nullptr); - } - TEST_LOG("ReferenceChainTracker::walkStaticFieldAnchors anchor " - "tag=%lld class=%s klass_id=%u parent=%lld root_kind=%u state=%u " - "field_index=%d", - (long long)resolved_tags[i], sig != nullptr ? sig : "", - entry.referrer_klass, - (long long)entry.parent_tag, (unsigned)entry.root_kind, - (unsigned)entry.state, (int)entry.referrer_field_index); - // TEMP DIAGNOSTIC (pod round 12): the LEAK_BUFFER wrapper shape is a - // Collections.synchronizedList value held by a static field. The - // wrapper gets walked (observed rounds 10-11) yet never intercepts, - // so for wrapper-class anchors additionally name the HOLDER class - // (the class owning the admitting static field - resolves the - // "which synchronized list is this" question) and trace the walk's - // admission sequence (descendFromAnchor's diag_trace). TEMP: remove - // once the pod answers the wrapper question. - wrapper_trace = - sig != nullptr && - (strstr(sig, "UnmodifiableRandomAccessList") != nullptr || - strstr(sig, "SynchronizedRandomAccessList") != nullptr || - strstr(sig, "SynchronizedList") != nullptr); - if (wrapper_trace && entry.referrer_class_tag < 0) { - jlong holder_tag = entry.referrer_class_tag; - jint holder_count = 0; - jobject *holder_objs = nullptr; - jlong *holder_tags = nullptr; - if (jvmti->GetObjectsWithTags(1, &holder_tag, &holder_count, - &holder_objs, &holder_tags) == - JVMTI_ERROR_NONE && - holder_count > 0) { - char *holder_sig = nullptr; - jvmti->GetClassSignature((jclass)holder_objs[0], &holder_sig, - nullptr); - TEST_LOG("ReferenceChainTracker::walkStaticFieldAnchors wrapper " - "anchor tag=%lld holder_class=%s field_index=%d", - (long long)resolved_tags[i], - holder_sig != nullptr ? holder_sig : "", - (int)entry.referrer_field_index); - if (holder_sig != nullptr) { - jvmti->Deallocate((unsigned char *)holder_sig); - } - } - if (holder_objs != nullptr) { - jvmti->Deallocate((unsigned char *)holder_objs); - } - if (holder_tags != nullptr) { - jvmti->Deallocate((unsigned char *)holder_tags); - } - } - if (sig != nullptr) { - jvmti->Deallocate((unsigned char *)sig); - } - } int remaining = budget - *edges_admitted; if (remaining <= 0) { jni->DeleteLocalRef(objects[i]); @@ -4134,8 +3906,7 @@ void ReferenceChainTracker::walkStaticFieldAnchors( int edges_before = *edges_admitted; descendFromAnchor(jvmti, jni, objects[i], resolved_tags[i], entry.depth, /*anchor_descend_class_tag=*/0, remaining, edges_admitted, - truncated, frontier_cap_hit, safepoint_ticks, - wrapper_trace); + truncated, frontier_cap_hit, safepoint_ticks); TEST_LOG("ReferenceChainTracker::walkStaticFieldAnchors anchor walk " "outcome tag=%lld edges=%d truncated=%d cap_hit=%d", (long long)resolved_tags[i], *edges_admitted - edges_before, @@ -4604,20 +4375,6 @@ void ReferenceChainTracker::runPassManualWalk(jvmtiEnv *jvmti, JNIEnv *jni, &static_field_cycle_complete, safepoint_ticks); expand_phase_edges_admitted += static_field_edges_admitted; *edges_admitted += static_field_edges_admitted; - // TEMP DIAGNOSTIC (see doc/temp/ investigation notes): split out how much - // of this pass's budget/deadline the static-field sweep's current chunk - // alone consumed, and whether that chunk completed / the lap wrapped - - // to distinguish "a chunk never finishes within the per-pass deadline" - // from "chunks finish but rotation/expansion still can't find the - // target". - TEST_LOG_SUMMARY("ReferenceChainTracker::runPassManualWalk static_field_phase " - "edges_admitted=%d truncated=%d frontier_cap_hit=%d " - "cycle_complete=%d sweep_cursor=%d " - "last_resolved_class_count=%d last_static_field_class_count=%d", - static_field_edges_admitted, (int)static_field_truncated, - (int)static_field_frontier_cap_hit, (int)static_field_cycle_complete, - _static_field_sweep_cursor, _last_resolved_class_count, - _last_static_field_class_count); if (static_field_truncated) { *truncated = true; *frontier_cap_hit = static_field_frontier_cap_hit; @@ -4725,35 +4482,10 @@ void ReferenceChainTracker::runPassManualWalk(jvmtiEnv *jvmti, JNIEnv *jni, std::vector static_anchor_tags = collectStaticFieldAnchorsForRotation(STATIC_ANCHOR_ROTATION_BUDGET); std::vector static_anchor_fifo_drained; - int static_anchor_fifo_drained_count = - drainStaticAnchorFifo(STATIC_ANCHOR_FIFO_DRAIN, static_anchor_fifo_drained); + drainStaticAnchorFifo(STATIC_ANCHOR_FIFO_DRAIN, static_anchor_fifo_drained); for (const AtRiskAnchor &at_risk : static_anchor_fifo_drained) { static_anchor_tags.push_back(at_risk.tag); } - // TEMP DIAGNOSTIC (see static_field_phase log above). The fifo fields are - // the round-10 verification channel for B': fifo_pushed_total sizes the - // at-risk population (the design's drain-rate argument was inferred, not - // measured). Round 16: fifo_quota_drops_total is the flood-classes' dropped - // pushes (should climb steadily on the pod while fifo_size stays far - // below the 1024 cap, so the wrapper's pushes land), and self_edge_skips - // is FrontierTable's self-edge guard count (should climb every pass that - // walks a Synchronized* holder - the LEAK_BUFFER wrapper's mutex==this - // edge proving the guard fires on the real object). - TEST_LOG_SUMMARY("ReferenceChainTracker::runPassManualWalk rotation_candidates " - "root_kind_tags=%zu leak_accumulation_tags=%zu stale_expanded_tags=%zu " - "static_anchor_tags=%zu static_anchor_fifo_size=%zu " - "static_anchor_fifo_drained=%d static_anchor_fifo_pushed_total=%llu " - "static_anchor_fifo_quota_drops_total=%llu " - "self_edge_skips=%llu watched_leak_klass_count=%d " - "leak_signatures=%zu leak_parents=%zu", - rotation_tags.size(), leak_accumulation_tags.size(), - stale_expanded_tags.size(), static_anchor_tags.size(), - _static_anchor_fifo.size(), static_anchor_fifo_drained_count, - (unsigned long long)_static_anchor_fifo_pushed, - (unsigned long long)_static_anchor_fifo_quota_drops, - (unsigned long long)_frontier->selfEdgeGuardSkips(), - _watched_leak_klass_count, - _leak_signature_totals.size(), _leak_parent_fanout.size()); if (rotation_tags.empty() && leak_accumulation_tags.empty() && stale_expanded_tags.empty() && static_anchor_tags.empty()) { return; @@ -4839,12 +4571,6 @@ void ReferenceChainTracker::runPassManualWalk(jvmtiEnv *jvmti, JNIEnv *jni, &rotation_frontier_cap_hit, safepoint_ticks); rotation_edges_admitted += queue_tier_edges_admitted; *edges_admitted += rotation_edges_admitted; - // TEMP DIAGNOSTIC (see static_field_phase log above). - TEST_LOG_SUMMARY("ReferenceChainTracker::runPassManualWalk rotation_phase " - "edges_admitted=%d truncated=%d frontier_cap_hit=%d " - "rotation_budget=%d", - rotation_edges_admitted, (int)rotation_truncated, - (int)rotation_frontier_cap_hit, rotation_budget); // OR, not overwrite: the ordinary expand phase above may have already set // these to true (real truncation/cap-hit left in _pending_expand), and a // rotation batch that happens to finish cleanly must not erase that - @@ -5055,16 +4781,6 @@ void ReferenceChainTracker::expandFrontier(jvmtiEnv *jvmti, JNIEnv *jni, break; } - // TEMP DIAGNOSTIC: verify adaptive batch_size is working - TEST_LOG("ReferenceChainTracker::expandFrontier gotw " - "batch_size=%zu resolved=%d edges=%d gotw_ms=%llu ema_call_ms=%llu " - "next_batch=%llu", - batch_size, resolved_count, ctx.edges_admitted, - (unsigned long long)(gotw_elapsed_ns / 1000000ULL), - (unsigned long long)(_gotw_ema_call_ns / 1000000ULL), - (unsigned long long)(_gotw_batch_size != 0 ? _gotw_batch_size - : GOTW_INITIAL_BATCH_SIZE)); - std::unordered_map live; for (jint i = 0; i < resolved_count; i++) { live[resolved_tags[i]] = resolved_objects[i]; @@ -5359,70 +5075,6 @@ void ReferenceChainTracker::admitStaticFieldRoots(jvmtiEnv *jvmti, JNIEnv *jni, } } - // TEMP DIAGNOSTIC (pod round 13): ground-truth probe for the - // LEAK_BUFFER wrapper. Every deduction path (root-attached admit -> - // index -> collector walk; chain-attached -> at-risk FIFO walk; - // already-admitted re-hit -> upgrade or push) should end in a walk - // log with class=...SynchronizedRandomAccessList, yet none is ever - // observed. This probe bypasses the sweep callback entirely: it reads - // ProfileAnalyzer.LEAK_BUFFER directly via JNI and reports the - // wrapper's CURRENT tag and frontier entry shape each time the class - // passes through a chunk. Runs BEFORE the sweep's FollowReferences for - // this chunk, so the first lap shows tag=0 (pre-admission) and later - // laps show the steady-state entry. Remove once the wrapper question - // is answered. - for (jint i = chunk_start; i < chunk_end; i++) { - char *probe_sig = nullptr; - if (jvmti->GetClassSignature(classes[i], &probe_sig, nullptr) != - JVMTI_ERROR_NONE) { - continue; - } - if (probe_sig == nullptr || - strstr(probe_sig, "ProfileAnalyzer") == nullptr) { - if (probe_sig != nullptr) { - jvmti->Deallocate((unsigned char *)probe_sig); - } - continue; - } - TEST_LOG("ReferenceChainTracker::admitStaticFieldRoots LEAK_BUFFER " - "probe: sweeping holder class %s at index %d", - probe_sig, (int)i); - jvmti->Deallocate((unsigned char *)probe_sig); - jfieldID leak_fid = - jni->GetStaticFieldID(classes[i], "LEAK_BUFFER", "Ljava/util/List;"); - if (jniExceptionCheck(jni) || leak_fid == nullptr) { - jni->ExceptionClear(); - TEST_LOG("ReferenceChainTracker::admitStaticFieldRoots LEAK_BUFFER " - "probe: GetStaticFieldID failed"); - continue; - } - jobject probe_wrapper = jni->GetStaticObjectField(classes[i], leak_fid); - if (jniExceptionCheck(jni) || probe_wrapper == nullptr) { - jni->ExceptionClear(); - TEST_LOG("ReferenceChainTracker::admitStaticFieldRoots LEAK_BUFFER " - "probe: static value null/exception"); - continue; - } - jlong probe_tag = 0; - jvmti->GetTag(probe_wrapper, &probe_tag); - FrontierEntry probe_entry{}; - bool probe_found = - probe_tag > 0 ? _frontier->lookup(probe_tag, &probe_entry) : false; - TEST_LOG("ReferenceChainTracker::admitStaticFieldRoots LEAK_BUFFER " - "probe wrapper_tag=%lld frontier_found=%d parent=%lld " - "root_kind=%u state=%u leak_tag=%lld depth=%u " - "referrer_klass=%u", - (long long)probe_tag, (int)probe_found, - (long long)(probe_found ? probe_entry.parent_tag : 0), - (unsigned)(probe_found ? probe_entry.root_kind : 0), - (unsigned)(probe_found ? probe_entry.state : 0), - (long long)(probe_found ? probe_entry.leak_tag : 0), - (unsigned)(probe_found ? probe_entry.depth : 0), - probe_found ? probe_entry.referrer_klass : 0); - jni->DeleteLocalRef(probe_wrapper); - break; // one holder class per chunk is enough - } - // GetLoadedClasses() returned a local ref for every class regardless of // chunk selection - free all of them here, not just the chunk. for (jint i = 0; i < class_count; i++) { @@ -5463,11 +5115,6 @@ void ReferenceChainTracker::admitStaticFieldRoots(jvmtiEnv *jvmti, JNIEnv *jni, // class. 32 covers a typical class's full constant-pool/interface set; // outlier classes are bounded so they cannot blow the chunk's deadline. ctx._class_other_cap = STATIC_FIELD_SWEEP_NON_STATIC_CAP_PER_CLASS; - // TEMP DIAGNOSTIC (see doc/temp/ investigation notes): see PassContext:: - // kind_counts's own comment. - int kind_counts[32]; - memset(kind_counts, 0, sizeof(kind_counts)); - ctx.kind_counts = kind_counts; jvmtiHeapCallbacks callbacks; memset(&callbacks, 0, sizeof(callbacks)); @@ -5477,15 +5124,6 @@ void ReferenceChainTracker::admitStaticFieldRoots(jvmtiEnv *jvmti, JNIEnv *jni, jvmti->FollowReferences(0, nullptr, holder, &callbacks, &ctx); *safepoint_ticks += TSC::ticks() - follow_start_ticks; jni->DeleteLocalRef(holder); - // TEMP DIAGNOSTIC (see doc/temp/ investigation notes): kind indices per - // jvmti.h's jvmtiHeapReferenceKind - 1=CLASS 2=FIELD 3=ARRAY_ELEMENT - // 4=CLASS_LOADER 5=SIGNERS 6=PROTECTION_DOMAIN 7=INTERFACE 8=STATIC_FIELD - // 9=CONSTANT_POOL 10=SUPERCLASS (21-27 are root kinds, not expected here). - TEST_LOG("ReferenceChainTracker::admitStaticFieldRoots kind_counts " - "k1=%d k2=%d k3=%d k4=%d k5=%d k6=%d k7=%d k8=%d k9=%d k10=%d", - kind_counts[1], kind_counts[2], kind_counts[3], kind_counts[4], - kind_counts[5], kind_counts[6], kind_counts[7], kind_counts[8], - kind_counts[9], kind_counts[10]); if (follow_err != JVMTI_ERROR_NONE) { return; } diff --git a/ddprof-lib/src/main/cpp/referenceChains.h b/ddprof-lib/src/main/cpp/referenceChains.h index 8a452d1729..fc3ffd00b6 100644 --- a/ddprof-lib/src/main/cpp/referenceChains.h +++ b/ddprof-lib/src/main/cpp/referenceChains.h @@ -391,19 +391,6 @@ class alignas(alignof(SpinLock)) FrontierTable { int _table_max_cap; FrontierEntry *_table; - // Cumulative improveChain()/reparentToDurableRoot() refusals of a - // parent==tag SELF-EDGE (see improveChain's own guard). A single - // callback-delivered self-edge trips BOTH sibling guards (improveChain - // refuses, then the already-admitted block's else-if offers the same - // self-parent to reparentToDurableRoot), so the count climbs 2 per - // delivered edge - on the hotdog pod, the LEAK_BUFFER wrapper's own - // rotation walk delivers its mutex == this edge every pass, so the - // counter climbing is the verification that the guard fires on the real - // wrapper. Relaxed-atomic: the heap-callback writers and the engine - // thread's per-pass log reader only need a monotonic tally, not - // synchronization. - std::atomic _self_edge_guard_skips{0}; - // Grows _table (doubling) until it holds at least `required_cap` slots or // _table_max_cap is reached. Must be called with _table_lock held // exclusively. Returns false (capacity exhausted) without partially @@ -421,11 +408,6 @@ class alignas(alignof(SpinLock)) FrontierTable { FrontierTable(const FrontierTable &) = delete; FrontierTable &operator=(const FrontierTable &) = delete; - // Cumulative self-edge guard refusals (see _self_edge_guard_skips). - u64 selfEdgeGuardSkips() const { - return _self_edge_guard_skips.load(std::memory_order_relaxed); - } - // Writes (parent_tag, referrer_klass, depth, state) into the slot for // `tag` (index = tag - 1), growing the table if needed. Returns false // without writing anything if `tag` is not positive, or the table is @@ -1532,11 +1514,6 @@ class ReferenceChainTracker { PriorityExpandSet _static_anchor_fifo_set; static constexpr size_t STATIC_ANCHOR_FIFO_CAP = PRIORITY_EXPAND_CAP; - // Cumulative at-risk pushes, for the per-pass TEST_LOG line (round-10 - // verification: sizes the at-risk population the design's drain-rate - // argument was inferred, not measured, from). - u64 _static_anchor_fifo_pushed = 0; - // Round 16 (pod round-15 measurement, ev-leaktag-onpod-round15-results): // the B' repair was DEAD on the pod - the FIFO sat cap-pinned at 1024 // because three classes flooded it (klass 1: 1396 pushes, klass 215: @@ -1555,11 +1532,6 @@ class ReferenceChainTracker { // contents (<= 1024 distinct classes), not by the search lifetime. std::unordered_map _static_anchor_fifo_klass_counts; static constexpr u32 STATIC_ANCHOR_ATRISK_PER_KLASS_CAP = 64; - // Cumulative quota drops (class at cap), for the per-pass TEST_LOG - // line: on the pod this should climb steadily with the flood classes' - // pushes while the wrapper's pushes stop dropping (the round-16 - // verification channel, alongside fifo_size dropping below 1024). - u64 _static_anchor_fifo_quota_drops = 0; // Index of root-attached STATIC_FIELD/JNI_GLOBAL frontier entries, // so collectStaticFieldAnchorsForRotation() iterates O(anchors) instead @@ -2765,8 +2737,7 @@ class ReferenceChainTracker { jlong anchor_tag, u32 anchor_depth, jlong anchor_descend_class_tag, int budget, int *edges_admitted, bool *truncated, - bool *frontier_cap_hit, u64 *safepoint_ticks, - bool diag_trace = false); + bool *frontier_cap_hit, u64 *safepoint_ticks); // Prong 1 of the candidate-scoped reach design (thread-retained taxonomy: // ThreadLocal-held caches and thread-owned collections): per pass, walk diff --git a/ddprof-lib/src/test/cpp/referenceChains_ut.cpp b/ddprof-lib/src/test/cpp/referenceChains_ut.cpp index 7dd6a525a8..b16143620d 100644 --- a/ddprof-lib/src/test/cpp/referenceChains_ut.cpp +++ b/ddprof-lib/src/test/cpp/referenceChains_ut.cpp @@ -185,8 +185,6 @@ class ReferenceChainsTestAccessor { t->_static_anchor_fifo.clear(); t->_static_anchor_fifo_set.clear(); t->_static_anchor_fifo_klass_counts.clear(); - t->_static_anchor_fifo_pushed = 0; - t->_static_anchor_fifo_quota_drops = 0; t->_static_anchor_fresh_queue.clear(); t->_last_pass_gc_finish_epoch = 0; t->_last_pass_ns = 0; @@ -698,21 +696,6 @@ class ReferenceChainsTestAccessor { ->_static_anchor_fifo_set.contains(tag); } - static u64 staticAnchorFifoPushedForTest() { - return ReferenceChainTracker::instance()->_static_anchor_fifo_pushed; - } - - static u64 staticAnchorFifoQuotaDropsForTest() { - return ReferenceChainTracker::instance() - ->_static_anchor_fifo_quota_drops; - } - - static u64 selfEdgeGuardSkipsForTest() { - return ReferenceChainTracker::instance() - ->frontierTable() - ->selfEdgeGuardSkips(); - } - static void walkStaticAnchorFifoForTest(jvmtiEnv *jvmti, JNIEnv *jni, const std::vector &tags, int budget, int *edges_admitted, @@ -6305,7 +6288,6 @@ TEST_F(ReferenceChainsBfsTest, DemotionPushFiresWhenImproveChainEvictsRootAttach EXPECT_EQ(1u, entry.depth); ASSERT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); EXPECT_TRUE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(105)); - EXPECT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest()); // Re-walking the same edge must NOT push twice: improveChain refuses // (new depth 1 is not > current 1), and the set dedupes regardless. @@ -6314,7 +6296,6 @@ TEST_F(ReferenceChainsBfsTest, DemotionPushFiresWhenImproveChainEvictsRootAttach ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, &mock_jni, &edges2); EXPECT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - EXPECT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest()); tracker->stop(); } @@ -6333,12 +6314,21 @@ TEST_F(ReferenceChainsBfsTest, SweepPushFiresOnStaticEdgeOntoChainAttachedHolder int classNode = addNode(); int holderNode = addNode(); + // A child reachable ONLY from the holder: nothing in the scripted + // graph walks holderNode except the FIFO-drained anchor walk, so the + // child's admission after runPass is the end-to-end evidence that the + // sweep pushed the holder, the rotation phase drained it, and + // walkStaticFieldAnchors walked it. + int holderChildNode = addNode(); addClass((void *)&node_tags[classNode], "Lcom/rc/statics/ChainBornHolder;"); script = { - // Only the sweep's static edge: nothing else reaches holderNode, so - // the only possible push is the static-edge-onto-chain-attached site. + // Only the sweep's static edge onto the holder, plus the holder's + // own child edge for the anchor walk to admit: nothing else reaches + // holderNode or holderChildNode, so the only possible push is the + // static-edge-onto-chain-attached site. {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, holderNode, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, holderNode, holderChildNode, -1}, }; // Seed the born-chain-attached shape the eviction leaves: holder already @@ -6347,6 +6337,9 @@ TEST_F(ReferenceChainsBfsTest, SweepPushFiresOnStaticEdgeOntoChainAttachedHolder // refusal and must fall into the B' push instead. FrontierTable *frontier = tracker->frontierTable(); node_tags[holderNode] = 105; + // The child is untagged (0): the anchor walk's admission assigns it a + // fresh frontier tag, observable via node_tags after the pass. + ASSERT_EQ(0, node_tags[holderChildNode]); ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( frontier, 105, 104, 1, FrontierEntryState::FRONTIER, 0)); ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( @@ -6356,17 +6349,27 @@ TEST_F(ReferenceChainsBfsTest, SweepPushFiresOnStaticEdgeOntoChainAttachedHolder bool truncated = true; ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest()) - << "static sweep edge onto the chain-attached holder never fed " - "the at-risk anchor FIFO"; // The rotation phase of the same pass drained the pushed tag into the - // anchor walk (the holder has no scripted subtree - the walk is a no-op, - // which is what makes the drained-empty assertion attributable to the - // drain rather than to a walk). + // anchor walk, and the walk admitted the holder's child - the push + // itself left no residue in the FIFO (drained empty) and never + // re-attributed the holder's entry (re-rooting is the documented + // refusal that motivated the FIFO in the first place). The child's + // admission is observed via tags_ever_assigned rather than node_tags: + // the frontier drained empty and the search COMPLETED in this same + // pass, so releaseSearchTags() has already cleared every live JVMTI + // tag (including the child's and the holder's) by the time runPass + // returns - the frontier table's own records survive that, the tag + // map does not. EXPECT_EQ(0u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + jlong childTag = tags_ever_assigned[holderChildNode]; + ASSERT_GT(childTag, 0) << "the FIFO-drained anchor walk never admitted " + "the holder's child"; + FrontierEntry childEntry{}; + ASSERT_TRUE(frontier->lookup(childTag, &childEntry)); + EXPECT_EQ(105, childEntry.parent_tag); + EXPECT_EQ(2u, childEntry.depth); // The holder's entry is untouched by the push - the B' feed records the - // at-risk shape, it never re-attributes the entry (re-rooting is the - // documented refusal that motivated the FIFO in the first place). + // at-risk shape, it never re-attributes the entry. FrontierEntry entry{}; ASSERT_TRUE(frontier->lookup(105, &entry)); EXPECT_EQ(104, entry.parent_tag); @@ -6602,37 +6605,25 @@ TEST_F(ReferenceChainsBfsTest, // A delivered self-edge trips BOTH sibling guards: improveChain refuses, // and the already-admitted block's else-if then offers the same - // self-parent to reparentToDurableRoot, which refuses and counts too. - // Two refusals per delivered edge is the designed accounting (the pod - // counter will climb 2 per wrapper walk). - u64 skips_before = - ReferenceChainsTestAccessor::selfEdgeGuardSkipsForTest(); + // self-parent to reparentToDurableRoot, which refuses too. int edges = 0; ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, &mock_jni, &edges); // The self-edge was delivered and refused: the entry keeps its - // root-attached shape (the collector's parent_tag == 0 eligibility), - // no demotion push fired, and the guard counted the refusal. + // root-attached shape (the collector's parent_tag == 0 eligibility) + // and no demotion push fired. FrontierEntry entry{}; ASSERT_TRUE(frontier->lookup(105, &entry)); EXPECT_EQ(0, entry.parent_tag); EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD, entry.root_kind); EXPECT_EQ(0u, entry.depth); EXPECT_EQ(0u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - EXPECT_EQ(2u, - ReferenceChainsTestAccessor::selfEdgeGuardSkipsForTest() - - skips_before); // The sibling guards and the direct table calls agree: a self-parent is - // refused by both improvement paths, unconditionally, and every - // refusal is counted (the callback's two refusals above are the first - // and second). + // refused by both improvement paths, unconditionally. EXPECT_FALSE(frontier->improveChain(105, 105, 0, 5, 0, -1, 0, 0)); EXPECT_FALSE(frontier->reparentToDurableRoot(105, 105, 0, -1, 0)); - EXPECT_EQ(4u, - ReferenceChainsTestAccessor::selfEdgeGuardSkipsForTest() - - skips_before); tracker->stop(); } @@ -6657,15 +6648,6 @@ TEST_F(ReferenceChainsBfsTest, AtRiskFifoPerClassQuotaKeepsFloodOut) { const u32 quota = ReferenceChainsTestAccessor::kAtRiskPerKlassCap; ASSERT_EQ(64u, quota); - // The push/quota-drop counters are CUMULATIVE across the tracker's - // lifetime (start() does not reset them - only the full - // search-restart reset does), so every assertion below is a DELTA - // from this baseline: order-immune to whatever the tests that ran - // before this one left behind (the round-15 singleton-state lesson). - const u64 pushed0 = - ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest(); - const u64 drops0 = - ReferenceChainsTestAccessor::staticAnchorFifoQuotaDropsForTest(); // The flood: only the first `quota` pushes of one class land. for (int i = 0; i < 70; i++) { @@ -6673,11 +6655,12 @@ TEST_F(ReferenceChainsBfsTest, AtRiskFifoPerClassQuotaKeepsFloodOut) { 1733); } EXPECT_EQ(quota, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - EXPECT_EQ(quota, ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest() - - pushed0); - EXPECT_EQ(70u - quota, - ReferenceChainsTestAccessor::staticAnchorFifoQuotaDropsForTest() - - drops0); + // The flood's first `quota` tags hold their slots and the excess is + // dropped at the quota check - absent from the FIFO, not queued. + EXPECT_TRUE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest( + 1000 + (int)quota - 1)); + EXPECT_FALSE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest( + 1000 + (int)quota)); // The wrapper's push (a different class) lands despite the flood - // exactly the push the pod dropped. @@ -6687,14 +6670,10 @@ TEST_F(ReferenceChainsBfsTest, AtRiskFifoPerClassQuotaKeepsFloodOut) { EXPECT_TRUE( ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(2000)); - // Tag dedupe is unchanged: the same tag never enters twice (and the - // duplicate is not counted as a quota drop, nor as a push). + // Tag dedupe is unchanged: the same tag never enters twice. ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(2000, 28366); EXPECT_EQ(quota + 1, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - EXPECT_EQ(quota + 1, - ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest() - - pushed0); // Full drain: order preserved (flood first, newcomer last), occupancy // erased with the entries - the next flood can land again (it never @@ -6714,9 +6693,6 @@ TEST_F(ReferenceChainsBfsTest, AtRiskFifoPerClassQuotaKeepsFloodOut) { 1733); } EXPECT_EQ(quota, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - EXPECT_EQ(2u * quota + 1, - ReferenceChainsTestAccessor::staticAnchorFifoPushedForTest() - - pushed0); // Partial drain at the real per-pass rate (16/pass): the flood's // occupancy is 64 - 16 = 48 after the drain, so its next push lands @@ -6764,9 +6740,9 @@ TEST_F(ReferenceChainsBfsTest, AtRiskFifoPerClassQuotaKeepsFloodOut) { } EXPECT_EQ(1024u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - EXPECT_EQ(0u, - ReferenceChainsTestAccessor::staticAnchorFifoQuotaDropsForTest() - - drops0 - (70u - quota) * 2); + // The 16 saturated classes sit exactly at their per-class quota (no + // quota drop is even possible at exactly `quota` pushes), so the + // newcomer's absence below is the CAP's doing, not the quota's. ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(20000, 28366); EXPECT_EQ(1024u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); From 45c17d35645a72f28347619d6ffecc41ffd3ec52 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Thu, 17 Sep 2026 14:40:52 +0200 Subject: [PATCH 13/18] Drop temporary-diagnostics note, document JMH tuning limits --- .../LiveHeapReferenceChains-Implementation.md | 39 ++++++++++++++----- 1 file changed, 30 insertions(+), 9 deletions(-) diff --git a/doc/architecture/LiveHeapReferenceChains-Implementation.md b/doc/architecture/LiveHeapReferenceChains-Implementation.md index f0d256217f..b016cd10d3 100644 --- a/doc/architecture/LiveHeapReferenceChains-Implementation.md +++ b/doc/architecture/LiveHeapReferenceChains-Implementation.md @@ -378,12 +378,33 @@ blocks on JFR I/O, and JFR writes never trigger a walk. - `ttl <= 0` disables the wall-clock TTL cutoff. - Unset values are auto-tuned (see §7). -## 11. Temporary diagnostics - -Some `TEST_LOG_SUMMARY` output and tallies in the static-field sweep and the -anchor walk (chunk-consumption splits, per-`reference_kind` callback tallies, -walked-anchor naming) are *temporary diagnostic* instrumentation from the -live pod debugging rounds (`.investigations/missing-refchains-on-hotdog/`, -16 rounds) that shaped this design. They are pending removal once the -on-pod investigation notes are fully distilled. Production behavior does not -depend on them. +## 11. Why the tuning pass is not a JMH benchmark + +The defaults above are placeholders pending a measurement pass. That pass +cannot be a JMH benchmark, for three reasons: + +1. **The cost is not per-Java-operation.** JMH measures the throughput and + latency of a benchmark method. This subsystem's cost is (a) STW + safepoint pauses from `VM_HeapWalkOperation` — global stalls + attributable to no benchmark method — and (b) CPU on a dedicated native + background thread. Both reach a Java workload only as indirect + throughput degradation mixed with GC noise; JMH can neither observe + nor attribute them directly. +2. **Activation is leak-signal-gated.** No pass runs until the population + rings fill (~10 GC epochs), trend hysteresis clears (5 consecutive + qualifying epochs), and the search gate opens. A seconds-scale JMH + iteration measures the feature idle. Exercising the chase needs minutes + of continuous leaking allocation; run-to-run variance is then dominated + by GC cadence and by *when* hysteresis cleared — exactly the steady-state + assumption JMH's fork/iteration statistics make. +3. **The knobs control quantities JMH cannot see.** `pausetarget`, + `budget`, cadence, backoff, and the pain budgets regulate per-pass + safepoint duration, pass-cost EMA, and refill rates — all directly + observable as `jdk.ExecuteVMOperation[HeapWalkOperation]` JFR durations + and the tracker's own pass telemetry, which is what the shipped `utils/` + repro/sweep/report tooling consumes. + +A coarse JMH A/B (leaking workload, `referencechains` off vs on) remains +possible with the repo's existing `ddprof-stresstest` JMH setup, and would +serve as an end-to-end throughput regression guard. It cannot tune these +defaults. Tuning them requires the JFR-based measurement pass above. From d0cb321ae677eae9fc789334f426072c55e97cf1 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Thu, 17 Sep 2026 16:57:04 +0200 Subject: [PATCH 14/18] Fix thread-object registry wiring and global-ref lifetime - registerThreadObject() was never called from Profiler::onThreadStart (only the one-time registerExistingThreads() snapshot), so threads born mid-recording never entered the tid->Thread registry and their ThreadLocalMap subgraphs were unreachable by walkCandidateThreadLocals(). - unregisterThreadObject()/register-replace DeleteGlobalRef'd under the registry lock while walkCandidateThreadLocals() can still hold a copied ref as a FollowReferences anchor - JNI use-after-free. Deletion is now deferred to _thread_refs_pending_delete and drained by releaseEndedThreadRefs() on the BFS thread (start of each pass) and from Profiler::stop() after stopThread() joined it. - tagAsRootForTest(): record the directly-tagged root as a discovered instance when a candidate slot watches its klass - the leak-tag redesign left a representative without a sampler-tracked instance no discovery channel; the seam test's tag now runs after the first poll. - getLeakTagInfo(): drop the bogus free-list membership check (a LIFO stack of indices is not an index-bounded region). - gtest: ThreadRefUnregisterDefersGlobalRefDeletion; failure-path logging in tagAsRootForTest. --- ddprof-lib/src/main/cpp/livenessTracker.cpp | 10 ++-- ddprof-lib/src/main/cpp/profiler.cpp | 13 +++++ ddprof-lib/src/main/cpp/referenceChains.cpp | 56 ++++++++++++++++++- ddprof-lib/src/main/cpp/referenceChains.h | 18 ++++++ .../src/test/cpp/referenceChains_ut.cpp | 39 +++++++++++++ .../ReferenceChainTestSeamsTest.java | 23 +++++++- 6 files changed, 148 insertions(+), 11 deletions(-) diff --git a/ddprof-lib/src/main/cpp/livenessTracker.cpp b/ddprof-lib/src/main/cpp/livenessTracker.cpp index 05dd686347..36cd9c8c71 100644 --- a/ddprof-lib/src/main/cpp/livenessTracker.cpp +++ b/ddprof-lib/src/main/cpp/livenessTracker.cpp @@ -360,12 +360,10 @@ bool LivenessTracker::getLeakTagInfo(jlong tag, u64 *out_call_trace_id, return false; } int idx = (int)(tag - LEAK_TAG_BASE); - if (idx >= _leak_tag_free_count && - _leak_tag_info[idx].call_trace_id == 0) { - return false; // tag is in free list - } - // Check if tag is still in use (not in free list) - // Simple check: if call_trace_id is 0 and tid is 0, it's been released + // releaseLeakTag() zeroes both fields, so a zero/zero slot means the tag + // was released (or never acquired) - any other state is in use. (A slot + // index comparison against _leak_tag_free_count proves nothing here: the + // free list is a LIFO stack of indices, not an index-bounded region.) if (_leak_tag_info[idx].call_trace_id == 0 && _leak_tag_info[idx].tid == 0) { return false; } diff --git a/ddprof-lib/src/main/cpp/profiler.cpp b/ddprof-lib/src/main/cpp/profiler.cpp index 58dcc20c6d..5609eda4ff 100644 --- a/ddprof-lib/src/main/cpp/profiler.cpp +++ b/ddprof-lib/src/main/cpp/profiler.cpp @@ -98,6 +98,14 @@ void Profiler::onThreadStart(jvmtiEnv *jvmti, JNIEnv *jni, jthread thread) { updateThreadName(jvmti, jni, thread, true); } + // Registers the tid -> Thread-object global ref that the reference-chain + // engine's walkCandidateThreadLocals() descends from for + // candidate-scoped ThreadLocalMap reach. No-op while reference chains are + // disabled (checked inside the tracker); jni/thread may be null on the + // internal pre-existing-threads call from start(), which the tracker + // also refuses. + ReferenceChainTracker::instance()->registerThreadObject(jni, tid, thread); + _cpu_engine->registerThread(tid); _wall_engine->registerThread(tid); } @@ -1907,6 +1915,11 @@ Error Profiler::stop() { if (ReferenceChainTracker::instance()->enabled()) { ReferenceChainTracker::instance()->stopThread(); ReferenceChainTracker::instance()->stop(); + // Drains the global refs of threads that ended during this recording + // (referenceChains.h, _thread_refs_pending_delete). Safe here: the BFS + // thread was joined by stopThread() above, so no walk phase can still + // hold a copied Thread-object ref. + ReferenceChainTracker::instance()->releaseEndedThreadRefs(VM::jni()); } // Stop the refresher BEFORE socket unpatch: the refresher calls // install_socket_hooks() which re-reads _socket_active before acquiring the diff --git a/ddprof-lib/src/main/cpp/referenceChains.cpp b/ddprof-lib/src/main/cpp/referenceChains.cpp index 9f408952c8..8d1b236d97 100644 --- a/ddprof-lib/src/main/cpp/referenceChains.cpp +++ b/ddprof-lib/src/main/cpp/referenceChains.cpp @@ -1559,6 +1559,9 @@ jlong ReferenceChainTracker::tagAsRootForTest(jvmtiEnv *jvmti, JNIEnv *jni, jobject obj) { if (_frontier == nullptr || jvmti == nullptr || jni == nullptr || obj == nullptr) { + TEST_LOG_SUMMARY("ReferenceChainTracker::tagAsRootForTest refused: " + "frontier=%p jvmti=%p jni=%p obj=%p", + (void *)_frontier, (void *)jvmti, (void *)jni, (void *)obj); return 0; } // Resolves the klass_id via the same GetClassSignature + @@ -1599,12 +1602,33 @@ jlong ReferenceChainTracker::tagAsRootForTest(jvmtiEnv *jvmti, JNIEnv *jni, // to reach/select it on its own. jlong tag = tagObject(jvmti, obj); if (tag == 0) { + TEST_LOG_SUMMARY("ReferenceChainTracker::tagAsRootForTest refused: " + "tagObject (SetTag) failed"); return 0; } if (!_frontier->insert(tag, 0, klass_id, 0)) { + TEST_LOG_SUMMARY("ReferenceChainTracker::tagAsRootForTest refused: " + "frontier insert failed tag=%lld klass_id=%u", + (long long)tag, klass_id); clearTag(jvmti, obj); return 0; } + // Discovery recording for this seam's decoupling contract. The leak-tag + // redesign ("no marker tags - using leak tags now", pollWatchedTargets()) + // left a representative without an ObjectSampler-tracked instance with no + // discovery channel at all: recordDiscoveredInstance() only fires from the + // leak-tag interception branch of heapReferenceCallback(), and + // tagLeakInstances() can only tag instances the sampler tracked. This + // seam's purpose is precisely to be decoupled from the sampler, so it + // records the directly-tagged root itself; pollWatchedTargets() then + // reconstructs the chain from the real frontier via + // buildDiscoveredInstanceChains()/buildChainEvent(). No-op while no + // candidate slot watches this klass yet (the caller must have driven at + // least one pollReferenceChainTargets0() first - see + // ReferenceChainTestSeamsTest's ordering comment). + if (_candidate_count > 0) { + recordDiscoveredInstance(klass_id, tag, false); + } return tag; } @@ -4094,7 +4118,9 @@ void ReferenceChainTracker::registerThreadObject(JNIEnv *jni, int tid, MutexLocker ml(_thread_objects_lock); auto it = _thread_objects.find(tid); if (it != _thread_objects.end()) { - jni->DeleteGlobalRef(it->second); + // Same deferred-deletion rule as unregisterThreadObject(): a walk may + // still hold a copy of the replaced ref. + _thread_refs_pending_delete.push_back(it->second); } _thread_objects[tid] = ref; } @@ -4106,11 +4132,31 @@ void ReferenceChainTracker::unregisterThreadObject(JNIEnv *jni, int tid) { MutexLocker ml(_thread_objects_lock); auto it = _thread_objects.find(tid); if (it != _thread_objects.end()) { - jni->DeleteGlobalRef(it->second); + // NOT DeleteGlobalRef() here: walkCandidateThreadLocals() may have + // already copied this jobject out of the map (lock released) and still + // be using it as a FollowReferences anchor - deleting a global ref + // invalidates it for every other JNI call (JNI spec), so deletion is + // deferred to releaseEndedThreadRefs() on the BFS thread (see + // _thread_refs_pending_delete's comment). + _thread_refs_pending_delete.push_back(it->second); _thread_objects.erase(it); } } +void ReferenceChainTracker::releaseEndedThreadRefs(JNIEnv *jni) { + if (jni == nullptr) { + return; + } + std::vector pending; + { + MutexLocker ml(_thread_objects_lock); + pending.swap(_thread_refs_pending_delete); + } + for (size_t i = 0; i < pending.size(); i++) { + jni->DeleteGlobalRef(pending[i]); + } +} + // --------------------------------------------------------------------------- // Manual walk driver - IterateOverReachableObjects root/stack-ref enumeration // plus expandFrontier()'s batched array-holder FollowReferences hop expansion. @@ -4219,6 +4265,12 @@ void ReferenceChainTracker::runPassManualWalk(jvmtiEnv *jvmti, JNIEnv *jni, "Heap-category calls and must not be made from " "GarbageCollectionStart/Finish"); + // Safe point to delete the global refs of threads that ended since the + // last drain: this runs on the BFS thread before any walk phase, and refs + // erased from _thread_objects (unregisterThreadObject()) can no longer be + // copied out by walkCandidateThreadLocals(), so no walk holds them. + releaseEndedThreadRefs(jni); + *safepoint_ticks = 0; // Shared wall-clock ceiling for this whole call's static-field sweep, diff --git a/ddprof-lib/src/main/cpp/referenceChains.h b/ddprof-lib/src/main/cpp/referenceChains.h index fc3ffd00b6..08be758a88 100644 --- a/ddprof-lib/src/main/cpp/referenceChains.h +++ b/ddprof-lib/src/main/cpp/referenceChains.h @@ -1006,6 +1006,16 @@ class ReferenceChainTracker { // outside the engine serialization the walk phases run under. Mutex _thread_objects_lock; std::unordered_map _thread_objects; + // Global refs of ended threads awaiting deletion. unregisterThreadObject() + // must NOT DeleteGlobalRef() directly: walkCandidateThreadLocals() copies + // the jobject out of _thread_objects under _thread_objects_lock, releases + // the lock, and can still be using it as a FollowReferences anchor when a + // concurrent ThreadEnd erases the entry - deleting there would be JNI + // use-after-free (once deleted, a global ref is invalid for every other + // JNI call). The erasing thread only enqueues; releaseEndedThreadRefs() + // drains the list on the BFS thread, at points where no walk phase holds + // a copied ref. + std::vector _thread_refs_pending_delete; // Auto-marked instances: when the BFS walk discovers ANY object whose // class matches a watched leak class (not just the pre-tagged @@ -2920,6 +2930,14 @@ class ReferenceChainTracker { void registerThreadObject(JNIEnv *jni, int tid, jthread thread); void unregisterThreadObject(JNIEnv *jni, int tid); + // Delete the global refs unregisterThreadObject() queued in + // _thread_refs_pending_delete (see that member's comment for why the + // deletion is deferred). Called only from points where no walk phase + // holds a copied Thread-object ref: the BFS thread at the start of each + // pass, and Profiler::stop() after stopThread() has joined the BFS + // thread. A jni of null (current thread not attached) is a no-op. + void releaseEndedThreadRefs(JNIEnv *jni); + // One-time sweep over the JVM's CURRENTLY LIVE threads at recording start, // registering each into the same tid -> Thread-object registry via // JVMThread::nativeThreadId(). Profiler::onThreadStart() cannot cover this diff --git a/ddprof-lib/src/test/cpp/referenceChains_ut.cpp b/ddprof-lib/src/test/cpp/referenceChains_ut.cpp index b16143620d..3294bd4d29 100644 --- a/ddprof-lib/src/test/cpp/referenceChains_ut.cpp +++ b/ddprof-lib/src/test/cpp/referenceChains_ut.cpp @@ -1493,6 +1493,9 @@ class ReferenceChainsBfsTest : public ::testing::Test { // (walkCandidateThreadLocals()'s fresh-anchor admission path). void *thread_class = nullptr; + // DeleteGlobalRef call count (see mock_DeleteGlobalRef). + int global_refs_deleted_ = 0; + jvmtiEnv *orig_jvmti = nullptr; static ReferenceChainsBfsTest *active_fixture; @@ -1534,6 +1537,7 @@ class ReferenceChainsBfsTest : public ::testing::Test { jni_tbl.GetObjectClass = &mock_GetObjectClass; jni_tbl.GetSuperclass = &mock_JniGetSuperclass; jni_tbl.NewGlobalRef = &mock_NewGlobalRef; + jni_tbl.DeleteGlobalRef = &mock_DeleteGlobalRef; jni_tbl.EnsureLocalCapacity = &mock_EnsureLocalCapacity; jni_tbl.NewObjectArray = &mock_NewObjectArray; jni_tbl.SetObjectArrayElement = &mock_SetObjectArrayElement; @@ -1785,6 +1789,12 @@ class ReferenceChainsBfsTest : public ::testing::Test { return obj; } + // Counts DeleteGlobalRef calls - the deferred thread-ref teardown test + // (ThreadRefUnregisterDefersGlobalRefDeletion) asserts on the count. + static void JNICALL mock_DeleteGlobalRef(JNIEnv *, jobject) { + active_fixture->global_refs_deleted_++; + } + static jint JNICALL mock_EnsureLocalCapacity(JNIEnv *, jint) { return JNI_OK; } @@ -5845,6 +5855,35 @@ TEST_F(ReferenceChainsBfsTest, ThreadWalkDescendsOnlyThreadLocalMapAndIntercepts tracker->stop(); } +// unregisterThreadObject() must defer the global-ref deletion to +// releaseEndedThreadRefs(): walkCandidateThreadLocals() copies the jobject +// out of _thread_objects under _thread_objects_lock, releases the lock, and +// can still be using it as a FollowReferences anchor when a concurrent +// ThreadEnd erases the entry - deleting there would be JNI use-after-free +// (see _thread_refs_pending_delete's comment). +TEST_F(ReferenceChainsBfsTest, ThreadRefUnregisterDefersGlobalRefDeletion) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int threadNode = addNode(); + tracker->registerThreadObject( + &mock_jni, 555, reinterpret_cast(&node_tags[threadNode])); + tracker->unregisterThreadObject(&mock_jni, 555); + // The erasing side only enqueues - no DeleteGlobalRef yet. + EXPECT_EQ(0, global_refs_deleted_); + + // The drain deletes exactly the queued ref, and draining an empty list + // is a no-op. + tracker->releaseEndedThreadRefs(&mock_jni); + EXPECT_EQ(1, global_refs_deleted_); + tracker->releaseEndedThreadRefs(&mock_jni); + EXPECT_EQ(1, global_refs_deleted_); + + tracker->stop(); +} + // Candidate-scoped reach, prong 2 (collectStaticFieldAnchorsForRotation()/ // walkStaticFieldAnchors()): the collector selects exactly the root-attached // static-holder entries with a wrapping cursor, and the walk reaches a leak diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainTestSeamsTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainTestSeamsTest.java index c2109d8d09..0d33e8fba2 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainTestSeamsTest.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ReferenceChainTestSeamsTest.java @@ -156,9 +156,6 @@ public void shouldReconstructChainForDirectlyTaggedRoot() { // calls, so no stack root is ever enumerated. chainTargetHolder = new ChainTarget(); - long tag = JavaProfiler.tagAsReferenceChainRoot0(chainTargetHolder); - assertTrue(tag > 0, "Expected tagAsReferenceChainRoot0 to assign a valid frontier tag"); - // Representative BEFORE seeding: with the aliasing seam, setting the representative is // what registers the synthetic->real id alias, so the seeding below lands re-keyed under // the target's real class id (seeding first still works - the synthetic entry gets @@ -179,6 +176,26 @@ public void shouldReconstructChainForDirectlyTaggedRoot() { CHAIN_TEST_KLASS_ID, JavaProfiler.getTid(), epoch * 3, epoch); } + // One poll BEFORE the direct tag: the first pollReferenceChainTargets0() + // admits the seeded klass as a watched candidate (canary slot), and the + // seam's discovery recording (tagAsRootForTest's recordDiscoveredInstance) + // only lands when a candidate slot already watches the klass - the + // leak-tag redesign removed the marker-tag path the original pre-poll tag + // used to ride. + JavaProfiler.pollReferenceChainTargets0(); + + // The target is referenced ONLY through the static holder - never bound to a + // live local of this frame. A local `target` variable would make the object a + // STACK_LOCAL GC root for the whole test body, and the noise gate (rightly) + // suppresses depth-0 chains rooted in a transient stack slot: the durable + // roots this fixture wants (the static holder, plus the JNI-global-weak-ref + // root LivenessTracker's representative itself creates) then lose the + // durability race only if a stack-local root exists at all. Accessing the + // object only via the static field leaves its frame slots empty between + // calls, so no stack root is ever enumerated. + long tag = JavaProfiler.tagAsReferenceChainRoot0(chainTargetHolder); + assertTrue(tag > 0, "Expected tagAsReferenceChainRoot0 to assign a valid frontier tag"); + // Drive pass+poll cycles until the chain event lands. One pass is NOT enough: // admitStaticFieldRoots() sweeps loaded classes in chunks (a few hundred per pass, // bounded by the budget) and the first pass truncates before reaching this test's From 0a35700ff9de4fcdeab42ed3c326967ed5557c40 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Thu, 17 Sep 2026 16:57:04 +0200 Subject: [PATCH 15/18] Fix stale docs; add late-thread thread-local leak e2e - arguments.h: 0 first-pass budget auto-scales from budget x AUTO_FIRST_PASS_BUDGET_MULTIPLIER, it does not fall back to the plain per-pass budget. - flightRecorder.h: recordReferenceChain is called from Profiler::dump()'s drain loop, not continuously from pollWatchedTargets(). - collection-summary.md: a terminal search state is not final - shouldRunPass() restarts from it. - ThreadLocalLeakScenario gains a lateThread mode (thread created after recording start) with a new threadlocal-leak-late launcher mode and shouldCorrelateLeakOnThreadCreatedAfterRecordingStart - the regression shape for the onThreadStart registration wiring. --- ddprof-lib/src/main/cpp/arguments.h | 7 +- ddprof-lib/src/main/cpp/flightRecorder.h | 14 ++-- .../datadoghq/profiler/ExternalLauncher.java | 17 +++++ .../ThreadLocalLeakReferenceChainTest.java | 66 +++++++++++++++++++ .../ThreadLocalLeakScenario.java | 50 +++++++++++++- doc/reference-chains-collection-summary.md | 2 +- 6 files changed, 143 insertions(+), 13 deletions(-) diff --git a/ddprof-lib/src/main/cpp/arguments.h b/ddprof-lib/src/main/cpp/arguments.h index 70a6ad97ba..a2044fa580 100644 --- a/ddprof-lib/src/main/cpp/arguments.h +++ b/ddprof-lib/src/main/cpp/arguments.h @@ -112,8 +112,11 @@ const int DEFAULT_REFERENCE_CHAINS_PAIN_BUDGET_PERCENT = 1; // DEFAULT_REFERENCE_CHAINS_BUDGET, which bounds every pass including the many // cheap, per-node expansion passes that follow, this only ever spends once // per search, so a much larger one-time ceiling is affordable. 0 (the -// default) means "no override - use the same budget as every other pass", -// preserving prior behavior for anyone not setting this explicitly. +// default) means "no override - auto-scale from the per-pass budget instead +// of falling back to it plainly": ReferenceChainTracker::start()'s tuning +// pass scales it from _budget (AUTO_FIRST_PASS_BUDGET_MULTIPLIER, capped at +// MAX_REFERENCE_CHAINS_FIRST_PASS_BUDGET) unless this knob was set +// explicitly. const int DEFAULT_REFERENCE_CHAINS_FIRST_PASS_BUDGET = 0; // Upper clamp for an explicit firstpassbudget override: like painbudget just // above, firstpassbudget was previously only floored at 0 with no ceiling. diff --git a/ddprof-lib/src/main/cpp/flightRecorder.h b/ddprof-lib/src/main/cpp/flightRecorder.h index 1a4936e3cd..b3a28c7550 100644 --- a/ddprof-lib/src/main/cpp/flightRecorder.h +++ b/ddprof-lib/src/main/cpp/flightRecorder.h @@ -568,13 +568,13 @@ class FlightRecorder { // Mirrors recordReferenceChainAbandoned() above exactly, for // ReferenceChainEvent instead. Called from Profiler::writeReferenceChain() - // (profiler.cpp), itself called from - // ReferenceChainTracker::pollWatchedTargets() (referenceChains.cpp) for - // each chain event discovered this poll cycle - unlike - // recordReferenceChainAbandoned() (only reached from dump()), chain events - // are produced continuously as candidates are discovered, not only at - // dump time, so this needs its own call site rather than piggybacking on - // dump()'s flush-on-dump pattern. + // (profiler.cpp), itself called from Profiler::dump()'s drain loop over + // the ReferenceChainTracker::drainPendingChainEvents() snapshot: the BFS + // scheduling thread only caches resolved chains in _resolved_chains + // (referenceChains.h) and each dump re-emits the cache, so chain events + // are written on dump()'s own thread, not from the tracker thread, and + // unlike recordReferenceChainAbandoned() (unbounded retry budget per + // event) the batch shares one deadline (writeReferenceChain()'s comment). void recordReferenceChain(int lock_index, ReferenceChainEvent *event); }; diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/ExternalLauncher.java b/ddprof-test/src/test/java/com/datadoghq/profiler/ExternalLauncher.java index a08674d004..dd18a91a58 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/ExternalLauncher.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/ExternalLauncher.java @@ -293,6 +293,23 @@ public static void main(String[] args) throws Exception { ThreadLocalLeakScenario.run(instance, commands, Paths.get(scratchPath)); System.out.flush(); System.exit(0); + } else if (args[0].equals("threadlocal-leak-late")) { + // Same packing as threadlocal-leak, but the leak thread is + // started only AFTER the recording began - the regression + // shape for Profiler::onThreadStart's registerThreadObject() + // wiring (ThreadLocalLeakScenario.run()'s lateThread comment). + String packed = args.length == 2 ? args[1] : ""; + int sep = packed.indexOf("|||"); + if (sep < 0) { + throw new IllegalArgumentException( + "threadlocal-leak-late requires \"|||\", got: " + packed); + } + String commands = packed.substring(0, sep); + String scratchPath = packed.substring(sep + "|||".length()); + JavaProfiler instance = JavaProfiler.getInstance(); + ThreadLocalLeakScenario.run(instance, commands, Paths.get(scratchPath), true); + System.out.flush(); + System.exit(0); } } finally { System.out.println("[ready]"); diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakReferenceChainTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakReferenceChainTest.java index 1bf3bf58cd..f72d4fa1ec 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakReferenceChainTest.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakReferenceChainTest.java @@ -111,6 +111,72 @@ void shouldCorrelateThreadLocalHeldLeakChain() throws Exception { } } + /** + * Same contract as {@link #shouldCorrelateThreadLocalHeldLeakChain()}, but + * the leak thread is started only AFTER the recording began. Regression + * test for the {@code Profiler::onThreadStart} → + * {@code registerThreadObject()} wiring: a mid-recording thread is + * invisible to {@code registerExistingThreads()}'s one-time snapshot, so + * only the onThreadStart registration lets + * {@code walkCandidateThreadLocals()} reach its ThreadLocalMap-held leak. + * With the wiring broken this child reports + * {@code ThreadLocalLeakScenario.NOT_FOUND_MARKER} and this test fails. + */ + @Test + void shouldCorrelateLeakOnThreadCreatedAfterRecordingStart() throws Exception { + assumeFalse(Platform.isJavaVersion(8)); + assumeFalse(Platform.isJ9()); + assumeFalse(Platform.isZing()); + + Path scratchDumpPath = Files.createTempFile("referencechains-tl-late-", ".jfr"); + Files.deleteIfExists(scratchDumpPath); + Path continuousJfrPath = Files.createTempFile("referencechains-tl-late-continuous-", ".jfr"); + Deque testLogTail = new ArrayDeque<>(); + try { + String startCommand = "start,memory=64:l:1.0,generations=true," + + "referencechains=true:hops=64:budget=200000:ttl=120000:framecap=2000000:" + + "pausetarget=60000" + + ",jfr,file=" + continuousJfrPath.toAbsolutePath(); + String packedCommand = startCommand + "|||" + scratchDumpPath.toAbsolutePath(); + + List jvmArgs = Collections.singletonList( + "-Dddprof_test.config=" + System.getProperty("ddprof_test.config")); + + AtomicReference resultLine = new AtomicReference<>(); + LaunchResult result = launch("threadlocal-leak-late", jvmArgs, packedCommand, + Collections.emptyMap(), + 150, + line -> { + if (line.startsWith(ThreadLocalLeakScenario.FOUND_MARKER) + || line.equals(ThreadLocalLeakScenario.NOT_FOUND_MARKER) + || line.startsWith(ThreadLocalLeakScenario.TAG_OUT_OF_POOL_MARKER) + || line.startsWith(ThreadLocalLeakScenario.NO_LIVE_OBJECT_MARKER)) { + resultLine.set(line); + } + if (line.startsWith("[TEST::INFO]")) { + testLogTail.addLast(line); + while (testLogTail.size() > 2000) { + testLogTail.removeFirst(); + } + } + return LineConsumerResult.CONTINUE; + }, + null); + + assertTrue(result.inTime, "Child process did not exit within the wait timeout"); + assertEquals(0, result.exitCode, "Child process exited with a non-zero code"); + assertNotNull(resultLine.get(), "Child process never printed a recognizable result " + + "marker on stdout" + diagnostics(testLogTail)); + String line = resultLine.get(); + assertTrue(line.startsWith(ThreadLocalLeakScenario.FOUND_MARKER), + "Thread-local leak held by a thread created after recording start was not " + + "correlated: " + line + diagnostics(testLogTail)); + } finally { + Files.deleteIfExists(scratchDumpPath); + Files.deleteIfExists(continuousJfrPath); + } + } + /** * A filtered summary of the child's TEST_LOG stream for failure messages - * same first-things-to-check selection as diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakScenario.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakScenario.java index 45f61e4211..5e92682af3 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakScenario.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/ThreadLocalLeakScenario.java @@ -106,6 +106,22 @@ byte[] take() { seedChunk = null; return chunk; } + + /** + * Bounded wait for the leak thread's first publish. Only the + * late-thread mode needs it: there the thread is started after the + * profiler (and thus after {@code profiler.execute()} returned), so the + * main thread cannot rely on the start-command's JNI work having + * scheduled the leak thread far enough to publish. + */ + boolean awaitPublication(long timeoutMs) throws InterruptedException { + long deadline = System.nanoTime() + + TimeUnit.MILLISECONDS.toNanos(timeoutMs); + while (seedChunk == null && System.nanoTime() < deadline) { + Thread.sleep(10); + } + return seedChunk != null; + } } /** Printed to stdout, followed by the correlated targetTag, on full success. */ @@ -126,8 +142,24 @@ private static final class ScanResult { int liveObjectMatchesForChainTag; } - public static void run(JavaProfiler profiler, String startCommand, Path scratchDumpPath) - throws Exception { + public static void run(JavaProfiler profiler, String startCommand, + Path scratchDumpPath) throws Exception { + run(profiler, startCommand, scratchDumpPath, false); + } + + /** + * @param lateThread when true, the leak thread is started only AFTER the + * recording began. This is the regression shape for + * {@code Profiler::onThreadStart}'s {@code registerThreadObject()} + * wiring (referenceChains.h): a thread born mid-recording is invisible + * to {@code registerExistingThreads()}'s one-time snapshot, so with + * the wiring broken the (klass, tid) qualification is seeded but + * {@code walkCandidateThreadLocals()} finds no Thread object for the + * tid and the ThreadLocalMap-held chain is never reached (the child + * prints {@link #NOT_FOUND_MARKER}). + */ + public static void run(JavaProfiler profiler, String startCommand, + Path scratchDumpPath, boolean lateThread) throws Exception { // Build the whole fixture before the profiler starts - see the class // comment and LeakingCacheScenario's seed-before-start rationale. for (int row = 0; row < FILLER.length; row++) { @@ -162,12 +194,24 @@ public static void run(JavaProfiler profiler, String startCommand, Path scratchD } }, "threadlocal-leak"); leakThread.setDaemon(true); - leakThread.start(); + if (!lateThread) { + leakThread.start(); + } if (startCommand != null && !startCommand.isEmpty()) { profiler.execute(startCommand); } + if (lateThread) { + leakThread.start(); + // The debug seeding below consumes the handoff immediately; the + // thread is only just started, so wait for its first publish. + if (!handoff.awaitPublication(30000)) { + System.out.println(NOT_FOUND_MARKER); + return; + } + } + boolean debugBuild = "debug".equals(System.getProperty("ddprof_test.config")); int leakTid = 0; if (debugBuild) { diff --git a/doc/reference-chains-collection-summary.md b/doc/reference-chains-collection-summary.md index 645c9c8285..0354a8cbe7 100644 --- a/doc/reference-chains-collection-summary.md +++ b/doc/reference-chains-collection-summary.md @@ -42,7 +42,7 @@ Once a klass is nominated, a **persistent background BFS thread** reconstructs a - **Every subsequent pass**: calls `expandFrontier()`, which resumes from a **persisted frontier** (the previous pass's boundary tags) instead of re-walking from roots. This is the resumability mechanism: each pass advances the frontier outward by one bounded increment and stops. - Each pass is capped by an **edge-admission budget** (`effectiveBudget`, e.g. `edges_admitted` capped at a configured value like 4000/200000/500 depending on test config) — `expandFrontier()`'s nested loops (`while (!ctx.truncated && progress)` outer, `for (jlong tag : candidate_tags)` inner) both check a truncation flag and bail out the moment the budget is exhausted, so a single pass's JVMTI-callback time is bounded regardless of heap size. - **Cooperative abort**: an `std::atomic _abort_pass_requested` flag, checked inside `heapReferenceCallback()` (the JVMTI callback invoked per edge), lets `stopThread()` interrupt an **in-flight** walk promptly — set before `pthread_kill(WAKEUP_SIGNAL)`/`pthread_join()`, cleared by `startThread()`. Without this, a `FollowReferences` call already in progress at JVM shutdown or profiler restart can't be interrupted, and `pthread_join()` blocks indefinitely (a real, previously-diagnosed shutdown hang). -- Search state is a small state machine: `RUNNING → {ABANDONED | COMPLETED}`. `RUNNING` can **restart itself** (fresh root walk) once a candidate's chain is found and its tags released, gated by `canAffordNewSearch()`'s **pacing budget** — self-throttling, not unconditional: a search won't restart back-to-back if it would blow the perturbation budget. Once a search reaches a terminal state (`ABANDONED`/`COMPLETED`) it stays there — restarts only happen from within `RUNNING`. +- Search state is a small state machine: `RUNNING → {ABANDONED | COMPLETED}`. `RUNNING` can **restart itself** (fresh root walk) once a candidate's chain is found and its tags released, gated by `canAffordNewSearch()`'s **pacing budget** — self-throttling, not unconditional: a search won't restart back-to-back if it would blow the perturbation budget. A terminal state (`ABANDONED`/`COMPLETED`) is not final: `shouldRunPass()` restarts the search from it via `restartSearch()` once the pain budget has drained and a leak indication is (still) present. - `runPass()` only moves to `COMPLETED` once the frontier is fully drained **and** `_watched_leak_klass_count == 0` (no klass currently under active leak watch, Component 4). A fully-drained frontier while a klass is still watched leaves `_search_state` at `RUNNING` instead: the walk has visited every reachable object once, but a leak-shaped klass keeps growing by **mutating an already-visited container** (e.g. appending to a `static final` collection field long after the walk first admitted it), which a one-time visit can never observe again. Rotation (Component 4) is what re-observes those already-`EXPANDED` entries on later passes. - A search is marked `ABANDONED` (with a reason code) if it runs out of frontier budget without completing — e.g. hitting a frontier-cap under a tiny configured budget. This is a deliberate, observable outcome, not a silent failure — surfaced so operators can distinguish "the walk gave up" from "the walk is still in progress." From 9a6857f5be6acdf77809cfae8160faba67df5bb6 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Thu, 17 Sep 2026 17:59:12 +0200 Subject: [PATCH 16/18] Prevent improveChain from creating cyclic parent chains On a cyclic heap reach the deeper path's referrer can be a descendant of the already-admitted entry being improved; re-parenting to it made the parent chains mutually recursive, and every chain through the cycle then failed to reconstruct (reconstructChain ran to its hop bound and reported failure). Observed as the thread-local leak scenario producing zero chain events: all discovered instances of the watched klass sat inside a cycle rooted at an early ThreadGroup/Thread admission. improveChain now walks the new parent's ancestor chain and refuses when it reaches the entry (bounded at 4096 hops - recorded depths can be understated by later ancestor improves). reparentToDurableRoot cannot create cycles (its new parent must be root-attached). Also: reconstructChain logs the failing hop and the first tag->parent pairs on the broken-chain/hop-bound exits, and improveChain logs its refusals (all level-gated diagnostics). --- ddprof-lib/src/main/cpp/referenceChains.cpp | 65 ++++++++++++++++++++- 1 file changed, 63 insertions(+), 2 deletions(-) diff --git a/ddprof-lib/src/main/cpp/referenceChains.cpp b/ddprof-lib/src/main/cpp/referenceChains.cpp index 8d1b236d97..933636aa19 100644 --- a/ddprof-lib/src/main/cpp/referenceChains.cpp +++ b/ddprof-lib/src/main/cpp/referenceChains.cpp @@ -351,6 +351,40 @@ bool FrontierTable::improveChain(jlong tag, jlong parent_tag, if (parent_tag == tag) { return false; } + // Ancestor-walk bound for the cycle guard below - see its comment. 16x + // the largest hop cap: a legitimate parent chain never comes close. + static constexpr int IMPROVE_CHAIN_GUARD_MAX_HOPS = 4096; + { + int guard_hops = 0; + jlong cur = parent_tag; + while (cur > 0 && guard_hops <= IMPROVE_CHAIN_GUARD_MAX_HOPS) { + if (cur == tag) { + TEST_LOG_SUMMARY("FrontierTable::improveChain refused: new parent " + "chain routes through the entry (cycle) tag=%lld " + "parent_tag=%lld depth=%u", + (long long)tag, (long long)parent_tag, depth); + return false; + } + FrontierEntry guard_entry{}; + if (!lookup(cur, &guard_entry)) { + break; + } + cur = guard_entry.parent_tag; + guard_hops++; + } + if (cur != 0) { + // The parent chain neither reached a root nor was fully verified + // within the guard bound - applying this improve could embed an + // unresolvable (cyclic or dangling) chain. Keep the existing entry. + TEST_LOG_SUMMARY("FrontierTable::improveChain refused: unverifiable " + "parent chain tag=%lld parent_tag=%lld depth=%u " + "walk_stopped_at=%lld", + (long long)tag, (long long)parent_tag, depth, + (long long)cur); + return false; + } + } + int idx = (int)(tag - 1); _table_lock.lock(); @@ -429,15 +463,20 @@ bool FrontierTable::reconstructChain(jlong target_tag, std::vector edges; jlong tag = target_tag; u8 root_kind = 0; + int hops = 0; // Bounded by maxCapacity(): every tag maps to a distinct slot (this table's // "tags/slots are never reused" invariant, see the class comment above), // so a well-formed parent_tag chain can visit at most maxCapacity() slots // before either reaching parent_tag == 0 or repeating a slot. - for (int hops = 0; hops <= maxCapacity() && tag != 0; hops++) { + for (; hops <= maxCapacity() && tag != 0; hops++) { if (!lookup(tag, &entry)) { // parent_tag pointed at a tag that was never inserted - should not // happen for a chain built entirely within one BFS pass, but do not // fabricate a partial chain silently. + TEST_LOG_SUMMARY("FrontierTable::reconstructChain broken chain: " + "target=%lld failed at hop=%d tag=%lld (parent tag never " + "inserted)", + (long long)target_tag, hops, (long long)tag); return false; } chain.push_back(entry.referrer_klass); @@ -467,7 +506,29 @@ bool FrontierTable::reconstructChain(jlong target_tag, if (tag != 0) { // Ran past the defensive hop bound without reaching a root-attached // entry (parent_tag == 0) - a corrupted/cyclic chain. Report failure - // rather than returning a truncated, possibly-misleading chain. + // rather than returning a truncated, possibly-misleading chain. The + // tag->parent dump names the cycle members (improveChain()'s cycle + // guard keeps new ones from forming, but a cycle written before that + // guard existed - or a dangling parent from a concurrent restart - + // still lands here). + { + jlong dbg = target_tag; + FrontierEntry dbg_e{}; + char pairs[256]; + size_t off = 0; + for (int d = 0; d < 12 && dbg != 0 && off < sizeof(pairs) - 24; d++) { + if (!lookup(dbg, &dbg_e)) { + break; + } + off += (size_t)snprintf(pairs + off, sizeof(pairs) - off, "%lld->%lld ", + (long long)dbg, (long long)dbg_e.parent_tag); + dbg = dbg_e.parent_tag; + } + TEST_LOG_SUMMARY("FrontierTable::reconstructChain hop bound: " + "target=%lld stuck at tag=%lld after %d hops - cyclic or " + "corrupt parent chain; hops: %.*s", + (long long)target_tag, (long long)tag, hops, (int)off, pairs); + } return false; } From becb1340f6b8431f5a0901665afb0d35c8407e62 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Thu, 17 Sep 2026 18:41:35 +0200 Subject: [PATCH 17/18] Append the root type to static-field-rooted reference chains The chain's root-side end was the static field's holder instance; the declaring class - the chain's actual root - was only present as the holder hop's edge label and the event's rootKind field. reconstructChain() now optionally returns the root-attached entry, and buildChainEvent()/buildCanaryChainEvent() append the declaring class (resolved from the root-attached entry's referrer_class_tag, captured at admission from the sweep's referrer_tag_ptr) as the chain's root-side terminal element, with one matching kind-only root edge so the _edges.size() == _chain.size() invariant recordReferenceChain() relies on to emit labels is preserved. Thread-rooted chains already end at the Thread instance; other root kinds have no further expressible root object. --- ddprof-lib/src/main/cpp/event.h | 7 +- ddprof-lib/src/main/cpp/referenceChains.cpp | 66 ++++++++++++++++++- ddprof-lib/src/main/cpp/referenceChains.h | 18 ++++- .../src/test/cpp/referenceChains_ut.cpp | 55 ++++++++++++++-- 4 files changed, 136 insertions(+), 10 deletions(-) diff --git a/ddprof-lib/src/main/cpp/event.h b/ddprof-lib/src/main/cpp/event.h index aafa3cc2c0..04a3840e98 100644 --- a/ddprof-lib/src/main/cpp/event.h +++ b/ddprof-lib/src/main/cpp/event.h @@ -97,7 +97,12 @@ class ObjectLivenessEvent : public Event { // BFS (referenceChains.h/.cpp). `_target_tag` is the FrontierTable tag the // chain was reconstructed for (FrontierTable::reconstructChain()); `_chain` // holds the referrer-klass StringDictionary ids it returns, in the same -// leaf(target)-to-root order. `_depth` is the target entry's own +// leaf(target)-to-root order. For a static-field-rooted chain the root-side +// end is the static field's holder instance followed by the declaring class +// (the ROOT TYPE, appended by buildChainEvent() from the root-attached +// entry's referrer_class_tag) - the chain then reads, root-first, as the +// root type retaining the holder through its static field, on down to the +// target. `_depth` is the target entry's own // FrontierEntry::depth (hop count from the search's root-side seed). // `_root_kind` is the jvmtiHeapReferenceKind of whichever edge first // admitted this chain into the frontier (FrontierEntry::root_kind, via diff --git a/ddprof-lib/src/main/cpp/referenceChains.cpp b/ddprof-lib/src/main/cpp/referenceChains.cpp index 933636aa19..5f75e56122 100644 --- a/ddprof-lib/src/main/cpp/referenceChains.cpp +++ b/ddprof-lib/src/main/cpp/referenceChains.cpp @@ -453,7 +453,8 @@ bool FrontierTable::reparentToDurableRoot(jlong tag, jlong new_parent_tag, bool FrontierTable::reconstructChain(jlong target_tag, std::vector *out_chain, u8 *out_root_kind, - std::vector *out_edges) { + std::vector *out_edges, + FrontierEntry *out_terminal) { FrontierEntry entry{}; if (!lookup(target_tag, &entry)) { return false; @@ -542,6 +543,11 @@ bool FrontierTable::reconstructChain(jlong target_tag, // that entry's own FrontierEntry::root_kind. *out_root_kind = root_kind; } + if (out_terminal != nullptr) { + // `entry` still holds the loop's last successful lookup - the + // root-attached entry that ended the walk. + *out_terminal = entry; + } return true; } @@ -5919,12 +5925,15 @@ bool ReferenceChainTracker::buildChainEvent(jvmtiEnv *jvmti, JNIEnv *jni, std::vector chain; std::vector edges; u8 root_kind = 0; - if (!_frontier->reconstructChain(target_tag, &chain, &root_kind, &edges)) { + FrontierEntry terminal{}; + if (!_frontier->reconstructChain(target_tag, &chain, &root_kind, &edges, + &terminal)) { TEST_LOG("ReferenceChainTracker::buildChainEvent false: " "reconstructChain failed for target_tag=%lld", (long long)target_tag); return false; } + appendStaticFieldRootType(terminal, &chain, &edges); TEST_LOG("ReferenceChainTracker::buildChainEvent target_tag=%lld chain_size=%zu " "chain[0]=%u depth=%u root_kind=%u leak_tag=%lld", (long long)target_tag, chain.size(), chain.empty() ? 0u : chain[0], @@ -5938,6 +5947,48 @@ bool ReferenceChainTracker::buildChainEvent(jvmtiEnv *jvmti, JNIEnv *jni, return true; } +// Appends the root TYPE as a chain element for a static-field-rooted chain: +// the frontier path's root-side end is the static field's HOLDER instance +// (the object stored in the field), but the chain's root is the DECLARING +// CLASS - the holder is "the field instance referenced by the root type", +// one hop below it. Without this element a root-first reading of the chain +// starts at the holder and the root type is only present as the rootKind +// event field and the holder hop's edge label. The declaring class's raw +// class tag is recorded on the root-attached entry at admission time +// (FrontierEntry::referrer_class_tag, captured from referrer_tag_ptr for +// root-attached static edges); it is resolved to a StringDictionary id here +// - this is why the append lives on the tracker, not in FrontierTable. +// Skipped when the root kind is not STATIC_FIELD (thread/JNI roots have no +// further expressible root object - the root-attached entry IS the root +// instance or its nearest class), when no declaring-class tag was captured +// (e.g. heapRootCallback-admitted static roots - the root callback carries +// no referrer_tag_ptr), or when the class tag no longer resolves (class +// unloaded). edges gains one matching entry (the root edge - kind label +// only, the field identity belongs to the holder hop) so the +// _edges.size() == _chain.size() invariant recordReferenceChain() relies on +// to emit labels at all is preserved. +void ReferenceChainTracker::appendStaticFieldRootType( + const FrontierEntry &terminal, std::vector *chain, + std::vector *edges) { + if (chain == nullptr || + terminal.root_kind != (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD || + terminal.referrer_class_tag == 0) { + return; + } + u32 root_klass = classTags()->resolve(terminal.referrer_class_tag); + if (root_klass == 0) { + return; + } + chain->push_back(root_klass); + if (edges != nullptr) { + ChainHopEdge root_edge{}; + root_edge.field_index = -1; + root_edge.edge_kind = terminal.root_kind; + root_edge.referrer_class_tag = 0; + edges->push_back(root_edge); + } +} + // Canary chain reconstruction (out of line for the same reason). The // canary's build outcomes are level-1: they are the chase's lifecycle // story, rare (bounded by the candidate count) and chase-relevant. @@ -5956,6 +6007,10 @@ bool ReferenceChainTracker::buildCanaryChainEvent(int candidate_idx, jlong frontier_tag = _candidate_frontier_tags[candidate_idx]; std::vector chain; u8 root_kind = 0; + // The root-attached entry the walk ends at - both branches below leave + // `entry` holding it (the walk's last lookup, or the candidate's own + // entry for a root-referenced candidate). + FrontierEntry terminal{}; if (parent_tag > 0) { // Walk parent_tag back to root through the frontier table. FrontierEntry entry{}; @@ -5976,6 +6031,7 @@ bool ReferenceChainTracker::buildCanaryChainEvent(int candidate_idx, chain.push_back(entry.referrer_klass); tag = entry.parent_tag; } + terminal = entry; } else if (parent_tag == 0 && frontier_tag > 0) { // Root-referenced candidate: chain is just [candidate_klass]. // root_kind was stored in the frontier entry at pruning time; @@ -5988,6 +6044,7 @@ bool ReferenceChainTracker::buildCanaryChainEvent(int candidate_idx, return false; } root_kind = entry.root_kind; + terminal = entry; } else { TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " "never pruned (candidate=%d parent_tag=%lld frontier_tag=%lld)", @@ -5999,6 +6056,11 @@ bool ReferenceChainTracker::buildCanaryChainEvent(int candidate_idx, chain.push_back(candidate_klass); // The chain was built root-to-parent; reverse to get candidate-to-root. std::reverse(chain.begin(), chain.end()); + // Same root-type element buildChainEvent() appends: the canary walk's + // terminal entry is the root-attached entry, and for a static-field root + // the declaring class belongs at the chain's root-side end (after the + // reverse). No-op for other root kinds. + appendStaticFieldRootType(terminal, &chain, nullptr); out->_target_tag = (u64)frontier_tag; out->_depth = _candidate_depths[candidate_idx]; out->_root_kind = root_kind; diff --git a/ddprof-lib/src/main/cpp/referenceChains.h b/ddprof-lib/src/main/cpp/referenceChains.h index 08be758a88..98613c0b90 100644 --- a/ddprof-lib/src/main/cpp/referenceChains.h +++ b/ddprof-lib/src/main/cpp/referenceChains.h @@ -551,9 +551,17 @@ class alignas(alignof(SpinLock)) FrontierTable { // first admitted this chain into the frontier, letting a caller label the // chain with why it is reachable at all (JNI global, thread stack, static // field, ...) instead of just how (the referrer_klass hops in *out_chain). + // + // `out_terminal` (if non-null) receives the root-attached entry itself - + // the chain's root-side end. Callers that need the ROOT TYPE as a chain + // element (buildChainEvent() appends the declaring class of a + // static-field-rooted chain) read FrontierEntry::referrer_class_tag from + // it - this table stores class tags, not StringDictionary ids, so the + // resolution stays with the tracker. bool reconstructChain(jlong target_tag, std::vector *out_chain, u8 *out_root_kind = nullptr, - std::vector *out_edges = nullptr); + std::vector *out_edges = nullptr, + FrontierEntry *out_terminal = nullptr); // Search restart (ReferenceChainTracker::restartSearch(), this class's own // header comment): marks every slot unoccupied again without releasing @@ -3303,6 +3311,14 @@ class ReferenceChainTracker { bool buildChainEvent(jvmtiEnv *jvmti, JNIEnv *jni, jlong target_tag, ReferenceChainEvent *out); + // Appends the root TYPE element (the declaring class, resolved from + // FrontierEntry::referrer_class_tag) to a static-field-rooted chain - see + // the definition's comment in referenceChains.cpp for the full rationale + // and the skip conditions. + void appendStaticFieldRootType(const FrontierEntry &terminal, + std::vector *chain, + std::vector *edges); + // Canary-search chain reconstruction: builds the chain for a canary // candidate from the per-candidate chain link recorded at // pruning time (_candidate_parent_tags[] etc.), walking diff --git a/ddprof-lib/src/test/cpp/referenceChains_ut.cpp b/ddprof-lib/src/test/cpp/referenceChains_ut.cpp index 3294bd4d29..7acebf75a7 100644 --- a/ddprof-lib/src/test/cpp/referenceChains_ut.cpp +++ b/ddprof-lib/src/test/cpp/referenceChains_ut.cpp @@ -2275,6 +2275,26 @@ TEST_F(ReferenceChainsBfsTest, DiscoversObjectRetainedOnlyByStaticField) { EXPECT_EQ(0, entry.parent_tag); // root-attached, not a child hop EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD, entry.root_kind); + // buildChainEvent() appends the root TYPE (the declaring class) as the + // chain's root-side end: the frontier path's terminal element is the + // static field's holder instance, one hop below the root type. The + // declaring class is resolved from the root-attached entry's + // referrer_class_tag (captured from the sweep's referrer_tag_ptr, a + // negative class tag). + ReferenceChainEvent event; + ASSERT_TRUE(tracker->buildChainEvent(&mock_jvmti, &mock_jni, target_tag, + &event)); + int expectedHolder = Profiler::instance()->lookupClass( + "com/rc/statics/Holder", strlen("com/rc/statics/Holder")); + ASSERT_NE(-1, expectedHolder); + ASSERT_EQ(2u, event._chain.size()); + EXPECT_EQ(chain[0], event._chain[0]); // the target's own class, unchanged + EXPECT_EQ((u32)expectedHolder, event._chain[1]); + // One edge per chain element (recordReferenceChain() drops ALL labels + // when the arrays' sizes diverge); the root-type hop's own edge is the + // unlabeled root edge (field_index -1). + ASSERT_EQ(event._chain.size(), event._edges.size()); + tracker->stop(); } @@ -6858,7 +6878,9 @@ TEST_F(ReferenceChainsBfsTest, HopEdgeLabelsDecodeSpecFieldOrdinals) { // Chain: [chunk(3)] <- Base.base_f(ordinal 0 over Base's space) <- // [value2(2), class Base] <- Holder.leakList(ordinal 4, the static root // edge with the declaring class as referrer) <- [static value(1), class - // Holder]. Interior hops decode against the PARENT entry's class_tag. + // Holder] <- [class Holder (the ROOT TYPE - buildChainEvent() appends + // the root-attached entry's referrer_class_tag for static-field roots)]. + // Interior hops decode against the PARENT entry's class_tag. ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( frontier, 1, 0, 0, FrontierEntryState::EDGE, JVMTI_HEAP_REFERENCE_STATIC_FIELD, @@ -6877,16 +6899,22 @@ TEST_F(ReferenceChainsBfsTest, HopEdgeLabelsDecodeSpecFieldOrdinals) { ReferenceChainEvent event; ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( &mock_jvmti, &mock_jni, /*target_tag=*/3, &event)); - ASSERT_EQ(3u, event._chain.size()); - ASSERT_EQ(3u, event._edges.size()) + ASSERT_EQ(4u, event._chain.size()); + ASSERT_EQ(4u, event._edges.size()) << "edge labels must align with the chain, one per hop"; // Leaf first: chunk is retained via Base.base_f (parent entry's class is // Base, ordinal 0 in Base's own space), then value2 via Holder's // holder_a (ordinal 3 = interface offset 2 + Base's 1 + own position 0), - // then the static root edge's field name leakList (ordinal 4). + // then the static root edge's field name leakList (ordinal 4), then the + // root-type hop (class Holder) - the root edge itself, kind label only. EXPECT_EQ("base_f", event._edges[0]); EXPECT_EQ("holder_a", event._edges[1]); EXPECT_EQ("leakList", event._edges[2]); + EXPECT_EQ("static_field", event._edges[3]); + int expectedRootType = Profiler::instance()->lookupClass( + "com/rc/labels/Holder", strlen("com/rc/labels/Holder")); + ASSERT_NE(-1, expectedRootType); + EXPECT_EQ((u32)expectedRootType, event._chain[3]); // Interface-referrer branch: ISink's own-field ordinals have NO // superclass-chain component (base = superinterfaces' fields only). @@ -6899,8 +6927,15 @@ TEST_F(ReferenceChainsBfsTest, HopEdgeLabelsDecodeSpecFieldOrdinals) { ReferenceChainEvent iface_event; ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( &mock_jvmti, &mock_jni, /*target_tag=*/4, &iface_event)); - ASSERT_EQ(1u, iface_event._edges.size()); + // Same root-type append as above: the root-side end gains ISink (the + // declaring class of the static field) plus its kind-only root edge. + ASSERT_EQ(2u, iface_event._edges.size()); EXPECT_EQ("CONST_B", iface_event._edges[0]); + EXPECT_EQ("static_field", iface_event._edges[1]); + int expectedSinkRoot = Profiler::instance()->lookupClass( + "com/rc/labels/ISink", strlen("com/rc/labels/ISink")); + ASSERT_NE(-1, expectedSinkRoot); + EXPECT_EQ((u32)expectedSinkRoot, iface_event._chain[1]); // Fail-safe: a referrer class that cannot be resolved degrades to the // edge KIND label, never a fabricated name. @@ -6913,8 +6948,16 @@ TEST_F(ReferenceChainsBfsTest, HopEdgeLabelsDecodeSpecFieldOrdinals) { ReferenceChainEvent degraded_event; ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( &mock_jvmti, &mock_jni, /*target_tag=*/5, °raded_event)); - ASSERT_EQ(1u, degraded_event._edges.size()); + // Both hops degrade to kind labels: the holder hop's referrer class + // (IBase) is deliberately unregistered from the decoder, and the + // appended root-type hop (IBase itself) carries no field identity. + ASSERT_EQ(2u, degraded_event._edges.size()); EXPECT_EQ("static_field", degraded_event._edges[0]); + EXPECT_EQ("static_field", degraded_event._edges[1]); + int expectedIbaseRoot = Profiler::instance()->lookupClass( + "com/rc/labels/IBase", strlen("com/rc/labels/IBase")); + ASSERT_NE(-1, expectedIbaseRoot); + EXPECT_EQ((u32)expectedIbaseRoot, degraded_event._chain[1]); tracker->stop(); } From 2bd422e84df9c317d5cb405672b22731f74b2a76 Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Thu, 17 Sep 2026 18:47:37 +0200 Subject: [PATCH 18/18] Stop racing the BFS thread in the urgent-OOM gate test The test asserted shouldRunPassForTest0() directly, but opening the gate spends the episode's one-shot urgency entitlement on the FIRST evaluation - and the BFS thread, whose cadence ramps to ~10ms once urgency latches, races the test's own call for it. The direct call only returned true when it won that race, failing deterministically on fast runners. Assert through the thread instead: seed the rising floor, then wait for referenceChainPassesRunForTest0() to advance. A pass only runs after shouldRunPass() returned true, and with an empty per-klass population table the urgent-OOM projection is the only possible trigger. --- .../AggressiveLeakReferenceChainTest.java | 38 +++++++++++++++---- 1 file changed, 31 insertions(+), 7 deletions(-) diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/AggressiveLeakReferenceChainTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/AggressiveLeakReferenceChainTest.java index b3a0cfd247..8cc7f9a440 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/AggressiveLeakReferenceChainTest.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/referencechains/AggressiveLeakReferenceChainTest.java @@ -65,11 +65,22 @@ private static void assumeDebugBuild() { * Seeds ten heap-floor-ring samples rising fast enough that {@code secondsToOOM()}'s projection * lands well under {@code OOM_URGENT_THRESHOLD_S} (300s), with LivenessTracker's per-klass * population table left empty throughout - {@code selectLeakCandidateKlassIds0()} returns - * nothing at any point in this test. Asserts the search-restart gate still opens, proving the - * urgent-OOM projection alone - not a per-klass candidate - is what let it through. + * nothing at any point in this test. Then waits for the BFS thread to actually run passes, + * proving the search-restart gate opened from the urgent-OOM projection alone. + * + *

    The gate is asserted through the live BFS thread rather than by calling + * {@code shouldRunPassForTest0()} directly: opening the gate spends the episode's one-shot + * urgency entitlement ({@code _urgent_search_spent}) on the FIRST evaluation, and the BFS + * thread - whose cadence ramps down to ~10ms once urgency latches - races the test's own + * call for that first evaluation. A direct call therefore only returns true when it wins + * that race, which made this test flaky by construction. A pass only ever runs after + * {@code shouldRunPass()} returned true, and with {@code generations=true} plus a provably + * empty per-klass population table the urgent-OOM projection is the only possible trigger - + * so "passes ran while zero candidates existed" is the same assertion, without the race. */ @Test - public void shouldOpenSearchGateOnAggressiveHeapWideGrowthWithNoLeakCandidate() { + public void shouldOpenSearchGateOnAggressiveHeapWideGrowthWithNoLeakCandidate() + throws InterruptedException { assumeDebugBuild(); JavaProfiler.setHeapFloorRecordingForTest0(false); JavaProfiler.resetKlassPopulationForTest0(); @@ -87,12 +98,25 @@ public void shouldOpenSearchGateOnAggressiveHeapWideGrowthWithNoLeakCandidate() int[] candidates = JavaProfiler.selectLeakCandidateKlassIds0(); assertTrue(candidates == null || candidates.length == 0, - "This test's own precondition: no per-klass candidate should exist, so a true result " + "This test's own precondition: no per-klass candidate should exist, so passes running " + "below can only come from the aggregate urgent-OOM bypass"); - assertTrue(JavaProfiler.shouldRunPassForTest0(), - "Expected the search-restart gate to open from the urgent heap-wide OOM projection " - + "alone, with zero per-klass leak candidate"); + int passesBefore = JavaProfiler.referenceChainPassesRunForTest0(); + // Idle BFS cadence is ~1s/pass; urgency ramps it to ~10ms after the latch. 20s covers the + // first evaluation landing up to one idle cadence after the seeding loop. + long deadline = System.currentTimeMillis() + 20_000; + while (JavaProfiler.referenceChainPassesRunForTest0() == passesBefore + && System.currentTimeMillis() < deadline) { + Thread.sleep(50); + } + assertTrue(JavaProfiler.referenceChainPassesRunForTest0() > passesBefore, + "Expected the BFS thread to run passes from the urgent heap-wide OOM projection alone, " + + "with zero per-klass leak candidate"); + + candidates = JavaProfiler.selectLeakCandidateKlassIds0(); + assertTrue(candidates == null || candidates.length == 0, + "The passes that just ran must still have had zero per-klass candidates - the " + + "aggregate urgent-OOM projection was the only possible trigger"); } finally { JavaProfiler.setMaxHeapBytesForTest0(-1); JavaProfiler.setHeapFloorRecordingForTest0(true);