diff --git a/ddprof-lib/src/main/cpp/counters.h b/ddprof-lib/src/main/cpp/counters.h index 0a6810669f..ec39e8f051 100644 --- a/ddprof-lib/src/main/cpp/counters.h +++ b/ddprof-lib/src/main/cpp/counters.h @@ -74,6 +74,10 @@ X(THREAD_NAMES_COUNT, "thread_names_count") \ X(THREAD_FILTER_PAGES, "thread_filter_pages") \ X(THREAD_FILTER_BYTES, "thread_filter_bytes") \ + X(THREAD_REGISTRY_CAPACITY_EXHAUSTED, "thread_registry_capacity_exhausted") \ + X(THREAD_REGISTRY_INDEX_FAILURES, "thread_registry_index_failures") \ + X(THREAD_REGISTRY_CONTEXT_RESET_RACE_DETECTED, "thread_registry_context_reset_race_detected") \ + X(THREAD_REGISTRY_HOOK_REREGISTRATION, "thread_registry_hook_reregistration") \ X(JMETHODID_SKIPPED, "jmethodid_skipped_count") \ X(CODECACHE_NATIVE_SIZE_BYTES, "codecache_native_size_bytes") \ X(CODECACHE_NATIVE_COUNT, "native_codecache_count") \ @@ -84,6 +88,9 @@ X(AGCT_BLOCKED_IN_VM, "agct_blocked_in_vm") \ X(SKIPPED_WALLCLOCK_UNWINDS, "skipped_wallclock_unwinds") \ X(WC_SIGNAL_SUPPRESSED_SAMPLED_RUN, "wc_signals_suppressed_sampled_run") \ + X(WC_PRECHECK_REGISTRY_LOOKUPS, "wc_precheck_registry_lookups") \ + X(WC_PRECHECK_CANDIDATES_REJECTED, "wc_precheck_candidates_rejected") \ + X(WC_PRECHECK_LOOKUP_BUDGET_EXHAUSTED, "wc_precheck_lookup_budget_exhausted") \ X(WC_UNOWNED_BLOCKED_SUPPRESSED, "wc_unowned_blocked_suppressed") \ X(WC_UNOWNED_BLOCKED_RECORDED, "wc_unowned_blocked_recorded") \ X(WC_SIGNAL_QUEUE_FULL, "wc_signals_queue_full") \ diff --git a/ddprof-lib/src/main/cpp/engine.h b/ddprof-lib/src/main/cpp/engine.h index c9e75e2986..d1e3def948 100644 --- a/ddprof-lib/src/main/cpp/engine.h +++ b/ddprof-lib/src/main/cpp/engine.h @@ -1,5 +1,6 @@ /* * Copyright 2017 Andrei Pangin + * Copyright 2026, Datadog, Inc. * * Licensed under the Apache License, Version 2.0 (the "License"); * you may not use this file except in compliance with the License. @@ -51,6 +52,12 @@ class Engine { virtual void stop(); virtual long interval() const { return 0L; } + // Whether this engine can keep the ThreadFilter registry populated and + // tracked when context filtering is disabled with an explicit empty + // `filter=` (e.g. to support wall-clock prechecks). Without a filter + // argument the filter defaults to "0", which enables context filtering. + virtual bool supportsUnfilteredThreadRegistryTracking() const { return false; } + virtual int registerThread(int tid) { return -1; } virtual void unregisterThread(int tid) {} diff --git a/ddprof-lib/src/main/cpp/frames.h b/ddprof-lib/src/main/cpp/frames.h index 15549e6e82..a0d0f498b3 100644 --- a/ddprof-lib/src/main/cpp/frames.h +++ b/ddprof-lib/src/main/cpp/frames.h @@ -1,9 +1,28 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ #ifndef _FRAMES_H #define _FRAMES_H #include +#include #include "vmEntry.h" +inline void copyJvmtiFrames(ASGCT_CallFrame *frames, + const jvmtiFrameInfo *jvmti_frames, + jint num_frames) { + // The source and destination commonly refer to the two views of the same + // CallTraceBuffer union. Read both source fields before either write. + for (jint i = 0; i < num_frames; ++i) { + jmethodID method = jvmti_frames[i].method; + jlocation location = jvmti_frames[i].location; + frames[i].method_id = method; + frames[i].bci = static_cast(location); + LP64_ONLY(frames[i].padding = 0;) + } +} + inline int makeFrame(ASGCT_CallFrame *frames, jint type, jmethodID id) { frames[0].bci = type; frames[0].method_id = id; diff --git a/ddprof-lib/src/main/cpp/javaApi.cpp b/ddprof-lib/src/main/cpp/javaApi.cpp index 90f6696755..18d290c313 100644 --- a/ddprof-lib/src/main/cpp/javaApi.cpp +++ b/ddprof-lib/src/main/cpp/javaApi.cpp @@ -34,6 +34,7 @@ #include "threadLocalData.inline.h" #include "tsc.h" #include "vmEntry.h" +#include "wallClock.h" #include #include #include @@ -163,6 +164,44 @@ Java_com_datadoghq_profiler_JavaProfiler_getSamples(JNIEnv *env, // some duplication between add and remove, though we want to avoid having an extra branch in the hot path +static ThreadFilter::SlotID ensureCurrentThreadFilterSlot( + ThreadFilter *thread_filter, ProfiledThread *current) { + int tid = current->tid(); + if (unlikely(tid < 0)) { + return -1; + } + + ThreadFilter::SlotID slot_id = current->filterSlotId(); + if (likely(slot_id >= 0)) { + if (likely(thread_filter->activeSlotForId(slot_id, tid) != nullptr)) { + return slot_id; + } + current->setFilterSlotId(-1); + } + + // Threads that existed before the recording started (and so never received + // a ThreadStart callback) bind their slot lazily here. If the tid is already + // indexed, registerThread(tid) returns that existing slot. + // + // This is the only place the filterThreadAdd0 (JavaCritical), parkEnter0 + // and blockEnter0 hooks can block on _registry_lock. It's bounded to at + // most once per thread lifetime (cold TLS) plus once per recording-epoch + // transition this thread observes (stale cached slot) - not a per-call cost. + // A thread that cannot get a slot because the registry is full retries here + // on every call, but registerThread() rejects it without taking the lock. + // THREAD_REGISTRY_HOOK_REREGISTRATION counts every attempt, including those + // lock-free rejections (tracked separately by + // THREAD_REGISTRY_CAPACITY_EXHAUSTED). If it fires per call while + // THREAD_REGISTRY_CAPACITY_EXHAUSTED stays flat, the "provably rare" + // assumption has broken and these hooks should be revisited. + Counters::increment(THREAD_REGISTRY_HOOK_REREGISTRATION); + slot_id = thread_filter->registerThread(tid); + if (slot_id >= 0) { + current->setFilterSlotId(slot_id); + } + return slot_id; +} + // JavaCritical is faster JNI, but more restrictive - parameters and return value have to be // primitives or arrays of primitive types. // We direct corresponding JNI calls to JavaCritical to make sure the parameters/return value @@ -180,25 +219,23 @@ JavaCritical_com_datadoghq_profiler_JavaProfiler_filterThreadAdd0() { return; } ThreadFilter *thread_filter = Profiler::instance()->threadFilter(); - if (unlikely(!thread_filter->enabled())) { + if (unlikely(!thread_filter->registryActive())) { return; } - int slot_id = current->filterSlotId(); - if (unlikely(slot_id == -1)) { - // Thread doesn't have a slot ID yet (e.g., main thread), so register it - // Happens when we are not enabled before thread start - slot_id = thread_filter->registerThread(); - current->setFilterSlotId(slot_id); - } - - if (unlikely(slot_id == -1)) { + int slot_id = ensureCurrentThreadFilterSlot(thread_filter, current); + if (unlikely(slot_id < 0)) { return; // Failed to register thread } - // Reset suppression state so a new thread occupying this slot does not inherit - // stale state from its predecessor. Must happen before add(). - thread_filter->resetSlotRunState(slot_id); - thread_filter->add(tid, slot_id); + if (unlikely(!thread_filter->add(tid, slot_id))) { + // The cached slot_id went stale between ensureCurrentThreadFilterSlot() + // and add(): an unfiltered recording restart reset the registry, so the + // slot no longer carries this thread's tid. + // Clear the cache so the next filterThreadAdd0()/parkEnter0()/blockEnter0() + // call re-runs ensureCurrentThreadFilterSlot()'s registerThread() path + // instead of leaving this thread permanently outside the context window. + current->setFilterSlotId(-1); + } } extern "C" DLLEXPORT void JNICALL @@ -213,13 +250,15 @@ JavaCritical_com_datadoghq_profiler_JavaProfiler_filterThreadRemove0() { return; } ThreadFilter *thread_filter = Profiler::instance()->threadFilter(); - if (unlikely(!thread_filter->enabled())) { + if (unlikely(!thread_filter->registryActive())) { return; } int slot_id = current->filterSlotId(); - if (unlikely(slot_id == -1)) { - // Thread doesn't have a slot ID yet - nothing to remove + if (unlikely(slot_id == -1 || + thread_filter->activeSlotForId(slot_id, tid) == nullptr)) { + // No slot yet, or a cached slot left over from an earlier recording - + // either way this thread is not in the context window, nothing to remove return; } thread_filter->remove(slot_id); @@ -307,6 +346,34 @@ Java_com_datadoghq_profiler_JavaProfiler_describeDebugCounters0( #endif // COUNTERS } +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfilerTestSupport_setForceWallStartFailureForTest0( + JNIEnv *env, jclass unused, jboolean force) { +#ifdef DEBUG + BaseWallClock::setForceStartFailureForTest(force); +#endif // DEBUG +} + +extern "C" DLLEXPORT jboolean JNICALL +Java_com_datadoghq_profiler_JavaProfilerTestSupport_isForceWallStartFailureArmedForTest0( + JNIEnv *env, jclass unused) { +#ifdef DEBUG + return BaseWallClock::isForceStartFailureForTest() ? JNI_TRUE : JNI_FALSE; +#else + // The setForceWallStartFailureForTest0 hook above is a no-op outside DEBUG, + // so the forced-failure toggle can never be armed here. Returning false lets + // callers (e.g. UnfilteredWallPrecheckFallbackTest) self-skip via + // Assumptions.assumeTrue rather than fail spuriously in release builds. + return JNI_FALSE; +#endif // DEBUG +} + +extern "C" DLLEXPORT jboolean JNICALL +Java_com_datadoghq_profiler_JavaProfilerTestSupport_isThreadRegistryActiveForTest0( + JNIEnv *env, jclass unused) { + return Profiler::instance()->threadFilter()->registryActive(); +} + extern "C" DLLEXPORT void JNICALL Java_com_datadoghq_profiler_JavaProfiler_recordSettingEvent0( JNIEnv *env, jclass unused, jstring name, jstring value, jstring unit) { @@ -383,11 +450,12 @@ Java_com_datadoghq_profiler_JavaProfiler_parkEnter0(JNIEnv *env, jclass unused) bool first_park = current->parkEnter(); ThreadFilter *tf = Profiler::instance()->threadFilter(); - if (first_park && tf->enabled()) { - ThreadFilter::SlotID slot_id = current->filterSlotId(); + if (first_park && tf->registryActive()) { + ThreadFilter::SlotID slot_id = ensureCurrentThreadFilterSlot(tf, current); if (slot_id >= 0) { + WallClockBlockTracker *tracker = Profiler::instance()->blockTracker(); current->setParkBlockToken( - tf->enterBlockedRun(slot_id, OSThreadState::CONDVAR_WAIT)); + tracker->enterBlockedRun(tf, slot_id, OSThreadState::CONDVAR_WAIT)); } } } @@ -405,10 +473,12 @@ Java_com_datadoghq_profiler_JavaProfiler_parkExit0( return; } ThreadFilter *tf = Profiler::instance()->threadFilter(); - if (tf->enabled()) { - ThreadFilter::SlotID slot_id = ThreadFilter::tokenSlotId(park_block_token); - if (current->filterSlotId() == slot_id) { - tf->exitBlockedRun(slot_id, ThreadFilter::tokenGeneration(park_block_token)); + if (tf->registryActive()) { + ThreadFilter::SlotID slot_id = WallClockBlockTracker::tokenSlotId(park_block_token); + if (tf->activeSlotForId(current->filterSlotId(), current->tid()) != nullptr && + current->filterSlotId() == slot_id) { + WallClockBlockTracker *tracker = Profiler::instance()->blockTracker(); + tracker->exitBlockedRun(slot_id, WallClockBlockTracker::tokenGeneration(park_block_token)); } } } @@ -435,14 +505,15 @@ Java_com_datadoghq_profiler_JavaProfiler_blockEnter0( return 0; } ThreadFilter *tf = Profiler::instance()->threadFilter(); - if (!tf->enabled()) { + if (!tf->registryActive()) { return 0; } - ThreadFilter::SlotID slot_id = current->filterSlotId(); + ThreadFilter::SlotID slot_id = ensureCurrentThreadFilterSlot(tf, current); if (slot_id < 0) { return 0; } - return static_cast(tf->enterBlockedRun(slot_id, decoded)); + WallClockBlockTracker *tracker = Profiler::instance()->blockTracker(); + return static_cast(tracker->enterBlockedRun(tf, slot_id, decoded)); } extern "C" DLLEXPORT void JNICALL @@ -457,14 +528,15 @@ Java_com_datadoghq_profiler_JavaProfiler_blockExit0( if (current == nullptr) { return; } - - ThreadFilter::SlotID slot_id = ThreadFilter::tokenSlotId(block_token); - if (current->filterSlotId() != slot_id) { + ThreadFilter *tf = Profiler::instance()->threadFilter(); + ThreadFilter::SlotID slot_id = WallClockBlockTracker::tokenSlotId(block_token); + if (current->filterSlotId() != slot_id || + tf->activeSlotForId(slot_id, current->tid()) == nullptr) { return; } - ThreadFilter *tf = Profiler::instance()->threadFilter(); - if (tf->enabled()) { - tf->exitBlockedRun(slot_id, ThreadFilter::tokenGeneration(block_token)); + if (tf->registryActive()) { + WallClockBlockTracker *tracker = Profiler::instance()->blockTracker(); + tracker->exitBlockedRun(slot_id, WallClockBlockTracker::tokenGeneration(block_token)); } } diff --git a/ddprof-lib/src/main/cpp/profiler.cpp b/ddprof-lib/src/main/cpp/profiler.cpp index 2c1b3271c3..f2b1c454e2 100644 --- a/ddprof-lib/src/main/cpp/profiler.cpp +++ b/ddprof-lib/src/main/cpp/profiler.cpp @@ -89,11 +89,12 @@ void Profiler::onThreadStart(jvmtiEnv *jvmti, JNIEnv *jni, jthread thread) { current->setJavaThread(true); int tid = current->tid(); - if (_thread_filter.enabled()) { - int slot_id = _thread_filter.registerThread(); + // Java lifecycle callbacks own registry allocation. The wall timer only + // looks up these entries and must never allocate slots for arbitrary OS + // threads returned by OS::listThreads(). + if (_thread_filter.registryActive()) { + int slot_id = _thread_filter.registerThread(tid); current->setFilterSlotId(slot_id); - _thread_filter.resetSlotRunState(slot_id); - _thread_filter.remove(slot_id); // Remove from filtering initially } if (thread != NULL) { updateThreadName(jvmti, jni, thread, true); @@ -113,9 +114,11 @@ void Profiler::onThreadEnd(jvmtiEnv *jvmti, JNIEnv *jni, jthread thread) { int slot_id = current->filterSlotId(); tid = current->tid(); - if (_thread_filter.enabled()) { - _thread_filter.unregisterThread(slot_id); + if (slot_id >= 0) { + _thread_filter.unregisterThread(slot_id, tid); current->setFilterSlotId(-1); + } else { + _thread_filter.unregisterThreadByTid(tid); } updateThreadName(jvmti, jni, thread, false); @@ -140,6 +143,7 @@ void Profiler::onThreadEnd(jvmtiEnv *jvmti, JNIEnv *jni, jthread thread) { } updateThreadName(jvmti, jni, thread, false); + _thread_filter.unregisterThreadByTid(tid); _cpu_engine->unregisterThread(tid); _wall_engine->unregisterThread(tid); LivenessTracker::instance()->releaseThreadLocalState(); @@ -591,14 +595,7 @@ u64 Profiler::recordJVMTISample(u64 counter, int tid, jthread thread, jint event if (VM::jvmti()->GetStackTrace(thread, 0, _max_stack_depth, jvmti_frames, &num_frames) == JVMTI_ERROR_NONE && num_frames > 0) { // Convert to AsyncGetCallTrace format. // Note: jvmti_frames and frames may overlap. - for (int i = 0; i < num_frames; i++) { - jint bci = jvmti_frames[i].location; - jmethodID mid = jvmti_frames[i].method; - frames[i].method_id = mid; - frames[i].bci = bci; - // see https://github.com/async-profiler/async-profiler/pull/1090 - LP64_ONLY(frames[i].padding = 0;) - } + copyJvmtiFrames(frames, jvmti_frames, num_frames); // On JDK 21+, GetStackTrace on a virtual thread returns only the VT's // logical stack; it stops at the continuation boundary and never includes // carrier-thread frames. Without a synthetic root the trace appears @@ -1584,24 +1581,40 @@ Error Profiler::start(Arguments &args, bool reset) { _safe_mode |= GC_TRACES | LAST_JAVA_PC; } - // TODO: Current way of setting filter is weird with the recent changes - _thread_filter.init(args._filter ? args._filter : "0"); - - // Minor optim: Register the current thread (start thread won't be called) + _cpu_engine = selectCpuEngine(args); + _wall_engine = selectWallEngine(args); + + const char *filter = args._filter != nullptr ? args._filter : "0"; + const bool track_unfiltered_wall = + (_event_mask & EM_WALL) != 0 && args._wall_precheck && + args._filter != nullptr && args._filter[0] == '\0' && + _wall_engine->supportsUnfilteredThreadRegistryTracking(); + _thread_filter.init(filter, track_unfiltered_wall); + + // Unfiltered init resets registrations before publishing registry admission. + // Context-filtered recordings retain identities but must clear membership. if (_thread_filter.enabled()) { _thread_filter.clearActive(); + } + + // Preserve the context-filter fast path, and extend it to unfiltered + // wall-precheck tracking: the calling thread must have a slot immediately, + // not only lazily via block/park/filterThreadAdd0 hooks. Other pre-existing + // Java threads are still registered lazily via those hooks or ThreadStart + // callbacks (both filter modes) - proactive registration here is only for + // the one thread that cannot update its own TLS from another thread's start(). + // A cached slot id is not proof of registration: an unfiltered restart resets + // the registry, leaving it pointing at a free slot or one now owned by another + // thread. activeSlotForId() rejects both (tid and recording-epoch mismatch). + if (_thread_filter.registryActive()) { ProfiledThread *current = ProfiledThread::initCurrentThreadSignalSafe(); assert(current != nullptr); - int slot_id = current->filterSlotId(); - if (slot_id < 0) { - slot_id = _thread_filter.registerThread(); - current->setFilterSlotId(slot_id); + int tid = current->tid(); + if (_thread_filter.activeSlotForId(current->filterSlotId(), tid) == nullptr) { + current->setFilterSlotId(_thread_filter.registerThread(tid)); } - _thread_filter.remove(slot_id); // Remove from filtering initially (matches onThreadStart behavior) } - _cpu_engine = selectCpuEngine(args); - _wall_engine = selectWallEngine(args); _cstack = args._cstack; _force_jmethodID = args._force_jmethodID; if (_cstack == CSTACK_DEFAULT) { @@ -1659,6 +1672,7 @@ Error Profiler::start(Arguments &args, bool reset) { _num_context_attributes = args._context_attributes.size(); error = _jfr.start(args, reset); if (error) { + _thread_filter.deactivateRecording(); switchLibraryTrap(false); _libs->stopRefresher(); return error; @@ -1716,6 +1730,7 @@ Error Profiler::start(Arguments &args, bool reset) { Log::warn("%s", error.message()); if (_event_mask == EM_NATIVEMEM) { // nativemem is the only requested mode: propagate the real error + _thread_filter.deactivateRecording(); disableEngines(); switchLibraryTrap(false); _libs->stopRefresher(); @@ -1739,6 +1754,12 @@ Error Profiler::start(Arguments &args, bool reset) { } } + // A recoverable wall-engine failure must not leave registry work enabled for + // unrelated engines that did start successfully. + if (track_unfiltered_wall && (activated & EM_WALL) == 0) { + _thread_filter.deactivateRecording(); + } + if (activated) { switchThreadEvents(JVMTI_ENABLE); @@ -1763,6 +1784,7 @@ Error Profiler::start(Arguments &args, bool reset) { return Error::OK; } // no engine was activated; perform cleanup + _thread_filter.deactivateRecording(); disableEngines(); switchLibraryTrap(false); _libs->stopRefresher(); @@ -1814,6 +1836,8 @@ Error Profiler::stop() { if (_event_mask & EM_CPU) _cpu_engine->stop(); + _thread_filter.deactivateRecording(); + switchLibraryTrap(false); switchThreadEvents(JVMTI_DISABLE); Libraries::instance()->refresh(); diff --git a/ddprof-lib/src/main/cpp/profiler.h b/ddprof-lib/src/main/cpp/profiler.h index 2d3d0bd09e..5512c3480a 100644 --- a/ddprof-lib/src/main/cpp/profiler.h +++ b/ddprof-lib/src/main/cpp/profiler.h @@ -29,6 +29,7 @@ #include "threadInfo.h" #include "trap.h" #include "vmEntry.h" +#include "wallClockBlockTracker.h" #include #include #include @@ -113,6 +114,7 @@ class alignas(alignof(SpinLock)) Profiler { // JavaProfiler.execute / ContextValueCache. std::atomic _context_value_dict_reset{false}; ThreadFilter _thread_filter; + WallClockBlockTracker _block_tracker; CallTraceStorage _call_trace_storage; FlightRecorder _jfr; Engine *_cpu_engine; @@ -238,6 +240,7 @@ class alignas(alignof(SpinLock)) Profiler { for (int i = 0; i < CONCURRENCY_LEVEL; i++) { _calltrace_buffer[i] = NULL; } + _thread_filter.setBlockTracker(&_block_tracker); } static inline Profiler *instance() { @@ -258,7 +261,6 @@ class alignas(alignof(SpinLock)) Profiler { // dump-time pass (which passes false), records the final name instead. void updateNativeThreadNames(bool defer_initializing = false); - inline void incFailure(int type) { if (type < ASGCT_FAILURE_TYPES) { atomicIncRelaxed(_failures[type]); @@ -312,6 +314,7 @@ class alignas(alignof(SpinLock)) Profiler { } u32 numContextAttributes() { return _num_context_attributes; } ThreadFilter *threadFilter() { return &_thread_filter; } + WallClockBlockTracker *blockTracker() { return &_block_tracker; } const char* cstack() const; int lookupClass(const char *key, size_t length); diff --git a/ddprof-lib/src/main/cpp/stackWalker.inline.h b/ddprof-lib/src/main/cpp/stackWalker.inline.h index 965f554cfc..13b67434ab 100644 --- a/ddprof-lib/src/main/cpp/stackWalker.inline.h +++ b/ddprof-lib/src/main/cpp/stackWalker.inline.h @@ -61,7 +61,15 @@ inline void fillFrame(ASGCT_CallFrame& frame, FrameTypeId type, int bci, jmethod // rescues only the padding-gap and zero-size-symbol cases, not zero-gap // adjacency (see BinarySearchPicksNextSymbolAtZeroGapBoundary). inline const void* attributionPC(const void* pc, bool pc_is_return_address) { - return pc_is_return_address ? (const void*)((const char*)pc - 1) : pc; + // A null walking pc (a zeroed return-address slot read during an optimistic + // unwind) has no code to attribute: findLibraryByAddress(nullptr) fails and + // no FDE matches, adjusted or not. Pointer arithmetic on nullptr is UB + // (UBSan: "applying non-zero offset to null pointer"), so pass it through + // unchanged -- the raw null is exactly what the pre-adjustment walkers used + // for lookup anyway. + return pc != nullptr && pc_is_return_address + ? (const void*)((const char*)pc - 1) + : pc; } // The walking pc and "was it loaded from a return-address slot?" are one diff --git a/ddprof-lib/src/main/cpp/threadFilter.cpp b/ddprof-lib/src/main/cpp/threadFilter.cpp index ea9e4acfa3..0e0235113b 100644 --- a/ddprof-lib/src/main/cpp/threadFilter.cpp +++ b/ddprof-lib/src/main/cpp/threadFilter.cpp @@ -21,9 +21,11 @@ #include "threadFilter.h" #include "arch.h" +#include "counters.h" #include "nativeMem.h" #include "os.h" #include "threadLocalData.h" +#include "wallClockBlockTracker.h" #include #include #include @@ -33,7 +35,8 @@ ThreadFilter::ShardHead ThreadFilter::_free_heads[ThreadFilter::kShardCount] {}; -ThreadFilter::ThreadFilter() : _enabled(false) { +ThreadFilter::ThreadFilter() + : _enabled(false), _registry_active(false), _track_unfiltered_wall(false) { // Initialize chunk pointers to null (lazy allocation) for (int i = 0; i < kMaxChunks; ++i) { _chunks[i].store(nullptr, std::memory_order_relaxed); @@ -54,6 +57,9 @@ ThreadFilter::ThreadFilter() : _enabled(false) { ThreadFilter::~ThreadFilter() { // Make the filter inert for any concurrent readers _enabled.store(false, std::memory_order_release); + _registry_active.store(false, std::memory_order_release); + _track_unfiltered_wall.store(false, std::memory_order_release); + _recording_epoch.store(0, std::memory_order_release); // Reset free-list heads and nodes first for (int s = 0; s < kShardCount; ++s) { _free_heads[s].head.store(-1, std::memory_order_relaxed); @@ -62,6 +68,12 @@ ThreadFilter::~ThreadFilter() { _free_list[i].value.store(-1, std::memory_order_relaxed); _free_list[i].next.store(-1, std::memory_order_relaxed); } + std::atomic* tid_index = _tid_index.exchange(nullptr, std::memory_order_acquire); + if (tid_index != nullptr) { + delete[] tid_index; + NativeMem::record(NM_THREAD_FILTER, + -(long long)(kTidIndexSize * sizeof(std::atomic))); + } // Publish 0 chunks to stop range scans (collect) _num_chunks.store(0, std::memory_order_release); // Detach and delete chunks. Record the decrement after the delete so the @@ -91,8 +103,9 @@ void ThreadFilter::initializeChunk(int chunk_idx) { // Allocate and initialize new chunk completely before swapping ChunkStorage* new_chunk = new ChunkStorage(); for (auto& slot : new_chunk->slots) { - slot.value.store(-1, std::memory_order_relaxed); - slot.active_block_state.store(OSThreadState::UNKNOWN, std::memory_order_relaxed); + slot.tid.store(-1, std::memory_order_relaxed); + slot.recording_epoch.store(0, std::memory_order_relaxed); + slot.context_window_state.store(0, std::memory_order_relaxed); } // Try to install it atomically @@ -106,15 +119,63 @@ void ThreadFilter::initializeChunk(int chunk_idx) { } } -ThreadFilter::SlotID ThreadFilter::registerThread() { - // If disabled, block new registrations - if (!_enabled.load(std::memory_order_acquire)) { +ThreadFilter::SlotID ThreadFilter::registerThread(int tid) { + // A slot's identity is its owner's native tid: add() and activeSlotForId() + // validate ownership against it, so a slot without one is unusable. + if (tid < 0 || !_registry_active.load(std::memory_order_acquire)) { + return -1; + } + // Callers retry registration on every hook call while they hold no slot + // (see ensureCurrentThreadFilterSlot() in javaApi.cpp), so once the + // registry is full, threads past capacity would otherwise serialize on + // _registry_lock on every filterThreadAdd0/parkEnter0/blockEnter0 call. + // Reject them lock-free instead. A tid that is already indexed must still + // reach the locked path, which returns (and refreshes) its existing slot. + if (unlikely(capacityExhausted()) && lookupSlotIdByTid(tid) < 0) { + Counters::increment(THREAD_REGISTRY_CAPACITY_EXHAUSTED); return -1; } +#ifdef UNIT_TEST + if (_post_active_check_hook != nullptr) { + _post_active_check_hook(_post_active_check_hook_arg); + } +#endif + std::lock_guard lock(_registry_lock); + // Re-check under the lock: deactivateRecording()/init() serialize their + // admission-closing store on this same lock, so this recheck closes the + // TOCTOU window between the fast-path check above and lock acquisition. + if (!_registry_active.load(std::memory_order_acquire)) { + return -1; + } + + SlotID existing = lookupSlotIdByTid(tid); + if (existing >= 0) { + RecordingEpoch epoch = recordingEpoch(); + if (epoch != 0) { + refreshSlotForRecording(existing, slotForId(existing), epoch); + } + return existing; + } + + RecordingEpoch epoch = recordingEpoch(); // First, try to get a slot from the free list (lock-free stack) SlotID reused_slot = popFromFreeList(); if (reused_slot >= 0) { + Slot* slot = slotForId(reused_slot); + slot->lifecycle_generation.fetch_add(1, std::memory_order_acq_rel); + slot->recording_epoch.store(0, std::memory_order_relaxed); + slot->context_window_state.store(0, std::memory_order_relaxed); + if (_block_tracker != nullptr) { + _block_tracker->resetSlot(reused_slot, OSThreadState::UNKNOWN); + } + if (!indexOrRollback(*slot, reused_slot, tid)) { + pushToFreeList(reused_slot); + return -1; + } + if (epoch != 0) { + slot->recording_epoch.store(epoch, std::memory_order_release); + } return reused_slot; } @@ -123,6 +184,7 @@ ThreadFilter::SlotID ThreadFilter::registerThread() { if (index >= kMaxThreads) { // Revert the increment and return failure _next_index.fetch_sub(1, std::memory_order_relaxed); + Counters::increment(THREAD_REGISTRY_CAPACITY_EXHAUSTED); return -1; } @@ -145,9 +207,206 @@ ThreadFilter::SlotID ThreadFilter::registerThread() { // Initialize the chunk if needed initializeChunk(chunk_idx); + Slot* slot = slotForId(index); + slot->lifecycle_generation.fetch_add(1, std::memory_order_acq_rel); + slot->recording_epoch.store(0, std::memory_order_relaxed); + slot->context_window_state.store(0, std::memory_order_relaxed); + if (_block_tracker != nullptr) { + _block_tracker->resetSlot(index, OSThreadState::UNKNOWN); + } + if (!indexOrRollback(*slot, index, tid)) { + pushToFreeList(index); + return -1; + } + if (epoch != 0) { + slot->recording_epoch.store(epoch, std::memory_order_release); + } + return index; } +void ThreadFilter::refreshSlotForRecording(SlotID slot_id, Slot* slot, RecordingEpoch epoch) { + if (slot == nullptr || epoch == 0 || slot->recordingEpoch() == epoch) { + return; + } + + // Make the retained identity ineligible before resetting its recording-local + // payload, then publish the new epoch only after the reset is complete. + slot->recording_epoch.store(0, std::memory_order_release); + + // add()/remove() only ever act on the calling thread's own slot + // (ensureCurrentThreadFilterSlot uses current->tid()), and this method is + // only reached via registerThread() re-registering the calling thread's + // own tid, so no other thread can be transitioning this slot's context + // window concurrently. The CAS (instead of a plain store) is defensive: + // it detects rather than silently clobbers a concurrent transition if + // that invariant is ever broken. + u64 current = slot->context_window_state.load(std::memory_order_acquire); + while (current != 0 && + !slot->context_window_state.compare_exchange_weak( + current, 0, std::memory_order_acq_rel, std::memory_order_acquire)) { + Counters::increment(THREAD_REGISTRY_CONTEXT_RESET_RACE_DETECTED); + } + + if (_block_tracker != nullptr) { + _block_tracker->resetSlot(slot_id, OSThreadState::UNKNOWN); + } + slot->recording_epoch.store(epoch, std::memory_order_release); +} + +// _tid_index is an open-addressing hash table mapping tid -> slot_id, used to +// dedupe registrations and support fast tid -> slot lookups (lookupSlotIdByTid). +// Each entry is one of: +// 0 empty — either never used, or reclaimed from a tombstone by +// unindexSlot() once provably safe (see below) +// -1 tombstone (previously occupied, now deleted; probing must +// continue past it, since a live entry may have landed +// further along the same probe chain while this slot was +// still occupied) +// slot_id + 1 occupied (offset by 1 so slot_id 0 doesn't collide with the +// "empty" sentinel) +// Invariant: a slot holds 0 only when no live entry's probe sequence can ever +// need to continue past it. indexSlot/unindexSlot/lookupSlotIdByTid all +// linearly probe from hashTid(tid) & kTidIndexMask, wrapping around the +// kTidIndexSize-slot table, but stop differently: indexSlot inserts at the +// first free slot it meets (value <= 0, i.e. empty or tombstone — it doesn't +// need to search past a tombstone, since callers already deduped via +// lookupSlotIdByTid under the same lock); lookupSlotIdByTid/unindexSlot must +// instead treat tombstones as transparent and keep probing past them, +// stopping only at a true empty slot (value == 0, by the invariant above) or +// a match. +bool ThreadFilter::indexSlot(SlotID slot_id, int tid) { + std::atomic* tid_index = _tid_index.load(std::memory_order_acquire); + if (tid_index == nullptr) return false; + unsigned start = hashTid(tid) & kTidIndexMask; + for (int probe = 0; probe < kTidIndexSize; ++probe) { + int index = (start + probe) & kTidIndexMask; + int value = tid_index[index].load(std::memory_order_acquire); + if (value <= 0) { + tid_index[index].store(slot_id + 1, std::memory_order_release); + return true; + } + if (value > 0) { + Slot* slot = slotForId(value - 1); + if (slot != nullptr && slot->nativeTid() == tid) { + return value - 1 == slot_id; + } + } + } + return false; +} + +void ThreadFilter::unindexSlot(SlotID slot_id, int tid) { + std::atomic* tid_index = _tid_index.load(std::memory_order_acquire); + if (tid < 0 || tid_index == nullptr) return; + unsigned start = hashTid(tid) & kTidIndexMask; + for (int probe = 0; probe < kTidIndexSize; ++probe) { + int index = (start + probe) & kTidIndexMask; + int value = tid_index[index].load(std::memory_order_acquire); + if (value == 0) return; + if (value == slot_id + 1) { + // Deleting `index`: by the invariant above, `next == 0` already + // means no live entry needs to probe past `next` — and therefore + // none needs to probe past `index` either — so it's safe to clear + // `index` straight to 0 instead of leaving a tombstone (-1). + // Otherwise `next` is occupied or itself a tombstone, so some + // entry may still rely on probing through `index` to reach it; + // `index` must stay a tombstone (-1) in that case. + int next = (index + 1) & kTidIndexMask; + int replacement = + tid_index[next].load(std::memory_order_acquire) == 0 ? 0 : -1; + tid_index[index].store(replacement, std::memory_order_release); + if (replacement == 0) { + // `index` is now 0, satisfying the invariant for it. Walk + // backward and reclaim any run of tombstones (-1) immediately + // preceding it into 0 too: each such tombstone only needed to + // stay non-zero to let probing reach `index` (or beyond), and + // that's no longer required now that `index` itself is 0. + // This keeps probe chains from growing unboundedly long as + // tids churn. + int previous = (index - 1) & kTidIndexMask; + while (tid_index[previous].load(std::memory_order_acquire) == -1) { + tid_index[previous].store(0, std::memory_order_release); + previous = (previous - 1) & kTidIndexMask; + } + } + return; + } + } +} + +void ThreadFilter::rollbackFailedIndex(Slot& slot) { + slot.tid.store(-1, std::memory_order_release); + Counters::increment(THREAD_REGISTRY_INDEX_FAILURES); +} + +bool ThreadFilter::indexOrRollback(Slot& slot, SlotID slot_id, int tid) { + slot.tid.store(tid, std::memory_order_release); + if (!indexSlot(slot_id, tid)) { + rollbackFailedIndex(slot); + return false; + } + return true; +} + +ThreadFilter::SlotID ThreadFilter::lookupSlotIdByTid(int tid) const { + std::atomic* tid_index = _tid_index.load(std::memory_order_acquire); + if (tid < 0 || tid_index == nullptr) return -1; + unsigned start = hashTid(tid) & kTidIndexMask; + for (int probe = 0; probe < kTidIndexSize; ++probe) { + int index = (start + probe) & kTidIndexMask; + int value = tid_index[index].load(std::memory_order_acquire); + if (value == 0) return -1; + if (value > 0) { + Slot* slot = slotForId(value - 1); + if (slot != nullptr && slot->nativeTid() == tid) { + return value - 1; + } + } + } + return -1; +} + +ThreadFilter::Slot* ThreadFilter::lookupByTid(int tid, SlotID* out_slot_id) const { + SlotID slot_id = lookupSlotIdByTid(tid); + if (out_slot_id != nullptr) { + *out_slot_id = slot_id; + } + return slot_id < 0 ? nullptr : slotForId(slot_id); +} + +ThreadFilter::Slot* ThreadFilter::lookupByTid(int tid, RecordingEpoch epoch, + SlotID* out_slot_id) const { + if (out_slot_id != nullptr) { + *out_slot_id = -1; + } + if (epoch == 0 || recordingEpoch() != epoch) { + return nullptr; + } + SlotID slot_id = -1; + Slot* slot = lookupByTid(tid, &slot_id); + if (slot == nullptr || slot->recordingEpoch() != epoch) { + return nullptr; + } + if (out_slot_id != nullptr) { + *out_slot_id = slot_id; + } + return slot; +} + +ThreadFilter::Slot* ThreadFilter::activeSlotForId(SlotID slot_id, + int tid) const { + Slot* slot = slotForId(slot_id); + if (slot == nullptr || slot->nativeTid() != tid) { + return nullptr; + } + RecordingEpoch epoch = recordingEpoch(); + if (epoch != 0 && slot->recordingEpoch() != epoch) { + return nullptr; + } + return slot; +} + void ThreadFilter::initFreeList() { // Initialize the free list storage for (int i = 0; i < kFreeListSize; ++i) { @@ -159,6 +418,7 @@ void ThreadFilter::initFreeList() { for (int s = 0; s < kShardCount; ++s) { _free_heads[s].head.store(-1, std::memory_order_relaxed); } + _free_count.store(0, std::memory_order_release); } bool ThreadFilter::accept(SlotID slot_id) const { @@ -174,15 +434,15 @@ bool ThreadFilter::accept(SlotID slot_id) const { // This is not a fast path like the add operation. ChunkStorage* chunk = _chunks[chunk_idx].load(std::memory_order_acquire); if (likely(chunk != nullptr)) { - return chunk->slots[slot_idx].value.load(std::memory_order_relaxed) != -1; + return chunk->slots[slot_idx].inContextWindow(); } return false; } -void ThreadFilter::add(int tid, SlotID slot_id) { +bool ThreadFilter::add(int tid, SlotID slot_id) { // PRECONDITION: slot_id must be from registerThread() or negative // Undefined behavior for invalid positive slot_ids (performance optimization) - if (slot_id < 0) return; + if (slot_id < 0) return false; int chunk_idx = slot_id >> kChunkShift; int slot_idx = slot_id & kChunkMask; @@ -190,8 +450,32 @@ void ThreadFilter::add(int tid, SlotID slot_id) { // Fast path: assume valid slot_id from registerThread() ChunkStorage* chunk = _chunks[chunk_idx].load(std::memory_order_acquire); if (likely(chunk != nullptr)) { - chunk->slots[slot_idx].value.store(tid, std::memory_order_release); + Slot& slot = chunk->slots[slot_idx]; + // registerThread() publishes the owner's tid before returning the slot, + // so a mismatch means the caller's cached slot_id went stale: an + // unfiltered init() reset the registry (clearing the tid) and the slot + // may since have been handed to another thread. Never write to it - + // enterContextWindow() relies on the owning thread being the only + // writer - and never re-publish the tid here, since after a reset the + // allocator considers this slot free and will hand it out again. + // The caller clears its cached slot_id and re-registers. + // + // A reset landing between this check and the store below can still let + // one stray transition through. That only happens in unfiltered mode + // (context-filter recordings never reset registrations), where context + // membership only gates owned-block suppression, and it only errs + // towards more samples: if the slot is still unassigned, + // registerThread() zeroes context_window_state before handing it out; + // if it already has a new owner, that owner is treated as in-context + // and the epoch bump disqualifies its in-flight owned block run, both + // of which disable suppression rather than enable it. + if (unlikely(tid < 0 || slot.nativeTid() != tid)) { + return false; + } + slot.enterContextWindow(); + return true; } + return false; } void ThreadFilter::remove(SlotID slot_id) { @@ -212,16 +496,87 @@ void ThreadFilter::remove(SlotID slot_id) { return; } - chunk->slots[slot_idx].value.store(-1, std::memory_order_release); + chunk->slots[slot_idx].exitContextWindow(); +} + +void ThreadFilter::unregisterThread(SlotID slot_id, int expected_tid) { + std::lock_guard lock(_registry_lock); + unregisterThreadLocked(slot_id, expected_tid); } -void ThreadFilter::unregisterThread(SlotID slot_id) { +void ThreadFilter::unregisterThreadLocked(SlotID slot_id, int expected_tid) { if (slot_id < 0) return; - remove(slot_id); - resetSlotRunState(slot_id); + Slot* slot = slotForId(slot_id); + if (slot == nullptr) return; + int tid = slot->nativeTid(); + if (expected_tid >= 0 && tid != expected_tid) return; + unindexSlot(slot_id, tid); + slot->recording_epoch.store(0, std::memory_order_release); + slot->tid.store(-1, std::memory_order_release); + slot->context_window_state.store(0, std::memory_order_release); + if (_block_tracker != nullptr) { + _block_tracker->resetSlot(slot_id, OSThreadState::UNKNOWN); + } pushToFreeList(slot_id); } +void ThreadFilter::unregisterThreadByTid(int tid) { + // Lock-free pre-check: avoids the mutex for the common case where this tid + // was never registered (e.g. the registry was never activated, so + // _tid_index is null, or registration failed because the registry was + // full). A hit here is always re-confirmed under the lock before anything + // is mutated. + if (lookupSlotIdByTid(tid) < 0) { + return; + } + std::lock_guard lock(_registry_lock); + SlotID slot_id = lookupSlotIdByTid(tid); + if (slot_id >= 0) { + unregisterThreadLocked(slot_id); + } +} + +void ThreadFilter::resetRegistrationsLocked() { + int num_chunks = _num_chunks.load(std::memory_order_acquire); + for (int chunk_idx = 0; chunk_idx < num_chunks; ++chunk_idx) { + ChunkStorage* chunk = _chunks[chunk_idx].load(std::memory_order_acquire); + if (chunk == nullptr) continue; + for (int slot_idx = 0; slot_idx < kChunkSize; ++slot_idx) { + Slot& slot = chunk->slots[slot_idx]; + if (slot.nativeTid() != -1) { + slot.lifecycle_generation.fetch_add(1, std::memory_order_acq_rel); + } + slot.recording_epoch.store(0, std::memory_order_release); + slot.tid.store(-1, std::memory_order_release); + slot.context_window_state.store(0, std::memory_order_release); + } + } + if (_block_tracker != nullptr) { + _block_tracker->resetAll(); + } + std::atomic* tid_index = _tid_index.load(std::memory_order_acquire); + if (tid_index != nullptr) { + for (int i = 0; i < kTidIndexSize; ++i) { + tid_index[i].store(0, std::memory_order_relaxed); + } + } + _next_index.store(0, std::memory_order_relaxed); + initFreeList(); +} + +void ThreadFilter::ensureTidIndexLocked() { + if (_tid_index.load(std::memory_order_acquire) != nullptr) return; + std::atomic* tid_index = new std::atomic[kTidIndexSize]; + for (int i = 0; i < kTidIndexSize; ++i) { + tid_index[i].store(0, std::memory_order_relaxed); + } + // Pairs with the acquire loads in the lock-free lookups, which must see + // the zeroed entries rather than uninitialized memory. + _tid_index.store(tid_index, std::memory_order_release); + NativeMem::record(NM_THREAD_FILTER, + (long long)(kTidIndexSize * sizeof(std::atomic))); +} + bool ThreadFilter::pushToFreeList(SlotID slot_id) { // Lock-free sharded Treiber stack push const int shard = shardOfSlot(slot_id); @@ -238,6 +593,7 @@ bool ThreadFilter::pushToFreeList(SlotID slot_id) { _free_list[i].next.store(old_head, std::memory_order_relaxed); } while (!head.compare_exchange_weak(old_head, i, std::memory_order_release, std::memory_order_relaxed)); + _free_count.fetch_add(1, std::memory_order_release); return true; } } @@ -265,6 +621,7 @@ ThreadFilter::SlotID ThreadFilter::popFromFreeList() { int id = _free_list[node].value.exchange(-1, std::memory_order_relaxed); _free_list[node].next.store(-1, std::memory_order_relaxed); + _free_count.fetch_sub(1, std::memory_order_release); return id; } } @@ -288,8 +645,8 @@ void ThreadFilter::collect(std::vector& tids) const { } for (const auto& slot : chunk->slots) { - int slot_tid = slot.value.load(std::memory_order_relaxed); - if (slot_tid != -1) { + int slot_tid = slot.nativeTid(); + if (slot_tid != -1 && slot.inContextWindow()) { tids.push_back(slot_tid); } } @@ -312,10 +669,13 @@ void ThreadFilter::collect(std::vector& entries) const { continue; } - for (auto& slot : chunk->slots) { - int slot_tid = slot.value.load(std::memory_order_acquire); - if (slot_tid != -1) { - entries.push_back({slot_tid, &slot}); + for (int slot_idx = 0; slot_idx < kChunkSize; ++slot_idx) { + Slot& slot = chunk->slots[slot_idx]; + int slot_tid = slot.nativeTid(); + if (slot_tid != -1 && slot.inContextWindow()) { + SlotID slot_id = (chunk_idx << kChunkShift) | slot_idx; + entries.push_back({slot_tid, &slot, slot_id, slot.lifecycleGeneration(), + slot.recordingEpoch()}); } } } @@ -330,61 +690,84 @@ void ThreadFilter::clearActive() { } for (auto& slot : chunk->slots) { - slot.value.store(-1, std::memory_order_release); - slot.clearActiveBlockRun(OSThreadState::UNKNOWN); + slot.exitContextWindow(); } } + if (_block_tracker != nullptr) { + _block_tracker->resetAll(); + } } void ThreadFilter::resetSlotRunState(SlotID slot_id) { - if (slot_id < 0) return; - int chunk_idx = slot_id >> kChunkShift; - int slot_idx = slot_id & kChunkMask; - ChunkStorage* chunk = _chunks[chunk_idx].load(std::memory_order_acquire); - if (chunk != nullptr) { - // Clear stale suppression state so a new thread in this slot cannot inherit - // its predecessor's active block or once-per-run sampled marker. - chunk->slots[slot_idx].clearActiveBlockRun(OSThreadState::UNKNOWN); - } + if (slot_id < 0 || _block_tracker == nullptr) return; + // Clear stale suppression state so a new thread in this slot cannot inherit + // its predecessor's active block or once-per-run sampled marker. + _block_tracker->resetSlot(slot_id, OSThreadState::UNKNOWN); } -u64 ThreadFilter::enterBlockedRun(SlotID slot_id, OSThreadState state, - BlockRunOwner owner) { - Slot* s = slotForId(slot_id); - if (s != nullptr) { - u32 generation = 0; - if (!s->trySetActiveBlockRun(state, owner, &generation)) { - return 0; +void ThreadFilter::init(const char* filter, bool track_unfiltered_wall) { + // Preserve the legacy filter contract: every non-empty value, including + // "0", enables context filtering. Empty filter disables filtering; the + // extra flag only retains metadata for unfiltered wall prechecks. + bool context_filter = filter != nullptr && strlen(filter) > 0; + bool unfiltered_tracking = track_unfiltered_wall && !context_filter; + // Close registration before clearing identities from a previous unfiltered + // recording. Other threads retain only stale slot IDs; activeSlotForId() + // rejects them until their next context or owned-block hook registers lazily. + // Serialized against registerThread()'s in-lock recheck of _registry_active + // so a thread can't publish a slot after admission has been closed here. + { + std::lock_guard lock(_registry_lock); + _registry_active.store(false, std::memory_order_release); + if (unfiltered_tracking || context_filter) { + ensureTidIndexLocked(); + } + if (unfiltered_tracking) { + resetRegistrationsLocked(); + } + } + RecordingEpoch epoch = 0; + if (unfiltered_tracking) { + epoch = _next_recording_epoch.fetch_add(1, std::memory_order_acq_rel) + 1; + if (epoch == 0) { + // Zero is reserved for inactive state. This can occur only after + // 2^64 recording starts; skip it rather than publishing ambiguity. + epoch = _next_recording_epoch.fetch_add(1, std::memory_order_acq_rel) + 1; } - return encodeBlockRunToken(slot_id, generation); } - return 0; + _recording_epoch.store(epoch, std::memory_order_release); + _track_unfiltered_wall.store(unfiltered_tracking, + std::memory_order_release); + _registry_active.store(unfiltered_tracking || context_filter, + std::memory_order_release); + _enabled.store(context_filter, std::memory_order_release); } -void ThreadFilter::exitBlockedRun(SlotID slot_id) { - Slot* s = slotForId(slot_id); - if (s != nullptr) { - s->clearActiveBlockRun(OSThreadState::RUNNABLE); - } +bool ThreadFilter::enabled() const { + return _enabled.load(std::memory_order_acquire); } -bool ThreadFilter::exitBlockedRun(SlotID slot_id, u32 generation) { - Slot* s = slotForId(slot_id); - if (s == nullptr || generation == 0 || s->blockGeneration() != generation) { - return false; - } - s->clearActiveBlockRun(OSThreadState::RUNNABLE); - return true; +bool ThreadFilter::registryActive() const { + return _registry_active.load(std::memory_order_acquire); } -void ThreadFilter::init(const char* filter) { - // Simple logic: any filter value (including "0") enables filtering - // Only explicitly registered threads via addThread() will be sampled - // Previously we had a syntax where we could manually force some thread IDs. - // This is no longer supported. - _enabled.store(filter != nullptr && strlen(filter) > 0, std::memory_order_release); +bool ThreadFilter::unfilteredWallTrackingActive() const { + return _track_unfiltered_wall.load(std::memory_order_acquire); } -bool ThreadFilter::enabled() const { - return _enabled.load(std::memory_order_acquire); +ThreadFilter::RecordingEpoch ThreadFilter::recordingEpoch() const { + return _recording_epoch.load(std::memory_order_acquire); +} + +void ThreadFilter::deactivateRecording() { + // Close producer admission before invalidating the recording epoch. + // Context-filtered recordings may retain slots; a new unfiltered recording + // resets them before reopening admission. Serialize against + // registerThread()'s in-lock recheck of _registry_active so a thread can't + // publish a slot after this admission-closing store has been observed. + std::lock_guard lock(_registry_lock); + _registry_active.store(false, std::memory_order_release); + _enabled.store(false, std::memory_order_release); + _track_unfiltered_wall.store(false, std::memory_order_release); + _recording_epoch.store(0, std::memory_order_release); } diff --git a/ddprof-lib/src/main/cpp/threadFilter.h b/ddprof-lib/src/main/cpp/threadFilter.h index 541249e4c1..e539fee182 100644 --- a/ddprof-lib/src/main/cpp/threadFilter.h +++ b/ddprof-lib/src/main/cpp/threadFilter.h @@ -21,22 +21,18 @@ #include #include #include +#include #include "arch.h" #include "threadState.h" struct ThreadEntry; // defined after ThreadFilter; carries a pointer to a ThreadFilter::Slot - -enum class BlockRunOwner : int { - NONE = 0, - JAVA = 1, - JVMTI = 2, - NATIVE = 3, -}; +class WallClockBlockTracker; // wall-clock owned/unowned block-run suppression state, see wallClockBlockTracker.h class ThreadFilter { public: using SlotID = int; + using RecordingEpoch = u64; // Optimized limits for reasonable memory usage static constexpr int kChunkSize = 256; @@ -45,168 +41,132 @@ class ThreadFilter { static constexpr int kMaxThreads = 2048; static constexpr int kMaxChunks = (kMaxThreads + kChunkSize - 1) / kChunkSize; // = 8 chunks // High-performance free list using Treiber stack, 64 shards - static constexpr int kFreeListSize = 1024; // power-of-two for fast modulo + static constexpr int kFreeListSize = kMaxThreads; static constexpr int kShardCount = 64; // power-of-two for fast modulo + static constexpr int kTidIndexSize = 8192; // 4x maximum live slots + static constexpr int kTidIndexMask = kTidIndexSize - 1; // One cache line per slot to avoid false sharing. Slot instances are never freed // (ChunkStorage is process-lifetime), so a captured Slot* is always dereferenceable. struct alignas(DEFAULT_CACHE_LINE_SIZE) Slot { - static constexpr u64 kUnownedBlockedFallbackRatio = 10; - - std::atomic unowned_blocked_pending_weight{0}; - std::atomic unowned_blocked_decision_count{0}; - std::atomic unowned_blocked_call_trace_id{0}; - std::atomic unowned_blocked_state{OSThreadState::UNKNOWN}; - std::atomic value{-1}; - std::atomic active_block_owner{static_cast(BlockRunOwner::NONE)}; - std::atomic block_generation{0}; - // Wall-clock once-per-run suppression state. The signal handler records the - // last sampled blocked state; the signal handler and timer thread read it to - // suppress duplicate samples, while lifecycle/block-exit paths reset it. - // Release/acquire on sampled_this_run pairs with relaxed last_sampled_state, - // following the standard flag+payload pattern. - std::atomic last_sampled_state{OSThreadState::UNKNOWN}; // 4 bytes - // Set by explicit block enter/exit hooks. It lets the timer skip sending a signal - // only while instrumentation still owns a suppressible blocking interval. - std::atomic active_block_state{OSThreadState::UNKNOWN}; - std::atomic sampled_this_run{false}; - char padding[2 * DEFAULT_CACHE_LINE_SIZE + // Packed as (epoch << 1) | in_context_window so a transition and its + // epoch change are observed atomically by owned-block admission and by + // the wall-clock suppression check. + std::atomic context_window_state{0}; + std::atomic lifecycle_generation{0}; + // Per-recording publication flag. A retained TID mapping is eligible for + // unfiltered suppression only when this value matches the registry's + // active recording epoch. The slot's context-window state and its + // WallClockBlockTracker::BlockState are reset before the epoch is + // release-published. + std::atomic recording_epoch{0}; + // Native identity and context-window membership are independent so an + // unfiltered wall recording can retain lifecycle metadata without + // changing ordinary thread selection. + std::atomic tid{-1}; + char padding[DEFAULT_CACHE_LINE_SIZE - sizeof(std::atomic) - sizeof(std::atomic) - sizeof(std::atomic) - - sizeof(std::atomic) - - sizeof(std::atomic) - - sizeof(std::atomic) - - sizeof(std::atomic) - - sizeof(std::atomic) - - sizeof(std::atomic) - - sizeof(std::atomic)]; + - sizeof(std::atomic)]; - inline bool sampledThisRun() const { - return sampled_this_run.load(std::memory_order_acquire); - } - inline OSThreadState lastSampledState() const { - return last_sampled_state.load(std::memory_order_relaxed); - } - inline void markSampledThisRun(OSThreadState state) { - last_sampled_state.store(state, std::memory_order_relaxed); - sampled_this_run.store(true, std::memory_order_release); - } - inline void resetSampledRun(OSThreadState state) { - resetUnownedBlockedSampling(); - last_sampled_state.store(state, std::memory_order_relaxed); - sampled_this_run.store(false, std::memory_order_release); - } - inline OSThreadState activeBlockState() const { - return active_block_state.load(std::memory_order_acquire); - } - inline void setActiveBlockState(OSThreadState state) { - active_block_state.store(state, std::memory_order_release); - } - inline BlockRunOwner activeBlockOwner() const { - return static_cast(active_block_owner.load(std::memory_order_acquire)); + inline int nativeTid() const { + return tid.load(std::memory_order_acquire); } - inline u32 blockGeneration() const { - return block_generation.load(std::memory_order_acquire); + inline u64 lifecycleGeneration() const { + return lifecycle_generation.load(std::memory_order_acquire); } - inline void resetUnownedBlockedSampling() { - unowned_blocked_pending_weight.store(0, std::memory_order_relaxed); - unowned_blocked_decision_count.store(0, std::memory_order_relaxed); - unowned_blocked_state.store(OSThreadState::UNKNOWN, std::memory_order_relaxed); - unowned_blocked_call_trace_id.store(0, std::memory_order_release); + inline RecordingEpoch recordingEpoch() const { + return recording_epoch.load(std::memory_order_acquire); } - inline bool shouldRecordUnownedBlockedSample() { - u64 decision = unowned_blocked_decision_count.fetch_add(1, std::memory_order_relaxed) + 1; - if ((decision % kUnownedBlockedFallbackRatio) == 1) { - return true; - } - unowned_blocked_pending_weight.fetch_add(1, std::memory_order_relaxed); - return false; + inline bool inContextWindow() const { + return (context_window_state.load(std::memory_order_acquire) & 1) != 0; } - inline u64 consumeUnownedBlockedWeight() { - return unowned_blocked_pending_weight.exchange(0, std::memory_order_relaxed) + 1; + inline u64 contextWindowEpoch() const { + return context_window_state.load(std::memory_order_acquire) >> 1; } - inline void restoreUnownedBlockedWeight(u64 weight) { - if (weight > 1) { - unowned_blocked_pending_weight.fetch_add(weight - 1, std::memory_order_relaxed); - } - } - inline void recordUnownedBlockedSample(u64 call_trace_id, OSThreadState state) { - unowned_blocked_state.store(state, std::memory_order_relaxed); - unowned_blocked_call_trace_id.store(call_trace_id, std::memory_order_release); - } - inline bool flushUnownedBlockedTail(u64& call_trace_id, u64& weight, - OSThreadState& state) { - call_trace_id = unowned_blocked_call_trace_id.exchange(0, std::memory_order_acq_rel); - weight = unowned_blocked_pending_weight.exchange(0, std::memory_order_relaxed); - state = unowned_blocked_state.exchange(OSThreadState::UNKNOWN, std::memory_order_relaxed); - unowned_blocked_decision_count.store(0, std::memory_order_relaxed); - if (call_trace_id == 0 || weight == 0 || state == OSThreadState::UNKNOWN) { - return false; - } + // add()/remove() only ever call these on the calling thread's own slot, + // so on that path there is no writer-writer race and the release store + // is sufficient for concurrent readers using the acquire loads in + // inContextWindow()/contextWindowEpoch(). Avoiding a CAS turns a locked + // RMW into a plain store on every context-filtered enter/exit. + // + // Other threads can also write this field: + // - unregisterThreadLocked()/resetRegistrationsLocked() zero it while + // tearing down or resetting the slot entirely. + // - clearActive() calls exitContextWindow() from the thread running + // Profiler::start(). Racing an owner's exit+enter, its store can move + // the epoch backwards. This is benign: clearActive() only runs in + // context-filter mode, and the epoch is only compared in unfiltered + // mode (WallClockBlockTracker's outside-context checks). + inline bool enterContextWindow() { + u64 current = context_window_state.load(std::memory_order_relaxed); + if ((current & 1) != 0) return false; + context_window_state.store(current + 3, std::memory_order_release); return true; } - inline bool trySetActiveBlockRun(OSThreadState state, BlockRunOwner owner, - u32* generation_out) { - int expected_owner = static_cast(BlockRunOwner::NONE); - if (!active_block_owner.compare_exchange_strong( - expected_owner, static_cast(owner), std::memory_order_acq_rel, - std::memory_order_acquire)) { - return false; - } - u32 generation = block_generation.fetch_add(1, std::memory_order_acq_rel) + 1; - resetUnownedBlockedSampling(); - last_sampled_state.store(OSThreadState::UNKNOWN, std::memory_order_relaxed); - sampled_this_run.store(false, std::memory_order_relaxed); - active_block_state.store(state, std::memory_order_release); - *generation_out = generation; + inline bool exitContextWindow() { + u64 current = context_window_state.load(std::memory_order_relaxed); + if ((current & 1) == 0) return false; + context_window_state.store(current + 1, std::memory_order_release); return true; } - inline void clearActiveBlockRun(OSThreadState state) { - active_block_state.store(OSThreadState::UNKNOWN, std::memory_order_release); - resetSampledRun(state); - active_block_owner.store(static_cast(BlockRunOwner::NONE), std::memory_order_release); + // Exposes the raw packed context-window state to WallClockBlockTracker, + // which reads it (twice, around its owner CAS) to validate that an owned + // block run did not span a context-window transition. + inline u64 rawContextWindowState() const { + return context_window_state.load(std::memory_order_acquire); } }; - static_assert(sizeof(Slot) == 2 * DEFAULT_CACHE_LINE_SIZE, "Slot must be exactly two cache lines"); - static_assert(std::atomic::is_always_lock_free, - "Slot OSThreadState fields must be lock-free for signal-handler safety"); - static_assert(std::atomic::is_always_lock_free, - "Slot::sampled_this_run must be lock-free for signal-handler safety"); + static_assert(sizeof(Slot) == DEFAULT_CACHE_LINE_SIZE, "Slot must fit exactly one cache line"); + static_assert(std::atomic::is_always_lock_free, + "Slot::recording_epoch must be lock-free for signal-handler safety"); ThreadFilter(); ~ThreadFilter(); - void init(const char* filter); + void init(const char* filter, bool track_unfiltered_wall = false); void initFreeList(); bool enabled() const; - // Hot path methods - slot_id MUST be from registerThread(), undefined behavior otherwise + bool registryActive() const; + bool unfilteredWallTrackingActive() const; + RecordingEpoch recordingEpoch() const; + // Hot path methods - slot_id MUST be from registerThread(), undefined behavior otherwise. + // add() is lock-free. It returns false, without touching the slot, when the slot no + // longer belongs to `tid` (its cached slot id went stale, e.g. an unfiltered + // recording restart reset the registry); callers must not assume membership on + // failure and should clear any cached slot id so registerThread() re-derives it. bool accept(SlotID slot_id) const; - void add(int tid, SlotID slot_id); + bool add(int tid, SlotID slot_id); void remove(SlotID slot_id); void collect(std::vector& tids) const; void collect(std::vector& entries) const; // Clears per-recording membership and suppression state while keeping // process-lifetime slot ownership intact. Threads must opt in again with add(). void clearActive(); + // Forwards to WallClockBlockTracker::resetSlot(). Not called from + // production code (registry-lifecycle paths call the tracker directly); + // only park_state_ut uses it. void resetSlotRunState(SlotID slot_id); - u64 enterBlockedRun(SlotID slot_id, OSThreadState state, - BlockRunOwner owner = BlockRunOwner::JAVA); - // Unconditional cleanup for reset/unregister paths only. Normal block - // lifecycles must use the generation-checked overload so they cannot clear - // another owner. - void exitBlockedRun(SlotID slot_id); - bool exitBlockedRun(SlotID slot_id, u32 generation); - static inline u64 encodeBlockRunToken(SlotID slot_id, u32 generation) { - return (static_cast(generation) << 32) | static_cast(slot_id + 1); - } - static inline SlotID tokenSlotId(u64 token) { - return static_cast(static_cast(token) - 1); - } - static inline u32 tokenGeneration(u64 token) { - return static_cast(token >> 32); + // Non-owning; wired up once by Profiler so ThreadFilter's own + // registry-lifecycle resets (registerThread/unregisterThread/ + // resetRegistrationsLocked/clearActive) also clear the tracker's + // parallel per-slot state. See wallClockBlockTracker.h. + void setBlockTracker(WallClockBlockTracker* tracker) { _block_tracker = tracker; } + +#ifdef UNIT_TEST + // Invoked by registerThread() immediately before _registry_lock is + // acquired, once the pre-lock _registry_active and capacity checks have + // passed - lets tests inject a deactivation between the lock-free + // _registry_active check and the in-lock recheck, and observe whether a + // call reached the lock at all. + using PostActiveCheckHook = void (*)(void*); + void setPostActiveCheckHookForTest(PostActiveCheckHook hook, void* arg) { + _post_active_check_hook = hook; + _post_active_check_hook_arg = arg; } +#endif // Returns nullptr if slot_id is invalid or its chunk has not been allocated. inline Slot* slotForId(SlotID slot_id) const { @@ -218,8 +178,15 @@ class ThreadFilter { return chunk != nullptr ? &chunk->slots[slot_idx] : nullptr; } - SlotID registerThread(); - void unregisterThread(SlotID slot_id); + // Returns the slot owned by native thread `tid` (allocating one if needed), + // or -1 if tid < 0, the registry is inactive, or it is full. + SlotID registerThread(int tid); + void unregisterThread(SlotID slot_id, int expected_tid = -1); + void unregisterThreadByTid(int tid); + Slot* lookupByTid(int tid, SlotID* out_slot_id = nullptr) const; + Slot* lookupByTid(int tid, RecordingEpoch epoch, SlotID* out_slot_id = nullptr) const; + Slot* activeSlotForId(SlotID slot_id, int tid) const; + void deactivateRecording(); private: @@ -235,6 +202,10 @@ class ThreadFilter { }; std::atomic _enabled{false}; + std::atomic _registry_active{false}; + std::atomic _track_unfiltered_wall{false}; + std::atomic _recording_epoch{0}; + std::atomic _next_recording_epoch{0}; // Lazily allocated storage for chunks std::atomic _chunks[kMaxChunks]; @@ -243,6 +214,31 @@ class ThreadFilter { // Lock-free slot allocation std::atomic _next_index{0}; std::unique_ptr _free_list; + // Number of slots currently linked into the free list. Maintained only by + // pushToFreeList()/popFromFreeList()/initFreeList(), all of which run under + // _registry_lock (or single-threaded construction), so it is exact under + // the lock and a hint outside it. Together with _next_index it lets + // registerThread() reject registrations against a full registry without + // taking _registry_lock (see capacityExhausted()). + std::atomic _free_count{0}; + // Entries contain slot_id + 1. Zero terminates a lookup probe; -1 is a + // tombstone left by unregister. The slot's published TID is the key. + // Allocated (kTidIndexSize entries) by the first init() that activates the + // registry and kept until destruction, so processes that never use a + // context filter or unfiltered precheck don't pay for it. Null until then: + // lookups find nothing, and nothing can be indexed while it is null because + // registration requires an active registry. + std::atomic*> _tid_index{nullptr}; + // Registration and teardown never run in a signal handler. Serializing + // writers prevents duplicate TID mappings while lookups remain lock-free. + std::mutex _registry_lock; + + WallClockBlockTracker* _block_tracker = nullptr; + +#ifdef UNIT_TEST + PostActiveCheckHook _post_active_check_hook = nullptr; + void* _post_active_check_hook_arg = nullptr; +#endif // Cache line aligned to prevent false sharing between shards struct alignas(DEFAULT_CACHE_LINE_SIZE) ShardHead { std::atomic head{-1}; }; @@ -254,12 +250,33 @@ class ThreadFilter { void initializeChunk(int chunk_idx); bool pushToFreeList(SlotID slot_id); SlotID popFromFreeList(); + // Lock-free hint: true when every slot index has been handed out and none + // is waiting in the free list, i.e. a new registration cannot succeed. + inline bool capacityExhausted() const { + return _next_index.load(std::memory_order_acquire) >= kMaxThreads && + _free_count.load(std::memory_order_acquire) == 0; + } + bool indexSlot(SlotID slot_id, int tid); + void unindexSlot(SlotID slot_id, int tid); + void rollbackFailedIndex(Slot& slot); + bool indexOrRollback(Slot& slot, SlotID slot_id, int tid); + void refreshSlotForRecording(SlotID slot_id, Slot* slot, RecordingEpoch epoch); + void resetRegistrationsLocked(); + void ensureTidIndexLocked(); + void unregisterThreadLocked(SlotID slot_id, int expected_tid = -1); + SlotID lookupSlotIdByTid(int tid) const; + static inline unsigned hashTid(int tid) { + return static_cast(tid) * 2654435761u; + } }; // Snapshot entry produced by ThreadFilter::collect for the wall-clock timer. struct ThreadEntry { int tid; ThreadFilter::Slot* slot; + ThreadFilter::SlotID slot_id; + u64 lifecycle_generation; + ThreadFilter::RecordingEpoch recording_epoch; }; #endif // _THREADFILTER_H diff --git a/ddprof-lib/src/main/cpp/threadState.h b/ddprof-lib/src/main/cpp/threadState.h index 786c96fb6f..cfa46edec3 100644 --- a/ddprof-lib/src/main/cpp/threadState.h +++ b/ddprof-lib/src/main/cpp/threadState.h @@ -32,4 +32,13 @@ enum class ExecutionMode : int { inline ExecutionMode getThreadExecutionMode(); inline OSThreadState getOSThreadState(); +// Shared by BaseWallClock's precheck path and WallClockBlockTracker::shouldSuppressOwnedBlock(): +// both need the same set of blocked/waiting states eligible for once-per-run suppression. +inline bool isPrecheckSuppressionState(OSThreadState state) { + return state == OSThreadState::SLEEPING || + state == OSThreadState::CONDVAR_WAIT || + state == OSThreadState::OBJECT_WAIT || + state == OSThreadState::MONITOR_WAIT; +} + #endif // _THREADSTATE_H diff --git a/ddprof-lib/src/main/cpp/wallClock.cpp b/ddprof-lib/src/main/cpp/wallClock.cpp index 742553bee0..0c9348de92 100644 --- a/ddprof-lib/src/main/cpp/wallClock.cpp +++ b/ddprof-lib/src/main/cpp/wallClock.cpp @@ -28,13 +28,9 @@ #include // For std::sort and std::binary_search std::atomic BaseWallClock::_enabled{false}; - -static inline bool isPrecheckSuppressionState(OSThreadState state) { - return state == OSThreadState::SLEEPING || - state == OSThreadState::CONDVAR_WAIT || - state == OSThreadState::OBJECT_WAIT || - state == OSThreadState::MONITOR_WAIT; -} +#if defined(UNIT_TEST) || defined(DEBUG) +std::atomic BaseWallClock::_force_start_failure_for_test{false}; +#endif static inline u64 loadSpanId(OtelThreadContextRecord* record) { u64 span_id = 0; @@ -59,11 +55,11 @@ static inline bool hasKnownActiveTraceContext(ProfiledThread* thread) { struct WallPrecheckResult { bool suppress = false; - ThreadFilter::Slot* slot_to_arm = nullptr; + WallClockBlockTracker::BlockState* slot_to_arm = nullptr; OSThreadState state_to_arm = OSThreadState::UNKNOWN; OSThreadState observed_state = OSThreadState::UNKNOWN; bool observed_state_valid = false; - ThreadFilter::Slot* unowned_weight_slot = nullptr; + WallClockBlockTracker::BlockState* unowned_weight_slot = nullptr; u64 unowned_weight = 1; bool flush_unowned_tail = false; u64 flush_call_trace_id = 0; @@ -76,19 +72,14 @@ static inline void incrementSuppressedSampledRun() { WallClockCounters::incrementSuppressedSampledRun(); } -static inline bool suppressAlreadySampledBlock(ThreadFilter::Slot* slot) { - if (slot == nullptr) { +bool BaseWallClock::suppressAlreadySampled(const ThreadEntry& entry) { + ThreadFilter* thread_filter = Profiler::instance()->threadFilter(); + WallClockBlockTracker* tracker = Profiler::instance()->blockTracker(); + if (!tracker->shouldSuppressOwnedBlock(thread_filter, entry)) { return false; } - OSThreadState block_state = slot->activeBlockState(); - if (slot->activeBlockOwner() != BlockRunOwner::NONE && - isPrecheckSuppressionState(block_state) && - slot->sampledThisRun() && - block_state == slot->lastSampledState()) { - incrementSuppressedSampledRun(); - return true; - } - return false; + incrementSuppressedSampledRun(); + return true; } static inline WallPrecheckResult prepareWallPrecheck(ProfiledThread* current, @@ -98,20 +89,37 @@ static inline WallPrecheckResult prepareWallPrecheck(ProfiledThread* current, return result; } - ThreadFilter::Slot* slot = - Profiler::instance()->threadFilter()->slotForId(current->filterSlotId()); + ThreadFilter* registry = Profiler::instance()->threadFilter(); + ThreadFilter::SlotID slot_id = current->filterSlotId(); + ThreadFilter::Slot* slot = registry->activeSlotForId(slot_id, current->tid()); + if (slot == nullptr) { return result; } - OSThreadState active_block_state = slot->activeBlockState(); - BlockRunOwner active_block_owner = slot->activeBlockOwner(); + WallClockBlockTracker* tracker = Profiler::instance()->blockTracker(); + WallClockBlockTracker::BlockState* block_slot = tracker->slotForId(slot_id); + if (block_slot == nullptr) { + return result; + } + + // In an unfiltered recording, context threads keep their normal MethodSample + // stream. Only owned blocks that remain outside the context window may replace + // repeated signals. + if (registry->unfilteredWallTrackingActive() && slot->inContextWindow()) { + return result; + } + + OSThreadState active_block_state = block_slot->activeBlockState(); + BlockRunOwner active_block_owner = block_slot->activeBlockOwner(); bool has_owned_block = active_block_owner != BlockRunOwner::NONE && - isPrecheckSuppressionState(active_block_state); + isPrecheckSuppressionState(active_block_state) && + (!registry->unfilteredWallTrackingActive() || + block_slot->activeBlockRemainedOutsideContextWindow(slot)); if (has_owned_block) { - if (slot->sampledThisRun() && - active_block_state == slot->lastSampledState()) { + if (block_slot->sampledThisRun() && + active_block_state == block_slot->lastSampledState()) { incrementSuppressedSampledRun(); result.suppress = true; return result; @@ -119,23 +127,30 @@ static inline WallPrecheckResult prepareWallPrecheck(ProfiledThread* current, // Arm only after the MethodSample has been successfully recorded. If the // JFR write is skipped due to lock contention, the next signal must retry // instead of losing the only stack for this blocked run. - result.slot_to_arm = slot; + result.slot_to_arm = block_slot; result.state_to_arm = active_block_state; return result; } + // Unfiltered tracking exists only to support explicit context and owned-block + // hooks. Keep unowned observations on ordinary per-signal sampling: the JVMTI + // path has no call_trace_id with which to replay a suppressed tail. + if (registry->unfilteredWallTrackingActive()) { + return result; + } + result.observed_state = getOSThreadState(); result.observed_state_valid = true; if (isPrecheckSuppressionState(result.observed_state)) { - if (!slot->shouldRecordUnownedBlockedSample()) { + if (!block_slot->shouldRecordUnownedBlockedSample()) { Counters::increment(WC_UNOWNED_BLOCKED_SUPPRESSED); result.suppress = true; return result; } - result.unowned_weight_slot = slot; - result.unowned_weight = slot->consumeUnownedBlockedWeight(); + result.unowned_weight_slot = block_slot; + result.unowned_weight = block_slot->consumeUnownedBlockedWeight(); } else { - result.flush_unowned_tail = slot->flushUnownedBlockedTail( + result.flush_unowned_tail = block_slot->flushUnownedBlockedTail( result.flush_call_trace_id, result.flush_weight, result.flush_state); } return result; @@ -306,6 +321,11 @@ void WallClockASGCT::signalHandler(int signo, siginfo_t *siginfo, void *ucontext } Error BaseWallClock::start(Arguments &args) { +#if defined(UNIT_TEST) || defined(DEBUG) + if (_force_start_failure_for_test.load(std::memory_order_acquire)) { + return Error("Forced wall engine start failure (unit test)"); + } +#endif int interval = args._event != NULL ? args._interval : args._wall; if (interval < 0) { return Error("interval must be positive"); @@ -315,7 +335,6 @@ Error BaseWallClock::start(Arguments &args) { _reservoir_size = args._wall_threads_per_tick ? args._wall_threads_per_tick : DEFAULT_WALL_THREADS_PER_TICK; - initialize(args); _running = true; @@ -329,6 +348,18 @@ Error BaseWallClock::start(Arguments &args) { void BaseWallClock::stop() { _running.store(false); + // start() can return before pthread_create() runs (e.g. the forced-failure + // test hook, or an early Error return for a bad interval), leaving _thread + // at its constructor sentinel of 0. Profiler::stop() calls every engine's + // stop() whenever its event mask bit was requested, regardless of whether + // start() actually activated it, so this guard must live here rather than + // at the call site. Skipping it crashes on musl: musl's pthread_kill/ + // pthread_join dereference the thread descriptor unconditionally, so a + // zero-valued pthread_t segfaults instead of returning an error like glibc + // does. + if (_thread == 0) { + return; + } // the thread join ensures we wait for the thread to finish before returning // (and possibly removing the object) pthread_kill(_thread, WAKEUP_SIGNAL); @@ -336,12 +367,56 @@ void BaseWallClock::stop() { if (res != 0) { Log::warn("Unable to join WallClock thread on stop %d", res); } + _thread = 0; } bool BaseWallClock::isEnabled() const { return _enabled.load(std::memory_order_acquire); } +WallClockCandidateOutcome BaseWallClock::sampleThreadCommon( + ThreadEntry entry, int& num_failures, int& threads_already_exited, + int& permission_denied, int& registry_lookups, bool lookup_registry_slot, + bool precheck, ThreadFilter* thread_filter, + ThreadFilter::RecordingEpoch recording_epoch) { + if (lookup_registry_slot && entry.slot == nullptr) { + registry_lookups++; + ThreadFilter::SlotID slot_id = -1; + ThreadFilter::Slot* slot = + thread_filter->lookupByTid(entry.tid, recording_epoch, &slot_id); + if (slot != nullptr) { + entry.slot = slot; + entry.slot_id = slot_id; + entry.lifecycle_generation = slot->lifecycleGeneration(); + entry.recording_epoch = slot->recordingEpoch(); + } + } + // Timer-thread fast path (wallprecheck=true): skip the kernel IPI entirely + // only when an explicit lifecycle hook still owns an already-sampled blocked + // run. Raw OS thread state is intentionally not used here because the timer + // thread cannot prove run boundaries for the target thread. + if (precheck && suppressAlreadySampled(entry)) { + return WallClockCandidateOutcome::PRECHECK_REJECTED; + } + if (!OS::sendSignalWithCookie(entry.tid, SIGVTALRM, SignalCookie::wallclock())) { + num_failures++; + if (errno != 0) { + if (errno == ESRCH) { + threads_already_exited++; + } else if (errno == EPERM) { + permission_denied++; + } else if (errno == EAGAIN) { + // Signal queue limit (RLIMIT_SIGPENDING) reached; not a permission error. + Counters::increment(WC_SIGNAL_QUEUE_FULL); + } else { + Log::debug("unexpected error %s", strerror(errno)); + } + } + return WallClockCandidateOutcome::SIGNAL_FAILED; + } + return WallClockCandidateOutcome::SIGNAL_SENT; +} + void WallClockASGCT::initialize(Arguments& args) { _collapsing = args._wall_collapsing; _precheck = args._wall_precheck; @@ -353,11 +428,15 @@ void WallClockASGCT::initialize(Arguments& args) { } void WallClockASGCT::timerLoop() { - // todo: re-allocating the vector every time is not efficient + ThreadFilter* thread_filter = Profiler::instance()->threadFilter(); + const bool lazy_backfill = + _precheck && thread_filter->unfilteredWallTrackingActive(); + const ThreadFilter::RecordingEpoch recording_epoch = + lazy_backfill ? thread_filter->recordingEpoch() : 0; auto collectThreads = [&](std::vector& entries) { // Get thread IDs from the filter if it's enabled // Otherwise list all threads in the system - if (Profiler::instance()->threadFilter()->enabled()) { + if (thread_filter->enabled()) { Profiler::instance()->threadFilter()->collect(entries); } else { const int refresher_tid = Libraries::instance()->refresherTid(); @@ -369,45 +448,28 @@ void WallClockASGCT::timerLoop() { // enough; we also want to avoid the kill() round-trip and any // pending-signal accumulation). if (tid != OS::threadId() && tid != refresher_tid) { - entries.push_back({tid, nullptr}); // no-filter: precheck fast path is skipped (null guards) + entries.push_back({tid, nullptr, -1, 0, 0}); } } delete thread_list; } }; - auto sampleThreads = [&](ThreadEntry entry, int& num_failures, int& threads_already_exited, - int& permission_denied) { - // Timer-thread fast path (wallprecheck=true): skip the kernel IPI entirely - // only when an explicit lifecycle hook still owns an already-sampled blocked - // run. Raw OS thread state is intentionally not used here because the timer - // thread cannot prove run boundaries for the target thread. - if (_precheck && suppressAlreadySampledBlock(entry.slot)) { - return false; - } - if (!OS::sendSignalWithCookie(entry.tid, SIGVTALRM, SignalCookie::wallclock())) { - num_failures++; - if (errno != 0) { - if (errno == ESRCH) { - threads_already_exited++; - } else if (errno == EPERM) { - permission_denied++; - } else if (errno == EAGAIN) { - // Signal queue limit (RLIMIT_SIGPENDING) reached; not a permission error. - Counters::increment(WC_SIGNAL_QUEUE_FULL); - } else { - Log::debug("unexpected error %s", strerror(errno)); - } - } - return false; - } - return true; + auto sampleThreads = [&](ThreadEntry entry, int& num_failures, + int& threads_already_exited, int& permission_denied, + int& registry_lookups, bool lookup_registry_slot) { + return sampleThreadCommon(entry, num_failures, threads_already_exited, + permission_denied, registry_lookups, + lookup_registry_slot, _precheck, thread_filter, + recording_epoch); }; auto doNothing = []() { }; - timerLoopCommon(collectThreads, sampleThreads, doNothing, _reservoir_size, _interval); + timerLoopCommon(collectThreads, sampleThreads, doNothing, + _reservoir_size, _interval, _precheck, + lazy_backfill); } // WallClockJvmti: mirrors WallClockASGCT's dispatch, but the signal handler @@ -501,9 +563,14 @@ void WallClockJvmti::initialize(Arguments &args) { } void WallClockJvmti::timerLoop() { + ThreadFilter* thread_filter = Profiler::instance()->threadFilter(); + const bool lazy_backfill = + _precheck && thread_filter->unfilteredWallTrackingActive(); + const ThreadFilter::RecordingEpoch recording_epoch = + lazy_backfill ? thread_filter->recordingEpoch() : 0; auto collectThreads = [&](std::vector &entries) { const int refresher_tid = Libraries::instance()->refresherTid(); - if (Profiler::instance()->threadFilter()->enabled()) { + if (thread_filter->enabled()) { Profiler::instance()->threadFilter()->collect(entries); } else { ThreadList *thread_list = OS::listThreads(); @@ -512,7 +579,7 @@ void WallClockJvmti::timerLoop() { // Exclude the wallclock timer thread itself and the Libraries // refresher (profiler-internal). if (tid != OS::threadId() && tid != refresher_tid) { - entries.push_back({tid, nullptr}); + entries.push_back({tid, nullptr, -1, 0, 0}); } } delete thread_list; @@ -520,31 +587,17 @@ void WallClockJvmti::timerLoop() { }; auto sampleThreads = [&](ThreadEntry entry, int &num_failures, - int &threads_already_exited, int &permission_denied) { - if (_precheck && suppressAlreadySampledBlock(entry.slot)) { - return false; - } - if (!OS::sendSignalWithCookie(entry.tid, SIGVTALRM, SignalCookie::wallclock())) { - num_failures++; - if (errno != 0) { - if (errno == ESRCH) { - threads_already_exited++; - } else if (errno == EPERM) { - permission_denied++; - } else if (errno == EAGAIN) { - // Signal queue limit (RLIMIT_SIGPENDING) reached — count as missed. - Counters::increment(WC_SIGNAL_QUEUE_FULL); - } else { - Log::debug("unexpected error %s", strerror(errno)); - } - } - return false; - } - return true; + int &threads_already_exited, int &permission_denied, + int ®istry_lookups, bool lookup_registry_slot) { + return sampleThreadCommon(entry, num_failures, threads_already_exited, + permission_denied, registry_lookups, + lookup_registry_slot, _precheck, thread_filter, + recording_epoch); }; auto doNothing = []() {}; timerLoopCommon(collectThreads, sampleThreads, doNothing, - _reservoir_size, _interval); + _reservoir_size, _interval, _precheck, + lazy_backfill); } diff --git a/ddprof-lib/src/main/cpp/wallClock.h b/ddprof-lib/src/main/cpp/wallClock.h index b598e40d0b..625bcaae4f 100644 --- a/ddprof-lib/src/main/cpp/wallClock.h +++ b/ddprof-lib/src/main/cpp/wallClock.h @@ -7,6 +7,7 @@ #ifndef _WALLCLOCK_H #define _WALLCLOCK_H +#include #include #include "engine.h" #include "nativeMem.h" @@ -17,6 +18,7 @@ #include "threadFilter.h" #include "threadState.h" #include "tsc.h" +#include "wallClockCandidateSelector.h" #include "wallClockCounters.h" #include "xorshift.h" @@ -25,6 +27,9 @@ class BaseWallClock : public Engine { static std::atomic _enabled; std::atomic _running; protected: + // Backfill rejected candidates without letting a population of already- + // suppressed blockers restore O(N) registry lookups on every wall tick. + static constexpr size_t PRECHECK_VISIT_BUDGET_MULTIPLIER = 4; long _interval; // Maximum number of threads sampled in one iteration. This limit serves as a // throttle when generating profiling signals. Otherwise applications with too @@ -32,7 +37,6 @@ class BaseWallClock : public Engine { // limit low enough helps to avoid contention on a spin lock inside // Profiler::recordSample(). int _reservoir_size; - pthread_t _thread; virtual void timerLoop() = 0; virtual void initialize(Arguments& args) {}; @@ -45,8 +49,22 @@ class BaseWallClock : public Engine { bool isEnabled() const; static bool inSyscall(void* ucontext); + // Shared by WallClockASGCT::timerLoop() and WallClockJvmti::timerLoop(): both + // engines resolve a registry slot, check precheck suppression, and send the + // sampling signal identically; only collectThreads() differs between them. + WallClockCandidateOutcome sampleThreadCommon( + ThreadEntry entry, int& num_failures, int& threads_already_exited, + int& permission_denied, int& registry_lookups, bool lookup_registry_slot, + bool precheck, ThreadFilter* thread_filter, + ThreadFilter::RecordingEpoch recording_epoch); + // Timer-thread precheck: true when an explicit lifecycle hook still owns an + // already-sampled blocked run for this thread, so no signal is needed. + static bool suppressAlreadySampled(const ThreadEntry& entry); + template - void timerLoopCommon(CollectThreadsFunc collectThreads, SampleThreadsFunc sampleThreads, CleanThreadFunc cleanThreads, int reservoirSize, u64 interval) { + void timerLoopCommon(CollectThreadsFunc collectThreads, SampleThreadsFunc sampleThreads, + CleanThreadFunc cleanThreads, int reservoirSize, u64 interval, + bool precheck = false, bool lazyBackfill = false) { if (!_enabled.load(std::memory_order_acquire)) { return; } @@ -57,6 +75,12 @@ class BaseWallClock : public Engine { // stream between recordings without needing an entropy source. u64 rng = xorshift::seed((u64)(uintptr_t)this, TSC::ticks()); + std::mt19937 candidate_generator; + if (lazyBackfill) { + candidate_generator.seed( + xorshift::seed((u64)(uintptr_t)&candidate_generator, TSC::ticks())); + } + std::vector threads; threads.reserve(reservoirSize); int self = OS::threadId(); @@ -81,19 +105,62 @@ class BaseWallClock : public Engine { while (_running.load(std::memory_order_relaxed)) { collectThreads(threads); NativeMem::setLive(NM_WALLCLOCK, (long long)threads.capacity() * sizeof(ThreadType)); + // The epoch's samplePoolSize counts every candidate, including threads + // the precheck suppresses below, so it means the same in both precheck + // modes: lazy backfill only discovers suppression while visiting. + const u32 num_samplable_threads = static_cast(threads.size()); + if (precheck && !lazyBackfill) { + // Drop suppressed threads before reservoir sampling so they do not + // consume reservoir slots. + threads.erase(std::remove_if(threads.begin(), threads.end(), + suppressAlreadySampled), + threads.end()); + } int num_failures = 0; int threads_already_exited = 0; int permission_denied = 0; + int registry_lookups = 0; u32 num_successful_samples = 0; - std::vector sample = reservoir.sample(threads); - for (ThreadType thread : sample) { - if (sampleThreads(thread, num_failures, threads_already_exited, permission_denied)) { - num_successful_samples++; + if (lazyBackfill) { + WallClockCandidateStats stats = selectWallClockCandidates( + threads, + static_cast(reservoirSize), + static_cast(reservoirSize) * + PRECHECK_VISIT_BUDGET_MULTIPLIER, + candidate_generator, + [&](ThreadType thread) { + WallClockCandidateOutcome outcome = sampleThreads( + thread, num_failures, threads_already_exited, + permission_denied, registry_lookups, true); + if (outcome == WallClockCandidateOutcome::SIGNAL_SENT) { + num_successful_samples++; + } + return outcome; + }); + if (stats.precheck_rejected > 0) { + Counters::increment(WC_PRECHECK_CANDIDATES_REJECTED, + stats.precheck_rejected); + } + if (stats.visit_limit_reached) { + Counters::increment(WC_PRECHECK_LOOKUP_BUDGET_EXHAUSTED); + } + } else { + std::vector sample = reservoir.sample(threads); + for (ThreadType thread : sample) { + WallClockCandidateOutcome outcome = sampleThreads( + thread, num_failures, threads_already_exited, + permission_denied, registry_lookups, false); + if (outcome == WallClockCandidateOutcome::SIGNAL_SENT) { + num_successful_samples++; + } } } + if (registry_lookups > 0) { + Counters::increment(WC_PRECHECK_REGISTRY_LOOKUPS, registry_lookups); + } - epoch.updateNumSamplableThreads(threads.size()); + epoch.updateNumSamplableThreads(num_samplable_threads); epoch.updateNumFailedSamples(num_failures); epoch.updateNumSuccessfulSamples(num_successful_samples); epoch.addNumSuppressedSampledRun(WallClockCounters::drainSuppressedSampledRun()); @@ -150,6 +217,18 @@ class BaseWallClock : public Engine { Error start(Arguments& args); void stop(); + +#if defined(UNIT_TEST) || defined(DEBUG) + static void setForceStartFailureForTest(bool force) { + _force_start_failure_for_test.store(force, std::memory_order_release); + } + static bool isForceStartFailureForTest() { + return _force_start_failure_for_test.load(std::memory_order_acquire); + } + private: + static std::atomic _force_start_failure_for_test; + public: +#endif }; class WallClockASGCT : public BaseWallClock { @@ -168,6 +247,7 @@ class WallClockASGCT : public BaseWallClock { const char* name() override { return "WallClock (ASGCT)"; } + bool supportsUnfilteredThreadRegistryTracking() const override { return true; } }; // Wall-clock engine that uses BaseWallClock's pthread reservoir sampling loop @@ -189,6 +269,7 @@ class WallClockJvmti : public BaseWallClock { const char* name() override { return "WallClock (JVMTI)"; } + bool supportsUnfilteredThreadRegistryTracking() const override { return true; } }; #endif // _WALLCLOCK_H diff --git a/ddprof-lib/src/main/cpp/wallClockBlockTracker.cpp b/ddprof-lib/src/main/cpp/wallClockBlockTracker.cpp new file mode 100644 index 0000000000..6f71b5d3f4 --- /dev/null +++ b/ddprof-lib/src/main/cpp/wallClockBlockTracker.cpp @@ -0,0 +1,123 @@ +/* + * Copyright 2026, Datadog, Inc + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +#include "wallClockBlockTracker.h" + +u64 WallClockBlockTracker::enterBlockedRun(ThreadFilter* registry, ThreadFilter::SlotID slot_id, + OSThreadState state, BlockRunOwner owner) { + ThreadFilter::Slot* identity_slot = registry->slotForId(slot_id); + BlockState* block_slot = slotForId(slot_id); + if (identity_slot == nullptr || block_slot == nullptr) { + return 0; + } + u32 generation = 0; + if (!block_slot->trySetActiveBlockRun(identity_slot, state, owner, &generation, + registry->unfilteredWallTrackingActive())) { + return 0; + } + return WallClockBlockTracker::encodeBlockRunToken(slot_id, generation); +} + +void WallClockBlockTracker::exitBlockedRun(ThreadFilter::SlotID slot_id) { + BlockState* s = slotForId(slot_id); + if (s != nullptr) { + s->resetSlot(OSThreadState::RUNNABLE); + } +} + +bool WallClockBlockTracker::exitBlockedRun(ThreadFilter::SlotID slot_id, u32 generation) { + BlockState* s = slotForId(slot_id); + if (s == nullptr || generation == 0 || s->blockGeneration() != generation) { + return false; + } + s->resetSlot(OSThreadState::RUNNABLE); + return true; +} + +void WallClockBlockTracker::resetSlot(ThreadFilter::SlotID slot_id, OSThreadState state) { + BlockState* s = slotForId(slot_id); + if (s != nullptr) { + s->resetSlot(state); + } +} + +void WallClockBlockTracker::resetAll() { + for (ThreadFilter::SlotID slot_id = 0; slot_id < ThreadFilter::kMaxThreads; ++slot_id) { + _slots[slot_id].resetSlot(OSThreadState::UNKNOWN); + } +} + +bool WallClockBlockTracker::shouldSuppressOwnedBlock(ThreadFilter* registry, + const ThreadEntry& entry) const { + ThreadFilter::Slot* identity_slot = entry.slot; + if (identity_slot == nullptr || identity_slot->nativeTid() != entry.tid || + identity_slot->lifecycleGeneration() != entry.lifecycle_generation) { + return false; + } + BlockState* slot = slotForId(entry.slot_id); + if (slot == nullptr) { + return false; + } + + const bool unfiltered_tracking = registry->unfilteredWallTrackingActive(); + ThreadFilter::RecordingEpoch epoch = 0; + if (unfiltered_tracking) { + epoch = registry->recordingEpoch(); + if (epoch == 0 || entry.recording_epoch != epoch || + identity_slot->recordingEpoch() != epoch) { + return false; + } + } + +#ifdef UNIT_TEST + if (_suppression_snapshot_hook != nullptr) { + _suppression_snapshot_hook(_suppression_snapshot_hook_arg); + } +#endif + + u32 block_generation = slot->blockGeneration(); + BlockRunOwner owner = slot->activeBlockOwner(); + OSThreadState state = slot->activeBlockState(); + bool context_eligible = + !unfiltered_tracking || slot->activeBlockRemainedOutsideContextWindow(identity_slot); + bool sampled = slot->sampledThisRun(); + OSThreadState last_sampled_state = + sampled ? slot->lastSampledState() : OSThreadState::UNKNOWN; + bool suppressible_state = isPrecheckSuppressionState(state); + if (owner == BlockRunOwner::NONE || !context_eligible || + !suppressible_state || !sampled || state != last_sampled_state) { + return false; + } + + // The payload is spread across independent atomics. Accept it only if the + // slot still represents the lifecycle and block run captured earlier in + // this wall-clock timer-thread pass, when `entry` was populated — either + // by ThreadFilter::collect() (inside timerLoop()'s collectThreads lambda) + // or by the lazy registry lookup at the top of sampleThreadCommon() — + // both in wallClock.cpp/.h (BaseWallClock::timerLoopCommon() itself is a + // template defined in wallClock.h). + if (slot->activeBlockOwner() != owner || + slot->blockGeneration() != block_generation || + identity_slot->nativeTid() != entry.tid || + identity_slot->lifecycleGeneration() != entry.lifecycle_generation) { + return false; + } + if (unfiltered_tracking && + (registry->recordingEpoch() != epoch || identity_slot->recordingEpoch() != epoch || + !slot->activeBlockRemainedOutsideContextWindow(identity_slot))) { + return false; + } + return true; +} diff --git a/ddprof-lib/src/main/cpp/wallClockBlockTracker.h b/ddprof-lib/src/main/cpp/wallClockBlockTracker.h new file mode 100644 index 0000000000..37bd5e3b6e --- /dev/null +++ b/ddprof-lib/src/main/cpp/wallClockBlockTracker.h @@ -0,0 +1,250 @@ +/* + * Copyright 2026, Datadog, Inc + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +#ifndef _WALLCLOCKBLOCKTRACKER_H +#define _WALLCLOCKBLOCKTRACKER_H + +#include +#include +#include + +#include "arch.h" +#include "threadFilter.h" +#include "threadState.h" + +enum class BlockRunOwner : int { + NONE = 0, + JAVA = 1, + JVMTI = 2, + NATIVE = 3, +}; + +// Wall-clock owned/unowned block-run suppression state, addressed in parallel +// to ThreadFilter's own slots by ThreadFilter::SlotID. Kept separate from +// ThreadFilter::Slot because this is wall-clock-engine policy (has this +// blocked interval already produced a sample?), not thread identity - see +// threadFilter.h for the registry itself. Methods that need registry facts +// (context-window state, unfiltered-tracking mode) take a ThreadFilter +// pointer/Slot pointer as an explicit parameter rather than storing one, so +// tests can pair a tracker with any ThreadFilter instance. +class WallClockBlockTracker { +public: + // One cache line per slot, mirroring ThreadFilter::Slot's own + // false-sharing avoidance. BlockState instances are process-lifetime + // (owned by WallClockBlockTracker's inline array), so a captured + // BlockState* is always dereferenceable. + struct alignas(DEFAULT_CACHE_LINE_SIZE) BlockState { + static constexpr u64 kUnownedBlockedFallbackRatio = 10; + + std::atomic unowned_blocked_pending_weight{0}; + std::atomic unowned_blocked_decision_count{0}; + std::atomic unowned_blocked_call_trace_id{0}; + std::atomic active_block_context_epoch{0}; + std::atomic unowned_blocked_state{OSThreadState::UNKNOWN}; + std::atomic active_block_owner{static_cast(BlockRunOwner::NONE)}; + std::atomic block_generation{0}; + // Wall-clock once-per-run suppression state. The signal handler records the + // last sampled blocked state; the signal handler and timer thread read it to + // suppress duplicate samples, while lifecycle/block-exit paths reset it. + // Release/acquire on sampled_this_run pairs with relaxed last_sampled_state, + // following the standard flag+payload pattern. + std::atomic last_sampled_state{OSThreadState::UNKNOWN}; + // Set by explicit block enter/exit hooks. It lets the timer skip sending a signal + // only while instrumentation still owns a suppressible blocking interval. + std::atomic active_block_state{OSThreadState::UNKNOWN}; + std::atomic sampled_this_run{false}; + char padding[DEFAULT_CACHE_LINE_SIZE + - sizeof(std::atomic) + - sizeof(std::atomic) + - sizeof(std::atomic) + - sizeof(std::atomic) + - sizeof(std::atomic) + - sizeof(std::atomic) + - sizeof(std::atomic) + - sizeof(std::atomic) + - sizeof(std::atomic) + - sizeof(std::atomic)]; + + inline bool sampledThisRun() const { + return sampled_this_run.load(std::memory_order_acquire); + } + inline OSThreadState lastSampledState() const { + return last_sampled_state.load(std::memory_order_relaxed); + } + inline void markSampledThisRun(OSThreadState state) { + last_sampled_state.store(state, std::memory_order_relaxed); + sampled_this_run.store(true, std::memory_order_release); + } + inline void resetSampledRun(OSThreadState state) { + resetUnownedBlockedSampling(); + last_sampled_state.store(state, std::memory_order_relaxed); + sampled_this_run.store(false, std::memory_order_release); + } + inline OSThreadState activeBlockState() const { + return active_block_state.load(std::memory_order_acquire); + } + inline void setActiveBlockState(OSThreadState state) { + active_block_state.store(state, std::memory_order_release); + } + inline BlockRunOwner activeBlockOwner() const { + return static_cast(active_block_owner.load(std::memory_order_acquire)); + } + inline u32 blockGeneration() const { + return block_generation.load(std::memory_order_acquire); + } + inline void resetUnownedBlockedSampling() { + unowned_blocked_pending_weight.store(0, std::memory_order_relaxed); + unowned_blocked_decision_count.store(0, std::memory_order_relaxed); + unowned_blocked_state.store(OSThreadState::UNKNOWN, std::memory_order_relaxed); + unowned_blocked_call_trace_id.store(0, std::memory_order_release); + } + inline bool shouldRecordUnownedBlockedSample() { + u64 decision = unowned_blocked_decision_count.fetch_add(1, std::memory_order_relaxed) + 1; + if ((decision % kUnownedBlockedFallbackRatio) == 1) { + return true; + } + unowned_blocked_pending_weight.fetch_add(1, std::memory_order_relaxed); + return false; + } + inline u64 consumeUnownedBlockedWeight() { + return unowned_blocked_pending_weight.exchange(0, std::memory_order_relaxed) + 1; + } + inline void restoreUnownedBlockedWeight(u64 weight) { + if (weight > 1) { + unowned_blocked_pending_weight.fetch_add(weight - 1, std::memory_order_relaxed); + } + } + inline void recordUnownedBlockedSample(u64 call_trace_id, OSThreadState state) { + unowned_blocked_state.store(state, std::memory_order_relaxed); + unowned_blocked_call_trace_id.store(call_trace_id, std::memory_order_release); + } + inline bool flushUnownedBlockedTail(u64& call_trace_id, u64& weight, + OSThreadState& state) { + call_trace_id = unowned_blocked_call_trace_id.exchange(0, std::memory_order_acq_rel); + weight = unowned_blocked_pending_weight.exchange(0, std::memory_order_relaxed); + state = unowned_blocked_state.exchange(OSThreadState::UNKNOWN, std::memory_order_relaxed); + unowned_blocked_decision_count.store(0, std::memory_order_relaxed); + if (call_trace_id == 0 || weight == 0 || state == OSThreadState::UNKNOWN) { + return false; + } + return true; + } + // identity_slot supplies the context-window state, which lives on + // ThreadFilter::Slot. See ThreadFilter::Slot::rawContextWindowState(). + inline bool trySetActiveBlockRun(ThreadFilter::Slot* identity_slot, OSThreadState state, + BlockRunOwner owner, u32* generation_out, + bool outside_context_required) { + u64 context_state = identity_slot->rawContextWindowState(); + if (outside_context_required && (context_state & 1) != 0) { + return false; + } + int expected_owner = static_cast(BlockRunOwner::NONE); + if (!active_block_owner.compare_exchange_strong( + expected_owner, static_cast(owner), std::memory_order_acq_rel, + std::memory_order_acquire)) { + return false; + } + if (outside_context_required && + identity_slot->rawContextWindowState() != context_state) { + active_block_owner.store(static_cast(BlockRunOwner::NONE), + std::memory_order_release); + return false; + } + u32 generation = block_generation.fetch_add(1, std::memory_order_acq_rel) + 1; + active_block_context_epoch.store(context_state >> 1, std::memory_order_relaxed); + resetUnownedBlockedSampling(); + last_sampled_state.store(OSThreadState::UNKNOWN, std::memory_order_relaxed); + sampled_this_run.store(false, std::memory_order_relaxed); + active_block_state.store(state, std::memory_order_release); + *generation_out = generation; + return true; + } + inline void resetSlot(OSThreadState state) { + active_block_state.store(OSThreadState::UNKNOWN, std::memory_order_release); + resetSampledRun(state); + active_block_owner.store(static_cast(BlockRunOwner::NONE), std::memory_order_release); + } + // identity_slot supplies the context-window state, which lives on + // ThreadFilter::Slot. See ThreadFilter::Slot::rawContextWindowState(). + inline bool activeBlockRemainedOutsideContextWindow(ThreadFilter::Slot* identity_slot) const { + u64 context_state = identity_slot->rawContextWindowState(); + return (context_state & 1) == 0 && + active_block_context_epoch.load(std::memory_order_acquire) == + (context_state >> 1); + } + }; + static_assert(sizeof(BlockState) == DEFAULT_CACHE_LINE_SIZE, + "BlockState must fit exactly one cache line"); + static_assert(std::atomic::is_always_lock_free, + "BlockState OSThreadState fields must be lock-free for signal-handler safety"); + static_assert(std::atomic::is_always_lock_free, + "BlockState::sampled_this_run must be lock-free for signal-handler safety"); + static_assert(std::atomic::is_always_lock_free, + "BlockState u64 fields must be lock-free for signal-handler safety"); + + // Returns nullptr if slot_id is out of range. Storage is a single eager + // allocation sized to ThreadFilter::kMaxThreads (unlike ThreadFilter's own + // lazily-chunked storage), so lookup is a direct array index. + inline BlockState* slotForId(ThreadFilter::SlotID slot_id) const { + if (slot_id < 0 || slot_id >= ThreadFilter::kMaxThreads) return nullptr; + return const_cast(&_slots[slot_id]); + } + + // Block-run tokens returned by enterBlockedRun() pack the generation in the + // upper 32 bits and slot_id + 1 in the lower 32 bits, so 0 means "no run". + static inline u64 encodeBlockRunToken(ThreadFilter::SlotID slot_id, u32 generation) { + return (static_cast(generation) << 32) | static_cast(slot_id + 1); + } + static inline ThreadFilter::SlotID tokenSlotId(u64 token) { + return static_cast(static_cast(token) - 1); + } + static inline u32 tokenGeneration(u64 token) { + return static_cast(token >> 32); + } + + u64 enterBlockedRun(ThreadFilter* registry, ThreadFilter::SlotID slot_id, + OSThreadState state, BlockRunOwner owner = BlockRunOwner::JAVA); + // Unconditional cleanup, used only by tests and the fuzzer (registry + // reset/unregister paths use resetSlot()). Normal block lifecycles must + // use the generation-checked overload so they cannot clear another owner. + void exitBlockedRun(ThreadFilter::SlotID slot_id); + bool exitBlockedRun(ThreadFilter::SlotID slot_id, u32 generation); + // Clears stale suppression state so a new/reused slot cannot inherit a + // predecessor's active block or once-per-run sampled marker. Called by + // ThreadFilter at its own registry-lifecycle decision points. + void resetSlot(ThreadFilter::SlotID slot_id, OSThreadState state); + void resetAll(); + // Reads the complete timer-side suppression payload and rejects it if slot + // identity or block lifecycle changes before final validation. + bool shouldSuppressOwnedBlock(ThreadFilter* registry, const ThreadEntry& entry) const; + +#ifdef UNIT_TEST + using SuppressionSnapshotHook = void (*)(void*); + void setSuppressionSnapshotHookForTest(SuppressionSnapshotHook hook, void* arg) { + _suppression_snapshot_hook = hook; + _suppression_snapshot_hook_arg = arg; + } +#endif + +private: + std::array _slots; + +#ifdef UNIT_TEST + SuppressionSnapshotHook _suppression_snapshot_hook = nullptr; + void* _suppression_snapshot_hook_arg = nullptr; +#endif +}; + +#endif // _WALLCLOCKBLOCKTRACKER_H diff --git a/ddprof-lib/src/main/cpp/wallClockCandidateSelector.h b/ddprof-lib/src/main/cpp/wallClockCandidateSelector.h new file mode 100644 index 0000000000..977fe139d5 --- /dev/null +++ b/ddprof-lib/src/main/cpp/wallClockCandidateSelector.h @@ -0,0 +1,75 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +#ifndef WALL_CLOCK_CANDIDATE_SELECTOR_H +#define WALL_CLOCK_CANDIDATE_SELECTOR_H + +#include +#include +#include +#include +#include + +enum class WallClockCandidateOutcome { + SIGNAL_SENT, + SIGNAL_FAILED, + PRECHECK_REJECTED, +}; + +struct WallClockCandidateStats { + size_t visited = 0; + size_t slots_consumed = 0; + size_t precheck_rejected = 0; + bool visit_limit_reached = false; +}; + +// Visits a uniformly randomized prefix without replacement. A precheck rejection +// is the only outcome that does not consume target capacity. This runs only on +// the wall-clock timer thread. +// +// Randomization matters because OS::listThreads() (the source of `candidates` +// in unfiltered lazy-backfill mode, where the context filter is disabled) +// returns threads in the same stable order every timer pass. visit_limit +// bounds how far into that list we're willing to walk to backfill past +// PRECHECK_REJECTED candidates, so it will not reach the entire pool when the +// pool is large. Walking a fixed-order prefix every interval would +// systematically favor whichever threads happen to sort first and starve +// others of ever being sampled. Reshuffling the visited prefix each pass +// instead gives every listed thread roughly equal odds of being sampled over +// time. +template +WallClockCandidateStats selectWallClockCandidates(std::vector& candidates, + size_t target_size, + size_t visit_limit, + URBG& generator, + Visitor&& visitor) { + WallClockCandidateStats stats; + if (target_size == 0 || candidates.empty() || visit_limit == 0) { + return stats; + } + + size_t max_visits = std::min(candidates.size(), visit_limit); + for (size_t i = 0; i < max_visits && stats.slots_consumed < target_size; ++i) { + std::uniform_int_distribution next(i, candidates.size() - 1); + size_t selected = next(generator); + if (selected != i) { + std::swap(candidates[i], candidates[selected]); + } + + stats.visited++; + WallClockCandidateOutcome outcome = visitor(candidates[i]); + if (outcome == WallClockCandidateOutcome::PRECHECK_REJECTED) { + stats.precheck_rejected++; + } else { + stats.slots_consumed++; + } + } + stats.visit_limit_reached = + stats.slots_consumed < target_size && + stats.visited == max_visits && max_visits < candidates.size(); + return stats; +} + +#endif // WALL_CLOCK_CANDIDATE_SELECTOR_H diff --git a/ddprof-lib/src/main/java/com/datadoghq/profiler/JavaProfilerTestSupport.java b/ddprof-lib/src/main/java/com/datadoghq/profiler/JavaProfilerTestSupport.java new file mode 100644 index 0000000000..69e435ea97 --- /dev/null +++ b/ddprof-lib/src/main/java/com/datadoghq/profiler/JavaProfilerTestSupport.java @@ -0,0 +1,45 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ +package com.datadoghq.profiler; + +/** + * Whitebox testing hooks for {@link JavaProfiler} internals. Kept in a separate + * class so these do not appear on JavaProfiler's public API surface. + */ +public final class JavaProfilerTestSupport { + private JavaProfilerTestSupport() {} + + /** + * Test-only hook (debug builds only): forces the next wall-clock engine + * start() call to fail. No-op in release builds. For whitebox testing. + */ + public static void setForceWallStartFailureForTest(boolean force) { + setForceWallStartFailureForTest0(force); + } + + /** + * Test-only accessor: whether the forced wall-engine start-failure toggle is + * currently armed. Returns {@code false} in release builds (where the setter + * is a no-op), so callers can self-skip via {@code Assumptions.assumeTrue}. + * For whitebox testing. + */ + public static boolean isForceWallStartFailureArmedForTest() { + return isForceWallStartFailureArmedForTest0(); + } + + /** + * Test-only accessor: whether the thread registry currently admits new + * registrations. For whitebox testing. + */ + public static boolean isThreadRegistryActiveForTest() { + return isThreadRegistryActiveForTest0(); + } + + private static native void setForceWallStartFailureForTest0(boolean force); + + private static native boolean isForceWallStartFailureArmedForTest0(); + + private static native boolean isThreadRegistryActiveForTest0(); +} diff --git a/ddprof-lib/src/test/cpp/frame_ut.cpp b/ddprof-lib/src/test/cpp/frame_ut.cpp index 951db75fb8..83779a8021 100644 --- a/ddprof-lib/src/test/cpp/frame_ut.cpp +++ b/ddprof-lib/src/test/cpp/frame_ut.cpp @@ -4,7 +4,9 @@ #include #include +#include #include "../../main/cpp/frame.h" +#include "../../main/cpp/frames.h" #include "../../main/cpp/gtest_crash_handler.h" // Test-only friend accessor for VM internals. It exists solely so these unit @@ -44,6 +46,26 @@ class GlobalSetup { static GlobalSetup global_setup; +TEST(CopyJvmtiFramesTest, PreservesOverlappingSourceFields) { + union { + ASGCT_CallFrame asgct[2]; + jvmtiFrameInfo jvmti[2]; + } buffer; + jmethodID first_method = + reinterpret_cast(static_cast(0x12340)); + jmethodID second_method = + reinterpret_cast(static_cast(0x56780)); + buffer.jvmti[0] = {first_method, 17}; + buffer.jvmti[1] = {second_method, 29}; + + copyJvmtiFrames(buffer.asgct, buffer.jvmti, 2); + + EXPECT_EQ(buffer.asgct[0].method_id, first_method); + EXPECT_EQ(buffer.asgct[0].bci, 17); + EXPECT_EQ(buffer.asgct[1].method_id, second_method); + EXPECT_EQ(buffer.asgct[1].bci, 29); +} + // ---- encode ---------------------------------------------------------------- TEST(FrameTypeEncodeTest, EncodedMarkerBitIsSet) { diff --git a/ddprof-lib/src/test/cpp/park_state_ut.cpp b/ddprof-lib/src/test/cpp/park_state_ut.cpp index 69f3792424..ff60f49e52 100644 --- a/ddprof-lib/src/test/cpp/park_state_ut.cpp +++ b/ddprof-lib/src/test/cpp/park_state_ut.cpp @@ -22,6 +22,7 @@ #include "threadLocalData.h" #include "threadFilter.h" #include "wallClock.h" +#include "wallClockBlockTracker.h" namespace { @@ -138,7 +139,7 @@ TEST(ProfiledThreadParkStateTest, ParkExitReturnsZeroTokenWhenBlockRunWasNotArme } TEST(WallClockOncePerRunFilterTest, SlotStateTransitions) { - ThreadFilter::Slot slot; + WallClockBlockTracker::BlockState slot; EXPECT_FALSE(slot.sampledThisRun()); EXPECT_EQ(OSThreadState::UNKNOWN, slot.lastSampledState()); @@ -182,17 +183,17 @@ TEST(WallClockOncePerRunFilterTest, SlotStateTransitions) { } TEST(WallClockOncePerRunFilterTest, UnownedBlockedFallbackCarriesWeight) { - ThreadFilter::Slot slot; + WallClockBlockTracker::BlockState slot; EXPECT_TRUE(slot.shouldRecordUnownedBlockedSample()); EXPECT_EQ(1ULL, slot.consumeUnownedBlockedWeight()); - for (u64 i = 1; i < ThreadFilter::Slot::kUnownedBlockedFallbackRatio; i++) { + for (u64 i = 1; i < WallClockBlockTracker::BlockState::kUnownedBlockedFallbackRatio; i++) { EXPECT_FALSE(slot.shouldRecordUnownedBlockedSample()); } EXPECT_TRUE(slot.shouldRecordUnownedBlockedSample()); - EXPECT_EQ(ThreadFilter::Slot::kUnownedBlockedFallbackRatio, + EXPECT_EQ(WallClockBlockTracker::BlockState::kUnownedBlockedFallbackRatio, slot.consumeUnownedBlockedWeight()); slot.restoreUnownedBlockedWeight(4); @@ -204,13 +205,13 @@ TEST(WallClockOncePerRunFilterTest, UnownedBlockedFallbackCarriesWeight) { } TEST(WallClockOncePerRunFilterTest, UnownedBlockedFallbackFlushesTailWeightWithRecordedStack) { - ThreadFilter::Slot slot; + WallClockBlockTracker::BlockState slot; ASSERT_TRUE(slot.shouldRecordUnownedBlockedSample()); EXPECT_EQ(1ULL, slot.consumeUnownedBlockedWeight()); slot.recordUnownedBlockedSample(42, OSThreadState::SLEEPING); - for (u64 i = 1; i < ThreadFilter::Slot::kUnownedBlockedFallbackRatio; i++) { + for (u64 i = 1; i < WallClockBlockTracker::BlockState::kUnownedBlockedFallbackRatio; i++) { EXPECT_FALSE(slot.shouldRecordUnownedBlockedSample()); } @@ -219,7 +220,7 @@ TEST(WallClockOncePerRunFilterTest, UnownedBlockedFallbackFlushesTailWeightWithR OSThreadState state = OSThreadState::UNKNOWN; EXPECT_TRUE(slot.flushUnownedBlockedTail(call_trace_id, weight, state)); EXPECT_EQ(42ULL, call_trace_id); - EXPECT_EQ(ThreadFilter::Slot::kUnownedBlockedFallbackRatio - 1, weight); + EXPECT_EQ(WallClockBlockTracker::BlockState::kUnownedBlockedFallbackRatio - 1, weight); EXPECT_EQ(OSThreadState::SLEEPING, state); EXPECT_FALSE(slot.flushUnownedBlockedTail(call_trace_id, weight, state)); @@ -228,12 +229,12 @@ TEST(WallClockOncePerRunFilterTest, UnownedBlockedFallbackFlushesTailWeightWithR } TEST(WallClockOncePerRunFilterTest, UnownedBlockedFallbackDoesNotFlushWithoutRecordedStack) { - ThreadFilter::Slot slot; + WallClockBlockTracker::BlockState slot; ASSERT_TRUE(slot.shouldRecordUnownedBlockedSample()); EXPECT_EQ(1ULL, slot.consumeUnownedBlockedWeight()); - for (u64 i = 1; i < ThreadFilter::Slot::kUnownedBlockedFallbackRatio; i++) { + for (u64 i = 1; i < WallClockBlockTracker::BlockState::kUnownedBlockedFallbackRatio; i++) { EXPECT_FALSE(slot.shouldRecordUnownedBlockedSample()); } @@ -242,19 +243,19 @@ TEST(WallClockOncePerRunFilterTest, UnownedBlockedFallbackDoesNotFlushWithoutRec OSThreadState state = OSThreadState::UNKNOWN; EXPECT_FALSE(slot.flushUnownedBlockedTail(call_trace_id, weight, state)); EXPECT_EQ(0ULL, call_trace_id); - EXPECT_EQ(ThreadFilter::Slot::kUnownedBlockedFallbackRatio - 1, weight); + EXPECT_EQ(WallClockBlockTracker::BlockState::kUnownedBlockedFallbackRatio - 1, weight); EXPECT_EQ(OSThreadState::UNKNOWN, state); EXPECT_TRUE(slot.shouldRecordUnownedBlockedSample()); } TEST(WallClockOncePerRunFilterTest, UnownedBlockedFallbackDoesNotFlushWithoutSavedState) { - ThreadFilter::Slot slot; + WallClockBlockTracker::BlockState slot; ASSERT_TRUE(slot.shouldRecordUnownedBlockedSample()); EXPECT_EQ(1ULL, slot.consumeUnownedBlockedWeight()); slot.recordUnownedBlockedSample(42, OSThreadState::SLEEPING); - for (u64 i = 1; i < ThreadFilter::Slot::kUnownedBlockedFallbackRatio; i++) { + for (u64 i = 1; i < WallClockBlockTracker::BlockState::kUnownedBlockedFallbackRatio; i++) { EXPECT_FALSE(slot.shouldRecordUnownedBlockedSample()); } @@ -265,12 +266,12 @@ TEST(WallClockOncePerRunFilterTest, UnownedBlockedFallbackDoesNotFlushWithoutSav OSThreadState state = OSThreadState::SLEEPING; EXPECT_FALSE(slot.flushUnownedBlockedTail(call_trace_id, weight, state)); EXPECT_EQ(42ULL, call_trace_id); - EXPECT_EQ(ThreadFilter::Slot::kUnownedBlockedFallbackRatio - 1, weight); + EXPECT_EQ(WallClockBlockTracker::BlockState::kUnownedBlockedFallbackRatio - 1, weight); EXPECT_EQ(OSThreadState::UNKNOWN, state); } TEST(WallClockOncePerRunFilterTest, UnownedBlockedTailStateConcurrentStress) { - ThreadFilter::Slot slot; + WallClockBlockTracker::BlockState slot; std::atomic start{false}; std::atomic invariant_failures{0}; std::vector workers; @@ -324,11 +325,13 @@ TEST(WallClockOncePerRunFilterTest, UnownedBlockedTailStateConcurrentStress) { TEST(WallClockOncePerRunFilterTest, FilterHelpersManageActiveBlockState) { ThreadFilter filter; + WallClockBlockTracker tracker; + filter.setBlockTracker(&tracker); filter.init("1"); - ThreadFilter::SlotID slot_id = filter.registerThread(); + ThreadFilter::SlotID slot_id = filter.registerThread(1234); - filter.enterBlockedRun(slot_id, OSThreadState::CONDVAR_WAIT); - ThreadFilter::Slot *slot = filter.slotForId(slot_id); + tracker.enterBlockedRun(&filter, slot_id, OSThreadState::CONDVAR_WAIT); + WallClockBlockTracker::BlockState *slot = tracker.slotForId(slot_id); ASSERT_NE(nullptr, slot); EXPECT_EQ(OSThreadState::CONDVAR_WAIT, slot->activeBlockState()); @@ -337,7 +340,7 @@ TEST(WallClockOncePerRunFilterTest, FilterHelpersManageActiveBlockState) { EXPECT_TRUE(slot->sampledThisRun() && slot->activeBlockState() == slot->lastSampledState()); - filter.exitBlockedRun(slot_id); + tracker.exitBlockedRun(slot_id); EXPECT_EQ(OSThreadState::UNKNOWN, slot->activeBlockState()); EXPECT_FALSE(slot->sampledThisRun()); EXPECT_EQ(OSThreadState::RUNNABLE, slot->lastSampledState()); @@ -347,10 +350,12 @@ TEST(WallClockOncePerRunFilterTest, FilterHelpersManageActiveBlockState) { // the new thread takes the slot (ThreadFilter::resetSlotRunState does this). TEST(WallClockOncePerRunFilterTest, ResetClearsArmedFlagOnSlotReuse) { ThreadFilter filter; + WallClockBlockTracker tracker; + filter.setBlockTracker(&tracker); filter.init("1"); - ThreadFilter::SlotID slot_id = filter.registerThread(); - filter.enterBlockedRun(slot_id, OSThreadState::CONDVAR_WAIT); - ThreadFilter::Slot *slot = filter.slotForId(slot_id); + ThreadFilter::SlotID slot_id = filter.registerThread(1234); + tracker.enterBlockedRun(&filter, slot_id, OSThreadState::CONDVAR_WAIT); + WallClockBlockTracker::BlockState *slot = tracker.slotForId(slot_id); ASSERT_NE(nullptr, slot); slot->markSampledThisRun(OSThreadState::CONDVAR_WAIT); EXPECT_TRUE(slot->sampledThisRun()); diff --git a/ddprof-lib/src/test/cpp/stress_threadLifecycle_ut.cpp b/ddprof-lib/src/test/cpp/stress_threadLifecycle_ut.cpp index 18bfa78d1f..001e13ef3d 100644 --- a/ddprof-lib/src/test/cpp/stress_threadLifecycle_ut.cpp +++ b/ddprof-lib/src/test/cpp/stress_threadLifecycle_ut.cpp @@ -115,7 +115,7 @@ static void churn_worker(ThreadFilter* filter, bool with_dump) { EXPECT_NE(nullptr, self); if (!self) return; - ThreadFilter::SlotID slot = filter->registerThread(); + ThreadFilter::SlotID slot = filter->registerThread(self->tid()); if (slot >= 0) { self->setFilterSlotId(slot); filter->add(self->tid(), slot); diff --git a/ddprof-lib/src/test/cpp/stress_wallClockBlockTracker_ut.cpp b/ddprof-lib/src/test/cpp/stress_wallClockBlockTracker_ut.cpp new file mode 100644 index 0000000000..8109835d0e --- /dev/null +++ b/ddprof-lib/src/test/cpp/stress_wallClockBlockTracker_ut.cpp @@ -0,0 +1,128 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + * + * Sustained multithreaded churn for ThreadFilter / WallClockBlockTracker, + * complementing the deterministic single-race gtests in threadFilter_ut.cpp + * and wallClockBlockTracker_ut.cpp (which each pin down one exact + * interleaving) and the single-threaded fuzz_threadFilter.cpp fuzzer (which + * cannot exercise concurrency at all). Many real OS threads register, + * context-enter, enter/exit a blocked run, then unregister in a tight loop + * while a separate thread concurrently restarts the registry — the same + * registry-to-tracker reset coupling flagged as a coordination risk with the + * in-flight TaskBlock work. Run under ASan/TSan, the only failure signal is a + * sanitizer report or crash; there is no useful single-threaded shadow model + * to assert against here, matching the convention in + * stress_threadLifecycle_ut.cpp. + */ +#include "gtest/gtest.h" + +#ifdef __linux__ + +#include "threadFilter.h" +#include "wallClockBlockTracker.h" +#include "threadLocalData.inline.h" +#include "../../main/cpp/gtest_crash_handler.h" + +#include +#include +#include + +static constexpr const char BLOCK_TRACKER_STRESS_TEST_NAME[] = "StressWallClockBlockTracker"; + +static constexpr int kChurnWorkers = 16; +static constexpr int kChurnIterations = 4000; + +static std::atomic g_run{false}; + +static void block_run_churn_worker(ThreadFilter* filter, WallClockBlockTracker* tracker) { + while (!g_run.load(std::memory_order_acquire)) { } + for (int i = 0; i < kChurnIterations && g_run.load(std::memory_order_relaxed); i++) { + ProfiledThread::initCurrentThread(); + ProfiledThread* self = ProfiledThread::current(); + EXPECT_NE(nullptr, self); + if (!self) return; + + ThreadFilter::SlotID slot = filter->registerThread(self->tid()); + if (slot >= 0) { + self->setFilterSlotId(slot); + filter->add(self->tid(), slot); + + OSThreadState state = (i & 1) ? OSThreadState::SLEEPING : OSThreadState::CONDVAR_WAIT; + u64 token = tracker->enterBlockedRun(filter, slot, state); + std::this_thread::yield(); + if (token != 0) { + // Either exit path may lose the race against a concurrent registry + // reset (which is expected and fine); neither may crash or corrupt. + if (i & 1) { + tracker->exitBlockedRun(slot, WallClockBlockTracker::tokenGeneration(token)); + } else { + tracker->exitBlockedRun(slot); + } + } + + filter->remove(slot); + filter->unregisterThread(slot); + } + self->setFilterSlotId(-1); + ProfiledThread::release(); + } +} + +// Periodically re-activates unfiltered tracking, driving +// resetRegistrationsLocked() -> WallClockBlockTracker::resetAll() while +// churn workers are concurrently mid-registration/mid-block-run. +static void registry_restart_thread(ThreadFilter* filter) { + while (g_run.load(std::memory_order_relaxed)) { + filter->init("", /*track_unfiltered_wall=*/true); + std::this_thread::yield(); + } +} + +TEST(StressWallClockBlockTracker, ChurnOnly) { + installGtestCrashHandler(); + + ThreadFilter filter; + WallClockBlockTracker tracker; + filter.setBlockTracker(&tracker); + filter.init("", /*track_unfiltered_wall=*/true); + + g_run.store(true, std::memory_order_release); + std::vector workers; + for (int t = 0; t < kChurnWorkers; t++) { + workers.emplace_back(block_run_churn_worker, &filter, &tracker); + } + for (auto& w : workers) { + w.join(); + } + g_run.store(false); + + restoreDefaultSignalHandlers(); + SUCCEED(); +} + +TEST(StressWallClockBlockTracker, ChurnDuringConcurrentRegistryRestart) { + installGtestCrashHandler(); + + ThreadFilter filter; + WallClockBlockTracker tracker; + filter.setBlockTracker(&tracker); + filter.init("", /*track_unfiltered_wall=*/true); + + g_run.store(true, std::memory_order_release); + std::thread restarter(registry_restart_thread, &filter); + std::vector workers; + for (int t = 0; t < kChurnWorkers; t++) { + workers.emplace_back(block_run_churn_worker, &filter, &tracker); + } + for (auto& w : workers) { + w.join(); + } + g_run.store(false); + restarter.join(); + + restoreDefaultSignalHandlers(); + SUCCEED(); +} + +#endif // __linux__ diff --git a/ddprof-lib/src/test/cpp/threadFilter_lifecycle_ut.cpp b/ddprof-lib/src/test/cpp/threadFilter_lifecycle_ut.cpp index dab8333f04..279784d33b 100644 --- a/ddprof-lib/src/test/cpp/threadFilter_lifecycle_ut.cpp +++ b/ddprof-lib/src/test/cpp/threadFilter_lifecycle_ut.cpp @@ -60,7 +60,7 @@ TEST(ThreadFilterLifecycle, UnregisterRacesClearActive) { ProfiledThread::release(); continue; } - ThreadFilter::SlotID slot = filter.registerThread(); + ThreadFilter::SlotID slot = filter.registerThread(self->tid()); if (slot >= 0) { filter.add(self->tid(), slot); filter.remove(slot); @@ -113,7 +113,7 @@ TEST(ThreadFilterLifecycle, RegisterRacesInit) { ProfiledThread::release(); continue; } - ThreadFilter::SlotID slot = filter.registerThread(); + ThreadFilter::SlotID slot = filter.registerThread(self->tid()); if (slot >= 0) { // filter may have been disabled between the enabled() check and // registerThread(); the slot is still valid and must be released. diff --git a/ddprof-lib/src/test/cpp/threadFilter_ut.cpp b/ddprof-lib/src/test/cpp/threadFilter_ut.cpp index 4608e11697..e0234135d2 100644 --- a/ddprof-lib/src/test/cpp/threadFilter_ut.cpp +++ b/ddprof-lib/src/test/cpp/threadFilter_ut.cpp @@ -1,5 +1,5 @@ /* - * Copyright 2025 Datadog, Inc + * Copyright 2025, 2026 Datadog, Inc * * Licensed under the Apache License, Version 2.0 (the "License"); * you may not use this file except in compliance with the License. @@ -15,7 +15,10 @@ */ #include +#include "counters.h" +#include "nativeMem.h" #include "threadFilter.h" +#include "wallClockBlockTracker.h" #include "../../main/cpp/gtest_crash_handler.h" #include #include @@ -34,25 +37,29 @@ class ThreadFilterTest : public ::testing::Test { installGtestCrashHandler(); filter = std::make_unique(); filter->init("enabled"); // Enable filtering with non-empty string + tracker = std::make_unique(); + filter->setBlockTracker(tracker.get()); } void TearDown() override { filter.reset(); + tracker.reset(); // Restore default signal handlers restoreDefaultSignalHandlers(); } std::unique_ptr filter; + std::unique_ptr tracker; }; // Basic functionality tests TEST_F(ThreadFilterTest, BasicRegisterAndAccept) { EXPECT_TRUE(filter->enabled()); - int slot_id = filter->registerThread(); + int slot_id = filter->registerThread(1234); EXPECT_GE(slot_id, 0); - // Initially should not accept (no tid added) + // Initially should not accept (not yet in the context window) EXPECT_FALSE(filter->accept(slot_id)); // Add tid and test accept @@ -86,7 +93,7 @@ TEST_F(ThreadFilterTest, EmptyStringDisablesFilter) { EXPECT_TRUE(empty_filter.accept(999999)); // When disabled, registerThread() blocks new registrations - EXPECT_EQ(empty_filter.registerThread(), -1); + EXPECT_EQ(empty_filter.registerThread(1234), -1); } TEST_F(ThreadFilterTest, InvalidSlotHandling) { @@ -106,7 +113,7 @@ TEST_F(ThreadFilterTest, ValidSlotIDContract) { std::vector slot_ids; for (int i = 0; i < 100; i++) { - int slot_id = filter->registerThread(); + int slot_id = filter->registerThread(i + 10000); ASSERT_GE(slot_id, 0) << "registerThread() returned invalid slot_id: " << slot_id; ASSERT_LT(slot_id, ThreadFilter::kMaxThreads) << "slot_id out of range: " << slot_id; @@ -130,10 +137,10 @@ TEST_F(ThreadFilterTest, MaxCapacityReached) { // Register up to the maximum for (int i = 0; i < ThreadFilter::kMaxThreads; i++) { - int slot_id = filter->registerThread(); + int slot_id = filter->registerThread(i + 1000); // Use unique tids if (slot_id >= 0) { slot_ids.push_back(slot_id); - filter->add(i + 1000, slot_id); // Use unique tids + filter->add(i + 1000, slot_id); } } @@ -143,9 +150,17 @@ TEST_F(ThreadFilterTest, MaxCapacityReached) { // Should have registered all slots EXPECT_EQ(slot_ids.size(), ThreadFilter::kMaxThreads); - // Next registration should fail - int overflow_slot = filter->registerThread(); + // Next registration should fail and make capacity loss observable +#ifdef COUNTERS + long long capacity_failures_before = + Counters::getCounter(THREAD_REGISTRY_CAPACITY_EXHAUSTED); +#endif + int overflow_slot = filter->registerThread(1000 + ThreadFilter::kMaxThreads); EXPECT_EQ(overflow_slot, -1); +#ifdef COUNTERS + EXPECT_EQ(capacity_failures_before + 1, + Counters::getCounter(THREAD_REGISTRY_CAPACITY_EXHAUSTED)); +#endif // Verify all registered slots work std::vector collected_tids; @@ -163,14 +178,14 @@ TEST_F(ThreadFilterTest, RecoveryAfterMaxCapacity) { // Fill to capacity for (int i = 0; i < ThreadFilter::kMaxThreads; i++) { - int slot_id = filter->registerThread(); + int slot_id = filter->registerThread(i + 2000); ASSERT_GE(slot_id, 0); slot_ids.push_back(slot_id); filter->add(i + 2000, slot_id); } // Should fail to register more - EXPECT_EQ(filter->registerThread(), -1); + EXPECT_EQ(filter->registerThread(100000), -1); // Unregister half the slots int slots_to_free = ThreadFilter::kMaxThreads / 2; @@ -182,17 +197,19 @@ TEST_F(ThreadFilterTest, RecoveryAfterMaxCapacity) { // Should be able to register new slots again std::vector new_slot_ids; for (int i = 0; i < slots_to_free; i++) { - int slot_id = filter->registerThread(); + int slot_id = filter->registerThread(i + 5000); EXPECT_GE(slot_id, 0) << "Failed to register slot " << i << " after freeing"; new_slot_ids.push_back(slot_id); - filter->add(i + 3000, slot_id); + // Keep replacement identities disjoint from the still-live 3024..4047 + // range. The registry intentionally rejects two slots for one native TID. + filter->add(i + 5000, slot_id); } // Verify we can still register up to capacity EXPECT_EQ(new_slot_ids.size(), slots_to_free); // Should fail again when at capacity - EXPECT_EQ(filter->registerThread(), -1); + EXPECT_EQ(filter->registerThread(100001), -1); // Verify collect works correctly std::vector collected_tids; @@ -210,7 +227,7 @@ TEST_F(ThreadFilterTest, FreeListStressTest) { // Register a batch for (int i = 0; i < batch_size; i++) { - int slot_id = filter->registerThread(); + int slot_id = filter->registerThread(iter * batch_size + i); ASSERT_GE(slot_id, 0); slot_ids.push_back(slot_id); filter->add(iter * batch_size + i, slot_id); @@ -247,7 +264,7 @@ TEST_F(ThreadFilterTest, ConcurrentMaxCapacityStress) { for (int t = 0; t < num_threads; t++) { threads.emplace_back([&, t]() { for (int i = 0; i < slots_per_thread + 10; i++) { // Try to over-register - int slot_id = filter->registerThread(); + int slot_id = filter->registerThread(t * 1000 + i); if (slot_id >= 0) { thread_slots[t].push_back(slot_id); filter->add(t * 1000 + i, slot_id); @@ -286,7 +303,7 @@ TEST_F(ThreadFilterTest, ChunkBoundaryBehavior) { int slots_to_register = ThreadFilter::kChunkSize * 3 + 10; // 3+ chunks for (int i = 0; i < slots_to_register; i++) { - int slot_id = filter->registerThread(); + int slot_id = filter->registerThread(i + 5000); ASSERT_GE(slot_id, 0) << "Failed at slot " << i; slot_ids.push_back(slot_id); filter->add(i + 5000, slot_id); @@ -317,7 +334,7 @@ TEST_F(ThreadFilterTest, ConcurrentAddRemoveAccept) { // Pre-register slots for each thread std::vector slot_ids(num_threads); for (int i = 0; i < num_threads; i++) { - slot_ids[i] = filter->registerThread(); + slot_ids[i] = filter->registerThread(i + 6000); ASSERT_GE(slot_ids[i], 0); } @@ -373,7 +390,7 @@ TEST_F(ThreadFilterTest, FreeListExhaustionRecovery) { // Register many slots for (int i = 0; i < ThreadFilter::kFreeListSize + 100; i++) { - int slot_id = filter->registerThread(); + int slot_id = filter->registerThread(i + 7000); if (slot_id >= 0) { slot_ids.push_back(slot_id); filter->add(i + 7000, slot_id); @@ -390,7 +407,7 @@ TEST_F(ThreadFilterTest, FreeListExhaustionRecovery) { // Try to register new slots - should reuse from free list std::vector new_slot_ids; for (int i = 0; i < 100; i++) { - int slot_id = filter->registerThread(); + int slot_id = filter->registerThread(i + 8000); EXPECT_GE(slot_id, 0) << "Failed to reuse slot " << i; new_slot_ids.push_back(slot_id); filter->add(i + 8000, slot_id); @@ -414,7 +431,7 @@ TEST_F(ThreadFilterTest, PerformanceRegression) { // Pre-register slots std::vector slot_ids; for (int i = 0; i < 100; i++) { - int slot_id = filter->registerThread(); + int slot_id = filter->registerThread(i); ASSERT_GE(slot_id, 0); slot_ids.push_back(slot_id); } @@ -424,7 +441,7 @@ TEST_F(ThreadFilterTest, PerformanceRegression) { // Perform many add/accept/remove operations for (int i = 0; i < num_operations; i++) { int slot_id = slot_ids[i % slot_ids.size()]; - filter->add(i, slot_id); + filter->add(i % slot_ids.size(), slot_id); bool accepted = filter->accept(slot_id); EXPECT_TRUE(accepted); filter->remove(slot_id); @@ -440,6 +457,36 @@ TEST_F(ThreadFilterTest, PerformanceRegression) { // Should be fast - less than 200ns per operation is reasonable for this complex test EXPECT_LT(duration.count() * 1000.0 / num_operations, 200.0); // 200ns per op max } + +// Isolates the cost of add()/remove() (i.e. Slot::enterContextWindow()/ +// exitContextWindow()) from accept()'s TID hashing, since this pair runs on +// every context-window transition for every context-filtered recording. +TEST_F(ThreadFilterTest, ContextWindowEnterExitPerformance) { + const int num_operations = 1000000; + + constexpr int tid = 4242; + int slot_id = filter->registerThread(tid); + ASSERT_GE(slot_id, 0); + + auto start = std::chrono::high_resolution_clock::now(); + + for (int i = 0; i < num_operations; i++) { + filter->add(tid, slot_id); + filter->remove(slot_id); + } + + auto end = std::chrono::high_resolution_clock::now(); + auto duration = std::chrono::duration_cast(end - start); + double ns_per_op = (double)duration.count() * 1000.0 / (num_operations * 2); + + fprintf(stderr, "ContextWindow enter/exit: %d pairs in %ld microseconds (%.2f ns/op)\n", + num_operations, duration.count(), ns_per_op); + + // A plain load + release store per call should stay well under a locked + // RMW's cost; this is a loose ceiling to catch a regression back to CAS + // or worse, not a tight performance contract. + EXPECT_LT(ns_per_op, 50.0); +} #endif // NDEBUG // Collect behavior with mixed states @@ -449,7 +496,7 @@ TEST_F(ThreadFilterTest, CollectMixedStates) { // Register slots and add some tids, leave others empty for (int i = 0; i < 50; i++) { - int slot_id = filter->registerThread(); + int slot_id = filter->registerThread(i + 9000); ASSERT_GE(slot_id, 0); slot_ids.push_back(slot_id); @@ -474,15 +521,15 @@ TEST_F(ThreadFilterTest, CollectMixedStates) { } TEST_F(ThreadFilterTest, ClearActiveDropsPreviousRecordingMembership) { - int stale_slot = filter->registerThread(); - int current_slot = filter->registerThread(); + int stale_slot = filter->registerThread(1111); + int current_slot = filter->registerThread(2222); ASSERT_GE(stale_slot, 0); ASSERT_GE(current_slot, 0); filter->add(1111, stale_slot); filter->add(2222, current_slot); - filter->enterBlockedRun(stale_slot, OSThreadState::SLEEPING); - ThreadFilter::Slot *stale = filter->slotForId(stale_slot); + tracker->enterBlockedRun(filter.get(), stale_slot, OSThreadState::SLEEPING); + WallClockBlockTracker::BlockState *stale = tracker->slotForId(stale_slot); ASSERT_NE(nullptr, stale); stale->markSampledThisRun(OSThreadState::SLEEPING); @@ -503,52 +550,653 @@ TEST_F(ThreadFilterTest, ClearActiveDropsPreviousRecordingMembership) { EXPECT_EQ(2222, collected_tids[0]); } -TEST_F(ThreadFilterTest, GenerationCheckedExitDoesNotClearAnotherOwner) { - int slot_id = filter->registerThread(); +// GenerationCheckedExitDoesNotClearAnotherOwner and NewGenerationRejectsStaleToken +// moved to wallClockBlockTracker_ut.cpp - they exercise pure WallClockBlockTracker +// generation-token behavior with no ThreadFilter-identity interaction. + +TEST_F(ThreadFilterTest, TokenRoundTripPreservesHighGenerationBit) { + ThreadFilter::SlotID slot_id = 7; + u32 generation = 0x80000001u; + u64 token = WallClockBlockTracker::encodeBlockRunToken(slot_id, generation); + int64_t java_token = static_cast(token); + + EXPECT_LT(java_token, 0); + EXPECT_EQ(slot_id, WallClockBlockTracker::tokenSlotId(static_cast(java_token))); + EXPECT_EQ(generation, WallClockBlockTracker::tokenGeneration(static_cast(java_token))); +} + +class ThreadRegistryTest : public ::testing::Test { +protected: + void SetUp() override { + registry.init("", true); + registry.setBlockTracker(&tracker); + } + + ThreadFilter registry; + WallClockBlockTracker tracker; +}; + +TEST_F(ThreadRegistryTest, UnfilteredTrackingSeparatesRegistrationFromContextWindow) { + + EXPECT_TRUE(registry.registryActive()); + EXPECT_TRUE(registry.unfilteredWallTrackingActive()); + EXPECT_FALSE(registry.enabled()); + + int slot_id = registry.registerThread(1234); ASSERT_GE(slot_id, 0); + ThreadFilter::Slot* slot = registry.slotForId(slot_id); + ASSERT_NE(nullptr, slot); + EXPECT_EQ(1234, slot->nativeTid()); + EXPECT_FALSE(slot->inContextWindow()); + EXPECT_EQ(slot, registry.lookupByTid(1234)); - u64 first_token = filter->enterBlockedRun(slot_id, OSThreadState::SLEEPING); - ASSERT_NE(0ULL, first_token); - EXPECT_EQ(0ULL, filter->enterBlockedRun(slot_id, OSThreadState::CONDVAR_WAIT)); + std::vector context; + registry.collect(context); + EXPECT_TRUE(context.empty()); + + registry.add(1234, slot_id); + registry.collect(context); + ASSERT_EQ(1u, context.size()); + EXPECT_EQ(1234, context[0].tid); + + registry.remove(slot_id); + EXPECT_FALSE(slot->inContextWindow()); + EXPECT_EQ(slot, registry.lookupByTid(1234)); + registry.collect(context); + EXPECT_TRUE(context.empty()); +} - ThreadFilter::Slot *slot = filter->slotForId(slot_id); +TEST_F(ThreadRegistryTest, RegisteringKnownTidReturnsExistingSlotWithoutMutation) { + constexpr int tid = 4321; + int slot_id = registry.registerThread(tid); + ASSERT_GE(slot_id, 0); + ThreadFilter::Slot* slot = registry.slotForId(slot_id); ASSERT_NE(nullptr, slot); - EXPECT_EQ(OSThreadState::SLEEPING, slot->activeBlockState()); - EXPECT_FALSE(filter->exitBlockedRun(slot_id, ThreadFilter::tokenGeneration(first_token) + 1)); - EXPECT_EQ(OSThreadState::SLEEPING, slot->activeBlockState()); + u64 token = tracker.enterBlockedRun(®istry, slot_id, OSThreadState::SLEEPING); + ASSERT_NE(0ULL, token); + WallClockBlockTracker::BlockState* block_slot = tracker.slotForId(slot_id); + ASSERT_NE(nullptr, block_slot); + block_slot->markSampledThisRun(OSThreadState::SLEEPING); + + u64 lifecycle_generation = slot->lifecycleGeneration(); + EXPECT_EQ(slot_id, registry.registerThread(tid)); + EXPECT_EQ(slot, registry.lookupByTid(tid)); + EXPECT_EQ(lifecycle_generation, slot->lifecycleGeneration()); + EXPECT_EQ(OSThreadState::SLEEPING, block_slot->activeBlockState()); + EXPECT_TRUE(block_slot->sampledThisRun()); + EXPECT_TRUE(tracker.exitBlockedRun( + slot_id, WallClockBlockTracker::tokenGeneration(token))); +} + +TEST_F(ThreadRegistryTest, UnregisterByTidForUnknownTidIsNoOp) { + constexpr int registered_tid = 1111; + constexpr int unknown_tid = 2222; + int slot_id = registry.registerThread(registered_tid); + ASSERT_GE(slot_id, 0); + + registry.unregisterThreadByTid(unknown_tid); - EXPECT_TRUE(filter->exitBlockedRun(slot_id, ThreadFilter::tokenGeneration(first_token))); - EXPECT_EQ(OSThreadState::UNKNOWN, slot->activeBlockState()); + // The unrelated, already-registered slot must be untouched. + EXPECT_NE(nullptr, registry.lookupByTid(registered_tid)); + EXPECT_EQ(nullptr, registry.lookupByTid(unknown_tid)); } -TEST_F(ThreadFilterTest, NewGenerationRejectsStaleToken) { - int slot_id = filter->registerThread(); +TEST_F(ThreadRegistryTest, UnregisterByTidFreesTheMatchingSlot) { + constexpr int tid = 3333; + int slot_id = registry.registerThread(tid); ASSERT_GE(slot_id, 0); + ASSERT_NE(nullptr, registry.lookupByTid(tid)); - u64 stale_token = filter->enterBlockedRun(slot_id, OSThreadState::SLEEPING); - ASSERT_NE(0ULL, stale_token); - EXPECT_TRUE(filter->exitBlockedRun(slot_id, ThreadFilter::tokenGeneration(stale_token))); + registry.unregisterThreadByTid(tid); - u64 current_token = filter->enterBlockedRun(slot_id, OSThreadState::CONDVAR_WAIT); - ASSERT_NE(0ULL, current_token); - EXPECT_NE(ThreadFilter::tokenGeneration(stale_token), - ThreadFilter::tokenGeneration(current_token)); + EXPECT_EQ(nullptr, registry.lookupByTid(tid)); + // The freed slot must be available for reuse rather than leaked. + int reused_slot_id = registry.registerThread(tid + 1); + EXPECT_EQ(slot_id, reused_slot_id); +} + +TEST_F(ThreadRegistryTest, LookupByTidPopulatesOutSlotId) { + constexpr int tid = 4444; + int slot_id = registry.registerThread(tid); + ASSERT_GE(slot_id, 0); - ThreadFilter::Slot *slot = filter->slotForId(slot_id); + ThreadFilter::SlotID found_slot_id = -1; + ThreadFilter::Slot* slot = registry.lookupByTid(tid, &found_slot_id); ASSERT_NE(nullptr, slot); - EXPECT_FALSE(filter->exitBlockedRun(slot_id, ThreadFilter::tokenGeneration(stale_token))); - EXPECT_EQ(OSThreadState::CONDVAR_WAIT, slot->activeBlockState()); - EXPECT_TRUE(filter->exitBlockedRun(slot_id, ThreadFilter::tokenGeneration(current_token))); + EXPECT_EQ(slot_id, found_slot_id); + + found_slot_id = -1; + EXPECT_EQ(nullptr, registry.lookupByTid(tid + 1, &found_slot_id)); + EXPECT_EQ(-1, found_slot_id); } -TEST_F(ThreadFilterTest, TokenRoundTripPreservesHighGenerationBit) { - ThreadFilter::SlotID slot_id = 7; - u32 generation = 0x80000001u; - u64 token = ThreadFilter::encodeBlockRunToken(slot_id, generation); - int64_t java_token = static_cast(token); +// add() validates ownership against the slot's published tid. The two tests +// below cover the ways a cached slot id goes stale: the slot belongs to another +// thread, or an unfiltered init() reset it (and it may since have been handed +// out again). +TEST_F(ThreadRegistryTest, AddRejectsSlotOwnedByAnotherTid) { + constexpr int tid = 5555; + int slot_a = registry.registerThread(tid); + int slot_b = registry.registerThread(tid + 1); + ASSERT_GE(slot_a, 0); + ASSERT_GE(slot_b, 0); + ASSERT_NE(slot_a, slot_b); + ThreadFilter::Slot* slot_b_ptr = registry.slotForId(slot_b); + ASSERT_NE(nullptr, slot_b_ptr); + u64 state_before = slot_b_ptr->rawContextWindowState(); - EXPECT_LT(java_token, 0); - EXPECT_EQ(slot_id, ThreadFilter::tokenSlotId(static_cast(java_token))); - EXPECT_EQ(generation, ThreadFilter::tokenGeneration(static_cast(java_token))); + // A stale cached slot id must never let one thread write another + // thread's context-window state. + EXPECT_FALSE(registry.add(tid, slot_b)); + EXPECT_FALSE(slot_b_ptr->inContextWindow()); + EXPECT_EQ(state_before, slot_b_ptr->rawContextWindowState()); + EXPECT_EQ(registry.slotForId(slot_a), registry.lookupByTid(tid)); + EXPECT_EQ(slot_b_ptr, registry.lookupByTid(tid + 1)); + + EXPECT_TRUE(registry.add(tid + 1, slot_b)); + EXPECT_TRUE(slot_b_ptr->inContextWindow()); +} + +TEST_F(ThreadRegistryTest, StaleAddAfterUnfilteredResetDoesNotResurrectSlot) { + constexpr int stale_tid = 7001; + constexpr int new_tid = 7002; + int stale_slot = registry.registerThread(stale_tid); + ASSERT_GE(stale_slot, 0); + ThreadFilter::Slot* slot = registry.slotForId(stale_slot); + ASSERT_NE(nullptr, slot); + + // A new unfiltered recording resets every registration; the thread still + // holds its old slot id in TLS. + registry.init("", true); + ASSERT_EQ(-1, slot->nativeTid()); + + // The allocator now considers the slot free: add() must not re-publish + // the stale tid into it. + EXPECT_FALSE(registry.add(stale_tid, stale_slot)); + EXPECT_EQ(-1, slot->nativeTid()); + EXPECT_EQ(nullptr, registry.lookupByTid(stale_tid)); + EXPECT_FALSE(slot->inContextWindow()); + + // Once the slot is handed to another thread, the stale owner still must + // not mark it in-context. + int reassigned = registry.registerThread(new_tid); + ASSERT_EQ(stale_slot, reassigned); + EXPECT_FALSE(registry.add(stale_tid, stale_slot)); + EXPECT_EQ(new_tid, slot->nativeTid()); + EXPECT_FALSE(slot->inContextWindow()); + EXPECT_EQ(nullptr, registry.lookupByTid(stale_tid)); + + // The stale thread re-registers into a slot of its own. + int fresh = registry.registerThread(stale_tid); + ASSERT_GE(fresh, 0); + EXPECT_NE(stale_slot, fresh); + EXPECT_TRUE(registry.add(stale_tid, fresh)); +} + +TEST_F(ThreadRegistryTest, RegisterThreadRejectsNegativeTid) { + EXPECT_EQ(-1, registry.registerThread(-1)); + EXPECT_FALSE(registry.add(-1, 0)); +} + +TEST_F(ThreadRegistryTest, ConcurrentSameTidRegistrationConvergesOnOneSlot) { + constexpr int thread_count = 32; + constexpr int tid = 8765; + std::atomic ready{0}; + std::atomic start{false}; + std::vector slots(thread_count, -1); + std::vector threads; + threads.reserve(thread_count); + + for (int i = 0; i < thread_count; ++i) { + threads.emplace_back([&, i] { + ready.fetch_add(1, std::memory_order_release); + while (!start.load(std::memory_order_acquire)) { + std::this_thread::yield(); + } + slots[i] = registry.registerThread(tid); + }); + } + + while (ready.load(std::memory_order_acquire) != thread_count) { + std::this_thread::yield(); + } + start.store(true, std::memory_order_release); + for (std::thread& thread : threads) { + thread.join(); + } + + ASSERT_GE(slots[0], 0); + for (int slot_id : slots) { + EXPECT_EQ(slots[0], slot_id); + } + ThreadFilter::Slot* slot = registry.slotForId(slots[0]); + ASSERT_NE(nullptr, slot); + EXPECT_EQ(tid, slot->nativeTid()); + EXPECT_EQ(slot, registry.lookupByTid(tid)); +} + +TEST_F(ThreadRegistryTest, ContextWindowTransitionsAreIdempotent) { + int slot_id = registry.registerThread(5678); + ASSERT_GE(slot_id, 0); + ThreadFilter::Slot* slot = registry.slotForId(slot_id); + ASSERT_NE(nullptr, slot); + + u64 initial_epoch = slot->contextWindowEpoch(); + registry.add(5678, slot_id); + EXPECT_TRUE(slot->inContextWindow()); + EXPECT_EQ(initial_epoch + 1, slot->contextWindowEpoch()); + + registry.add(5678, slot_id); + EXPECT_EQ(initial_epoch + 1, slot->contextWindowEpoch()); + + registry.remove(slot_id); + EXPECT_FALSE(slot->inContextWindow()); + EXPECT_EQ(initial_epoch + 2, slot->contextWindowEpoch()); + + registry.remove(slot_id); + EXPECT_EQ(initial_epoch + 2, slot->contextWindowEpoch()); +} + +TEST_F(ThreadRegistryTest, SlotReuseChangesLifecycleGenerationAndTidMapping) { + int slot_id = registry.registerThread(1111); + ASSERT_GE(slot_id, 0); + ThreadFilter::Slot* slot = registry.slotForId(slot_id); + ASSERT_NE(nullptr, slot); + u64 first_generation = slot->lifecycleGeneration(); + + registry.unregisterThread(slot_id); + EXPECT_EQ(nullptr, registry.lookupByTid(1111)); + + int reused_id = registry.registerThread(2222); + ASSERT_EQ(slot_id, reused_id); + EXPECT_GT(slot->lifecycleGeneration(), first_generation); + EXPECT_EQ(nullptr, registry.lookupByTid(1111)); + EXPECT_EQ(slot, registry.lookupByTid(2222)); +} + +TEST_F(ThreadRegistryTest, ContextTransitionInvalidatesOwnedRunSuppression) { + int slot_id = registry.registerThread(3333); + ASSERT_GE(slot_id, 0); + ThreadFilter::Slot* slot = registry.slotForId(slot_id); + ASSERT_NE(nullptr, slot); + + u64 token = tracker.enterBlockedRun(®istry, slot_id, OSThreadState::SLEEPING); + ASSERT_NE(0u, token); + WallClockBlockTracker::BlockState* block_slot = tracker.slotForId(slot_id); + ASSERT_NE(nullptr, block_slot); + block_slot->markSampledThisRun(OSThreadState::SLEEPING); + EXPECT_TRUE(block_slot->activeBlockRemainedOutsideContextWindow(slot)); + + registry.add(3333, slot_id); + registry.remove(slot_id); + EXPECT_FALSE(block_slot->activeBlockRemainedOutsideContextWindow(slot)); + + ThreadEntry entry{3333, slot, slot_id, slot->lifecycleGeneration(), + slot->recordingEpoch()}; + EXPECT_FALSE(tracker.shouldSuppressOwnedBlock(®istry, entry)); +} + +TEST_F(ThreadRegistryTest, UnfilteredSuppressionValidatesIdentityAndLifecycle) { + int slot_id = registry.registerThread(4444); + ASSERT_GE(slot_id, 0); + ThreadFilter::Slot* slot = registry.slotForId(slot_id); + ASSERT_NE(nullptr, slot); + + u64 token = tracker.enterBlockedRun(®istry, slot_id, OSThreadState::SLEEPING); + ASSERT_NE(0u, token); + WallClockBlockTracker::BlockState* block_slot = tracker.slotForId(slot_id); + ASSERT_NE(nullptr, block_slot); + block_slot->markSampledThisRun(OSThreadState::SLEEPING); + ThreadEntry entry{4444, slot, slot_id, slot->lifecycleGeneration(), + slot->recordingEpoch()}; + EXPECT_TRUE(tracker.shouldSuppressOwnedBlock(®istry, entry)); + + ThreadEntry wrong_tid{4445, slot, slot_id, entry.lifecycle_generation, + entry.recording_epoch}; + EXPECT_FALSE(tracker.shouldSuppressOwnedBlock(®istry, wrong_tid)); + ThreadEntry stale_generation{4444, slot, slot_id, entry.lifecycle_generation + 1, + entry.recording_epoch}; + EXPECT_FALSE(tracker.shouldSuppressOwnedBlock(®istry, stale_generation)); + + EXPECT_TRUE(tracker.exitBlockedRun( + slot_id, WallClockBlockTracker::tokenGeneration(token))); + EXPECT_FALSE(tracker.shouldSuppressOwnedBlock(®istry, entry)); +} + +TEST_F(ThreadRegistryTest, ContextFilteredSuppressionPreservesHistoricalEligibility) { + registry.init("0"); + int slot_id = registry.registerThread(5555); + ASSERT_GE(slot_id, 0); + registry.add(5555, slot_id); + ThreadFilter::Slot* slot = registry.slotForId(slot_id); + ASSERT_NE(nullptr, slot); + + u64 token = tracker.enterBlockedRun(®istry, slot_id, OSThreadState::SLEEPING); + ASSERT_NE(0u, token); + WallClockBlockTracker::BlockState* block_slot = tracker.slotForId(slot_id); + ASSERT_NE(nullptr, block_slot); + block_slot->markSampledThisRun(OSThreadState::SLEEPING); + ThreadEntry entry{5555, slot, slot_id, slot->lifecycleGeneration(), + slot->recordingEpoch()}; + EXPECT_TRUE(tracker.shouldSuppressOwnedBlock(®istry, entry)); +} + +TEST_F(ThreadRegistryTest, ConcurrentTidReuseInvalidatesSuppressionSnapshot) { + constexpr int tid = 5601; + int slot_id = registry.registerThread(tid); + ASSERT_GE(slot_id, 0); + ThreadFilter::Slot* slot = registry.slotForId(slot_id); + ASSERT_NE(nullptr, slot); + ASSERT_NE(0u, tracker.enterBlockedRun(®istry, slot_id, OSThreadState::SLEEPING)); + tracker.slotForId(slot_id)->markSampledThisRun(OSThreadState::SLEEPING); + ThreadEntry stale{tid, slot, slot_id, slot->lifecycleGeneration(), + slot->recordingEpoch()}; + + struct SnapshotPause { + std::atomic reached{false}; + std::atomic resume{false}; + } pause; + tracker.setSuppressionSnapshotHookForTest( + [](void* raw) { + SnapshotPause* pause = static_cast(raw); + pause->reached.store(true, std::memory_order_release); + while (!pause->resume.load(std::memory_order_acquire)) { + std::this_thread::yield(); + } + }, + &pause); + + std::atomic suppressed{true}; + std::thread reader([&] { + suppressed.store(tracker.shouldSuppressOwnedBlock(®istry, stale), + std::memory_order_release); + }); + auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(5); + while (!pause.reached.load(std::memory_order_acquire) && + std::chrono::steady_clock::now() < deadline) { + std::this_thread::yield(); + } + if (!pause.reached.load(std::memory_order_acquire)) { + pause.resume.store(true, std::memory_order_release); + reader.join(); + tracker.setSuppressionSnapshotHookForTest(nullptr, nullptr); + GTEST_FAIL() << "Suppression reader did not reach the snapshot barrier"; + } + + registry.unregisterThread(slot_id, tid); + int reused_id = registry.registerThread(tid); + ThreadFilter::Slot* reused = registry.slotForId(reused_id); + u64 new_token = tracker.enterBlockedRun(®istry, reused_id, OSThreadState::SLEEPING); + if (reused != nullptr && new_token != 0) { + tracker.slotForId(reused_id)->markSampledThisRun(OSThreadState::SLEEPING); + } + + pause.resume.store(true, std::memory_order_release); + reader.join(); + tracker.setSuppressionSnapshotHookForTest(nullptr, nullptr); + + ASSERT_EQ(slot_id, reused_id); + ASSERT_NE(nullptr, reused); + ASSERT_NE(0u, new_token); + EXPECT_FALSE(suppressed.load(std::memory_order_acquire)); +} + +TEST_F(ThreadRegistryTest, TidIndexRemainsReusableAcrossLongThreadChurn) { + for (int tid = 1; tid <= ThreadFilter::kTidIndexSize * 3; ++tid) { + int slot_id = registry.registerThread(tid); + ASSERT_GE(slot_id, 0) << "tid=" << tid; + ThreadFilter::Slot* slot = registry.slotForId(slot_id); + ASSERT_EQ(slot, registry.lookupByTid(tid)); + registry.unregisterThread(slot_id); + ASSERT_EQ(nullptr, registry.lookupByTid(tid)); + } +} + +TEST_F(ThreadRegistryTest, ConfigurationSeparatesFilterAndUnfilteredTracking) { + registry.init("0", false); + EXPECT_TRUE(registry.enabled()); + EXPECT_TRUE(registry.registryActive()); + EXPECT_FALSE(registry.unfilteredWallTrackingActive()); + + registry.init("", false); + EXPECT_FALSE(registry.enabled()); + EXPECT_FALSE(registry.registryActive()); + EXPECT_FALSE(registry.unfilteredWallTrackingActive()); + + registry.init("", true); + EXPECT_FALSE(registry.enabled()); + EXPECT_TRUE(registry.registryActive()); + EXPECT_TRUE(registry.unfilteredWallTrackingActive()); +} + +TEST_F(ThreadRegistryTest, NewUnfilteredRecordingReclaimsRetainedSlot) { + constexpr int tid = 6101; + int slot_id = registry.registerThread(tid); + ASSERT_GE(slot_id, 0); + ThreadFilter::Slot* slot = registry.slotForId(slot_id); + ASSERT_NE(nullptr, slot); + u64 first_lifecycle_generation = slot->lifecycleGeneration(); + ThreadFilter::RecordingEpoch first_epoch = registry.recordingEpoch(); + ASSERT_NE(0u, first_epoch); + EXPECT_EQ(slot, registry.lookupByTid(tid, first_epoch)); + + u64 token = tracker.enterBlockedRun(®istry, slot_id, OSThreadState::SLEEPING); + ASSERT_NE(0u, token); + tracker.slotForId(slot_id)->markSampledThisRun(OSThreadState::SLEEPING); + ThreadEntry stale{tid, slot, slot_id, slot->lifecycleGeneration(), + slot->recordingEpoch()}; + ASSERT_TRUE(tracker.shouldSuppressOwnedBlock(®istry, stale)); + + registry.init("", true); + ThreadFilter::RecordingEpoch second_epoch = registry.recordingEpoch(); + ASSERT_NE(first_epoch, second_epoch); + EXPECT_EQ(nullptr, registry.lookupByTid(tid, first_epoch)); + EXPECT_EQ(nullptr, registry.lookupByTid(tid, second_epoch)); + EXPECT_EQ(nullptr, registry.lookupByTid(tid)); + EXPECT_EQ(-1, slot->nativeTid()); + EXPECT_GT(slot->lifecycleGeneration(), first_lifecycle_generation); + EXPECT_FALSE(tracker.shouldSuppressOwnedBlock(®istry, stale)); + + EXPECT_EQ(slot_id, registry.registerThread(tid)); + EXPECT_EQ(slot, registry.lookupByTid(tid, second_epoch)); + WallClockBlockTracker::BlockState* block_slot = tracker.slotForId(slot_id); + EXPECT_FALSE(block_slot->sampledThisRun()); + EXPECT_EQ(OSThreadState::UNKNOWN, block_slot->activeBlockState()); + EXPECT_EQ(BlockRunOwner::NONE, block_slot->activeBlockOwner()); +} + +TEST_F(ThreadRegistryTest, NewUnfilteredRecordingReclaimsFullCapacity) { + for (int i = 0; i < ThreadFilter::kMaxThreads; ++i) { + ASSERT_GE(registry.registerThread(10000 + i), 0) << "tid index=" << i; + } + ASSERT_EQ(-1, registry.registerThread(20000)); + + registry.init("", true); + + EXPECT_EQ(nullptr, registry.lookupByTid(10000)); + for (int i = 0; i < ThreadFilter::kMaxThreads; ++i) { + ASSERT_GE(registry.registerThread(30000 + i), 0) << "tid index=" << i; + } + EXPECT_EQ(-1, registry.registerThread(40000)); +} + +TEST_F(ThreadRegistryTest, ExpectedTidProtectsReusedSlotDuringTeardown) { + int slot_id = registry.registerThread(6105); + ASSERT_GE(slot_id, 0); + registry.unregisterThread(slot_id, 9999); + EXPECT_NE(nullptr, registry.lookupByTid(6105)); + + registry.unregisterThread(slot_id, 6105); + EXPECT_EQ(nullptr, registry.lookupByTid(6105)); +} + +TEST_F(ThreadRegistryTest, DeactivationMakesSlotsIneligibleWithoutClearingStorage) { + constexpr int tid = 6106; + int slot_id = registry.registerThread(tid); + ASSERT_GE(slot_id, 0); + ThreadFilter::Slot* slot = registry.slotForId(slot_id); + ASSERT_NE(nullptr, slot); + ThreadFilter::RecordingEpoch epoch = registry.recordingEpoch(); + + registry.deactivateRecording(); + EXPECT_FALSE(registry.registryActive()); + EXPECT_FALSE(registry.unfilteredWallTrackingActive()); + EXPECT_EQ(0u, registry.recordingEpoch()); + EXPECT_EQ(nullptr, registry.lookupByTid(tid, epoch)); + EXPECT_EQ(slot, registry.lookupByTid(tid)); + EXPECT_EQ(-1, registry.registerThread(7777)); +} + +TEST_F(ThreadRegistryTest, RegisterThreadRechecksActiveAfterLockAcquired) { + // Simulates a concurrent deactivateRecording() landing in the window + // between registerThread()'s pre-lock _registry_active check and its + // acquisition of _registry_lock. + registry.setPostActiveCheckHookForTest( + [](void* arg) { static_cast(arg)->deactivateRecording(); }, + ®istry); + int slot_id = registry.registerThread(8888); + registry.setPostActiveCheckHookForTest(nullptr, nullptr); + + EXPECT_EQ(-1, slot_id); +} + +// Counts registerThread() calls that reach _registry_lock acquisition. +static void countLockAttempt(void* arg) { + static_cast*>(arg)->fetch_add(1, std::memory_order_relaxed); +} + +static void fillRegistry(ThreadFilter& registry, int first_tid, std::vector* slot_ids = nullptr) { + for (int i = 0; i < ThreadFilter::kMaxThreads; ++i) { + int slot_id = registry.registerThread(first_tid + i); + ASSERT_GE(slot_id, 0) << "tid index=" << i; + if (slot_ids != nullptr) { + slot_ids->push_back(slot_id); + } + } +} + +TEST_F(ThreadRegistryTest, FullRegistryRejectsUnknownTidWithoutLocking) { + fillRegistry(registry, 10000); + + std::atomic lock_attempts{0}; + registry.setPostActiveCheckHookForTest(countLockAttempt, &lock_attempts); +#ifdef COUNTERS + long long capacity_failures_before = + Counters::getCounter(THREAD_REGISTRY_CAPACITY_EXHAUSTED); +#endif + // Threads past capacity retry on every hook call; none may reach the lock. + for (int i = 0; i < 100; ++i) { + EXPECT_EQ(-1, registry.registerThread(20000 + (i % 4))); + } + registry.setPostActiveCheckHookForTest(nullptr, nullptr); + + EXPECT_EQ(0, lock_attempts.load()); +#ifdef COUNTERS + // Rejections stay observable even though they skip the locked path. + EXPECT_EQ(capacity_failures_before + 100, + Counters::getCounter(THREAD_REGISTRY_CAPACITY_EXHAUSTED)); +#endif +} + +TEST_F(ThreadRegistryTest, FullRegistryStillReturnsExistingSlotForKnownTid) { + std::vector slot_ids; + fillRegistry(registry, 10000, &slot_ids); + + // A thread whose cached slot id was cleared must still recover its own + // slot through the locked path while the registry is full. + EXPECT_EQ(slot_ids[5], registry.registerThread(10005)); + EXPECT_EQ(slot_ids[ThreadFilter::kMaxThreads - 1], + registry.registerThread(10000 + ThreadFilter::kMaxThreads - 1)); +} + +TEST_F(ThreadRegistryTest, FreedSlotReopensRegistrationAfterExhaustion) { + fillRegistry(registry, 10000); + ASSERT_EQ(-1, registry.registerThread(20000)); + + registry.unregisterThreadByTid(10000); + int reused = registry.registerThread(20000); + EXPECT_GE(reused, 0); + EXPECT_NE(nullptr, registry.lookupByTid(20000)); + + // Full again once the freed slot has been taken. + std::atomic lock_attempts{0}; + registry.setPostActiveCheckHookForTest(countLockAttempt, &lock_attempts); + EXPECT_EQ(-1, registry.registerThread(20001)); + registry.setPostActiveCheckHookForTest(nullptr, nullptr); + EXPECT_EQ(0, lock_attempts.load()); +} + +TEST_F(ThreadRegistryTest, UnfilteredResetReopensRegistrationAfterExhaustion) { + fillRegistry(registry, 10000); + ASSERT_EQ(-1, registry.registerThread(20000)); + + registry.init("", true); + + std::atomic lock_attempts{0}; + registry.setPostActiveCheckHookForTest(countLockAttempt, &lock_attempts); + EXPECT_GE(registry.registerThread(20000), 0); + registry.setPostActiveCheckHookForTest(nullptr, nullptr); + EXPECT_EQ(1, lock_attempts.load()); +} + +TEST_F(ThreadFilterTest, FullContextFilterRejectsWithoutLocking) { + // The default context-filter mode shares the same registry and retry path. + for (int i = 0; i < ThreadFilter::kMaxThreads; ++i) { + ASSERT_GE(filter->registerThread(30000 + i), 0) << "tid index=" << i; + } + + std::atomic lock_attempts{0}; + filter->setPostActiveCheckHookForTest(countLockAttempt, &lock_attempts); + EXPECT_EQ(-1, filter->registerThread(40000)); + filter->setPostActiveCheckHookForTest(nullptr, nullptr); + EXPECT_EQ(0, lock_attempts.load()); +} + +// The tid index lives inside the process-lifetime Profiler's ThreadFilter, so +// embedding it would keep it resident from library load even for recordings +// that never activate the registry (no context filter, no unfiltered precheck). +static constexpr long long kTidIndexBytes = + (long long)(ThreadFilter::kTidIndexSize * sizeof(std::atomic)); + +TEST(ThreadFilterTidIndexStorageTest, FilterEmbedsNoTidIndex) { + EXPECT_LT((long long)sizeof(ThreadFilter), kTidIndexBytes); +} + +TEST(ThreadFilterTidIndexStorageTest, TidIndexIsAllocatedOnFirstRegistryActivationOnly) { + ThreadFilter filter; + const long long constructed = NativeMem::live(NM_THREAD_FILTER); + + filter.init(""); // no context filter, no unfiltered tracking: registry stays inactive + ASSERT_FALSE(filter.registryActive()); + EXPECT_EQ(constructed, NativeMem::live(NM_THREAD_FILTER)); + + filter.init("1"); + ASSERT_TRUE(filter.registryActive()); + EXPECT_EQ(constructed + kTidIndexBytes, NativeMem::live(NM_THREAD_FILTER)); + + // Later activations, in either registry mode, reuse the same index. + filter.init("", true); + ASSERT_TRUE(filter.registryActive()); + filter.init("1"); + EXPECT_EQ(constructed + kTidIndexBytes, NativeMem::live(NM_THREAD_FILTER)); + + ThreadFilter::SlotID slot_id = filter.registerThread(9001); + ASSERT_GE(slot_id, 0); + ThreadFilter::SlotID found = -1; + EXPECT_NE(nullptr, filter.lookupByTid(9001, &found)); + EXPECT_EQ(slot_id, found); +} + +TEST(ThreadFilterTidIndexStorageTest, TidLookupsBeforeActivationFindNothing) { + ThreadFilter filter; + const long long constructed = NativeMem::live(NM_THREAD_FILTER); + ThreadFilter::SlotID found = 123; + + EXPECT_EQ(nullptr, filter.lookupByTid(9101, &found)); + EXPECT_EQ(-1, found); + EXPECT_EQ(nullptr, filter.lookupByTid(9101, 1, &found)); + filter.unregisterThreadByTid(9101); // ThreadEnd path, runs for every exiting thread + EXPECT_EQ(-1, filter.registerThread(9101)); + EXPECT_EQ(constructed, NativeMem::live(NM_THREAD_FILTER)); } diff --git a/ddprof-lib/src/test/cpp/wallClockBlockTracker_ut.cpp b/ddprof-lib/src/test/cpp/wallClockBlockTracker_ut.cpp new file mode 100644 index 0000000000..d56af858a6 --- /dev/null +++ b/ddprof-lib/src/test/cpp/wallClockBlockTracker_ut.cpp @@ -0,0 +1,172 @@ +/* + * Copyright 2026, Datadog, Inc + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include +#include "threadFilter.h" +#include "wallClockBlockTracker.h" +#include "../../main/cpp/gtest_crash_handler.h" +#include + +// Test name for crash handler +static constexpr char WALLCLOCK_BLOCK_TRACKER_TEST_NAME[] = "WallClockBlockTrackerTest"; + +class WallClockBlockTrackerTest : public ::testing::Test { +protected: + void SetUp() override { + installGtestCrashHandler(); + filter = std::make_unique(); + filter->init("enabled"); + tracker = std::make_unique(); + filter->setBlockTracker(tracker.get()); + } + + void TearDown() override { + filter.reset(); + tracker.reset(); + restoreDefaultSignalHandlers(); + } + + std::unique_ptr filter; + std::unique_ptr tracker; +}; + +TEST_F(WallClockBlockTrackerTest, GenerationCheckedExitDoesNotClearAnotherOwner) { + int slot_id = filter->registerThread(1234); + ASSERT_GE(slot_id, 0); + + u64 first_token = tracker->enterBlockedRun(filter.get(), slot_id, OSThreadState::SLEEPING); + ASSERT_NE(0ULL, first_token); + EXPECT_EQ(0ULL, tracker->enterBlockedRun(filter.get(), slot_id, OSThreadState::CONDVAR_WAIT)); + + WallClockBlockTracker::BlockState* block_slot = tracker->slotForId(slot_id); + ASSERT_NE(nullptr, block_slot); + EXPECT_EQ(OSThreadState::SLEEPING, block_slot->activeBlockState()); + + EXPECT_FALSE(tracker->exitBlockedRun(slot_id, WallClockBlockTracker::tokenGeneration(first_token) + 1)); + EXPECT_EQ(OSThreadState::SLEEPING, block_slot->activeBlockState()); + + EXPECT_TRUE(tracker->exitBlockedRun(slot_id, WallClockBlockTracker::tokenGeneration(first_token))); + EXPECT_EQ(OSThreadState::UNKNOWN, block_slot->activeBlockState()); +} + +TEST_F(WallClockBlockTrackerTest, NewGenerationRejectsStaleToken) { + int slot_id = filter->registerThread(1234); + ASSERT_GE(slot_id, 0); + + u64 stale_token = tracker->enterBlockedRun(filter.get(), slot_id, OSThreadState::SLEEPING); + ASSERT_NE(0ULL, stale_token); + EXPECT_TRUE(tracker->exitBlockedRun(slot_id, WallClockBlockTracker::tokenGeneration(stale_token))); + + u64 current_token = tracker->enterBlockedRun(filter.get(), slot_id, OSThreadState::CONDVAR_WAIT); + ASSERT_NE(0ULL, current_token); + EXPECT_NE(WallClockBlockTracker::tokenGeneration(stale_token), + WallClockBlockTracker::tokenGeneration(current_token)); + + WallClockBlockTracker::BlockState* block_slot = tracker->slotForId(slot_id); + ASSERT_NE(nullptr, block_slot); + EXPECT_FALSE(tracker->exitBlockedRun(slot_id, WallClockBlockTracker::tokenGeneration(stale_token))); + EXPECT_EQ(OSThreadState::CONDVAR_WAIT, block_slot->activeBlockState()); + EXPECT_TRUE(tracker->exitBlockedRun(slot_id, WallClockBlockTracker::tokenGeneration(current_token))); +} + +// The four tests below verify that ThreadFilter's registry-lifecycle reset +// points each clear the wired WallClockBlockTracker's parallel per-slot +// state, since the two classes coordinate via one non-owning pointer +// (ThreadFilter::setBlockTracker()) rather than a shared struct. +// +// refreshSlotForRecording()'s own resetSlot() call is not covered here: its +// reset branch only fires when a slot's published recording epoch differs +// from the registry's current one while the slot's tid mapping survives - +// unreachable through sequential registerThread()/init() calls, since every +// transition into unfiltered tracking mode (the only way the epoch changes) +// goes through resetRegistrationsLocked() first, which already clears the +// tid mapping. It is only reachable via a registration racing a concurrent +// init(), which the existing ConcurrentSameTidRegistrationConvergesOnOneSlot +// stress test in threadFilter_ut.cpp already exercises for the identity side +// of that race. + +TEST_F(WallClockBlockTrackerTest, RegisterThreadReuseResetsBlockState) { + int slot_id = filter->registerThread(1001); + ASSERT_GE(slot_id, 0); + ASSERT_NE(0ULL, tracker->enterBlockedRun(filter.get(), slot_id, OSThreadState::SLEEPING)); + tracker->slotForId(slot_id)->markSampledThisRun(OSThreadState::SLEEPING); + + filter->unregisterThread(slot_id); + int reused_id = filter->registerThread(1002); + ASSERT_EQ(slot_id, reused_id); + + WallClockBlockTracker::BlockState* block_slot = tracker->slotForId(reused_id); + EXPECT_EQ(OSThreadState::UNKNOWN, block_slot->activeBlockState()); + EXPECT_FALSE(block_slot->sampledThisRun()); + EXPECT_EQ(BlockRunOwner::NONE, block_slot->activeBlockOwner()); +} + +TEST_F(WallClockBlockTrackerTest, RegisterThreadNewSlotStartsWithClearBlockState) { + int slot_id = filter->registerThread(2001); + ASSERT_GE(slot_id, 0); + WallClockBlockTracker::BlockState* block_slot = tracker->slotForId(slot_id); + ASSERT_NE(nullptr, block_slot); + EXPECT_EQ(OSThreadState::UNKNOWN, block_slot->activeBlockState()); + EXPECT_FALSE(block_slot->sampledThisRun()); + EXPECT_EQ(BlockRunOwner::NONE, block_slot->activeBlockOwner()); +} + +TEST_F(WallClockBlockTrackerTest, UnregisterThreadResetsBlockState) { + int slot_id = filter->registerThread(4001); + ASSERT_GE(slot_id, 0); + ASSERT_NE(0ULL, tracker->enterBlockedRun(filter.get(), slot_id, OSThreadState::SLEEPING)); + tracker->slotForId(slot_id)->markSampledThisRun(OSThreadState::SLEEPING); + + filter->unregisterThread(slot_id); + + WallClockBlockTracker::BlockState* block_slot = tracker->slotForId(slot_id); + EXPECT_EQ(OSThreadState::UNKNOWN, block_slot->activeBlockState()); + EXPECT_FALSE(block_slot->sampledThisRun()); + EXPECT_EQ(BlockRunOwner::NONE, block_slot->activeBlockOwner()); +} + +TEST_F(WallClockBlockTrackerTest, InitResetRegistrationsLockedResetsBlockState) { + ThreadFilter registry; + registry.init("", true); + registry.setBlockTracker(tracker.get()); + + int slot_id = registry.registerThread(5001); + ASSERT_GE(slot_id, 0); + ASSERT_NE(0u, tracker->enterBlockedRun(®istry, slot_id, OSThreadState::SLEEPING)); + tracker->slotForId(slot_id)->markSampledThisRun(OSThreadState::SLEEPING); + + // Re-entering unfiltered tracking mode drives resetRegistrationsLocked(), + // which must clear every slot's block state via a single resetAll() call. + registry.init("", true); + + WallClockBlockTracker::BlockState* block_slot = tracker->slotForId(slot_id); + EXPECT_EQ(OSThreadState::UNKNOWN, block_slot->activeBlockState()); + EXPECT_FALSE(block_slot->sampledThisRun()); +} + +TEST_F(WallClockBlockTrackerTest, ClearActiveResetsBlockState) { + int slot_id = filter->registerThread(6001); + ASSERT_GE(slot_id, 0); + filter->add(6001, slot_id); + ASSERT_NE(0ULL, tracker->enterBlockedRun(filter.get(), slot_id, OSThreadState::SLEEPING)); + tracker->slotForId(slot_id)->markSampledThisRun(OSThreadState::SLEEPING); + + filter->clearActive(); + + WallClockBlockTracker::BlockState* block_slot = tracker->slotForId(slot_id); + EXPECT_EQ(OSThreadState::UNKNOWN, block_slot->activeBlockState()); + EXPECT_FALSE(block_slot->sampledThisRun()); +} diff --git a/ddprof-lib/src/test/cpp/wallClockCandidateSelector_ut.cpp b/ddprof-lib/src/test/cpp/wallClockCandidateSelector_ut.cpp new file mode 100644 index 0000000000..5ade786b17 --- /dev/null +++ b/ddprof-lib/src/test/cpp/wallClockCandidateSelector_ut.cpp @@ -0,0 +1,198 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +#include + +#include "wallClockCandidateSelector.h" + +#include +#include +#include +#include + +static std::vector makeCandidates(size_t count) { + std::vector candidates(count); + std::iota(candidates.begin(), candidates.end(), 0); + return candidates; +} + +TEST(WallClockCandidateSelectorTest, VisitsOnlyTargetSizeWithoutRejections) { + std::vector candidates = makeCandidates(1000); + std::mt19937 generator(1234); + std::set selected; + + WallClockCandidateStats stats = selectWallClockCandidates( + candidates, 10, 40, generator, [&](int tid) { + selected.insert(tid); + return WallClockCandidateOutcome::SIGNAL_SENT; + }); + + EXPECT_EQ(10u, stats.visited); + EXPECT_EQ(10u, stats.slots_consumed); + EXPECT_EQ(0u, stats.precheck_rejected); + EXPECT_EQ(10u, selected.size()); +} + +TEST(WallClockCandidateSelectorTest, PrecheckRejectedCandidatesAreBackfilled) { + std::vector candidates = makeCandidates(100); + std::mt19937 generator(42); + std::set selected; + + WallClockCandidateStats stats = selectWallClockCandidates( + candidates, 8, 100, generator, [&](int tid) { + if ((tid & 1) == 0) { + return WallClockCandidateOutcome::PRECHECK_REJECTED; + } + selected.insert(tid); + return WallClockCandidateOutcome::SIGNAL_SENT; + }); + + EXPECT_EQ(8u, stats.slots_consumed); + EXPECT_EQ(stats.slots_consumed + stats.precheck_rejected, stats.visited); + EXPECT_EQ(8u, selected.size()); + for (int tid : selected) { + EXPECT_EQ(1, tid & 1); + } +} + +TEST(WallClockCandidateSelectorTest, AllPrecheckRejectedCandidatesRespectVisitLimit) { + std::vector candidates = makeCandidates(257); + std::mt19937 generator(7); + std::set visited; + + WallClockCandidateStats stats = selectWallClockCandidates( + candidates, 10, 40, generator, [&](int tid) { + visited.insert(tid); + return WallClockCandidateOutcome::PRECHECK_REJECTED; + }); + + EXPECT_EQ(40u, stats.visited); + EXPECT_EQ(0u, stats.slots_consumed); + EXPECT_EQ(40u, stats.precheck_rejected); + EXPECT_EQ(40u, visited.size()); + EXPECT_TRUE(stats.visit_limit_reached); +} + +TEST(WallClockCandidateSelectorTest, SignalFailureConsumesCapacityWithoutBackfill) { + std::vector candidates{1, 2, 3, 4, 5}; + std::mt19937 generator(17); + int callbacks = 0; + + WallClockCandidateStats stats = selectWallClockCandidates( + candidates, 3, 12, generator, [&](int) { + callbacks++; + return WallClockCandidateOutcome::SIGNAL_FAILED; + }); + + EXPECT_EQ(3, callbacks); + EXPECT_EQ(3u, stats.visited); + EXPECT_EQ(3u, stats.slots_consumed); + EXPECT_EQ(0u, stats.precheck_rejected); +} + +TEST(WallClockCandidateSelectorTest, EmptyBoundsDoNoWork) { + std::vector candidates{1, 2, 3}; + std::vector empty; + std::mt19937 generator(1); + int callbacks = 0; + auto visitor = [&](int) { + callbacks++; + return WallClockCandidateOutcome::SIGNAL_SENT; + }; + + WallClockCandidateStats zero_target = + selectWallClockCandidates(candidates, 0, 3, generator, visitor); + WallClockCandidateStats empty_input = + selectWallClockCandidates(empty, 3, 3, generator, visitor); + WallClockCandidateStats zero_visits = + selectWallClockCandidates(candidates, 3, 0, generator, visitor); + + EXPECT_EQ(0, callbacks); + EXPECT_EQ(0u, zero_target.visited); + EXPECT_EQ(0u, empty_input.visited); + EXPECT_EQ(0u, zero_visits.visited); +} + +TEST(WallClockCandidateSelectorTest, FixedSeedProducesDeterministicTraversal) { + std::vector first = makeCandidates(50); + std::vector second = first; + std::mt19937 first_generator(2026); + std::mt19937 second_generator(2026); + std::vector first_result; + std::vector second_result; + + selectWallClockCandidates(first, 12, 24, first_generator, [&](int tid) { + first_result.push_back(tid); + return WallClockCandidateOutcome::SIGNAL_SENT; + }); + selectWallClockCandidates(second, 12, 24, second_generator, [&](int tid) { + second_result.push_back(tid); + return WallClockCandidateOutcome::SIGNAL_SENT; + }); + + EXPECT_EQ(first_result, second_result); +} + +TEST(WallClockCandidateSelectorTest, RandomizedPrefixRemainsFairAcrossCandidates) { + constexpr int candidate_count = 20; + constexpr int sample_size = 4; + constexpr int rounds = 10000; + std::vector candidates(candidate_count); + std::vector selections(candidate_count, 0); + std::mt19937 generator(2026); + + for (int round = 0; round < rounds; ++round) { + std::iota(candidates.begin(), candidates.end(), 0); + WallClockCandidateStats stats = selectWallClockCandidates( + candidates, sample_size, sample_size, generator, [&](int candidate) { + selections[candidate]++; + return WallClockCandidateOutcome::SIGNAL_SENT; + }); + ASSERT_EQ(sample_size, stats.slots_consumed); + } + + constexpr int expected = rounds * sample_size / candidate_count; + for (int count : selections) { + EXPECT_NEAR(expected, count, expected / 10); + } +} + +TEST(WallClockCandidateSelectorTest, ExhaustingInputDoesNotReportVisitLimit) { + std::vector candidates{1, 2, 3}; + std::mt19937 generator(8); + + WallClockCandidateStats stats = selectWallClockCandidates( + candidates, 5, 20, generator, + [](int) { return WallClockCandidateOutcome::PRECHECK_REJECTED; }); + + EXPECT_EQ(3u, stats.visited); + EXPECT_EQ(3u, stats.precheck_rejected); + EXPECT_FALSE(stats.visit_limit_reached); +} + +TEST(WallClockCandidateSelectorTest, MixedAcceptRejectExhaustsPoolBelowTargetSize) { + std::vector candidates = makeCandidates(20); + std::mt19937 generator(99); + std::set selected; + + // Accept every third candidate (0, 3, 6, ..., 18 -> 7 acceptances out of 20), + // so the full pool is visited (exhausted) without reaching target_size=10, + // and well within visit_limit=100 -- this is pool exhaustion, not budget + // truncation. + WallClockCandidateStats stats = selectWallClockCandidates( + candidates, 10, 100, generator, [&](int tid) { + if (tid % 3 != 0) { + return WallClockCandidateOutcome::PRECHECK_REJECTED; + } + selected.insert(tid); + return WallClockCandidateOutcome::SIGNAL_SENT; + }); + + EXPECT_EQ(20u, stats.visited); + EXPECT_EQ(7u, stats.slots_consumed); + EXPECT_EQ(13u, stats.precheck_rejected); + EXPECT_EQ(7u, selected.size()); + EXPECT_FALSE(stats.visit_limit_reached); +} diff --git a/ddprof-lib/src/test/cpp/wallClock_ut.cpp b/ddprof-lib/src/test/cpp/wallClock_ut.cpp new file mode 100644 index 0000000000..5ae8c69c79 --- /dev/null +++ b/ddprof-lib/src/test/cpp/wallClock_ut.cpp @@ -0,0 +1,23 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +#include +#include "arguments.h" +#include "wallClock.h" + +// profiler.cpp's Profiler::start() falls back to ThreadFilter::deactivateRecording() +// when the wall engine fails to start. This proves BaseWallClock::start() can +// genuinely return a non-OK Error, i.e. that the fallback's trigger condition +// is reachable in production. +TEST(WallClockStartTest, StartReturnsErrorWhenForced) { + WallClockASGCT wall_clock; + Arguments args; + + BaseWallClock::setForceStartFailureForTest(true); + Error error = wall_clock.start(args); + BaseWallClock::setForceStartFailureForTest(false); + + EXPECT_TRUE(error); +} diff --git a/ddprof-lib/src/test/cpp/wallprecheck_args_ut.cpp b/ddprof-lib/src/test/cpp/wallprecheck_args_ut.cpp index ebde382e01..39a2720113 100644 --- a/ddprof-lib/src/test/cpp/wallprecheck_args_ut.cpp +++ b/ddprof-lib/src/test/cpp/wallprecheck_args_ut.cpp @@ -5,6 +5,21 @@ #include #include "arguments.h" +#include "engine.h" +#include "j9/j9WallClock.h" +#include "wallClock.h" + +TEST(WallPrecheckCapabilityTest, OnlySupportingWallEnginesAdvertiseUnfilteredTracking) { + Engine engine; + J9WallClock j9; + WallClockASGCT asgct; + WallClockJvmti jvmti; + + EXPECT_FALSE(engine.supportsUnfilteredThreadRegistryTracking()); + EXPECT_FALSE(j9.supportsUnfilteredThreadRegistryTracking()); + EXPECT_TRUE(asgct.supportsUnfilteredThreadRegistryTracking()); + EXPECT_TRUE(jvmti.supportsUnfilteredThreadRegistryTracking()); +} TEST(WallPrecheckArgsTest, DefaultsToDisabled) { Arguments args; @@ -68,3 +83,17 @@ TEST(WallPrecheckArgsTest, EnabledWithinLongerArgString) { EXPECT_TRUE(args._wall_precheck); } +TEST(WallPrecheckArgsTest, OmittedFilterRemainsNull) { + Arguments args; + + EXPECT_EQ(nullptr, args._filter); +} + +TEST(WallPrecheckArgsTest, ExplicitEmptyFilterIsPreserved) { + Arguments args; + Error error = args.parse("filter="); + + EXPECT_FALSE(error); + ASSERT_NE(nullptr, args._filter); + EXPECT_STREQ("", args._filter); +} diff --git a/ddprof-lib/src/test/fuzz/README.md b/ddprof-lib/src/test/fuzz/README.md index db69ed045d..abb925bd0e 100644 --- a/ddprof-lib/src/test/fuzz/README.md +++ b/ddprof-lib/src/test/fuzz/README.md @@ -172,6 +172,26 @@ running the wrong number of times (double free or leak across `clear()`, an over the next `get()`, and non-bit-exact round-trips for the `double` specialization (NaN, infinities, subnormals, -0.0). +### fuzz_threadFilter.cpp +**Target**: `ThreadFilter` / `WallClockBlockTracker` - the wall-clock thread +identity registry and its block-run suppression sidecar (see +`threadFilter.h` / `wallClockBlockTracker.h`). + +Drives `registerThread` / `unregisterThread` / `add` / `remove` / +`enterBlockedRun` / `exitBlockedRun` / `init` (registry restart) over a small, +fixed tid domain so slot reuse and registration collisions are frequent. A +shadow model tracks which tid should currently own which slot and traps on +any divergence. + +**Expected bugs**: a still-live tid handed a second slot, two live tids +sharing one slot (the single-owner invariant reviewers flagged as at risk +from `add()`'s unchecked lazy-index fallback), a registry restart +(`init()`) failing to clear a slot's block-run state via +`WallClockBlockTracker::resetAll()`, a stale or forged generation token +incorrectly clearing a newer block run, and any out-of-bounds/use-after-free +in the lazy chunk allocation paths that slot reuse and registry resets +exercise. + ## Corpus Seed corpus files are in `corpus//`. These provide starting points diff --git a/ddprof-lib/src/test/fuzz/fuzz_callTraceStorage.cpp b/ddprof-lib/src/test/fuzz/fuzz_callTraceStorage.cpp index 091780f6d9..592f7d2ae3 100644 --- a/ddprof-lib/src/test/fuzz/fuzz_callTraceStorage.cpp +++ b/ddprof-lib/src/test/fuzz/fuzz_callTraceStorage.cpp @@ -77,7 +77,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { } else if (op < 0xC0) { // processTraces() — verify I1 and I2 std::unordered_set seen; - g_storage->processTraces([&](const std::unordered_set& traces) { + g_storage->processTraces([&](const CallTraceSet& traces) { for (CallTrace* t : traces) { if (t) seen.insert(t->trace_id); } diff --git a/ddprof-lib/src/test/fuzz/fuzz_threadFilter.cpp b/ddprof-lib/src/test/fuzz/fuzz_threadFilter.cpp new file mode 100644 index 0000000000..0016e5d1f4 --- /dev/null +++ b/ddprof-lib/src/test/fuzz/fuzz_threadFilter.cpp @@ -0,0 +1,205 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + * + * libFuzzer target for ThreadFilter / WallClockBlockTracker: the thread + * identity registry and its wall-clock block-run suppression sidecar (see + * threadFilter.h / wallClockBlockTracker.h). + * + * Input bytes are consumed as a stream of operations against a small, fixed + * tid domain (deliberately narrow so registration/reuse collisions are + * frequent): + * op < 0x30: registerThread(tid) + * op < 0x50: unregisterThread(slot_id) for a tracked tid + * op < 0x60: add(tid, slot_id) - context-window enter + * op < 0x70: remove(slot_id) - context-window exit + * op < 0x90: enterBlockedRun(tracker, slot_id, state) + * op < 0xB0: exitBlockedRun(slot_id, generation) - generation-checked; an + * extra fuzzed byte decides whether the real generation is used + * or corrupted, to exercise the stale-token rejection path + * op < 0xD0: exitBlockedRun(slot_id) - unconditional + * op < 0xF0: unregisterThreadByTid(tid) + * op >= 0xF0: init("", true) - simulates a recording restart, which must + * reset both the registry and (via resetRegistrationsLocked -> + * WallClockBlockTracker::resetAll()) every slot's block state + * + * Each op byte is followed by one more byte selecting the tid (and, for + * enterBlockedRun and the generation-checked exit, a third byte selecting + * the OSThreadState / generation-corruption decision respectively). + * + * Invariants verified (violation -> __builtin_trap() -> ASan/fuzzer crash): + * I1. Single-owner TID mapping: registerThread() rediscovering a tid that + * is still live must return the exact slot it was already given, and + * no two live tids may ever be mapped to the same slot_id. This is the + * same invariant flagged in review as being at risk from + * ThreadFilter::add()'s unchecked lazy-index fallback. + * I2. activeSlotForId(slot_id, tid) is non-null if and only if the shadow + * model still considers tid the current owner of slot_id. + * I3. After init() resets the registry, every slot this run ever put into + * a blocked run reports UNKNOWN active-block-state and + * sampled-this-run == false (resetAll() must have actually run). + * I4. A generation-checked exitBlockedRun() called with a generation other + * than the block run's current one must return false and must not + * change that slot's active-block-state (a stale/forged token must + * never clear a newer or already-cleared run). + * + * ASan+UBSan (enabled by the fuzz build config) additionally catch any + * out-of-bounds chunk/tid-index access or use-after-free surfaced by the + * lazy chunk allocation paths as this drives slot reuse and registry resets. + */ + +#include +#include +#include +#include + +#include "threadFilter.h" +#include "wallClockBlockTracker.h" + +namespace { + +// Deliberately small so the same handful of tids are registered/unregistered +// repeatedly, forcing slot reuse rather than spreading across kMaxThreads. +constexpr int kTidDomain = 16; + +ThreadFilter* g_filter = nullptr; +WallClockBlockTracker* g_tracker = nullptr; + +// Shadow model: tid -> slot_id for tids the driver believes are currently +// registered (mirrors only calls this fuzzer itself made). +std::unordered_map g_owner; +// slot_id -> last block-run token, so exitBlockedRun(slot_id, generation) has +// a real generation to check instead of always failing on 0. +std::unordered_map g_last_token; +// Every slot that has ever entered a blocked run this run, checked against +// resetAll() after an init() restart. +std::unordered_set g_ever_blocked; + +void resetShadowState() { + g_owner.clear(); + g_last_token.clear(); +} + +} // namespace + +extern "C" int LLVMFuzzerInitialize(int* /*argc*/, char*** /*argv*/) { + g_filter = new ThreadFilter(); + g_tracker = new WallClockBlockTracker(); + g_filter->setBlockTracker(g_tracker); + g_filter->init("", /*track_unfiltered_wall=*/true); + return 0; +} + +extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { + if (size < 2) return 0; + + size_t pos = 0; + auto nextByte = [&]() -> uint8_t { return pos < size ? data[pos++] : 0; }; + + while (pos < size) { + uint8_t op = nextByte(); + int tid = 1 + (nextByte() % kTidDomain); + + if (op < 0x30) { + ThreadFilter::SlotID slot_id = g_filter->registerThread(tid); + if (slot_id >= 0) { + auto it = g_owner.find(tid); + if (it != g_owner.end() && it->second != slot_id) { + // A still-live tid must always rediscover the same slot. + __builtin_trap(); + } + for (auto& kv : g_owner) { + if (kv.first != tid && kv.second == slot_id) { + // Two live tids must never share a slot_id (I1). + __builtin_trap(); + } + } + g_owner[tid] = slot_id; + } + } else if (op < 0x50) { + auto it = g_owner.find(tid); + if (it != g_owner.end()) { + g_filter->unregisterThread(it->second, tid); + g_owner.erase(it); + } + } else if (op < 0x60) { + auto it = g_owner.find(tid); + if (it != g_owner.end()) { + g_filter->add(tid, it->second); + } + } else if (op < 0x70) { + auto it = g_owner.find(tid); + if (it != g_owner.end()) { + g_filter->remove(it->second); + } + } else if (op < 0x90) { + auto it = g_owner.find(tid); + if (it != g_owner.end()) { + OSThreadState state = static_cast(1 + (nextByte() % 9)); + u64 token = g_tracker->enterBlockedRun(g_filter, it->second, state); + if (token != 0) { + g_last_token[it->second] = token; + g_ever_blocked.insert(it->second); + } + } + } else if (op < 0xB0) { + auto it = g_owner.find(tid); + if (it != g_owner.end()) { + auto tok_it = g_last_token.find(it->second); + if (tok_it != g_last_token.end()) { + uint8_t corruption_byte = nextByte(); + u32 correct_generation = WallClockBlockTracker::tokenGeneration(tok_it->second); + bool corrupt = (corruption_byte & 0x1) != 0; + u32 generation = corrupt + ? correct_generation + 1 + (corruption_byte >> 1) + : correct_generation; + + WallClockBlockTracker::BlockState* block_slot = g_tracker->slotForId(it->second); + OSThreadState state_before = + block_slot ? block_slot->activeBlockState() : OSThreadState::UNKNOWN; + bool exited = g_tracker->exitBlockedRun(it->second, generation); + + if (generation != correct_generation) { + // I4: a mismatched generation must be rejected and must not + // touch the slot's active-block-state. + if (exited || (block_slot && block_slot->activeBlockState() != state_before)) { + __builtin_trap(); + } + } else if (exited) { + g_last_token.erase(it->second); + } + } + } + } else if (op < 0xD0) { + auto it = g_owner.find(tid); + if (it != g_owner.end()) { + g_tracker->exitBlockedRun(it->second); + } + } else if (op < 0xF0) { + g_filter->unregisterThreadByTid(tid); + g_owner.erase(tid); + } else { + g_filter->init("", /*track_unfiltered_wall=*/true); + resetShadowState(); + + for (ThreadFilter::SlotID slot_id : g_ever_blocked) { + WallClockBlockTracker::BlockState* block_slot = g_tracker->slotForId(slot_id); + if (block_slot == nullptr) continue; + if (block_slot->activeBlockState() != OSThreadState::UNKNOWN || + block_slot->sampledThisRun()) { + // resetAll() must have cleared every slot on registry reset (I3). + __builtin_trap(); + } + } + } + + // I2: cheap enough to check after every op. + for (auto& kv : g_owner) { + if (g_filter->activeSlotForId(kv.second, kv.first) == nullptr) { + __builtin_trap(); + } + } + } + + return 0; +} diff --git a/ddprof-stresstest/src/chaos/README.md b/ddprof-stresstest/src/chaos/README.md index 74e5f19d4b..c69f54d5f0 100644 --- a/ddprof-stresstest/src/chaos/README.md +++ b/ddprof-stresstest/src/chaos/README.md @@ -16,6 +16,7 @@ the runner script. | `alloc-storm` | Java alloc engine + GOT-patched libc malloc/free | | `trace-context` | `setTraceContext`/`clearTraceContext` racing signals, span ID propagation | | `reapply-context-value` | concurrent per-slot `setContextValue`/`clearContextValue` churn interleaved with span activation, racing the all-native record's write/read window | +| `park-block-churn` | `ThreadFilter`/`WallClockBlockTracker` registry+block-run races: sleep/park/wait/contended-monitor blocking (some interrupted mid-block) on short-lived threads, racing slot reuse and teardown against the owned/unowned block-run suppression path | ## Deferred diff --git a/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/Main.java b/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/Main.java index 7e846c02d1..8cc43a57b6 100644 --- a/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/Main.java +++ b/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/Main.java @@ -94,6 +94,8 @@ private static Antagonist create(String name) { return new DumpStormAntagonist(); case "reapply-context-value": return new ReapplyContextValueAntagonist(); + case "park-block-churn": + return new ParkBlockChurnAntagonist(); // Deferred: dlopen-churn (needs per-arch dummy .so built in CI prep). default: throw new IllegalArgumentException("unknown antagonist: " + name); diff --git a/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/ParkBlockChurnAntagonist.java b/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/ParkBlockChurnAntagonist.java new file mode 100644 index 0000000000..fe454eeeda --- /dev/null +++ b/ddprof-stresstest/src/chaos/java/com/datadoghq/profiler/chaos/ParkBlockChurnAntagonist.java @@ -0,0 +1,143 @@ +/* + * Copyright 2026, Datadog, Inc + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + */ +package com.datadoghq.profiler.chaos; + +import java.time.Duration; +import java.util.ArrayList; +import java.util.List; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.atomic.AtomicLong; +import java.util.concurrent.locks.LockSupport; + +/** + * Continuously spawns short-lived threads that block (Thread.sleep, + * LockSupport.parkNanos, Object.wait, contended synchronized) and then die, + * some interrupted mid-block. dd-trace-java's instrumentation of these calls + * drives the native park/block JNI hooks (JavaProfiler#parkEnter/#parkExit/ + * #blockEnter/#blockExit), which none of the other antagonists exercise. + * + *

Targets: registry slot reuse racing an in-flight blocked-run generation + * token (ThreadFilter::registerThread/unregisterThread vs. + * WallClockBlockTracker::enterBlockedRun/exitBlockedRun), block-exit running + * after JVMTI ThreadEnd has already reclaimed the slot, and interrupted + * parks (early/EINTR-style wakeups) racing the same exit path. + */ +public final class ParkBlockChurnAntagonist implements Antagonist { + + private final int concurrentThreads; + private final int blockMillis; + + private volatile boolean running; + private Thread driver; + private final AtomicLong totalBlocks = new AtomicLong(); + private final Object monitor = new Object(); + + public ParkBlockChurnAntagonist() { + this(48, 3); + } + + public ParkBlockChurnAntagonist(int concurrentThreads, int blockMillis) { + this.concurrentThreads = concurrentThreads; + this.blockMillis = blockMillis; + } + + @Override + public String name() { + return "park-block-churn"; + } + + @Override + public void start() { + running = true; + driver = new Thread(this::loop, "chaos-park-block-churn"); + driver.setDaemon(true); + driver.start(); + } + + @Override + public void stopGracefully(Duration timeout) { + running = false; + try { + driver.join(timeout.toMillis()); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + } + } + + private void loop() { + int round = 0; + while (running) { + List batch = new ArrayList<>(concurrentThreads); + for (int i = 0; i < concurrentThreads && running; i++) { + final int mode = (round + i) % 4; + Thread t = new Thread(() -> block(mode)); + t.setDaemon(true); + t.start(); + batch.add(t); + if ((i & 0x3) == 0) { + interruptShortly(t); + } + } + for (Thread t : batch) { + try { + t.join(1_000L); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + return; + } + } + round++; + } + } + + /** Wakes {@code t} mid-block so its exit hook races against interruption. */ + private void interruptShortly(Thread t) { + Thread interruptor = new Thread(() -> { + try { + Thread.sleep(1L); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + return; + } + t.interrupt(); + }); + interruptor.setDaemon(true); + interruptor.start(); + } + + private void block(int mode) { + try { + switch (mode) { + case 0: + Thread.sleep(blockMillis); + break; + case 1: + LockSupport.parkNanos(TimeUnit.MILLISECONDS.toNanos(blockMillis)); + break; + case 2: + synchronized (monitor) { + monitor.wait(blockMillis); + } + break; + default: + // Contend the same monitor from many concurrent threads so + // some entrants queue up on MONITOR_WAIT rather than the + // OBJECT_WAIT path exercised by case 2. + synchronized (monitor) { + Thread.sleep(1L); + } + break; + } + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + } + totalBlocks.incrementAndGet(); + } +} diff --git a/ddprof-stresstest/src/jmh/java/com/datadoghq/profiler/WallClockPrecheckBenchmarkHooks.java b/ddprof-stresstest/src/jmh/java/com/datadoghq/profiler/WallClockPrecheckBenchmarkHooks.java new file mode 100644 index 0000000000..ac96ec5b85 --- /dev/null +++ b/ddprof-stresstest/src/jmh/java/com/datadoghq/profiler/WallClockPrecheckBenchmarkHooks.java @@ -0,0 +1,21 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler; + +/** Exposes package-scoped owned-block hooks to the wall-clock overhead benchmark. */ +public final class WallClockPrecheckBenchmarkHooks { + private WallClockPrecheckBenchmarkHooks() {} + + /** Marks the current benchmark worker as entering an owned sleeping interval. */ + public static long enterSleeping(JavaProfiler profiler) { + return profiler.blockEnter(7); + } + + /** Closes an interval returned by {@link #enterSleeping(JavaProfiler)}. */ + public static void exit(JavaProfiler profiler, long token) { + profiler.blockExit(token); + } +} diff --git a/ddprof-stresstest/src/jmh/java/com/datadoghq/profiler/stresstest/scenarios/throughput/WallClockPrecheckOverheadBenchmark.java b/ddprof-stresstest/src/jmh/java/com/datadoghq/profiler/stresstest/scenarios/throughput/WallClockPrecheckOverheadBenchmark.java new file mode 100644 index 0000000000..1af41a979f --- /dev/null +++ b/ddprof-stresstest/src/jmh/java/com/datadoghq/profiler/stresstest/scenarios/throughput/WallClockPrecheckOverheadBenchmark.java @@ -0,0 +1,145 @@ +/* + * Copyright 2026, Datadog, Inc. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package com.datadoghq.profiler.stresstest.scenarios.throughput; + +import com.datadoghq.profiler.JavaProfiler; +import com.datadoghq.profiler.WallClockPrecheckBenchmarkHooks; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.ArrayList; +import java.util.List; +import java.util.concurrent.CountDownLatch; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.locks.LockSupport; +import org.openjdk.jmh.annotations.Benchmark; +import org.openjdk.jmh.annotations.BenchmarkMode; +import org.openjdk.jmh.annotations.Fork; +import org.openjdk.jmh.annotations.Level; +import org.openjdk.jmh.annotations.Measurement; +import org.openjdk.jmh.annotations.Mode; +import org.openjdk.jmh.annotations.OutputTimeUnit; +import org.openjdk.jmh.annotations.Param; +import org.openjdk.jmh.annotations.Scope; +import org.openjdk.jmh.annotations.Setup; +import org.openjdk.jmh.annotations.State; +import org.openjdk.jmh.annotations.TearDown; +import org.openjdk.jmh.annotations.Warmup; + +/** + * Measures steady-state wall-clock timer overhead as an owned-block thread population grows. + * + *

The implementation separately records registry lookup work in the {@code + * wc_precheck_registry_lookups} debug counter and bounds candidate visits to four times {@code + * walltpt} per tick. This benchmark does not report that counter; it compares {@code + * precheck=false} and {@code precheck=true} at each population to detect timer-loop throughput + * regressions independently of one-time startup registration. + */ +@BenchmarkMode(Mode.Throughput) +@OutputTimeUnit(TimeUnit.MILLISECONDS) +@Fork(value = 1, warmups = 0, jvmArgsAppend = "-Xss256k") +@Warmup(iterations = 3, time = 2) +@Measurement(iterations = 5, time = 3) +@State(Scope.Benchmark) +public class WallClockPrecheckOverheadBenchmark { + @Param({"false", "true"}) + public boolean precheck; + + @Param({"100", "500", "1000"}) + public int threadCount; + + private final List workers = new ArrayList<>(); + private volatile boolean running; + private JavaProfiler profiler; + private Path recording; + + /** Creates the requested thread population before starting the profiler. */ + @Setup(Level.Trial) + public void setup() throws Exception { + running = true; + CountDownLatch ready = new CountDownLatch(threadCount); + CountDownLatch profilerStarted = new CountDownLatch(1); + CountDownLatch armed = new CountDownLatch(threadCount); + for (int i = 0; i < threadCount; ++i) { + Thread worker = + new Thread( + () -> { + ready.countDown(); + long token = 0; + boolean armedReported = false; + try { + profilerStarted.await(); + token = WallClockPrecheckBenchmarkHooks.enterSleeping(profiler); + armed.countDown(); + armedReported = true; + while (running) { + LockSupport.parkNanos(TimeUnit.SECONDS.toNanos(1)); + } + } catch (InterruptedException interrupted) { + Thread.currentThread().interrupt(); + } finally { + if (!armedReported) { + armed.countDown(); + } + WallClockPrecheckBenchmarkHooks.exit(profiler, token); + } + }, + "wall-precheck-benchmark-" + i); + worker.setDaemon(true); + worker.start(); + workers.add(worker); + } + if (!ready.await(30, TimeUnit.SECONDS)) { + throw new IllegalStateException("Timed out creating benchmark workers"); + } + + profiler = JavaProfiler.getInstance(); + recording = Files.createTempFile("wall-precheck-overhead-", ".jfr"); + profiler.execute( + "start,wall=1ms,walltpt=16,filter=,wallprecheck=" + + precheck + + ",jfr,file=" + + recording.toAbsolutePath()); + profilerStarted.countDown(); + if (!armed.await(30, TimeUnit.SECONDS)) { + throw new IllegalStateException("Timed out arming benchmark workers"); + } + } + + /** Stops profiling and releases every background worker. */ + @TearDown(Level.Trial) + public void tearDown() throws Exception { + running = false; + for (Thread worker : workers) { + LockSupport.unpark(worker); + } + for (Thread worker : workers) { + worker.join(TimeUnit.SECONDS.toMillis(5)); + } + workers.clear(); + if (profiler != null) { + profiler.stop(); + } + if (recording != null) { + Files.deleteIfExists(recording); + } + } + + /** Provides stable foreground work whose throughput captures timer-loop interference. */ + @Benchmark + public void foregroundWork() { + org.openjdk.jmh.infra.Blackhole.consumeCPU(1_000); + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/AbstractProfilerTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/AbstractProfilerTest.java index b4e988c582..be8cf339ea 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/AbstractProfilerTest.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/AbstractProfilerTest.java @@ -209,6 +209,7 @@ public void setupProfiler(TestInfo testInfo) throws Exception { jfrDump = Files.createTempFile(rootDir, testInfo.getTestMethod().map(m -> m.getDeclaringClass().getSimpleName() + "_" + m.getName()).orElse("unknown") + (testConfig.isEmpty() ? "" : "-" + testConfig.replace('/', '_')), ".jfr"); profiler = JavaProfiler.getInstance(); + beforeProfilerStart(); String command = "start," + getAmendedProfilerCommand() + ",jfr,file=" + jfrDump.toAbsolutePath(); cpuInterval = command.contains("cpu") ? parseInterval(command, "cpu") : (command.contains("interval") ? parseInterval(command, "interval") : Duration.ZERO); wallInterval = parseInterval(command, "wall"); @@ -244,6 +245,17 @@ public void cleanup() throws Exception { protected void before() throws Exception { } + /** + * Runs after the profiler instance is available but before the recording starts. + * + *

Tests may override this hook when their setup must predate profiler thread-event + * registration. + * + * @throws Exception if setup fails + */ + protected void beforeProfilerStart() throws Exception { + } + protected void after() throws Exception { } @@ -294,6 +306,7 @@ protected void runTests(Runnable... runnables) throws InterruptedException { public final void stopProfiler() { if (!stopped) { profiler.stop(); + profiler.clearTraceContext(); stopped = true; checkConfig(); } diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/context/AllNativeContextTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/context/AllNativeContextTest.java index 70007fe03e..b641b7d74c 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/context/AllNativeContextTest.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/context/AllNativeContextTest.java @@ -63,6 +63,7 @@ public void cleanup() { profiler.stop(); profilerStarted = false; } + profiler.clearTraceContext(); } private void start() throws IOException { diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/memleak/JMethodIDInvalidationStressTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/memleak/JMethodIDInvalidationStressTest.java index 443fbd635b..64ff214cdf 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/memleak/JMethodIDInvalidationStressTest.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/memleak/JMethodIDInvalidationStressTest.java @@ -30,6 +30,7 @@ import java.lang.reflect.Method; import java.nio.file.Files; import java.nio.file.Path; +import java.nio.file.StandardCopyOption; import java.util.ArrayList; import java.util.List; import java.util.Map; @@ -110,6 +111,7 @@ public void testProfilerSurvivesConcurrentClassUnloadDuringDump() throws Excepti Path baseFile = tempFile("jmethodid-churn-base"); Path dumpFile = tempFile("jmethodid-churn-dump"); + Path firedDumpFile = null; AtomicBoolean running = new AtomicBoolean(true); List churnThreads = new ArrayList<>(); @@ -148,9 +150,23 @@ public void testProfilerSurvivesConcurrentClassUnloadDuringDump() throws Excepti // with class unloading, not just after it. long deadline = System.currentTimeMillis() + DURATION_MILLIS; int dumps = 0; + long prevSkipped = before.getOrDefault("jmethodid_skipped_count", 0L); while (System.currentTimeMillis() < deadline) { profiler.dump(dumpFile); dumps++; + // The stale-jmethodID branch (JMETHODID_SKIPPED) and the '' label it + // serializes are produced in the same fillJavaMethodInfo call, so a dump whose + // window observed the counter crossing is guaranteed to carry the branch's + // output. Snapshot it: the assertion cannot use just the last dump file, since + // the counter also fires during background JFR buffer flushes between dumps, + // and a stale trace still present at an early dump can be evicted from the + // call-trace storage by later churn before the final dump is written. + long skipped = profiler.getDebugCounters().getOrDefault("jmethodid_skipped_count", 0L); + if (firedDumpFile == null && skipped > prevSkipped) { + firedDumpFile = tempFile("jmethodid-churn-fired"); + Files.copy(dumpFile, firedDumpFile, StandardCopyOption.REPLACE_EXISTING); + } + prevSkipped = skipped; Thread.sleep(50); } @@ -162,6 +178,21 @@ public void testProfilerSurvivesConcurrentClassUnloadDuringDump() throws Excepti // Reaching this line means the profiler survived the whole churn window. Map after = profiler.getDebugCounters(); + long skippedDelta = after.getOrDefault("jmethodid_skipped_count", 0L) + - before.getOrDefault("jmethodid_skipped_count", 0L); + long unreadableLineTableDelta = after.getOrDefault("line_number_table_unreadable", 0L) + - before.getOrDefault("line_number_table_unreadable", 0L); + + // If the counter only fired during background JFR buffer flushes between dumps, + // no dump window observed it and no snapshot was taken. One more dump + // re-serializes any trace still carrying that stale jmethodID -- its + // '' MethodInfo is cached from the first resolution -- so the label + // assertion below stays checkable. + if (skippedDelta > 0 && firedDumpFile == null) { + firedDumpFile = tempFile("jmethodid-churn-fired"); + profiler.dump(firedDumpFile); + } + profiler.stop(); assertTrue(Files.size(dumpFile) > 0, @@ -174,11 +205,6 @@ public void testProfilerSurvivesConcurrentClassUnloadDuringDump() throws Excepti + "can't tell 'no stale jmethodID was hit' apart from 'churn never ran at all' " + "(e.g. a regression in generateChurnClassBytecode or IsolatedClassLoader)."); - long skippedDelta = after.getOrDefault("jmethodid_skipped_count", 0L) - - before.getOrDefault("jmethodid_skipped_count", 0L); - long unreadableLineTableDelta = after.getOrDefault("line_number_table_unreadable", 0L) - - before.getOrDefault("line_number_table_unreadable", 0L); - // Whether the stale-jmethodID race actually gets hit within the churn window is // JVM/host-discretionary (see class javadoc); treat "never observed" as an aborted // run rather than a failure so a healthy host that just didn't race tightly enough @@ -206,7 +232,10 @@ public void testProfilerSurvivesConcurrentClassUnloadDuringDump() throws Excepti + " JVMTI-resolution-failure branch ran; skipping to avoid a spurious failure."); // Assert that the recording produced by that branch uses the '' label and // never the legacy 'jvmtiError' one -- this fails if the label is reverted to 'jvmtiError'. - assertUnloadedFrameLabel(dumpFile); + // With skippedDelta > 0, firedDumpFile is always set: either the dump whose window + // observed the counter crossing (label emitted in that same fillJavaMethodInfo call) + // or the extra post-churn dump taken above. + assertUnloadedFrameLabel(firedDumpFile); } finally { running.set(false); for (Thread t : churnThreads) { @@ -233,6 +262,12 @@ public void testProfilerSurvivesConcurrentClassUnloadDuringDump() throws Excepti Files.deleteIfExists(dumpFile); } catch (IOException ignored) { } + if (firedDumpFile != null) { + try { + Files.deleteIfExists(firedDumpFile); + } catch (IOException ignored) { + } + } } } diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/ConcurrentOwnedBlockChurnTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/ConcurrentOwnedBlockChurnTest.java new file mode 100644 index 0000000000..f745104494 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/ConcurrentOwnedBlockChurnTest.java @@ -0,0 +1,140 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.wallclock; + +import static org.junit.jupiter.api.Assertions.assertTrue; + +import com.datadoghq.profiler.AbstractProfilerTest; +import com.datadoghq.profiler.JfrEvent; +import com.datadoghq.profiler.JfrEvents; +import com.datadoghq.profiler.Platform; +import com.datadoghq.profiler.ProfilerOwnedBlockHooks; +import java.util.ArrayList; +import java.util.List; +import java.util.concurrent.atomic.AtomicLong; +import org.junit.jupiter.api.Assumptions; +import org.junitpioneer.jupiter.RetryingTest; + +/** + * Verifies owned-block once-per-run suppression stays correct under real thread churn, unlike + * {@link UnfilteredWallPrecheckTest} and {@link UnfilteredWallPrecheckRestartTest}, which each + * pin down one deterministic worker/interleaving. Many short-lived threads concurrently register + * a registry slot, arm an owned SLEEPING block, block, and die (forcing JVMTI ThreadEnd / slot + * reuse) while the wall-clock engine is sampling every thread ({@code filter=}), so this + * specifically races {@code ThreadFilter} slot reuse against {@code WallClockBlockTracker} + * enter/exit - the coupling introduced by splitting the tracker out of the registry. Unlike the + * chaos {@code park-block-churn} antagonist (crash-only signal), this asserts on actual sample + * counts and debug counters, so it also catches silent over-/under-suppression regressions. + */ +public class ConcurrentOwnedBlockChurnTest extends AbstractProfilerTest { + private static final int OSTHREAD_STATE_SLEEPING = 7; + private static final long BLOCK_MILLIS = 60; + private static final int WORKERS_PER_ROUND = 24; + private static final int ROUNDS = 8; + private static final String SUPPRESSED_RUN_COUNTER = "wc_signals_suppressed_sampled_run"; + + private static final String WORKER_NAME_PREFIX = "owned-block-churn-"; + + private final AtomicLong armedRuns = new AtomicLong(); + + @Override + protected String getProfilerCommand() { + return "wall=1ms,filter=,wallprecheck=true"; + } + + @Override + protected boolean isPlatformSupported() { + return !Platform.isJ9(); + } + + @Override + protected void withTestAssumptions() { + Assumptions.assumeTrue( + Platform.isJavaVersionAtLeast(11), + "Sleeping-state precheck assertions are stable on JDK 11+"); + } + + /** + * Spawns many short-lived owned-block workers across several rounds, then verifies suppression + * collapsed the resulting signals into roughly one sample per armed run rather than either + * sampling every signal (under-suppression) or losing every armed run (over-suppression). + * + * @throws InterruptedException if a worker join is interrupted + */ + @RetryingTest(3) + public void churnedOwnedBlocksAreSuppressedRoughlyOncePerRun() throws InterruptedException { + for (int round = 0; round < ROUNDS; round++) { + List workers = new ArrayList<>(WORKERS_PER_ROUND); + for (int i = 0; i < WORKERS_PER_ROUND; i++) { + Thread worker = new Thread(this::runOwnedBlock, WORKER_NAME_PREFIX + round + "-" + i); + worker.setDaemon(true); + workers.add(worker); + } + for (Thread worker : workers) { + worker.start(); + } + for (Thread worker : workers) { + worker.join(); + } + } + + stopProfiler(); + + long totalArmed = armedRuns.get(); + assertTrue(totalArmed > 0, "Expected at least some owned-block runs to arm"); + + long workerSamples = countWorkerSamples(); + assertTrue( + workerSamples > 0, + "Expected at least one sample across " + totalArmed + " armed owned-block runs"); + assertTrue( + workerSamples <= totalArmed * 3, + "Expected roughly one sample per armed run (under-suppression / a slot-reuse " + + "race leaking extra samples), got " + + workerSamples + + " samples for " + + totalArmed + + " armed runs"); + + long suppressedAfter = suppressedSignals(); + if (suppressedAfter >= 0) { + assertTrue( + suppressedAfter > 0, + "Expected the owned-block once-per-run suppression counter to have fired at least " + + "once under sustained churn"); + } + } + + private void runOwnedBlock() { + long token = ProfilerOwnedBlockHooks.blockEnter(profiler, OSTHREAD_STATE_SLEEPING); + if (token != 0) { + armedRuns.incrementAndGet(); + } + try { + Thread.sleep(BLOCK_MILLIS); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + } finally { + ProfilerOwnedBlockHooks.blockExit(profiler, token); + } + } + + private long countWorkerSamples() { + long count = 0; + JfrEvents events = verifyEvents("datadog.MethodSample", false); + for (JfrEvent item : events) { + String threadName = item.getThreadName("eventThread"); + if (threadName != null && threadName.startsWith(WORKER_NAME_PREFIX)) { + count++; + } + } + return count; + } + + private long suppressedSignals() { + return profiler.getDebugCounters().getOrDefault(SUPPRESSED_RUN_COUNTER, -1L); + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/J9WallClockPrecheckCapabilityTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/J9WallClockPrecheckCapabilityTest.java new file mode 100644 index 0000000000..f91eeef578 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/J9WallClockPrecheckCapabilityTest.java @@ -0,0 +1,36 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.wallclock; + +import static org.junit.jupiter.api.Assertions.assertEquals; + +import com.datadoghq.profiler.AbstractProfilerTest; +import com.datadoghq.profiler.Platform; +import com.datadoghq.profiler.ProfilerOwnedBlockHooks; +import org.junit.jupiter.api.Test; + +/** Verifies that unsupported J9 wall sampling does not activate unfiltered precheck tracking. */ +public class J9WallClockPrecheckCapabilityTest extends AbstractProfilerTest { + private static final int OSTHREAD_STATE_SLEEPING = 7; + + /** Ensures owned-block hooks stay inactive when the selected wall engine cannot consume them. */ + @Test + public void unsupportedEngineDoesNotActivateRegistry() { + long token = ProfilerOwnedBlockHooks.blockEnter(profiler, OSTHREAD_STATE_SLEEPING); + + assertEquals(0L, token, "J9WallClock must not activate unfiltered precheck tracking"); + } + + @Override + protected boolean isPlatformSupported() { + return Platform.isJ9(); + } + + @Override + protected String getProfilerCommand() { + return "wall=1ms,wallsampler=jvmti,filter=,wallprecheck=true"; + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/JvmtiBasedPrecheckTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/JvmtiBasedPrecheckTest.java index 5190e7da61..c8f6a501e7 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/JvmtiBasedPrecheckTest.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/JvmtiBasedPrecheckTest.java @@ -59,6 +59,6 @@ protected String getProfilerCommand() { @Override protected String getPrecheckDisabledProfilerCommand() { - return "wall=1ms,wallprecheck=false,filter=0,jvmtistacks=true"; + return "wall=1ms,wallprecheck=false,jvmtistacks=true"; } } diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/JvmtiBasedUnfilteredWallPrecheckTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/JvmtiBasedUnfilteredWallPrecheckTest.java new file mode 100644 index 0000000000..1c5491e3a9 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/JvmtiBasedUnfilteredWallPrecheckTest.java @@ -0,0 +1,44 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.wallclock; + +import static org.junit.jupiter.api.Assertions.assertTrue; + +import java.util.Map; +import org.junit.jupiter.api.Assumptions; + +/** Runs unfiltered owned-block precheck coverage through delegated JVMTI stack collection. */ +public class JvmtiBasedUnfilteredWallPrecheckTest extends UnfilteredWallPrecheckTest { + private boolean jvmtiDelegationAvailable; + private long requestedBefore; + + @Override + protected void before() { + Map counters = profiler.getDebugCounters(); + Assumptions.assumeTrue( + counters.getOrDefault("jvmti_stacks_init_ok", 0L) > 0, + "HotSpot RequestStackTrace JVMTI extension is not available"); + jvmtiDelegationAvailable = true; + requestedBefore = counters.getOrDefault("jvmti_stacks_requested", 0L); + } + + @Override + protected void after() { + if (!jvmtiDelegationAvailable) { + return; + } + long requestedAfter = + profiler.getDebugCounters().getOrDefault("jvmti_stacks_requested", 0L); + assertTrue( + requestedAfter > requestedBefore, + "Expected wallclock jvmtistacks path to request delegated stack traces"); + } + + @Override + protected String getProfilerCommand() { + return super.getProfilerCommand() + ",jvmtistacks=true"; + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/PrecheckEfficiencyTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/PrecheckEfficiencyTest.java index 050d762c4f..0f150fb773 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/PrecheckEfficiencyTest.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/PrecheckEfficiencyTest.java @@ -296,6 +296,9 @@ public void realisticServiceWorkload() throws Exception { @Override protected String getProfilerCommand() { + // The workload deliberately has no tracing context; it relies on the + // default context-filter scope (filter="0") plus each worker thread + // explicitly registering itself via registerCurrentThreadForWallClockProfiling(). return "wall=1ms"; } } diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/PrecheckTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/PrecheckTest.java index 10b7da59df..8f5133bea0 100644 --- a/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/PrecheckTest.java +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/PrecheckTest.java @@ -28,6 +28,8 @@ */ public class PrecheckTest extends AbstractProfilerTest { private static final int OSTHREAD_STATE_SLEEPING = 7; + private static final int POOL_WORKERS = 4; + private static final long POOL_SLEEP_MILLIS = 300; private static final String TAIL_WEIGHT_THREAD = "precheck-tail-weight"; private static final int TAIL_WEIGHT_ITERATIONS = 50; private static final int TAIL_WEIGHT_SLEEP_MILLIS = 6; @@ -65,6 +67,57 @@ public void testSleepingThreadIsNotSampled() throws InterruptedException { } } + /** + * Verifies that {@code samplePoolSize} in {@code datadog.WallClockSamplingEpoch} still counts + * threads whose owned blocked run is already suppressed. The timer drops those threads before + * reservoir sampling, but the pool size must be taken before that step so it keeps counting + * every candidate, as it does in unfiltered recordings. + * + * @throws InterruptedException if a worker is interrupted + */ + @Test + public void samplePoolSizeCountsSuppressedThreads() throws InterruptedException { + Assumptions.assumeTrue(!Platform.isJ9()); + Assumptions.assumeTrue(Platform.isJavaVersionAtLeast(11)); + + // Fewer workers than the default reservoir (16 threads per tick), so every + // unsuppressed worker is signaled and armed on its first tick. + Thread[] workers = new Thread[POOL_WORKERS]; + for (int i = 0; i < POOL_WORKERS; i++) { + workers[i] = new Thread(() -> { + registerCurrentThreadForWallClockProfiling(); + long token = ProfilerOwnedBlockHooks.blockEnter(profiler, OSTHREAD_STATE_SLEEPING); + try { + Thread.sleep(POOL_SLEEP_MILLIS); + } catch (InterruptedException ignored) { + } finally { + ProfilerOwnedBlockHooks.blockExit(profiler, token); + profiler.removeThread(); + } + }, "precheck-pool-" + i); + workers[i].start(); + } + for (Thread worker : workers) { + worker.join(); + } + + stopProfiler(); + + // While all workers sit in suppressed runs, each tick suppresses every one of them. + // Before the fix those ticks reported a pool size of 0. + boolean sawFullPoolWhileSuppressing = false; + for (JfrEvent epoch : verifyEvents("datadog.WallClockSamplingEpoch")) { + if (epoch.getLong("numSuppressedSampledRun", 0) >= POOL_WORKERS + && epoch.getLong("samplePoolSize", 0) >= POOL_WORKERS) { + sawFullPoolWhileSuppressing = true; + break; + } + } + assertTrue(sawFullPoolWhileSuppressing, + "Expected an epoch that suppressed all " + POOL_WORKERS + + " workers while still counting them in samplePoolSize"); + } + @Test public void unownedSleepingThreadIsNotExactOncePerRunSuppressed() throws Exception { Assumptions.assumeTrue(!Platform.isJ9()); @@ -124,6 +177,7 @@ public void tracedSleepingThreadIsSampled() throws InterruptedException { Assumptions.assumeTrue(Platform.isJavaVersionAtLeast(11)); registerCurrentThreadForWallClockProfiling(); + Map countersBefore = profiler.getDebugCounters(); profiler.setTraceContext(0x5100L, 0x5101L, 0L, 0x5101L, -1, null, -1, null); try { Thread.sleep(300); @@ -138,9 +192,11 @@ public void tracedSleepingThreadIsSampled() throws InterruptedException { assertTrue(sampleCount >= 10, "Expected normal MethodSample volume for traced sleep, got: " + sampleCount); - Map counters = profiler.getDebugCounters(); - if (counters.containsKey("wc_signals_suppressed_sampled_run")) { - assertEquals(0L, counters.get("wc_signals_suppressed_sampled_run"), + if (countersBefore.containsKey("wc_signals_suppressed_sampled_run")) { + long suppressedBefore = countersBefore.get("wc_signals_suppressed_sampled_run"); + long suppressedAfter = profiler.getDebugCounters() + .getOrDefault("wc_signals_suppressed_sampled_run", 0L); + assertEquals(suppressedBefore, suppressedAfter, "wc_signals_suppressed_sampled_run must not increment for traced sleep"); } } @@ -187,11 +243,15 @@ private void leaveClearedInitializedContext() { @Override protected String getProfilerCommand() { + // This suite verifies sampling and suppression for threads outside a + // tracing-context window. It relies on the default context-filter + // scope (filter="0") plus each worker thread explicitly registering + // itself via registerCurrentThreadForWallClockProfiling()/addThread(). return "wall=1ms,wallprecheck=true"; } protected String getPrecheckDisabledProfilerCommand() { - return "wall=1ms,wallprecheck=false,filter=0"; + return "wall=1ms,wallprecheck=false"; } private WeightedSamples weightedSamplesForThread(String threadName) { diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/UnfilteredWallPrecheckFallbackTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/UnfilteredWallPrecheckFallbackTest.java new file mode 100644 index 0000000000..449bf11897 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/UnfilteredWallPrecheckFallbackTest.java @@ -0,0 +1,59 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.wallclock; + +import static org.junit.jupiter.api.Assertions.assertFalse; + +import com.datadoghq.profiler.AbstractProfilerTest; +import com.datadoghq.profiler.JavaProfilerTestSupport; +import org.junit.jupiter.api.Assumptions; +import org.junit.jupiter.api.Test; + +/** + * Regression test for the {@code Profiler::start()} fallback that closes thread-registry + * admission when the wall engine fails to activate in unfiltered wall-precheck tracking mode, so + * unrelated engines that did start successfully (e.g. {@code cpu=}) don't keep paying registry + * overhead for the rest of the recording. + */ +public class UnfilteredWallPrecheckFallbackTest extends AbstractProfilerTest { + + @Override + protected void beforeProfilerStart() throws Exception { + super.beforeProfilerStart(); + // In effect only for this test's start() call; cleared at the top of the test method below. + JavaProfilerTestSupport.setForceWallStartFailureForTest(true); + // The forced-start-failure toggle is a DEBUG-only native hook (no-op in + // release builds). Self-skip when it isn't armed so the test doesn't fail + // spuriously in release: without a real wall-engine start failure, registry + // admission stays open and the assertion below cannot hold. + Assumptions.assumeTrue( + JavaProfilerTestSupport.isForceWallStartFailureArmedForTest(), + "force-wall-start-failure test hook is a no-op outside DEBUG native builds;" + + " the Profiler::start() fallback under test only fires when the wall engine" + + " can be made to fail, which requires the DEBUG-only hook -- skipping"); + } + + @Override + protected String getProfilerCommand() { + // Empty filter + wallprecheck=true selects unfiltered wall-precheck tracking + // (Profiler::start()'s track_unfiltered_wall). cpu proves an unrelated engine + // keeps running despite the wall engine's forced start failure. Bare (interval-less) + // event names so AbstractProfilerTest.checkConfig()'s interval assertions -- which + // only apply when an explicit "cpu=" / "wall=" interval was requested -- are skipped; + // the wall engine's configured interval is never published because start() fails + // before reaching that point. + return "cpu,wall,filter=,wallprecheck=true"; + } + + @Test + public void wallEngineFailureClosesRegistryAdmission() { + JavaProfilerTestSupport.setForceWallStartFailureForTest(false); + assertFalse( + JavaProfilerTestSupport.isThreadRegistryActiveForTest(), + "wall engine failed to activate in unfiltered-wall-precheck mode; registry admission" + + " must be closed so unrelated engines (cpu=) don't keep paying registry overhead"); + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/UnfilteredWallPrecheckRestartTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/UnfilteredWallPrecheckRestartTest.java new file mode 100644 index 0000000000..afc78ad543 --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/UnfilteredWallPrecheckRestartTest.java @@ -0,0 +1,127 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.wallclock; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotEquals; + +import com.datadoghq.profiler.AbstractProfilerTest; +import com.datadoghq.profiler.Platform; +import com.datadoghq.profiler.ProfilerOwnedBlockHooks; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.concurrent.ExecutorService; +import java.util.concurrent.Executors; +import java.util.concurrent.Future; +import org.junit.jupiter.api.Test; + +/** Verifies that unfiltered wall registry activation does not leak across recordings. */ +public class UnfilteredWallPrecheckRestartTest extends AbstractProfilerTest { + private static final int OSTHREAD_STATE_SLEEPING = 7; + + /** Exercises enabled, disabled, CPU-only, and re-enabled tracking in one process. */ + @Test + public void recordingRestartsReconfigureUnfilteredTracking() throws Exception { + assertOwnedBlockArmed(); + stopProfiler(); + + runRecording("wall=1ms,filter=,wallprecheck=false", false); + runRecording("cpu=1ms,filter=,wallprecheck=true", false); + runRecording("wall=1ms,filter=,wallprecheck=true", true); + } + + /** Verifies epoch refresh and lazy registration for workers that survive a stopped gap. */ + @Test + public void workerLifecyclesRemainSafeAcrossStoppedGap() throws Exception { + ExecutorService survivingWorker = Executors.newSingleThreadExecutor(); + ExecutorService stoppedGapWorker = null; + Path recording = null; + boolean restarted = false; + try { + long oldToken = enterBlock(survivingWorker); + assertNotEquals(0L, oldToken, "Expected the initial worker run to be armed"); + + stopProfiler(); + stoppedGapWorker = Executors.newSingleThreadExecutor(); + // Force creation while JVMTI lifecycle callbacks are disabled. + assertEquals(0L, enterBlock(stoppedGapWorker)); + + recording = Files.createTempFile("unfiltered-wall-worker-restart-", ".jfr"); + profiler.execute( + "start," + getProfilerCommand() + ",jfr,file=" + recording.toAbsolutePath()); + restarted = true; + + long newToken = enterBlock(survivingWorker); + assertNotEquals(0L, newToken, "Expected the surviving worker to refresh its slot"); + exitBlock(survivingWorker, oldToken); + assertEquals( + 0L, + enterBlock(survivingWorker), + "A token from the previous recording cleared the current worker run"); + exitBlock(survivingWorker, newToken); + + long stoppedGapToken = enterBlock(stoppedGapWorker); + assertNotEquals( + 0L, stoppedGapToken, "Expected the stopped-gap worker to register lazily after restart"); + exitBlock(stoppedGapWorker, stoppedGapToken); + } finally { + if (restarted) { + profiler.stop(); + } + survivingWorker.shutdownNow(); + if (stoppedGapWorker != null) { + stoppedGapWorker.shutdownNow(); + } + if (recording != null) { + Files.deleteIfExists(recording); + } + } + } + + @Override + protected String getProfilerCommand() { + return "wall=1ms,filter=,wallprecheck=true"; + } + + @Override + protected boolean isPlatformSupported() { + return !Platform.isJ9(); + } + + private void assertOwnedBlockArmed() { + long token = ProfilerOwnedBlockHooks.blockEnter(profiler, OSTHREAD_STATE_SLEEPING); + assertNotEquals(0L, token, "Expected unfiltered wall precheck to arm the owned block"); + ProfilerOwnedBlockHooks.blockExit(profiler, token); + } + + private void runRecording(String command, boolean expectArmed) throws Exception { + Path recording = Files.createTempFile("unfiltered-wall-restart-", ".jfr"); + profiler.execute("start," + command + ",jfr,file=" + recording.toAbsolutePath()); + try { + long token = ProfilerOwnedBlockHooks.blockEnter(profiler, OSTHREAD_STATE_SLEEPING); + if (expectArmed) { + assertNotEquals(0L, token, "Expected unfiltered wall tracking after restart"); + ProfilerOwnedBlockHooks.blockExit(profiler, token); + } else { + assertEquals(0L, token, "Registry tracking leaked into " + command); + } + } finally { + profiler.stop(); + Files.deleteIfExists(recording); + } + } + + private long enterBlock(ExecutorService worker) throws Exception { + Future result = + worker.submit( + () -> ProfilerOwnedBlockHooks.blockEnter(profiler, OSTHREAD_STATE_SLEEPING)); + return result.get(); + } + + private void exitBlock(ExecutorService worker, long token) throws Exception { + worker.submit(() -> ProfilerOwnedBlockHooks.blockExit(profiler, token)).get(); + } +} diff --git a/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/UnfilteredWallPrecheckTest.java b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/UnfilteredWallPrecheckTest.java new file mode 100644 index 0000000000..2b7c98fb3a --- /dev/null +++ b/ddprof-test/src/test/java/com/datadoghq/profiler/wallclock/UnfilteredWallPrecheckTest.java @@ -0,0 +1,277 @@ +/* + * Copyright 2026, Datadog, Inc. + * SPDX-License-Identifier: Apache-2.0 + */ + +package com.datadoghq.profiler.wallclock; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertSame; +import static org.junit.jupiter.api.Assertions.assertTrue; + +import com.datadoghq.profiler.AbstractProfilerTest; +import com.datadoghq.profiler.JfrEvent; +import com.datadoghq.profiler.JfrEvents; +import com.datadoghq.profiler.Platform; +import com.datadoghq.profiler.ProfilerOwnedBlockHooks; +import java.util.concurrent.ExecutorService; +import java.util.concurrent.Executors; +import java.util.concurrent.Future; +import java.util.concurrent.FutureTask; +import java.util.concurrent.TimeUnit; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.Assumptions; +import org.junitpioneer.jupiter.RetryingTest; + +/** Verifies owned-block prechecks when legacy {@code filter=} samples every thread. */ +public class UnfilteredWallPrecheckTest extends AbstractProfilerTest { + private static final int OSTHREAD_STATE_SLEEPING = 7; + private static final long SLEEP_MILLIS = 300; + private static final String PRE_EXISTING_THREAD_NAME = "unfiltered-precheck-existing"; + private static final String SUPPRESSED_RUN_COUNTER = "wc_signals_suppressed_sampled_run"; + private static final String UNOWNED_SUPPRESSED_COUNTER = "wc_unowned_blocked_suppressed"; + + private ExecutorService preExistingWorker; + private Thread preExistingThread; + + /** + * Verifies that an untraced thread's owned sleeping run is sampled once and then suppressed. + * + * @throws Exception if the worker cannot complete + */ + @RetryingTest(3) + public void sleepingThreadOutsideContextWindowIsOwnedBlockSuppressed() throws Exception { + long suppressedBefore = suppressedSignals(); + assertTrue( + runPreExistingSleepingWorker(false) != 0, + "Expected native blockEnter to arm SLEEPING state"); + + stopProfiler(); + assertSuppressedSamples(PRE_EXISTING_THREAD_NAME); + assertOwnedBlockSuppressionObserved(suppressedBefore); + } + + /** + * Verifies that entering the context window prevents owned-block suppression in an + * unfiltered recording. + * + * @throws Exception if the worker cannot complete + */ + @RetryingTest(3) + public void sleepingThreadInsideContextWindowIsNotOverSuppressed() throws Exception { + assertTrue( + runPreExistingSleepingWorker(true) != 0, + "Expected native blockEnter to arm SLEEPING state"); + + stopProfiler(); + + long sampleCount = samplesForThread(PRE_EXISTING_THREAD_NAME); + assertTrue( + sampleCount >= 10, + "Expected normal MethodSample volume inside the context window, got: " + sampleCount); + } + + /** + * Verifies that a pre-existing thread can lazily bind its slot through the park hook. + * + * @throws Exception if the worker cannot complete + */ + @RetryingTest(3) + public void parkedPreExistingThreadOutsideContextWindowIsOwnedBlockSuppressed() + throws Exception { + long suppressedBefore = suppressedSignals(); + runPreExistingParkedWorker(); + + stopProfiler(); + assertSuppressedSamples(PRE_EXISTING_THREAD_NAME); + assertOwnedBlockSuppressionObserved(suppressedBefore); + } + + /** + * Verifies that a thread which already owns a registry slot still receives ordinary per-signal + * sampling outside an explicitly owned block. + * + * @throws Exception if the worker cannot complete + */ + @RetryingTest(3) + public void unownedSleepAfterOwnedBlockUsesNormalSampling() throws Exception { + long unownedSuppressedBefore = unownedSuppressedSignals(); + assertTrue( + runPreExistingUnownedSleepingWorker() != 0, + "Expected the setup block to register the worker"); + long unownedSuppressedAfter = unownedSuppressedSignals(); + + stopProfiler(); + + long sampleCount = samplesForThread(PRE_EXISTING_THREAD_NAME); + assertTrue( + sampleCount >= 10, + "Expected normal MethodSample volume for an unowned sleep, got: " + sampleCount); + if (unownedSuppressedBefore >= 0) { + assertEquals( + unownedSuppressedBefore, + unownedSuppressedAfter, + "Unfiltered mode must not use observation-only unowned suppression"); + } + } + + /** + * Retains coverage for post-start threads, whose owned-block hook must register a slot lazily. + * + * @throws Exception if the worker cannot complete + */ + @RetryingTest(3) + public void postStartSleepingThreadLazilyRegistersOwnedBlock() throws Exception { + String threadName = "unfiltered-precheck-post-start"; + assertTrue( + runPostStartSleepingWorker(threadName) != 0, + "Expected the owned-block hook to register and arm SLEEPING state"); + + stopProfiler(); + assertSuppressedSamples(threadName); + } + + @Override + protected void beforeProfilerStart() throws Exception { + preExistingWorker = + Executors.newSingleThreadExecutor( + task -> { + Thread worker = new Thread(task, PRE_EXISTING_THREAD_NAME); + worker.setDaemon(true); + return worker; + }); + preExistingThread = preExistingWorker.submit(Thread::currentThread).get(); + } + + /** Stops the worker that was deliberately created before profiler startup. */ + @AfterEach + public void stopPreExistingWorker() throws InterruptedException { + if (preExistingWorker == null) { + return; + } + preExistingWorker.shutdownNow(); + assertTrue( + preExistingWorker.awaitTermination(5, TimeUnit.SECONDS), + "Pre-existing wall-clock worker did not terminate"); + } + + @Override + protected boolean isPlatformSupported() { + return !Platform.isJ9(); + } + + @Override + protected void withTestAssumptions() { + Assumptions.assumeTrue( + Platform.isJavaVersionAtLeast(11), + "Sleeping-state precheck assertions are stable on JDK 11+"); + } + + @Override + protected String getProfilerCommand() { + return "wall=1ms,filter=,wallprecheck=true"; + } + + private long runPreExistingSleepingWorker(boolean enterContextWindowDuringBlock) + throws Exception { + Future sleep = + preExistingWorker.submit( + () -> { + assertSame(preExistingThread, Thread.currentThread()); + return runSleepingBlock(enterContextWindowDuringBlock); + }); + return sleep.get(); + } + + private long runPostStartSleepingWorker(String threadName) throws Exception { + FutureTask sleep = new FutureTask<>(() -> runSleepingBlock(false)); + Thread worker = new Thread(sleep, threadName); + worker.start(); + return sleep.get(); + } + + private long runPreExistingUnownedSleepingWorker() throws Exception { + return preExistingWorker + .submit( + () -> { + assertSame(preExistingThread, Thread.currentThread()); + long token = + ProfilerOwnedBlockHooks.blockEnter(profiler, OSTHREAD_STATE_SLEEPING); + ProfilerOwnedBlockHooks.blockExit(profiler, token); + Thread.sleep(SLEEP_MILLIS); + return token; + }) + .get(); + } + + private long runSleepingBlock(boolean enterContextWindowDuringBlock) throws Exception { + long token = ProfilerOwnedBlockHooks.blockEnter(profiler, OSTHREAD_STATE_SLEEPING); + if (enterContextWindowDuringBlock) { + profiler.addThread(); + } + try { + Thread.sleep(SLEEP_MILLIS); + return token; + } finally { + ProfilerOwnedBlockHooks.blockExit(profiler, token); + if (enterContextWindowDuringBlock) { + profiler.removeThread(); + } + } + } + + private void runPreExistingParkedWorker() throws Exception { + preExistingWorker + .submit( + () -> { + assertSame(preExistingThread, Thread.currentThread()); + ProfilerOwnedBlockHooks.parkEnter(profiler); + try { + long deadline = System.nanoTime() + TimeUnit.MILLISECONDS.toNanos(SLEEP_MILLIS); + while (System.nanoTime() < deadline) { + // Keep the OS thread runnable so suppression must come from the owned park marker. + } + } finally { + ProfilerOwnedBlockHooks.parkExit( + profiler, System.identityHashCode(preExistingThread), 0L); + } + return null; + }) + .get(); + } + + private void assertOwnedBlockSuppressionObserved(long suppressedBefore) { + if (suppressedBefore >= 0) { + assertTrue( + suppressedSignals() > suppressedBefore, + "Expected owned-block once-per-run suppression counter to increase"); + } + } + + private long suppressedSignals() { + return profiler.getDebugCounters().getOrDefault(SUPPRESSED_RUN_COUNTER, -1L); + } + + private long unownedSuppressedSignals() { + return profiler.getDebugCounters().getOrDefault(UNOWNED_SUPPRESSED_COUNTER, -1L); + } + + private void assertSuppressedSamples(String threadName) { + long sampleCount = samplesForThread(threadName); + assertTrue(sampleCount > 0, "Expected the owned block run to be sampled once"); + assertTrue( + sampleCount < 10, + "Expected nearly no samples from owned block thread, got: " + sampleCount); + } + + private long samplesForThread(String threadName) { + long count = 0; + JfrEvents events = verifyEvents("datadog.MethodSample", false); + for (JfrEvent item : events) { + if (threadName.equals(item.getThreadName("eventThread"))) { + count++; + } + } + return count; + } +} diff --git a/utils/run-chaos-harness.sh b/utils/run-chaos-harness.sh index b1a62139c7..7df8664e4e 100755 --- a/utils/run-chaos-harness.sh +++ b/utils/run-chaos-harness.sh @@ -137,11 +137,11 @@ case $CONFIG in profiler) ENABLEMENT="-Ddd.profiling.enabled=true -Ddd.trace.enabled=false" # @Trace is a no-op without the tracer, so trace-context is excluded here. - DEFAULT_ANTAGONISTS="thread-churn,alloc-storm,vthread-churn,classloader-churn,bounded-pool,context-hop,consumer-group,hidden-class-churn,direct-memory,weakref-wave,dump-storm,reapply-context-value" + DEFAULT_ANTAGONISTS="thread-churn,alloc-storm,vthread-churn,classloader-churn,bounded-pool,context-hop,consumer-group,hidden-class-churn,direct-memory,weakref-wave,dump-storm,reapply-context-value,park-block-churn" ;; profiler+tracer) ENABLEMENT="-Ddd.profiling.enabled=true -Ddd.trace.enabled=true" - DEFAULT_ANTAGONISTS="thread-churn,alloc-storm,vthread-churn,classloader-churn,trace-context,bounded-pool,context-hop,consumer-group,hidden-class-churn,direct-memory,weakref-wave,dump-storm,reapply-context-value" + DEFAULT_ANTAGONISTS="thread-churn,alloc-storm,vthread-churn,classloader-churn,trace-context,bounded-pool,context-hop,consumer-group,hidden-class-churn,direct-memory,weakref-wave,dump-storm,reapply-context-value,park-block-churn" ;; *) echo "Unknown configuration: $CONFIG (valid: profiler, profiler+tracer)" >&2